#!/usr/bin/env python3 """ CSV Import Script for Student Data This script imports student data from a CSV file (ISO-8859-1 encoded) into the SQLite database. It handles: - Encoding conversion from ISO-8859-1 to UTF-8 - Splitting "Apellidos, Nombre" into separate fields - Date format conversion from DD/MM/YYYY to YYYY-MM-DD - Discarding the "Estudio" field as per requirements - Validation and error handling Usage: python import_students.py [--database ] [--dry-run] Examples: python import_students.py Datos_programa.csv python import_students.py Datos_programa.csv --database tablets.db python import_students.py Datos_programa.csv --dry-run """ import argparse import csv import os import re import sqlite3 import sys from datetime import datetime from typing import Dict, List, Optional, Tuple # Default database file DEFAULT_DATABASE = 'tablets.db' # CSV field mapping CSV_FIELDS = [ 'order_number', 'full_name', 'birth_date', 'cial_code', 'nif_nie_passport', 'registration_number', 'file_number', 'gender', 'study_group', 'study' # This will be discarded ] def parse_arguments(): """Parse command line arguments.""" parser = argparse.ArgumentParser( description='Import student data from CSV file to SQLite database' ) parser.add_argument( 'csv_file', help='Path to the CSV file to import (ISO-8859-1 encoded)' ) parser.add_argument( '--database', '-d', default=DEFAULT_DATABASE, help=f'Path to SQLite database file (default: {DEFAULT_DATABASE})' ) parser.add_argument( '--dry-run', action='store_true', help='Validate and preview data without importing' ) parser.add_argument( '--delimiter', default=',', help='CSV delimiter (default: comma)' ) parser.add_argument( '--quotechar', default='"', help='CSV quote character (default: double quote)' ) parser.add_argument( '--encoding', default='iso-8859-1', help='CSV file encoding (default: iso-8859-1)' ) parser.add_argument( '--skip-errors', action='store_true', help='Skip rows with errors instead of stopping' ) parser.add_argument( '--verbose', '-v', action='store_true', help='Show detailed processing information' ) return parser.parse_args() def read_csv_file(filepath: str, encoding: str, delimiter: str, quotechar: str) -> List[Dict[str, str]]: """Read CSV file and return list of dictionaries.""" """ Read a CSV file with specified encoding and return its contents as a list of dictionaries. Args: filepath: Path to the CSV file encoding: File encoding (e.g., 'iso-8859-1') delimiter: CSV field delimiter quotechar: CSV quote character Returns: List of dictionaries where keys are header names and values are field values """ if not os.path.exists(filepath): raise FileNotFoundError(f"CSV file not found: {filepath}") if not os.path.isfile(filepath): raise ValueError(f"Path is not a file: {filepath}") rows = [] with open(filepath, 'r', encoding=encoding, newline='') as csvfile: reader = csv.DictReader(csvfile, delimiter=delimiter, quotechar=quotechar) # Convert header names to lowercase and strip whitespace if reader.fieldnames: reader.fieldnames = [name.strip().lower() for name in reader.fieldnames] for row in reader: rows.append(row) return rows def split_full_name(full_name: str) -> Tuple[str, str]: """ Split a full name in format 'APELLIDOS, NOMBRE' into last_name and first_name. Args: full_name: Full name string in format 'Last Name, First Name' Returns: Tuple of (last_name, first_name) """ if not full_name or full_name.strip() == '': return '', '' # Remove surrounding quotes if present full_name = full_name.strip().strip('"').strip("'") # Split on comma followed by optional whitespace parts = re.split(r',\s*', full_name, maxsplit=1) if len(parts) == 2: last_name = parts[0].strip() first_name = parts[1].strip() else: # If no comma, assume it's just a name last_name = '' first_name = full_name.strip() return last_name, first_name def convert_date(date_str: str) -> Optional[str]: """ Convert date from DD/MM/YYYY to YYYY-MM-DD format. Args: date_str: Date string in DD/MM/YYYY format Returns: Date string in YYYY-MM-DD format, or None if invalid """ if not date_str or date_str.strip() == '': return None date_str = date_str.strip() try: # Parse DD/MM/YYYY day, month, year = re.split(r'[/-]', date_str) day = int(day) month = int(month) year = int(year) # Validate date datetime(year, month, day) # Format as YYYY-MM-DD return f"{year:04d}-{month:02d}-{day:02d}" except (ValueError, IndexError): return None def validate_gender(gender: str) -> Optional[str]: """ Validate and normalize gender field. Args: gender: Gender string (M, F, etc.) Returns: Normalized gender ('M', 'F', or 'O') or None if invalid """ if not gender or gender.strip() == '': return None gender = gender.strip().upper() if gender in ['M', 'H', 'MALE', 'HOMBRE']: return 'M' elif gender in ['F', 'MUJER', 'FEMALE']: return 'F' elif gender in ['O', 'OTHER', 'OTRO']: return 'O' else: return None def process_row(row: Dict[str, str], verbose: bool = False) -> Optional[Dict[str, Optional[str]]]: """ Process a single CSV row and convert it to database format. Args: row: Dictionary with CSV field names as keys verbose: Whether to print processing details Returns: Dictionary with processed data ready for database insertion, or None if invalid """ # Normalize the row keys to lowercase and remove special characters normalized_row = {} for key, value in row.items(): # Remove special characters and normalize normalized_key = key.lower().replace('º', '').replace('ó', 'o').replace('á', 'a').replace('í', 'i').replace('ú', 'u').replace('ñ', 'n').replace('.', '').replace(',', '').strip() normalized_row[normalized_key] = value # Map CSV fields to our expected field names mapped_row = {} field_mapping = { 'n or': 'order_number', 'apellidos nombre': 'full_name', 'fecha nac': 'birth_date', 'cial': 'cial_code', 'nif/nie/pas': 'nif_nie_passport', 'registro': 'registration_number', 'expediente': 'file_number', 'sexo': 'gender', 'grupo': 'study_group', 'estudio': 'study' } for csv_key, our_key in field_mapping.items(): if csv_key in normalized_row: mapped_row[our_key] = normalized_row[csv_key] else: mapped_row[our_key] = '' # Extract and process fields try: # Order number order_number = mapped_row.get('order_number', '') if order_number.strip() == '': order_number = None else: try: order_number = int(order_number.strip()) except ValueError: order_number = None # Full name and split into last_name, first_name full_name = mapped_row.get('full_name', '') last_name, first_name = split_full_name(full_name) # Birth date conversion birth_date_str = mapped_row.get('birth_date', '') birth_date = convert_date(birth_date_str) # CIAL code cial_code = mapped_row.get('cial_code', '').strip() if cial_code == '': cial_code = None # NIF/NIE/Passport nif_nie_passport = mapped_row.get('nif_nie_passport', '').strip() if nif_nie_passport == '': nif_nie_passport = None # Registration number registration_number = mapped_row.get('registration_number', '') if registration_number.strip() == '': registration_number = None else: try: registration_number = int(registration_number.strip()) except ValueError: registration_number = None # File number file_number = mapped_row.get('file_number', '').strip() if file_number == '': file_number = None # Gender gender = mapped_row.get('gender', '') gender = validate_gender(gender) # Study group study_group = mapped_row.get('study_group', '').strip() if study_group == '': study_group = None # Create result dictionary result = { 'order_number': order_number, 'cial_code': cial_code, 'nif_nie_passport': nif_nie_passport, 'registration_number': registration_number, 'file_number': file_number, 'first_name': first_name if first_name else None, 'last_name': last_name if last_name else None, 'full_name': full_name.strip() if full_name.strip() else None, 'birth_date': birth_date, 'gender': gender, 'study_group': study_group, 'created_at': datetime.now().strftime('%Y-%m-%d %H:%M:%S'), 'updated_at': datetime.now().strftime('%Y-%m-%d %H:%M:%S') } if verbose: print(f" Processed: {result}") return result except Exception as e: print(f" Error processing row: {e}") return None def validate_row(data: Dict[str, Optional[str]]) -> Tuple[bool, List[str]]: """ Validate processed row data. Args: data: Processed row data Returns: Tuple of (is_valid, list_of_errors) """ errors = [] # Check required fields if not data.get('cial_code'): errors.append("CIAL code is required") if not data.get('first_name'): errors.append("First name is required") if not data.get('last_name'): errors.append("Last name is required") # Check date format birth_date = data.get('birth_date') if birth_date: try: datetime.strptime(birth_date, '%Y-%m-%d') except ValueError: errors.append(f"Invalid birth date format: {birth_date}") # Check gender gender = data.get('gender') if gender and gender not in ['M', 'F', 'O']: errors.append(f"Invalid gender: {gender}") return len(errors) == 0, errors def create_students_table(conn: sqlite3.Connection) -> bool: """ Create the students table if it doesn't exist. Args: conn: SQLite database connection Returns: True if table was created or already exists, False on error """ try: cursor = conn.cursor() cursor.execute(''' CREATE TABLE IF NOT EXISTS students ( id INTEGER PRIMARY KEY AUTOINCREMENT, order_number INTEGER, cial_code TEXT UNIQUE NOT NULL, nif_nie_passport TEXT UNIQUE, registration_number INTEGER, file_number TEXT, first_name TEXT NOT NULL, last_name TEXT NOT NULL, full_name TEXT NOT NULL, birth_date TEXT, gender TEXT CHECK(gender IN ('M', 'F', 'O')), study_group TEXT, created_at TEXT DEFAULT CURRENT_TIMESTAMP, updated_at TEXT DEFAULT CURRENT_TIMESTAMP, CONSTRAINT unique_identification UNIQUE (cial_code, nif_nie_passport) ) ''') # Create indexes cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_cial ON students(cial_code) ''') cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_nif ON students(nif_nie_passport) ''') cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_name ON students(last_name, first_name) ''') cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_birth_date ON students(birth_date) ''') cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_gender ON students(gender) ''') cursor.execute(''' CREATE INDEX IF NOT EXISTS idx_students_group ON students(study_group) ''') conn.commit() return True except sqlite3.Error as e: print(f"Error creating students table: {e}") conn.rollback() return False def insert_student(conn: sqlite3.Connection, data: Dict[str, Optional[str]], verbose: bool = False) -> Tuple[bool, Optional[int]]: """ Insert a student record into the database. Args: conn: SQLite database connection data: Student data dictionary verbose: Whether to print details Returns: Tuple of (success, student_id or None) """ try: cursor = conn.cursor() # Prepare SQL and parameters sql = ''' INSERT INTO students ( order_number, cial_code, nif_nie_passport, registration_number, file_number, first_name, last_name, full_name, birth_date, gender, study_group, created_at, updated_at ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) ''' params = ( data['order_number'], data['cial_code'], data['nif_nie_passport'], data['registration_number'], data['file_number'], data['first_name'], data['last_name'], data['full_name'], data['birth_date'], data['gender'], data['study_group'], data['created_at'], data['updated_at'] ) cursor.execute(sql, params) student_id = cursor.lastrowid conn.commit() if verbose: print(f" Inserted student ID: {student_id}") return True, student_id except sqlite3.IntegrityError as e: if verbose: print(f" Duplicate entry (skipped): {e}") conn.rollback() return False, None except sqlite3.Error as e: print(f" Database error: {e}") conn.rollback() return False, None def check_duplicate(conn: sqlite3.Connection, cial_code: str, nif_nie_passport: Optional[str]) -> bool: """ Check if a student with the same CIAL code or NIF/NIE already exists. Args: conn: SQLite database connection cial_code: CIAL code to check nif_nie_passport: NIF/NIE/Passport to check Returns: True if duplicate exists, False otherwise """ try: cursor = conn.cursor() # Check by CIAL code cursor.execute("SELECT id FROM students WHERE cial_code = ?", (cial_code,)) if cursor.fetchone(): return True # Check by NIF/NIE if provided if nif_nie_passport and nif_nie_passport.strip(): cursor.execute("SELECT id FROM students WHERE nif_nie_passport = ?", (nif_nie_passport,)) if cursor.fetchone(): return True return False except sqlite3.Error as e: print(f"Error checking for duplicates: {e}") return False def import_csv(csv_file: str, database: str, dry_run: bool = False, skip_errors: bool = False, verbose: bool = False) -> Dict[str, int]: """ Main import function. Args: csv_file: Path to CSV file database: Path to SQLite database dry_run: If True, validate without importing skip_errors: If True, skip invalid rows instead of stopping verbose: If True, print detailed information Returns: Dictionary with import statistics """ stats = { 'total_rows': 0, 'processed': 0, 'valid': 0, 'invalid': 0, 'imported': 0, 'duplicates': 0, 'errors': 0 } # Read CSV file try: if verbose: print(f"Reading CSV file: {csv_file}") rows = read_csv_file(csv_file, encoding='iso-8859-1', delimiter=',', quotechar='"') stats['total_rows'] = len(rows) if verbose: print(f"Found {len(rows)} rows in CSV file") except Exception as e: print(f"Error reading CSV file: {e}") stats['errors'] += 1 return stats # Connect to database try: conn = sqlite3.connect(database) conn.execute("PRAGMA foreign_keys = ON") if not dry_run: # Create students table if it doesn't exist if not create_students_table(conn): print("Failed to create students table") stats['errors'] += 1 return stats except sqlite3.Error as e: print(f"Error connecting to database: {e}") stats['errors'] += 1 return stats # Process each row for i, row in enumerate(rows, 1): if verbose: print(f"\nProcessing row {i}/{len(rows)}:") print(f" Raw: {row}") # Process row processed = process_row(row, verbose) if processed is None: stats['errors'] += 1 if not skip_errors: print(f"Error processing row {i}, stopping.") break else: if verbose: print(f" Skipping row {i} due to processing error") continue stats['processed'] += 1 # Validate row is_valid, errors = validate_row(processed) if not is_valid: stats['invalid'] += 1 if verbose: print(f" Invalid row {i}: {errors}") if not skip_errors: print(f"Error validating row {i}, stopping.") break else: continue stats['valid'] += 1 if dry_run: # Just print what would be imported if verbose: print(f" [DRY RUN] Would import: {processed['full_name']} ({processed['cial_code']})") else: # Check for duplicates if check_duplicate(conn, processed['cial_code'], processed.get('nif_nie_passport')): stats['duplicates'] += 1 if verbose: print(f" Duplicate found for CIAL: {processed['cial_code']}, skipping") continue # Insert into database success, student_id = insert_student(conn, processed, verbose) if success: stats['imported'] += 1 else: stats['duplicates'] += 1 # Close database connection if not dry_run: conn.close() return stats def print_statistics(stats: Dict[str, int], dry_run: bool = False): """Print import statistics.""" print("\n" + "=" * 50) print("IMPORT STATISTICS") print("=" * 50) print(f"Total rows in CSV: {stats['total_rows']}") print(f"Rows processed: {stats['processed']}") print(f"Valid rows: {stats['valid']}") print(f"Invalid rows: {stats['invalid']}") if not dry_run: print(f"Rows imported: {stats['imported']}") print(f"Duplicates skipped: {stats['duplicates']}") print(f"Errors: {stats['errors']}") if dry_run: print("\n[DRY RUN] No data was imported to the database.") print("=" * 50) def main(): """Main entry point.""" args = parse_arguments() print(f"Importing student data from: {args.csv_file}") print(f"Database: {args.database}") print(f"Dry run: {args.dry_run}") print(f"Skip errors: {args.skip_errors}") print() # Perform import stats = import_csv( csv_file=args.csv_file, database=args.database, dry_run=args.dry_run, skip_errors=args.skip_errors, verbose=args.verbose ) # Print statistics print_statistics(stats, args.dry_run) # Exit with appropriate code if stats['errors'] > 0: sys.exit(1) elif stats['invalid'] > 0 and not args.skip_errors: sys.exit(1) else: sys.exit(0) if __name__ == '__main__': main()