GestionTablets/scripts/import_students.py

703 lines
21 KiB
Python

#!/usr/bin/env python3
"""
CSV Import Script for Student Data
This script imports student data from a CSV file (ISO-8859-1 encoded) into the SQLite database.
It handles:
- Encoding conversion from ISO-8859-1 to UTF-8
- Splitting "Apellidos, Nombre" into separate fields
- Date format conversion from DD/MM/YYYY to YYYY-MM-DD
- Discarding the "Estudio" field as per requirements
- Validation and error handling
Usage:
python import_students.py <csv_file> [--database <db_file>] [--dry-run]
Examples:
python import_students.py Datos_programa.csv
python import_students.py Datos_programa.csv --database tablets.db
python import_students.py Datos_programa.csv --dry-run
"""
import argparse
import csv
import os
import re
import sqlite3
import sys
from datetime import datetime
from typing import Dict, List, Optional, Tuple
# Default database file
DEFAULT_DATABASE = 'tablets.db'
# CSV field mapping
CSV_FIELDS = [
'order_number',
'full_name',
'birth_date',
'cial_code',
'nif_nie_passport',
'registration_number',
'file_number',
'gender',
'study_group',
'study' # This will be discarded
]
def parse_arguments():
"""Parse command line arguments."""
parser = argparse.ArgumentParser(
description='Import student data from CSV file to SQLite database'
)
parser.add_argument(
'csv_file',
help='Path to the CSV file to import (ISO-8859-1 encoded)'
)
parser.add_argument(
'--database', '-d',
default=DEFAULT_DATABASE,
help=f'Path to SQLite database file (default: {DEFAULT_DATABASE})'
)
parser.add_argument(
'--dry-run',
action='store_true',
help='Validate and preview data without importing'
)
parser.add_argument(
'--delimiter',
default=',',
help='CSV delimiter (default: comma)'
)
parser.add_argument(
'--quotechar',
default='"',
help='CSV quote character (default: double quote)'
)
parser.add_argument(
'--encoding',
default='iso-8859-1',
help='CSV file encoding (default: iso-8859-1)'
)
parser.add_argument(
'--skip-errors',
action='store_true',
help='Skip rows with errors instead of stopping'
)
parser.add_argument(
'--verbose', '-v',
action='store_true',
help='Show detailed processing information'
)
return parser.parse_args()
def read_csv_file(filepath: str, encoding: str, delimiter: str, quotechar: str) -> List[Dict[str, str]]:
"""Read CSV file and return list of dictionaries."""
"""
Read a CSV file with specified encoding and return its contents as a list of dictionaries.
Args:
filepath: Path to the CSV file
encoding: File encoding (e.g., 'iso-8859-1')
delimiter: CSV field delimiter
quotechar: CSV quote character
Returns:
List of dictionaries where keys are header names and values are field values
"""
if not os.path.exists(filepath):
raise FileNotFoundError(f"CSV file not found: {filepath}")
if not os.path.isfile(filepath):
raise ValueError(f"Path is not a file: {filepath}")
rows = []
with open(filepath, 'r', encoding=encoding, newline='') as csvfile:
reader = csv.DictReader(csvfile, delimiter=delimiter, quotechar=quotechar)
# Convert header names to lowercase and strip whitespace
if reader.fieldnames:
reader.fieldnames = [name.strip().lower() for name in reader.fieldnames]
for row in reader:
rows.append(row)
return rows
def split_full_name(full_name: str) -> Tuple[str, str]:
"""
Split a full name in format 'APELLIDOS, NOMBRE' into last_name and first_name.
Args:
full_name: Full name string in format 'Last Name, First Name'
Returns:
Tuple of (last_name, first_name)
"""
if not full_name or full_name.strip() == '':
return '', ''
# Remove surrounding quotes if present
full_name = full_name.strip().strip('"').strip("'")
# Split on comma followed by optional whitespace
parts = re.split(r',\s*', full_name, maxsplit=1)
if len(parts) == 2:
last_name = parts[0].strip()
first_name = parts[1].strip()
else:
# If no comma, assume it's just a name
last_name = ''
first_name = full_name.strip()
return last_name, first_name
def convert_date(date_str: str) -> Optional[str]:
"""
Convert date from DD/MM/YYYY to YYYY-MM-DD format.
Args:
date_str: Date string in DD/MM/YYYY format
Returns:
Date string in YYYY-MM-DD format, or None if invalid
"""
if not date_str or date_str.strip() == '':
return None
date_str = date_str.strip()
try:
# Parse DD/MM/YYYY
day, month, year = re.split(r'[/-]', date_str)
day = int(day)
month = int(month)
year = int(year)
# Validate date
datetime(year, month, day)
# Format as YYYY-MM-DD
return f"{year:04d}-{month:02d}-{day:02d}"
except (ValueError, IndexError):
return None
def validate_gender(gender: str) -> Optional[str]:
"""
Validate and normalize gender field.
Args:
gender: Gender string (M, F, etc.)
Returns:
Normalized gender ('M', 'F', or 'O') or None if invalid
"""
if not gender or gender.strip() == '':
return None
gender = gender.strip().upper()
if gender in ['M', 'H', 'MALE', 'HOMBRE']:
return 'M'
elif gender in ['F', 'MUJER', 'FEMALE']:
return 'F'
elif gender in ['O', 'OTHER', 'OTRO']:
return 'O'
else:
return None
def process_row(row: Dict[str, str], verbose: bool = False) -> Optional[Dict[str, Optional[str]]]:
"""
Process a single CSV row and convert it to database format.
Args:
row: Dictionary with CSV field names as keys
verbose: Whether to print processing details
Returns:
Dictionary with processed data ready for database insertion, or None if invalid
"""
# Normalize the row keys to lowercase and remove special characters
normalized_row = {}
for key, value in row.items():
# Remove special characters and normalize
normalized_key = key.lower().replace('º', '').replace('ó', 'o').replace('á', 'a').replace('í', 'i').replace('ú', 'u').replace('ñ', 'n').replace('.', '').replace(',', '').strip()
normalized_row[normalized_key] = value
# Map CSV fields to our expected field names
mapped_row = {}
field_mapping = {
'n or': 'order_number',
'apellidos nombre': 'full_name',
'fecha nac': 'birth_date',
'cial': 'cial_code',
'nif/nie/pas': 'nif_nie_passport',
'registro': 'registration_number',
'expediente': 'file_number',
'sexo': 'gender',
'grupo': 'study_group',
'estudio': 'study'
}
for csv_key, our_key in field_mapping.items():
if csv_key in normalized_row:
mapped_row[our_key] = normalized_row[csv_key]
else:
mapped_row[our_key] = ''
# Extract and process fields
try:
# Order number
order_number = mapped_row.get('order_number', '')
if order_number.strip() == '':
order_number = None
else:
try:
order_number = int(order_number.strip())
except ValueError:
order_number = None
# Full name and split into last_name, first_name
full_name = mapped_row.get('full_name', '')
last_name, first_name = split_full_name(full_name)
# Birth date conversion
birth_date_str = mapped_row.get('birth_date', '')
birth_date = convert_date(birth_date_str)
# CIAL code
cial_code = mapped_row.get('cial_code', '').strip()
if cial_code == '':
cial_code = None
# NIF/NIE/Passport
nif_nie_passport = mapped_row.get('nif_nie_passport', '').strip()
if nif_nie_passport == '':
nif_nie_passport = None
# Registration number
registration_number = mapped_row.get('registration_number', '')
if registration_number.strip() == '':
registration_number = None
else:
try:
registration_number = int(registration_number.strip())
except ValueError:
registration_number = None
# File number
file_number = mapped_row.get('file_number', '').strip()
if file_number == '':
file_number = None
# Gender
gender = mapped_row.get('gender', '')
gender = validate_gender(gender)
# Study group
study_group = mapped_row.get('study_group', '').strip()
if study_group == '':
study_group = None
# Create result dictionary
result = {
'order_number': order_number,
'cial_code': cial_code,
'nif_nie_passport': nif_nie_passport,
'registration_number': registration_number,
'file_number': file_number,
'first_name': first_name if first_name else None,
'last_name': last_name if last_name else None,
'full_name': full_name.strip() if full_name.strip() else None,
'birth_date': birth_date,
'gender': gender,
'study_group': study_group,
'created_at': datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
'updated_at': datetime.now().strftime('%Y-%m-%d %H:%M:%S')
}
if verbose:
print(f" Processed: {result}")
return result
except Exception as e:
print(f" Error processing row: {e}")
return None
def validate_row(data: Dict[str, Optional[str]]) -> Tuple[bool, List[str]]:
"""
Validate processed row data.
Args:
data: Processed row data
Returns:
Tuple of (is_valid, list_of_errors)
"""
errors = []
# Check required fields
if not data.get('cial_code'):
errors.append("CIAL code is required")
if not data.get('first_name'):
errors.append("First name is required")
if not data.get('last_name'):
errors.append("Last name is required")
# Check date format
birth_date = data.get('birth_date')
if birth_date:
try:
datetime.strptime(birth_date, '%Y-%m-%d')
except ValueError:
errors.append(f"Invalid birth date format: {birth_date}")
# Check gender
gender = data.get('gender')
if gender and gender not in ['M', 'F', 'O']:
errors.append(f"Invalid gender: {gender}")
return len(errors) == 0, errors
def create_students_table(conn: sqlite3.Connection) -> bool:
"""
Create the students table if it doesn't exist.
Args:
conn: SQLite database connection
Returns:
True if table was created or already exists, False on error
"""
try:
cursor = conn.cursor()
cursor.execute('''
CREATE TABLE IF NOT EXISTS students (
id INTEGER PRIMARY KEY AUTOINCREMENT,
order_number INTEGER,
cial_code TEXT UNIQUE NOT NULL,
nif_nie_passport TEXT UNIQUE,
registration_number INTEGER,
file_number TEXT,
first_name TEXT NOT NULL,
last_name TEXT NOT NULL,
full_name TEXT NOT NULL,
birth_date TEXT,
gender TEXT CHECK(gender IN ('M', 'F', 'O')),
study_group TEXT,
created_at TEXT DEFAULT CURRENT_TIMESTAMP,
updated_at TEXT DEFAULT CURRENT_TIMESTAMP,
CONSTRAINT unique_identification UNIQUE (cial_code, nif_nie_passport)
)
''')
# Create indexes
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_cial ON students(cial_code)
''')
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_nif ON students(nif_nie_passport)
''')
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_name ON students(last_name, first_name)
''')
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_birth_date ON students(birth_date)
''')
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_gender ON students(gender)
''')
cursor.execute('''
CREATE INDEX IF NOT EXISTS idx_students_group ON students(study_group)
''')
conn.commit()
return True
except sqlite3.Error as e:
print(f"Error creating students table: {e}")
conn.rollback()
return False
def insert_student(conn: sqlite3.Connection, data: Dict[str, Optional[str]], verbose: bool = False) -> Tuple[bool, Optional[int]]:
"""
Insert a student record into the database.
Args:
conn: SQLite database connection
data: Student data dictionary
verbose: Whether to print details
Returns:
Tuple of (success, student_id or None)
"""
try:
cursor = conn.cursor()
# Prepare SQL and parameters
sql = '''
INSERT INTO students (
order_number, cial_code, nif_nie_passport, registration_number,
file_number, first_name, last_name, full_name, birth_date,
gender, study_group, created_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
'''
params = (
data['order_number'],
data['cial_code'],
data['nif_nie_passport'],
data['registration_number'],
data['file_number'],
data['first_name'],
data['last_name'],
data['full_name'],
data['birth_date'],
data['gender'],
data['study_group'],
data['created_at'],
data['updated_at']
)
cursor.execute(sql, params)
student_id = cursor.lastrowid
conn.commit()
if verbose:
print(f" Inserted student ID: {student_id}")
return True, student_id
except sqlite3.IntegrityError as e:
if verbose:
print(f" Duplicate entry (skipped): {e}")
conn.rollback()
return False, None
except sqlite3.Error as e:
print(f" Database error: {e}")
conn.rollback()
return False, None
def check_duplicate(conn: sqlite3.Connection, cial_code: str, nif_nie_passport: Optional[str]) -> bool:
"""
Check if a student with the same CIAL code or NIF/NIE already exists.
Args:
conn: SQLite database connection
cial_code: CIAL code to check
nif_nie_passport: NIF/NIE/Passport to check
Returns:
True if duplicate exists, False otherwise
"""
try:
cursor = conn.cursor()
# Check by CIAL code
cursor.execute("SELECT id FROM students WHERE cial_code = ?", (cial_code,))
if cursor.fetchone():
return True
# Check by NIF/NIE if provided
if nif_nie_passport and nif_nie_passport.strip():
cursor.execute("SELECT id FROM students WHERE nif_nie_passport = ?", (nif_nie_passport,))
if cursor.fetchone():
return True
return False
except sqlite3.Error as e:
print(f"Error checking for duplicates: {e}")
return False
def import_csv(csv_file: str, database: str, dry_run: bool = False,
skip_errors: bool = False, verbose: bool = False) -> Dict[str, int]:
"""
Main import function.
Args:
csv_file: Path to CSV file
database: Path to SQLite database
dry_run: If True, validate without importing
skip_errors: If True, skip invalid rows instead of stopping
verbose: If True, print detailed information
Returns:
Dictionary with import statistics
"""
stats = {
'total_rows': 0,
'processed': 0,
'valid': 0,
'invalid': 0,
'imported': 0,
'duplicates': 0,
'errors': 0
}
# Read CSV file
try:
if verbose:
print(f"Reading CSV file: {csv_file}")
rows = read_csv_file(csv_file, encoding='iso-8859-1', delimiter=',', quotechar='"')
stats['total_rows'] = len(rows)
if verbose:
print(f"Found {len(rows)} rows in CSV file")
except Exception as e:
print(f"Error reading CSV file: {e}")
stats['errors'] += 1
return stats
# Connect to database
try:
conn = sqlite3.connect(database)
conn.execute("PRAGMA foreign_keys = ON")
if not dry_run:
# Create students table if it doesn't exist
if not create_students_table(conn):
print("Failed to create students table")
stats['errors'] += 1
return stats
except sqlite3.Error as e:
print(f"Error connecting to database: {e}")
stats['errors'] += 1
return stats
# Process each row
for i, row in enumerate(rows, 1):
if verbose:
print(f"\nProcessing row {i}/{len(rows)}:")
print(f" Raw: {row}")
# Process row
processed = process_row(row, verbose)
if processed is None:
stats['errors'] += 1
if not skip_errors:
print(f"Error processing row {i}, stopping.")
break
else:
if verbose:
print(f" Skipping row {i} due to processing error")
continue
stats['processed'] += 1
# Validate row
is_valid, errors = validate_row(processed)
if not is_valid:
stats['invalid'] += 1
if verbose:
print(f" Invalid row {i}: {errors}")
if not skip_errors:
print(f"Error validating row {i}, stopping.")
break
else:
continue
stats['valid'] += 1
if dry_run:
# Just print what would be imported
if verbose:
print(f" [DRY RUN] Would import: {processed['full_name']} ({processed['cial_code']})")
else:
# Check for duplicates
if check_duplicate(conn, processed['cial_code'], processed.get('nif_nie_passport')):
stats['duplicates'] += 1
if verbose:
print(f" Duplicate found for CIAL: {processed['cial_code']}, skipping")
continue
# Insert into database
success, student_id = insert_student(conn, processed, verbose)
if success:
stats['imported'] += 1
else:
stats['duplicates'] += 1
# Close database connection
if not dry_run:
conn.close()
return stats
def print_statistics(stats: Dict[str, int], dry_run: bool = False):
"""Print import statistics."""
print("\n" + "=" * 50)
print("IMPORT STATISTICS")
print("=" * 50)
print(f"Total rows in CSV: {stats['total_rows']}")
print(f"Rows processed: {stats['processed']}")
print(f"Valid rows: {stats['valid']}")
print(f"Invalid rows: {stats['invalid']}")
if not dry_run:
print(f"Rows imported: {stats['imported']}")
print(f"Duplicates skipped: {stats['duplicates']}")
print(f"Errors: {stats['errors']}")
if dry_run:
print("\n[DRY RUN] No data was imported to the database.")
print("=" * 50)
def main():
"""Main entry point."""
args = parse_arguments()
print(f"Importing student data from: {args.csv_file}")
print(f"Database: {args.database}")
print(f"Dry run: {args.dry_run}")
print(f"Skip errors: {args.skip_errors}")
print()
# Perform import
stats = import_csv(
csv_file=args.csv_file,
database=args.database,
dry_run=args.dry_run,
skip_errors=args.skip_errors,
verbose=args.verbose
)
# Print statistics
print_statistics(stats, args.dry_run)
# Exit with appropriate code
if stats['errors'] > 0:
sys.exit(1)
elif stats['invalid'] > 0 and not args.skip_errors:
sys.exit(1)
else:
sys.exit(0)
if __name__ == '__main__':
main()