import os
import re
import time
from datetime import datetime
from typing import Dict, List, Optional, Tuple, Any
import db

SUPPORTED_EXTENSIONS = {".jpg", ".jpeg", ".png", ".pdf", ".tif", ".tiff"}

def parse_date_from_string(text: str) -> Optional[str]:
    """Extrae una fecha en formato YYYY-MM-DD de una cadena (YYYYMMDD, YYYY-MM-DD, YYYY_MM_DD)"""
    # Formato YYYYMMDD (ej: 20260904)
    m = re.search(r'\b(20\d{2})(0[1-9]|1[0-2])(0[1-9]|[12]\d|3[01])\b', text)
    if m:
        return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
    
    # Formato YYYY-MM-DD o YYYY_MM_DD
    m = re.search(r'\b(20\d{2})[-_](0[1-9]|1[0-2])[-_](0[1-9]|[12]\d|3[01])\b', text)
    if m:
        return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
    
    return None

def normalize_doc_type(type_raw: str) -> str:
    """Normaliza el nombre del tipo documental en formato legible Capitalizado"""
    t = type_raw.strip().lower()
    if t in ("factura", "facturas"):
        return "Facturas"
    if t in ("contrato", "contratos"):
        return "Contratos"
    if t in ("orden", "ordenes", "órdenes"):
        return "Ordenes"
    if t in ("certificado", "certificados"):
        return "Certificados"
    if t in ("solicitud", "solicitudes"):
        return "Solicitudes"
    if t in ("pagare", "pagares", "pagarés"):
        return "Pagarés"
    
    # Capitalizar cualquier otro tipo documental futuro
    return type_raw.strip().title()

def extract_metadata_from_path(rel_path: str, full_path: str) -> Dict[str, Any]:
    """
    Extrae metadatos a partir de la ruta relativa y las propiedades del archivo.
    Soporta estructuras:
    - {documentosdigitales}/{tipo_documento}/{YYYYMMDD}/{cliente_id}.jpg
    - {tipo_documento}/{YYYYMMDD}/{cliente_id}.jpg
    - {tipo_documento}/{cliente_id}.jpg
    - {cableoperador}/{agencia}/{YYYYMMDD}/{cliente_id}.jpg
    - {YYYYMMDD}/{cliente_id}.jpg
    """
    normalized_rel = rel_path.replace("\\", "/")
    parts = [p for p in normalized_rel.split("/") if p]
    filename = parts[-1] if parts else os.path.basename(full_path)
    stem, ext = os.path.splitext(filename)

    # 1. Extraer ID del cliente: primer bloque numérico en el stem del archivo (ej: 625036)
    customer_id = stem
    num_match = re.search(r'(\d+)', stem)
    if num_match:
        customer_id = num_match.group(1)

    # 2. Extraer Tipo Documental
    # Si 'documentosdigitales' está en la ruta, el siguiente nivel es el tipo documental
    doc_type = "Facturas"
    normalized_full = full_path.replace("\\", "/")
    full_parts = [p for p in normalized_full.split("/") if p]

    found_type = None
    if "documentosdigitales" in parts:
        idx = parts.index("documentosdigitales")
        if idx + 1 < len(parts) - 1:
            found_type = parts[idx + 1]
    elif "documentosdigitales" in full_parts:
        idx = full_parts.index("documentosdigitales")
        if idx + 1 < len(full_parts) - 1:
            found_type = full_parts[idx + 1]
    
    # Si no se encontró mediante 'documentosdigitales', buscar en los directorios previos que no sean fecha
    if not found_type and len(parts) > 1:
        for p in parts[:-1]:
            if not parse_date_from_string(p) and not p.isdigit() and p.lower() not in ("storage", "facturas_1", "facturas_2", "facturas_3", "facturas_4", "facturas_5", "facturas_6", "facturas_7", "facturas_8", "facturas_9"):
                found_type = p
                break

    if found_type:
        doc_type = normalize_doc_type(found_type)

    # 3. Extraer Fecha del documento (inspeccionando carpetas padre y luego el nombre del archivo)
    doc_date = None
    for part in reversed(parts[:-1]):
        doc_date = parse_date_from_string(part)
        if doc_date:
            break
    
    if not doc_date:
        doc_date = parse_date_from_string(stem)

    # Si no se encuentra fecha en el path, usar la fecha de modificación del archivo
    file_stat = os.stat(full_path)
    file_mtime = file_stat.st_mtime
    file_size = file_stat.st_size

    if not doc_date:
        doc_date = datetime.fromtimestamp(file_mtime).strftime("%Y-%m-%d")

    # 4. Año y Mes
    date_obj = datetime.strptime(doc_date, "%Y-%m-%d")
    year = date_obj.year
    month = date_obj.month

    # 5. Cableoperador / Empresa
    cableoperador = ""
    agencia = ""

    # Detección de empresa a partir del path (ej: cableexito, infobox)
    for p in full_parts:
        low = p.lower()
        if low in ("cableexito", "cableéxito"):
            cableoperador = "Cableéxito"
            break
        elif low in ("infobox"):
            cableoperador = "Infobox"
            break
        elif low not in ("var", "www", "html", "devs", "documentosarchivo", "documentosdigitales", "storage", "app", "c:", "d:"):
            if "documentosdigitales" in full_parts and full_parts.index(p) == full_parts.index("documentosdigitales") - 1:
                cableoperador = p.title()
                break

    if not cableoperador and len(parts) >= 3:
        if parts[0].lower() not in ("contratos", "facturas", "ordenes", "certificados", "documentosdigitales"):
            cableoperador = parts[0].title()

    return {
        "customer_id": str(customer_id).strip(),
        "document_date": doc_date,
        "year": year,
        "month": month,
        "document_type": doc_type,
        "cableoperador": cableoperador,
        "agencia": agencia,
        "file_name": filename,
        "file_relative_path": normalized_rel,
        "file_full_path": os.path.abspath(full_path),
        "file_size": file_size,
        "file_mtime": file_mtime
    }

def get_configured_storage_paths() -> List[str]:
    """Obtiene la lista de hasta 9 rutas de almacenamiento configuradas desde variables de entorno"""
    paths: List[str] = []

    # 1. Variable STORAGE_PATHS (lista separada por coma o punto y coma)
    raw_paths = os.getenv("STORAGE_PATHS", "")
    if raw_paths:
        for p in re.split(r'[,;]', raw_paths):
            p = p.strip()
            if p and p not in paths:
                paths.append(p)

    # 2. Variables individuales STORAGE_PATH_1 .. STORAGE_PATH_9
    for i in range(1, 10):
        var_name = f"STORAGE_PATH_{i}"
        val = os.getenv(var_name)
        if val and val.strip() and val.strip() not in paths:
            paths.append(val.strip())

    # 3. Variables de Host HOST_INVOICES_PATH_1 .. HOST_INVOICES_PATH_9 (útil si corre local fuera de Docker)
    for i in range(1, 10):
        var_name = f"HOST_INVOICES_PATH_{i}"
        val = os.getenv(var_name)
        if val and val.strip() and val.strip() not in paths and os.path.exists(val.strip()):
            paths.append(val.strip())

    # 4. Fallbacks
    if not paths:
        legacy = os.getenv("STORAGE_PATH") or os.getenv("HOST_INVOICES_PATH")
        if legacy and legacy.strip():
            paths.append(legacy.strip())

    if not paths:
        default_test = os.path.join(os.path.dirname(__file__), "test_facturas")
        paths.append(default_test)

    # Filtrar solo rutas que no estén vacías
    return [p for p in paths if p]

def scan_storage_directory(storage_paths: Optional[Any] = None, batch_size: int = 500) -> Dict[str, Any]:
    """
    Realiza un escaneo incremental de hasta 9 carpetas de almacenamiento multi-documento y actualiza la base de datos SQLite.
    """
    if storage_paths is None:
        paths = get_configured_storage_paths()
    elif isinstance(storage_paths, str):
        paths = [p.strip() for p in re.split(r'[,;]', storage_paths) if p.strip()]
    elif isinstance(storage_paths, (list, tuple)):
        paths = list(storage_paths)
    else:
        paths = get_configured_storage_paths()

    start_time = time.time()
    db.init_db()
    indexed_mtimes = db.get_indexed_mtimes()

    scanned_count = 0
    updated_count = 0
    skipped_count = 0
    batch: List[Dict[str, Any]] = []
    scanned_dirs: List[str] = []

    for idx, storage_path in enumerate(paths, start=1):
        if not storage_path or not os.path.exists(storage_path):
            continue

        scanned_dirs.append(storage_path)
        dir_prefix = f"dir_{idx}"

        for root, _, files in os.walk(storage_path):
            for f in files:
                ext = os.path.splitext(f)[1].lower()
                if ext not in SUPPORTED_EXTENSIONS:
                    continue

                scanned_count += 1
                full_path = os.path.join(root, f)
                rel_path = os.path.relpath(full_path, storage_path).replace("\\", "/")
                
                # Identificador único con prefijo de directorio para evitar colisiones entre carpetas
                unique_rel_key = f"[{dir_prefix}]/{rel_path}" if len(paths) > 1 else rel_path

                try:
                    mtime = os.path.getmtime(full_path)
                except OSError:
                    continue

                # Verificación incremental: Si ya existe en DB y no ha cambiado el mtime, omitir
                if unique_rel_key in indexed_mtimes and abs(indexed_mtimes[unique_rel_key] - mtime) < 0.001:
                    skipped_count += 1
                    continue

                # Parsear metadatos y agregar al lote
                try:
                    metadata = extract_metadata_from_path(rel_path, full_path)
                    metadata["file_relative_path"] = unique_rel_key
                    batch.append(metadata)
                except Exception as e:
                    print(f"[INDEXER] Error al procesar archivo {full_path}: {e}")
                    continue

                if len(batch) >= batch_size:
                    updated_count += db.upsert_invoices_batch(batch)
                    batch.clear()

    # Procesar remanente
    if batch:
        updated_count += db.upsert_invoices_batch(batch)
        batch.clear()

    elapsed = round(time.time() - start_time, 3)
    db.set_meta("last_scan_time", datetime.now().strftime("%Y-%m-%d %H:%M:%S"))
    db.set_meta("last_scan_count", str(updated_count))
    db.set_meta("total_scanned_files", str(scanned_count))
    db.set_meta("storage_paths_active", ",".join(scanned_dirs))

    print(f"[INDEXER] Escaneo completado en {elapsed}s sobre {len(scanned_dirs)} directorio(s): {scanned_count} archivos escaneados, {updated_count} indexados/actualizados, {skipped_count} sin cambios.")
    
    return {
        "scanned": scanned_count,
        "added_or_updated": updated_count,
        "skipped": skipped_count,
        "elapsed_seconds": elapsed,
        "directories_scanned": scanned_dirs,
        "stats": db.get_stats(),
        "status": "success"
    }

if __name__ == "__main__":
    scan_storage_directory()
