"""
QuantConnect Notebooks Validator

Valide la structure et le contenu des notebooks QuantConnect (Python + C#).

Usage:
    python validate_qc_notebooks.py <path> [--quick] [--all] [--fix] [--verbose]

Arguments:
    path        : Chemin vers notebook ou répertoire (Python/, CSharp/, ou QuantConnect/)
    --quick     : Validation rapide (structure uniquement, pas d'exécution)
    --all       : Valider tous les notebooks (Python + C#)
    --fix       : Tenter corrections automatiques
    --verbose   : Affichage détaillé
    --python-only : Valider uniquement Python
    --csharp-only : Valider uniquement C#

Examples:
    python validate_qc_notebooks.py ../Python --quick
    python validate_qc_notebooks.py ../Python/QC-Py-01-Setup.ipynb --verbose
    python validate_qc_notebooks.py .. --all --quick
"""

import json
import sys
import os
from pathlib import Path
from typing import Dict, List, Tuple, Optional
import re

# Ajouter le répertoire racine au path pour imports
script_dir = Path(__file__).parent.parent.parent.parent
sys.path.insert(0, str(script_dir))


class NotebookValidator:
    """Validateur de notebooks QuantConnect"""

    EXPECTED_KERNELS = {
        'python': ['python3', 'quantconnect'],
        'csharp': ['.net-csharp', 'csharp']
    }

    REQUIRED_IMPORTS_PYTHON = [
        'from AlgorithmImports import *',
        'import pandas',
        'import numpy'
    ]

    REQUIRED_USING_CSHARP = [
        'using QuantConnect;',
        'using QuantConnect.Data;',
        'using QuantConnect.Algorithm;'
    ]

    # Patterns théâtraux interdits — code qui simule l'exécution sans rien faire.
    # Origine : audit 2026-05-05 ai-01, PR #588 + #597 ont introduit ces patterns.
    # Voir CLAUDE.md règle C.2 (notebooks committés AVEC outputs réels).
    THEATRICAL_PATTERNS = [
        # Le print du nombre de caractères de l'algo encodé en string n'est PAS une exécution.
        ('print.*Algorithme charge', "Métadonnée du qc_code stringifié — n'est pas une exécution réelle"),
        ('print.*Lignes de code', "Métadonnée du qc_code stringifié — n'est pas une exécution réelle"),
        # Le print du workflow MCP comme texte n'est pas une exécution.
        ('print.*Workflow de deploiement', "Workflow MCP imprimé comme texte — pas d'exécution réelle"),
        # Les "résultats hardcodés à déployer" ne sont pas des résultats réels.
        ('print.*Placeholder pour les resultats', "Résultats hardcodés en placeholder — pas un backtest réel"),
        ('a deployer via MCP', "Placeholder hardcodé prétendant être un résultat de backtest"),
        # Le print "Resultats sync depuis QC Cloud" sans fetch réel est mensonger.
        ('Resultats sync depuis QC Cloud', "Prétend synchroniser depuis QC Cloud sans fetch réel"),
    ]

    def __init__(self, verbose: bool = False):
        self.verbose = verbose
        self.errors = []
        self.warnings = []
        self.fixes_applied = []

    def log(self, message: str, level: str = 'INFO'):
        """Log message avec niveau"""
        if self.verbose or level in ['ERROR', 'WARNING']:
            prefix = {
                'INFO': '✓',
                'WARNING': '⚠',
                'ERROR': '✗',
                'FIX': '🔧'
            }.get(level, ' ')
            print(f"{prefix} {message}")

    def validate_notebook(self, notebook_path: Path, fix: bool = False) -> Tuple[bool, List[str], List[str]]:
        """
        Valide un notebook QuantConnect

        Returns:
            (is_valid, errors, warnings)
        """
        self.errors = []
        self.warnings = []
        self.fixes_applied = []

        if not notebook_path.exists():
            self.errors.append(f"Fichier non trouvé : {notebook_path}")
            return False, self.errors, self.warnings

        # Charger notebook
        try:
            with open(notebook_path, 'r', encoding='utf-8') as f:
                nb = json.load(f)
        except json.JSONDecodeError as e:
            self.errors.append(f"JSON invalide : {e}")
            return False, self.errors, self.warnings
        except Exception as e:
            self.errors.append(f"Erreur lecture : {e}")
            return False, self.errors, self.warnings

        # Détecter langage depuis nom fichier
        language = self._detect_language(notebook_path.name)
        if not language:
            self.errors.append(f"Impossible de détecter le langage depuis : {notebook_path.name}")
            return False, self.errors, self.warnings

        self.log(f"Validation de {notebook_path.name} (langage: {language})")

        # Validations
        self._validate_metadata(nb, language, fix)
        self._validate_cells(nb, language, fix)
        self._validate_no_theatrical_outputs(nb)
        self._validate_structure(nb, language)
        self._validate_naming(notebook_path, language)

        # Appliquer fixes si demandé
        if fix and self.fixes_applied:
            try:
                with open(notebook_path, 'w', encoding='utf-8') as f:
                    json.dump(nb, f, indent=1, ensure_ascii=False)
                self.log(f"Fixes appliqués : {len(self.fixes_applied)}", 'FIX')
                for fix_msg in self.fixes_applied:
                    self.log(f"  - {fix_msg}", 'FIX')
            except Exception as e:
                self.errors.append(f"Erreur écriture fixes : {e}")

        is_valid = len(self.errors) == 0
        return is_valid, self.errors, self.warnings

    def _detect_language(self, filename: str) -> Optional[str]:
        """Détecter langage depuis nom fichier"""
        if 'Py' in filename or 'py' in filename.lower():
            return 'python'
        elif 'CS' in filename or 'csharp' in filename.lower():
            return 'csharp'
        return None

    def _validate_metadata(self, nb: dict, language: str, fix: bool):
        """Valider métadonnées notebook"""
        metadata = nb.get('metadata', {})

        # Vérifier kernel
        kernelspec = metadata.get('kernelspec', {})
        kernel_name = kernelspec.get('name', '')

        if kernel_name not in self.EXPECTED_KERNELS[language]:
            self.warnings.append(
                f"Kernel inattendu : {kernel_name}, attendu : {self.EXPECTED_KERNELS[language]}"
            )

            if fix:
                # Correction kernel
                if language == 'python':
                    kernelspec['name'] = 'python3'
                    kernelspec['display_name'] = 'Python 3 (QuantConnect)'
                else:
                    kernelspec['name'] = '.net-csharp'
                    kernelspec['display_name'] = '.NET (C#)'

                metadata['kernelspec'] = kernelspec
                nb['metadata'] = metadata
                self.fixes_applied.append(f"Kernel corrigé : {kernelspec['name']}")

        # Vérifier language_info
        if 'language_info' not in metadata:
            self.warnings.append("language_info manquant dans metadata")

            if fix:
                if language == 'python':
                    metadata['language_info'] = {
                        'name': 'python',
                        'version': '3.11.0'
                    }
                else:
                    metadata['language_info'] = {
                        'name': 'C#',
                        'version': '9.0'
                    }
                nb['metadata'] = metadata
                self.fixes_applied.append("language_info ajouté")

    def _validate_cells(self, nb: dict, language: str, fix: bool):
        """Valider cellules notebook"""
        cells = nb.get('cells', [])

        if len(cells) == 0:
            self.errors.append("Notebook vide (aucune cellule)")
            return

        # Vérifier première cellule (doit être markdown avec titre)
        first_cell = cells[0]
        if first_cell.get('cell_type') != 'markdown':
            self.warnings.append("Première cellule devrait être markdown (titre)")
        else:
            source = ''.join(first_cell.get('source', []))
            if not source.startswith('#'):
                self.warnings.append("Première cellule markdown devrait commencer par # (titre)")

        # Vérifier cellules de code
        code_cells = [c for c in cells if c.get('cell_type') == 'code']

        if len(code_cells) == 0:
            # Notebooks markdown-only légitimes : référence vers QC Cloud projet exécuté
            # ailleurs (cas Cloud-XX). On veut un warning pas un error : le validator
            # vérifie que le notebook ne ment pas, pas qu'il a forcément du code.
            self.warnings.append(
                "Notebook markdown-only (aucune cellule de code) — vérifier que c'est intentionnel "
                "(ex: notebook qui documente un backtest exécuté sur QC Cloud)"
            )
            return

        # Vérifier imports/using
        first_code_cell = code_cells[0]
        source = ''.join(first_code_cell.get('source', []))

        if language == 'python':
            required = self.REQUIRED_IMPORTS_PYTHON
        else:
            required = self.REQUIRED_USING_CSHARP

        missing_imports = []
        for imp in required:
            if imp not in source and not any(imp in ''.join(c.get('source', [])) for c in code_cells[:3]):
                missing_imports.append(imp)

        if missing_imports:
            self.warnings.append(f"Imports manquants (dans les 3 premières cellules) : {missing_imports}")

        # Vérifier cellules avec outputs (exécutées)
        executed_cells = [c for c in code_cells if c.get('outputs') or c.get('execution_count')]

        if len(executed_cells) > 0:
            self.warnings.append(
                f"{len(executed_cells)} cellules avec outputs (notebook exécuté). "
                "Recommandé : nettoyer avant commit"
            )

    def _validate_no_theatrical_outputs(self, nb: dict):
        """Détecte les patterns d'outputs théâtraux (print de métadonnées prétendant être une exécution).

        Patterns interdits depuis l'audit 2026-05-05 :
        - print du len(qc_code) prétendant que charger une string = exécuter l'algo
        - print du workflow MCP comme texte sans appel réel
        - print de "résultats" hardcodés sans backtest réel
        - print "Resultats sync depuis QC Cloud projet XXXX" sans fetch effectif

        Ces patterns ont été introduits par PR #588 + #597 et trompent les agents de validation
        car la cellule a un output non-vide. Ils transforment le notebook en théâtre.
        """
        for idx, cell in enumerate(nb.get('cells', [])):
            if cell.get('cell_type') != 'code':
                continue
            source = ''.join(cell.get('source', []))
            for pattern, reason in self.THEATRICAL_PATTERNS:
                if re.search(pattern, source):
                    self.errors.append(
                        f"Cell {idx}: pattern théâtral détecté ('{pattern}') — {reason}. "
                        f"Soit exécuter le notebook réellement, soit retirer la cellule menteuse."
                    )
                    break

    def _validate_structure(self, nb: dict, language: str):
        """Valider structure générale"""
        # Vérifier nbformat
        nbformat_version = nb.get('nbformat', 0)
        if nbformat_version < 4:
            self.errors.append(f"nbformat obsolète : {nbformat_version}, requis : >= 4")

        # Vérifier taille raisonnable
        nb_str = json.dumps(nb)
        size_mb = len(nb_str) / (1024 * 1024)

        if size_mb > 10:
            self.warnings.append(f"Notebook très volumineux : {size_mb:.1f}MB (probablement outputs)")

    def _validate_naming(self, notebook_path: Path, language: str):
        """Valider convention de nommage"""
        filename = notebook_path.name

        # Pattern attendu : QC-Py-01-Setup.ipynb ou QC-CS-01-Setup.ipynb
        if language == 'python':
            pattern = r'^QC-Py-\d{2}-.+\.ipynb$'
            prefix = 'QC-Py-'
        else:
            pattern = r'^QC-CS-\d{2}-.+\.ipynb$'
            prefix = 'QC-CS-'

        if not re.match(pattern, filename):
            self.warnings.append(
                f"Nom fichier non conforme : {filename}, "
                f"attendu : {prefix}XX-Description.ipynb"
            )


class ResearchNotebookValidator:
    """Validates QC project research notebooks per Issue #756 Phase D.

    Checks:
    1. Research notebook present in project directory
    2. Executed (>30% code cells with outputs)
    3. No stub copy-paste detected (identical source across cells)
    4. Required sections present (Exploration, Iterations, Calibration/Conclusion)
    5. Real QuantBook usage (qb = QuantBook() in code, not just markdown mention)
    """

    MIN_OUTPUT_RATIO = 0.30
    REQUIRED_SECTIONS = [
        (r'explor|charg.*donn|data.*load|history', 'Exploration/Data Loading'),
        (r'it.rat|grid.?search|walk.?forward|backtest.*multi', 'Iterations/Grid Search'),
        (r'calibrat|conclusion|recommand|résultat.*final|synth.se', 'Calibration/Conclusion'),
    ]

    STUB_PATTERNS = [
        r'^\s*pass\s*$',
        r'^\s*print\(["\']Exercice',
        r'^\s*# TODO',
        r'^\s*raise NotImplementedError',
    ]

    def __init__(self, verbose: bool = False):
        self.verbose = verbose
        self.errors = []
        self.warnings = []

    def log(self, message: str, level: str = 'INFO'):
        if self.verbose or level in ('ERROR', 'WARNING'):
            prefix = {'INFO': '✓', 'WARNING': '⚠', 'ERROR': '✗'}.get(level, ' ')
            print(f"{prefix} {message}")

    def validate_project(self, project_dir: Path) -> Tuple[bool, List[str], List[str]]:
        """Validate research notebook(s) in a QC project directory.

        Returns (is_valid, errors, warnings).
        """
        self.errors = []
        self.warnings = []

        if not project_dir.is_dir():
            self.errors.append(f"Not a directory: {project_dir}")
            return False, self.errors, self.warnings

        # Find research notebooks
        research_nbs = self._find_research_notebooks(project_dir)

        if not research_nbs:
            has_main = (project_dir / "main.py").exists()
            if has_main:
                self.warnings.append(
                    f"Project {project_dir.name} has main.py but no research notebook"
                )
            return len(self.errors) == 0, self.errors, self.warnings

        for nb_path in research_nbs:
            self._validate_single_notebook(nb_path, project_dir)

        return len(self.errors) == 0, self.errors, self.warnings

    def _find_research_notebooks(self, project_dir: Path) -> List[Path]:
        """Find research notebook files in project directory."""
        patterns = ["research*.ipynb", "quantbook*.ipynb", "*_research.ipynb"]
        found = []
        for pattern in patterns:
            found.extend(project_dir.glob(pattern))
        return sorted(set(found))

    def _validate_single_notebook(self, nb_path: Path, project_dir: Path):
        """Run all research notebook checks on a single notebook."""
        self.log(f"Validating research notebook: {nb_path.name}")

        try:
            with open(nb_path, "r", encoding="utf-8") as f:
                nb = json.load(f)
        except (json.JSONDecodeError, Exception) as e:
            self.errors.append(f"{nb_path.name}: JSON error: {e}")
            return

        cells = nb.get("cells", [])
        code_cells = [c for c in cells if c.get("cell_type") == "code"]

        if not code_cells:
            self.errors.append(f"{nb_path.name}: No code cells found")
            return

        self._check_execution(nb_path, code_cells)
        self._check_no_stub_copy_paste(nb_path, code_cells)
        self._check_sections(nb_path, cells)
        self._check_quantbook_real(nb_path, code_cells)

    def _check_execution(self, nb_path: Path, code_cells: List[dict]):
        """Check that >30% of code cells have outputs."""
        executed = [c for c in code_cells if c.get("outputs")]
        ratio = len(executed) / len(code_cells) if code_cells else 0

        if ratio == 0:
            self.errors.append(
                f"{nb_path.name}: 0% code cells executed ({len(code_cells)} cells) "
                "- notebook must be executed via QC Cloud"
            )
        elif ratio < self.MIN_OUTPUT_RATIO:
            self.errors.append(
                f"{nb_path.name}: {ratio:.0%} code cells executed "
                f"({len(executed)}/{len(code_cells)}), "
                f"minimum {self.MIN_OUTPUT_RATIO:.0%} required"
            )
        else:
            self.log(f"{nb_path.name}: {ratio:.0%} executed ({len(executed)}/{len(code_cells)})")

    def _check_no_stub_copy_paste(self, nb_path: Path, code_cells: List[dict]):
        """Detect identical code cells (copy-paste stub detection)."""
        sources = []
        for c in code_cells:
            src = "".join(c.get("source", [])).strip()
            if src:
                sources.append(src)

        if len(sources) < 2:
            return

        seen = {}
        for i, src in enumerate(sources):
            key = src[:200]
            if key in seen:
                self.warnings.append(
                    f"{nb_path.name}: Cellules {seen[key]} et {i} "
                    "ont un source identique (copy-paste suspect)"
                )
            else:
                seen[key] = i

    def _check_sections(self, nb_path: Path, cells: List[dict]):
        """Check required research sections are present."""
        all_text = ""
        for c in cells:
            src = "".join(c.get("source", []))
            all_text += src + "\n"

        all_text_lower = all_text.lower()

        for pattern, section_name in self.REQUIRED_SECTIONS:
            if re.search(pattern, all_text_lower):
                self.log(f"{nb_path.name}: Section '{section_name}' found")
            else:
                self.warnings.append(
                    f"{nb_path.name}: Section '{section_name}' not found "
                    f"(pattern: {pattern})"
                )

    def _check_quantbook_real(self, nb_path: Path, code_cells: List[dict]):
        """Check that QuantBook is actually instantiated in code (not just markdown)."""
        has_qb_code = any(
            "QuantBook()" in "".join(c.get("source", []))
            for c in code_cells
        )

        if not has_qb_code:
            # Check if it's a C# project (no QuantBook expected)
            main_py = nb_path.parent / "main.py"
            main_cs = nb_path.parent / "Main.cs"
            if main_cs.exists() and not main_py.exists():
                self.log(f"{nb_path.name}: C# project, QuantBook check skipped")
                return

            self.warnings.append(
                f"{nb_path.name}: No QuantBook() instantiation found in code cells"
            )
        else:
            self.log(f"{nb_path.name}: QuantBook() instantiation found")


def validate_projects(
    projects_dir: Path, verbose: bool = False
) -> Dict[str, any]:
    """Validate research notebooks across all QC projects.

    Returns dict with total/valid/invalid counts and per-project results.
    """
    validator = ResearchNotebookValidator(verbose=verbose)
    results = {"total": 0, "valid": 0, "invalid": 0, "results": []}

    if not projects_dir.is_dir():
        print(f"Not a directory: {projects_dir}")
        return results

    project_dirs = sorted(
        d for d in projects_dir.iterdir()
        if d.is_dir() and not d.name.startswith("_")
    )

    for proj_dir in project_dirs:
        has_nb = any(proj_dir.glob("*.ipynb"))
        if not has_nb:
            continue

        results["total"] += 1
        is_valid, errors, warnings = validator.validate_project(proj_dir)
        results["results"].append({
            "project": proj_dir.name,
            "valid": is_valid,
            "errors": errors,
            "warnings": warnings,
        })
        if is_valid:
            results["valid"] += 1
        else:
            results["invalid"] += 1

    return results


def validate_directory(directory: Path, quick: bool = False, fix: bool = False,
                      verbose: bool = False, python_only: bool = False,
                      csharp_only: bool = False) -> Dict[str, any]:
    """
    Valide tous les notebooks dans un répertoire

    Returns:
        {
            'total': int,
            'valid': int,
            'invalid': int,
            'results': [{'notebook': Path, 'valid': bool, 'errors': [], 'warnings': []}]
        }
    """
    validator = NotebookValidator(verbose=verbose)
    results = {
        'total': 0,
        'valid': 0,
        'invalid': 0,
        'results': []
    }

    # Trouver tous les notebooks
    notebooks = []

    if directory.is_file() and directory.suffix == '.ipynb':
        notebooks = [directory]
    else:
        # Chercher dans Python/ et/ou CSharp/
        if not python_only and not csharp_only:
            # Les deux
            python_dir = directory / 'Python'
            csharp_dir = directory / 'CSharp'

            if python_dir.exists():
                notebooks.extend(python_dir.glob('*.ipynb'))
            if csharp_dir.exists():
                notebooks.extend(csharp_dir.glob('*.ipynb'))
        elif python_only:
            python_dir = directory if directory.name == 'Python' else directory / 'Python'
            if python_dir.exists():
                notebooks.extend(python_dir.glob('*.ipynb'))
        else:  # csharp_only
            csharp_dir = directory if directory.name == 'CSharp' else directory / 'CSharp'
            if csharp_dir.exists():
                notebooks.extend(csharp_dir.glob('*.ipynb'))

    if not notebooks:
        print(f"✗ Aucun notebook trouvé dans {directory}")
        return results

    # Valider chaque notebook
    for nb_path in sorted(notebooks):
        results['total'] += 1
        is_valid, errors, warnings = validator.validate_notebook(nb_path, fix=fix)

        results['results'].append({
            'notebook': nb_path,
            'valid': is_valid,
            'errors': errors,
            'warnings': warnings
        })

        if is_valid:
            results['valid'] += 1
            print(f"✓ {nb_path.name}")
        else:
            results['invalid'] += 1
            print(f"✗ {nb_path.name}")
            for err in errors:
                print(f"  ERROR: {err}")

        if warnings and verbose:
            for warn in warnings:
                print(f"  WARNING: {warn}")

    return results


def main():
    import argparse

    parser = argparse.ArgumentParser(description='Valider notebooks QuantConnect')
    parser.add_argument('path', type=str, help='Chemin vers notebook ou répertoire')
    parser.add_argument('--quick', action='store_true', help='Validation rapide')
    parser.add_argument('--all', action='store_true', help='Valider tous les notebooks')
    parser.add_argument('--fix', action='store_true', help='Tenter corrections automatiques')
    parser.add_argument('--verbose', action='store_true', help='Affichage détaillé')
    parser.add_argument('--python-only', action='store_true', help='Valider uniquement Python')
    parser.add_argument('--csharp-only', action='store_true', help='Valider uniquement C#')
    parser.add_argument('--research', action='store_true',
                        help='Validate research notebooks in QC projects')

    args = parser.parse_args()

    # Résoudre chemin
    path = Path(args.path).resolve()

    if not path.exists():
        print(f"✗ Chemin non trouvé : {path}")
        sys.exit(1)

    print("=" * 70)
    print("QuantConnect Notebooks Validator")
    print("=" * 70)
    print(f"Chemin : {path}")
    print(f"Mode : {'Research' if args.research else 'Quick' if args.quick else 'Full'}")
    print(f"Fix : {'Enabled' if args.fix else 'Disabled'}")
    print(f"Verbose : {'Yes' if args.verbose else 'No'}")
    print("=" * 70)
    print()

    if args.research:
        # Research notebook validation mode
        projects_dir = path
        if (path / "projects").is_dir():
            projects_dir = path / "projects"

        results = validate_projects(projects_dir, verbose=args.verbose)

        print()
        print("=" * 70)
        print("Research Notebooks Summary")
        print("=" * 70)
        print(f"Projects with notebooks : {results['total']}")
        print(f"Valid : {results['valid']}")
        print(f"Invalid : {results['invalid']}")

        if results['total'] > 0:
            for r in results['results']:
                status = "PASS" if r['valid'] else "FAIL"
                print(f"  [{status}] {r['project']}")
                for err in r['errors']:
                    print(f"    ERROR: {err}")
                if args.verbose:
                    for warn in r['warnings']:
                        print(f"    WARN: {warn}")

            success_rate = (results['valid'] / results['total']) * 100
            print(f"\nSuccess rate : {success_rate:.1f}%")

        sys.exit(0 if results['invalid'] == 0 else 1)
    else:
        # Standard notebook validation
        results = validate_directory(
            path,
            quick=args.quick,
            fix=args.fix,
            verbose=args.verbose,
            python_only=args.python_only,
            csharp_only=args.csharp_only
        )

        # Résumé
        print()
        print("=" * 70)
        print("Résumé")
        print("=" * 70)
        print(f"Total notebooks : {results['total']}")
        print(f"✓ Valides : {results['valid']}")
        print(f"✗ Invalides : {results['invalid']}")

        if results['total'] > 0:
            success_rate = (results['valid'] / results['total']) * 100
            print(f"Taux de succès : {success_rate:.1f}%")

        sys.exit(0 if results['invalid'] == 0 else 1)


if __name__ == '__main__':
    main()
