#!/usr/bin/env python3
"""Prepare zero-copy AI MD power cases from Ocean FTP directories.

This helper maps a user-facing FTP path like:
    /UL/log/...
to the server-local path:
    /mnt/vsftp/test/UL/log/...

It scans for directories that contain paired power CSV and raw mdlog ZIP files,
creates one symlinked case workspace per pair, and writes a manifest with the
exact command needed by skill_code_ai_md_power.
"""

from __future__ import annotations

import argparse
import csv
import json
import os
import re
import shutil
import sys
from dataclasses import dataclass
from datetime import datetime
from pathlib import Path
from typing import Dict, List, Optional, Tuple


SERVER_FTP_ROOT = Path("/mnt/vsftp/test")
CSV_PATTERN = re.compile(
    r"(?P<prefix>.+?)_(?P<date>\d{6})_(?P<time>\d{6})_(?P<band>[A-Za-z0-9]+)_(?P<atten>[A-Za-z0-9]+)_(?P<rate>[A-Za-z0-9]+)\.csv$",
    re.IGNORECASE,
)
ZIP_PATTERN = re.compile(
    r"log_(?P<date>\d{8})_(?P<time>\d{6})_(?P<band>[A-Za-z0-9]+)_(?P<atten>[A-Za-z0-9]+)_(?P<rate>[A-Za-z0-9]+)\.zip$",
    re.IGNORECASE,
)


@dataclass(frozen=True)
class FileMeta:
    path: Path
    band: str
    atten: str
    rate: str
    timestamp: datetime


@dataclass(frozen=True)
class CasePair:
    source_dir: Path
    csv_path: Path
    zip_path: Path
    case_name: str
    relative_group: Path


def _normalize_rate(raw_value: str) -> str:
    text = str(raw_value).strip().lower()
    if text.endswith("m"):
        text = text[:-1]
    return text


def _sanitize_name(text: str) -> str:
    sanitized = re.sub(r"[^0-9A-Za-z._-]+", "_", text.strip())
    sanitized = re.sub(r"_+", "_", sanitized).strip("._")
    return sanitized or "case"


def map_ftp_path(user_path: str) -> Path:
    raw = str(user_path).strip()
    if not raw:
        raise ValueError("empty ftp path")

    path = Path(raw)
    if str(path).startswith(str(SERVER_FTP_ROOT)):
        return path

    if raw.startswith("/UL/") or raw.startswith("/log/"):
        return SERVER_FTP_ROOT / raw.lstrip("/")

    raise ValueError(
        "unsupported path. Expected '/UL/log/...', '/log/...', or '/mnt/vsftp/test/...': "
        f"{raw}"
    )


def parse_csv_meta(path: Path) -> Optional[FileMeta]:
    match = CSV_PATTERN.match(path.name)
    if not match:
        return None
    groups = match.groupdict()
    timestamp = datetime.strptime(groups["date"] + groups["time"], "%y%m%d%H%M%S")
    return FileMeta(
        path=path,
        band=groups["band"].lower(),
        atten=groups["atten"].lower(),
        rate=_normalize_rate(groups["rate"]),
        timestamp=timestamp,
    )


def parse_zip_meta(path: Path) -> Optional[FileMeta]:
    match = ZIP_PATTERN.match(path.name)
    if not match:
        return None
    groups = match.groupdict()
    timestamp = datetime.strptime(groups["date"] + groups["time"], "%Y%m%d%H%M%S")
    return FileMeta(
        path=path,
        band=groups["band"].lower(),
        atten=groups["atten"].lower(),
        rate=_normalize_rate(groups["rate"]),
        timestamp=timestamp,
    )


def discover_candidate_dirs(source_root: Path) -> List[Path]:
    candidate_dirs: List[Path] = []
    for root, _, files in os.walk(source_root):
        has_csv = any(name.lower().endswith(".csv") for name in files)
        has_zip = any(name.lower().endswith(".zip") for name in files)
        if has_csv and has_zip:
            candidate_dirs.append(Path(root))
    return sorted(candidate_dirs)


def _pair_key(meta: FileMeta) -> Tuple[str, str, str]:
    return meta.band, meta.atten, meta.rate


def pair_case_files(source_root: Path, candidate_dir: Path) -> Tuple[List[CasePair], List[Path], List[Path]]:
    csv_files = sorted(path for path in candidate_dir.iterdir() if path.is_file() and path.suffix.lower() == ".csv")
    zip_files = sorted(path for path in candidate_dir.iterdir() if path.is_file() and path.suffix.lower() == ".zip")

    csv_meta_list = [(path, parse_csv_meta(path)) for path in csv_files]
    zip_meta_list = [(path, parse_zip_meta(path)) for path in zip_files]

    zip_pool: Dict[Path, FileMeta] = {
        path: meta for path, meta in zip_meta_list if meta is not None
    }
    paired: List[CasePair] = []
    unmatched_csv: List[Path] = []

    for csv_path, csv_meta in csv_meta_list:
        if csv_meta is None:
            unmatched_csv.append(csv_path)
            continue

        compatible = [
            (zip_path, zip_meta)
            for zip_path, zip_meta in zip_pool.items()
            if _pair_key(zip_meta) == _pair_key(csv_meta)
        ]
        if not compatible:
            unmatched_csv.append(csv_path)
            continue

        compatible.sort(
            key=lambda item: (
                abs((item[1].timestamp - csv_meta.timestamp).total_seconds()),
                item[1].timestamp,
                item[0].name,
            )
        )
        zip_path, _ = compatible[0]
        del zip_pool[zip_path]

        relative_group = candidate_dir.relative_to(source_root)
        case_name = _sanitize_name(
            f"{relative_group.name}_{csv_meta.band}_{csv_meta.atten}_{csv_meta.rate}_{csv_meta.timestamp:%y%m%d_%H%M%S}"
        )
        paired.append(
            CasePair(
                source_dir=candidate_dir,
                csv_path=csv_path,
                zip_path=zip_path,
                case_name=case_name,
                relative_group=relative_group,
            )
        )

    unmatched_zip = sorted(zip_pool.keys())
    return paired, unmatched_csv, unmatched_zip


def ensure_link(src: Path, dst: Path) -> None:
    if dst.exists() or dst.is_symlink():
        dst.unlink()
    try:
        os.symlink(str(src), str(dst))
    except OSError:
        shutil.copy2(src, dst)


def prepare_cases(
    source_root: Path,
    workspace_root: Path,
    main_script_path: Path,
    python_bin: str,
) -> Dict[str, object]:
    candidate_dirs = discover_candidate_dirs(source_root)
    workspace_root.mkdir(parents=True, exist_ok=True)

    all_pairs: List[CasePair] = []
    unmatched_csv: List[str] = []
    unmatched_zip: List[str] = []

    for candidate_dir in candidate_dirs:
        pairs, missing_csv, missing_zip = pair_case_files(source_root, candidate_dir)
        all_pairs.extend(pairs)
        unmatched_csv.extend(str(path) for path in missing_csv)
        unmatched_zip.extend(str(path) for path in missing_zip)

    cases_manifest: List[Dict[str, str]] = []
    for pair in all_pairs:
        case_root = workspace_root / pair.relative_group / pair.case_name
        input_dir = case_root / "input"
        runs_dir = case_root / "runs"
        input_dir.mkdir(parents=True, exist_ok=True)
        runs_dir.mkdir(parents=True, exist_ok=True)

        csv_link = input_dir / pair.csv_path.name
        zip_link = input_dir / pair.zip_path.name
        ensure_link(pair.csv_path, csv_link)
        ensure_link(pair.zip_path, zip_link)

        run_cmd = (
            f"{python_bin} {main_script_path} -d {input_dir} --modem-zip {pair.zip_path.name}"
        )
        cases_manifest.append(
            {
                "source_dir": str(pair.source_dir),
                "csv_path": str(pair.csv_path),
                "zip_path": str(pair.zip_path),
                "case_root": str(case_root),
                "input_dir": str(input_dir),
                "run_cmd": run_cmd,
            }
        )

    manifest_csv = workspace_root / "cases_manifest.csv"
    manifest_json = workspace_root / "cases_manifest.json"

    with manifest_csv.open("w", encoding="utf-8", newline="") as fp:
        writer = csv.DictWriter(
            fp,
            fieldnames=["source_dir", "csv_path", "zip_path", "case_root", "input_dir", "run_cmd"],
        )
        writer.writeheader()
        writer.writerows(cases_manifest)

    manifest_payload = {
        "source_root": str(source_root),
        "workspace_root": str(workspace_root),
        "case_count": len(cases_manifest),
        "candidate_dir_count": len(candidate_dirs),
        "unmatched_csv": unmatched_csv,
        "unmatched_zip": unmatched_zip,
        "cases": cases_manifest,
    }
    manifest_json.write_text(
        json.dumps(manifest_payload, ensure_ascii=False, indent=2),
        encoding="utf-8",
    )
    return manifest_payload


def parse_args() -> argparse.Namespace:
    parser = argparse.ArgumentParser(
        description="Prepare zero-copy AI MD power case workspaces from Ocean FTP logs."
    )
    parser.add_argument(
        "--ftp-path",
        required=True,
        help="FTP-facing path like /UL/log/... or server-local /mnt/vsftp/test/...",
    )
    parser.add_argument(
        "--workspace-root",
        required=True,
        help="Workspace root where symlinked cases and manifests will be created.",
    )
    parser.add_argument(
        "--main-script-path",
        default="scripts/2_data_process_tool/main.py",
        help="Path to the deployed skill_code_ai_md_power main.py used to build run commands.",
    )
    parser.add_argument(
        "--python-bin",
        default=sys.executable,
        help="Python interpreter used to build runnable manifest commands.",
    )
    return parser.parse_args()


def main() -> None:
    args = parse_args()
    source_root = map_ftp_path(args.ftp_path)
    if not source_root.is_dir():
        raise FileNotFoundError(f"source root does not exist: {source_root}")

    workspace_root = Path(args.workspace_root).resolve()
    main_script_path = Path(args.main_script_path)
    if not main_script_path.is_absolute():
        main_script_path = (Path.cwd() / main_script_path).resolve()

    payload = prepare_cases(
        source_root=source_root,
        workspace_root=workspace_root,
        main_script_path=main_script_path,
        python_bin=str(Path(args.python_bin).expanduser().absolute()),
    )

    print(f"[SOURCE] {payload['source_root']}")
    print(f"[WORKSPACE] {payload['workspace_root']}")
    print(f"[CANDIDATE_DIRS] {payload['candidate_dir_count']}")
    print(f"[CASES] {payload['case_count']}")
    print(f"[MANIFEST_CSV] {workspace_root / 'cases_manifest.csv'}")
    print(f"[MANIFEST_JSON] {workspace_root / 'cases_manifest.json'}")
    if payload["unmatched_csv"]:
        print(f"[UNMATCHED_CSV] {len(payload['unmatched_csv'])}")
    if payload["unmatched_zip"]:
        print(f"[UNMATCHED_ZIP] {len(payload['unmatched_zip'])}")


if __name__ == "__main__":
    main()
