#!/usr/bin/env python3

import csv
import re
from pathlib import Path
from typing import List, Sequence, Tuple

MARKDOWN_RULE_HEADING = re.compile(r"^####\s+`?([A-Za-z0-9_./-]+)`?(?:\s+(.*))?$")
MARKDOWN_RULE_PREFIX = re.compile(r"^(?:Rule|Source|规则|来源)\s*[:：]\s*(.*)$", re.IGNORECASE)


def detect_delimiter(file_path: Path) -> str:
    first_line = file_path.read_text(encoding="utf-8").splitlines()[0]
    if "\t" in first_line:
        return "\t"
    return ","


def load_delimited_rules(path: Path) -> List[Tuple[str, str]]:
    delimiter = detect_delimiter(path)
    rows: List[Tuple[str, str]] = []
    with path.open("r", encoding="utf-8", newline="") as fp:
        reader = csv.reader(fp, delimiter=delimiter)
        for raw_row in reader:
            if not raw_row:
                continue
            row = [cell.strip() for cell in raw_row]
            if len(row) < 2 or not row[0]:
                continue
            rows.append((row[0], row[1]))
    return rows


def normalize_markdown_rule_lines(lines: Sequence[str]) -> str:
    normalized: List[str] = []
    for line in lines:
        text = line.strip()
        if not text:
            if normalized:
                break
            continue
        match = MARKDOWN_RULE_PREFIX.match(text)
        if match:
            text = match.group(1).strip()
        if text:
            normalized.append(text)
    return " ".join(normalized)


def load_markdown_rules(path: Path) -> List[Tuple[str, str]]:
    rows: List[Tuple[str, str]] = []
    current_field = ""
    current_lines: List[str] = []

    def flush_current() -> None:
        if current_field:
            rows.append((current_field, normalize_markdown_rule_lines(current_lines)))

    for raw_line in path.read_text(encoding="utf-8").splitlines():
        match = MARKDOWN_RULE_HEADING.match(raw_line.strip())
        if match:
            flush_current()
            current_field = match.group(1).strip()
            heading_tail = (match.group(2) or "").strip()
            current_lines = [heading_tail] if heading_tail else []
            continue
        if current_field:
            current_lines.append(raw_line)
    flush_current()
    return rows


def load_transposed_rules(file_path: str) -> List[Tuple[str, str]]:
    path = Path(file_path)
    if path.suffix.lower() == ".md":
        return load_markdown_rules(path)
    return load_delimited_rules(path)


def split_dual_rule(rule_text: str) -> Tuple[str, str]:
    parts = [part.strip() for part in rule_text.split("|", 1)]
    if len(parts) == 1:
        return parts[0], ""
    return parts[0], parts[1]


def build_horizontal_rules(
    rows: Sequence[Tuple[str, str]],
) -> Tuple[List[str], List[str], List[str]]:
    headers: List[str] = []
    primary_row: List[str] = []
    secondary_row: List[str] = []
    for field_name, rule_text in rows:
        primary, secondary = split_dual_rule(rule_text)
        headers.append(field_name)
        primary_row.append(primary)
        secondary_row.append(secondary)
    return headers, primary_row, secondary_row


def write_horizontal_rules(
    output_path: str,
    headers: Sequence[str],
    primary_row: Sequence[str],
    secondary_row: Sequence[str],
) -> None:
    path = Path(output_path)
    path.parent.mkdir(parents=True, exist_ok=True)
    with path.open("w", encoding="utf-8", newline="") as fp:
        writer = csv.writer(fp)
        writer.writerow(list(headers))
        writer.writerow(list(primary_row))
        writer.writerow(list(secondary_row))


def write_markdown_rules(output_path: str, rows: Sequence[Tuple[str, str]]) -> None:
    path = Path(output_path)
    path.parent.mkdir(parents=True, exist_ok=True)
    lines: List[str] = []
    for field_name, rule_text in rows:
        lines.append(f"#### {field_name}")
        lines.append("")
        if rule_text:
            lines.append(rule_text)
        lines.append("")
    path.write_text("\n".join(lines).rstrip() + "\n", encoding="utf-8")
