"""Generate separate, deterministic CSV/GeoJSON artifacts and provenance."""

from __future__ import annotations

import csv
import hashlib
import json
from dataclasses import asdict
from pathlib import Path
from typing import Any
from urllib.parse import quote

from mountain_twin.trails.gpx import ParsedRoute
from mountain_twin.trails.normalize import normalize_route


def write_csv(path: Path, rows: list[dict[str, Any]], fields: list[str]) -> None:
    """Write UTF-8/LF CSV with fixed six-decimal floats and blank missing values."""
    with path.open("w", encoding="utf-8", newline="") as stream:
        writer = csv.DictWriter(stream, fieldnames=fields, lineterminator="\n")
        writer.writeheader()
        for row in rows:
            writer.writerow(
                {
                    key: f"{value:.6f}" if isinstance(value, float) else value
                    for key, value in row.items()
                }
            )


def write_json(path: Path, value: Any) -> None:
    """Write stable JSON with nonfinite numbers prohibited."""
    path.write_text(
        json.dumps(value, ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False) + "\n",
        encoding="utf-8",
    )


def generate(routes: list[ParsedRoute], output: Path, data_dir: Path) -> None:
    """Generate into a new/empty directory; reject overlap with preserved inputs."""
    output = output.resolve()
    protected = [data_dir.resolve() / name for name in ("processed", "geojson", "raw")]
    for path in protected:
        path = path.resolve()
        if output == path or path in output.parents or output in path.parents:
            raise ValueError(f"output overlaps preserved input: {path}")
    if output.exists() and (not output.is_dir() or any(output.iterdir())):
        raise ValueError(f"output must be new or empty: {output}")
    if not routes:
        raise ValueError("no routes to generate")
    ordered = sorted(
        routes,
        key=lambda r: (
            r.metadata.route_group != "tmb",
            r.metadata.tmb_day or 0,
            r.metadata.route_id,
        ),
    )
    normalized = [normalize_route(route) for route in ordered]
    summaries = [item[0] for item in normalized]
    points = [point for item in normalized for point in item[1]]
    output.mkdir(parents=True, exist_ok=True)
    (output / "geojson").mkdir()
    write_csv(output / "routes_summary.csv", summaries, list(summaries[0]))
    write_csv(
        output / "tmb_routes_summary.csv",
        [r for r in summaries if r["route_group"] == "tmb"],
        list(summaries[0]),
    )
    write_csv(output / "trail_points_master.csv", points, list(points[0]))
    sources = []
    for route, (_, _, geojson) in zip(ordered, normalized):
        filename = quote(route.metadata.route_id, safe="") + ".geojson"
        write_json(output / "geojson" / filename, geojson)
        sources.append(
            {
                "metadata": asdict(route.metadata),
                "source_sha256": route.source_sha256,
                "geojson": "geojson/" + filename,
            }
        )
    artifacts = {
        str(p.relative_to(output)): hashlib.sha256(p.read_bytes()).hexdigest()
        for p in sorted(output.rglob("*"))
        if p.is_file()
    }
    write_json(
        output / "provenance.json",
        {
            "pipeline_version": 1,
            "sources": sources,
            "artifact_sha256": artifacts,
            "calculation_policy": {
                "distance": "haversine sphere, radius 6371008.8 m; no segment bridging",
                "smoothing": "centered median 5, clipped at segment/missing-elevation boundaries",
                "missing_elevation": "no interpolation; incomplete route gain/loss null",
                "grade_min_segment_distance_m": 1.0,
                "csv_float_decimals": 6,
            },
            "solar": "not generated; preserved v0.1 solar CSV is legacy/untrusted after timezone fix",
        },
    )
