"""Parse free-text maritime noon reports into structured rows.

Noon reports arrive as email bodies typed by whoever had the watch, so the field
order and the spelling drift constantly. This pulls the numbers out, keeps the
units straight, and flags anything it could not read rather than guessing.

    python3 noon_report_parser.py reports/*.txt --csv noon.csv

Standard library only.
"""

from __future__ import annotations

import argparse
import csv
import re
import sys
from dataclasses import asdict, dataclass, fields
from pathlib import Path

# Each field: the canonical name, and the aliases seen in the wild.
PATTERNS: dict[str, str] = {
    "vessel":       r"(?:vessel|ship|mv)\s*[:\-]\s*(?P<v>[^\n,;]+)",
    "date":         r"(?:date|dated)\s*[:\-]\s*(?P<v>[\d]{4}-[\d]{2}-[\d]{2})",
    "latitude":     r"(?:lat|latitude)\s*[:\-]\s*(?P<v>[\d]{1,3}[^\n,;]{0,12}[NS])",
    "longitude":    r"(?:lon|long|longitude)\s*[:\-]\s*(?P<v>[\d]{1,3}[^\n,;]{0,12}[EW])",
    "distance_nm":  r"(?:dist(?:ance)?(?:\s*run)?)\s*[:\-]?\s+(?P<v>[\d.]+)",
    "avg_speed_kn": r"(?:avg\s*speed|average\s*speed|speed)\s*[:\-]?\s+(?P<v>[\d.]+)",
    "hsfo_mt":      r"(?:hsfo|hfo)\s*[:\-]?\s+(?P<v>[\d.]+)",
    "vlsfo_mt":     r"(?:vlsfo|lsfo)\s*[:\-]?\s+(?P<v>[\d.]+)",
    "mgo_mt":       r"(?:mgo|dgo|diesel)\s*[:\-]?\s+(?P<v>[\d.]+)",
    "rob_fw_mt":    r"(?:rob\s*fw|fresh\s*water)\s*[:\-]?\s+(?P<v>[\d.]+)",
}

NUMERIC = {"distance_nm", "avg_speed_kn", "hsfo_mt", "vlsfo_mt", "mgo_mt", "rob_fw_mt"}


@dataclass
class NoonReport:
    source: str = ""
    vessel: str = ""
    date: str = ""
    latitude: str = ""
    longitude: str = ""
    distance_nm: float | None = None
    avg_speed_kn: float | None = None
    hsfo_mt: float | None = None
    vlsfo_mt: float | None = None
    mgo_mt: float | None = None
    rob_fw_mt: float | None = None
    missing: str = ""


def parse_report(text: str, source: str) -> NoonReport:
    report = NoonReport(source=source)
    missing: list[str] = []

    for name, pattern in PATTERNS.items():
        match = re.search(pattern, text, re.IGNORECASE)
        if not match:
            missing.append(name)
            continue
        raw = match.group("v").strip()
        if name in NUMERIC:
            try:
                setattr(report, name, float(raw))
            except ValueError:
                missing.append(name)
        else:
            setattr(report, name, raw)

    report.missing = ";".join(missing)
    return report


def sanity_check(report: NoonReport) -> list[str]:
    """Cheap physical plausibility checks — catches transposed digits."""
    warnings: list[str] = []
    if report.avg_speed_kn is not None and not 0 <= report.avg_speed_kn <= 30:
        warnings.append(f"{report.source}: speed {report.avg_speed_kn} kn is implausible")
    if report.distance_nm is not None and not 0 <= report.distance_nm <= 800:
        warnings.append(f"{report.source}: distance {report.distance_nm} nm is implausible")
    if (
        report.distance_nm is not None
        and report.avg_speed_kn is not None
        and report.avg_speed_kn > 0
    ):
        implied = report.distance_nm / report.avg_speed_kn
        if not 20 <= implied <= 28:
            warnings.append(
                f"{report.source}: distance/speed implies a {implied:.1f} h day"
            )
    return warnings


def main(argv: list[str] | None = None) -> int:
    parser = argparse.ArgumentParser(description="Parse noon reports into CSV.")
    parser.add_argument("files", nargs="+", type=Path, help="Noon report text files")
    parser.add_argument("--csv", type=Path, help="Write rows to this CSV")
    args = parser.parse_args(argv)

    reports: list[NoonReport] = []
    warnings: list[str] = []

    for path in args.files:
        try:
            text = path.read_text(encoding="utf-8", errors="replace")
        except OSError as exc:
            print(f"Skipping {path}: {exc}", file=sys.stderr)
            continue
        report = parse_report(text, path.name)
        reports.append(report)
        warnings.extend(sanity_check(report))

    if not reports:
        print("No reports parsed.", file=sys.stderr)
        return 1

    if args.csv:
        with args.csv.open("w", newline="", encoding="utf-8") as handle:
            writer = csv.DictWriter(handle, fieldnames=[f.name for f in fields(NoonReport)])
            writer.writeheader()
            writer.writerows(asdict(r) for r in reports)
        print(f"Wrote {len(reports)} row(s) to {args.csv}")
    else:
        for report in reports:
            print(asdict(report))

    incomplete = [r for r in reports if r.missing]
    if incomplete:
        print(f"\n{len(incomplete)} report(s) had unreadable fields:", file=sys.stderr)
        for r in incomplete:
            print(f"  {r.source}: {r.missing}", file=sys.stderr)

    for warning in warnings:
        print(f"WARNING {warning}", file=sys.stderr)

    return 0


if __name__ == "__main__":
    raise SystemExit(main())
