#!/usr/bin/env python3
from __future__ import annotations

import csv
from collections import Counter
from pathlib import Path


ROOT = Path(__file__).resolve().parents[1]
INDEX_PATH = ROOT / "findings" / "data" / "index.csv"

EXPECTED_HEADERS = [
    "dataset_id",
    "path",
    "kind",
    "source",
    "format",
    "producer",
    "consumers",
    "status",
    "notes",
]

ALLOWED_KINDS = {"raw", "intermediate", "derived", "artifact", "reference"}
ALLOWED_STATUSES = {"active", "partial", "stale", "deprecated"}
ALLOWED_PRODUCER_PREFIXES = {
    "scripts/",
    "manual_analysis",
    "manual_extract",
    "manual_export",
    "manual_research",
}


def load_rows() -> list[dict[str, str]]:
    with INDEX_PATH.open(newline="") as handle:
        reader = csv.DictReader(handle)
        if reader.fieldnames != EXPECTED_HEADERS:
            raise ValueError(f"unexpected headers: {reader.fieldnames}")
        return list(reader)


def producer_ok(value: str) -> bool:
    return any(value == prefix or value.startswith(prefix) for prefix in ALLOWED_PRODUCER_PREFIXES)


def validate_rows(rows: list[dict[str, str]]) -> list[str]:
    errors: list[str] = []
    seen_ids: set[str] = set()
    seen_paths: set[str] = set()

    for row in rows:
        dataset_id = row["dataset_id"]
        path = row["path"]

        if dataset_id in seen_ids:
            errors.append(f"{dataset_id}: duplicate dataset_id")
        seen_ids.add(dataset_id)

        if path in seen_paths:
            errors.append(f"{dataset_id}: duplicate path={path}")
        seen_paths.add(path)

        if row["kind"] not in ALLOWED_KINDS:
            errors.append(f"{dataset_id}: invalid kind={row['kind']}")
        if row["status"] not in ALLOWED_STATUSES:
            errors.append(f"{dataset_id}: invalid status={row['status']}")

        full_path = ROOT / path
        if not full_path.exists():
            errors.append(f"{dataset_id}: missing path={path}")

        producer = row["producer"]
        if not producer_ok(producer):
            errors.append(f"{dataset_id}: invalid producer={producer}")
        elif producer.startswith("scripts/") and not (ROOT / producer).exists():
            errors.append(f"{dataset_id}: producer script missing={producer}")

        consumers = [item for item in row["consumers"].split("|") if item]
        if not consumers:
            errors.append(f"{dataset_id}: no consumers listed")
        for consumer in consumers:
            if not (ROOT / consumer).exists():
                errors.append(f"{dataset_id}: missing consumer={consumer}")

        if not row["notes"].strip():
            errors.append(f"{dataset_id}: empty notes")

    return errors


def print_report(rows: list[dict[str, str]]) -> None:
    print(f"rows={len(rows)}")
    for field in ["kind", "source", "format", "status"]:
        counts = Counter(row[field] for row in rows)
        parts = " ".join(f"{key}={counts[key]}" for key in sorted(counts))
        print(f"{field}: {parts}")


def main() -> int:
    try:
        rows = load_rows()
    except Exception as exc:
        print(f"data-inventory validation failed: {exc}")
        return 2

    errors = validate_rows(rows)
    if errors:
        print("data-inventory validation failed")
        for error in errors:
            print(f"- {error}")
        return 1

    print("data-inventory validation OK")
    print_report(rows)
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
