#!/usr/bin/env python3
from __future__ import annotations

import csv
from collections import Counter
from pathlib import Path


ROOT = Path(__file__).resolve().parents[1]
INDEX_PATH = ROOT / "artifacts" / "index.csv"

EXPECTED_HEADERS = [
    "artifact_id",
    "path",
    "kind",
    "producer",
    "inputs",
    "status",
    "notes",
]

ALLOWED_KINDS = {"report", "inventory"}
ALLOWED_STATUSES = {"active", "partial", "stale", "deprecated"}
ALLOWED_PRODUCER_PREFIXES = {
    "scripts/",
    "manual_analysis",
}


def load_rows() -> list[dict[str, str]]:
    with INDEX_PATH.open(newline="") as handle:
        reader = csv.DictReader(handle)
        if reader.fieldnames != EXPECTED_HEADERS:
            raise ValueError(f"unexpected headers: {reader.fieldnames}")
        return list(reader)


def producer_ok(value: str) -> bool:
    return any(value == prefix or value.startswith(prefix) for prefix in ALLOWED_PRODUCER_PREFIXES)


def validate_rows(rows: list[dict[str, str]]) -> list[str]:
    errors: list[str] = []
    seen_ids: set[str] = set()
    seen_paths: set[str] = set()

    for row in rows:
        artifact_id = row["artifact_id"]
        path = row["path"]

        if artifact_id in seen_ids:
            errors.append(f"{artifact_id}: duplicate artifact_id")
        seen_ids.add(artifact_id)

        if path in seen_paths:
            errors.append(f"{artifact_id}: duplicate path={path}")
        seen_paths.add(path)

        if row["kind"] not in ALLOWED_KINDS:
            errors.append(f"{artifact_id}: invalid kind={row['kind']}")
        if row["status"] not in ALLOWED_STATUSES:
            errors.append(f"{artifact_id}: invalid status={row['status']}")

        full_path = ROOT / path
        if not full_path.exists():
            errors.append(f"{artifact_id}: missing path={path}")

        producer = row["producer"]
        if not producer_ok(producer):
            errors.append(f"{artifact_id}: invalid producer={producer}")
        elif producer.startswith("scripts/") and not (ROOT / producer).exists():
            errors.append(f"{artifact_id}: producer script missing={producer}")

        inputs = [item for item in row["inputs"].split("|") if item]
        if not inputs:
            errors.append(f"{artifact_id}: no inputs listed")
        for input_path in inputs:
            if not (ROOT / input_path).exists():
                errors.append(f"{artifact_id}: missing input={input_path}")

        if not row["notes"].strip():
            errors.append(f"{artifact_id}: empty notes")

    return errors


def print_report(rows: list[dict[str, str]]) -> None:
    print(f"rows={len(rows)}")
    for field in ["kind", "status"]:
        counts = Counter(row[field] for row in rows)
        parts = " ".join(f"{key}={counts[key]}" for key in sorted(counts))
        print(f"{field}: {parts}")


def main() -> int:
    try:
        rows = load_rows()
    except Exception as exc:
        print(f"artifact-inventory validation failed: {exc}")
        return 2

    errors = validate_rows(rows)
    if errors:
        print("artifact-inventory validation failed")
        for error in errors:
            print(f"- {error}")
        return 1

    print("artifact-inventory validation OK")
    print_report(rows)
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
