#!/usr/bin/env python3
import csv
import re
import json
import os
import sys

BATCHES_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "batches")
OUTPUT_CSV = os.path.join(os.path.dirname(os.path.abspath(__file__)), "data", "batch_env_vars.csv")

ENV_BATCH_IDS = [
    "d40378c5",
    "55394532",
    "ed4c4df0",
    "43de8dd5",
    "912fe119",
    "4ff65c32",
]

JSON_BATCH_IDS = ["4ff65c32"]


def find_batch_files(batches_dir, batch_ids):
    files = []
    for fname in sorted(os.listdir(batches_dir)):
        if fname.endswith(".grepped.txt"):
            continue
        for bid in batch_ids:
            if fname.startswith(f"batch-{bid}") and fname.endswith(".txt"):
                is_rerun = "(1)" in fname
                files.append({
                    "path": os.path.join(batches_dir, fname),
                    "filename": fname,
                    "batch_id_short": bid,
                    "is_rerun": is_rerun,
                })
    return files


def parse_header(text):
    m_bid = re.search(r"BATCH ID:\s*(.+)", text)
    m_total = re.search(r"Total Tasks:\s*(\d+)", text)
    m_success = re.search(r"Successful Tasks:\s*(\d+)", text)
    m_date = re.search(r"Generated:\s*(.+)", text)
    return {
        "batch_id": m_bid.group(1).strip() if m_bid else "",
        "total_tasks": int(m_total.group(1)) if m_total else 0,
        "successful_tasks": int(m_success.group(1)) if m_success else 0,
        "generated": m_date.group(1).strip() if m_date else "",
    }


def split_tasks(text):
    parts = re.split(r"={50,}", text)
    tasks = []
    for part in parts[1:]:
        if re.search(r"TASK\s+\d+", part):
            tasks.append(part.strip())
    return tasks


def parse_task_meta(task_text):
    m_num = re.search(r"TASK\s+(\d+)", task_text)
    m_tid = re.search(r"Task ID:\s*(.+)", task_text)
    m_agent = re.search(r"Agent:\s*(.+)", task_text)
    m_created = re.search(r"Created:\s*(.+)", task_text)
    m_updated = re.search(r"Updated:\s*(.+)", task_text)
    return {
        "task_num": int(m_num.group(1)) if m_num else 0,
        "task_id": m_tid.group(1).strip() if m_tid else "",
        "agent_id": m_agent.group(1).strip() if m_agent else "",
        "created": m_created.group(1).strip() if m_created else "",
        "updated": m_updated.group(1).strip() if m_updated else "",
    }


def extract_output(task_text):
    m = re.search(r"\nOUTPUT:\n(.*?)(?:\nVALUE:|\Z)", task_text, re.DOTALL)
    if m:
        return m.group(1).strip()
    return ""


def parse_standard_env(output_text):
    rows = []
    current_container = None
    for line in output_text.split("\n"):
        line = line.rstrip()
        m_container = re.match(r"Container:\s*(\S+)", line)
        if m_container:
            current_container = m_container.group(1)
            continue
        if current_container and "=" in line and not line.startswith("OCI runtime"):
            key, _, value = line.partition("=")
            key = key.strip()
            if key and re.match(r"^[A-Za-z_][A-Za-z0-9_.*-]*$", key):
                rows.append((current_container, key, value))
    return rows


def parse_json_env(output_text):
    rows = []
    chunks = re.split(r"Container:\s*(\S+)", output_text)
    i = 1
    while i < len(chunks) - 1:
        container_id = chunks[i]
        json_text = chunks[i + 1].strip()
        i += 2
        if not json_text or json_text.startswith("OCI runtime"):
            continue
        try:
            data = json.loads(json_text)
            if isinstance(data, list):
                data = data[0]
            env_list = data.get("Config", {}).get("Env", [])
            for entry in env_list:
                if "=" in entry:
                    key, _, value = entry.partition("=")
                    rows.append((container_id, key, value))
        except (json.JSONDecodeError, AttributeError, IndexError):
            pass
    return rows


def main():
    batch_files = find_batch_files(BATCHES_DIR, ENV_BATCH_IDS)
    print(f"Found {len(batch_files)} batch files to process")

    all_rows = []

    for bf in batch_files:
        print(f"  Parsing {bf['filename']}...")
        with open(bf["path"], "r", encoding="utf-8", errors="replace") as f:
            text = f.read()

        header = parse_header(text)
        tasks = split_tasks(text)
        is_json = bf["batch_id_short"] in JSON_BATCH_IDS
        rerun_tag = "(rerun)" if bf["is_rerun"] else ""

        for task_text in tasks:
            meta = parse_task_meta(task_text)
            output = extract_output(task_text)
            if not output:
                continue

            if is_json:
                env_rows = parse_json_env(output)
            else:
                env_rows = parse_standard_env(output)

            for container_id, key, value in env_rows:
                all_rows.append({
                    "batch_id": header["batch_id"],
                    "batch_id_short": bf["batch_id_short"],
                    "batch_date": header["generated"],
                    "rerun": rerun_tag,
                    "task_num": meta["task_num"],
                    "task_id": meta["task_id"],
                    "agent_id": meta["agent_id"],
                    "created": meta["created"],
                    "updated": meta["updated"],
                    "container_id": container_id,
                    "env_key": key,
                    "env_value": value,
                })

    os.makedirs(os.path.dirname(OUTPUT_CSV), exist_ok=True)

    fieldnames = [
        "batch_id", "batch_id_short", "batch_date", "rerun",
        "task_num", "task_id", "agent_id", "created", "updated",
        "container_id", "env_key", "env_value",
    ]

    with open(OUTPUT_CSV, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=fieldnames)
        writer.writeheader()
        writer.writerows(all_rows)

    unique_batches = len(set(r["batch_id"] for r in all_rows))
    unique_agents = len(set(r["agent_id"] for r in all_rows))
    unique_containers = len(set(r["container_id"] for r in all_rows))
    unique_keys = len(set(r["env_key"] for r in all_rows))

    print(f"\nDone: {len(all_rows)} env vars → {OUTPUT_CSV}")
    print(f"  Batches: {unique_batches}")
    print(f"  Agents: {unique_agents}")
    print(f"  Containers: {unique_containers}")
    print(f"  Unique keys: {unique_keys}")


if __name__ == "__main__":
    main()
