#!/usr/bin/env python3
"""
R-4: Aggregate credentials.csv from 96 dumps (Windows Credential Manager).
Parses, decodes base64, categorizes, and exports to CSV.
"""

import csv
import os
import base64
import json
import sys
from collections import Counter, defaultdict

DUMPS_DIR = "/home/user/Work/Research/ir-assessment/findings/dumps"
OUTPUT_CSV = "/home/user/Work/Research/ir-assessment/findings/data/dumps_creds_aggregated.csv"

def try_b64_decode(s):
    if not s:
        return s
    try:
        decoded = base64.b64decode(s)
        text = decoded.decode("utf-8", errors="strict")
        if all(c == "\x00" or c.isprintable() for c in text):
            return text.replace("\x00", "")
        return s
    except Exception:
        return s

def classify_target(target):
    tl = target.lower()
    if "sso_pop" in tl:
        return "SSO_POP"
    if "xblgrts" in tl or "xboxlive" in tl or "xbl|" in tl:
        return "Xbox/XBL"
    if "skype" in tl:
        return "Skype"
    if "roblox" in tl:
        return "Roblox"
    if "github" in tl:
        return "GitHub"
    if "docker" in tl:
        return "Docker"
    if "gitlab" in tl or "gitea" in tl:
        return "GitLab/Gitea"
    if "git:" in tl or "git.oat" in tl or "git.icosa" in tl:
        return "Git"
    if "azure" in tl or "visualstudio" in tl:
        return "Azure DevOps"
    if "nordvpn" in tl:
        return "NordVPN"
    if "nordpass" in tl:
        return "NordPass"
    if "bitwarden" in tl:
        return "Bitwarden"
    if "mozilla" in tl or "firefox" in tl:
        return "Firefox"
    if "outlook" in tl or "office365" in tl or "microsoftoffice" in tl:
        return "Microsoft Office"
    if "onedrive" in tl:
        return "OneDrive"
    if "pgadmin" in tl or "postgres" in tl:
        return "PostgreSQL"
    if "ssms" in tl or "mssql" in tl:
        return "MSSQL"
    if "twitch" in tl:
        return "Twitch"
    if "element.io" in tl or "matrix" in tl:
        return "Matrix/Element"
    if "mcl|" in tl or "mclms|" in tl or "mojang" in tl:
        return "Minecraft/Mojang"
    if "adobe" in tl:
        return "Adobe"
    if "logi" in tl:
        return "Logitech"
    if "lenovo" in tl:
        return "Lenovo"
    if "yandex" in tl:
        return "Yandex"
    if "splice" in tl:
        return "Splice"
    if "xivlauncher" in tl or "ffxv" in tl:
        return "Gaming"
    if "fchat" in tl:
        return "FChat"
    if "xsolla" in tl:
        return "Xsolla"
    if "virtualapp" in tl:
        return "VirtualApp"
    if "elgato" in tl or "streamdeck" in tl:
        return "Elgato"
    if "voicemod" in tl:
        return "Voicemod"
    if "eve-launcher" in tl:
        return "Gaming"
    if "toontown" in tl:
        return "Gaming"
    return "Other"

def classify_criticality(category, target, credential_decoded):
    if category in ("GitHub", "Docker", "GitLab/Gitea", "Git", "Azure DevOps"):
        return "HIGH"
    if category in ("PostgreSQL", "MSSQL", "Microsoft Office"):
        return "HIGH"
    if category in ("Bitwarden", "NordPass"):
        return "MEDIUM"
    if category in ("NordVPN",):
        return "MEDIUM"
    if category in ("Matrix/Element",):
        return "MEDIUM"
    if "192.168" in target or "10." in target:
        return "HIGH"
    if category in ("SSO_POP", "Xbox/XBL", "Skype", "Roblox", "Minecraft/Mojang",
                     "Adobe", "Logitech", "Lenovo", "VirtualApp", "Elgato", "Voicemod"):
        return "LOW"
    return "LOW"

def main():
    rows = []
    stats = Counter()
    category_counts = Counter()
    dump_counts = Counter()

    for dump_id in sorted(os.listdir(DUMPS_DIR)):
        cred_file = os.path.join(DUMPS_DIR, dump_id, "credentials.csv")
        if not os.path.isfile(cred_file):
            continue

        try:
            with open(cred_file, "r", errors="replace") as fh:
                reader = csv.DictReader(fh)
                for row in reader:
                    target = row.get("target", "")
                    cred = row.get("credential", "")
                    user = row.get("username", "")

                    stats["total"] += 1
                    category = classify_target(target)
                    category_counts[category] += 1
                    dump_counts[dump_id] += 1

                    cred_decoded = try_b64_decode(cred) if len(cred) < 200 else ""
                    is_token = len(cred) > 200

                    criticality = classify_criticality(category, target, cred_decoded)

                    rows.append({
                        "dump_id": dump_id,
                        "category": category,
                        "criticality": criticality,
                        "target": target,
                        "username": user,
                        "credential_raw": cred,
                        "credential_decoded": cred_decoded if cred_decoded != cred else "",
                        "is_token": is_token,
                    })
        except Exception as e:
            print(f"Error processing {dump_id}: {e}", file=sys.stderr)

    with open(OUTPUT_CSV, "w", newline="") as fh:
        writer = csv.DictWriter(fh, fieldnames=[
            "dump_id", "category", "criticality", "target", "username",
            "credential_raw", "credential_decoded", "is_token"
        ])
        writer.writeheader()
        writer.writerows(rows)

    print(f"Total rows: {stats['total']}")
    print(f"Output: {OUTPUT_CSV}")
    print(f"Dumps: {len(dump_counts)}")
    print(f"\n=== Category distribution ===")
    for cat, count in category_counts.most_common():
        print(f"  {cat}: {count}")

    high_rows = [r for r in rows if r["criticality"] == "HIGH"]
    print(f"\n=== HIGH criticality entries: {len(high_rows)} ===")
    for r in high_rows:
        decoded = r["credential_decoded"] or r["credential_raw"]
        print(f"  [{r['dump_id']}] {r['category']}: {r['target'][:60]}  user={r['username'][:30]}  cred={decoded[:60]}")

if __name__ == "__main__":
    main()
