#!/usr/bin/env python3
"""
L16: Full secret mining across ALL exfil sources
Covers: DB dumps, source code, home dirs
Output: L16_secrets.json + L16_secrets.tsv
"""
import os, re, glob, gzip, tarfile, json, csv, hashlib, sys
from collections import defaultdict

BASE = "/root/ir-assessment/redteam/gitlab_npontutechnologies_com"
os.chdir(BASE)

PATTERNS = {
    "aws_key": r"AKIA[0-9A-Z]{16}",
    "github_token": r"ghp_[0-9a-zA-Z]{36}",
    "gitlab_token": r"glpat-[0-9a-zA-Z\-_]{20}",
    "slack_token": r"xox[baprs]-[0-9a-zA-Z\-]{10,48}",
    "google_api": r"AIza[0-9A-Za-z\-_]{35}",
    "jwt_secret": r"(?i)jwt.{0,15}secret.{0,10}['\"]([^'\"]{8,})['\"]",
    "app_key": r"(?i)app.{0,10}key.{0,10}['\"](base64:[^'\"]{16,}|[^'\"]{32,})['\"]",
    "db_password": r"(?i)(db_|database_|mysql_|pg_|postgres_).{0,10}(pass|pwd|password).{0,10}['\"]([^'\"\s]{6,})['\"]",
    "ssh_private": r"-----BEGIN (?:RSA |EC |OPENSSH |DSA )?PRIVATE KEY-----",
    "connection_string": r"(?i)(mysql|postgres|pgsql|mongodb|redis)://[^\\s\"']+",
    "api_key_generic": r"(?i)(api[_-]?key|apikey|secret[_-]?key)['\"]?\s*[:=]\s*['\"]([0-9a-zA-Z\-_\.]{16,})['\"]",
    "bearer_token": r"Bearer\s+[0-9a-zA-Z\-_\.]{20,}",
    "tgbot_token": r"\d{9,10}:[0-9A-Za-z_-]{35}",
    "stripe_key": r"sk_live_[0-9a-zA-Z]{24}",
    "mail_password": r"(?i)(mail|smtp).{0,15}(pass|pwd|password).{0,10}['\"]([^'\"\s]{6,})['\"]",
    "encryption_key": r"(?i)(encrypt|cipher|aes).{0,15}key.{0,10}['\"]([^'\"]{16,})['\"]",
}

COMPILED = {k: re.compile(v) for k, v in PATTERNS.items()}
RESULTS = []

def scan_content(content, source, filename):
    for pname, pat in COMPILED.items():
        try:
            for m in pat.finditer(content):
                line = content[:m.start()].count('\n') + 1
                val = m.group(0)
                # dedupe key
                h = hashlib.sha256(f"{pname}|{val}".encode()).hexdigest()[:16]
                RESULTS.append({
                    "pattern": pname,
                    "source": source,
                    "file": filename,
                    "line": line,
                    "value": val[:200],
                    "hash": h,
                })
        except Exception:
            pass

def scan_text_file(path, source):
    try:
        with open(path, 'r', errors='ignore') as f:
            scan_content(f.read(), source, path)
    except:
        pass

def scan_gz_file(path, source):
    try:
        with gzip.open(path, 'rt', errors='ignore') as f:
            scan_content(f.read(), source, path)
    except:
        pass

def scan_tar_gz(path, source):
    try:
        with tarfile.open(path, 'r:gz') as tf:
            for m in tf.getmembers():
                if not m.isfile() or m.size > 5*1024*1024:
                    continue
                if not m.name.endswith(('.py','.php','.js','.json','.yml','.yaml','.env','.env.example','.xml','.ini','.cfg','.conf','.txt','.md','.sh','.sql','.ts','.vue','.env.local','.env.production','.env.staging','.htaccess','.gitconfig','.npmrc','.dockercfg','.dockerconfigjson')):
                    continue
                try:
                    f = tf.extractfile(m)
                    if f:
                        scan_content(f.read().decode('utf-8', errors='ignore'), source, f"{path}:{m.name}")
                except:
                    pass
    except Exception as e:
        print(f"  tar error {path}: {e}", file=sys.stderr)

def main():
    # 1. DB dumps (gz sql)
    print("[*] Scanning DB dumps...")
    for d in ["exfil", "exfil_full", "exfil_all", "exfil_hellio", "exfil_hostB_pg", "exfil_mysql_remaining", "exfil_ecobank_cvms"]:
        if not os.path.isdir(d):
            continue
        for f in glob.glob(f"{d}/**/*.gz", recursive=True) + glob.glob(f"{d}/**/*.sql", recursive=True) + glob.glob(f"{d}/**/*.txt", recursive=True):
            if f.endswith('.gz'):
                scan_gz_file(f, d)
            else:
                scan_text_file(f, d)
        print(f"  {d}: done")

    # 2. Source code selective (tar.gz)
    print("[*] Scanning source selective...")
    for f in glob.glob("exfil_hostB_source_selective/*.tar.gz"):
        scan_tar_gz(f, "source_selective")
    print("  source_selective: done")

    # 3. Source code full (tar.gz) — stream scan, this is the big one
    print("[*] Scanning source full (54GB)...")
    for f in glob.glob("exfil_hostB_source_full/*.tar.gz"):
        scan_tar_gz(f, "source_full")
    print("  source_full: done")

    # 4. Home dirs
    print("[*] Scanning home dirs...")
    for f in glob.glob("exfil_hostA_home/*"):
        if os.path.isfile(f):
            if f.endswith('.gz'):
                scan_gz_file(f, "home_a")
            else:
                scan_text_file(f, "home_a")
    print("  home_a: done")

    # 5. Existing harvested secrets for cross-reference
    print("[*] Cross-referencing with L7 deep secrets...")
    if os.path.exists("L7_deep_secrets.tsv"):
        with open("L7_deep_secrets.tsv", errors='ignore') as f:
            scan_content(f.read(), "L7_crossref", "L7_deep_secrets.tsv")

    # Dedupe
    seen = set()
    unique = []
    for r in RESULTS:
        if r['hash'] not in seen:
            seen.add(r['hash'])
            unique.append(r)

    # Write outputs
    with open("L16_secrets.json", "w") as f:
        json.dump(unique, f, indent=2)

    with open("L16_secrets.tsv", "w", newline='') as f:
        w = csv.DictWriter(f, fieldnames=["pattern","source","file","line","value","hash"], delimiter='\t')
        w.writeheader()
        w.writerows(unique)

    print(f"\n[+] TOTAL unique secrets: {len(unique)}")
    by_pattern = defaultdict(int)
    by_source = defaultdict(int)
    for r in unique:
        by_pattern[r['pattern']] += 1
        by_source[r['source']] += 1

    print("\n=== BY PATTERN ===")
    for p, c in sorted(by_pattern.items(), key=lambda x: -x[1]):
        print(f"  {p:25s} {c}")

    print("\n=== BY SOURCE ===")
    for s, c in sorted(by_source.items(), key=lambda x: -x[1]):
        print(f"  {s:35s} {c}")

if __name__ == "__main__":
    main()
