#!/usr/bin/env python3
"""
L16v2: Fast secret mining — .env, config, key files only (skip full SQL data)
"""
import os, re, glob, gzip, tarfile, json, csv, hashlib, sys
from collections import defaultdict

BASE = "/root/ir-assessment/redteam/gitlab_npontutechnologies_com"
os.chdir(BASE)

PATTERNS = {
    "aws_key": r"AKIA[0-9A-Z]{16}",
    "github_token": r"ghp_[0-9a-zA-Z]{36}",
    "gitlab_token": r"glpat-[0-9a-zA-Z\-_]{20}",
    "google_api": r"AIza[0-9A-Za-z\-_]{35}",
    "tgbot": r"\d{9,10}:[0-9A-Za-z_-]{35}",
    "stripe": r"sk_live_[0-9a-zA-Z]{24}",
    "jwt_secret": r"(?i)jwt.{0,15}secret.{0,10}['\"]([^'\"]{8,})['\"]",
    "app_key": r"(?i)app.{0,10}key.{0,10}['\"](base64:[^'\"]{16,}|[^'\"]{32,})['\"]",
    "db_password": r"(?i)(db_|database_|mysql_|pg_|postgres_).{0,10}(pass|pwd|password).{0,10}['\"]([^'\"\s]{6,})['\"]",
    "ssh_private": r"-----BEGIN (?:RSA |EC |OPENSSH |DSA )?PRIVATE KEY-----",
    "connection_string": r"(?i)(mysql|postgres|pgsql|mongodb|redis)://[^\\s\"']+",
    "api_key": r"(?i)(api[_-]?key|apikey|secret[_-]?key)['\"]?\s*[:=]\s*['\"]([0-9a-zA-Z\-_\.]{16,})['\"]",
    "bearer": r"Bearer\s+[0-9a-zA-Z\-_\.]{20,}",
    "mail_password": r"(?i)(mail|smtp).{0,15}(pass|pwd|password).{0,10}['\"]([^'\"\s]{6,})['\"]",
}

COMPILED = {k: re.compile(v) for k, v in PATTERNS.items()}
RESULTS = []
SEEN = set()

def add(pattern, source, file, line, value):
    h = hashlib.sha256(f"{pattern}|{value}".encode()).hexdigest()[:16]
    if h not in SEEN:
        SEEN.add(h)
        RESULTS.append({"pattern": pattern, "source": source, "file": file, "line": line, "value": value[:200], "hash": h})

def scan_text(content, source, filename):
    for pname, pat in COMPILED.items():
        try:
            for m in pat.finditer(content):
                line = content[:m.start()].count('\n') + 1
                add(pname, source, filename, line, m.group(0))
        except:
            pass

def scan_file(path, source):
    try:
        if path.endswith('.gz'):
            with gzip.open(path, 'rt', errors='ignore') as f:
                scan_text(f.read(), source, path)
        else:
            with open(path, 'r', errors='ignore') as f:
                scan_text(f.read(), source, path)
    except:
        pass

def scan_tar_gz(path, source):
    try:
        with tarfile.open(path, 'r:gz') as tf:
            for m in tf.getmembers():
                if not m.isfile() or m.size > 2*1024*1024:
                    continue
                # only config/key files
                if not re.search(r'\.(env|env\..*|ini|cfg|conf|config|yml|yaml|json|xml|php|py|sh|sql|txt|md|properties|credentials|secret|key|pem|ppk)$', m.name, re.I):
                    continue
                try:
                    f = tf.extractfile(m)
                    if f:
                        scan_text(f.read().decode('utf-8', errors='ignore'), source, f"{path}:{m.name}")
                except:
                    pass
    except Exception as e:
        print(f"tar error: {e}", file=sys.stderr)

def main():
    # 1. Source selective (fast — configs only)
    print("[*] source selective...")
    for f in glob.glob("exfil_hostB_source_selective/*.tar.gz"):
        scan_tar_gz(f, "source_selective")
        print(f"  {os.path.basename(f)}")
    
    # 2. Source full (stream, configs only)
    print("[*] source full (configs only)...")
    for f in glob.glob("exfil_hostB_source_full/*.tar.gz"):
        scan_tar_gz(f, "source_full")
    
    # 3. Home dirs
    print("[*] home dirs...")
    for f in glob.glob("exfil_hostA_home/*") + glob.glob("exfil_hostB_home/*"):
        if os.path.isfile(f):
            scan_file(f, "home")
    
    # 4. Key DB samples (not full data — just schema/lookup tables)
    print("[*] DB schema samples...")
    for f in glob.glob("exfil_all/pg_*/.done"):
        pass  # skip markers
    
    # Write
    with open("L16_secrets.json", "w") as f:
        json.dump(RESULTS, f, indent=2)
    with open("L16_secrets.tsv", "w", newline='') as f:
        w = csv.DictWriter(f, fieldnames=["pattern","source","file","line","value","hash"], delimiter='\t')
        w.writeheader()
        w.writerows(RESULTS)
    
    print(f"\n[+] TOTAL: {len(RESULTS)} unique")
    by_p = defaultdict(int)
    for r in RESULTS:
        by_p[r['pattern']] += 1
    for p, c in sorted(by_p.items(), key=lambda x: -x[1]):
        print(f"  {p:20s} {c}")

if __name__ == "__main__":
    main()
