#!/usr/bin/env python3
"""Scan all cloned GitLab repos for secrets.
- greps for common secret patterns
- looks at .env, config/*.yml, docker-compose*.yml, application.properties, settings.py, database.yml
- finds hardcoded DB connection strings, SMTP passwords, JWT secrets
Output: _secrets_scan.md
"""
import os, re, json, base64

OUTDIR = "/root/ir-assessment/redteam/camelsoft/gitlab_repos"

# ---- secret regex patterns ----
PATTERNS = [
    ("GitLab PAT", re.compile(r"glpat-[A-Za-z0-9_\-]{15,}")),
    ("GitHub PAT", re.compile(r"ghp_[A-Za-z0-9]{30,}")),
    ("GitHub fine-grained", re.compile(r"github_pat_[A-Za-z0-9_]{30,}")),
    ("AWS Access Key", re.compile(r"AKIA[0-9A-Z]{16}")),
    ("AWS Secret Key", re.compile(r"(?i)aws[_-]?secret[_-]?access[_-]?key['\"\s:=]+[A-Za-z0-9/+=]{40}")),
    ("SendGrid key", re.compile(r"SG\.[A-Za-z0-9_\-]{20,}\.[A-Za-z0-9_\-]{30,}")),
    ("Stripe key", re.compile(r"(?:sk|pk)_(?:test|live)_[A-Za-z0-9]{20,}")),
    ("sk- OpenAI key", re.compile(r"sk-[A-Za-z0-9]{20,}(?:[A-Za-z0-9]{20,})?")),
    ("Slack token", re.compile(r"xox[abprs]-[A-Za-z0-9\-]{10,}")),
    ("Google API key", re.compile(r"AIza[0-9A-Za-z_\-]{30,}")),
    ("JWT secret assign", re.compile(r"(?i)jwt[_-]?(?:secret|key)['\"\s:=]+[A-Za-z0-9_\-\.]{8,}")),
    ("Private key block", re.compile(r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----")),
    ("Telegram bot token", re.compile(r"\b\d{6,12}:[A-Za-z0-9_-]{30,}\b")),
    ("Firebase web key", re.compile(r"AIzaSy[A-Za-z0-9_\-]{20,}")),
    ("MongoDB connection", re.compile(r"mongodb(?:\+srv)?://[^\s\"'<>]+")),
    ("Postgres connection", re.compile(r"postgres(?:ql)?://[^\s\"'<>]+")),
    ("MySQL connection", re.compile(r"mysql://[^\s\"'<>]+")),
    ("Redis URL", re.compile(r"redis://[^\s\"'<>]+")),
    ("HTTP basic auth URL", re.compile(r"https?://[^\s\"'<>]*:[^\s\"'<>]+@[^\s\"'<>]+")),
    ("Stripe webhook", re.compile(r"whsec_[A-Za-z0-9]+")),
    ("Twilio SID", re.compile(r"AC[a-z0-9]{32}", re.I)),
    ("Coinbase/Gcp sa", re.compile(r"\"type\":\s*\"service_account\"")),
    ("Generic password assign", re.compile(r"(?i)(?:password|passwd|pwd|pass)\s*[:=]\s*['\"]?[^\s'\"]{3,}")),
    ("Generic secret assign", re.compile(r"(?i)(?:secret|api[_-]?key|apikey|token|access[_-]?key)\s*[:=]\s*['\"]?[^\s'\"]{6,}")),
    ("Base64 long blob", re.compile(r"\beyJ[A-Za-z0-9_\-]{20,}\.[A-Za-z0-9_\-]{20,}\.[A-Za-z0-9_\-]{20,}\b")),  # JWT
    ("Bearer token", re.compile(r"Bearer\s+[A-Za-z0-9_\-\.]{20,}")),
]

# ---- config file patterns ----
CONFIG_GLOBS = [
    ".env", ".env.*", "*.env",
    "docker-compose*.yml", "docker-compose*.yaml",
    "application.yml", "application.yaml", "application*.properties",
    "application*.yml", "application*.yaml",
    "settings.py", "settings*.py", "local_settings.py",
    "database.yml", "database.yaml", "db.yml",
    "config.yml", "config.yaml", "config.json",
    "secrets.yml", "secrets.yaml",
    "credentials", "credentials.json",
    "*.config", "config.*",
    "appsettings*.json", "appsettings*.yml",
    ".envrc",
    "*.key", "*.pem", "*.pfx",
    "id_rsa*", "id_ed25519*",
]

# extensions to scan fully
SCAN_EXT = {".yml", ".yaml", ".json", ".env", ".properties", ".conf", ".cfg", ".ini",
            ".toml", ".py", ".js", ".ts", ".jsx", ".tsx", ".php", ".rb", ".go", ".rs",
            ".java", ".kt", ".cs", ".swift", ".config", ".sh", ".bash", ".zsh",
            ".xml", ".gradle", ".tf", ".hcl"}

# files by basename to scan fully regardless of ext
SCAN_BASENAMES = {"dockerfile", "makefile", ".env", ".envrc", "config", "settings",
                  "application", "database", "credentials", "secrets", "env"}

SKIP_DIRS = {".git", "node_modules", "vendor", "bower_components", "dist", "build",
             "target", "__pycache__", ".gradle", ".idea", ".vscode", ".next",
             "coverage", ".cache", "tmp", "logs", "Pods", "Carthage", "DerivedData",
             "venv", ".venv", "env", ".eggs"}

MAX_FILE = 2_000_000  # skip files > 2MB
MAX_FINDINGS_PER_FILE = 200

def is_text(b):
    try:
        b.decode("utf-8"); return True
    except Exception:
        return False

def scan_file(filepath, findings, repo_name, relpath):
    try:
        size = os.path.getsize(filepath)
    except OSError:
        return
    if size == 0 or size > MAX_FILE:
        return
    try:
        with open(filepath, "rb") as f:
            raw = f.read()
        if not is_text(raw):
            return
        text = raw.decode("utf-8", errors="replace")
    except Exception:
        return
    lines = text.split("\n")
    file_hits = 0
    for i, line in enumerate(lines, 1):
        if file_hits >= MAX_FINDINGS_PER_FILE:
            break
        for label, pat in PATTERNS:
            for m in pat.finditer(line):
                val = m.group(0)
                # filter obvious false positives
                if "password" in label.lower() or "secret" in label.lower() or "token" in label.lower() and "Generic" in label:
                    v = val.split("=", 1)[-1].split(":", 1)[-1].strip().strip("'\"")
                    if v.lower() in ("true", "false", "null", "none", "", "$password", "$env", "${", "your", "example", "changeme", "xxx", "password"):
                        continue
                    if v.startswith("${") or v.startswith("$(") or v.startswith("env["):
                        continue
                    if "placeholder" in v.lower() or "example" in v.lower() or "your_" in v.lower():
                        continue
                findings.append({
                    "repo": repo_name, "file": relpath, "line_no": i,
                    "type": label, "match": val, "line": line.strip()[:500]
                })
                file_hits += 1
                break  # one match type per line to reduce noise

def walk_repo(repo_dir, repo_name, findings):
    for root, dirs, files in os.walk(repo_dir):
        dirs[:] = [d for d in dirs if d not in SKIP_DIRS and not d.startswith(".git")]
        for fn in files:
            fp = os.path.join(root, fn)
            relp = os.path.relpath(fp, repo_dir)
            ext = os.path.splitext(fn)[1].lower()
            base = fn.lower()
            # always scan config-ish files and SCAN_EXT
            scan = False
            if ext in SCAN_EXT:
                scan = True
            if base in SCAN_BASENAMES or base.startswith(".env") or base.startswith("docker-compose"):
                scan = True
            # match config globs
            for g in CONFIG_GLOBS:
                if g.startswith("*."):
                    if base.endswith(g[1:]) and g != "*.config":
                        scan = True; break
                elif g.endswith(".*"):
                    if base.startswith(g[:-2]):
                        scan = True; break
                elif "*" in g:
                    # simple wildcard match
                    rg = re.compile("^" + re.escape(g).replace("\\*", ".*") + "$")
                    if rg.match(base):
                        scan = True; break
                else:
                    if base == g.lower():
                        scan = True; break
            # scan all .env* and *.key/*.pem
            if base.startswith(".env") or ext in (".key", ".pem", ".pfx", ".env"):
                scan = True
            if scan:
                scan_file(fp, findings, repo_name, relp)

def main():
    # find all cloned repo dirs (dirs containing .git, two-segment names)
    repos = []
    for entry in sorted(os.listdir(OUTDIR)):
        full = os.path.join(OUTDIR, entry)
        if os.path.isdir(full) and os.path.isdir(os.path.join(full, ".git")):
            repos.append((entry, full))
    print(f"Scanning {len(repos)} repos...", flush=True)

    all_findings = []
    for i, (rname, rdir) in enumerate(repos, 1):
        repo_findings = []
        walk_repo(rdir, rname, repo_findings)
        all_findings.extend(repo_findings)
        if repo_findings:
            print(f"  [{i}/{len(repos)}] {rname}: {len(repo_findings)} findings", flush=True)

    # group by repo
    by_repo = {}
    for f in all_findings:
        by_repo.setdefault(f["repo"], []).append(f)

    # write markdown
    md = ["# GitLab Repositories — Secret Scan Results\n",
          f"Scanned **{len(repos)}** repos. Total findings: **{len(all_findings)}** across **{len(by_repo)}** repos.\n"]
    for rname in sorted(by_repo.keys()):
        fs = by_repo[rname]
        md.append(f"\n## {rname} ({len(fs)} findings)\n")
        seen = set()
        for f in sorted(fs, key=lambda x: (x["file"], x["line_no"])):
            key = (f["file"], f["line_no"], f["type"])
            if key in seen:
                continue
            seen.add(key)
            md.append(f"- `{f['file']}:{f['line_no']}` — **{f['type']}**")
            md.append(f"  ```")
            md.append(f"  {f['line']}")
            md.append(f"  ```")
    with open(os.path.join(OUTDIR, "_secrets_scan.md"), "w") as f:
        f.write("\n".join(md))
    # also save JSON
    with open(os.path.join(OUTDIR, "_secrets_scan.json"), "w") as f:
        json.dump(all_findings, f, indent=2)

    # summary by type
    types = {}
    for f in all_findings:
        types[f["type"]] = types.get(f["type"], 0) + 1
    print("\n=== SCAN COMPLETE ===")
    print(f"Total findings: {len(all_findings)} across {len(by_repo)} repos")
    print("By type:")
    for t, c in sorted(types.items(), key=lambda x: -x[1]):
        print(f"  {t}: {c}")

if __name__ == "__main__":
    main()
