#!/usr/bin/env python3
"""L2 raw-file secrets scan for gitlab.npontutechnologies.com (engagement one-off).

User-level token (non-admin), 82 member projects, GitLab 14.1.2.
Walks each project's recursive repository tree (default branch), fetches raw
text files, regex-hunts secrets. Per-project checkpoints in L2_raw_scan/ so a
kill loses nothing and re-runs resume. Read-only GETs only.

Pitfall guards baked in (l2-scope-enumeration skill):
- 11: request failure != empty page; a failed tree page aborts the project
  with explicit error, NOT a checkpoint. 0-file project with non-null
  default_branch = anomaly, not checkpointed.
- 12: checkpoints are the ground truth; top-level JSON rewritten each run.
- G00: checkpoint/resume by project id; ThreadPoolExecutor with modest
  workers (target is a single nginx, be polite).
- no-masking: full matching lines kept (bounded 2000 chars/file output cap,
  full line values, no truncation of secrets).
"""
import json, re, ssl, sys, time, urllib.error, urllib.parse, urllib.request
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

BASE = 'https://gitlab.npontutechnologies.com'
ROOT = Path('/root/ir-assessment/redteam/gitlab_npontutechnologies_com')
CKPT = ROOT / 'L2_raw_scan'
OUT_JSON = ROOT / 'L2_raw_secrets.json'
OUT_TSV = ROOT / 'L2_raw_secrets.tsv'
LOG = ROOT / 'raw_scan.log'

CTX = ssl.create_default_context(); CTX.check_hostname = False; CTX.verify_mode = ssl.CERT_NONE
TOK = (ROOT / '.token').read_text().strip()
H = {'Authorization': f'Bearer {TOK}', 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) ir-assessment-l2'}

MAX_FILE = 200 * 1024
WORKERS = 4

TEXT_EXT = re.compile(r'\.(env|txt|md|yml|yaml|json|xml|ini|cfg|conf|config|php|py|js|ts|rb|go|java|sh|bash|sql|'
                      r'properties|gradle|htaccess|pem|key|pub|crt|cer|p12|tf|tfvars|dockerfile|'
                      r'html|htm|css|vue|jsx|tsx|pl|pm|c|h|cpp|cs|swift|kt|rs|toml|lock|example|sample|old|bak|dist|template|tpl)$', re.I)
SKIP_DIR = re.compile(r'(^|/)(vendor|node_modules|\.git|\.svn|__pycache__|\.idea|\.vscode|dist/assets|'
                      r'public/assets/(plugins|vendor)|Lib/site-packages|storage/framework|bootstrap/cache)/', re.I)
SKIP_NAME = re.compile(r'(package-lock\.json|composer\.lock|yarn\.lock|Gemfile\.lock|poetry\.lock|'
                       r'mix-manifest\.json|\.min\.js|\.min\.css|\.map)$', re.I)

# Secret regexes — assignment-style + high-entropy tokens
RX = [
 ('glpat', re.compile(r'glpat-[\w-]{20,}')),
 ('tgbot', re.compile(r'\d{8,10}:AA[\w-]{33}')),
 ('jwt', re.compile(r'eyJ[\w-]{10,}\.[\w-]{10,}\.[\w-]{5,}')),
 ('privkey', re.compile(r'-----BEGIN [A-Z0-9 ]*PRIVATE KEY-----')),
 ('akia', re.compile(r'AKIA[0-9A-Z]{16}')),
 ('google_api', re.compile(r'AIza[0-9A-Za-z_-]{35}')),
 ('slack', re.compile(r'xox[baprs]-[0-9A-Za-z-]{10,}')),
 ('mailgun', re.compile(r'key-[0-9a-f]{32}')),
 ('assign', re.compile(r'(?i)(password|passwd|pwd|secret|api[_-]?key|apikey|token|access[_-]?key|secret[_-]?key|'
                       r'private[_-]?key|client[_-]?secret|app[_-]?key|auth|smtp|smtp_password)\s*[:=]\s*'
                       r'["\']?([^\s"\'#,;]{6,120})')),
]
PLACEHOLDER = re.compile(r'(?i)^(change[-_ ]?me.*|placeholder|example.*|your[-_ ].*|<.*>|\$\{.*\}|\$[A-Z_]+|'
                         r'tobemodified|dummy|sample|null|none|redacted|\*{3,}|x{3,}|test|password|secret|'
                         r'token|changeme|admin|root|insert.*|todo.*|fixme.*)$')


def get(url, raw=False):
    """Return (status, body). body=str (raw) or parsed JSON. Never raises."""
    r = urllib.request.Request(url, headers=H)
    for attempt in (1, 2, 3):
        try:
            with urllib.request.urlopen(r, timeout=40, context=CTX) as resp:
                data = resp.read()
                return resp.status, (data if raw else json.loads(data))
        except urllib.error.HTTPError as e:
            if e.code in (429, 500, 502, 503) and attempt < 3:
                time.sleep(2 * attempt); continue
            return e.code, ''
        except Exception:
            if attempt < 3:
                time.sleep(2 * attempt); continue
            return 0, 'exception'
    return 0, 'exception'


def scan_file(pid, path):
    url = f'{BASE}/api/v4/projects/{pid}/repository/files/{urllib.parse.quote(path, safe="")}/raw?ref=HEAD'
    st, body = get(url, raw=True)
    if st != 200 or not isinstance(body, (bytes, bytearray)):
        return None, st
    try:
        text = body.decode('utf-8', errors='ignore')
    except Exception:
        return None, 'decode'
    hits = []
    for name, rx in RX:
        for m in rx.finditer(text):
            if name == 'assign':
                val = m.group(2)
                if PLACEHOLDER.match(val):
                    continue
                line_start = text.rfind('\n', 0, m.start()) + 1
                line_end = text.find('\n', m.end())
                line = text[line_start: line_end if line_end != -1 else len(text)][:500]
                hits.append({'kind': 'assign', 'key': m.group(1)[:40], 'value': val, 'line': line})
            else:
                hits.append({'kind': name, 'value': m.group(0)[:200]})
    return hits, st


def scan_project(proj):
    pid, path, default_branch = proj['id'], proj['path'], proj.get('default_branch')
    ckpt_file = CKPT / f'{pid}.json'
    if ckpt_file.exists():
        return pid, 'cached'
    # full recursive tree with pagination; failure anywhere = project error (pitfall 11)
    files, page = [], 1
    while True:
        st, batch = get(f'{BASE}/api/v4/projects/{pid}/repository/tree?recursive=true&per_page=100&page={page}')
        if st != 200:
            return pid, f'error: tree page {page} status {st}'
        if not batch:
            break
        files += [f['path'] for f in batch if f.get('type') == 'blob']
        if len(batch) < 100:
            break
        page += 1
    if not files and default_branch:
        return pid, 'anomaly: 0 files with default_branch'  # pitfall 16 — not checkpointed
    cands = [f for f in files
             if (TEXT_EXT.search(f) or f.lower().endswith(('.env', '.env.old', '.env.bak', 'dockerfile', '.htpasswd')))
             and not SKIP_DIR.search(f) and not SKIP_NAME.search(f)]
    # extra: dotfiles like .env with no ext
    cands += [f for f in files if re.search(r'/(?:\.env[\w.]*)$', f, re.I) and f not in cands]
    findings, errors = [], 0
    for f in cands[:400]:  # generous cap; monorepos truncated get flagged below
        st_head = None
        # size guard via raw fetch anyway (GitLab raw has no HEAD size here); files >MAX skipped post-read
        hits, st = scan_file(pid, f)
        if hits is None:
            if st not in (404,):
                errors += 1
            continue
        if hits:
            findings.append({'file': f, 'hits': hits})
    rec = {'proj': path, 'id': pid, 'default_branch': default_branch,
           'total_files': len(files), 'scanned': len(cands), 'errors': errors,
           'truncated': len(cands) > 400, 'findings': findings}
    ckpt_file.write_text(json.dumps(rec, ensure_ascii=False))
    return pid, f'done: {len(files)} files, {len(cands)} scanned, {len(findings)} finding-files'


def main():
    CKPT.mkdir(exist_ok=True)
    projects = json.loads((ROOT / 'L2.json').read_text())['projects']
    # need default_branch -> fetch project list fresh (cheap, 1 call)
    st, plist = get(f'{BASE}/api/v4/projects?membership=true&per_page=100')
    meta = {p['id']: p.get('default_branch') for p in plist} if st == 200 else {}
    for p in projects:
        p['default_branch'] = meta.get(p['id'])
    log = LOG.open('a')
    t0 = time.time()
    log.write(f'=== run {time.strftime("%Y-%m-%d %H:%M:%S UTC", time.gmtime())} projects={len(projects)} ===\n')
    done = 0
    with ThreadPoolExecutor(max_workers=WORKERS) as ex:
        futs = {ex.submit(scan_project, p): p for p in projects}
        for f in as_completed(futs):
            p = futs[f]
            try:
                pid, msg = f.result()
            except Exception as e:
                pid, msg = p['id'], f'exception: {e}'
            done += 1
            line = f'[{done}/{len(projects)}] {p["path"]} -> {msg}'
            log.write(line + '\n'); log.flush()
            print(line, flush=True)
    # aggregate
    all_recs = []
    for cf in sorted(CKPT.glob('*.json')):
        try:
            all_recs.append(json.loads(cf.read_text()))
        except Exception:
            pass
    OUT_JSON.write_text(json.dumps({'base': BASE, 'scanned_at': time.strftime('%Y-%m-%d %H:%M:%S UTC', time.gmtime()),
                                    'projects': all_recs}, ensure_ascii=False, indent=1))
    with OUT_TSV.open('w') as fh:
        fh.write('proj\tfile\tkind\tkey\tvalue\tline\n')
        def _t(v): return ('' if v is None else str(v)).replace('\r', '\\r').replace('\n', '\\n').replace('\t', '\\t')
        for r in all_recs:
            for fnd in r.get('findings', []):
                for h in fnd['hits']:
                    fh.write('\t'.join([_t(r['proj']), _t(fnd['file']), _t(h.get('kind')),
                                        _t(h.get('key', '')), _t(h.get('value', '')), _t(h.get('line', ''))]) + '\n')
    n_find = sum(len(r.get('findings', [])) for r in all_recs)
    log.write(f'=== done in {time.time()-t0:.0f}s: {len(all_recs)} projects checkpointed, {n_find} finding-files ===\n')
    print(f'[+] DONE: {len(all_recs)}/{len(projects)} projects, {n_find} finding-files -> {OUT_JSON} / {OUT_TSV}')


if __name__ == '__main__':
    main()
