#!/usr/bin/env python3
"""Fast structural census: for each bucket -> region (via Expect: 100-continue probe
or first list call), top-level prefixes with sample keys + LastModified, and
first-1000-keys aggregate. NO deep pagination. Decision-support grade: shows
what data classes exist per bucket, approximate scale (from prefix sample), and
date ranges. Read-only."""
import hashlib, hmac, datetime, ssl, urllib.request, urllib.error, urllib.parse
import json, os, re, time
AK="AKIAWBJWGCCLDPFUDWO6"; SK="64ikpW0jhL/sYoKRE8kcIE8uDjdcO9QwC0ITnFWQ"; SVC="s3"
OUT="/root/ir-assessment/redteam/gitlab_visionstory_cn/visionstory_s3/census_full"
CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE
BUCKETS=["transfer.visionstory","ses-inbox-415114858646","visionstory-logs",
 "webapp-visionstory","web-visionstory-ai","web.darlite.me","webapp-darlite",
 "affiliate-visionstory-ai","affiliate-darlite-me","cdn-visionstory-logs",
 "proxy-web-static-prod-us-west-2","proxy-web-static-prod"]
def sigv4(method,host,path,query="",region="us-west-2",body=b""):
    t=datetime.datetime.now(datetime.timezone.utc); amz=t.strftime("%Y%m%dT%H%M%SZ"); day=t.strftime("%Y%m%d")
    ph=hashlib.sha256(body).hexdigest()
    headers={"host":host,"x-amz-content-sha256":ph,"x-amz-date":amz}
    signed=";".join(sorted(headers))
    pairs=sorted(urllib.parse.parse_qsl(query,keep_blank_values=True))
    qs="&".join(f"{urllib.parse.quote(k,safe='-_.~')}={urllib.parse.quote(v,safe='-_.~')}" for k,v in pairs)
    canon=f"{method}\n{path}\n{qs}\n"+"".join(f"{k}:{v}\n" for k,v in sorted(headers.items()))+f"\n{signed}\n{ph}"
    scope=f"{day}/{region}/{SVC}/aws4_request"
    sts=f"AWS4-HMAC-SHA256\n{amz}\n{scope}\n{hashlib.sha256(canon.encode()).hexdigest()}"
    def h(k,m): return hmac.new(k,m.encode(),hashlib.sha256).digest()
    ks=h(h(h(h(("AWS4"+SK).encode(),day),region),SVC),"aws4_request")
    sig=hmac.new(ks,sts.encode(),hashlib.sha256).hexdigest()
    auth=f"AWS4-HMAC-SHA256 Credential={AK}/{scope}, SignedHeaders={signed}, Signature={sig}"
    url=f"https://{host}{path}"+(f"?{query}" if query else "")
    req=urllib.request.Request(url,headers={**headers,"Authorization":auth},method=method)
    for a in range(3):
        try:
            with urllib.request.urlopen(req,timeout=30,context=CTX) as r: return r.status,dict(r.headers),r.read().decode("utf-8","replace")
        except urllib.error.HTTPError as e: return e.code,dict(e.headers),e.read().decode("utf-8","replace")[:400]
        except Exception as e:
            if a==2: return -1,{},str(e)[:150]
            time.sleep(1)
def parse_common(xml):
    out=[]
    for m in re.finditer(r"<Contents>(.*?)</Contents>",xml,re.S):
        b=m.group(1); g=lambda t:(re.search(f"<{t}>(.*?)</{t}>",b,re.S) or [None,None])[1]
        out.append({"key":g("Key"),"last_modified":g("LastModified"),"size":int(g("Size") or 0)})
    return out
def one(bucket):
    # discover region: hit global endpoint, follow 301
    region="us-west-2"; host=f"{bucket}.s3.{region}.amazonaws.com"
    st,hd,body=sigv4("GET",host,"/","list-type=2&max-keys=1000&delimiter=%2F",region)
    if st in (301,400):
        # region from x-amz-bucket-region header or Endpoint body
        r=hd.get("x-amz-bucket-region") or hd.get("X-Amz-Bucket-Region")
        if not r:
            m=re.search(r"<Region>(.*?)</Region>",body); r=m.group(1) if m else None
        if r:
            region=r; host=f"{bucket}.s3.{region}.amazonaws.com"
            st,hd,body=sigv4("GET",host,"/","list-type=2&max-keys=1000&delimiter=%2F",region)
    if st!=200:
        return {"bucket":bucket,"error":f"{st} {body[:200]}"}
    prefixes=re.findall(r"<CommonPrefixes>.*?<Prefix>(.*?)</Prefix>.*?</CommonPrefixes>",body,re.S)
    keys=parse_common(body)
    truncated="<IsTruncated>true</IsTruncated>" in body
    kb=sum(k["size"] for k in keys)
    dates=[k["last_modified"] for k in keys if k["last_modified"]]
    # per-prefix: list first page of each to get sample + date range
    pd={}
    for p in prefixes[:15]:
        st2,hd2,b2=sigv4("GET",host,"/",f"list-type=2&max-keys=1000&prefix={urllib.parse.quote(p)}",region)
        if st2!=200: pd[p]={"error":st2}; continue
        ks=parse_common(b2)
        pb=sum(k["size"] for k in ks)
        ds=[k["last_modified"] for k in ks if k["last_modified"]]
        pd[p]={"sample_count":len(ks),"sample_bytes":pb,
               "oldest":min(ds) if ds else None,"newest":max(ds) if ds else None,
               "truncated":"<IsTruncated>true</IsTruncated>" in b2,
               "samples":[k["key"] for k in ks[:5]]}
    return {"bucket":bucket,"region":region,
            "top_prefixes":prefixes,"root_keys_first_page":len(keys),
            "root_bytes_first_page":kb,
            "root_oldest":min(dates) if dates else None,
            "root_newest":max(dates) if dates else None,
            "root_listing_truncated":truncated,
            "prefix_detail":pd}
res=[]
for b in BUCKETS:
    d=one(b); res.append(d)
    if "error" in d: print(f"{b}: ERROR {d['error'][:120]}",flush=True)
    else:
        print(f"{b} [{d['region']}]: {len(d['top_prefixes'])} prefixes, root+sample {d['root_bytes_first_page']/1e6:.1f}MB first-page, trunc={d['root_listing_truncated']}",flush=True)
json.dump(res,open(os.path.join(OUT,"_fast_structure.json"),"w"),indent=1,ensure_ascii=False)
print("saved _fast_structure.json")
