#!/usr/bin/env python3
"""Light census for the 12 non-cdn visionstory buckets (cdn already covered by
894,972-object partial dump). Per-bucket: object count, total bytes, LastModified
range. Read-only ListObjectsV2. Output: census_full/_others12.json"""
import hashlib, hmac, datetime, ssl, urllib.request, urllib.error, urllib.parse
import json, os, re, time
AK="AKIAWBJWGCCLDPFUDWO6"; SK="64ikpW0jhL/sYoKRE8kcIE8uDjdcO9QwC0ITnFWQ"; SVC="s3"
OUT="/root/ir-assessment/redteam/gitlab_visionstory_cn/visionstory_s3/census_full"
CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE
BUCKETS=[("transfer.visionstory","us-west-2"),("ses-inbox-415114858646","us-west-2"),
 ("visionstory-logs","us-west-2"),("webapp-visionstory","us-west-2"),
 ("web-visionstory-ai","us-west-2"),("web.darlite.me","us-west-2"),
 ("webapp-darlite","us-west-2"),("affiliate-visionstory-ai","us-west-2"),
 ("affiliate-darlite-me","us-west-2"),("cdn-visionstory-logs","us-west-2"),
 ("proxy-web-static-prod-us-west-2","us-west-2"),("proxy-web-static-prod","us-east-1")]
def sigv4(method,host,path,query="",region="us-west-2",body=b""):
    t=datetime.datetime.now(datetime.timezone.utc); amz=t.strftime("%Y%m%dT%H%M%SZ"); day=t.strftime("%Y%m%d")
    ph=hashlib.sha256(body).hexdigest()
    headers={"host":host,"x-amz-content-sha256":ph,"x-amz-date":amz}
    signed=";".join(sorted(headers))
    pairs=sorted(urllib.parse.parse_qsl(query,keep_blank_values=True))
    qs="&".join(f"{urllib.parse.quote(k,safe='-_.~')}={urllib.parse.quote(v,safe='-_.~')}" for k,v in pairs)
    canon=f"{method}\n{path}\n{qs}\n"+"".join(f"{k}:{v}\n" for k,v in sorted(headers.items()))+f"\n{signed}\n{ph}"
    scope=f"{day}/{region}/{SVC}/aws4_request"
    sts=f"AWS4-HMAC-SHA256\n{amz}\n{scope}\n{hashlib.sha256(canon.encode()).hexdigest()}"
    def h(k,m): return hmac.new(k,m.encode(),hashlib.sha256).digest()
    ks=h(h(h(h(("AWS4"+SK).encode(),day),region),SVC),"aws4_request")
    sig=hmac.new(ks,sts.encode(),hashlib.sha256).hexdigest()
    auth=f"AWS4-HMAC-SHA256 Credential={AK}/{scope}, SignedHeaders={signed}, Signature={sig}"
    url=f"https://{host}{path}"+(f"?{query}" if query else "")
    req=urllib.request.Request(url,headers={**headers,"Authorization":auth},method=method)
    for a in range(4):
        try:
            with urllib.request.urlopen(req,timeout=60,context=CTX) as r: return r.status,r.read().decode("utf-8","replace")
        except urllib.error.HTTPError as e: return e.code,e.read().decode("utf-8","replace")[:300]
        except Exception as e:
            if a==3: return -1,str(e)[:150]
            time.sleep(2**a)
def parse(xml):
    out=[]
    for m in re.finditer(r"<Contents>(.*?)</Contents>",xml,re.S):
        b=m.group(1); g=lambda t:(re.search(f"<{t}>(.*?)</{t}>",b,re.S) or [None,None])[1]
        out.append({"key":g("Key"),"last_modified":g("LastModified"),"size":int(g("Size") or 0)})
    return out
def census(bucket,region):
    host=f"{bucket}.s3.{region}.amazonaws.com"
    n=nb=0; old=new=None; tok=""; pages=0
    while True:
        q="list-type=2&max-keys=1000"+(f"&continuation-token={urllib.parse.quote(tok)}" if tok else "")
        st,body=sigv4("GET",host,"/",q,region); pages+=1
        if st==301:
            m=re.search(r"<Endpoint>(.*?)</Endpoint>",body)
            if m:
                host=m.group(1); r2=re.search(r"s3[.-]([a-z-]+-\d+)\.amazonaws",host)
                if r2: region=r2.group(1)
                continue
        if st!=200: return {"bucket":bucket,"error":f"{st} {body[:150]}"}
        for o in parse(body):
            n+=1; nb+=o["size"]; lm=o["last_modified"]
            old=lm if (old is None or (lm and lm<old)) else old
            new=lm if (new is None or (lm and lm>new)) else new
        if "<IsTruncated>true</IsTruncated>" in body:
            m=re.search(r"<NextContinuationToken>(.*?)</NextContinuationToken>",body)
            tok=m.group(1) if m else ""
            if not tok: break
        else: break
    return {"bucket":bucket,"region":region,"objects":n,"bytes":nb,"oldest":old,"newest":new,"pages":pages}
res=[]
for b,r in BUCKETS:
    d=census(b,r); res.append(d)
    if "error" in d: print(f"{b}: ERROR {d['error']}",flush=True)
    else: print(f"{b}: {d['objects']:>7} obj {d['bytes']/1e9:>9.2f} GB  {str(d['oldest'])[:10]}..{str(d['newest'])[:10]} ({d['pages']} pages)",flush=True)
json.dump(res,open(os.path.join(OUT,"_others12.json"),"w"),indent=1)
print("saved _others12.json")
