#!/usr/bin/env python3
"""Content sampling from cdn-visionstory big prefixes (operator 'go' 2026-08-18).

Per prefix: 20 random + 5 newest objects, size-capped (<15MB each) to keep
traffic tiny. Saves to samples/ with sha256 + INDEX.json. Read-only GET calls.
"""
import hashlib, hmac, datetime, ssl, urllib.request, urllib.error, urllib.parse
import json, os, random, time

AK="AKIAWBJWGCCLDPFUDWO6"; SK="64ikpW0jhL/sYoKRE8kcIE8uDjdcO9QwC0ITnFWQ"; SVC="s3"
REGION="us-west-2"; BUCKET="cdn-visionstory"
BASE="/root/ir-assessment/redteam/gitlab_visionstory_cn/visionstory_s3"
DUMP=f"{BASE}/census_full/cdn-visionstory_objects_partial_859k.jsonl"
OUT=f"{BASE}/samples"
SIZE_CAP = 15*1024*1024   # 15MB per object hard cap
CTX=ssl.create_default_context(); CTX.check_hostname=False; CTX.verify_mode=ssl.CERT_NONE
random.seed(42)

BIG = ["podcast","avatar_story","mat_design","ppt_video","ad_video","openapi_asset","creative_video"]

def sigv4(method,host,path,query="",region=REGION,body=b""):
    t=datetime.datetime.now(datetime.timezone.utc); amz=t.strftime("%Y%m%dT%H%M%SZ"); day=t.strftime("%Y%m%d")
    ph=hashlib.sha256(body).hexdigest()
    headers={"host":host,"x-amz-content-sha256":ph,"x-amz-date":amz}
    signed=";".join(sorted(headers))
    pairs=sorted(urllib.parse.parse_qsl(query,keep_blank_values=True))
    qs="&".join(f"{urllib.parse.quote(k,safe='-_.~')}={urllib.parse.quote(v,safe='-_.~')}" for k,v in pairs)
    canon=f"{method}\n{path}\n{qs}\n"+"".join(f"{k}:{v}\n" for k,v in sorted(headers.items()))+f"\n{signed}\n{ph}"
    scope=f"{day}/{region}/{SVC}/aws4_request"
    sts=f"AWS4-HMAC-SHA256\n{amz}\n{scope}\n{hashlib.sha256(canon.encode()).hexdigest()}"
    def h(k,m): return hmac.new(k,m.encode(),hashlib.sha256).digest()
    ks=h(h(h(h(("AWS4"+SK).encode(),day),region),SVC),"aws4_request")
    sig=hmac.new(ks,sts.encode(),hashlib.sha256).hexdigest()
    auth=f"AWS4-HMAC-SHA256 Credential={AK}/{scope}, SignedHeaders={signed}, Signature={sig}"
    url=f"https://{host}{path}"+(f"?{query}" if query else "")
    req=urllib.request.Request(url,headers={**headers,"Authorization":auth},method=method)
    for a in range(3):
        try:
            with urllib.request.urlopen(req,timeout=60,context=CTX) as r:
                return r.status, r.read()
        except urllib.error.HTTPError as e:
            return e.code, e.read()[:300]
        except Exception as e:
            if a==2: return -1, str(e).encode()[:150]
            time.sleep(1)

# ---- pick samples from local dump ----
bypref = {p: [] for p in BIG}
for line in open(DUMP):
    o = json.loads(line)
    p = o["key"].lstrip("/").split("/",1)[0]
    if p in bypref and 0 < o["size"] <= SIZE_CAP:
        bypref[p].append((o["key"], o["size"], o["last_modified"]))

picks = []
for p in BIG:
    lst = bypref[p]
    if not lst: continue
    newest = sorted(lst, key=lambda x: x[2], reverse=True)[:5]
    rnd = random.sample(lst, min(20, len(lst)))
    seen = set()
    for k,s,lm in newest + rnd:
        if k not in seen:
            seen.add(k); picks.append((p,k,s,lm))
print(f"selected {len(picks)} objects across {len(BIG)} prefixes")

# ---- download ----
os.makedirs(OUT, exist_ok=True)
host=f"{BUCKET}.s3.{REGION}.amazonaws.com"
index=[]; total=0; fails=0
for p,key,size,lm in picks:
    st, data = sigv4("GET", host, "/"+urllib.parse.quote(key))
    if st != 200:
        fails+=1; print(f"  FAIL {st} {key}"); continue
    sha = hashlib.sha256(data).hexdigest()
    fn = f"{p}__{key.replace('/','_')[-80:]}"
    with open(os.path.join(OUT,fn),"wb") as fh: fh.write(data)
    index.append({"prefix":p,"key":key,"size":size,"last_modified":lm,
                  "sha256":sha,"file":fn,"content_type_guess":None})
    total+=len(data)
    print(f"  [{p}] {len(data):>9,} B  {key[:70]}")

json.dump({"bucket":BUCKET,"account":"415114858646","purpose":"content sampling + proof-of-access",
           "size_cap_per_object":SIZE_CAP,"objects":index,"total_bytes":total,
           "fails":fails,"generated":datetime.datetime.now(datetime.timezone.utc).isoformat()},
          open(os.path.join(OUT,"INDEX.json"),"w"), indent=1, ensure_ascii=False)
print(f"\ndone: {len(index)} files, {total/1e6:.1f} MB, fails={fails}")
print("-> samples/INDEX.json")
