igneum/tools/attack/f9-grind/summarise.py
2026-10-07 10:02:44 +00:00

104 lines
5.9 KiB
Python

#!/usr/bin/env python3
"""Summaries of the F9 census outputs. summarise.py edges <part.tsv...> | hotset <part.tsv...>"""
import sys
from collections import Counter, defaultdict
MIN_DISTINCT = 245760
MAX_SAT = 164
BIAS = 136
def edges(paths):
cand = 0; agree = 0; dis = []; seeds = 0; chosen_differs = 0; cf_exh = 0; mh_exh = 0
first_fail = Counter(); rejected_cf = 0; rejected_mh = 0
# margins of accepted candidates on each stand-in: min distinct, max sat, max bias, max near_const
marg = {"cf": [0, 0, 0, 10**9], "mh": [0, 0, 0, 10**9]} # max sat, max bias, max near_const, min distinct
diff_stats = defaultdict(list)
for p in paths:
for line in open(p):
if line.startswith("#"):
continue
f = line.rstrip("\n").split("\t")
if f[0] == "S":
seeds += 1
a, b = int(f[2]), int(f[3])
if a < 0: cf_exh += 1
if b < 0: mh_exh += 1
if a != b: chosen_differs += 1
continue
cand += 1
cf, mh = f[2:10], f[10:18]
if cf[1] == "static":
first_fail["static"] += 1; rejected_cf += 1; rejected_mh += 1; agree += 1
continue
if cf[0] == "R": rejected_cf += 1; first_fail["cf:" + cf[1]] += 1
if mh[0] == "R": rejected_mh += 1; first_fail["mh:" + mh[1]] += 1
for name, m in (("cf", cf), ("mh", mh)):
if m[0] == "A":
s = marg[name]
s[0] = max(s[0], int(m[5])); s[1] = max(s[1], int(m[6])); s[2] = max(s[2], int(m[3])); s[3] = min(s[3], int(m[7]))
if f[18] == "1":
agree += 1
else:
dis.append((int(f[0]), int(f[1]), cf, mh))
print(f"seeds {seeds}, candidates {cand}, agree {agree}, disagree {len(dis)}, seeds whose chosen attempt differs {chosen_differs}, exhausted cf {cf_exh} mh {mh_exh}")
print(f"rejected: closed-form {rejected_cf}, memory-hard {rejected_mh}; first failing condition counts: {dict(first_fail)}")
for name in ("cf", "mh"):
s = marg[name]
print(f"accepted margins on {name}: max saturated {s[0]} (limit {MAX_SAT}), max bias {s[1]} (limit {BIAS}), max near-constant |ones-1024| {s[2]} (1024 = constant), min distinct sum {s[3]} (bound {MIN_DISTINCT}, mean {s[3]/2048:.3f})")
print("disagreements (seed, attempt, side that accepts, condition, closed-form metrics, memory-hard metrics):")
kinds = Counter()
for seed, att, cf, mh in dis:
side = "closed-form accepts" if cf[0] == "A" else "memory-hard accepts"
cond = mh[1] if cf[0] == "A" else cf[1]
kinds[(side, cond)] += 1
print(f" {seed}\t{att}\t{side}\t{cond}\tcf: cb={cf[2]} nc={cf[3]} lc={cf[4]} sat={cf[5]} bias={cf[6]} dist={cf[7]}\tmh: cb={mh[2]} nc={mh[3]} lc={mh[4]} sat={mh[5]} bias={mh[6]} dist={mh[7]}")
print("by kind:", dict(kinds))
def hotset(paths):
n = 0; prog = set(); any_hot = Counter(); flagged = Counter(); worst = {}
taint = Counter(); taint_total_max = 0
share_hist = Counter(); const_max = 0; skew_max = 0; mind_min = 10**9; top_max = 0
hot_programs = set(); flag_programs = set()
worst_share = (0.0, None)
mind_hist = Counter(); top_hist = Counter()
for p in paths:
for line in open(p):
if line.startswith("#"):
continue
f = line.rstrip("\n").split("\t")
if len(f) < 20:
continue
seed, attempt, init = int(f[0]), int(f[1]), f[2]
n += 1; prog.add(seed)
t0, tt = int(f[3]), int(f[4])
if init == "seed":
taint[t0] += 1; taint_total_max = max(taint_total_max, tt)
cb, sk, mind, top = int(f[5]), int(f[6]), int(f[7]), int(f[8])
const_max = max(const_max, cb); skew_max = max(skew_max, sk); mind_min = min(mind_min, mind); top_max = max(top_max, top)
share = float(f[18]); anyh = f[19] == "hot"; flag = f[20] == "FLAG"
if init == "seed":
mind_hist[min(mind // 8 * 8, 2048)] += 1
top_hist[1 if top <= 1 else 2 if top <= 2 else 4 if top <= 4 else 8 if top <= 8 else 16 if top <= 16 else 64 if top <= 64 else 256 if top <= 256 else 2048] += 1
if anyh: any_hot[init] += 1; hot_programs.add(seed)
if flag: flagged[init] += 1; flag_programs.add(seed)
b = "0" if share == 0 else ("<0.1%" if share < 0.001 else "<0.5%" if share < 0.005 else "<1%" if share < 0.01 else ">=1%")
share_hist[(init, b)] += 1
if share > worst_share[0]: worst_share = (share, (seed, attempt, init, line.strip()))
print(f"rows {n}, programs {len(prog)}; programs with any hot bucket (strict) {len(hot_programs)} (rows by init {dict(any_hot)}); programs flagged at the gate reading (hot share >= 1% or 7 constant bits) {len(flag_programs)} (rows {dict(flagged)})")
print(f"max site constant bits {const_max}, max skewed bits {skew_max}, min site distinct {mind_min}, max site top-address count {top_max}")
print("hot share histogram (init, band) -> rows:", dict(sorted(share_hist.items())))
print("worst hot share:", worst_share)
print("init-determined loads in iteration 0 (count -> programs):", dict(sorted(taint.items())), "max total", taint_total_max)
print("min site distinct (seed init), floor-8 bins -> programs:", dict(sorted(mind_hist.items())))
print("max site top-address count (seed init), upper bin -> programs:", dict(sorted(top_hist.items())))
share_sum = 0.0; share_n = 0
for p in paths:
for line in open(p):
if line.startswith("#"): continue
f = line.rstrip("\n").split("\t")
if len(f) >= 20 and f[2] == "seed":
share_sum += float(f[18]); share_n += 1
print(f"mean hot share over programs (seed init): {share_sum / max(share_n, 1):.6f}")
if __name__ == "__main__":
{"edges": edges, "hotset": hotset}[sys.argv[1]](sys.argv[2:])