104 lines
5.9 KiB
Python
104 lines
5.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Summaries of the F9 census outputs. summarise.py edges <part.tsv...> | hotset <part.tsv...>"""
|
|
import sys
|
|
from collections import Counter, defaultdict
|
|
|
|
MIN_DISTINCT = 245760
|
|
MAX_SAT = 164
|
|
BIAS = 136
|
|
|
|
def edges(paths):
|
|
cand = 0; agree = 0; dis = []; seeds = 0; chosen_differs = 0; cf_exh = 0; mh_exh = 0
|
|
first_fail = Counter(); rejected_cf = 0; rejected_mh = 0
|
|
# margins of accepted candidates on each stand-in: min distinct, max sat, max bias, max near_const
|
|
marg = {"cf": [0, 0, 0, 10**9], "mh": [0, 0, 0, 10**9]} # max sat, max bias, max near_const, min distinct
|
|
diff_stats = defaultdict(list)
|
|
for p in paths:
|
|
for line in open(p):
|
|
if line.startswith("#"):
|
|
continue
|
|
f = line.rstrip("\n").split("\t")
|
|
if f[0] == "S":
|
|
seeds += 1
|
|
a, b = int(f[2]), int(f[3])
|
|
if a < 0: cf_exh += 1
|
|
if b < 0: mh_exh += 1
|
|
if a != b: chosen_differs += 1
|
|
continue
|
|
cand += 1
|
|
cf, mh = f[2:10], f[10:18]
|
|
if cf[1] == "static":
|
|
first_fail["static"] += 1; rejected_cf += 1; rejected_mh += 1; agree += 1
|
|
continue
|
|
if cf[0] == "R": rejected_cf += 1; first_fail["cf:" + cf[1]] += 1
|
|
if mh[0] == "R": rejected_mh += 1; first_fail["mh:" + mh[1]] += 1
|
|
for name, m in (("cf", cf), ("mh", mh)):
|
|
if m[0] == "A":
|
|
s = marg[name]
|
|
s[0] = max(s[0], int(m[5])); s[1] = max(s[1], int(m[6])); s[2] = max(s[2], int(m[3])); s[3] = min(s[3], int(m[7]))
|
|
if f[18] == "1":
|
|
agree += 1
|
|
else:
|
|
dis.append((int(f[0]), int(f[1]), cf, mh))
|
|
print(f"seeds {seeds}, candidates {cand}, agree {agree}, disagree {len(dis)}, seeds whose chosen attempt differs {chosen_differs}, exhausted cf {cf_exh} mh {mh_exh}")
|
|
print(f"rejected: closed-form {rejected_cf}, memory-hard {rejected_mh}; first failing condition counts: {dict(first_fail)}")
|
|
for name in ("cf", "mh"):
|
|
s = marg[name]
|
|
print(f"accepted margins on {name}: max saturated {s[0]} (limit {MAX_SAT}), max bias {s[1]} (limit {BIAS}), max near-constant |ones-1024| {s[2]} (1024 = constant), min distinct sum {s[3]} (bound {MIN_DISTINCT}, mean {s[3]/2048:.3f})")
|
|
print("disagreements (seed, attempt, side that accepts, condition, closed-form metrics, memory-hard metrics):")
|
|
kinds = Counter()
|
|
for seed, att, cf, mh in dis:
|
|
side = "closed-form accepts" if cf[0] == "A" else "memory-hard accepts"
|
|
cond = mh[1] if cf[0] == "A" else cf[1]
|
|
kinds[(side, cond)] += 1
|
|
print(f" {seed}\t{att}\t{side}\t{cond}\tcf: cb={cf[2]} nc={cf[3]} lc={cf[4]} sat={cf[5]} bias={cf[6]} dist={cf[7]}\tmh: cb={mh[2]} nc={mh[3]} lc={mh[4]} sat={mh[5]} bias={mh[6]} dist={mh[7]}")
|
|
print("by kind:", dict(kinds))
|
|
|
|
def hotset(paths):
|
|
n = 0; prog = set(); any_hot = Counter(); flagged = Counter(); worst = {}
|
|
taint = Counter(); taint_total_max = 0
|
|
share_hist = Counter(); const_max = 0; skew_max = 0; mind_min = 10**9; top_max = 0
|
|
hot_programs = set(); flag_programs = set()
|
|
worst_share = (0.0, None)
|
|
mind_hist = Counter(); top_hist = Counter()
|
|
for p in paths:
|
|
for line in open(p):
|
|
if line.startswith("#"):
|
|
continue
|
|
f = line.rstrip("\n").split("\t")
|
|
if len(f) < 20:
|
|
continue
|
|
seed, attempt, init = int(f[0]), int(f[1]), f[2]
|
|
n += 1; prog.add(seed)
|
|
t0, tt = int(f[3]), int(f[4])
|
|
if init == "seed":
|
|
taint[t0] += 1; taint_total_max = max(taint_total_max, tt)
|
|
cb, sk, mind, top = int(f[5]), int(f[6]), int(f[7]), int(f[8])
|
|
const_max = max(const_max, cb); skew_max = max(skew_max, sk); mind_min = min(mind_min, mind); top_max = max(top_max, top)
|
|
share = float(f[18]); anyh = f[19] == "hot"; flag = f[20] == "FLAG"
|
|
if init == "seed":
|
|
mind_hist[min(mind // 8 * 8, 2048)] += 1
|
|
top_hist[1 if top <= 1 else 2 if top <= 2 else 4 if top <= 4 else 8 if top <= 8 else 16 if top <= 16 else 64 if top <= 64 else 256 if top <= 256 else 2048] += 1
|
|
if anyh: any_hot[init] += 1; hot_programs.add(seed)
|
|
if flag: flagged[init] += 1; flag_programs.add(seed)
|
|
b = "0" if share == 0 else ("<0.1%" if share < 0.001 else "<0.5%" if share < 0.005 else "<1%" if share < 0.01 else ">=1%")
|
|
share_hist[(init, b)] += 1
|
|
if share > worst_share[0]: worst_share = (share, (seed, attempt, init, line.strip()))
|
|
print(f"rows {n}, programs {len(prog)}; programs with any hot bucket (strict) {len(hot_programs)} (rows by init {dict(any_hot)}); programs flagged at the gate reading (hot share >= 1% or 7 constant bits) {len(flag_programs)} (rows {dict(flagged)})")
|
|
print(f"max site constant bits {const_max}, max skewed bits {skew_max}, min site distinct {mind_min}, max site top-address count {top_max}")
|
|
print("hot share histogram (init, band) -> rows:", dict(sorted(share_hist.items())))
|
|
print("worst hot share:", worst_share)
|
|
print("init-determined loads in iteration 0 (count -> programs):", dict(sorted(taint.items())), "max total", taint_total_max)
|
|
print("min site distinct (seed init), floor-8 bins -> programs:", dict(sorted(mind_hist.items())))
|
|
print("max site top-address count (seed init), upper bin -> programs:", dict(sorted(top_hist.items())))
|
|
share_sum = 0.0; share_n = 0
|
|
for p in paths:
|
|
for line in open(p):
|
|
if line.startswith("#"): continue
|
|
f = line.rstrip("\n").split("\t")
|
|
if len(f) >= 20 and f[2] == "seed":
|
|
share_sum += float(f[18]); share_n += 1
|
|
print(f"mean hot share over programs (seed init): {share_sum / max(share_n, 1):.6f}")
|
|
|
|
if __name__ == "__main__":
|
|
{"edges": edges, "hotset": hotset}[sys.argv[1]](sys.argv[2:])
|