igneum/tools/ci/p01-vectors.py

262 lines
23 KiB
Python
Executable file

#!/usr/bin/env python3
"""P01 part A, the million-vector campaign per backend (the founder's Test and Acceptance Standard, POW-01; 8 October 2026).
The GPU side is the real worker (the kit's CUDA, OpenCL or Metal binary) driven as the miner drives it: `job` lines on stdin
with an all-pass share target, so every nonce comes back as `found <job> <nonce> <hash>`; the CPU side is the reference list
the hash lane writes with `igneum-pow hash-bound --count N` (one `<nonce> <hash>` per line, hex or decimal nonce, 16-hex hash).
Every found hash is compared to the reference's hash for that nonce; the evidence JSON carries the counts, the first ten
disagreements, the pack id, the program id, the device line and the manifest sha; exit 1 on any disagreement or any nonce
the worker never answered.
tools/ci/p01-vectors.py --worker <bin> [--worker-arg=X ...] --pack <dir> --reference <file> --count 1000000
(a worker argument that itself starts with "--" must be given as --worker-arg=--serve; argparse reads "--worker-arg --serve" as two options)
[--start 0] [--job-nonces 1048576] [--prehash <64 hex>] [--manifest <sha>] --out <evidence.json>
tools/ci/p01-vectors.py --worker <bin> [--worker-arg X ...] --job-context <job-context.json> --phase 1|2|3 --out <evidence.json>
tools/ci/p01-vectors.py --worker <bin> --worker-arg=--serve --job-context <file> --phase N --prepare-first --out <evidence.json>
(a worker with no --pack flag, the Metal worker: its first pack goes by a prepare line before any job)
tools/ci/p01-vectors.py --self-test
The job context (the same-work test, docs/plans/igneum-2.0-same-work-test.md) names two packs and two references (day D and
day D+1), the prehash, the range width (range_log2, 20) and the boundary nonce; phase N runs nonces [(N-1)*2^w, N*2^w): phase 1
under day D, phase 3 under day D+1, phase 2 under day D up to the boundary nonce and day D+1 from it, no job line crossing the
boundary, and the evidence carries the boundary block (the nonce, the program id and day bytes on each side, the two hashes
either side) that the readers are compared on. A worker holds one (epoch, day) pair, so before any job the driver sends the
worker `prepare <seed bytes> <day D+1 bytes> <pack dir of day D+1>` and waits for its `prepared` line (a `prepare-failed` line or
a ready line saying `prepare 0` is an error: the day switch cannot be driven), keeps feeding day D jobs meanwhile (the worker loads
and reports the pair only between stdin lines, so a silent wait deadlocks: the 22:26 to 22:47 UK hang on the 5090), polls it with
a line a second when day D is spent, and sends the first day D+1 job only after `prepared`, as the miner does at a real day switch.
The job line is pool.rs's: `job <seq> <prehash hex> <share_target64 hex16> <start nonce> <nonces> <epoch seed bytes hex> <day bytes hex> class=<c> era=<era hex>`.
"""
import argparse, json, os, subprocess, sys, tempfile, time
def read_reference(path):
ref = {}
with open(path) as f:
for line in f:
p = line.split()
if len(p) < 2 or p[0].startswith('#'): continue
n = int(p[0], 16) if p[0].lower().startswith('0x') else int(p[0]); h = p[1].lower().replace('0x', '')
ref[n] = h.zfill(16)
return ref
def pack_fields(pack):
j = json.load(open(os.path.join(pack, 'program.json')))
cls = j.get('program_class', '?'); era = j.get('era_seed_bytes', '')
return j.get('seed_bytes', ''), j['dataset']['day_bytes'], cls, era, j.get('program_id', '?'), j.get('generator', '?')
def segments_for(a):
"""[(start, end, pack dir, reference path)] with no job crossing a segment edge; one segment for the plain form."""
if not a.job_context: return [(a.start, a.start + a.count, a.pack, a.reference)], None
jc = json.load(open(a.job_context)); w = int(jc.get('range_log2', 20)); n = int(a.phase)
if n not in (1, 2, 3): raise SystemExit('p01-vectors: --phase must be 1, 2 or 3')
base = os.path.dirname(os.path.abspath(a.job_context))
pth = lambda x: x if os.path.isabs(x) else os.path.join(base, x)
packs, refs = jc['packs'], jc['references']
s0, e0 = (n - 1) << w, n << w
if n == 1: segs = [(s0, e0, pth(packs['D']), pth(refs['D']))]
elif n == 3: segs = [(s0, e0, pth(packs['D1']), pth(refs['D1']))]
else:
b = int(jc['boundary_nonce'])
if not (s0 < b < e0): raise SystemExit(f'p01-vectors: the boundary nonce {b} is not inside phase 2 [{s0}, {e0})')
segs = [(s0, b, pth(packs['D']), pth(refs['D'])), (b, e0, pth(packs['D1']), pth(refs['D1']))]
a.start, a.count = s0, e0 - s0
if jc.get('prehash'): a.prehash = jc['prehash']
if jc.get('manifest') and not a.manifest: a.manifest = jc['manifest']
return segs, jc
def run(a):
segs, jc = segments_for(a)
ref = {}; seg_fields = []
for (ss, se, pack, rpath) in segs:
r = read_reference(rpath); ref.update({k: v for k, v in r.items() if ss <= k < se}); seg_fields.append((ss, se, pack_fields(pack)))
seed_bytes, day_bytes, cls, era, program_id, generator = seg_fields[0][2]
def fields_at(n):
for (ss, se, f) in seg_fields:
if ss <= n < se: return ss, se, f
return None
p = subprocess.Popen([a.worker] + a.worker_arg, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1)
ready = None; t0 = time.time()
for line in p.stdout:
if line.startswith('ready'): ready = line.strip(); break
if time.time() - t0 > 120: break
if not ready: p.kill(); return {'error': 'the worker never said ready'}, 2
# which segments the worker holds: a worker started on its --pack holds segment 1; a worker with no --pack flag (the Metal
# worker) takes its first pack by a prepare line (--prepare-first); the day D+1 segment always comes by prepare. The worker
# builds a pair in a background thread and loads and reports it only between stdin lines (proto-cuda/host.cu: "a finished
# nvcc step is loaded here, between lines"), as the miner's steady stream of jobs gives it; so the driver keeps feeding the
# jobs it can, never sends a job on a segment before that segment's prepared line, and when it has nothing to send it polls
# the worker with a line a second (the worker answers "info ignored", or the prepared line first). One prepare is in flight at
# a time (the worker refuses a second while one runs). A worker without prepare support cannot be driven across a boundary.
# the pack the worker holds is read from its ready line ("pack igneum-epoch/<seed>/day/<day bytes>", the CUDA and OpenCL
# workers started on --pack); a segment on any other day is prepared, whichever phase (phase 3's worker started on the
# day D pack refused every day D+1 job as a day-seed mismatch at 22:56 UK: missing 1,048,576, no disagreement). A ready
# line without a pack label means the worker holds the first segment's pack unless --prepare-first says it holds none.
rt = ready.split(); resident = next((t for t in rt[1:] if 'igneum-epoch/' in t and '/day/' in t), None)
def holds(f):
if a.prepare_first: return False
if resident is None: return f is seg_fields[0][2]
return resident.endswith('/day/' + f[1]) and ('/' + f[0] + '/') in resident
held = {i: holds(seg_fields[i][2]) for i in range(len(seg_fields))}
to_prepare = [i for i in range(len(seg_fields)) if not held[i]]; in_flight = None; prepare_sent = None; prepare_lines = []
if to_prepare and ' prepare 0' in ready: p.kill(); return {'error': f'the worker has no prepare support (ready line: {ready}); a pack by prepare or a day switch needs it'}, 2
def send_prepare():
nonlocal in_flight, prepare_sent
if in_flight is None and to_prepare:
i = to_prepare.pop(0); sb, db, c, er, _pid, _gen = seg_fields[i][2]
p.stdin.write(f"prepare {sb} {db} {segs[i][2]} class={c} era={er}\n"); p.stdin.flush(); in_flight = i; prepare_sent = time.time()
send_prepare()
got = {}; agree = 0; disagree = []; seq = 0; start = a.start; end = a.start + a.count; pending = set()
def feed():
nonlocal seq, start
while start < end and len(pending) < 4:
ss, se, (sb, db, c, er, _pid, _gen) = fields_at(start)
idx = next(i for i, (s1, e1, _f) in enumerate(seg_fields) if s1 <= start < e1)
if not held[idx]:
# no job on a segment before the worker says prepared for it; with nothing pending, poll the worker
if not pending:
if time.time() - prepare_sent > 900: p.kill(); raise SystemExit(f'p01-vectors: the worker never said prepared for segment {idx + 1} (900 s)')
time.sleep(1); p.stdin.write("ping\n"); p.stdin.flush()
return
n = min(a.job_nonces, end - start, se - start); seq += 1 # a job never crosses a segment edge (the day boundary)
p.stdin.write(f"job {seq} {a.prehash} {'f'*16} {start} {n} {sb} {db} class={c} era={er}\n"); p.stdin.flush()
pending.add(seq); start += n
feed()
for line in p.stdout:
it = line.split()
if not it: continue
if it[0] == 'found' and len(it) >= 4:
n = int(it[2]); h = it[3].lower().replace('0x', '').zfill(16); got[n] = h
r = ref.get(n)
if r == h: agree += 1
else:
if len(disagree) < 10: disagree.append({'nonce': n, 'gpu': h, 'cpu': r})
disagree_count[0] += 1
elif it[0] == 'prepared' and len(it) >= 3:
prepare_lines.append(line.strip())
for i, (_s1, _e1, f) in enumerate(seg_fields):
if f[1] == it[2] and f[0] == it[1]: held[i] = True
in_flight = None; send_prepare(); feed()
elif it[0] == 'prepare-failed':
p.kill(); return {'error': f'the worker refused a pair: {line.strip()[:300]}'}, 2
elif it[0] == 'info' and 'ignored' in line and not pending and start < end and not held[next(i for i, (s1, e1, _f) in enumerate(seg_fields) if s1 <= start < e1)]:
feed() # the poll's answer: ask again after a second
elif it[0] in ('done', 'error') and len(it) >= 2:
try: pending.discard(int(it[1]))
except ValueError: pass
if it[0] == 'error': errors.append(line.strip()[:200])
if start >= end and not pending: break
feed()
try: p.stdin.close()
except Exception: pass
p.wait(timeout=30)
missing = [n for n in range(a.start, end) if n not in got]
pack_name = os.path.basename(os.path.abspath(a.pack)) if a.pack else ', '.join(os.path.basename(os.path.abspath(s[2])) for s in segs)
ev = {'case': 'POW-01', 'profile': 'P01', 'pack': pack_name, 'program_id': program_id, 'generator': generator, 'class': cls,
'worker': a.worker, 'worker_args': a.worker_arg, 'device_line': ready, 'manifest_sha': a.manifest, 'prehash': a.prehash,
'nonces': {'start': a.start, 'count': a.count}, 'answered': len(got), 'agree': agree, 'disagree': disagree_count[0], 'missing': len(missing),
'first_disagreements': disagree, 'first_missing': missing[:10], 'worker_errors': errors[:10], 'seconds': round(time.time() - t0, 1), 'at': time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())}
if jc is not None:
ev['job_context'] = os.path.abspath(a.job_context); ev['phase'] = int(a.phase); ev['object'] = jc.get('object')
ev['segments'] = [{'start': ss, 'end': se, 'program_id': f[4], 'day_bytes': f[1], 'class': f[2], 'pack': os.path.basename(os.path.abspath(sg[2]))} for (ss, se, f), sg in zip(seg_fields, segs)]
if prepare_lines: ev['prepare_lines'] = prepare_lines; ev['prepare_line'] = prepare_lines[-1]
if len(seg_fields) == 2:
b = seg_fields[1][0]; fb, fa = seg_fields[0][2], seg_fields[1][2]
ev['boundary'] = {'nonce': b, 'program_id_before': fb[4], 'program_id_after': fa[4], 'day_bytes_before': fb[1], 'day_bytes_after': fa[1],
'hash_before': got.get(b - 1), 'hash_after': got.get(b), 'reference_before': ref.get(b - 1), 'reference_after': ref.get(b),
'switched': bool(fb[1] != fa[1] and got.get(b - 1) == ref.get(b - 1) and got.get(b) == ref.get(b))}
ev['verdict'] = 'PASS' if (agree == a.count and not disagree and not missing and (jc is None or len(seg_fields) < 2 or ev['boundary']['switched'])) else 'FAIL'
return ev, (0 if ev['verdict'] == 'PASS' else 1)
disagree_count = [0]; errors = []
def self_test():
d = tempfile.mkdtemp(); fails = 0
pack = os.path.join(d, 'pack'); os.makedirs(pack)
json.dump({'seed_bytes': 'aa' * 32, 'program_class': 'v5', 'era_seed_bytes': 'bb' * 32, 'program_id': '0x1', 'generator': 5, 'dataset': {'day_bytes': 'cc' * 19, 'log2_words': 20}}, open(os.path.join(pack, 'program.json'), 'w'))
# a fake worker: hashes nonce n as (n * 0x9e3779b97f4a7c15) mod 2^64; nonce 7 wrong when WRONG=1; nonce 9 never answered when DROP=1
w = os.path.join(d, 'worker.py')
open(w, 'w').write('''import sys, os
res0 = os.environ.get("RESIDENT", "cc" * 19)
print("ready fake-gpu 0 pack " + ("igneum-epoch/" + "aa" * 32 + "/day/" + res0 if res0 != "none" else "none") + " prepare " + ("0" if os.environ.get("NOPREPARE") == "1" else "1"), flush=True)
pending_prepare = None; prep_lines_left = 0
resident = {os.environ.get("RESIDENT", "cc" * 19)} # the pack's day (the real worker starts on its --pack); another day needs prepare first, as the real worker (day seed mismatch otherwise)
for line in sys.stdin:
p = line.split()
if pending_prepare and prep_lines_left > 0: prep_lines_left -= 1 # PREPSLOW=n: the nvcc step takes n more lines (the driver must poll)
elif pending_prepare: # as the real worker: a finished prepare is loaded and reported between lines, on the next line of any kind
q = pending_prepare; pending_prepare = None
if os.environ.get("PREPFAIL") == "1": print(f"prepare-failed {q[0]} {q[1]} nvcc exit 1", flush=True)
else: resident.add(q[1]); print(f"prepared {q[0]} {q[1]} 12.0 nvcc 10.0 cache 1.0 dataset 1.0 resident 2 programs 2 datasets", flush=True)
if p[0] == "prepare": pending_prepare = (p[1], p[2]); prep_lines_left = int(os.environ.get("PREPSLOW", "0")); print("info prepare started", flush=True); continue
if p[0] == "ping": print("info ignored: ping", flush=True); continue
if p[0] != "job": continue
seq, start, n = int(p[1]), int(p[4]), int(p[5])
if p[7] not in resident: print(f"error {seq} day seed mismatch: the job's day seed {p[7]} is not resident", flush=True); print(f"done {seq} 0 0", flush=True); continue
for k in range(start, start + n):
if os.environ.get("DROP") == "1" and k == 9: continue
h = (k * 0x9e3779b97f4a7c15) % (1 << 64)
if p[7] == "dd" * 19: h ^= 0xdd00dd00dd00dd00 # day D+1's bytes hash differently (a reader on the wrong day disagrees)
if os.environ.get("WRONG") == "1" and k == 7: h ^= 1
print(f"found {seq} {k} {h:016x}", flush=True)
print(f"done {seq} {n} 1.0", flush=True)
''')
ref = os.path.join(d, 'ref.txt'); open(ref, 'w').write(''.join(f"{k} {(k * 0x9e3779b97f4a7c15) % (1 << 64):016x}\n" for k in range(0, 20)))
base = ['--worker', sys.executable, '--worker-arg', w, '--pack', pack, '--reference', ref, '--count', '20', '--job-nonces', '8', '--manifest', 'deadbeef']
def go(env, tag):
out = os.path.join(d, f'{tag}.json'); e = dict(os.environ); e.update(env)
r = subprocess.run([sys.executable, __file__] + base + ['--out', out], env=e, capture_output=True, text=True, timeout=120); return r.returncode, json.load(open(out)) if os.path.exists(out) else {}
rc, ev = go({}, 'ok')
if not (rc == 0 and ev.get('verdict') == 'PASS' and ev['agree'] == 20 and ev['disagree'] == 0 and ev['missing'] == 0): print(f"self-test failed: a clean run was not PASS: rc={rc} {ev}"); fails = 1
rc, ev = go({'WRONG': '1'}, 'wrong')
if not (rc == 1 and ev.get('verdict') == 'FAIL' and ev['disagree'] == 1 and ev['first_disagreements'][0]['nonce'] == 7): print(f"self-test failed: one wrong hash was not a FAIL naming nonce 7: rc={rc} {ev.get('first_disagreements')}"); fails = 1
rc, ev = go({'DROP': '1'}, 'drop')
if not (rc == 1 and ev['missing'] == 1 and ev['first_missing'] == [9]): print(f"self-test failed: an unanswered nonce was not a FAIL naming nonce 9: rc={rc} {ev.get('first_missing')}"); fails = 1
# the job-context form: two packs (day D cc.., day D+1 dd..), two references, range_log2 4 (16 nonces a phase), boundary 24 inside phase 2
pack2 = os.path.join(d, 'pack2'); os.makedirs(pack2)
json.dump({'seed_bytes': 'aa' * 32, 'program_class': 'v5', 'era_seed_bytes': 'bb' * 32, 'program_id': '0x2', 'generator': 5, 'dataset': {'day_bytes': 'dd' * 19, 'log2_words': 20}}, open(os.path.join(pack2, 'program.json'), 'w'))
refD = os.path.join(d, 'refD.txt'); open(refD, 'w').write(''.join(f"{k} {(k * 0x9e3779b97f4a7c15) % (1 << 64):016x}\n" for k in range(0, 48)))
refD1 = os.path.join(d, 'refD1.txt'); open(refD1, 'w').write(''.join(f"{k} {((k * 0x9e3779b97f4a7c15) % (1 << 64)) ^ 0xdd00dd00dd00dd00:016x}\n" for k in range(0, 48)))
jc = os.path.join(d, 'job-context.json')
json.dump({'object': 'fake', 'prehash': '00' * 31 + '01', 'range_log2': 4, 'boundary_nonce': 24, 'manifest': 'deadbeef', 'packs': {'D': pack, 'D1': pack2}, 'references': {'D': refD, 'D1': refD1}}, open(jc, 'w'))
def gojc(phase, tag, boundary=None, env=None, extra=()):
if boundary is not None:
j = json.load(open(jc)); j['boundary_nonce'] = boundary; json.dump(j, open(jc, 'w'))
out = os.path.join(d, f'{tag}.json'); e = dict(os.environ); e.update(env or {})
r = subprocess.run([sys.executable, __file__, '--worker', sys.executable, '--worker-arg', w, '--job-context', jc, '--phase', str(phase), '--job-nonces', '8', '--out', out] + list(extra), capture_output=True, text=True, env=e, timeout=120)
return r.returncode, (json.load(open(out)) if os.path.exists(out) else {}), r.stdout + r.stderr
for ph, s0, pid in ((1, 0, '0x1'), (3, 32, '0x2')):
rc, ev, o = gojc(ph, f'jc{ph}') # the worker holds the day D pack in every phase, as the pods run it; phase 3 needs the day D+1 prepare
if not (rc == 0 and ev.get('verdict') == 'PASS' and ev['nonces'] == {'start': s0, 'count': 16} and ev['segments'][0]['program_id'] == pid and 'boundary' not in ev and (ph == 1) == ('prepare_lines' not in ev)): print(f"self-test failed: phase {ph} of the job context was not a PASS on its range and pack: rc={rc} {ev.get('nonces')} {ev.get('segments')} {o[-200:]}"); fails = 1
rc, ev, o = gojc(2, 'jc2')
b = ev.get('boundary', {})
if not (rc == 0 and ev.get('verdict') == 'PASS' and ev['nonces'] == {'start': 16, 'count': 16} and b.get('nonce') == 24 and b.get('program_id_before') == '0x1' and b.get('program_id_after') == '0x2' and b.get('switched') is True and len(ev['segments']) == 2): print(f"self-test failed: phase 2 did not switch pack at the boundary nonce 24 with the boundary block: rc={rc} {b} {o[-200:]}"); fails = 1
if not (b.get('switched') is True and str(ev.get('prepare_line', '')).startswith('prepared ' + 'aa' * 32 + ' ' + 'dd' * 19) and ev['segments'][1]['pack'] == 'pack2'): print(f"self-test failed: phase 2 did not prepare the day D+1 pair before the first job on it: {ev.get('prepare_line')} {ev.get('segments')}"); fails = 1
rc, ev, o = gojc(2, 'jc2slow', env={'PREPSLOW': '3'})
b = ev.get('boundary', {})
if not (rc == 0 and ev.get('verdict') == 'PASS' and b.get('switched') is True and str(ev.get('prepare_line', '')).startswith('prepared ')): print(f"self-test failed: a worker whose prepare finishes three lines later (the driver must poll it) was not driven across the boundary: rc={rc} {b} {o[-200:]}"); fails = 1
rc, ev, o = gojc(2, 'jc2first', env={'RESIDENT': 'none', 'PREPSLOW': '2'}, extra=['--prepare-first'])
b = ev.get('boundary', {})
if not (rc == 0 and ev.get('verdict') == 'PASS' and b.get('switched') is True and len(ev.get('prepare_lines', [])) == 2 and ev['prepare_lines'][0].split()[2] == 'cc' * 19): print(f"self-test failed: --prepare-first (a worker with no pack at start) did not prepare day D before the first job and day D+1 before the boundary: rc={rc} {ev.get('prepare_lines')} {o[-200:]}"); fails = 1
rc, ev, o = gojc(2, 'jc2noprep', env={'NOPREPARE': '1'})
if rc != 2 or 'no prepare support' not in o: print(f"self-test failed: a worker without prepare support was not refused for phase 2: rc={rc} {o[-200:]}"); fails = 1
rc, ev, o = gojc(2, 'jc2prepfail', env={'PREPFAIL': '1'})
if rc != 2 or 'refused a pair' not in o: print(f"self-test failed: a prepare-failed line was not an error: rc={rc} {o[-200:]}"); fails = 1
rc, ev, o = gojc(2, 'jc2bad', boundary=40)
if rc == 0 or 'not inside phase 2' not in o: print(f"self-test failed: a boundary outside phase 2 was not refused: rc={rc} {o[-200:]}"); fails = 1
if not fails: print('self-test passed: a clean million-shape run is PASS with the counts; one wrong hash is FAIL naming the nonce, the gpu and cpu hashes; an unanswered nonce is FAIL naming it; the job-context form runs each phase on its range and pack, splits phase 2 at the boundary nonce with the boundary block, and refuses a boundary outside phase 2; the job lines carry the pack fields and the all-pass target')
return fails
if __name__ == '__main__':
if '--self-test' in sys.argv: sys.exit(self_test())
ap = argparse.ArgumentParser(); ap.add_argument('--worker', required=True); ap.add_argument('--worker-arg', action='append', default=[]); ap.add_argument('--pack', default='')
ap.add_argument('--reference', default=''); ap.add_argument('--count', type=int, default=1_000_000); ap.add_argument('--start', type=int, default=0); ap.add_argument('--job-nonces', type=int, default=1 << 20)
ap.add_argument('--prehash', default='00' * 31 + '01'); ap.add_argument('--manifest', default=''); ap.add_argument('--out', required=True)
ap.add_argument('--job-context', default=''); ap.add_argument('--phase', type=int, default=0)
ap.add_argument('--prepare-first', action='store_true', help='the worker holds no pack at start (no --pack flag, the Metal worker): the first pack goes by a prepare line before any job')
a = ap.parse_args()
if a.job_context and not a.phase: ap.error('--job-context needs --phase 1|2|3')
if not a.job_context and not (a.pack and a.reference): ap.error('--pack and --reference, or --job-context with --phase')
ev, rc = run(a)
os.makedirs(os.path.dirname(os.path.abspath(a.out)), exist_ok=True); json.dump(ev, open(a.out, 'w'), indent=2)
if 'error' in ev: print(f"p01-vectors: ERROR: {ev['error']} -> {a.out}"); sys.exit(rc)
print(f"p01-vectors: {ev.get('verdict', 'ERROR')}: answered {ev.get('answered')} agree {ev.get('agree')} disagree {ev.get('disagree')} missing {ev.get('missing')} in {ev.get('seconds')} s -> {a.out}")
sys.exit(rc)