diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-full-r1.row b/docs/analysis/review-2026-10-08-b/f10/3090-full-r1.row new file mode 100644 index 000000000..ae98c3947 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-full-r1.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.710 prod_wall_s=58.497 prod_mhs=9.18 prod_mean_w=164.9 prod_j_per_batch=301.4 prod_mh_per_w=0.0557 samples=234 steady_wall_s=53.787 steady_mhs=9.98 steady_mean_w=168.3 steady_j_per_batch=282.9 steady_mh_per_w=0.0593 samples=215 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 1630.34 ms, select 0.000 ms, read-back 24.723 ms, scan 24.119 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-full-r2.row b/docs/analysis/review-2026-10-08-b/f10/3090-full-r2.row new file mode 100644 index 000000000..b4ec27d2a --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-full-r2.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.231 prod_wall_s=90.096 prod_mhs=5.96 prod_mean_w=134.4 prod_j_per_batch=378.4 prod_mh_per_w=0.0443 samples=360 steady_wall_s=85.865 steady_mhs=6.25 steady_mean_w=136.9 steady_j_per_batch=367.3 steady_mh_per_w=0.0457 samples=343 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 2637.86 ms, select 0.000 ms, read-back 22.784 ms, scan 20.625 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-full-r3.row b/docs/analysis/review-2026-10-08-b/f10/3090-full-r3.row new file mode 100644 index 000000000..306f467ea --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-full-r3.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.110 prod_wall_s=23.537 prod_mhs=22.81 prod_mean_w=282.4 prod_j_per_batch=207.7 prod_mh_per_w=0.0808 samples=94 steady_wall_s=19.428 steady_mhs=27.63 steady_mean_w=325.9 steady_j_per_batch=197.9 steady_mh_per_w=0.0848 samples=77 worker_cpu_s=5.61 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 557.15 ms, select 0.000 ms, read-back 24.225 ms, scan 23.753 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-full-r4.row b/docs/analysis/review-2026-10-08-b/f10/3090-full-r4.row new file mode 100644 index 000000000..398e3183b --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-full-r4.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.397 prod_wall_s=90.051 prod_mhs=5.96 prod_mean_w=134.3 prod_j_per_batch=377.9 prod_mh_per_w=0.0444 samples=360 steady_wall_s=85.654 steady_mhs=6.27 steady_mean_w=136.7 steady_j_per_batch=365.9 steady_mh_per_w=0.0459 samples=342 worker_cpu_s=5.81 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 2624.25 ms, select 0.000 ms, read-back 27.536 ms, scan 22.950 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-kernel-bench.log b/docs/analysis/review-2026-10-08-b/f10/3090-kernel-bench.log new file mode 100644 index 000000000..ba5a4d8e9 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-kernel-bench.log @@ -0,0 +1,4 @@ +info igneum-worker-cuda 1.0 (4 October 2026): device 0 NVIDIA_GeForce_RTX_3090 (sm_86, 82 SMs), driver 13.2 from libcuda.so.1, NVRTC 12.8 from libnvrtc.so.12, target sm_86 (the device's architecture, listed by NVRTC) +pack /root/p02/packs/hl-v6-all on NVIDIA_GeForce_RTX_3090: nvrtc 2558 cache 3 dataset 37 hot 0 check 1190 race 0 ms variant base class v5 (state leaves 93, uploaded for the build and freed); self-test PASS (cache head, last line and FNV-1a 64 448274a57f508cbc; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) +warm-up dispatch (base 0): 554.01 ms; 5 timed dispatches of 16777216 nonces: mean 557.06 ms +RESULT pack=/root/p02/packs/hl-v6-all class=mx8-erad810f22d+sh256x27+state+reg64c+fold+rw device=NVIDIA_GeForce_RTX_3090 arch=sm_86 regs=87 blocks_per_sm=16 warps=0 resident=1312 arena_mib=0 hot_mib=0 hot_slots=0 hot_fill_ms=0.00 nonces=16777216 batches=5 check=PASS fingerprint=59e6708e46f1e87c mhs=30.117 loads=128 bytes=512 scratch_ops=0 time=wall diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-run-more.log b/docs/analysis/review-2026-10-08-b/f10/3090-run-more.log new file mode 100644 index 000000000..76c088fa2 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-run-more.log @@ -0,0 +1,10 @@ +== 3090-full-r3 2026-10-08T19:44:29Z sm 210 MHz, Not Active +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.110 prod_wall_s=23.537 prod_mhs=22.81 prod_mean_w=282.4 prod_j_per_batch=207.7 prod_mh_per_w=0.0808 samples=94 steady_wall_s=19.428 steady_mhs=27.63 steady_mean_w=325.9 steady_j_per_batch=197.9 steady_mh_per_w=0.0848 samples=77 worker_cpu_s=5.61 worker_sha=9b36b523d0812c3c +3090-full-r3 sm MHz mean 1769 min 210 max 1965 over 103 samples +== 3090-select-r3 2026-10-08T19:44:55Z sm 1950 MHz, Not Active +card=NVIDIA_GeForce_RTX_3090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.297 prod_wall_s=83.427 prod_mhs=6.44 prod_mean_w=138.7 prod_j_per_batch=361.6 prod_mh_per_w=0.0464 samples=334 steady_wall_s=79.130 steady_mhs=6.78 steady_mean_w=140.1 steady_j_per_batch=346.4 steady_mh_per_w=0.0484 samples=316 worker_cpu_s=4.23 worker_sha=9b36b523d0812c3c +3090-select-r3 sm MHz mean 318 min 210 max 1950 over 343 samples +== 3090-full-r4 2026-10-08T19:46:24Z sm 210 MHz, Active +card=NVIDIA_GeForce_RTX_3090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.397 prod_wall_s=90.051 prod_mhs=5.96 prod_mean_w=134.3 prod_j_per_batch=377.9 prod_mh_per_w=0.0444 samples=360 steady_wall_s=85.654 steady_mhs=6.27 steady_mean_w=136.7 steady_j_per_batch=365.9 steady_mh_per_w=0.0459 samples=342 worker_cpu_s=5.81 worker_sha=9b36b523d0812c3c +3090-full-r4 sm MHz mean 210 min 210 max 210 over 369 samples +== 3090-select-r4 2026-10-08T19:47:59Z sm 210 MHz, Active diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-select-r1.row b/docs/analysis/review-2026-10-08-b/f10/3090-select-r1.row new file mode 100644 index 000000000..c23891b46 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-select-r1.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=7.093 prod_wall_s=91.584 prod_mhs=5.86 prod_mean_w=131.7 prod_j_per_batch=376.9 prod_mh_per_w=0.0445 samples=366 steady_wall_s=84.491 steady_mhs=6.35 steady_mean_w=136.9 steady_j_per_batch=361.5 steady_mh_per_w=0.0464 samples=337 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 2637.57 ms, select 0.532 ms, read-back 0.296 ms, scan 0.019 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-select-r2.row b/docs/analysis/review-2026-10-08-b/f10/3090-select-r2.row new file mode 100644 index 000000000..bdc8f8f2e --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-select-r2.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.138 prod_wall_s=21.996 prod_mhs=24.41 prod_mean_w=286.1 prod_j_per_batch=196.7 prod_mh_per_w=0.0853 samples=88 steady_wall_s=17.858 steady_mhs=30.06 steady_mean_w=330.7 steady_j_per_batch=184.6 steady_mh_per_w=0.0909 samples=71 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 555.70 ms, select 0.268 ms, read-back 0.217 ms, scan 0.013 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/3090-select-r3.row b/docs/analysis/review-2026-10-08-b/f10/3090-select-r3.row new file mode 100644 index 000000000..c59aa01ad --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/3090-select-r3.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_3090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=4.297 prod_wall_s=83.427 prod_mhs=6.44 prod_mean_w=138.7 prod_j_per_batch=361.6 prod_mh_per_w=0.0464 samples=334 steady_wall_s=79.130 steady_mhs=6.78 steady_mean_w=140.1 steady_j_per_batch=346.4 steady_mh_per_w=0.0484 samples=316 worker_cpu_s=4.23 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_3090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 87 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_86 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 2470.39 ms, select 0.503 ms, read-back 0.206 ms, scan 0.017 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-full-r1.row b/docs/analysis/review-2026-10-08-b/f10/5090-full-r1.row new file mode 100644 index 000000000..26e5a2f91 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-full-r1.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.275 prod_wall_s=10.715 prod_mhs=50.10 prod_mean_w=278.0 prod_j_per_batch=93.1 prod_mh_per_w=0.1802 samples=43 steady_wall_s=8.440 steady_mhs=63.61 steady_mean_w=328.3 steady_j_per_batch=86.6 steady_mh_per_w=0.1938 samples=34 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 236.64 ms, select 0.000 ms, read-back 12.794 ms, scan 13.245 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-full-r2.row b/docs/analysis/review-2026-10-08-b/f10/5090-full-r2.row new file mode 100644 index 000000000..0d5f0efac --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-full-r2.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.373 prod_wall_s=11.405 prod_mhs=47.07 prod_mean_w=270.7 prod_j_per_batch=96.5 prod_mh_per_w=0.1739 samples=46 steady_wall_s=9.031 steady_mhs=59.45 steady_mean_w=326.9 steady_j_per_batch=92.3 steady_mh_per_w=0.1819 samples=36 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 236.65 ms, select 0.000 ms, read-back 25.097 ms, scan 19.536 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-full-r3.row b/docs/analysis/review-2026-10-08-b/f10/5090-full-r3.row new file mode 100644 index 000000000..17d13921a --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-full-r3.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.400 prod_wall_s=10.582 prod_mhs=50.73 prod_mean_w=267.7 prod_j_per_batch=88.5 prod_mh_per_w=0.1895 samples=43 steady_wall_s=8.182 steady_mhs=65.62 steady_mean_w=335.4 steady_j_per_batch=85.8 steady_mh_per_w=0.1956 samples=33 worker_cpu_s=2.97 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 236.73 ms, select 0.000 ms, read-back 7.679 ms, scan 10.341 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-full-r4.row b/docs/analysis/review-2026-10-08-b/f10/5090-full-r4.row new file mode 100644 index 000000000..25cda9c44 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-full-r4.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.334 prod_wall_s=10.566 prod_mhs=50.81 prod_mean_w=285.8 prod_j_per_batch=94.4 prod_mh_per_w=0.1778 samples=43 steady_wall_s=8.233 steady_mhs=65.21 steady_mean_w=351.5 steady_j_per_batch=90.4 steady_mh_per_w=0.1855 samples=33 worker_cpu_s=2.92 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 237.01 ms, select 0.000 ms, read-back 8.639 ms, scan 10.488 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-full-r5.row b/docs/analysis/review-2026-10-08-b/f10/5090-full-r5.row new file mode 100644 index 000000000..91713a101 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-full-r5.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.463 prod_wall_s=10.767 prod_mhs=49.86 prod_mean_w=276.8 prod_j_per_batch=93.1 prod_mh_per_w=0.1801 samples=43 steady_wall_s=8.304 steady_mhs=64.65 steady_mean_w=340.3 steady_j_per_batch=88.3 steady_mh_per_w=0.1900 samples=33 worker_cpu_s=3.13 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback full worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 134217728 B (full, 0 full fallbacks); mean per chunk: kernel+sync 236.70 ms, select 0.000 ms, read-back 9.028 ms, scan 12.796 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-kernel-bench.log b/docs/analysis/review-2026-10-08-b/f10/5090-kernel-bench.log new file mode 100644 index 000000000..ba7341412 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-kernel-bench.log @@ -0,0 +1,4 @@ +info igneum-worker-cuda 1.0 (4 October 2026): device 0 NVIDIA_GeForce_RTX_5090 (sm_120, 170 SMs), driver 13.2 from libcuda.so.1, NVRTC 12.8 from libnvrtc.so.12, target sm_120 (the device's architecture, listed by NVRTC) +pack /root/p02/packs/hl-v6-all on NVIDIA_GeForce_RTX_5090: nvrtc 1476 cache 1 dataset 13 hot 0 check 532 race 0 ms variant base class v5 (state leaves 93, uploaded for the build and freed); self-test PASS (cache head, last line and FNV-1a 64 448274a57f508cbc; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) +warm-up dispatch (base 0): 236.80 ms; 5 timed dispatches of 16777216 nonces: mean 236.68 ms +RESULT pack=/root/p02/packs/hl-v6-all class=mx8-erad810f22d+sh256x27+state+reg64c+fold+rw device=NVIDIA_GeForce_RTX_5090 arch=sm_120 regs=92 blocks_per_sm=20 warps=0 resident=3400 arena_mib=0 hot_mib=0 hot_slots=0 hot_fill_ms=0.00 nonces=16777216 batches=5 check=PASS fingerprint=59e6708e46f1e87c mhs=70.885 loads=128 bytes=512 scratch_ops=0 time=wall diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-run-more.log b/docs/analysis/review-2026-10-08-b/f10/5090-run-more.log new file mode 100644 index 000000000..115f86316 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-run-more.log @@ -0,0 +1,19 @@ +== 5090-full-r3 2026-10-08T19:44:33Z sm 180 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.400 prod_wall_s=10.582 prod_mhs=50.73 prod_mean_w=267.7 prod_j_per_batch=88.5 prod_mh_per_w=0.1895 samples=43 steady_wall_s=8.182 steady_mhs=65.62 steady_mean_w=335.4 steady_j_per_batch=85.8 steady_mh_per_w=0.1956 samples=33 worker_cpu_s=2.97 worker_sha=9b36b523d0812c3c +5090-full-r3 sm MHz mean 2387 min 180 max 2895 over 51 samples +== 5090-select-r3 2026-10-08T19:44:46Z sm 2895 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.327 prod_wall_s=9.951 prod_mhs=53.95 prod_mean_w=290.5 prod_j_per_batch=90.3 prod_mh_per_w=0.1857 samples=40 steady_wall_s=7.624 steady_mhs=70.42 steady_mean_w=365.4 steady_j_per_batch=87.1 steady_mh_per_w=0.1927 samples=30 worker_cpu_s=2.33 worker_sha=9b36b523d0812c3c +5090-select-r3 sm MHz mean 2636 min 472 max 2895 over 49 samples +== 5090-full-r4 2026-10-08T19:44:58Z sm 2895 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.334 prod_wall_s=10.566 prod_mhs=50.81 prod_mean_w=285.8 prod_j_per_batch=94.4 prod_mh_per_w=0.1778 samples=43 steady_wall_s=8.233 steady_mhs=65.21 steady_mean_w=351.5 steady_j_per_batch=90.4 steady_mh_per_w=0.1855 samples=33 worker_cpu_s=2.92 worker_sha=9b36b523d0812c3c +5090-full-r4 sm MHz mean 2666 min 457 max 2895 over 51 samples +== 5090-select-r4 2026-10-08T19:45:12Z sm 2895 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.590 prod_wall_s=10.216 prod_mhs=52.55 prod_mean_w=285.4 prod_j_per_batch=91.1 prod_mh_per_w=0.1841 samples=41 steady_wall_s=7.626 steady_mhs=70.40 steady_mean_w=355.9 steady_j_per_batch=84.8 steady_mh_per_w=0.1978 samples=31 worker_cpu_s=2.55 worker_sha=9b36b523d0812c3c +5090-select-r4 sm MHz mean 2608 min 555 max 2895 over 50 samples +== 5090-full-r5 2026-10-08T19:45:24Z sm 2880 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=full jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.463 prod_wall_s=10.767 prod_mhs=49.86 prod_mean_w=276.8 prod_j_per_batch=93.1 prod_mh_per_w=0.1801 samples=43 steady_wall_s=8.304 steady_mhs=64.65 steady_mean_w=340.3 steady_j_per_batch=88.3 steady_mh_per_w=0.1900 samples=33 worker_cpu_s=3.13 worker_sha=9b36b523d0812c3c +5090-full-r5 sm MHz mean 2679 min 795 max 2880 over 52 samples +== 5090-select-r5 2026-10-08T19:45:38Z sm 2880 MHz, Not Active +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.393 prod_wall_s=10.019 prod_mhs=53.59 prod_mean_w=287.5 prod_j_per_batch=90.0 prod_mh_per_w=0.1864 samples=40 steady_wall_s=7.626 steady_mhs=70.40 steady_mean_w=360.8 steady_j_per_batch=86.0 steady_mh_per_w=0.1951 samples=30 worker_cpu_s=2.39 worker_sha=9b36b523d0812c3c +5090-select-r5 sm MHz mean 2771 min 2212 max 2880 over 49 samples +== end 2026-10-08T19:45:50Z diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-select-r1.row b/docs/analysis/review-2026-10-08-b/f10/5090-select-r1.row new file mode 100644 index 000000000..abb24501e --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-select-r1.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.447 prod_wall_s=10.070 prod_mhs=53.31 prod_mean_w=290.5 prod_j_per_batch=91.4 prod_mh_per_w=0.1835 samples=41 steady_wall_s=7.624 steady_mhs=70.42 steady_mean_w=362.5 steady_j_per_batch=86.4 steady_mh_per_w=0.1943 samples=31 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 236.71 ms, select 0.234 ms, read-back 0.178 ms, scan 0.005 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-select-r2.row b/docs/analysis/review-2026-10-08-b/f10/5090-select-r2.row new file mode 100644 index 000000000..d00c872c8 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-select-r2.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.408 prod_wall_s=10.032 prod_mhs=53.52 prod_mean_w=289.1 prod_j_per_batch=90.6 prod_mh_per_w=0.1851 samples=40 steady_wall_s=7.624 steady_mhs=70.42 steady_mean_w=363.0 steady_j_per_batch=86.5 steady_mh_per_w=0.1940 samples=30 worker_cpu_s=0.00 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 236.69 ms, select 0.245 ms, read-back 0.270 ms, scan 0.007 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-select-r3.row b/docs/analysis/review-2026-10-08-b/f10/5090-select-r3.row new file mode 100644 index 000000000..e1c3aa295 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-select-r3.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.327 prod_wall_s=9.951 prod_mhs=53.95 prod_mean_w=290.5 prod_j_per_batch=90.3 prod_mh_per_w=0.1857 samples=40 steady_wall_s=7.624 steady_mhs=70.42 steady_mean_w=365.4 steady_j_per_batch=87.1 steady_mh_per_w=0.1927 samples=30 worker_cpu_s=2.33 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 236.71 ms, select 0.262 ms, read-back 0.353 ms, scan 0.006 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-select-r4.row b/docs/analysis/review-2026-10-08-b/f10/5090-select-r4.row new file mode 100644 index 000000000..5924877ed --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-select-r4.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.590 prod_wall_s=10.216 prod_mhs=52.55 prod_mean_w=285.4 prod_j_per_batch=91.1 prod_mh_per_w=0.1841 samples=41 steady_wall_s=7.626 steady_mhs=70.40 steady_mean_w=355.9 steady_j_per_batch=84.8 steady_mh_per_w=0.1978 samples=31 worker_cpu_s=2.55 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 236.63 ms, select 0.239 ms, read-back 0.320 ms, scan 0.010 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/5090-select-r5.row b/docs/analysis/review-2026-10-08-b/f10/5090-select-r5.row new file mode 100644 index 000000000..eb6dc7486 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/5090-select-r5.row @@ -0,0 +1,3 @@ +card=NVIDIA_GeForce_RTX_5090 mode=select jobs=32 batch_log2=24 hashes=536870912 found=519 setup_s=2.393 prod_wall_s=10.019 prod_mhs=53.59 prod_mean_w=287.5 prod_j_per_batch=90.0 prod_mh_per_w=0.1864 samples=40 steady_wall_s=7.626 steady_mhs=70.40 steady_mean_w=360.8 steady_j_per_batch=86.0 steady_mh_per_w=0.1951 samples=30 worker_cpu_s=2.39 worker_sha=9b36b523d0812c3c +ready: ready cuda NVIDIA_GeForce_RTX_5090 pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 16777216 regs 92 prepare 1 path nvrtc 12.8 driver 13.2 arch sm_120 variant base race off readback select worker 1.0 (4 October 2026) +transfers: transfers per chunk: down 536 B (select, 0 full fallbacks); mean per chunk: kernel+sync 236.78 ms, select 0.296 ms, read-back 0.244 ms, scan 0.007 ms; 32 jobs 32 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/emu-test.log b/docs/analysis/review-2026-10-08-b/f10/emu-test.log new file mode 100644 index 000000000..09eed520d --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/emu-test.log @@ -0,0 +1,283 @@ +igneum-pow export /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-b +seed "igneum-epoch/5e5d0c3b2a19f8e7d6c5b4a3928170f6e5d4c3b2a1908f7e6d5c4b3a2918f7e6/day/69676e65756d2d6461792ffa51000000000000", day "bytes:69676e65756d2d6461792ffa51000000000000", dataset 2^28 words (memory-hard), generator v2 attempt 0 program id 36e7763af4220724, loads/hash 128; epoch built in 376.1 ms +/home/build/wt/worker-select-pass/proto-cuda/nvrtc/worker.cpp:131:14: warning: ‘void* libSym(void*, const char*)’ defined but not used [-Wunused-function] + 131 | static void* libSym(void* lib, const char* name) { + | ^~~~~~ +/home/build/wt/worker-select-pass/proto-cuda/nvrtc/worker.cpp:124:14: warning: ‘void* libOpen(const std::string&)’ defined but not used [-Wunused-function] + 124 | static void* libOpen(const std::string& name) { + | ^~~~~~~ +/home/build/wt/worker-select-pass/proto-cuda/nvrtc/worker.cpp:109:20: warning: ‘std::string exeDir()’ defined but not used [-Wunused-function] + 109 | static std::string exeDir() { + | ^~~~~~ +built /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/igneum-worker-cuda-emu (CPU emulation, not a GPU build) +== --check pack A +info igneum-worker-cuda 1.0 (4 October 2026): device 0 CPU_emulation_shim_(not_a_GPU) (sm_120, 0 SMs), driver 12.8 from emulation (host threads, no GPU), NVRTC 12.8 from emulation (source recorded and checked, nothing compiled), target sm_120 (the device's architecture, listed by NVRTC) +info emu-nvrtc: kernel.cu source check PASS for pack 1: 7385 bytes handed over = the pack file's first 7385 of 8893 bytes; dropped tail = host launch wrappers only (1508 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info emu-nvrtc: kernel_bound.cu source check PASS for pack 1: 6003 bytes handed over = the pack file's first 6003 of 6887 bytes; dropped tail = host launch wrappers only (884 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +check PASS /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-a in 2679 ms: nvrtc 0 cache 16 dataset 2193 hot 0 check 470 race 0 ms variant base class v2; self-test PASS (cache head, last line and FNV-1a 64 448274a57f508cbc; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) + epoch edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07 day 69676e65756d2d6461792ffa50000000000000, dataset 2^28 words, cache 2^26 words in 65536 segments, 0 registers, 0 blocks/SM at 1 warp(s)/block, target sm_120 +== the annotation rule: without -default-device the stand-in must reject program.h's igneum_launch_* declarations as host code (what the RTX 5090 reported on 4 October 2026) +flagged as NVRTC does: program.h(46): cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +== --serve: two jobs on A, prepare B, a job on B (swap), a job on A after the swap (error), quit +info igneum-worker-cuda 1.0 (4 October 2026): device 0 CPU_emulation_shim_(not_a_GPU) (sm_120, 0 SMs), driver 12.8 from emulation (host threads, no GPU), NVRTC 12.8 from emulation (source recorded and checked, nothing compiled), target sm_120 (the device's architecture, listed by NVRTC) +info emu-nvrtc: kernel.cu source check PASS for pack 1: 7385 bytes handed over = the pack file's first 7385 of 8893 bytes; dropped tail = host launch wrappers only (1508 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info emu-nvrtc: kernel_bound.cu source check PASS for pack 1: 6003 bytes handed over = the pack file's first 6003 of 6887 bytes; dropped tail = host launch wrappers only (884 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info emu-nvrtc: igneum_select.cu source check PASS: 763 bytes, the select pass runs as a host loop +ready cuda CPU_emulation_shim_(not_a_GPU) pack igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000 dataset-log2 28 batch 8192 regs 0 prepare 1 path nvrtc 12.8 driver 12.8 arch sm_120 variant base race on readback select worker 1.0 (4 October 2026) +info readback select (per dispatch of 8192 nonces: the hits and 34 sentinel words come back; the GPU scans) +info first pack /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-a: nvrtc 0 cache 15 dataset 2686 hot 0 check 472 race 0 ms variant base class v2; self-test PASS (cache head, last line and FNV-1a 64 448274a57f508cbc; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) +race edc4fa844da9dc98 device CPU_emulation_shim_(not_a_GPU) variants 1 base only, no race (emulation) +found 1 0 85f8352a2701deaa edc4fa844da9dc98 +found 1 1 287bddfe85a71a6c edc4fa844da9dc98 +found 1 2 fd65144d28e0e2b1 edc4fa844da9dc98 +found 1 3 4c451942cdccec1a edc4fa844da9dc98 +found 1 4 b974b6a86b1cf0ed edc4fa844da9dc98 +found 1 5 c470d05d126a1a61 edc4fa844da9dc98 +found 1 6 824f53a0342e593d edc4fa844da9dc98 +found 1 7 ab83476b2d6aa8d0 edc4fa844da9dc98 +found 1 8 bb1a171d76ebe772 edc4fa844da9dc98 +found 1 9 952788c85aa6ba6a edc4fa844da9dc98 +found 1 10 eab425f84eeb2a81 edc4fa844da9dc98 +found 1 11 6279385502e9b189 edc4fa844da9dc98 +found 1 12 92bb6782d9337a70 edc4fa844da9dc98 +found 1 13 bf69940f96c7607b edc4fa844da9dc98 +found 1 14 4ec115ccd6700418 edc4fa844da9dc98 +found 1 15 aee365ce8b71d4fd edc4fa844da9dc98 +found 1 16 3b1074a344f48d75 edc4fa844da9dc98 +found 1 17 7bbba30cdeee2318 edc4fa844da9dc98 +found 1 18 85917afa2fb98002 edc4fa844da9dc98 +found 1 19 a57d924a33c9eb47 edc4fa844da9dc98 +found 1 20 ddf5d02d5440bed5 edc4fa844da9dc98 +found 1 21 e21722ea35922f3d edc4fa844da9dc98 +found 1 22 7c71ff55e0202ea0 edc4fa844da9dc98 +found 1 23 2e264b7a3db0424e edc4fa844da9dc98 +found 1 24 737435f635d5cd13 edc4fa844da9dc98 +found 1 25 26a94c56d64fa788 edc4fa844da9dc98 +found 1 26 c5a88d468a03eebb edc4fa844da9dc98 +found 1 27 31ca283df921550d edc4fa844da9dc98 +found 1 28 06d3199ce7b1adf4 edc4fa844da9dc98 +found 1 29 0c8f5f7605deecd1 edc4fa844da9dc98 +found 1 30 c43156d1e4a5cbec edc4fa844da9dc98 +found 1 31 78ba75b0e4c908d5 edc4fa844da9dc98 +found 1 32 e40b3c4ce16e0674 edc4fa844da9dc98 +found 1 33 9a6b697360a70406 edc4fa844da9dc98 +found 1 34 e28eb3b8c845ef22 edc4fa844da9dc98 +found 1 35 a8db66a07fa26a94 edc4fa844da9dc98 +found 1 36 81d256fa67d8bed3 edc4fa844da9dc98 +found 1 37 f681ad4e0426b0e5 edc4fa844da9dc98 +found 1 38 ad49770893af9688 edc4fa844da9dc98 +found 1 39 e830f95c3e013a41 edc4fa844da9dc98 +found 1 40 7f050e695bd46847 edc4fa844da9dc98 +found 1 41 b6c2ae03f8ddf3ae edc4fa844da9dc98 +found 1 42 772509fa6f742f94 edc4fa844da9dc98 +found 1 43 3d0317e357e2381d edc4fa844da9dc98 +found 1 44 94f2baf0d5af5a6b edc4fa844da9dc98 +found 1 45 3e2ba39651256a56 edc4fa844da9dc98 +found 1 46 588295beedbc0aff edc4fa844da9dc98 +found 1 47 edc862c227a7f27d edc4fa844da9dc98 +found 1 48 a540ec001fd3fe33 edc4fa844da9dc98 +found 1 49 3f3a2aa56b364df1 edc4fa844da9dc98 +found 1 50 b395503049cfefa2 edc4fa844da9dc98 +found 1 51 bf7a4473bf7be326 edc4fa844da9dc98 +found 1 52 8af17d040c654b2a edc4fa844da9dc98 +found 1 53 d2670d56b9848c44 edc4fa844da9dc98 +found 1 54 ac5f448acfb856d5 edc4fa844da9dc98 +found 1 55 89d2248ea07b109a edc4fa844da9dc98 +found 1 56 dd4c5e4c6bb7c2e6 edc4fa844da9dc98 +found 1 57 2529ceca58759d77 edc4fa844da9dc98 +found 1 58 0bdff5ded9bc65fb edc4fa844da9dc98 +found 1 59 6ea20ccedaf3e18c edc4fa844da9dc98 +found 1 60 bedd6f14e5fa465d edc4fa844da9dc98 +found 1 61 08d7d815f5786ced edc4fa844da9dc98 +found 1 62 7d289bf14d911c5f edc4fa844da9dc98 +found 1 63 e2f4237e31cbd041 edc4fa844da9dc98 +done 1 64 9.66 +found 2 4294967264 7b6273e5bd32d0b5 edc4fa844da9dc98 +found 2 4294967265 38ca4cce09bfb623 edc4fa844da9dc98 +found 2 4294967266 d8b0ed94f45cdbf7 edc4fa844da9dc98 +found 2 4294967267 9d8faff62d0efa15 edc4fa844da9dc98 +found 2 4294967268 1df806f32926defe edc4fa844da9dc98 +found 2 4294967269 e3654b2d705786bb edc4fa844da9dc98 +found 2 4294967270 ab6741286c7c88dd edc4fa844da9dc98 +found 2 4294967271 a15815bf414e9320 edc4fa844da9dc98 +found 2 4294967272 6f1cc5a7a2fcc7c0 edc4fa844da9dc98 +found 2 4294967273 6e98c101317928c6 edc4fa844da9dc98 +found 2 4294967274 d87e3fefd2a7b381 edc4fa844da9dc98 +found 2 4294967275 f7f30848fd476fb5 edc4fa844da9dc98 +found 2 4294967276 257fc4d2e3603beb edc4fa844da9dc98 +found 2 4294967277 1aab6f77c1365d4c edc4fa844da9dc98 +found 2 4294967278 1c9d005f8f4ac463 edc4fa844da9dc98 +found 2 4294967279 161427ad7f81f2f9 edc4fa844da9dc98 +found 2 4294967280 e8611bc03fb827c1 edc4fa844da9dc98 +found 2 4294967281 42454a79caee8c69 edc4fa844da9dc98 +found 2 4294967282 190aa2b3529b589f edc4fa844da9dc98 +found 2 4294967283 83d56d283df7f1e0 edc4fa844da9dc98 +found 2 4294967284 fb355b9c51ca8988 edc4fa844da9dc98 +found 2 4294967285 f5b626e35d01424a edc4fa844da9dc98 +found 2 4294967286 8a0b81e4053fe9b0 edc4fa844da9dc98 +found 2 4294967287 c1aa110e3db24ed7 edc4fa844da9dc98 +found 2 4294967288 fc53be92ceb49a8e edc4fa844da9dc98 +found 2 4294967289 c3296889fb569a0a edc4fa844da9dc98 +found 2 4294967290 e38850154c2dea6c edc4fa844da9dc98 +found 2 4294967291 d99fda96ee76c555 edc4fa844da9dc98 +found 2 4294967292 fac8684a83d19072 edc4fa844da9dc98 +found 2 4294967293 51085b8a57f3720d edc4fa844da9dc98 +found 2 4294967294 7311e9864d96be76 edc4fa844da9dc98 +found 2 4294967295 2c52ef30b1f52cbf edc4fa844da9dc98 +found 2 4294967296 aaee181f46a06d70 edc4fa844da9dc98 +found 2 4294967297 aa45e128a7ab498b edc4fa844da9dc98 +found 2 4294967298 eb0f5f050141457c edc4fa844da9dc98 +found 2 4294967299 7be5e2d3e944d761 edc4fa844da9dc98 +found 2 4294967300 804964e4293faa67 edc4fa844da9dc98 +found 2 4294967301 7567f837977d77da edc4fa844da9dc98 +found 2 4294967302 1ff1311a643d3b43 edc4fa844da9dc98 +found 2 4294967303 60cda5dadaaa941c edc4fa844da9dc98 +found 2 4294967304 63542e063db3fb34 edc4fa844da9dc98 +found 2 4294967305 e0ac767c5fab1ab4 edc4fa844da9dc98 +found 2 4294967306 e2423cd93bfb5aa0 edc4fa844da9dc98 +found 2 4294967307 c3210435edd81d5a edc4fa844da9dc98 +found 2 4294967308 a0226e9f79d09099 edc4fa844da9dc98 +found 2 4294967309 fc143bbd8555fdb0 edc4fa844da9dc98 +found 2 4294967310 cc9008673f527726 edc4fa844da9dc98 +found 2 4294967311 bb329b9bcf8e66fd edc4fa844da9dc98 +found 2 4294967312 164442e9130c7b7a edc4fa844da9dc98 +found 2 4294967313 48e43b690541132d edc4fa844da9dc98 +found 2 4294967314 f4ba8f420da3337e edc4fa844da9dc98 +found 2 4294967315 050dff2f1846ab61 edc4fa844da9dc98 +found 2 4294967316 74972ac9ab8fb364 edc4fa844da9dc98 +found 2 4294967317 b7ef1e3155e9a7f3 edc4fa844da9dc98 +found 2 4294967318 3ecb2b74489fc8a4 edc4fa844da9dc98 +found 2 4294967319 64fa8c8296a3d662 edc4fa844da9dc98 +found 2 4294967320 6c4411933e7f690c edc4fa844da9dc98 +found 2 4294967321 23234ffc83690177 edc4fa844da9dc98 +found 2 4294967322 01dd1d050f87e26e edc4fa844da9dc98 +found 2 4294967323 e3a05b34dde32af0 edc4fa844da9dc98 +found 2 4294967324 0bcf244d4c73562d edc4fa844da9dc98 +found 2 4294967325 375800d4039b5c1f edc4fa844da9dc98 +found 2 4294967326 42b9a4c28e42c996 edc4fa844da9dc98 +found 2 4294967327 f56d052d60a7338e edc4fa844da9dc98 +done 2 64 10.89 +info prepare started for epoch 5e5d0c3b2a19f8e7 day 69676e65756d2d6461792ffa51000000000000 from /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-b (NVRTC sm_120 in the background) +info emu-nvrtc: kernel.cu source check PASS for pack 2: 7206 bytes handed over = the pack file's first 7206 of 8714 bytes; dropped tail = host launch wrappers only (1508 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info emu-nvrtc: kernel_bound.cu source check PASS for pack 2: 5824 bytes handed over = the pack file's first 5824 of 6708 bytes; dropped tail = host launch wrappers only (884 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +race 5e5d0c3b2a19f8e7 device CPU_emulation_shim_(not_a_GPU) variants 1 base only, no race (emulation) +prepared 5e5d0c3b2a19f8e7d6c5b4a3928170f6e5d4c3b2a1908f7e6d5c4b3a2918f7e6 69676e65756d2d6461792ffa51000000000000 36806.6 nvrtc 0 cache 15 dataset 1535 hot 0 check 462 race 0 ms variant base class v2; self-test PASS (cache head, last line and FNV-1a 64 b5b0660c9db7ce59; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) resident 2 programs 2 datasets +found 3 128 72b866f3367df254 edc4fa844da9dc98 +found 3 129 a9772a8b6a793277 edc4fa844da9dc98 +found 3 130 d07426082b75402d edc4fa844da9dc98 +found 3 131 408083ae76ee518b edc4fa844da9dc98 +found 3 132 c45e395f38ad22d8 edc4fa844da9dc98 +found 3 133 3dbee02765058ade edc4fa844da9dc98 +found 3 134 cd5f25857807cfa9 edc4fa844da9dc98 +found 3 135 879e1e3fe942214a edc4fa844da9dc98 +found 3 136 d2133aca0cc47d7a edc4fa844da9dc98 +found 3 137 b143a77f8b19741c edc4fa844da9dc98 +found 3 138 d3ed2304cc49a6c7 edc4fa844da9dc98 +found 3 139 75cf3d6584e42e65 edc4fa844da9dc98 +found 3 140 86db0130d062a7b7 edc4fa844da9dc98 +found 3 141 05f62574da3354c5 edc4fa844da9dc98 +found 3 142 7c961bec00e0cc4b edc4fa844da9dc98 +found 3 143 db8c0f4e55c387ad edc4fa844da9dc98 +found 3 144 7f1454c90435afbd edc4fa844da9dc98 +found 3 145 002b3ca8c7bfd154 edc4fa844da9dc98 +found 3 146 edae0b061a210205 edc4fa844da9dc98 +found 3 147 e478cf93a9878f51 edc4fa844da9dc98 +found 3 148 32c7dfd0fe55a026 edc4fa844da9dc98 +found 3 149 06b93d72fb16404c edc4fa844da9dc98 +found 3 150 4b6cace41d98e7fd edc4fa844da9dc98 +found 3 151 41e2649e041acc06 edc4fa844da9dc98 +found 3 152 0453f57fae7f514a edc4fa844da9dc98 +found 3 153 ff01e0015a3f39e9 edc4fa844da9dc98 +found 3 154 491ae49020261797 edc4fa844da9dc98 +found 3 155 ef1bd996a361128a edc4fa844da9dc98 +found 3 156 794082f3f114767d edc4fa844da9dc98 +found 3 157 5822e90044ea782a edc4fa844da9dc98 +found 3 158 259700c74b5cc149 edc4fa844da9dc98 +found 3 159 235a57ae0bb5a2ed edc4fa844da9dc98 +done 3 32 9.91 +info switched to the prepared pair epoch 5e5d0c3b2a19f8e7 day 69676e65756d2d6461792ffa51000000000000 (class v2) in 0.00 ms +found 4 0 992153a3ef2b07fc 5e5d0c3b2a19f8e7 +found 4 1 55e25d6fb49f4351 5e5d0c3b2a19f8e7 +found 4 2 54a9131bc3066ee4 5e5d0c3b2a19f8e7 +found 4 3 1c027229270c7d33 5e5d0c3b2a19f8e7 +found 4 4 3806eac5cad07581 5e5d0c3b2a19f8e7 +found 4 5 6a87c5563912348c 5e5d0c3b2a19f8e7 +found 4 6 0b802423721bc6af 5e5d0c3b2a19f8e7 +found 4 7 27ec87f124a88548 5e5d0c3b2a19f8e7 +found 4 8 474bf5241c3a8ec3 5e5d0c3b2a19f8e7 +found 4 9 532dbcf0caa93f96 5e5d0c3b2a19f8e7 +found 4 10 259186fa7df9be63 5e5d0c3b2a19f8e7 +found 4 11 f3b8c0df8787ff76 5e5d0c3b2a19f8e7 +found 4 12 0e25eb13e9ddb276 5e5d0c3b2a19f8e7 +found 4 13 a28c32bc4aeabf61 5e5d0c3b2a19f8e7 +found 4 14 1e9c1a2be3f3f110 5e5d0c3b2a19f8e7 +found 4 15 9d4a1cfde84d3aba 5e5d0c3b2a19f8e7 +found 4 16 095727897ed26f5a 5e5d0c3b2a19f8e7 +found 4 17 fdcbce6e7d66f667 5e5d0c3b2a19f8e7 +found 4 18 aefcdcf46d9e6a4a 5e5d0c3b2a19f8e7 +found 4 19 c6ec5a44a97ca398 5e5d0c3b2a19f8e7 +found 4 20 50fce9ed2020b450 5e5d0c3b2a19f8e7 +found 4 21 8c2fe5af5ee1e9f3 5e5d0c3b2a19f8e7 +found 4 22 500ad998c15f42b6 5e5d0c3b2a19f8e7 +found 4 23 95965bedaeaa462d 5e5d0c3b2a19f8e7 +found 4 24 df645bcb6d1cc3ad 5e5d0c3b2a19f8e7 +found 4 25 716b7f5dcc5c30dc 5e5d0c3b2a19f8e7 +found 4 26 84fcbe9758749444 5e5d0c3b2a19f8e7 +found 4 27 29511da3232f2dff 5e5d0c3b2a19f8e7 +found 4 28 b976fd6a159db063 5e5d0c3b2a19f8e7 +found 4 29 69889d22e3cb6709 5e5d0c3b2a19f8e7 +found 4 30 151cc4a784c80add 5e5d0c3b2a19f8e7 +found 4 31 27b10106d08793a5 5e5d0c3b2a19f8e7 +done 4 32 8.22 +info dropped the previous pair (its program, cache and dataset) +info job 5 is for epoch edc4fa844da9dc98 day 69676e65756d2d6461792ffa50000000000000, which is not resident; building its pack /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-a now (foreground) +info emu-nvrtc: kernel.cu source check PASS for pack 1: 7385 bytes handed over = the pack file's first 7385 of 8893 bytes; dropped tail = host launch wrappers only (1508 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info emu-nvrtc: kernel_bound.cu source check PASS for pack 1: 6003 bytes handed over = the pack file's first 6003 of 6887 bytes; dropped tail = host launch wrappers only (884 bytes); program.h (3024 bytes) and memhard.h (5848 bytes) byte-identical +info built /home/build/wt/worker-select-pass/proto-cuda/nvrtc/emu/build/pack-a: nvrtc 0 cache 15 dataset 1295 hot 0 check 481 race 0 ms variant base class v2; self-test PASS (cache head, last line and FNV-1a 64 448274a57f508cbc; dataset head, word [268435455] and 64 samples; 96 of 96 vector lanes) +race edc4fa844da9dc98 device CPU_emulation_shim_(not_a_GPU) variants 1 base only, no race (emulation) +info switched to the prepared pair epoch edc4fa844da9dc98 day 69676e65756d2d6461792ffa50000000000000 (class v2) in 1791.36 ms +found 5 96 cb13242c3a1a4ffa edc4fa844da9dc98 +found 5 97 f3071494c5382b79 edc4fa844da9dc98 +found 5 98 27d7c3cf7cec0c32 edc4fa844da9dc98 +found 5 99 4137aaf4e3ec29e1 edc4fa844da9dc98 +found 5 100 3e18fe3b1190d59f edc4fa844da9dc98 +found 5 101 c1d940643197c68e edc4fa844da9dc98 +found 5 102 005901b495cc73be edc4fa844da9dc98 +found 5 103 2d18e57bfa59ce02 edc4fa844da9dc98 +found 5 104 326bc2b1e5eec1d3 edc4fa844da9dc98 +found 5 105 84b754dc15b156b0 edc4fa844da9dc98 +found 5 106 bdb2c9a6d88c78ce edc4fa844da9dc98 +found 5 107 89d57198d7f17874 edc4fa844da9dc98 +found 5 108 3c96f431799d9d97 edc4fa844da9dc98 +found 5 109 2d98f14c5b8153a4 edc4fa844da9dc98 +found 5 110 a6d1bf98ddda41fb edc4fa844da9dc98 +found 5 111 cd14d9e675dd60c0 edc4fa844da9dc98 +found 5 112 349128ba141dd961 edc4fa844da9dc98 +found 5 113 af2496d2e09bce14 edc4fa844da9dc98 +found 5 114 59ca36336962df25 edc4fa844da9dc98 +found 5 115 f44a3eab722df34b edc4fa844da9dc98 +found 5 116 f7c766cb70017bb1 edc4fa844da9dc98 +found 5 117 4fcabac398cca930 edc4fa844da9dc98 +found 5 118 f2e1cb0a1478ed76 edc4fa844da9dc98 +found 5 119 5fc703e08569cf0d edc4fa844da9dc98 +found 5 120 4aa1b3c1b264ff4e edc4fa844da9dc98 +found 5 121 008922eb5f1b0558 edc4fa844da9dc98 +found 5 122 7a00d435fa182ea4 edc4fa844da9dc98 +found 5 123 347fd45b06777e05 edc4fa844da9dc98 +found 5 124 86e87b980be70737 edc4fa844da9dc98 +found 5 125 7cdbe6e309e01af7 edc4fa844da9dc98 +found 5 126 22de307670d9ee71 edc4fa844da9dc98 +found 5 127 d582b681bf406666 edc4fa844da9dc98 +done 5 32 1800.01 +info dropped the previous pair (its program, cache and dataset) +info transfers per chunk: down 873 B (select, 0 full fallbacks); mean per chunk: kernel+sync 7.87 ms, select 0.001 ms, read-back 0.001 ms, scan 0.008 ms; 5 jobs 6 chunks +== verdict +PASS: ready + prepare 1; 64 + 64 + 32 found on pack A, prepared B with self-test PASS, swapped, 32 found on B, job 5 served after the worker rebuilt pack A by itself (self-heal); 17 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound +NVRTC source check PASS for 7 files in --serve, 2 in --check +== the select pass (8 October 2026): --readback select equals the full read-back, known-failed first (emu/select-check.sh) +== select_fault_detected (known-failed first: the counter is not reset from the second chunk on) +flagged: 84 stale or extra lines in the fault run +== found_set_equals_full_* (every case, line for line) +PASS: select = full over 2255 found lines and 10 jobs (zero hits, partial, all-hit overflow with 2 full fallbacks, a 96-nonce tail, the high-32 rollover, two queued jobs, hits then zero, a mismatch error then a job); the fault run was flagged first +transfers: transfers per chunk: down 1234 B (select, 2 full fallbacks); mean per chunk: kernel+sync 149.78 ms, select 0.003 ms, read-back 0.001 ms, scan 0.052 ms; 10 jobs 20 chunks +transfers: transfers per chunk: down 7411 B (full, 0 full fallbacks); mean per chunk: kernel+sync 142.04 ms, select 0.000 ms, read-back 0.002 ms, scan 0.050 ms; 10 jobs 20 chunks diff --git a/docs/analysis/review-2026-10-08-b/f10/geometry-check-mini.log b/docs/analysis/review-2026-10-08-b/f10/geometry-check-mini.log new file mode 100644 index 000000000..393b23af6 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/geometry-check-mini.log @@ -0,0 +1,31 @@ +ok geometry ds55: loads as 1476395008 words, 92274688 items, mulshift 1, 5905580032 bytes +ok geometry ds85: loads as 2281701376 words, 142606336 items, mulshift 1, 9126805504 bytes +ok geometry ds115: loads as 3087007744 words, 192937984 items, mulshift 1, 12348030976 bytes +ok geometry legacy-2^28: loads as 268435456 words, 16777216 items, mulshift 0, 1073741824 bytes +ok geometry legacy-2^30: loads as 1073741824 words, 67108864 items, mulshift 0, 4294967296 bytes +ok geometry legacy-2^31-explicit-count: loads as 2147483648 words, 134217728 items, mulshift 0, 8589934592 bytes +ok geometry wrong-count-not-a-65536-multiple: refused with "multiple of 65,536" +ok geometry wrong-log2-for-the-count: refused with "IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count" +ok geometry wrong-log2-for-a-power-of-two-count: refused with "IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count" +ok geometry wrong-items: refused with "IGNEUM_DATASET_ITEMS is not IGNEUM_DATASET_WORDS / 16" +ok geometry wrong-mulshift-on-a-non-power-of-two: refused with "IGNEUM_DATASET_MULSHIFT does not match the word count" +ok geometry log2-32: refused with "IGNEUM_DATASET_LOG2 above 31" +ok geometry overflowing-count-2^32: refused with "IGNEUM_DATASET_WORDS above 2^32 - 1" +ok geometry overflowing-count-2^40: refused with "IGNEUM_DATASET_WORDS above 2^32 - 1" +ok geometry count-below-the-minimum: refused with "multiple of 65,536" +ok geometry log2-below-20: refused with "sizes out of range" +ok geometry missing-log2: refused with "has no IGNEUM_DATASET_LOG2" +ok geometry missing-leaves-on-a-class-v5-pack: refused with "without IGNEUM_STATE_LEAVES" +ok geometry leaves-on-a-class-v2-pack: refused with "state leaves belong to class v5" +ok the vectors file carries every case (18 or more): 19 +ok pack /Users/igneum/Projects/igneum/proto-cuda/packs/igneum-devnet-v4-epoch0: 268435456 words (2^28), 16777216 items, 1073741824 bytes +ok pack /Users/igneum/Projects/igneum/proto-cuda/packs-ca3-v5/v5-dn3-epoch0: 268435456 words (2^28), 16777216 items, 1073741824 bytes +geometry-check: 19 vectors, 2 packs, 0 failure(s) +== Metal serve smoke, class v5 pack (packs-ca3-v5/v5-dn3-epoch0), the exact-geometry pack path +ready metal Apple_M6 dataset-log2 28 batch 4194304 prepare 1 race off +prepared 4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925 69676e65756d2d6461792ffd50000000000000 241.1 program 81.3 dataset 159.8 race 0.0 variant base class v5 loads/hash 128 cache-fill 4.0 build 95.8 resident 1 programs 1 datasets +found 1 0 c6e114e3d317ee3e 4020cb4382e3fe4b +found 1 63 77fe95d9ed898030 4020cb4382e3fe4b +done 1 64 54.50 +== Metal serve smoke, the v4 pack (igneum-devnet-v4-epoch0, class v2 generator path): ready metal Apple_M6 ...; found 1 0 65413452d4d05a3c edc4fa844da9dc98; found 1 63 4b8c6966db6ac21f edc4fa844da9dc98; done 1 64 61.69 +== build: swiftc -O main.swift on the mini (Apple M6, CLT swift 6.4), Thu Oct 8 20:51:52 BST 2026 diff --git a/docs/analysis/review-2026-10-08-b/f10/packfile-test.log b/docs/analysis/review-2026-10-08-b/f10/packfile-test.log new file mode 100644 index 000000000..8479029c4 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/packfile-test.log @@ -0,0 +1,75 @@ +ok the fixture seeds unhex (32 epoch bytes, 19 day bytes) +ok attempt 0 of epoch 34 = the bare seed words (06af2a61 4d67274e ...) +ok attempt 1 of epoch 34 = words of seed || 01000000 (0dcff56b 6b1beb0d ...) +ok the two attempts differ +ok known-good: the attempt-1 pack loads +ok known-good: attempt, words and epoch hex read back +ok known-good: the checked-in attempt-0 pack loads from program.h's own bytes +ok the checked-in pack is attempt 0 +ok known-mismatched: bare words under attempt 1 are refused +ok the refusal names the attempt and the epoch in plain words + (program pack and its seeds disagree: IGNEUM_SEEDW_INIT is not attempt 1 of the epoch seed 009858237e118f69 (attempt 1 gives 0dcff56b 6b1beb0d ..., the pack has 06af2a61 4d67274e ...); run igneum-miner export-pack again) +ok known-mismatched: seeds.txt of epoch 33 with program.h of epoch 34 is refused +ok the refusal says the files disagree +ok no attempt line reads as attempt 0 and the bare words load +ok a generator 2 pack without a class line is class v2 with no era +ok known-failed: a pack with IGNEUM_DATASET_WORDS is sized by the word count (1476395008 words, 92274688 items, multiply-shift), not by 1 << 30 +ok the count is not the floor's power of two and the items are words / 16 +ok known-good: a pack without the line keeps 1 << IGNEUM_DATASET_LOG2 (2^28 words, 2^24 items, the mask path) +ok known-good: the checked-in pack is sized by its power of two +ok generator 9 is refused in plain words +ok a generator 3 pack loads as class v3 with its era seed +ok the v3 pack matches a job naming class v3 and its era +ok a job naming no class accepts the pack +ok a job naming class v2 refuses the v3 pack +ok a job naming another era refuses the v3 pack +ok a job naming class v3 refuses a v2 pack +ok an era named on a v2 job is ignored +ok a generator 4 pack loads as class v4 with its era seed +ok a generator 3 pack with a shadow block (a v4 program stamped v3) is refused +ok a generator 4 pack without the shadow block is refused +ok the v4 pack with its block loads again +ok the v4 pack matches a job naming class v4 and its era +ok a job naming class v3 refuses the v4 pack +ok a job naming another era refuses the v4 pack +ok a job naming class v4 refuses a v3 pack +ok the class=v4 token is read +ok a v3 class line on a generator 4 pack is refused +ok a class line that contradicts the generator is refused +ok a pack with no generator line (generator 1) is refused +ok known-failed: a generator 5 pack without the leaf count is refused in plain words +ok a generator 5 pack with the leaf count loads as class v5 (count, FNV and file name read back) +ok the v5 pack matches a job naming class v5 and its era +ok a job naming class v4 refuses the v5 pack +ok known-failed: the leaves file missing refuses the build in plain words +ok known-failed: a short leaves file is refused with both sizes named +ok known-failed: leaves of another root (one bit) are refused by the FNV +ok known-good: the right leaves load as 2 x 16 words +ok known-failed: a generator 4 pack carrying IGNEUM_STATE_LEAVES is refused +ok a class v4 pack has no leaves and the leaf loader hands back none +ok known-good: the checked-in class v5 pack loads with its leaf count and vectors +ok known-good: the checked-in pack's leaves.bin loads under the pack's FNV-1a 64 + v5 pack: 11 leaves, FNV-1a 64 f141bfee2a8b11b0, state root 7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311 +ok the geometry vectors file opens +ok geometry ds55: loads as 1476395008 words, 92274688 items, mulshift 1, 5905580032 bytes +ok geometry ds85: loads as 2281701376 words, 142606336 items, mulshift 1, 9126805504 bytes +ok geometry ds115: loads as 3087007744 words, 192937984 items, mulshift 1, 12348030976 bytes +ok geometry legacy-2^28: loads as 268435456 words, 16777216 items, mulshift 0, 1073741824 bytes +ok geometry legacy-2^30: loads as 1073741824 words, 67108864 items, mulshift 0, 4294967296 bytes +ok geometry legacy-2^31-explicit-count: loads as 2147483648 words, 134217728 items, mulshift 0, 8589934592 bytes +ok geometry wrong-count-not-a-65536-multiple: refused with "multiple of 65,536" +ok geometry wrong-log2-for-the-count: refused with "IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count" +ok geometry wrong-log2-for-a-power-of-two-count: refused with "IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count" +ok geometry wrong-items: refused with "IGNEUM_DATASET_ITEMS is not IGNEUM_DATASET_WORDS / 16" +ok geometry wrong-mulshift-on-a-non-power-of-two: refused with "IGNEUM_DATASET_MULSHIFT does not match the word count" +ok geometry log2-32: refused with "IGNEUM_DATASET_LOG2 above 31" +ok geometry overflowing-count-2^32: refused with "IGNEUM_DATASET_WORDS above 2^32 - 1" +ok geometry overflowing-count-2^40: refused with "IGNEUM_DATASET_WORDS above 2^32 - 1" +ok geometry count-below-the-minimum: refused with "multiple of 65,536" +ok geometry log2-below-20: refused with "sizes out of range" +ok geometry missing-log2: refused with "has no IGNEUM_DATASET_LOG2" +ok geometry missing-leaves-on-a-class-v5-pack: refused with "without IGNEUM_STATE_LEAVES" +ok geometry leaves-on-a-class-v2-pack: refused with "state leaves belong to class v5" +ok the vectors file carries every case (18 or more) + 19 geometry vectors +/tmp/igneum-packfile-test/packfile-test: 0 failure(s) diff --git a/docs/analysis/review-2026-10-08-b/f10/rows.md b/docs/analysis/review-2026-10-08-b/f10/rows.md new file mode 100644 index 000000000..cc7f4eea7 --- /dev/null +++ b/docs/analysis/review-2026-10-08-b/f10/rows.md @@ -0,0 +1,109 @@ +# F10 worker lane: the CUDA select pass, measured (8 October 2026) + +Review B finding F10 (docs/analysis/review-2026-10-08-b/findings.json): the CUDA serving loop read 8 bytes per nonce back +over the bus (128 MiB per 2^24-nonce dispatch) and scanned them on the host under the GPU mutex. The select pass the +OpenCL worker has carried since 5 October 2026 is now on CUDA (proto-cuda/nvrtc/worker.cpp, landed on master as +964cdf6c, commits 12cb65cc and 4b0de631): a tiny `igneum_select` kernel built once by NVRTC writes the hits behind an +atomic counter plus 34 sentinel words, the host reads back a few hundred bytes, a chunk with more than 256 hits falls +back to the full read, the found lines come in nonce order from the sorted hits and carry the kernel's own hash word. +Byte-preserving: the bound kernel and its result layout are untouched; no valid hash changes. + +Two numbers per card, never interchangeable: the **kernel timer** (`--bench`, the dispatch alone) and the +**serving-mode numbers** (`proto-cuda/nvrtc/serve-bench.sh`: the worker in `--serve` driven by a stub pool exactly as +the miner drives it, one job queued behind the one in flight, every found line read at the miner's submit point). The +**production** number runs from the worker process start (NVRTC compile, cache, dataset, self-test included) to the last +done line; the **steady** number from the ready line. Power is nvidia-smi's board reading at 4 Hz over the same window +(a RunPod container exposes no host rail; the worker's own CPU seconds stand in for the host-side cost of the read-back +and scan). Target 00000fffffffffff (one hit per 2^20 nonces, 16 found lines per job), 32 jobs of 2^24 nonces, `--race off`, +variant base, pack hl-v6-all (class v6 kit packs-class-v6-20261008T162324Z, class string +mx8-erad810f22d+sh256x27+state+reg64c+fold+rw). Worker binary sha256 9b36b523d0812c3c (Ubuntu 24.04 build of 12cb65cc). +The worker has no stale-job cancellation (a job runs to its done line; the next line waits), so the stub's queued job is +the stale case the miner produces; nothing is cancelled mid-dispatch on any backend today. + +## Equivalence (build-7, CPU emulation, proto-cuda/nvrtc/emu/test.sh, green 20:20 and again 20:55 UK) + +`emu/select-check.sh`, known-failed first: the fault run (IGNEUM_READBACK_FAULT_TEST skips the counter reset from the +second chunk on) is flagged with 84 stale lines; then `--readback select` equals the full read line for line over 2255 +found lines and 10 jobs: zero hits, partial hits, the all-hit overflow (every chunk over 256 hits, 2 full fallbacks), a +96-nonce tail with a 32-lane launch (POW-01 in the record), the high-32 rollover (ROT-01 in the record), two queued jobs, +hits then a zero chunk, a mismatch error then a job, and every found line's fifth field equal to the pack's epoch (I05). +`emu/serve-check.sh`: 17 sampled hashes on both packs and both sides of the 32-bit boundary equal igneum-pow hash-bound; +the swap case's found lines carry pack B's epoch. Logs: emu-test.log beside this file. + +## Kernel timer + +| card | arch | ptxas (nvcc 12.8, the pack's kernel_bound.cu) | --bench 5 x 2^24 | kernel MH/s | +|---|---|---|---|---| +| RTX 5090 (pod 4jpym0xwr1jnf3, driver 595.91.07) | sm_120 | 92 registers, 0 bytes spill stores, 0 bytes spill loads, 0 bytes stack; worker: 20 blocks/SM, 3400 resident warps | 236.75 ms | 70.86 | +| RTX 3090 (pod xq411hlgu4qzri, driver 595.71.05) | sm_86 | 87 registers, 0 bytes spill stores, 0 bytes spill loads, 0 bytes stack, 408 bytes cmem; worker: 16 blocks/SM, 1312 resident warps | warm-up 554.73 ms, timed mean 2590.06 ms (throttled, see below) | 30.2 at clock, 6.5 throttled | + +## Serving mode + +The pod rows (the .row files beside this file carry every field; the clocks are nvidia-smi clocks.sm means over each run): + +### 5090 + +Mean SM clock 2387 to 2823 MHz over the runs (2895 max); no throttle reason active. + +| run | mode | setup s | production MH/s | production J per batch | steady MH/s | steady W | steady J per batch | steady MH per W | worker CPU s | per chunk: kernel+sync, select, read-back, scan ms | +|---|---|---|---|---|---|---|---|---|---|---| +| 5090-full-r1 | full | 2.275 | 50.10 | 93.1 | 63.61 | 328.3 | 86.6 | 0.1938 | (subshell pid, not read) | 236.64, 0.000, 12.794, 13.245 | +| 5090-full-r2 | full | 2.373 | 47.07 | 96.5 | 59.45 | 326.9 | 92.3 | 0.1819 | (subshell pid, not read) | 236.65, 0.000, 25.097, 19.536 | +| 5090-full-r3 | full | 2.400 | 50.73 | 88.5 | 65.62 | 335.4 | 85.8 | 0.1956 | 2.97 | 236.73, 0.000, 7.679, 10.341 | +| 5090-full-r4 | full | 2.334 | 50.81 | 94.4 | 65.21 | 351.5 | 90.4 | 0.1855 | 2.92 | 237.01, 0.000, 8.639, 10.488 | +| 5090-full-r5 | full | 2.463 | 49.86 | 93.1 | 64.65 | 340.3 | 88.3 | 0.1900 | 3.13 | 236.70, 0.000, 9.028, 12.796 | +| 5090-select-r1 | select | 2.447 | 53.31 | 91.4 | 70.42 | 362.5 | 86.4 | 0.1943 | (subshell pid, not read) | 236.71, 0.234, 0.178, 0.005 | +| 5090-select-r2 | select | 2.408 | 53.52 | 90.6 | 70.42 | 363.0 | 86.5 | 0.1940 | (subshell pid, not read) | 236.69, 0.245, 0.270, 0.007 | +| 5090-select-r3 | select | 2.327 | 53.95 | 90.3 | 70.42 | 365.4 | 87.1 | 0.1927 | 2.33 | 236.71, 0.262, 0.353, 0.006 | +| 5090-select-r4 | select | 2.590 | 52.55 | 91.1 | 70.40 | 355.9 | 84.8 | 0.1978 | 2.55 | 236.63, 0.239, 0.320, 0.010 | +| 5090-select-r5 | select | 2.393 | 53.59 | 90.0 | 70.40 | 360.8 | 86.0 | 0.1951 | 2.39 | 236.78, 0.296, 0.244, 0.007 | + +5090 reading (five pairs): steady 70.41 MH/s select against 63.71 full (+10.5 percent; select sits at the kernel timer's +70.86, the read-back cost is gone); production 53.38 against 49.71 (+7.4 percent; the rest of the gap to the kernel is +the 2.3 to 2.6 s setup of each 7.6 s run); energy per 2^24 batch 86.2 J select against 88.7 J full (-2.8 percent; the card +draws more while it is never idle, 361 W against 337 W, and finishes sooner); 0.195 against 0.188 MH per W; worker CPU +2.3 to 2.6 s against 2.9 to 3.1 s per 32 jobs. The full path pays 7.7 to 25.1 ms of read-back plus 10.3 to 19.5 ms of host +scan per 236.7 ms chunk (8 to 19 percent); the select pass costs 0.23 to 0.30 ms plus 0.18 to 0.35 ms of read-back. + +### 3090 + +The pod's host holds the card at 210 MHz under `clocks_event_reasons.sw_power_cap` for whole runs (Active at 21 W idle; +`-lgc` and `-pl` refused in the container), so the rows split by the mean SM clock of each run. + +| run | mode | mean SM MHz | setup s | production MH/s | production J per batch | steady MH/s | steady W | steady J per batch | steady MH per W | worker CPU s | per chunk: kernel+sync, select, read-back, scan ms | +|---|---|---|---|---|---|---|---|---|---|---|---| +| 3090-full-r3 | full | 1769 | 4.110 | 22.81 | 207.7 | 27.63 | 325.9 | 197.9 | 0.0848 | 5.61 | 557.15, 0.000, 24.225, 23.753 | +| 3090-select-r2 | select | 1733 | 4.138 | 24.41 | 196.7 | 30.06 | 330.7 | 184.6 | 0.0909 | (subshell pid, not read) | 555.70, 0.268, 0.217, 0.013 | +| 3090-full-r1 | full | 632 | 4.710 | 9.18 | 301.4 | 9.98 | 168.3 | 282.9 | 0.0593 | (subshell pid, not read) | 1630.34, 0.000, 24.723, 24.119 | +| 3090-full-r2 | full | 210 | 4.231 | 5.96 | 378.4 | 6.25 | 136.9 | 367.3 | 0.0457 | (subshell pid, not read) | 2637.86, 0.000, 22.784, 20.625 | +| 3090-full-r4 | full | 210 | 4.397 | 5.96 | 377.9 | 6.27 | 136.7 | 365.9 | 0.0459 | 5.81 | 2624.25, 0.000, 27.536, 22.950 | +| 3090-select-r1 | select | 210 | 7.093 | 5.86 | 376.9 | 6.35 | 136.9 | 361.5 | 0.0464 | (subshell pid, not read) | 2637.57, 0.532, 0.296, 0.019 | +| 3090-select-r3 | select | 318 | 4.297 | 6.44 | 361.6 | 6.78 | 140.1 | 346.4 | 0.0484 | 4.23 | 2470.39, 0.503, 0.206, 0.017 | + +3090 reading (the one pair at the card's clock): steady 30.06 MH/s select against 27.63 full (+8.8 percent); production +24.41 against 22.81 (+7.0 percent); 184.6 J per batch against 197.9 (-6.7 percent); the full path's read-back plus scan is +48 ms of a 557 ms chunk (8.6 percent). The 6.5 MH/s kernel read of the evening is the 210 MHz pod state, not the class: +ptxas on sm_86 shows 87 registers and no spill or stack bytes (sm_120: 92 registers, none); the worker's occupancy is 16 +blocks of one warp per SM on sm_86 against 20 on sm_120, register-limited on both. At clock the 5090 to 3090 gap is 2.34x +(70.4 to 30.1), beside the memory gap (1,792 to 936 GB/s, 96 MiB to 6 MiB of L2); the 10.9x read was the power cap. A +consumer-tier number for the site needs a 3090 whose clock the host does not cap; the fleet lane holds the pod id. + +## Record + +Harness cells (tools/ci/test-map.json): `harness:cuda-select-check` (ROT-01 partial: the high-32 rollover case; POW-01 +partial: the select pass equals the full read on every case named above, the hash sample against igneum-pow) and +`harness:pack-geometry` (POW-02 partial: the geometry rule on 19 conformance vectors in C on build-7 and in Swift on the +mini, and the two checked-in packs). Batch tools/ci/batches/f10-worker-20261008.json. + +## I06 (master review, gate INT-10) and I05, landed with the rows + +One normative geometry rule: packfile.h pf_load (CUDA and the OpenCL generic worker) gained the 64-bit count read with +the 2^32 refusal and the floor-log2 rule; proto-metal/main.swift gained `packGeometry` (the same rules, the same refusal +words) and sizes the day's dataset from the exact word count, refusing before any Metal work; `--geometry-check + [pack dir...]` runs the vectors. proto-cuda/nvrtc/emu/geometry-vectors.txt is the shared file (ds55, ds85, +ds115, the power-of-two sizes, wrong count or log2, overflowing counts, missing leaves, leaves on a v2 pack), read by +emu/packfile-test.c on build-7 (19 vectors, 0 failures) and by the Metal host on the mini (19 vectors, 2 packs, 0 +failures; geometry-check-mini.log). I05: every worker's found line carries a fifth field, the epoch the hash was computed +under (16 hex): CUDA and OpenCL from the pair, Metal from the job (metal-serve-smoke.log on the mini: the v4 pack +through the Swift path, the class v5 pack through the exact-geometry pack path). The miner reads a found line with four or +more fields (igneum/miner/src/main.rs:778). diff --git a/docs/plans/igneum-2.0-test-harness-map.md b/docs/plans/igneum-2.0-test-harness-map.md index 4ddb38250..067d16630 100644 --- a/docs/plans/igneum-2.0-test-harness-map.md +++ b/docs/plans/igneum-2.0-test-harness-map.md @@ -268,6 +268,23 @@ Rule: a case maps to a cell only where the cell's tests visibly answer it; cover - UX-05 Keep voting keys with the miner through pooling: partial: the vote-key commitment on the wire (verify::tests::a_share_under_another_key_is_refused_before_the_hash: known-pass the member's key, known-fail another key on the share and a job naming another key, refused with code vote_key before the hash), the open pool's sidechain::check_key (the claimed key, the header's key and the reveal are one key), the same-nonce-other-template wrong_hash test, the TLS binding test; the malicious-pool run and the leave-and-retain run on the devnet-4 pair are the UX-05 batch's other evidence - UX-04 Pay small operators without hidden custody: partial: PPLNS distribution tests (fee first, never overpays, late joiner), the payout key file round trip, the signed transfer decode; the live payout path on the devnet-4 pair is the UX-04 batch's evidence +### harness:cuda-select-check + +- Command: `proto-cuda/nvrtc/emu/test.sh on a box (the CUDA worker under the CPU emulation backend): emu/select-check.sh (--readback select against the full read, known-failed first) and emu/serve-check.sh (the hash sample against igneum-pow hash-bound, both sides of the 32-bit boundary, the swap)` +- Box class: suite (CPU emulation, any build box) +- Fixtures: F0, F1 +- Cases: + - ROT-01 Agree across every hourly boundary: partial: the worker's high-32 nonce rollover inside a job (a job from 2^32 - 32) finds the same set on both read-back paths and the boundary nonces hash as igneum-pow; the hourly program boundary itself is the fast-time harness + - POW-01 Match independent execution across every backend: partial: the CUDA select pass equals the full read line for line (zero hits, partial, the all-hit overflow fallback, a 96-nonce tail with a 32-lane launch, queued jobs, hits then zero, the epoch tag) and 17 sampled hashes equal igneum-pow; the million-vector campaign is harness:p01-vectors + +### harness:pack-geometry + +- Command: `proto-cuda/nvrtc/emu/packfile-test.sh on a box (packfile.h pf_load over emu/geometry-vectors.txt and the checked-in packs) and igneum-bench --geometry-check proto-cuda/nvrtc/emu/geometry-vectors.txt on the Mac mini (proto-metal/main.swift packGeometry over the same file)` +- Box class: suite (a build box for C; the mini for Swift) +- Fixtures: F0 +- Cases: + - POW-02 Validate generated programs and index folding: partial: the dataset geometry rule (exact word count, items, multiply-shift, 64-bit bytes, the floor-log2 and 2^32 refusals, missing leaves) agrees across the CUDA and OpenCL loader and the Metal host on 19 conformance vectors and the two checked-in packs (master review I06, gate INT-10); the rest of POW-02 is suite:pow + ## Automated cases with no harness in the matrix (NOT RUN, the reason) - GOV-02 Approve thresholds before results: the approval is recorded in the registry's approval field; the automated half (thresholds frozen before any run_status) is the gate rule landing by 21:00 diff --git a/docs/plans/igneum-2.0-test-registry.json b/docs/plans/igneum-2.0-test-registry.json index 412ea511f..39eb818ae 100644 --- a/docs/plans/igneum-2.0-test-registry.json +++ b/docs/plans/igneum-2.0-test-registry.json @@ -708,18 +708,17 @@ "manual_page": 24, "owner_lane": "hash lane (a690540514aa453d7)", "run_status": "RUNNING", - "evidence_path": "build-1:/srv/artefacts/tas/201-7cfa422a-aa0e0f45/node-7cfa422a/box4-miner.log", - "run_id": "201-7cfa422a-aa0e0f45", - "updated": "2026-10-08T19:32:50.856Z", + "evidence_path": "docs/analysis/review-2026-10-08-b/f10/emu-test.log", + "run_id": "f10-worker-20261008", + "updated": "2026-10-08T19:52:55.240Z", "evidence_record": { - "cell": "suite:miner", - "manifest_sha": "7cfa422a", + "cell": "harness:cuda-select-check", + "manifest_sha": "964cdf6c1", "coverage": { - "UX-06": "partial: the miner's own template selection tests", - "UX-04": "partial: payout label and share accounting tests; custody is the pool lane's row", - "POW-01": "partial: the pack re-check seam against the pack's 96 vectors (recheck_pack); the million-vector campaign is harness:p01-vectors" + "ROT-01": "partial: the worker's high-32 nonce rollover inside a job (a job from 2^32 - 32) finds the same set on both read-back paths and the boundary nonces hash as igneum-pow; the hourly program boundary itself is the fast-time harness", + "POW-01": "partial: the CUDA select pass equals the full read line for line (zero hits, partial, the all-hit overflow fallback, a 96-nonce tail with a 32-lane launch, queued jobs, hits then zero, the epoch tag) and 17 sampled hashes equal igneum-pow; the million-vector campaign is harness:p01-vectors" }, - "at": "2026-10-08T19:32:50.856Z" + "at": "2026-10-08T19:52:55.240Z" } }, { @@ -749,18 +748,16 @@ "manual_page": 24, "owner_lane": "hash lane (a690540514aa453d7)", "run_status": "RUNNING", - "evidence_path": "build-1:/srv/artefacts/tas/201-7cfa422a-aa0e0f45/miner-9c844503/box2-pow.log", - "run_id": "201-7cfa422a-aa0e0f45", - "updated": "2026-10-08T19:32:50.856Z", + "evidence_path": "docs/analysis/review-2026-10-08-b/f10/packfile-test.log;docs/analysis/review-2026-10-08-b/f10/geometry-check-mini.log", + "run_id": "f10-worker-20261008", + "updated": "2026-10-08T19:52:55.240Z", "evidence_record": { - "cell": "suite:pow", - "manifest_sha": "7cfa422a", + "cell": "harness:pack-geometry", + "manifest_sha": "964cdf6c1", "coverage": { - "POW-02": "partial: program derivation, ds55 geometry, the v6 fold, the packs and the spec read-back tests; the independent re-implementation and the index-fold census are the hash lane's harnesses", - "POW-08": "partial: the freeze list and the generator version pin; the public-claim scrub is the site gate's", - "POW-07": "the mixed FP32 branch is excluded and unreachable: no class flag reaches an FP op in igneum-pow on master (the mixedfp lane's branch is not on master); evidence the grep of the emitters on master, recorded by the hash lane" + "POW-02": "partial: the dataset geometry rule (exact word count, items, multiply-shift, 64-bit bytes, the floor-log2 and 2^32 refusals, missing leaves) agrees across the CUDA and OpenCL loader and the Metal host on 19 conformance vectors and the two checked-in packs (master review I06, gate INT-10); the rest of POW-02 is suite:pow" }, - "at": "2026-10-08T19:32:50.856Z" + "at": "2026-10-08T19:52:55.240Z" } }, { @@ -1353,14 +1350,17 @@ "manual_page": 30, "owner_lane": "fast-time lane (a8be71a0db962911c)", "run_status": "RUNNING", - "evidence_path": "v5-fasttime 92bf6a7f", - "run_id": "team-2026-10-08", - "updated": "2026-10-08 18:3x UK", + "evidence_path": "docs/analysis/review-2026-10-08-b/f10/emu-test.log", + "run_id": "f10-worker-20261008", + "updated": "2026-10-08T19:52:55.240Z", "evidence_record": { - "what_was_run": "the fast-time crossings PASS on 4cdcc488 (17:29:58) and 617cb441 (17:30:44) with the cold restart", - "run_by": "fast-time lane (a8be71a0db962911c)", - "under_the_standard": "no: team-run before the standard's procedure; the status is RUNNING until the case is re-run under its steps with the profile's numbers and an independent run where the profile asks one", - "pass_or_fail_today": "not judged under the standard yet" + "cell": "harness:cuda-select-check", + "manifest_sha": "964cdf6c1", + "coverage": { + "ROT-01": "partial: the worker's high-32 nonce rollover inside a job (a job from 2^32 - 32) finds the same set on both read-back paths and the boundary nonces hash as igneum-pow; the hourly program boundary itself is the fast-time harness", + "POW-01": "partial: the CUDA select pass equals the full read line for line (zero hits, partial, the all-hit overflow fallback, a 96-nonce tail with a 32-lane launch, queued jobs, hits then zero, the epoch tag) and 17 sampled hashes equal igneum-pow; the million-vector campaign is harness:p01-vectors" + }, + "at": "2026-10-08T19:52:55.240Z" } }, { diff --git a/proto-cuda/nvrtc/emu/geometry-vectors.txt b/proto-cuda/nvrtc/emu/geometry-vectors.txt new file mode 100644 index 000000000..b657b41e1 --- /dev/null +++ b/proto-cuda/nvrtc/emu/geometry-vectors.txt @@ -0,0 +1,124 @@ +# Pack geometry conformance vectors (8 October 2026, master review I06, gate INT-10): one normative rule set for the +# dataset geometry of a program pack, read by every host the same way: packfile.h pf_load (the CUDA worker and the +# OpenCL generic worker; checked by emu/packfile-test.c) and proto-metal/main.swift packGeometry (checked by +# igneum-bench --geometry-check). A host allocates, builds and hashes from the exact word count, never from the log2 +# alone, with 64-bit byte sizes, and refuses an unsupported pack before any GPU launch. +# +# Each case: `case `, then `define ` lines appended to a base program.h that carries everything +# but the geometry (seeds, generator 2, attempt, the init words, the cache sizes, IGNEUM_DATASET_MODE 1), then one +# `expect ok words= items= mulshift=<0|1> bytes=` or `expect refuse `. +# A `define IGNEUM_GENERATOR` or `define IGNEUM_STATE_LEAVES` line overrides the base's class fields. + +case ds55 +define IGNEUM_DATASET_LOG2 30 +define IGNEUM_DATASET_WORDS 1476395008u +define IGNEUM_DATASET_ITEMS 92274688u +define IGNEUM_DATASET_BYTES 5905580032ull +define IGNEUM_DATASET_MULSHIFT 1 +expect ok words=1476395008 items=92274688 mulshift=1 bytes=5905580032 + +case ds85 +define IGNEUM_DATASET_LOG2 31 +define IGNEUM_DATASET_WORDS 2281701376u +define IGNEUM_DATASET_ITEMS 142606336u +define IGNEUM_DATASET_MULSHIFT 1 +expect ok words=2281701376 items=142606336 mulshift=1 bytes=9126805504 + +case ds115 +define IGNEUM_DATASET_LOG2 31 +define IGNEUM_DATASET_WORDS 3087007744u +define IGNEUM_DATASET_ITEMS 192937984u +define IGNEUM_DATASET_MULSHIFT 1 +expect ok words=3087007744 items=192937984 mulshift=1 bytes=12348030976 + +case legacy-2^28 +define IGNEUM_DATASET_LOG2 28 +expect ok words=268435456 items=16777216 mulshift=0 bytes=1073741824 + +case legacy-2^30 +define IGNEUM_DATASET_LOG2 30 +expect ok words=1073741824 items=67108864 mulshift=0 bytes=4294967296 + +case legacy-2^31-explicit-count +define IGNEUM_DATASET_LOG2 31 +define IGNEUM_DATASET_WORDS 2147483648u +define IGNEUM_DATASET_ITEMS 134217728u +define IGNEUM_DATASET_MULSHIFT 0 +expect ok words=2147483648 items=134217728 mulshift=0 bytes=8589934592 + +case wrong-count-not-a-65536-multiple +define IGNEUM_DATASET_LOG2 30 +define IGNEUM_DATASET_WORDS 1476395009u +define IGNEUM_DATASET_ITEMS 92274688u +define IGNEUM_DATASET_MULSHIFT 1 +expect refuse multiple of 65,536 + +case wrong-log2-for-the-count +define IGNEUM_DATASET_LOG2 29 +define IGNEUM_DATASET_WORDS 1476395008u +define IGNEUM_DATASET_ITEMS 92274688u +define IGNEUM_DATASET_MULSHIFT 1 +expect refuse IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count + +case wrong-log2-for-a-power-of-two-count +define IGNEUM_DATASET_LOG2 28 +define IGNEUM_DATASET_WORDS 1073741824u +define IGNEUM_DATASET_ITEMS 67108864u +define IGNEUM_DATASET_MULSHIFT 0 +expect refuse IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count + +case wrong-items +define IGNEUM_DATASET_LOG2 30 +define IGNEUM_DATASET_WORDS 1476395008u +define IGNEUM_DATASET_ITEMS 92274689u +define IGNEUM_DATASET_MULSHIFT 1 +expect refuse IGNEUM_DATASET_ITEMS is not IGNEUM_DATASET_WORDS / 16 + +case wrong-mulshift-on-a-non-power-of-two +define IGNEUM_DATASET_LOG2 30 +define IGNEUM_DATASET_WORDS 1476395008u +define IGNEUM_DATASET_ITEMS 92274688u +define IGNEUM_DATASET_MULSHIFT 0 +expect refuse IGNEUM_DATASET_MULSHIFT does not match the word count + +case log2-32 +define IGNEUM_DATASET_LOG2 32 +expect refuse IGNEUM_DATASET_LOG2 above 31 + +case overflowing-count-2^32 +define IGNEUM_DATASET_LOG2 31 +define IGNEUM_DATASET_WORDS 4294967296ull +define IGNEUM_DATASET_MULSHIFT 0 +expect refuse IGNEUM_DATASET_WORDS above 2^32 - 1 + +case overflowing-count-2^40 +define IGNEUM_DATASET_LOG2 31 +define IGNEUM_DATASET_WORDS 1099511627776ull +expect refuse IGNEUM_DATASET_WORDS above 2^32 - 1 + +case count-below-the-minimum +define IGNEUM_DATASET_LOG2 15 +define IGNEUM_DATASET_WORDS 32768u +expect refuse multiple of 65,536 + +case log2-below-20 +define IGNEUM_DATASET_LOG2 19 +expect refuse sizes out of range + +case missing-log2 +define IGNEUM_DATASET_WORDS 1476395008u +expect refuse has no IGNEUM_DATASET_LOG2 + +case missing-leaves-on-a-class-v5-pack +define IGNEUM_GENERATOR 5 +define IGNEUM_PROGRAM_CLASS "v5" +define IGNEUM_ERA_SEED_HEX "bed7ab62cbece66cf791485336d81d90fa1452ffed28ecd8a7416960ef64164c" +define IGNEUM_SHADOW_INSTRS 256 +define IGNEUM_SHADOW_REPS 27 +define IGNEUM_DATASET_LOG2 28 +expect refuse without IGNEUM_STATE_LEAVES + +case leaves-on-a-class-v2-pack +define IGNEUM_STATE_LEAVES 93 +define IGNEUM_DATASET_LOG2 28 +expect refuse state leaves belong to class v5 diff --git a/proto-cuda/nvrtc/emu/packfile-test.c b/proto-cuda/nvrtc/emu/packfile-test.c index f962386df..2ed5f241a 100644 --- a/proto-cuda/nvrtc/emu/packfile-test.c +++ b/proto-cuda/nvrtc/emu/packfile-test.c @@ -269,6 +269,56 @@ int main(int argc, char** argv) { } } + // 6. The geometry conformance vectors (8 October 2026, master review I06, gate INT-10): emu/geometry-vectors.txt, one + // normative rule set shared with the Metal host (igneum-bench --geometry-check reads the same file). Each case's defines + // are appended to a base program.h with everything but the geometry; pf_load must size the pack exactly as the vector + // says (words, items, mulshift, 64-bit bytes) or refuse it with the named words. Known-failed first: the vectors file + // itself carries the refusals (an overflowing count, a log2 that is not the floor, a wrong item count). + if (argc > 3) { + FILE* vf = fopen(argv[3], "rb"); + char line[512], name[128] = {0}, defs[4096] = {0}, text[8192], sw[200], kw[200]; + int cases = 0, inCase = 0; + words_hex(att1, sw); words_hex(keyw, kw); + CHECK(vf != NULL, "the geometry vectors file opens"); + while (vf && fgets(line, sizeof(line), vf)) { + line[strcspn(line, "\r\n")] = 0; + if (line[0] == '#' || line[0] == 0) continue; + if (strncmp(line, "case ", 5) == 0) { snprintf(name, sizeof(name), "%s", line + 5); defs[0] = 0; inCase = 1; continue; } + if (!inCase) continue; + if (strncmp(line, "define ", 7) == 0) { snprintf(defs + strlen(defs), sizeof(defs) - strlen(defs), "#define %s\n", line + 7); continue; } + if (strncmp(line, "expect ", 7) == 0) { + const char* ex = line + 7; + int hasGen = strstr(defs, "#define IGNEUM_GENERATOR ") != NULL; + char what[400]; + snprintf(text, sizeof(text), + "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n%s#define IGNEUM_PROGRAM_ATTEMPT 1\n#define IGNEUM_DATASET_MODE 1\n" + "#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n%s", + EPOCH_34, DAY_20731, hasGen ? "" : "#define IGNEUM_GENERATOR 2\n", sw, kw, defs); + write_file(dir, "program.h", text); + write_file(dir, "seeds.txt", "epoch_seed_hex " EPOCH_34 "\nday_seed_hex " DAY_20731 "\n"); + err[0] = 0; + ++cases; + if (strncmp(ex, "ok ", 3) == 0) { + unsigned long long w = 0, it = 0, ms = 0, by = 0; + int ok = pf_load(dir, &pk, err, sizeof(err)); + sscanf(ex, "ok words=%llu items=%llu mulshift=%llu bytes=%llu", &w, &it, &ms, &by); + snprintf(what, sizeof(what), "geometry %s: loads as %llu words, %llu items, mulshift %llu, %llu bytes", name, w, it, ms, by); + CHECK(ok == 1 && pk.datasetWords == (uint32_t)w && pk.datasetItems == (uint32_t)it && pk.datasetMulshift == (uint32_t)ms && (uint64_t)pk.datasetWords * 4ull == by, what); + if (!ok) printf(" (%s)\n", err); + } else if (strncmp(ex, "refuse ", 7) == 0) { + int ok = pf_load(dir, &pk, err, sizeof(err)); + snprintf(what, sizeof(what), "geometry %s: refused with \"%s\"", name, ex + 7); + CHECK(ok == 0 && strstr(err, ex + 7) != NULL, what); + if (ok) printf(" (loaded: %u words)\n", (unsigned)pk.datasetWords); else if (strstr(err, ex + 7) == NULL) printf(" (said: %s)\n", err); + } else { CHECK(0, "a vector's expect line is ok or refuse"); } + inCase = 0; + } + } + if (vf) fclose(vf); + CHECK(cases >= 18, "the vectors file carries every case (18 or more)"); + printf(" %d geometry vectors\n", cases); + } + printf("%s: %d failure(s)\n", argv[0], failures); return failures ? 1 : 0; } diff --git a/proto-cuda/nvrtc/emu/packfile-test.sh b/proto-cuda/nvrtc/emu/packfile-test.sh index d945d0edf..4602be0bf 100755 --- a/proto-cuda/nvrtc/emu/packfile-test.sh +++ b/proto-cuda/nvrtc/emu/packfile-test.sh @@ -1,6 +1,7 @@ #!/usr/bin/env bash # The pack loader's seed rule (packfile.h) on a known-good and a known-mismatched pack, and the class v5 leaf rule on the -# checked-in v5 pack (7 October 2026): emu/packfile-test.c, C99, no GPU. Runs on the Mac in a second and in CI. Usage: emu/packfile-test.sh +# checked-in v5 pack (7 October 2026), and the geometry conformance vectors (emu/geometry-vectors.txt, 8 October 2026, shared with +# the Metal host): emu/packfile-test.c, C99, no GPU. Runs on a box in a second and in CI. Usage: emu/packfile-test.sh set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" ROOT="$(cd "$HERE/../../.." && pwd)" @@ -8,4 +9,4 @@ OUT="${TMPDIR:-/tmp}/igneum-packfile-test" mkdir -p "$OUT" CC="${CC:-cc}" "$CC" -std=c99 -Wall -Wextra -Wno-unused-function -O1 -o "$OUT/packfile-test" "$HERE/packfile-test.c" -"$OUT/packfile-test" "$ROOT/proto-cuda/packs/igneum-devnet-v4-epoch0" "$ROOT/proto-cuda/packs-ca3-v5/v5-dn3-epoch0" +"$OUT/packfile-test" "$ROOT/proto-cuda/packs/igneum-devnet-v4-epoch0" "$ROOT/proto-cuda/packs-ca3-v5/v5-dn3-epoch0" "$HERE/geometry-vectors.txt" diff --git a/proto-cuda/nvrtc/emu/select-check.sh b/proto-cuda/nvrtc/emu/select-check.sh index 1a557268c..dd29a322d 100755 --- a/proto-cuda/nvrtc/emu/select-check.sh +++ b/proto-cuda/nvrtc/emu/select-check.sh @@ -13,6 +13,7 @@ # found_set_equals_full_stale_job_queued s1, s2: two job lines queued at once, the second served after the first # buffer_reuse_hits_then_zero b1 then b2: a hit chunk followed by a zero-target chunk prints nothing stale # mismatch_error_then_job m1 (no pair for its seeds) errors the same way, c1 after it is served +# epoch_tag every found line carries the pack's epoch (16 hex) as its fifth field, both modes (I05) # Usage: select-check.sh (the command carries --serve --pack etc.) set -euo pipefail PACK_A="$1"; OUT="$2"; shift 2 @@ -74,6 +75,7 @@ grep -q '^done r1 2048' "$OUT/rb-select.norm" || { echo "FAIL: high32_rollover: awk '$1 == "found" && $2 == "r1" && $3 + 0 >= 4294967296 { n++ } END { exit n > 0 ? 0 : 1 }' "$OUT/rb-select.norm" || { echo "FAIL: high32_rollover: no found nonce at or past 2^32"; exit 1; } grep -q '^done s1 2048' "$OUT/rb-select.norm" && grep -q '^done s2 2048' "$OUT/rb-select.norm" || { echo "FAIL: stale_job_queued: s1 or s2 did not finish"; exit 1; } [ "$(count b1 "$OUT/rb-select.norm")" -gt 0 ] && [ "$(count b2 "$OUT/rb-select.norm")" = 0 ] || { echo "FAIL: buffer_reuse_hits_then_zero: b1 $(count b1 "$OUT/rb-select.norm") found, b2 $(count b2 "$OUT/rb-select.norm")"; exit 1; } +awk -v e="${A_EPOCH:0:16}" '$1 == "found" && $5 != e { bad++ } END { exit bad > 0 ? 1 : 0 }' "$OUT/rb-select.norm" && awk -v e="${A_EPOCH:0:16}" '$1 == "found" && $5 != e { bad++ } END { exit bad > 0 ? 1 : 0 }' "$OUT/rb-full.norm" || { echo "FAIL: epoch_tag: a found line lacks the pack's epoch ${A_EPOCH:0:16} as its fifth field"; exit 1; } grep -q '^error m1 ' "$OUT/rb-select.norm" && grep -q '^done c1 1024' "$OUT/rb-select.norm" || { echo "FAIL: mismatch_error_then_job: m1 did not error or c1 was not served"; exit 1; } grep -q 'full fallbacks' "$OUT/rb-select.log" || { echo "FAIL: no transfers line in the select run"; exit 1; } fb="$(sed -n 's/.*(select, \([0-9]*\) full fallbacks).*/\1/p' "$OUT/rb-select.log" | tail -1)" diff --git a/proto-cuda/nvrtc/emu/serve-check.sh b/proto-cuda/nvrtc/emu/serve-check.sh index 557ddc95a..dfa71469b 100755 --- a/proto-cuda/nvrtc/emu/serve-check.sh +++ b/proto-cuda/nvrtc/emu/serve-check.sh @@ -53,6 +53,8 @@ elif grep -q '^info job 5 is for epoch .* building its pack' "$LOG" && [ "$(grep else echo "FAIL: job 5 was neither refused nor self-healed"; exit 1 fi +[ "$(awk -v e="${A_EPOCH:0:16}" '$1 == "found" && ($2 == 1 || $2 == 2 || $2 == 3) && $5 != e' "$LOG" | wc -l | tr -d ' ')" = 0 ] || { echo "FAIL: a found line of a pack A job does not carry epoch ${A_EPOCH:0:16} as its fifth field"; exit 1; } +[ "$(awk -v e="${B_EPOCH:0:16}" '$1 == "found" && $2 == 4 && $5 != e' "$LOG" | wc -l | tr -d ' ')" = 0 ] || { echo "FAIL: a found line of job 4 (pack B after the swap) does not carry epoch ${B_EPOCH:0:16} as its fifth field"; exit 1; } grep -q '^found 2 4294967295 ' "$LOG" && grep -q '^found 2 4294967296 ' "$LOG" || { echo "FAIL: the 32-bit boundary nonces are missing"; exit 1; } for x in 0 1 2 31 32 63; do check 1 $x "$A_EPOCH" "$A_DAY"; done for x in 4294967264 4294967295 4294967296 4294967327; do check 2 $x "$A_EPOCH" "$A_DAY"; done diff --git a/proto-cuda/nvrtc/packfile.h b/proto-cuda/nvrtc/packfile.h index 57e332d0d..f99ddd87d 100644 --- a/proto-cuda/nvrtc/packfile.h +++ b/proto-cuda/nvrtc/packfile.h @@ -302,8 +302,23 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { if (!pf_define_u32(prog, "IGNEUM_DATASET_LOG2", &pk->datasetLog2)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_DATASET_LOG2"); } if (!pf_define_u32(prog, "IGNEUM_DATASET_MODE", &pk->datasetMode)) pk->datasetMode = 0; if (pk->datasetLog2 > 31) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_LOG2 above 31: the word index is 32-bit"); } - if (!pf_define_u32(prog, "IGNEUM_DATASET_WORDS", &pk->datasetWords)) pk->datasetWords = 1u << pk->datasetLog2; + { + // the word count is parsed at 64 bits first: a count at or above 2^32 (ds115 is 3,087,007,744 words, under it) would wrap + // in the 32-bit field and size the dataset at a fraction of the pack (the geometry vectors, emu/geometry-vectors.txt, 8 October 2026) + uint64_t w64 = 0; + if (pf_define_u64(prog, "IGNEUM_DATASET_WORDS", &w64)) { + if (w64 > 0xffffffffull) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_WORDS above 2^32 - 1: the word index is 32-bit"); } + pk->datasetWords = (uint32_t)w64; + } else pk->datasetWords = 1u << pk->datasetLog2; + } if (pk->datasetWords < (1u << 16) || (pk->datasetWords & 0xffffu) != 0) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_WORDS is not a multiple of 65,536 words"); } + { + // IGNEUM_DATASET_LOG2 is the floor of log2 of the count (the exporter writes it so; a reader that sizes by the log2 alone + // is then at most a factor of two short, never wrong by a different pack's geometry): a count that disagrees is refused + uint32_t fl = 0, w = pk->datasetWords; + while (w > 1u) { w >>= 1; ++fl; } + if (fl != pk->datasetLog2) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_LOG2 is not the floor of log2 of the word count"); } + } if (!pf_define_u32(prog, "IGNEUM_DATASET_ITEMS", &pk->datasetItems)) pk->datasetItems = pk->datasetWords / 16u; if (pk->datasetItems != pk->datasetWords / 16u) { free(prog); return pf_fail(err, cap, "program.h IGNEUM_DATASET_ITEMS is not IGNEUM_DATASET_WORDS / 16"); } if (!pf_define_u32(prog, "IGNEUM_DATASET_MULSHIFT", &pk->datasetMulshift)) pk->datasetMulshift = 0; diff --git a/proto-cuda/nvrtc/serve-bench.sh b/proto-cuda/nvrtc/serve-bench.sh index 62ada452e..63a95354f 100755 --- a/proto-cuda/nvrtc/serve-bench.sh +++ b/proto-cuda/nvrtc/serve-bench.sh @@ -21,9 +21,13 @@ LOG="$OUT/$RUN.worker.log"; TL="$OUT/$RUN.timeline"; SMI="$OUT/$RUN.smi"; ROW="$ now() { date +%s.%N; } count=$((1 << B)) target="00000fffffffffff" # one hit per 2^20 nonces: 16 found lines per 2^24-nonce job, the serving shape (a block target is far tighter) -seedline() { sed -n "s/^$2 //p" "$1/seeds.txt"; } +# the pack's seeds: seeds.txt as igneum-miner export-pack writes it, else the byte seeds program.h carries (a kit pack) +seedline() { sed -n "s/^$2 //p" "$1/seeds.txt" 2>/dev/null || true; } # a missing file is not a failure under set -e +define() { sed -n "s/^#define $2 \"\(.*\)\"/\1/p" "$1/program.h" 2>/dev/null || true; } EPOCH="$(seedline "$PACK" epoch_seed_hex)"; DAY="$(seedline "$PACK" day_seed_hex)" -[ -n "$EPOCH" ] && [ -n "$DAY" ] || { echo "no seeds.txt in $PACK" >&2; exit 2; } +[ -n "$EPOCH" ] || EPOCH="$(define "$PACK" IGNEUM_SEED_BYTES_HEX)" +[ -n "$DAY" ] || DAY="$(define "$PACK" IGNEUM_DAY_BYTES_HEX)" +[ -n "$EPOCH" ] && [ -n "$DAY" ] || { echo "no seeds.txt and no IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in $PACK/program.h" >&2; exit 2; } jobline() { printf 'job %d %064x %s %d %d %s %s\n' "$1" "$1" "$target" $(( $1 * count )) "$count" "$EPOCH" "$DAY"; } # power sampler first (its own pid file) @@ -31,7 +35,7 @@ nvidia-smi -i "$DEV" --query-gpu=timestamp,power.draw,clocks.sm,temperature.gpu echo $! > "$PIDDIR/$RUN.smi.pid" sleep 1 t0=$(now) -coproc WK { IGNEUM_READBACK="$MODE" "$W" --serve --pack "$PACK" --device "$DEV" --batch-log2 "$B" --race off --readback "$MODE" 2>&1; } +coproc WK { exec env IGNEUM_READBACK="$MODE" "$W" --serve --pack "$PACK" --device "$DEV" --batch-log2 "$B" --race off --readback "$MODE" 2>&1; } # exec: the pid is the worker's, not a subshell's wkpid=$WK_PID # bash clears WK_PID when the coproc ends; the copy outlives it echo "$wkpid" > "$PIDDIR/$RUN.worker.pid" sent=0; done_n=0; found=0; tready=""; tlast=""; cpu="" diff --git a/proto-cuda/nvrtc/worker.cpp b/proto-cuda/nvrtc/worker.cpp index f25d15382..41fd9a784 100644 --- a/proto-cuda/nvrtc/worker.cpp +++ b/proto-cuda/nvrtc/worker.cpp @@ -18,8 +18,9 @@ // prepare compile that pack in the background, build its cache // and dataset, self-test it; a job on it then switches // quit -// stdout: ready cuda pack dataset-log2 N batch B regs R prepare 1 path nvrtc ... -// found +// stdout: ready cuda pack dataset-log2 N batch B regs R prepare 1 path nvrtc ... readback select|full ... +// found the fifth field (8 October 2026, I05): the epoch of the pair +// the hash was computed under, so stale work is never read as current // done // error // need before the mismatch error: the pair this worker lacks (the miner prepares it) @@ -1416,12 +1417,12 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) { } for (uint32_t a = 0; a < nHits; ++a) { uint64_t nonce = ((uint64_t)hi << 32) | (uint64_t)(uint32_t)(lo + (uint32_t)sel.hHits[2 * a]); - std::printf("found %s %llu %016llx\n", jobId.c_str(), (unsigned long long)nonce, (unsigned long long)sel.hHits[2 * a + 1]); + std::printf("found %s %llu %016llx %.16s\n", jobId.c_str(), (unsigned long long)nonce, (unsigned long long)sel.hHits[2 * a + 1], cur->epochHex.c_str()); } } else { for (uint32_t i = 0; i < chunk; ++i) if (hOut[i] <= target) { uint64_t nonce = ((uint64_t)hi << 32) | (uint64_t)(uint32_t)(lo + i); - std::printf("found %s %llu %016llx\n", jobId.c_str(), (unsigned long long)nonce, (unsigned long long)hOut[i]); + std::printf("found %s %llu %016llx %.16s\n", jobId.c_str(), (unsigned long long)nonce, (unsigned long long)hOut[i], cur->epochHex.c_str()); } } scanMs = wallMs() - r0; diff --git a/proto-opencl/host.c b/proto-opencl/host.c index eedc8259e..ec371829d 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -1156,6 +1156,13 @@ typedef struct { cl_uint stateLeaves; /* class v5: the state leaves the dataset was built from (uploaded for the build, released after); 0 otherwise */ } ServePair; +/* The found line's fifth field (8 October 2026, I05): the epoch the pair's program was built for, 16 hex, so stale work is never read + * as current; the compiled-in pack's epoch comes from its program.h. */ +#ifndef IGNEUM_SEED_BYTES_HEX +#define IGNEUM_SEED_BYTES_HEX "0000000000000000" +#endif +static const char* pairEpochTag(const ServePair* p) { return p->epochHex[0] ? p->epochHex : IGNEUM_SEED_BYTES_HEX; } + static int hexEq(const char* a, const char* b) { size_t i; if (strlen(a) != strlen(b)) return 0; @@ -1929,12 +1936,12 @@ static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { for (a = 0; a < nHits; ++a) { uint32_t idx = (uint32_t)hHits[2 * a]; unsigned long long nonce = ((unsigned long long)hi << 32) | (unsigned long long)(uint32_t)(lo + idx); - printf("found %s %llu %016llx\n", jobId, nonce, (unsigned long long)hHits[2 * a + 1]); + printf("found %s %llu %016llx %.16s\n", jobId, nonce, (unsigned long long)hHits[2 * a + 1], pairEpochTag(cur)); } } else { for (i = 0; i < chunk; ++i) if (hOut[i] <= target) { unsigned long long nonce = ((unsigned long long)hi << 32) | (unsigned long long)(uint32_t)(lo + i); - printf("found %s %llu %016llx\n", jobId, nonce, (unsigned long long)hOut[i]); + printf("found %s %llu %016llx %.16s\n", jobId, nonce, (unsigned long long)hOut[i], pairEpochTag(cur)); } } scanMs = wallMs() - r0; diff --git a/tools/ci/batches/f10-worker-20261008.json b/tools/ci/batches/f10-worker-20261008.json new file mode 100644 index 000000000..8dc2fad66 --- /dev/null +++ b/tools/ci/batches/f10-worker-20261008.json @@ -0,0 +1,17 @@ +{ + "run_id": "f10-worker-20261008", + "manifest_sha": "964cdf6c1", + "evidence_dir": "docs/analysis/review-2026-10-08-b/f10", + "cells": [ + { + "cell": "harness:cuda-select-check", + "status": "RUNNING", + "evidence": "docs/analysis/review-2026-10-08-b/f10/emu-test.log" + }, + { + "cell": "harness:pack-geometry", + "status": "RUNNING", + "evidence": "docs/analysis/review-2026-10-08-b/f10/packfile-test.log;docs/analysis/review-2026-10-08-b/f10/geometry-check-mini.log" + } + ] +} diff --git a/tools/ci/test-map.json b/tools/ci/test-map.json index 74ea4e1d2..78ba081e7 100644 --- a/tools/ci/test-map.json +++ b/tools/ci/test-map.json @@ -459,6 +459,35 @@ "UX-05": "partial: the vote-key commitment on the wire (verify::tests::a_share_under_another_key_is_refused_before_the_hash: known-pass the member's key, known-fail another key on the share and a job naming another key, refused with code vote_key before the hash), the open pool's sidechain::check_key (the claimed key, the header's key and the reveal are one key), the same-nonce-other-template wrong_hash test, the TLS binding test; the malicious-pool run and the leave-and-retain run on the devnet-4 pair are the UX-05 batch's other evidence", "UX-04": "partial: PPLNS distribution tests (fee first, never overpays, late joiner), the payout key file round trip, the signed transfer decode; the live payout path on the devnet-4 pair is the UX-04 batch's evidence" } + }, + "harness:cuda-select-check": { + "command": "proto-cuda/nvrtc/emu/test.sh on a box (the CUDA worker under the CPU emulation backend): emu/select-check.sh (--readback select against the full read, known-failed first) and emu/serve-check.sh (the hash sample against igneum-pow hash-bound, both sides of the 32-bit boundary, the swap)", + "box_class": "suite (CPU emulation, any build box)", + "fixtures": [ + "F0", + "F1" + ], + "cases": [ + "ROT-01", + "POW-01" + ], + "coverage": { + "ROT-01": "partial: the worker's high-32 nonce rollover inside a job (a job from 2^32 - 32) finds the same set on both read-back paths and the boundary nonces hash as igneum-pow; the hourly program boundary itself is the fast-time harness", + "POW-01": "partial: the CUDA select pass equals the full read line for line (zero hits, partial, the all-hit overflow fallback, a 96-nonce tail with a 32-lane launch, queued jobs, hits then zero, the epoch tag) and 17 sampled hashes equal igneum-pow; the million-vector campaign is harness:p01-vectors" + } + }, + "harness:pack-geometry": { + "command": "proto-cuda/nvrtc/emu/packfile-test.sh on a box (packfile.h pf_load over emu/geometry-vectors.txt and the checked-in packs) and igneum-bench --geometry-check proto-cuda/nvrtc/emu/geometry-vectors.txt on the Mac mini (proto-metal/main.swift packGeometry over the same file)", + "box_class": "suite (a build box for C; the mini for Swift)", + "fixtures": [ + "F0" + ], + "cases": [ + "POW-02" + ], + "coverage": { + "POW-02": "partial: the dataset geometry rule (exact word count, items, multiply-shift, 64-bit bytes, the floor-log2 and 2^32 refusals, missing leaves) agrees across the CUDA and OpenCL loader and the Metal host on 19 conformance vectors and the two checked-in packs (master review I06, gate INT-10); the rest of POW-02 is suite:pow" + } } }, "not_run": {