diff --git a/docs/plans/read-width.md b/docs/plans/read-width.md index 60a4d3da8..b9c04b2e5 100644 --- a/docs/plans/read-width.md +++ b/docs/plans/read-width.md @@ -51,6 +51,23 @@ Probe ceilings at 1024 MiB (G dependent reads/s): RTX 5090 4 B 17.5, 16 B 18.0, | mix50-35-15 (6 programs, min / median / max) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0, spread 18.8% | 17.45 / 18.76 / 18.83, 7.4% | 25.36 / 27.26 / 28.43, 11.3% | 6.1x | 5,939 / 8,192 expected | 0.620 | | mix25-50-25 (6 programs) | 2,240 to 5,024 | 95.9 / 107.3 / 119.8, 22.3% | 17.84 / 18.45 / 18.85, 5.5% | 23.21 / 24.68 / 25.21, 8.1% | 5.8x | 7,168 / 8,192 expected | 0.614 | +### 4.1 Per watt and per pound (consequences review C11) + +The runs carried no power sampling; the watts are the telemetry entry's (`docs/bench-log.md`, opencl-rdna4-telemetry, 5 October 2026: the RTX 5090 at 307.6 W under its 450 W cap for 122.3 MH/s, the RX 9070 XT at 199 W of its 304 W rating for about 17.8 MH/s, both on v2 with the shader clock at its top and the die waiting on memory), held constant across classes because every class is memory-bound on both cards (approximate: a class that moves more bytes per hash draws somewhat more at the memory controller, unmeasured). The Mac's GPU power is not measurable without root (`powermetrics`) and is taken as about 50 W (approximate, from memory). Prices are UK list, approximate, from memory. + +| Class | RTX 5090 MH/W (at 307.6 W) | RX 9070 XT MH/W (at 199 W) | 5090 / 9070 per watt | M5 Max MH/W (at about 50 W GPU, approximate) | 5090 MH per pound (at about 1,900, approximate) | 9070 XT MH per pound (at about 570, approximate) | +|---|---|---|---|---|---|---| +| v2 (w4) | 0.442 | 0.091 | 4.9x | 0.55 | 0.072 | 0.032 | +| w16 | 0.454 | 0.090 | 5.1x | 0.57 | 0.074 | 0.031 | +| w64 | 0.234 | 0.088 | 2.6x | 0.57 | 0.038 | 0.031 | +| w64x4 | 0.895 | 0.378 | 2.4x | 2.19 | 0.145 | 0.132 | +| mix50-35-15 (median) | 0.371 | 0.094 | 3.9x | 0.55 | 0.060 | 0.033 | +| mix25-50-25 (median) | 0.349 | 0.093 | 3.8x | 0.49 | 0.056 | 0.032 | +| scr8k32 | 0.397 | 0.071 | 5.6x | 0.98 | 0.064 | 0.025 | +| scr2k32 | 0.372 | 0.074 | 5.1x | 0.52 | 0.060 | 0.026 | + +Reading: whatever width is chosen, an AMD home miner keeps about a seventh of a 5090's rate and pays about 4.5x the electricity per hash, because every width costs the 9070 XT the same 2.4 G line fetches a second; per pound of card the 5090 is 2.2x the 9070 XT at v2 and w16 (0.072 against 0.032 MH/s per pound) and 4.9x per watt; only w64x4 narrows the per-pound gap (0.145 against 0.132), and that class fails the width rule. The consequence for the decision (D6, the project lead's): AMD's line width is not a read-width question at all; it is the card's random-access rate, and the levers that act on it (the 64 MB Infinity Cache against the dataset size, the memory path) are v3-or-3.0 questions outside this experiment. + Scratch, variant 5 (N persistent warps; GPU cost against the persistent control scr0k32; working set = 1 GiB + 256 MiB + 128 MiB output + N x size): | Class | RMW share | dataset B/hash | scratch B/hash read + written | RTX 5090 MH/s, 2,048 warps of 4,080 resident (vs control, share) | M5 Max Metal (vs control) | RX 9070 XT, 4,096 warps | working set 5090 / 9070 / Mac | diff --git a/proto-metal/packbench.swift b/proto-metal/packbench.swift index 3110bbe12..c708f5c28 100644 --- a/proto-metal/packbench.swift +++ b/proto-metal/packbench.swift @@ -204,6 +204,9 @@ print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)") print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)") print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall") +let footprintMiB = device.currentAllocatedSize / 1048576 +let recommendedMiB = device.recommendedMaxWorkingSetSize / 1048576 +print("resident footprint after the timed batches: currentAllocatedSize \(footprintMiB) MiB (recommendedMaxWorkingSetSize \(recommendedMiB) MiB)") let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN -print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") +print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN * 32 * scratchWordsPerLane * 4 / 1048576 : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) footprint_mib=\(footprintMiB) recommended_mib=\(recommendedMiB) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") exit(overall ? 0 : 1)