From f1ddf0b39700fc15b85d2ca7645c0e9082b82775 Mon Sep 17 00:00:00 2001 From: igneum-josh Date: Thu, 8 Oct 2026 15:33:29 +0100 Subject: [PATCH] shadow-k: the adversary's 64-register core (the live-state analysis: 95 percent of the window necessary across every wait under the fold-reads-all rule; the state's cheaper forms modelled; the gated rows pending), the gated variants' configs, livestate.py Co-Authored-By: Claude Fable 5.1 --- docs/analysis/class-v6/floor/shadow-k.md | 50 +++++++++++++++ tools/chip-model/rtl/Makefile | 4 +- tools/chip-model/rtl/flow/collect.py | 4 +- tools/chip-model/rtl/flow/core8g.mk | 18 ++++++ tools/chip-model/rtl/flow/core8g.sdc | 10 +++ tools/chip-model/rtl/flow/core8r64g.mk | 18 ++++++ tools/chip-model/rtl/flow/core8r64g.sdc | 10 +++ tools/chip-model/rtl/flow/designs.txt | 2 + tools/chip-model/rtl/flow/livestate.py | 79 ++++++++++++++++++++++++ 9 files changed, 192 insertions(+), 3 deletions(-) create mode 100644 tools/chip-model/rtl/flow/core8g.mk create mode 100644 tools/chip-model/rtl/flow/core8g.sdc create mode 100644 tools/chip-model/rtl/flow/core8r64g.mk create mode 100644 tools/chip-model/rtl/flow/core8r64g.sdc create mode 100644 tools/chip-model/rtl/flow/livestate.py diff --git a/docs/analysis/class-v6/floor/shadow-k.md b/docs/analysis/class-v6/floor/shadow-k.md index a5c561c52..dbbeb4478 100644 --- a/docs/analysis/class-v6/floor/shadow-k.md +++ b/docs/analysis/class-v6/floor/shadow-k.md @@ -203,6 +203,56 @@ takes back on the card's own node (3.6x to 2.4x), which is the design. Node-for- k 0.78 and the 64-register core at 1.09, so "near 0.9" is reached node-for-node by the window alone; what it does not survive is the node step a chip project would buy (an N3 core gives back 0.4x, an N2 core 0.8x). +### 4c. The adversary's 64-register core: is the window a defence? (the coordinator's order, 15:3x UK) + +Every row in this section is a MODEL of a chip core, never a lower bound on what a chip maker can build; the +synthesis gives the cost of the circuit as drawn, and a better circuit is always possible. + +**The live-state analysis** (`tools/chip-model/rtl/flow/livestate.py`, run on build-3: programs drawn as the +core testbench draws them, the class v4 op weights, a load on one instruction in 16 as the dependent memory wait, +dst and src uniform over the window, the result fold reading every register at the end of the block; 64 drawn +programs, 1,024 waits per row): + +| Window R | Live values at a wait (mean, min to max) | Of which necessary (reach a later address or the result, transitively) | Dead writes per block | +|---|---|---|---| +| 8 (the class ISA) | 7.0 of 8 (6 to 7) | 6.9 | 2.9 percent | +| 32 | 30.5 of 32 (29 to 31) | 30.0 | 2.9 percent | +| 64 | 61.7 of 64 (59 to 63) | 61.0 | 2.8 percent | +| 64 at a 1,024-instruction block | 61.5 of 64 (58 to 63) | 60.6 | 3.1 percent | + +So under a fold that reads every register, 95 percent of the window is live AND necessary across every memory +wait: the adversary cannot shrink the state it keeps by liveness, and recomputing a value instead of keeping it +costs the dependent chain that produced it (every value feeds the result transitively). The window is a +defence ONLY because of the fold rule; a fold that read 8 of the 64 registers would let the chip drop the rest +(the dead fraction would rise toward the fraction never read before the fold), so the fold-reads-all rule is the +design rule that goes with the window. + +**What the adversary can do with the state it must keep** is make it cheaper per access, not smaller. The +GPU-shaped row (4a, 64 registers in flops, every flop clocked every cycle, three 64:1 read muxes) is 9.7 pJ per +lane-op at ASAP7. The forms a chip maker would use: + +| Form of the 64-register state (per lane, 256 bytes) | pJ per lane-op ASAP7 | N3 | k at the lock (N3) | Label | +|---|---|---|---|---| +| flops, no clock gating, 64:1 read muxes (the 4a row) | 9.7 | 4.9 | 0.78 | synthesised; a model | +| flops with the register-file clock gated (one of 64 registers written per cycle; the ICG cells allowed back in and inferred by Yosys) | ROW_CORE8R64G | | | synthesised; a model | +| the same gating on the 32-register base, for the penalty | ROW_CORE8G | | | synthesised; a model | +| latch-based register file (the clocked element halved; about 30 percent under the gated flop file, approximate) | about 0.7 x the gated row | | | modelled | +| SRAM-banked state shared across time-multiplexed lanes (one execution port serving many lanes' streams in turn, each lane's 256 bytes in a bank of a few KB): three 32-bit operand reads and one write per lane-op from a small macro at about 1.0 to 1.5 pJ per 32-bit access at 7 nm (approximate: the JSSC 2026 3 nm macro reads at 3.9 pJ per access at 563 kbit; a few-KB bank is a third of that; claimed), plus the units and the read network (the base core's 4.5 pJ combinational term) | about 8.5 to 10.5 | 4.3 to 5.3 | 0.69 to 0.85 | modelled: NOT cheaper than the gated flop file; the GPU's own register file is SRAM-banked because it is 256 KB per SM, and at 256 bytes per lane flops with gating win | +| values recomputed instead of kept | not available: 95 percent of the window is necessary (above) | | | measured on drawn programs | + +The defence, then, is the penalty that remains after the adversary's best form: the gated 64-register file +against the gated 32-register file (the two synthesised rows above when they land; the analytic estimate from +the ungated rows is 9.7 - 2.5 = 7.2 against 6.9 - 1.2 = 5.7 pJ per lane-op at ASAP7, a penalty of about 1.5 pJ, +0.75 at N3, +0.12 of k at the lock, approximate until the gated rows replace it). On the 32-lane core the same +penalty applies per lane (the register file does not amortise), so the window moves the 32-lane core from k 0.45 +to about 0.57 at the lock at N3 (0.63 to about 0.80 node-for-node). + +**The GPU side** (the hash lane's hand): the compiled allocation of the 64-register measurement pack (ptxas +registers per thread, local-memory spill bytes, occupancy) and the rate beside the 8-register base, clock +18:00 UK; until then the modelled reading stands: about 110 of 255 registers per thread, occupancy about half, +the rate expected to hold under the latency-bound chain (the 5090 hides about 330,000 ops per hash before compute +binds) and the energy to move little, the per-lane register traffic the unmeasured term. ROW_GPU_REG64 + ## 5. The chip edge at the measured k `E_chip = E_mem + N_ops x e_chip` (absolute: the chip's shadow cost is 102,100 x 3.5 pJ = 0.36 microjoules per hash diff --git a/tools/chip-model/rtl/Makefile b/tools/chip-model/rtl/Makefile index dea28cc08..2544642d0 100644 --- a/tools/chip-model/rtl/Makefile +++ b/tools/chip-model/rtl/Makefile @@ -24,7 +24,7 @@ DOCKER := $(LEASEPFX) docker run --rm -u $(UID_GID) -e HOME=/tmp -e NUM_CORES= ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG) SIM := $(DOCKER) -w /work $(SIM_IMG) -DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all +DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt) # per-family simulation tags (the op field fixed per row where the family has several ops) @@ -44,6 +44,8 @@ SIMS_core8r64 := mix SIMS_core8i1k := mix SIMS_core8sel := mix SIMS_core32all := mix +SIMS_core8g := mix +SIMS_core8r64g := mix .PHONY: rows table clean diff --git a/tools/chip-model/rtl/flow/collect.py b/tools/chip-model/rtl/flow/collect.py index 842842019..cc9a1adb7 100644 --- a/tools/chip-model/rtl/flow/collect.py +++ b/tools/chip-model/rtl/flow/collect.py @@ -6,7 +6,7 @@ import re, sys, os, csv work = sys.argv[1] if len(sys.argv) > 1 else '.' # ops per cycle per design (the per-op divisor) and the GPU row each family is read against -OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32} +OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8} # 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a GPU = { 'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2), @@ -17,7 +17,7 @@ GPU = { 'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4), 'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed) 'tile:mix': (4.1, 2.2), - 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 + 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 } # per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed: # N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent), diff --git a/tools/chip-model/rtl/flow/core8g.mk b/tools/chip-model/rtl/flow/core8g.mk new file mode 100644 index 000000000..c2601b8b8 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8g.mk @@ -0,0 +1,18 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8 +export DESIGN_NICKNAME = core8g +export VERILOG_FILES = /work/rtl/core_v6_8.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core8g.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core8g +export SYNTH_MEMORY_MAX_BITS = 2000000 +# the adversary's register file: clock gating inferred (the ICG cells allowed back in) +export INFER_CLKGATES = 1 +export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF* diff --git a/tools/chip-model/rtl/flow/core8g.sdc b/tools/chip-model/rtl/flow/core8g.sdc new file mode 100644 index 000000000..10f3be2bd --- /dev/null +++ b/tools/chip-model/rtl/flow/core8g.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8 +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/core8r64g.mk b/tools/chip-model/rtl/flow/core8r64g.mk new file mode 100644 index 000000000..e43857538 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8r64g.mk @@ -0,0 +1,18 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8r64), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8r64 +export DESIGN_NICKNAME = core8r64g +export VERILOG_FILES = /work/rtl/core_v6_8r64.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core8r64g.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core8r64g +export SYNTH_MEMORY_MAX_BITS = 2000000 +# the adversary's register file: clock gating inferred (the ICG cells allowed back in) +export INFER_CLKGATES = 1 +export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF* diff --git a/tools/chip-model/rtl/flow/core8r64g.sdc b/tools/chip-model/rtl/flow/core8r64g.sdc new file mode 100644 index 000000000..ffb92a9cd --- /dev/null +++ b/tools/chip-model/rtl/flow/core8r64g.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8r64 +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/designs.txt b/tools/chip-model/rtl/flow/designs.txt index af4f9566e..6fbd648cf 100644 --- a/tools/chip-model/rtl/flow/designs.txt +++ b/tools/chip-model/rtl/flow/designs.txt @@ -14,3 +14,5 @@ core8r64 core_v6_8r64 1500 core8i1k core_v6_8i1k 1500 core8sel core_v6_8sel 1500 core32all core_v6_32all 1500 +core8g core_v6_8 1500 +core8r64g core_v6_8r64 1500 diff --git a/tools/chip-model/rtl/flow/livestate.py b/tools/chip-model/rtl/flow/livestate.py new file mode 100644 index 000000000..f729ddd05 --- /dev/null +++ b/tools/chip-model/rtl/flow/livestate.py @@ -0,0 +1,79 @@ +#!/usr/bin/env python3 +"""Live-state analysis of a drawn shadow program on an R-register window (the coordinator's order, 15:3x UK). +The program is drawn as the core testbench draws it: NPROG instructions with the class v4 op weights, a load on +one instruction in 16 (the dependent memory wait), dst/src/src2 uniform over the R registers, and the result +fold reading every register at the end of the block. For every load (wait) the script reports the live set: +registers whose current value is read later (by an instruction, a later address, or the fold) before being +overwritten, split into those that feed a later ADDRESS or the RESULT and those that die inside an arithmetic +block. Dead writes (overwritten before any read) are counted too. Usage: livestate.py R [NPROG] [seeds]""" +import random, sys +R = int(sys.argv[1]) if len(sys.argv) > 1 else 64 +N = int(sys.argv[2]) if len(sys.argv) > 2 else 256 +SEEDS = int(sys.argv[3]) if len(sys.argv) > 3 else 64 +W = [('add',12),('xor',10),('mul',8),('mad',8),('shfl',8),('rotl',7),('sub',6),('mulhi',6),('rotr',6),('or',4)] +ops = [o for o,w in W for _ in range(w)] +def draw(rng): + prog = [] + for k in range(N): + op = 'load' if k % 16 == 15 else rng.choice(ops) + d, s, s2 = rng.randrange(R), rng.randrange(R), rng.randrange(R) + reads = [s] if op not in ('load',) else [s] # the load's address comes from src + if op in ('add','xor','mul','mad','sub','or','rotr','shfl','mulhi','rotl'): reads.append(d) # dst is read too (r[d] op= ...) + if op == 'mad': reads.append(s2) + if op == 'rotl': reads = [d] + prog.append((op, d, reads)) + return prog +tot_live = tot_addr = tot_dead = tot_waits = 0 +live_min, live_max = R, 0 +for seed in range(SEEDS): + rng = random.Random(seed) + prog = draw(rng) + # a value's "version" = (reg, write index); the fold at the end reads every register + # forward pass: for each instruction i and register r, next read of r's current value before its next write + n = len(prog) + # necessity: a version is NECESSARY if it reaches an address (a load's src) or the fold, transitively + # compute transitively by backward dataflow over versions + writes_at = {} # (i) -> reg written + # build version ids: version of reg r valid after instruction i + cur = {r: ('init', r) for r in range(R)} + uses = {} # version -> list of (consumer index, consumer version or 'addr'/'fold') + versions = set(cur.values()) + deps = {} # version -> set of versions it reads + for i, (op, d, reads) in enumerate(prog): + srcs = [cur[r] for r in reads] + if op == 'load': + v = ('load', i); deps[v] = set() # the returned word: its ADDRESS depends on srcs + for s in srcs: uses.setdefault(s, []).append(('addr', i)) + else: + v = (op, i); deps[v] = set(srcs) + for s in srcs: uses.setdefault(s, []).append(('op', i)) + cur[d] = v; versions.add(v) + fold = set(cur.values()) + # necessary = reaches an address or the fold + necessary = set(fold) + for v, us in uses.items(): + if any(k == 'addr' for k, _ in us): necessary.add(v) + changed = True + while changed: + changed = False + for v in list(versions): + if v in necessary: + for s in deps.get(v, ()): + if s not in necessary: necessary.add(s); changed = True + # per wait: the versions live at the load (written before it, read after it) + last_read = {} + for v, us in uses.items(): + last_read[v] = max(i for _, i in us) + for v in fold: last_read[v] = n + written_at = {v: (v[1] if v[0] != 'init' else -1) for v in versions} + for i, (op, d, reads) in enumerate(prog): + if op != 'load': continue + live = [v for v in versions if written_at[v] < i and last_read.get(v, -1) > i] + nec = [v for v in live if v in necessary] + tot_live += len(live); tot_addr += len(nec); tot_waits += 1 + live_min = min(live_min, len(live)); live_max = max(live_max, len(live)) + dead = sum(1 for v in versions if v[0] not in ('init',) and v not in uses and v not in fold) + tot_dead += dead +print(f'R = {R}, NPROG = {N}, {SEEDS} drawn programs, {tot_waits} waits') +print(f'live values at a wait: mean {tot_live/tot_waits:.1f} of {R} (min {live_min}, max {live_max}); of which necessary (reach a later address or the result): {tot_addr/tot_waits:.1f}') +print(f'dead writes (overwritten before any read): {tot_dead/SEEDS:.1f} per {N}-instruction block ({100*tot_dead/SEEDS/N:.1f} percent)')