From 0dc59c302e8cbf5fbde92d38eebcf591c0e35c23 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 14:47:21 +0000 Subject: [PATCH] shadow-k: the window's measured GPU side (no spill, 83 and 67 percent occupancy, at most 5 percent per load); the time-multiplexed core, the connected-state draw, the simplified FP32 units and the pod mode Co-Authored-By: Claude Fable 5.1 Documents-only replay of 26844fde7 (86e5b0fb248a565d23560406a79961746a5f66bf) for the box mirror master --- docs/analysis/class-v6/floor/shadow-k.md | 20 +++- tools/chip-model/rtl/Makefile | 11 +- tools/chip-model/rtl/flow/collect.py | 8 +- tools/chip-model/rtl/flow/coretm.mk | 18 +++ tools/chip-model/rtl/flow/coretm.sdc | 10 ++ tools/chip-model/rtl/flow/cs64.mk | 18 +++ tools/chip-model/rtl/flow/cs64.sdc | 10 ++ tools/chip-model/rtl/flow/designs.txt | 3 + tools/chip-model/rtl/flow/fp32.mk | 15 +++ tools/chip-model/rtl/flow/fp32.sdc | 10 ++ tools/chip-model/rtl/flow/pod.sh | 9 ++ tools/chip-model/rtl/rtl/core_tm.v | 82 ++++++++++++++ tools/chip-model/rtl/rtl/core_tm_8r64.v | 7 ++ tools/chip-model/rtl/rtl/core_v6_8cs64.v | 7 ++ tools/chip-model/rtl/rtl/fp32_units.v | 123 +++++++++++++++++++++ tools/chip-model/rtl/tb/tb_core_common.vh | 22 ++++ tools/chip-model/rtl/tb/tb_core_tm_8r64.v | 6 + tools/chip-model/rtl/tb/tb_core_v6_8cs64.v | 7 ++ tools/chip-model/rtl/tb/tb_lane_fp32.v | 22 ++++ 19 files changed, 402 insertions(+), 6 deletions(-) create mode 100644 tools/chip-model/rtl/flow/coretm.mk create mode 100644 tools/chip-model/rtl/flow/coretm.sdc create mode 100644 tools/chip-model/rtl/flow/cs64.mk create mode 100644 tools/chip-model/rtl/flow/cs64.sdc create mode 100644 tools/chip-model/rtl/flow/fp32.mk create mode 100644 tools/chip-model/rtl/flow/fp32.sdc create mode 100755 tools/chip-model/rtl/flow/pod.sh create mode 100644 tools/chip-model/rtl/rtl/core_tm.v create mode 100644 tools/chip-model/rtl/rtl/core_tm_8r64.v create mode 100644 tools/chip-model/rtl/rtl/core_v6_8cs64.v create mode 100644 tools/chip-model/rtl/rtl/fp32_units.v create mode 100644 tools/chip-model/rtl/tb/tb_core_tm_8r64.v create mode 100644 tools/chip-model/rtl/tb/tb_core_v6_8cs64.v create mode 100644 tools/chip-model/rtl/tb/tb_lane_fp32.v diff --git a/docs/analysis/class-v6/floor/shadow-k.md b/docs/analysis/class-v6/floor/shadow-k.md index dbbeb4478..6241a51e8 100644 --- a/docs/analysis/class-v6/floor/shadow-k.md +++ b/docs/analysis/class-v6/floor/shadow-k.md @@ -251,7 +251,25 @@ to about 0.57 at the lock at N3 (0.63 to about 0.80 node-for-node). registers per thread, local-memory spill bytes, occupancy) and the rate beside the 8-register base, clock 18:00 UK; until then the modelled reading stands: about 110 of 255 registers per thread, occupancy about half, the rate expected to hold under the latency-bound chain (the 5090 hides about 330,000 ops per hash before compute -binds) and the energy to move little, the per-lane register traffic the unmeasured term. ROW_GPU_REG64 +binds) and the energy to move little, the per-lane register traffic the unmeasured term. + +The measured GPU side (the hash lane, 16:1x to 16:4x UK, RunPod secure pods, driver 580, the kit worker, 250 x +2^24, nvidia-smi 1 Hz; ptxas from nvcc 12.8 -Xptxas -v on the pack's kernel; pods destroyed, USD 1.22): + +| Card, pack | MH/s | W | microjoules per hash | registers per thread (ptxas) | spill | blocks per SM (occupancy) | per load | Label | +|---|---|---|---|---|---|---|---|---| +| 5090, the base (mx8-devnet-epoch0) | 141.74 | 303.1 | 2.139 | 30 | 0 B | 24 (4,080 warps) | 16.7 nJ | measured | +| 5090, the window, arithmetic-only (hl-reg64, twice the base's work by construction) | 80.38 | 308.6 | 3.839 | 96 | 0 B | 20 (83 percent) | 15.0 nJ | measured: level per unit of work | +| 5090, the window, full chain (hl-reg64c: every load's address mixes all 64 registers) | 70.96 | 320.3 | 4.513 | 88 | 0 B | 20 (83 percent) | 17.6 nJ (+5 percent) | measured | +| 4090, the base | 62.67 | 208.9 | 3.333 | 29 | 0 B | 24 | 26.0 nJ | measured | +| 4090, the window, arithmetic-only | 31.57 | 210.3 | 6.663 | 104 | 0 B | 16 (67 percent) | 26.0 nJ | measured: level | +| 4090, the window, full chain | 31.38 | 216.5 | 6.898 | 87 | 0 B | 20 (83 percent) | 27.0 nJ (+4 percent) | measured | + +So the card's side of the window defence is at most 5 percent per load: no spill on either card in either form, 88 +to 104 registers per thread, occupancy 67 to 83 percent, and the rate per unit of work held within 5 percent under +the latency-bound chain. The sound class form is the full chain (the arithmetic-only fold fails the liveness rule; +class string `+reg64c`, pack hl-v6-win with `check_window_liveness` in its suite). The chip's side (this section's +gated rows) therefore carries the whole defence. ## 5. The chip edge at the measured k diff --git a/tools/chip-model/rtl/Makefile b/tools/chip-model/rtl/Makefile index 2544642d0..bd774375a 100644 --- a/tools/chip-model/rtl/Makefile +++ b/tools/chip-model/rtl/Makefile @@ -23,8 +23,14 @@ CPUSET = $(if $(LEASE_ON),--cpuset-cpus {cpuset},) DOCKER := $(LEASEPFX) docker run --rm -u $(UID_GID) -e HOME=/tmp -e NUM_CORES=$(THREADS) $(CPUSET) -v $(WORK):/work ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG) SIM := $(DOCKER) -w /work $(SIM_IMG) +# NO_DOCKER=1: the box IS the ORFS image (a rented pod started from openroad/orfs:latest with iverilog installed +# and /work a symlink to this directory); the same targets run natively. +ifeq ($(NO_DOCKER),1) +ORFS := env NUM_CORES=$(THREADS) bash -c 'cd /OpenROAD-flow-scripts/flow && exec "$$@"' -- +SIM := env +endif -DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g +DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g coretm cs64 fp32 top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt) # per-family simulation tags (the op field fixed per row where the family has several ops) @@ -46,6 +52,9 @@ SIMS_core8sel := mix SIMS_core32all := mix SIMS_core8g := mix SIMS_core8r64g := mix +SIMS_coretm := mix +SIMS_cs64 := mix +SIMS_fp32 := mix fadd:+op=0 fmul:+op=1 ffma:+op=2 fcvt:+op=3 .PHONY: rows table clean diff --git a/tools/chip-model/rtl/flow/collect.py b/tools/chip-model/rtl/flow/collect.py index cc9a1adb7..79e01aff7 100644 --- a/tools/chip-model/rtl/flow/collect.py +++ b/tools/chip-model/rtl/flow/collect.py @@ -6,7 +6,7 @@ import re, sys, os, csv work = sys.argv[1] if len(sys.argv) > 1 else '.' # ops per cycle per design (the per-op divisor) and the GPU row each family is read against -OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8} +OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8, 'coretm': 1, 'cs64': 8, 'fp32': 1} # 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a GPU = { 'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2), @@ -17,7 +17,7 @@ GPU = { 'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4), 'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed) 'tile:mix': (4.1, 2.2), - 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 + 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), 'coretm:mix': (11.3, 6.2), 'cs64:mix': (11.3, 6.2), 'fp32:mix': (9.2, 5.2), 'fp32:fadd': (9.2, 5.2), 'fp32:fmul': (9.2, 5.2), 'fp32:ffma': (9.2, 5.2), 'fp32:fcvt': (9.2, 5.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 } # per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed: # N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent), @@ -67,10 +67,10 @@ for d in OPS: pairs = [('vcd:s150', 'vcd:s600', 150, 600), ('vcd:short', 'vcd:synth', 150, 800), ('vcd:s150', 'vcd:s400', 150, 400)] for sh, lg, cs, cl in pairs: if sh in rows and lg in rows: - rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else LOAD_CYCLES) + rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else (452 if d == 'cs64' else LOAD_CYCLES)) for tag, r in rows.items(): sub = tag.split(':')[1] if ':' in tag else 'prop' - key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld') else f'{d}:mix' + key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld','fadd','fmul','ffma','fcvt') else f'{d}:mix' gpu = GPU.get(key, (None, None)) pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ pj_dyn = (r['internal'] + r['switching']) * period / OPS[d] diff --git a/tools/chip-model/rtl/flow/coretm.mk b/tools/chip-model/rtl/flow/coretm.mk new file mode 100644 index 000000000..ee4daff66 --- /dev/null +++ b/tools/chip-model/rtl/flow/coretm.mk @@ -0,0 +1,18 @@ +# ORFS design config for the programmable shadow core (core8: core_tm_8r64), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_tm_8r64 +export DESIGN_NICKNAME = coretm +export VERILOG_FILES = /work/rtl/core_tm_8r64.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/coretm.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/coretm +export SYNTH_MEMORY_MAX_BITS = 2000000 +# the adversary's register file: clock gating inferred (the ICG cells allowed back in) +export INFER_CLKGATES = 1 +export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF* diff --git a/tools/chip-model/rtl/flow/coretm.sdc b/tools/chip-model/rtl/flow/coretm.sdc new file mode 100644 index 000000000..904853df5 --- /dev/null +++ b/tools/chip-model/rtl/flow/coretm.sdc @@ -0,0 +1,10 @@ +current_design core_tm_8r64 +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/cs64.mk b/tools/chip-model/rtl/flow/cs64.mk new file mode 100644 index 000000000..66f9c9f51 --- /dev/null +++ b/tools/chip-model/rtl/flow/cs64.mk @@ -0,0 +1,18 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8cs64), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8cs64 +export DESIGN_NICKNAME = cs64 +export VERILOG_FILES = /work/rtl/core_v6_8cs64.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/cs64.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/cs64 +export SYNTH_MEMORY_MAX_BITS = 2000000 +# the adversary's register file: clock gating inferred (the ICG cells allowed back in) +export INFER_CLKGATES = 1 +export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF* diff --git a/tools/chip-model/rtl/flow/cs64.sdc b/tools/chip-model/rtl/flow/cs64.sdc new file mode 100644 index 000000000..8ab79b15b --- /dev/null +++ b/tools/chip-model/rtl/flow/cs64.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8cs64 +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/designs.txt b/tools/chip-model/rtl/flow/designs.txt index 6fbd648cf..292d1983d 100644 --- a/tools/chip-model/rtl/flow/designs.txt +++ b/tools/chip-model/rtl/flow/designs.txt @@ -16,3 +16,6 @@ core8sel core_v6_8sel 1500 core32all core_v6_32all 1500 core8g core_v6_8 1500 core8r64g core_v6_8r64 1500 +coretm core_tm_8r64 1500 +cs64 core_v6_8cs64 1500 +fp32 lane_fp32 2000 diff --git a/tools/chip-model/rtl/flow/fp32.mk b/tools/chip-model/rtl/flow/fp32.mk new file mode 100644 index 000000000..cb508f48a --- /dev/null +++ b/tools/chip-model/rtl/flow/fp32.mk @@ -0,0 +1,15 @@ +# ORFS design config for the mul shadow-core family (top lane_fp32), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = lane_fp32 +export DESIGN_NICKNAME = fp32 +export VERILOG_FILES = /work/rtl/fp32_units.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/fp32.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/fp32 + diff --git a/tools/chip-model/rtl/flow/fp32.sdc b/tools/chip-model/rtl/flow/fp32.sdc new file mode 100644 index 000000000..b729bbae1 --- /dev/null +++ b/tools/chip-model/rtl/flow/fp32.sdc @@ -0,0 +1,10 @@ +current_design lane_fp32 +set clk_name core_clock +set clk_port_name clk +set clk_period 2000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/pod.sh b/tools/chip-model/rtl/flow/pod.sh new file mode 100755 index 000000000..8344fc4ab --- /dev/null +++ b/tools/chip-model/rtl/flow/pod.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +# pod.sh: bootstrap a rented pod started from openroad/orfs:latest (RunPod, root): iverilog, /work -> this dir. +set -euo pipefail +cd "$(dirname "$0")/.." +export DEBIAN_FRONTEND=noninteractive +command -v iverilog >/dev/null || { apt-get update -qq >/dev/null 2>&1; apt-get install -y -qq iverilog rsync python3 >/dev/null 2>&1; } +[ -e /work ] || ln -s "$(pwd)" /work +export PATH=/OpenROAD-flow-scripts/tools/install/OpenROAD/bin:/OpenROAD-flow-scripts/tools/install/yosys/bin:$PATH +echo "pod ready: $(nproc) cores, $(free -g | awk '/Mem/{print $2}') GB, yosys $(yosys -V | cut -d' ' -f2), $(which iverilog)" diff --git a/tools/chip-model/rtl/rtl/core_tm.v b/tools/chip-model/rtl/rtl/core_tm.v new file mode 100644 index 000000000..3e702511b --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_tm.v @@ -0,0 +1,82 @@ +// The adversary's time-multiplexed core: ONE execution port (every class unit, once) serving LANES lanes' instruction +// streams round-robin, each lane's state (REGS x 32-bit) kept in its own bank; the imem and sequencer shared. +// One lane-op per cycle. Compared with core_v6 at the same LANES x REGS this removes LANES-1 copies of the units and +// keeps the register state and the imem: the energy per lane-op is the state's cost plus one unit set's. +// The butterfly shuffle across lanes needs every lane's source register in the same cycle, so the shuffle reads the +// bank-wide source column (as the SIMD core does) and the lane in turn takes its word. Loads return on ld_val. +`include "lane_common.vh" +module core_tm #(parameter LANES = 8, parameter LOG_LANES = 3, parameter REGS = 64, parameter LOG_REGS = 6, + parameter IW = 40, parameter IMEM_LOG = 8) ( + input clk, input rst, input run, + input prog_we, input [9:0] prog_addr, input [IW-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel, + input [31:0] ld_val, + output [31:0] addr, output [31:0] out); + localparam IMEM = 1 << IMEM_LOG; + reg [IW-1:0] imem [0:IMEM-1]; + reg [IMEM_LOG-1:0] pc; reg [IMEM_LOG-1:0] n_q; reg [IW-1:0] ir; reg [LOG_LANES-1:0] lane; + reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q; + integer i; + // the sequencer: the same instruction is issued to each lane in turn (LANES cycles per instruction) + always @(posedge clk) begin + if (prog_we) imem[prog_addr[IMEM_LOG-1:0]] <= prog_data; + if (rst) begin pc <= 0; ir <= 0; lane <= 0; n_q <= {IMEM_LOG{1'b1}}; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end + else begin + if (cfg_en) begin n_q <= cfg_n[IMEM_LOG-1:0]; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end + if (run) begin + if (lane == {LOG_LANES{1'b1}}) begin ir <= imem[pc]; pc <= (pc == n_q) ? {IMEM_LOG{1'b0}} : pc + 1'b1; end + lane <= lane + 1'b1; + end + end + end + wire [3:0] op = ir[3:0]; + wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[4+LOG_REGS +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[4+2*LOG_REGS +: LOG_REGS]; + wire [4:0] imm = ir[4+3*LOG_REGS +: 5]; wire [7:0] aux = ir[9+3*LOG_REGS +: 8]; + wire is_load = (op == 4'd12); + wire [4:0] rn = (imm == 0) ? 5'd1 : imm; + wire [LOG_LANES-1:0] smask = imm[LOG_LANES-1:0]; + // the banked state: one bank per lane, read through the lane select (a chip's SRAM bank select) + reg [31:0] rf [0:LANES*REGS-1]; + wire [31:0] d = rf[lane*REGS + dst]; + wire [31:0] s = rf[lane*REGS + src]; + wire [31:0] s2 = rf[lane*REGS + src2]; + wire [31:0] sx = rf[(lane ^ smask)*REGS + src]; // the shuffle partner's source word + // the one execution port + function [7:0] pick; input [63:0] b; input [3:0] k; reg [7:0] v; + begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end + endfunction + wire [4:0] sn = (s[4:0] == 0) ? 5'd1 : s[4:0]; + wire mad = (op == 4'd8); + wire [63:0] p = (mad ? s : d) * (mad ? s2 : s); + wire [63:0] bytes = {s, d}; wire [15:0] sel = {aux, aux}; + wire [31:0] prm = {pick(bytes, sel[15:12]), pick(bytes, sel[11:8]), pick(bytes, sel[7:4]), pick(bytes, sel[3:0])}; + reg [31:0] lp; integer b; + always @* for (b = 0; b < 32; b = b + 1) lp[b] = aux[{d[b], s[b], s2[b]}]; + reg [31:0] r; + always @* begin + case (op) + 4'd0, 4'd13: r = d + s; + 4'd1, 4'd15: r = d - s; + 4'd2, 4'd14: r = d ^ s; + 4'd3: r = d | s; + 4'd4: r = `ROTL32(d, rn); + 4'd5: r = `ROTR32(d, sn); + 4'd6: r = p[31:0]; + 4'd7: r = p[63:32]; + 4'd8: r = p[31:0] + d; + 4'd9: r = d ^ sx; + 4'd10: r = prm; + 4'd11: r = lp; + default: r = ld_val ^ (32'h9e3779b9 * (lane + 1)); + endcase + end + wire [31:0] fx = s * m_q; + wire [4:0] frn = (r_q == 0) ? 5'd1 : r_q; + wire [31:0] fy = `ROTL32(fx, frn); + assign addr = is_load ? (((fy & wm_q) | off_q) & mask_q) : 32'd0; + always @(posedge clk) begin + if (rst) begin for (i = 0; i < LANES*REGS; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end + else if (run) rf[lane*REGS + dst] <= r; + end + assign out = r; +endmodule diff --git a/tools/chip-model/rtl/rtl/core_tm_8r64.v b/tools/chip-model/rtl/rtl/core_tm_8r64.v new file mode 100644 index 000000000..5db91edc4 --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_tm_8r64.v @@ -0,0 +1,7 @@ +`include "core_tm.v" +module core_tm_8r64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [39:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_tm #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .cfg_sel(cfg_sel), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_8cs64.v b/tools/chip-model/rtl/rtl/core_v6_8cs64.v new file mode 100644 index 000000000..4d482e2c8 --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_v6_8cs64.v @@ -0,0 +1,7 @@ +`include "core_v6.v" +module core_v6_8cs64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [40-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_v6 #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40), .IMEM_LOG(9)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/rtl/fp32_units.v b/tools/chip-model/rtl/rtl/fp32_units.v new file mode 100644 index 000000000..73a76c0a4 --- /dev/null +++ b/tools/chip-model/rtl/rtl/fp32_units.v @@ -0,0 +1,123 @@ +// The adversary's simplified FP32 units for the mixed-resource lane's candidate (class-v6-mixedfp): inputs are +// f(x) = as_float((x & 0x807FFFFF) | ((96 + ((x >> 23) & 63)) << 23)): never zero, denormal, NaN or Inf; exponents +// in [96, 159]; results normal or +0 (exact cancellation). The units drop NaN/Inf/denormal handling and the flags, +// keep the full 24-bit mantissa path, a full alignment and a full normaliser (the mantissas are uniform), RNE. +// Each lane reads two or three registers of an 8 x 32-bit window, applies f(), computes, xors the bits into dst. +`include "lane_common.vh" +// ---- the shared pieces ---- +module fp_unpack(input [31:0] x, output s, output [8:0] e, output [23:0] m); + assign s = x[31]; + assign e = 9'd96 + {3'b0, x[28:23]}; // the masked exponent, 96..159 + assign m = {1'b1, x[22:0]}; +endmodule +module lzc48(input [47:0] v, output reg [5:0] n); // leading-zero count (v != 0) + integer i; always @* begin n = 6'd47; for (i = 47; i >= 0; i = i - 1) if (v[i]) begin n = 6'd47 - i; i = -1; end end +endmodule +module lzc32(input [31:0] v, output reg [5:0] n); + integer i; always @* begin n = 6'd31; for (i = 31; i >= 0; i = i - 1) if (v[i]) begin n = 6'd31 - i; i = -1; end end +endmodule +// ---- the FMA: fma(a, b, c) = a*b + c, one rounding (RNE), exponents in the lane's ranges ---- +module fp_fma(input [31:0] a, input [31:0] b, input [31:0] c, output [31:0] y); + wire sa, sb, sc; wire [8:0] ea, eb, ec; wire [23:0] ma, mb, mc; + fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb); fp_unpack uc(c, sc, ec, mc); + wire [47:0] prod = ma * mb; // 48-bit product, binary point after bit 46 + wire sp = sa ^ sb; + wire [9:0] ep = {1'b0, ea} + {1'b0, eb} - 10'd127; // product exponent (bias kept), 65..192 + // align the addend to the product: a 100-bit window keeps full precision for exponent gaps up to about 96 (the lane's bound) + wire [9:0] diff = (ep >= {1'b0, ec}) ? ep - {1'b0, ec} : {1'b0, ec} - ep; + wire prod_big = (ep >= {1'b0, ec}); + wire [99:0] pw = {2'b0, prod, 50'b0}; + wire [99:0] cw = {2'b0, mc, 24'b0, 50'b0}; // the addend at the product's scale when exponents equal + wire [6:0] sh = (diff > 10'd99) ? 7'd99 : diff[6:0]; + wire [99:0] smw = prod_big ? (cw >> sh) : (pw >> sh); + wire [99:0] bgw = prod_big ? pw : cw; + wire sbig = prod_big ? sp : sc; wire ssmall = prod_big ? sc : sp; + wire [9:0] ebig = prod_big ? ep : {1'b0, ec}; + wire [100:0] sum = (sbig == ssmall) ? ({1'b0, bgw} + {1'b0, smw}) : ({1'b0, bgw} - {1'b0, smw}); + wire [100:0] mag = sum[100] ? (~sum + 1'b1) : sum; // two's complement when the subtraction went negative + wire ssum = sum[100] ? ssmall : sbig; + // normalise: find the leading one in the 101-bit magnitude + reg [6:0] lz; integer i; + always @* begin lz = 7'd100; for (i = 100; i >= 0; i = i - 1) if (mag[i]) begin lz = 7'd100 - i; i = -1; end end + wire [100:0] norm = mag << lz; // leading one at bit 100 + wire [23:0] mant = norm[100:77]; + wire guard = norm[76]; wire sticky = |norm[75:0]; + wire round_up = guard & (sticky | mant[0]); + wire [24:0] mr = {1'b0, mant} + round_up; + wire carry = mr[24]; + wire [9:0] eres = ebig + 10'd2 - lz + carry; // the leading one of bgw sat at bit 98 (two headroom bits) + wire zero = (mag == 0); + wire [7:0] eout = eres[7:0]; + assign y = zero ? 32'h0 : {ssum, eout, carry ? mr[23:1] : mr[22:0]}; +endmodule +// ---- the adder and the multiplier as their own units ---- +module fp_add(input [31:0] a, input [31:0] b, output [31:0] y); + wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb; + fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb); + wire abig = (ea > eb) || (ea == eb && ma >= mb); + wire [8:0] ebig = abig ? ea : eb; wire [8:0] esm = abig ? eb : ea; + wire [23:0] mbig = abig ? ma : mb; wire [23:0] msm = abig ? mb : ma; + wire sbig = abig ? sa : sb; wire ssm = abig ? sb : sa; + wire [8:0] diff = ebig - esm; wire [6:0] sh = (diff > 9'd70) ? 7'd70 : diff[6:0]; + wire [73:0] bw = {1'b0, mbig, 49'b0}; wire [73:0] sw = {1'b0, msm, 49'b0} >> sh; + wire [74:0] sum = (sbig == ssm) ? ({1'b0, bw} + {1'b0, sw}) : ({1'b0, bw} - {1'b0, sw}); + reg [6:0] lz; integer i; + always @* begin lz = 7'd74; for (i = 74; i >= 0; i = i - 1) if (sum[i]) begin lz = 7'd74 - i; i = -1; end end + wire [74:0] norm = sum << lz; + wire [23:0] mant = norm[74:51]; wire guard = norm[50]; wire sticky = |norm[49:0]; + wire round_up = guard & (sticky | mant[0]); + wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24]; + wire [9:0] eres = {1'b0, ebig} + 10'd1 - lz + carry; + wire zero = (sum == 0); + assign y = zero ? 32'h0 : {sbig, eres[7:0], carry ? mr[23:1] : mr[22:0]}; +endmodule +module fp_mul(input [31:0] a, input [31:0] b, output [31:0] y); + wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb; + fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb); + wire [47:0] prod = ma * mb; + wire top = prod[47]; + wire [23:0] mant = top ? prod[47:24] : prod[46:23]; + wire guard = top ? prod[23] : prod[22]; wire sticky = top ? |prod[22:0] : |prod[21:0]; + wire round_up = guard & (sticky | mant[0]); + wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24]; + wire [9:0] eres = {1'b0, ea} + {1'b0, eb} - 10'd127 + top + carry; + assign y = {sa ^ sb, eres[7:0], carry ? mr[23:1] : mr[22:0]}; +endmodule +// ---- int32 to float, RNE ---- +module fp_cvt(input [31:0] a, output [31:0] y); + wire s = a[31]; wire [31:0] mag = s ? (~a + 1'b1) : a; + wire [5:0] lz; lzc32 l(mag, lz); + wire [31:0] norm = mag << lz; // leading one at bit 31 + wire [23:0] mant = norm[31:8]; wire guard = norm[7]; wire sticky = |norm[6:0]; + wire round_up = guard & (sticky | mant[0]); + wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24]; + wire [7:0] e = 8'd127 + 8'd31 - lz + carry; + assign y = (mag == 0) ? 32'h0 : {s, e, carry ? mr[23:1] : mr[22:0]}; +endmodule +// ---- the lane: op 0 fadd, 1 fmul, 2 ffma, 3 fcvt; d ^= bits(result) ---- +module lane_fp32( + input clk, input rst, + input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q; + integer i; + always @(posedge clk) begin + if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end + else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end + end + wire [31:0] d = rf[dst_q]; wire [31:0] s = rf[src_q]; wire [31:0] s2 = rf[src2_q]; + wire [31:0] ya, ym, yf, yc; + fp_add A(d, s, ya); + fp_mul M(d, s, ym); + fp_fma F(s, s2, d, yf); + fp_cvt C(s, yc); + reg [31:0] res; + always @* case (op_q) 2'd0: res = d ^ ya; 2'd1: res = d ^ ym; 2'd2: res = d ^ yf; default: res = d ^ yc; endcase + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 9; end + else rf[dst_q] <= ld_q ? ld_val_q : res; + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/tb/tb_core_common.vh b/tools/chip-model/rtl/tb/tb_core_common.vh index 38e969897..5a6f88d70 100644 --- a/tools/chip-model/rtl/tb/tb_core_common.vh +++ b/tools/chip-model/rtl/tb/tb_core_common.vh @@ -21,10 +21,32 @@ module tb; @(negedge clk); cfg_en = 1; cfg_n = `NPROG - 1; cfg_sel = `SEL; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff; @(negedge clk); cfg_en = 0; // the program: `NPROG instructions drawn with the class v4 weights +`ifdef CS + // the connected-state draw: step = load + 27-instruction spine block; `NPROG = 16 x 28 = 448 + begin : cs + integer st, q, last0, last1, last2, last3, mreg, areg, dreg, sreg, s2reg; + areg = 0; mreg = 1; + for (st = 0; st < `NPROG / 28; st = st + 1) begin + @(negedge clk); w = {$random, $random}; mreg = $random & 63; + prog_we = 1; prog_addr = st*28; prog_data = {w[`IW-1:4], 4'd12}; prog_data[4 +: 6] = mreg; prog_data[10 +: 6] = areg; // load: dst m_j, src a_j + last0 = mreg; last1 = mreg; last2 = mreg; last3 = mreg; + for (q = 0; q < 27; q = q + 1) begin + @(negedge clk); w = {$random, $random}; opc = draw_op($random); dreg = $random & 63; + case ($random & 3) 0: sreg = last0; 1: sreg = last1; 2: sreg = last2; default: sreg = last3; endcase + s2reg = last0; + if (q == 26 && !(opc == 4'd0 || opc == 4'd1 || opc == 4'd2 || opc == 4'd8 || opc == 4'd9)) opc = 4'd0; // the last instruction injects + prog_we = 1; prog_addr = st*28 + 1 + q; prog_data = {w[`IW-1:4], opc}; prog_data[4 +: 6] = dreg; prog_data[10 +: 6] = sreg; prog_data[16 +: 6] = s2reg; + last3 = last2; last2 = last1; last1 = last0; last0 = dreg; + end + areg = last0; + end + end +`else for (k = 0; k < `NPROG; k = k + 1) begin @(negedge clk); w = {$random, $random}; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random); prog_we = 1; prog_addr = k; prog_data = {w[`IW-1:4], opc}; end +`endif @(negedge clk); prog_we = 0; run = 1; for (n = 0; n < cycles; n = n + 1) begin @(negedge clk); ld_val = $random; acc = acc ^ out ^ addr; diff --git a/tools/chip-model/rtl/tb/tb_core_tm_8r64.v b/tools/chip-model/rtl/tb/tb_core_tm_8r64.v new file mode 100644 index 000000000..954ae207f --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_tm_8r64.v @@ -0,0 +1,6 @@ +`define TOP core_tm_8r64 +`define HALF 750 +`define IW 40 +`define NPROG 256 +`define SEL 64'hfedcba9876543210 +`include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_8cs64.v b/tools/chip-model/rtl/tb/tb_core_v6_8cs64.v new file mode 100644 index 000000000..83d71eb03 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_v6_8cs64.v @@ -0,0 +1,7 @@ +`define TOP core_v6_8cs64 +`define HALF 750 +`define IW 40 +`define NPROG 448 +`define CS 1 +`define SEL 64'hfedcba9876543210 +`include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_lane_fp32.v b/tools/chip-model/rtl/tb/tb_lane_fp32.v new file mode 100644 index 000000000..721a54e69 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_fp32.v @@ -0,0 +1,22 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + lane_fp32 dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, fixed_op, cycles; reg [31:0] acc = 0; + always #1000 clk = ~clk; + // a reference check of the units against the host's float arithmetic is the mixed lane's own (the ranges are its); + // this bench drives random registers and reports the checksum + initial begin + if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1; + if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random; + ld_en = (($random & 7) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule