shadow-k: the window's measured GPU side (no spill, 83 and 67 percent occupancy, at most 5 percent per load); the time-multiplexed core, the connected-state draw, the simplified FP32 units and the pod mode

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-josh 2026-10-08 15:47:21 +01:00
parent f1ddf0b397
commit 26844fde7f
19 changed files with 402 additions and 6 deletions

View file

@ -251,7 +251,25 @@ to about 0.57 at the lock at N3 (0.63 to about 0.80 node-for-node).
registers per thread, local-memory spill bytes, occupancy) and the rate beside the 8-register base, clock
18:00 UK; until then the modelled reading stands: about 110 of 255 registers per thread, occupancy about half,
the rate expected to hold under the latency-bound chain (the 5090 hides about 330,000 ops per hash before compute
binds) and the energy to move little, the per-lane register traffic the unmeasured term. ROW_GPU_REG64
binds) and the energy to move little, the per-lane register traffic the unmeasured term.
The measured GPU side (the hash lane, 16:1x to 16:4x UK, RunPod secure pods, driver 580, the kit worker, 250 x
2^24, nvidia-smi 1 Hz; ptxas from nvcc 12.8 -Xptxas -v on the pack's kernel; pods destroyed, USD 1.22):
| Card, pack | MH/s | W | microjoules per hash | registers per thread (ptxas) | spill | blocks per SM (occupancy) | per load | Label |
|---|---|---|---|---|---|---|---|---|
| 5090, the base (mx8-devnet-epoch0) | 141.74 | 303.1 | 2.139 | 30 | 0 B | 24 (4,080 warps) | 16.7 nJ | measured |
| 5090, the window, arithmetic-only (hl-reg64, twice the base's work by construction) | 80.38 | 308.6 | 3.839 | 96 | 0 B | 20 (83 percent) | 15.0 nJ | measured: level per unit of work |
| 5090, the window, full chain (hl-reg64c: every load's address mixes all 64 registers) | 70.96 | 320.3 | 4.513 | 88 | 0 B | 20 (83 percent) | 17.6 nJ (+5 percent) | measured |
| 4090, the base | 62.67 | 208.9 | 3.333 | 29 | 0 B | 24 | 26.0 nJ | measured |
| 4090, the window, arithmetic-only | 31.57 | 210.3 | 6.663 | 104 | 0 B | 16 (67 percent) | 26.0 nJ | measured: level |
| 4090, the window, full chain | 31.38 | 216.5 | 6.898 | 87 | 0 B | 20 (83 percent) | 27.0 nJ (+4 percent) | measured |
So the card's side of the window defence is at most 5 percent per load: no spill on either card in either form, 88
to 104 registers per thread, occupancy 67 to 83 percent, and the rate per unit of work held within 5 percent under
the latency-bound chain. The sound class form is the full chain (the arithmetic-only fold fails the liveness rule;
class string `+reg64c`, pack hl-v6-win with `check_window_liveness` in its suite). The chip's side (this section's
gated rows) therefore carries the whole defence.
## 5. The chip edge at the measured k

View file

@ -23,8 +23,14 @@ CPUSET = $(if $(LEASE_ON),--cpuset-cpus {cpuset},)
DOCKER := $(LEASEPFX) docker run --rm -u $(UID_GID) -e HOME=/tmp -e NUM_CORES=$(THREADS) $(CPUSET) -v $(WORK):/work
ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG)
SIM := $(DOCKER) -w /work $(SIM_IMG)
# NO_DOCKER=1: the box IS the ORFS image (a rented pod started from openroad/orfs:latest with iverilog installed
# and /work a symlink to this directory); the same targets run natively.
ifeq ($(NO_DOCKER),1)
ORFS := env NUM_CORES=$(THREADS) bash -c 'cd /OpenROAD-flow-scripts/flow && exec "$$@"' --
SIM := env
endif
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g coretm cs64 fp32
top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt)
# per-family simulation tags (the op field fixed per row where the family has several ops)
@ -46,6 +52,9 @@ SIMS_core8sel := mix
SIMS_core32all := mix
SIMS_core8g := mix
SIMS_core8r64g := mix
SIMS_coretm := mix
SIMS_cs64 := mix
SIMS_fp32 := mix fadd:+op=0 fmul:+op=1 ffma:+op=2 fcvt:+op=3
.PHONY: rows table clean

View file

@ -6,7 +6,7 @@ import re, sys, os, csv
work = sys.argv[1] if len(sys.argv) > 1 else '.'
# ops per cycle per design (the per-op divisor) and the GPU row each family is read against
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8}
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8, 'coretm': 1, 'cs64': 8, 'fp32': 1}
# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a
GPU = {
'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2),
@ -17,7 +17,7 @@ GPU = {
'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4),
'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed)
'tile:mix': (4.1, 2.2),
'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), 'coretm:mix': (11.3, 6.2), 'cs64:mix': (11.3, 6.2), 'fp32:mix': (9.2, 5.2), 'fp32:fadd': (9.2, 5.2), 'fp32:fmul': (9.2, 5.2), 'fp32:ffma': (9.2, 5.2), 'fp32:fcvt': (9.2, 5.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
}
# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed:
# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent),
@ -67,10 +67,10 @@ for d in OPS:
pairs = [('vcd:s150', 'vcd:s600', 150, 600), ('vcd:short', 'vcd:synth', 150, 800), ('vcd:s150', 'vcd:s400', 150, 400)]
for sh, lg, cs, cl in pairs:
if sh in rows and lg in rows:
rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else LOAD_CYCLES)
rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else (452 if d == 'cs64' else LOAD_CYCLES))
for tag, r in rows.items():
sub = tag.split(':')[1] if ':' in tag else 'prop'
key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld') else f'{d}:mix'
key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld','fadd','fmul','ffma','fcvt') else f'{d}:mix'
gpu = GPU.get(key, (None, None))
pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ
pj_dyn = (r['internal'] + r['switching']) * period / OPS[d]

View file

@ -0,0 +1,18 @@
# ORFS design config for the programmable shadow core (core8: core_tm_8r64), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = core_tm_8r64
export DESIGN_NICKNAME = coretm
export VERILOG_FILES = /work/rtl/core_tm_8r64.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/coretm.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/coretm
export SYNTH_MEMORY_MAX_BITS = 2000000
# the adversary's register file: clock gating inferred (the ICG cells allowed back in)
export INFER_CLKGATES = 1
export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF*

View file

@ -0,0 +1,10 @@
current_design core_tm_8r64
set clk_name core_clock
set clk_port_name clk
set clk_period 1500
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,18 @@
# ORFS design config for the programmable shadow core (core8: core_v6_8cs64), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = core_v6_8cs64
export DESIGN_NICKNAME = cs64
export VERILOG_FILES = /work/rtl/core_v6_8cs64.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/cs64.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/cs64
export SYNTH_MEMORY_MAX_BITS = 2000000
# the adversary's register file: clock gating inferred (the ICG cells allowed back in)
export INFER_CLKGATES = 1
export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF*

View file

@ -0,0 +1,10 @@
current_design core_v6_8cs64
set clk_name core_clock
set clk_port_name clk
set clk_period 1500
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -16,3 +16,6 @@ core8sel core_v6_8sel 1500
core32all core_v6_32all 1500
core8g core_v6_8 1500
core8r64g core_v6_8r64 1500
coretm core_tm_8r64 1500
cs64 core_v6_8cs64 1500
fp32 lane_fp32 2000

View file

@ -0,0 +1,15 @@
# ORFS design config for the mul shadow-core family (top lane_fp32), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_fp32
export DESIGN_NICKNAME = fp32
export VERILOG_FILES = /work/rtl/fp32_units.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/fp32.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/fp32

View file

@ -0,0 +1,10 @@
current_design lane_fp32
set clk_name core_clock
set clk_port_name clk
set clk_period 2000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,9 @@
#!/usr/bin/env bash
# pod.sh: bootstrap a rented pod started from openroad/orfs:latest (RunPod, root): iverilog, /work -> this dir.
set -euo pipefail
cd "$(dirname "$0")/.."
export DEBIAN_FRONTEND=noninteractive
command -v iverilog >/dev/null || { apt-get update -qq >/dev/null 2>&1; apt-get install -y -qq iverilog rsync python3 >/dev/null 2>&1; }
[ -e /work ] || ln -s "$(pwd)" /work
export PATH=/OpenROAD-flow-scripts/tools/install/OpenROAD/bin:/OpenROAD-flow-scripts/tools/install/yosys/bin:$PATH
echo "pod ready: $(nproc) cores, $(free -g | awk '/Mem/{print $2}') GB, yosys $(yosys -V | cut -d' ' -f2), $(which iverilog)"

View file

@ -0,0 +1,82 @@
// The adversary's time-multiplexed core: ONE execution port (every class unit, once) serving LANES lanes' instruction
// streams round-robin, each lane's state (REGS x 32-bit) kept in its own bank; the imem and sequencer shared.
// One lane-op per cycle. Compared with core_v6 at the same LANES x REGS this removes LANES-1 copies of the units and
// keeps the register state and the imem: the energy per lane-op is the state's cost plus one unit set's.
// The butterfly shuffle across lanes needs every lane's source register in the same cycle, so the shuffle reads the
// bank-wide source column (as the SIMD core does) and the lane in turn takes its word. Loads return on ld_val.
`include "lane_common.vh"
module core_tm #(parameter LANES = 8, parameter LOG_LANES = 3, parameter REGS = 64, parameter LOG_REGS = 6,
parameter IW = 40, parameter IMEM_LOG = 8) (
input clk, input rst, input run,
input prog_we, input [9:0] prog_addr, input [IW-1:0] prog_data,
input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel,
input [31:0] ld_val,
output [31:0] addr, output [31:0] out);
localparam IMEM = 1 << IMEM_LOG;
reg [IW-1:0] imem [0:IMEM-1];
reg [IMEM_LOG-1:0] pc; reg [IMEM_LOG-1:0] n_q; reg [IW-1:0] ir; reg [LOG_LANES-1:0] lane;
reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q;
integer i;
// the sequencer: the same instruction is issued to each lane in turn (LANES cycles per instruction)
always @(posedge clk) begin
if (prog_we) imem[prog_addr[IMEM_LOG-1:0]] <= prog_data;
if (rst) begin pc <= 0; ir <= 0; lane <= 0; n_q <= {IMEM_LOG{1'b1}}; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end
else begin
if (cfg_en) begin n_q <= cfg_n[IMEM_LOG-1:0]; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end
if (run) begin
if (lane == {LOG_LANES{1'b1}}) begin ir <= imem[pc]; pc <= (pc == n_q) ? {IMEM_LOG{1'b0}} : pc + 1'b1; end
lane <= lane + 1'b1;
end
end
end
wire [3:0] op = ir[3:0];
wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[4+LOG_REGS +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[4+2*LOG_REGS +: LOG_REGS];
wire [4:0] imm = ir[4+3*LOG_REGS +: 5]; wire [7:0] aux = ir[9+3*LOG_REGS +: 8];
wire is_load = (op == 4'd12);
wire [4:0] rn = (imm == 0) ? 5'd1 : imm;
wire [LOG_LANES-1:0] smask = imm[LOG_LANES-1:0];
// the banked state: one bank per lane, read through the lane select (a chip's SRAM bank select)
reg [31:0] rf [0:LANES*REGS-1];
wire [31:0] d = rf[lane*REGS + dst];
wire [31:0] s = rf[lane*REGS + src];
wire [31:0] s2 = rf[lane*REGS + src2];
wire [31:0] sx = rf[(lane ^ smask)*REGS + src]; // the shuffle partner's source word
// the one execution port
function [7:0] pick; input [63:0] b; input [3:0] k; reg [7:0] v;
begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end
endfunction
wire [4:0] sn = (s[4:0] == 0) ? 5'd1 : s[4:0];
wire mad = (op == 4'd8);
wire [63:0] p = (mad ? s : d) * (mad ? s2 : s);
wire [63:0] bytes = {s, d}; wire [15:0] sel = {aux, aux};
wire [31:0] prm = {pick(bytes, sel[15:12]), pick(bytes, sel[11:8]), pick(bytes, sel[7:4]), pick(bytes, sel[3:0])};
reg [31:0] lp; integer b;
always @* for (b = 0; b < 32; b = b + 1) lp[b] = aux[{d[b], s[b], s2[b]}];
reg [31:0] r;
always @* begin
case (op)
4'd0, 4'd13: r = d + s;
4'd1, 4'd15: r = d - s;
4'd2, 4'd14: r = d ^ s;
4'd3: r = d | s;
4'd4: r = `ROTL32(d, rn);
4'd5: r = `ROTR32(d, sn);
4'd6: r = p[31:0];
4'd7: r = p[63:32];
4'd8: r = p[31:0] + d;
4'd9: r = d ^ sx;
4'd10: r = prm;
4'd11: r = lp;
default: r = ld_val ^ (32'h9e3779b9 * (lane + 1));
endcase
end
wire [31:0] fx = s * m_q;
wire [4:0] frn = (r_q == 0) ? 5'd1 : r_q;
wire [31:0] fy = `ROTL32(fx, frn);
assign addr = is_load ? (((fy & wm_q) | off_q) & mask_q) : 32'd0;
always @(posedge clk) begin
if (rst) begin for (i = 0; i < LANES*REGS; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end
else if (run) rf[lane*REGS + dst] <= r;
end
assign out = r;
endmodule

View file

@ -0,0 +1,7 @@
`include "core_tm.v"
module core_tm_8r64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [39:0] prog_data,
input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel,
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
core_tm #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .cfg_sel(cfg_sel), .ld_val(ld_val), .addr(addr), .out(out));
endmodule

View file

@ -0,0 +1,7 @@
`include "core_v6.v"
module core_v6_8cs64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [40-1:0] prog_data,
input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
core_v6 #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40), .IMEM_LOG(9)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
endmodule

View file

@ -0,0 +1,123 @@
// The adversary's simplified FP32 units for the mixed-resource lane's candidate (class-v6-mixedfp): inputs are
// f(x) = as_float((x & 0x807FFFFF) | ((96 + ((x >> 23) & 63)) << 23)): never zero, denormal, NaN or Inf; exponents
// in [96, 159]; results normal or +0 (exact cancellation). The units drop NaN/Inf/denormal handling and the flags,
// keep the full 24-bit mantissa path, a full alignment and a full normaliser (the mantissas are uniform), RNE.
// Each lane reads two or three registers of an 8 x 32-bit window, applies f(), computes, xors the bits into dst.
`include "lane_common.vh"
// ---- the shared pieces ----
module fp_unpack(input [31:0] x, output s, output [8:0] e, output [23:0] m);
assign s = x[31];
assign e = 9'd96 + {3'b0, x[28:23]}; // the masked exponent, 96..159
assign m = {1'b1, x[22:0]};
endmodule
module lzc48(input [47:0] v, output reg [5:0] n); // leading-zero count (v != 0)
integer i; always @* begin n = 6'd47; for (i = 47; i >= 0; i = i - 1) if (v[i]) begin n = 6'd47 - i; i = -1; end end
endmodule
module lzc32(input [31:0] v, output reg [5:0] n);
integer i; always @* begin n = 6'd31; for (i = 31; i >= 0; i = i - 1) if (v[i]) begin n = 6'd31 - i; i = -1; end end
endmodule
// ---- the FMA: fma(a, b, c) = a*b + c, one rounding (RNE), exponents in the lane's ranges ----
module fp_fma(input [31:0] a, input [31:0] b, input [31:0] c, output [31:0] y);
wire sa, sb, sc; wire [8:0] ea, eb, ec; wire [23:0] ma, mb, mc;
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb); fp_unpack uc(c, sc, ec, mc);
wire [47:0] prod = ma * mb; // 48-bit product, binary point after bit 46
wire sp = sa ^ sb;
wire [9:0] ep = {1'b0, ea} + {1'b0, eb} - 10'd127; // product exponent (bias kept), 65..192
// align the addend to the product: a 100-bit window keeps full precision for exponent gaps up to about 96 (the lane's bound)
wire [9:0] diff = (ep >= {1'b0, ec}) ? ep - {1'b0, ec} : {1'b0, ec} - ep;
wire prod_big = (ep >= {1'b0, ec});
wire [99:0] pw = {2'b0, prod, 50'b0};
wire [99:0] cw = {2'b0, mc, 24'b0, 50'b0}; // the addend at the product's scale when exponents equal
wire [6:0] sh = (diff > 10'd99) ? 7'd99 : diff[6:0];
wire [99:0] smw = prod_big ? (cw >> sh) : (pw >> sh);
wire [99:0] bgw = prod_big ? pw : cw;
wire sbig = prod_big ? sp : sc; wire ssmall = prod_big ? sc : sp;
wire [9:0] ebig = prod_big ? ep : {1'b0, ec};
wire [100:0] sum = (sbig == ssmall) ? ({1'b0, bgw} + {1'b0, smw}) : ({1'b0, bgw} - {1'b0, smw});
wire [100:0] mag = sum[100] ? (~sum + 1'b1) : sum; // two's complement when the subtraction went negative
wire ssum = sum[100] ? ssmall : sbig;
// normalise: find the leading one in the 101-bit magnitude
reg [6:0] lz; integer i;
always @* begin lz = 7'd100; for (i = 100; i >= 0; i = i - 1) if (mag[i]) begin lz = 7'd100 - i; i = -1; end end
wire [100:0] norm = mag << lz; // leading one at bit 100
wire [23:0] mant = norm[100:77];
wire guard = norm[76]; wire sticky = |norm[75:0];
wire round_up = guard & (sticky | mant[0]);
wire [24:0] mr = {1'b0, mant} + round_up;
wire carry = mr[24];
wire [9:0] eres = ebig + 10'd2 - lz + carry; // the leading one of bgw sat at bit 98 (two headroom bits)
wire zero = (mag == 0);
wire [7:0] eout = eres[7:0];
assign y = zero ? 32'h0 : {ssum, eout, carry ? mr[23:1] : mr[22:0]};
endmodule
// ---- the adder and the multiplier as their own units ----
module fp_add(input [31:0] a, input [31:0] b, output [31:0] y);
wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb;
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb);
wire abig = (ea > eb) || (ea == eb && ma >= mb);
wire [8:0] ebig = abig ? ea : eb; wire [8:0] esm = abig ? eb : ea;
wire [23:0] mbig = abig ? ma : mb; wire [23:0] msm = abig ? mb : ma;
wire sbig = abig ? sa : sb; wire ssm = abig ? sb : sa;
wire [8:0] diff = ebig - esm; wire [6:0] sh = (diff > 9'd70) ? 7'd70 : diff[6:0];
wire [73:0] bw = {1'b0, mbig, 49'b0}; wire [73:0] sw = {1'b0, msm, 49'b0} >> sh;
wire [74:0] sum = (sbig == ssm) ? ({1'b0, bw} + {1'b0, sw}) : ({1'b0, bw} - {1'b0, sw});
reg [6:0] lz; integer i;
always @* begin lz = 7'd74; for (i = 74; i >= 0; i = i - 1) if (sum[i]) begin lz = 7'd74 - i; i = -1; end end
wire [74:0] norm = sum << lz;
wire [23:0] mant = norm[74:51]; wire guard = norm[50]; wire sticky = |norm[49:0];
wire round_up = guard & (sticky | mant[0]);
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
wire [9:0] eres = {1'b0, ebig} + 10'd1 - lz + carry;
wire zero = (sum == 0);
assign y = zero ? 32'h0 : {sbig, eres[7:0], carry ? mr[23:1] : mr[22:0]};
endmodule
module fp_mul(input [31:0] a, input [31:0] b, output [31:0] y);
wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb;
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb);
wire [47:0] prod = ma * mb;
wire top = prod[47];
wire [23:0] mant = top ? prod[47:24] : prod[46:23];
wire guard = top ? prod[23] : prod[22]; wire sticky = top ? |prod[22:0] : |prod[21:0];
wire round_up = guard & (sticky | mant[0]);
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
wire [9:0] eres = {1'b0, ea} + {1'b0, eb} - 10'd127 + top + carry;
assign y = {sa ^ sb, eres[7:0], carry ? mr[23:1] : mr[22:0]};
endmodule
// ---- int32 to float, RNE ----
module fp_cvt(input [31:0] a, output [31:0] y);
wire s = a[31]; wire [31:0] mag = s ? (~a + 1'b1) : a;
wire [5:0] lz; lzc32 l(mag, lz);
wire [31:0] norm = mag << lz; // leading one at bit 31
wire [23:0] mant = norm[31:8]; wire guard = norm[7]; wire sticky = |norm[6:0];
wire round_up = guard & (sticky | mant[0]);
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
wire [7:0] e = 8'd127 + 8'd31 - lz + carry;
assign y = (mag == 0) ? 32'h0 : {s, e, carry ? mr[23:1] : mr[22:0]};
endmodule
// ---- the lane: op 0 fadd, 1 fmul, 2 ffma, 3 fcvt; d ^= bits(result) ----
module lane_fp32(
input clk, input rst,
input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q;
integer i;
always @(posedge clk) begin
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end
else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end
end
wire [31:0] d = rf[dst_q]; wire [31:0] s = rf[src_q]; wire [31:0] s2 = rf[src2_q];
wire [31:0] ya, ym, yf, yc;
fp_add A(d, s, ya);
fp_mul M(d, s, ym);
fp_fma F(s, s2, d, yf);
fp_cvt C(s, yc);
reg [31:0] res;
always @* case (op_q) 2'd0: res = d ^ ya; 2'd1: res = d ^ ym; 2'd2: res = d ^ yf; default: res = d ^ yc; endcase
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 9; end
else rf[dst_q] <= ld_q ? ld_val_q : res;
end
assign out = res;
endmodule

View file

@ -21,10 +21,32 @@ module tb;
@(negedge clk); cfg_en = 1; cfg_n = `NPROG - 1; cfg_sel = `SEL; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff;
@(negedge clk); cfg_en = 0;
// the program: `NPROG instructions drawn with the class v4 weights
`ifdef CS
// the connected-state draw: step = load + 27-instruction spine block; `NPROG = 16 x 28 = 448
begin : cs
integer st, q, last0, last1, last2, last3, mreg, areg, dreg, sreg, s2reg;
areg = 0; mreg = 1;
for (st = 0; st < `NPROG / 28; st = st + 1) begin
@(negedge clk); w = {$random, $random}; mreg = $random & 63;
prog_we = 1; prog_addr = st*28; prog_data = {w[`IW-1:4], 4'd12}; prog_data[4 +: 6] = mreg; prog_data[10 +: 6] = areg; // load: dst m_j, src a_j
last0 = mreg; last1 = mreg; last2 = mreg; last3 = mreg;
for (q = 0; q < 27; q = q + 1) begin
@(negedge clk); w = {$random, $random}; opc = draw_op($random); dreg = $random & 63;
case ($random & 3) 0: sreg = last0; 1: sreg = last1; 2: sreg = last2; default: sreg = last3; endcase
s2reg = last0;
if (q == 26 && !(opc == 4'd0 || opc == 4'd1 || opc == 4'd2 || opc == 4'd8 || opc == 4'd9)) opc = 4'd0; // the last instruction injects
prog_we = 1; prog_addr = st*28 + 1 + q; prog_data = {w[`IW-1:4], opc}; prog_data[4 +: 6] = dreg; prog_data[10 +: 6] = sreg; prog_data[16 +: 6] = s2reg;
last3 = last2; last2 = last1; last1 = last0; last0 = dreg;
end
areg = last0;
end
end
`else
for (k = 0; k < `NPROG; k = k + 1) begin
@(negedge clk); w = {$random, $random}; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random);
prog_we = 1; prog_addr = k; prog_data = {w[`IW-1:4], opc};
end
`endif
@(negedge clk); prog_we = 0; run = 1;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk); ld_val = $random; acc = acc ^ out ^ addr;

View file

@ -0,0 +1,6 @@
`define TOP core_tm_8r64
`define HALF 750
`define IW 40
`define NPROG 256
`define SEL 64'hfedcba9876543210
`include "tb_core_common.vh"

View file

@ -0,0 +1,7 @@
`define TOP core_v6_8cs64
`define HALF 750
`define IW 40
`define NPROG 448
`define CS 1
`define SEL 64'hfedcba9876543210
`include "tb_core_common.vh"

View file

@ -0,0 +1,22 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
lane_fp32 dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, fixed_op, cycles; reg [31:0] acc = 0;
always #1000 clk = ~clk;
// a reference check of the units against the host's float arithmetic is the mixed lane's own (the ranges are its);
// this bench drives random registers and reports the checksum
initial begin
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random;
ld_en = (($random & 7) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule