shadow-k: the window's measured GPU side (no spill, 83 and 67 percent occupancy, at most 5 percent per load); the time-multiplexed core, the connected-state draw, the simplified FP32 units and the pod mode
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Documents-only replay of26844fde7(86e5b0fb24) for the box mirror master
This commit is contained in:
parent
eb4bcb98e0
commit
0dc59c302e
19 changed files with 402 additions and 6 deletions
|
|
@ -251,7 +251,25 @@ to about 0.57 at the lock at N3 (0.63 to about 0.80 node-for-node).
|
|||
registers per thread, local-memory spill bytes, occupancy) and the rate beside the 8-register base, clock
|
||||
18:00 UK; until then the modelled reading stands: about 110 of 255 registers per thread, occupancy about half,
|
||||
the rate expected to hold under the latency-bound chain (the 5090 hides about 330,000 ops per hash before compute
|
||||
binds) and the energy to move little, the per-lane register traffic the unmeasured term. ROW_GPU_REG64
|
||||
binds) and the energy to move little, the per-lane register traffic the unmeasured term.
|
||||
|
||||
The measured GPU side (the hash lane, 16:1x to 16:4x UK, RunPod secure pods, driver 580, the kit worker, 250 x
|
||||
2^24, nvidia-smi 1 Hz; ptxas from nvcc 12.8 -Xptxas -v on the pack's kernel; pods destroyed, USD 1.22):
|
||||
|
||||
| Card, pack | MH/s | W | microjoules per hash | registers per thread (ptxas) | spill | blocks per SM (occupancy) | per load | Label |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| 5090, the base (mx8-devnet-epoch0) | 141.74 | 303.1 | 2.139 | 30 | 0 B | 24 (4,080 warps) | 16.7 nJ | measured |
|
||||
| 5090, the window, arithmetic-only (hl-reg64, twice the base's work by construction) | 80.38 | 308.6 | 3.839 | 96 | 0 B | 20 (83 percent) | 15.0 nJ | measured: level per unit of work |
|
||||
| 5090, the window, full chain (hl-reg64c: every load's address mixes all 64 registers) | 70.96 | 320.3 | 4.513 | 88 | 0 B | 20 (83 percent) | 17.6 nJ (+5 percent) | measured |
|
||||
| 4090, the base | 62.67 | 208.9 | 3.333 | 29 | 0 B | 24 | 26.0 nJ | measured |
|
||||
| 4090, the window, arithmetic-only | 31.57 | 210.3 | 6.663 | 104 | 0 B | 16 (67 percent) | 26.0 nJ | measured: level |
|
||||
| 4090, the window, full chain | 31.38 | 216.5 | 6.898 | 87 | 0 B | 20 (83 percent) | 27.0 nJ (+4 percent) | measured |
|
||||
|
||||
So the card's side of the window defence is at most 5 percent per load: no spill on either card in either form, 88
|
||||
to 104 registers per thread, occupancy 67 to 83 percent, and the rate per unit of work held within 5 percent under
|
||||
the latency-bound chain. The sound class form is the full chain (the arithmetic-only fold fails the liveness rule;
|
||||
class string `+reg64c`, pack hl-v6-win with `check_window_liveness` in its suite). The chip's side (this section's
|
||||
gated rows) therefore carries the whole defence.
|
||||
|
||||
## 5. The chip edge at the measured k
|
||||
|
||||
|
|
|
|||
|
|
@ -23,8 +23,14 @@ CPUSET = $(if $(LEASE_ON),--cpuset-cpus {cpuset},)
|
|||
DOCKER := $(LEASEPFX) docker run --rm -u $(UID_GID) -e HOME=/tmp -e NUM_CORES=$(THREADS) $(CPUSET) -v $(WORK):/work
|
||||
ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG)
|
||||
SIM := $(DOCKER) -w /work $(SIM_IMG)
|
||||
# NO_DOCKER=1: the box IS the ORFS image (a rented pod started from openroad/orfs:latest with iverilog installed
|
||||
# and /work a symlink to this directory); the same targets run natively.
|
||||
ifeq ($(NO_DOCKER),1)
|
||||
ORFS := env NUM_CORES=$(THREADS) bash -c 'cd /OpenROAD-flow-scripts/flow && exec "$$@"' --
|
||||
SIM := env
|
||||
endif
|
||||
|
||||
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g
|
||||
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all core8g core8r64g coretm cs64 fp32
|
||||
top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt)
|
||||
|
||||
# per-family simulation tags (the op field fixed per row where the family has several ops)
|
||||
|
|
@ -46,6 +52,9 @@ SIMS_core8sel := mix
|
|||
SIMS_core32all := mix
|
||||
SIMS_core8g := mix
|
||||
SIMS_core8r64g := mix
|
||||
SIMS_coretm := mix
|
||||
SIMS_cs64 := mix
|
||||
SIMS_fp32 := mix fadd:+op=0 fmul:+op=1 ffma:+op=2 fcvt:+op=3
|
||||
|
||||
.PHONY: rows table clean
|
||||
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ import re, sys, os, csv
|
|||
|
||||
work = sys.argv[1] if len(sys.argv) > 1 else '.'
|
||||
# ops per cycle per design (the per-op divisor) and the GPU row each family is read against
|
||||
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8}
|
||||
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32, 'core8g': 8, 'core8r64g': 8, 'coretm': 1, 'cs64': 8, 'fp32': 1}
|
||||
# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a
|
||||
GPU = {
|
||||
'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2),
|
||||
|
|
@ -17,7 +17,7 @@ GPU = {
|
|||
'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4),
|
||||
'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed)
|
||||
'tile:mix': (4.1, 2.2),
|
||||
'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
|
||||
'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), 'core8g:mix': (11.3, 6.2), 'core8r64g:mix': (11.3, 6.2), 'coretm:mix': (11.3, 6.2), 'cs64:mix': (11.3, 6.2), 'fp32:mix': (9.2, 5.2), 'fp32:fadd': (9.2, 5.2), 'fp32:fmul': (9.2, 5.2), 'fp32:ffma': (9.2, 5.2), 'fp32:fcvt': (9.2, 5.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
|
||||
}
|
||||
# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed:
|
||||
# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent),
|
||||
|
|
@ -67,10 +67,10 @@ for d in OPS:
|
|||
pairs = [('vcd:s150', 'vcd:s600', 150, 600), ('vcd:short', 'vcd:synth', 150, 800), ('vcd:s150', 'vcd:s400', 150, 400)]
|
||||
for sh, lg, cs, cl in pairs:
|
||||
if sh in rows and lg in rows:
|
||||
rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else LOAD_CYCLES)
|
||||
rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else (452 if d == 'cs64' else LOAD_CYCLES))
|
||||
for tag, r in rows.items():
|
||||
sub = tag.split(':')[1] if ':' in tag else 'prop'
|
||||
key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld') else f'{d}:mix'
|
||||
key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld','fadd','fmul','ffma','fcvt') else f'{d}:mix'
|
||||
gpu = GPU.get(key, (None, None))
|
||||
pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ
|
||||
pj_dyn = (r['internal'] + r['switching']) * period / OPS[d]
|
||||
|
|
|
|||
18
tools/chip-model/rtl/flow/coretm.mk
Normal file
18
tools/chip-model/rtl/flow/coretm.mk
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
# ORFS design config for the programmable shadow core (core8: core_tm_8r64), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = core_tm_8r64
|
||||
export DESIGN_NICKNAME = coretm
|
||||
export VERILOG_FILES = /work/rtl/core_tm_8r64.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/coretm.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/coretm
|
||||
export SYNTH_MEMORY_MAX_BITS = 2000000
|
||||
# the adversary's register file: clock gating inferred (the ICG cells allowed back in)
|
||||
export INFER_CLKGATES = 1
|
||||
export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF*
|
||||
10
tools/chip-model/rtl/flow/coretm.sdc
Normal file
10
tools/chip-model/rtl/flow/coretm.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design core_tm_8r64
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
18
tools/chip-model/rtl/flow/cs64.mk
Normal file
18
tools/chip-model/rtl/flow/cs64.mk
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
# ORFS design config for the programmable shadow core (core8: core_v6_8cs64), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = core_v6_8cs64
|
||||
export DESIGN_NICKNAME = cs64
|
||||
export VERILOG_FILES = /work/rtl/core_v6_8cs64.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/cs64.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/cs64
|
||||
export SYNTH_MEMORY_MAX_BITS = 2000000
|
||||
# the adversary's register file: clock gating inferred (the ICG cells allowed back in)
|
||||
export INFER_CLKGATES = 1
|
||||
export DONT_USE_CELLS = *x1p*_ASAP7* *xp*_ASAP7* SDF*
|
||||
10
tools/chip-model/rtl/flow/cs64.sdc
Normal file
10
tools/chip-model/rtl/flow/cs64.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design core_v6_8cs64
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
|
|
@ -16,3 +16,6 @@ core8sel core_v6_8sel 1500
|
|||
core32all core_v6_32all 1500
|
||||
core8g core_v6_8 1500
|
||||
core8r64g core_v6_8r64 1500
|
||||
coretm core_tm_8r64 1500
|
||||
cs64 core_v6_8cs64 1500
|
||||
fp32 lane_fp32 2000
|
||||
|
|
|
|||
15
tools/chip-model/rtl/flow/fp32.mk
Normal file
15
tools/chip-model/rtl/flow/fp32.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the mul shadow-core family (top lane_fp32), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_fp32
|
||||
export DESIGN_NICKNAME = fp32
|
||||
export VERILOG_FILES = /work/rtl/fp32_units.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/fp32.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/fp32
|
||||
|
||||
10
tools/chip-model/rtl/flow/fp32.sdc
Normal file
10
tools/chip-model/rtl/flow/fp32.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_fp32
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 2000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
9
tools/chip-model/rtl/flow/pod.sh
Executable file
9
tools/chip-model/rtl/flow/pod.sh
Executable file
|
|
@ -0,0 +1,9 @@
|
|||
#!/usr/bin/env bash
|
||||
# pod.sh: bootstrap a rented pod started from openroad/orfs:latest (RunPod, root): iverilog, /work -> this dir.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
command -v iverilog >/dev/null || { apt-get update -qq >/dev/null 2>&1; apt-get install -y -qq iverilog rsync python3 >/dev/null 2>&1; }
|
||||
[ -e /work ] || ln -s "$(pwd)" /work
|
||||
export PATH=/OpenROAD-flow-scripts/tools/install/OpenROAD/bin:/OpenROAD-flow-scripts/tools/install/yosys/bin:$PATH
|
||||
echo "pod ready: $(nproc) cores, $(free -g | awk '/Mem/{print $2}') GB, yosys $(yosys -V | cut -d' ' -f2), $(which iverilog)"
|
||||
82
tools/chip-model/rtl/rtl/core_tm.v
Normal file
82
tools/chip-model/rtl/rtl/core_tm.v
Normal file
|
|
@ -0,0 +1,82 @@
|
|||
// The adversary's time-multiplexed core: ONE execution port (every class unit, once) serving LANES lanes' instruction
|
||||
// streams round-robin, each lane's state (REGS x 32-bit) kept in its own bank; the imem and sequencer shared.
|
||||
// One lane-op per cycle. Compared with core_v6 at the same LANES x REGS this removes LANES-1 copies of the units and
|
||||
// keeps the register state and the imem: the energy per lane-op is the state's cost plus one unit set's.
|
||||
// The butterfly shuffle across lanes needs every lane's source register in the same cycle, so the shuffle reads the
|
||||
// bank-wide source column (as the SIMD core does) and the lane in turn takes its word. Loads return on ld_val.
|
||||
`include "lane_common.vh"
|
||||
module core_tm #(parameter LANES = 8, parameter LOG_LANES = 3, parameter REGS = 64, parameter LOG_REGS = 6,
|
||||
parameter IW = 40, parameter IMEM_LOG = 8) (
|
||||
input clk, input rst, input run,
|
||||
input prog_we, input [9:0] prog_addr, input [IW-1:0] prog_data,
|
||||
input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel,
|
||||
input [31:0] ld_val,
|
||||
output [31:0] addr, output [31:0] out);
|
||||
localparam IMEM = 1 << IMEM_LOG;
|
||||
reg [IW-1:0] imem [0:IMEM-1];
|
||||
reg [IMEM_LOG-1:0] pc; reg [IMEM_LOG-1:0] n_q; reg [IW-1:0] ir; reg [LOG_LANES-1:0] lane;
|
||||
reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q;
|
||||
integer i;
|
||||
// the sequencer: the same instruction is issued to each lane in turn (LANES cycles per instruction)
|
||||
always @(posedge clk) begin
|
||||
if (prog_we) imem[prog_addr[IMEM_LOG-1:0]] <= prog_data;
|
||||
if (rst) begin pc <= 0; ir <= 0; lane <= 0; n_q <= {IMEM_LOG{1'b1}}; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end
|
||||
else begin
|
||||
if (cfg_en) begin n_q <= cfg_n[IMEM_LOG-1:0]; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end
|
||||
if (run) begin
|
||||
if (lane == {LOG_LANES{1'b1}}) begin ir <= imem[pc]; pc <= (pc == n_q) ? {IMEM_LOG{1'b0}} : pc + 1'b1; end
|
||||
lane <= lane + 1'b1;
|
||||
end
|
||||
end
|
||||
end
|
||||
wire [3:0] op = ir[3:0];
|
||||
wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[4+LOG_REGS +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[4+2*LOG_REGS +: LOG_REGS];
|
||||
wire [4:0] imm = ir[4+3*LOG_REGS +: 5]; wire [7:0] aux = ir[9+3*LOG_REGS +: 8];
|
||||
wire is_load = (op == 4'd12);
|
||||
wire [4:0] rn = (imm == 0) ? 5'd1 : imm;
|
||||
wire [LOG_LANES-1:0] smask = imm[LOG_LANES-1:0];
|
||||
// the banked state: one bank per lane, read through the lane select (a chip's SRAM bank select)
|
||||
reg [31:0] rf [0:LANES*REGS-1];
|
||||
wire [31:0] d = rf[lane*REGS + dst];
|
||||
wire [31:0] s = rf[lane*REGS + src];
|
||||
wire [31:0] s2 = rf[lane*REGS + src2];
|
||||
wire [31:0] sx = rf[(lane ^ smask)*REGS + src]; // the shuffle partner's source word
|
||||
// the one execution port
|
||||
function [7:0] pick; input [63:0] b; input [3:0] k; reg [7:0] v;
|
||||
begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end
|
||||
endfunction
|
||||
wire [4:0] sn = (s[4:0] == 0) ? 5'd1 : s[4:0];
|
||||
wire mad = (op == 4'd8);
|
||||
wire [63:0] p = (mad ? s : d) * (mad ? s2 : s);
|
||||
wire [63:0] bytes = {s, d}; wire [15:0] sel = {aux, aux};
|
||||
wire [31:0] prm = {pick(bytes, sel[15:12]), pick(bytes, sel[11:8]), pick(bytes, sel[7:4]), pick(bytes, sel[3:0])};
|
||||
reg [31:0] lp; integer b;
|
||||
always @* for (b = 0; b < 32; b = b + 1) lp[b] = aux[{d[b], s[b], s2[b]}];
|
||||
reg [31:0] r;
|
||||
always @* begin
|
||||
case (op)
|
||||
4'd0, 4'd13: r = d + s;
|
||||
4'd1, 4'd15: r = d - s;
|
||||
4'd2, 4'd14: r = d ^ s;
|
||||
4'd3: r = d | s;
|
||||
4'd4: r = `ROTL32(d, rn);
|
||||
4'd5: r = `ROTR32(d, sn);
|
||||
4'd6: r = p[31:0];
|
||||
4'd7: r = p[63:32];
|
||||
4'd8: r = p[31:0] + d;
|
||||
4'd9: r = d ^ sx;
|
||||
4'd10: r = prm;
|
||||
4'd11: r = lp;
|
||||
default: r = ld_val ^ (32'h9e3779b9 * (lane + 1));
|
||||
endcase
|
||||
end
|
||||
wire [31:0] fx = s * m_q;
|
||||
wire [4:0] frn = (r_q == 0) ? 5'd1 : r_q;
|
||||
wire [31:0] fy = `ROTL32(fx, frn);
|
||||
assign addr = is_load ? (((fy & wm_q) | off_q) & mask_q) : 32'd0;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < LANES*REGS; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end
|
||||
else if (run) rf[lane*REGS + dst] <= r;
|
||||
end
|
||||
assign out = r;
|
||||
endmodule
|
||||
7
tools/chip-model/rtl/rtl/core_tm_8r64.v
Normal file
7
tools/chip-model/rtl/rtl/core_tm_8r64.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`include "core_tm.v"
|
||||
module core_tm_8r64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [39:0] prog_data,
|
||||
input cfg_en, input [9:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [63:0] cfg_sel,
|
||||
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
|
||||
core_tm #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
|
||||
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .cfg_sel(cfg_sel), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
endmodule
|
||||
7
tools/chip-model/rtl/rtl/core_v6_8cs64.v
Normal file
7
tools/chip-model/rtl/rtl/core_v6_8cs64.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`include "core_v6.v"
|
||||
module core_v6_8cs64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [40-1:0] prog_data,
|
||||
input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
|
||||
core_v6 #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40), .IMEM_LOG(9)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
|
||||
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
endmodule
|
||||
123
tools/chip-model/rtl/rtl/fp32_units.v
Normal file
123
tools/chip-model/rtl/rtl/fp32_units.v
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
// The adversary's simplified FP32 units for the mixed-resource lane's candidate (class-v6-mixedfp): inputs are
|
||||
// f(x) = as_float((x & 0x807FFFFF) | ((96 + ((x >> 23) & 63)) << 23)): never zero, denormal, NaN or Inf; exponents
|
||||
// in [96, 159]; results normal or +0 (exact cancellation). The units drop NaN/Inf/denormal handling and the flags,
|
||||
// keep the full 24-bit mantissa path, a full alignment and a full normaliser (the mantissas are uniform), RNE.
|
||||
// Each lane reads two or three registers of an 8 x 32-bit window, applies f(), computes, xors the bits into dst.
|
||||
`include "lane_common.vh"
|
||||
// ---- the shared pieces ----
|
||||
module fp_unpack(input [31:0] x, output s, output [8:0] e, output [23:0] m);
|
||||
assign s = x[31];
|
||||
assign e = 9'd96 + {3'b0, x[28:23]}; // the masked exponent, 96..159
|
||||
assign m = {1'b1, x[22:0]};
|
||||
endmodule
|
||||
module lzc48(input [47:0] v, output reg [5:0] n); // leading-zero count (v != 0)
|
||||
integer i; always @* begin n = 6'd47; for (i = 47; i >= 0; i = i - 1) if (v[i]) begin n = 6'd47 - i; i = -1; end end
|
||||
endmodule
|
||||
module lzc32(input [31:0] v, output reg [5:0] n);
|
||||
integer i; always @* begin n = 6'd31; for (i = 31; i >= 0; i = i - 1) if (v[i]) begin n = 6'd31 - i; i = -1; end end
|
||||
endmodule
|
||||
// ---- the FMA: fma(a, b, c) = a*b + c, one rounding (RNE), exponents in the lane's ranges ----
|
||||
module fp_fma(input [31:0] a, input [31:0] b, input [31:0] c, output [31:0] y);
|
||||
wire sa, sb, sc; wire [8:0] ea, eb, ec; wire [23:0] ma, mb, mc;
|
||||
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb); fp_unpack uc(c, sc, ec, mc);
|
||||
wire [47:0] prod = ma * mb; // 48-bit product, binary point after bit 46
|
||||
wire sp = sa ^ sb;
|
||||
wire [9:0] ep = {1'b0, ea} + {1'b0, eb} - 10'd127; // product exponent (bias kept), 65..192
|
||||
// align the addend to the product: a 100-bit window keeps full precision for exponent gaps up to about 96 (the lane's bound)
|
||||
wire [9:0] diff = (ep >= {1'b0, ec}) ? ep - {1'b0, ec} : {1'b0, ec} - ep;
|
||||
wire prod_big = (ep >= {1'b0, ec});
|
||||
wire [99:0] pw = {2'b0, prod, 50'b0};
|
||||
wire [99:0] cw = {2'b0, mc, 24'b0, 50'b0}; // the addend at the product's scale when exponents equal
|
||||
wire [6:0] sh = (diff > 10'd99) ? 7'd99 : diff[6:0];
|
||||
wire [99:0] smw = prod_big ? (cw >> sh) : (pw >> sh);
|
||||
wire [99:0] bgw = prod_big ? pw : cw;
|
||||
wire sbig = prod_big ? sp : sc; wire ssmall = prod_big ? sc : sp;
|
||||
wire [9:0] ebig = prod_big ? ep : {1'b0, ec};
|
||||
wire [100:0] sum = (sbig == ssmall) ? ({1'b0, bgw} + {1'b0, smw}) : ({1'b0, bgw} - {1'b0, smw});
|
||||
wire [100:0] mag = sum[100] ? (~sum + 1'b1) : sum; // two's complement when the subtraction went negative
|
||||
wire ssum = sum[100] ? ssmall : sbig;
|
||||
// normalise: find the leading one in the 101-bit magnitude
|
||||
reg [6:0] lz; integer i;
|
||||
always @* begin lz = 7'd100; for (i = 100; i >= 0; i = i - 1) if (mag[i]) begin lz = 7'd100 - i; i = -1; end end
|
||||
wire [100:0] norm = mag << lz; // leading one at bit 100
|
||||
wire [23:0] mant = norm[100:77];
|
||||
wire guard = norm[76]; wire sticky = |norm[75:0];
|
||||
wire round_up = guard & (sticky | mant[0]);
|
||||
wire [24:0] mr = {1'b0, mant} + round_up;
|
||||
wire carry = mr[24];
|
||||
wire [9:0] eres = ebig + 10'd2 - lz + carry; // the leading one of bgw sat at bit 98 (two headroom bits)
|
||||
wire zero = (mag == 0);
|
||||
wire [7:0] eout = eres[7:0];
|
||||
assign y = zero ? 32'h0 : {ssum, eout, carry ? mr[23:1] : mr[22:0]};
|
||||
endmodule
|
||||
// ---- the adder and the multiplier as their own units ----
|
||||
module fp_add(input [31:0] a, input [31:0] b, output [31:0] y);
|
||||
wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb;
|
||||
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb);
|
||||
wire abig = (ea > eb) || (ea == eb && ma >= mb);
|
||||
wire [8:0] ebig = abig ? ea : eb; wire [8:0] esm = abig ? eb : ea;
|
||||
wire [23:0] mbig = abig ? ma : mb; wire [23:0] msm = abig ? mb : ma;
|
||||
wire sbig = abig ? sa : sb; wire ssm = abig ? sb : sa;
|
||||
wire [8:0] diff = ebig - esm; wire [6:0] sh = (diff > 9'd70) ? 7'd70 : diff[6:0];
|
||||
wire [73:0] bw = {1'b0, mbig, 49'b0}; wire [73:0] sw = {1'b0, msm, 49'b0} >> sh;
|
||||
wire [74:0] sum = (sbig == ssm) ? ({1'b0, bw} + {1'b0, sw}) : ({1'b0, bw} - {1'b0, sw});
|
||||
reg [6:0] lz; integer i;
|
||||
always @* begin lz = 7'd74; for (i = 74; i >= 0; i = i - 1) if (sum[i]) begin lz = 7'd74 - i; i = -1; end end
|
||||
wire [74:0] norm = sum << lz;
|
||||
wire [23:0] mant = norm[74:51]; wire guard = norm[50]; wire sticky = |norm[49:0];
|
||||
wire round_up = guard & (sticky | mant[0]);
|
||||
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
|
||||
wire [9:0] eres = {1'b0, ebig} + 10'd1 - lz + carry;
|
||||
wire zero = (sum == 0);
|
||||
assign y = zero ? 32'h0 : {sbig, eres[7:0], carry ? mr[23:1] : mr[22:0]};
|
||||
endmodule
|
||||
module fp_mul(input [31:0] a, input [31:0] b, output [31:0] y);
|
||||
wire sa, sb; wire [8:0] ea, eb; wire [23:0] ma, mb;
|
||||
fp_unpack ua(a, sa, ea, ma); fp_unpack ub(b, sb, eb, mb);
|
||||
wire [47:0] prod = ma * mb;
|
||||
wire top = prod[47];
|
||||
wire [23:0] mant = top ? prod[47:24] : prod[46:23];
|
||||
wire guard = top ? prod[23] : prod[22]; wire sticky = top ? |prod[22:0] : |prod[21:0];
|
||||
wire round_up = guard & (sticky | mant[0]);
|
||||
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
|
||||
wire [9:0] eres = {1'b0, ea} + {1'b0, eb} - 10'd127 + top + carry;
|
||||
assign y = {sa ^ sb, eres[7:0], carry ? mr[23:1] : mr[22:0]};
|
||||
endmodule
|
||||
// ---- int32 to float, RNE ----
|
||||
module fp_cvt(input [31:0] a, output [31:0] y);
|
||||
wire s = a[31]; wire [31:0] mag = s ? (~a + 1'b1) : a;
|
||||
wire [5:0] lz; lzc32 l(mag, lz);
|
||||
wire [31:0] norm = mag << lz; // leading one at bit 31
|
||||
wire [23:0] mant = norm[31:8]; wire guard = norm[7]; wire sticky = |norm[6:0];
|
||||
wire round_up = guard & (sticky | mant[0]);
|
||||
wire [24:0] mr = {1'b0, mant} + round_up; wire carry = mr[24];
|
||||
wire [7:0] e = 8'd127 + 8'd31 - lz + carry;
|
||||
assign y = (mag == 0) ? 32'h0 : {s, e, carry ? mr[23:1] : mr[22:0]};
|
||||
endmodule
|
||||
// ---- the lane: op 0 fadd, 1 fmul, 2 ffma, 3 fcvt; d ^= bits(result) ----
|
||||
module lane_fp32(
|
||||
input clk, input rst,
|
||||
input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end
|
||||
else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] d = rf[dst_q]; wire [31:0] s = rf[src_q]; wire [31:0] s2 = rf[src2_q];
|
||||
wire [31:0] ya, ym, yf, yc;
|
||||
fp_add A(d, s, ya);
|
||||
fp_mul M(d, s, ym);
|
||||
fp_fma F(s, s2, d, yf);
|
||||
fp_cvt C(s, yc);
|
||||
reg [31:0] res;
|
||||
always @* case (op_q) 2'd0: res = d ^ ya; 2'd1: res = d ^ ym; 2'd2: res = d ^ yf; default: res = d ^ yc; endcase
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 9; end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : res;
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
|
|
@ -21,10 +21,32 @@ module tb;
|
|||
@(negedge clk); cfg_en = 1; cfg_n = `NPROG - 1; cfg_sel = `SEL; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff;
|
||||
@(negedge clk); cfg_en = 0;
|
||||
// the program: `NPROG instructions drawn with the class v4 weights
|
||||
`ifdef CS
|
||||
// the connected-state draw: step = load + 27-instruction spine block; `NPROG = 16 x 28 = 448
|
||||
begin : cs
|
||||
integer st, q, last0, last1, last2, last3, mreg, areg, dreg, sreg, s2reg;
|
||||
areg = 0; mreg = 1;
|
||||
for (st = 0; st < `NPROG / 28; st = st + 1) begin
|
||||
@(negedge clk); w = {$random, $random}; mreg = $random & 63;
|
||||
prog_we = 1; prog_addr = st*28; prog_data = {w[`IW-1:4], 4'd12}; prog_data[4 +: 6] = mreg; prog_data[10 +: 6] = areg; // load: dst m_j, src a_j
|
||||
last0 = mreg; last1 = mreg; last2 = mreg; last3 = mreg;
|
||||
for (q = 0; q < 27; q = q + 1) begin
|
||||
@(negedge clk); w = {$random, $random}; opc = draw_op($random); dreg = $random & 63;
|
||||
case ($random & 3) 0: sreg = last0; 1: sreg = last1; 2: sreg = last2; default: sreg = last3; endcase
|
||||
s2reg = last0;
|
||||
if (q == 26 && !(opc == 4'd0 || opc == 4'd1 || opc == 4'd2 || opc == 4'd8 || opc == 4'd9)) opc = 4'd0; // the last instruction injects
|
||||
prog_we = 1; prog_addr = st*28 + 1 + q; prog_data = {w[`IW-1:4], opc}; prog_data[4 +: 6] = dreg; prog_data[10 +: 6] = sreg; prog_data[16 +: 6] = s2reg;
|
||||
last3 = last2; last2 = last1; last1 = last0; last0 = dreg;
|
||||
end
|
||||
areg = last0;
|
||||
end
|
||||
end
|
||||
`else
|
||||
for (k = 0; k < `NPROG; k = k + 1) begin
|
||||
@(negedge clk); w = {$random, $random}; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random);
|
||||
prog_we = 1; prog_addr = k; prog_data = {w[`IW-1:4], opc};
|
||||
end
|
||||
`endif
|
||||
@(negedge clk); prog_we = 0; run = 1;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk); ld_val = $random; acc = acc ^ out ^ addr;
|
||||
|
|
|
|||
6
tools/chip-model/rtl/tb/tb_core_tm_8r64.v
Normal file
6
tools/chip-model/rtl/tb/tb_core_tm_8r64.v
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
`define TOP core_tm_8r64
|
||||
`define HALF 750
|
||||
`define IW 40
|
||||
`define NPROG 256
|
||||
`define SEL 64'hfedcba9876543210
|
||||
`include "tb_core_common.vh"
|
||||
7
tools/chip-model/rtl/tb/tb_core_v6_8cs64.v
Normal file
7
tools/chip-model/rtl/tb/tb_core_v6_8cs64.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`define TOP core_v6_8cs64
|
||||
`define HALF 750
|
||||
`define IW 40
|
||||
`define NPROG 448
|
||||
`define CS 1
|
||||
`define SEL 64'hfedcba9876543210
|
||||
`include "tb_core_common.vh"
|
||||
22
tools/chip-model/rtl/tb/tb_lane_fp32.v
Normal file
22
tools/chip-model/rtl/tb/tb_lane_fp32.v
Normal file
|
|
@ -0,0 +1,22 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
lane_fp32 dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, fixed_op, cycles; reg [31:0] acc = 0;
|
||||
always #1000 clk = ~clk;
|
||||
// a reference check of the units against the host's float arithmetic is the mixed lane's own (the ranges are its);
|
||||
// this bench drives random registers and reports the checksum
|
||||
initial begin
|
||||
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random;
|
||||
ld_en = (($random & 7) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
Loading…
Reference in a new issue