chip-model: the shadow-k floor lane's RTL, testbenches and ORFS flow (ASAP7) for every class v6 mixer family

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-josh 2026-10-08 12:59:10 +01:00
parent 7618e72981
commit 9009ed0f40
43 changed files with 869 additions and 0 deletions

View file

@ -0,0 +1,60 @@
# Shadow-k floor lane: RTL -> ASAP7 (Yosys + OpenROAD, the ORFS docker image) -> pJ per op from a
# random-input gate-level simulation. Reproduces every row of docs/analysis/class-v6/floor/shadow-k.md.
#
# make row-arx synthesise, place, route, simulate and report one family
# make rows every family (serially; use -j for parallel on the box)
# make table collect every power log into table.md and table.csv
#
# Runs on build-4 or build-3 (docker, the build user in the docker group). Never on the Mac.
# Images: openroad/orfs:latest (yosys 0.68, OpenROAD) and orfs-sim:latest (the same plus iverilog,
# built once with: docker run --name b -u root openroad/orfs:latest bash -c 'apt-get update && apt-get install -y iverilog' && docker commit b orfs-sim:latest).
WORK ?= $(abspath .)
ORFS_IMG ?= openroad/orfs:latest
SIM_IMG ?= orfs-sim:latest
UID_GID := $(shell id -u):$(shell id -g)
DOCKER := docker run --rm -u $(UID_GID) -e HOME=/tmp -v $(WORK):/work
ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG)
SIM := $(DOCKER) -w /work $(SIM_IMG)
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile
top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt)
# per-family simulation tags (the op field fixed per row where the family has several ops)
SIMS_arx := mix add:+op=0 sub:+op=1 xor:+op=2 or:+op=3 rotl:+op=4 rotr:+op=5
SIMS_mul := mix mul:+op=0 mulhi:+op=1 mad:+op=2
SIMS_prmt := mix
SIMS_lop3 := mix
SIMS_fold := mix
SIMS_shfl := mix
SIMS_xbar := mix
SIMS_scratch := mix
SIMS_tile := mix
.PHONY: rows table clean
rows: $(addprefix row-,$(DESIGNS))
row-%: power-%
@echo "row $* done: $(WORK)/out/$*/logs/asap7/$*/base/power.log"
flow-%:
mkdir -p out/$* logs
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk 2>&1 | tee logs/flow-$*.log
test -f out/$*/results/asap7/$*/base/6_final.v
sim-%: flow-%
mkdir -p sim/$*
$(SIM) bash /work/flow/gl2sim.sh $* $(call top,$*) 2>&1 | tee logs/gl2sim-$*.log
for t in $(SIMS_$*); do tag=$${t%%:*}; args=$${t#*:}; [ "$$args" = "$$t" ] && args=""; \
$(SIM) bash /work/flow/sim.sh $* $$tag $$args 2>&1 | tee logs/sim-$*-$$tag.log; done
power-%: sim-%
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power run 2>&1 | tee logs/power-$*.log
table:
python3 flow/collect.py $(WORK) > table.md
@echo wrote table.md and table.csv
clean:
rm -rf out sim logs table.md table.csv

View file

@ -0,0 +1,15 @@
# ORFS design config for the arx shadow-core family (top lane_arx), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_arx
export DESIGN_NICKNAME = arx
export VERILOG_FILES = /work/rtl/lane_arx.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/arx.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/arx

View file

@ -0,0 +1,10 @@
current_design lane_arx
set clk_name core_clock
set clk_port_name clk
set clk_period 1000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,72 @@
#!/usr/bin/env python3
"""Collect the ORFS power logs into the shadow-k table: pJ per op per family at ASAP7 (TC corner),
scaled to N5, N3 and N2, against the 5090's measured pJ per counted op (counter-asic-4-research.md 15.1a).
Usage: collect.py <work dir> (writes table.csv beside, prints table.md)."""
import re, sys, os, csv
work = sys.argv[1] if len(sys.argv) > 1 else '.'
# ops per cycle per design (the per-op divisor) and the GPU row each family is read against
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512}
# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a
GPU = {
'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2),
'arx:rotl': (11.3, 6.2), 'arx:rotr': (11.3, 6.2),
'mul:mix': (13.9, 8.3), 'mul:mul': (13.9, 8.3), 'mul:mad': (13.9, 8.3), 'mul:mulhi': (39.6, 21.0),
'prmt:mix': (22.3, 11.5), 'lop3:mix': (24.1, 13.0),
'fold:mix': (13.9, 8.3), # the fold is a multiply plus a rotate and masks: read against int_mul
'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4),
'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed)
'tile:mix': (4.1, 2.2), # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
}
# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed:
# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent),
# N3E -> N2 x0.72 (TSMC: 25 to 30 percent). Sources in shadow-k.md section 3.
SCALE = {'ASAP7': 1.0, 'N5': 0.70, 'N3': 0.70 * 0.72, 'N2': 0.70 * 0.72 * 0.72}
def parse_log(path):
txt = open(path, errors='replace').read()
period = None
m = re.search(r'FLOORK clock_period_ps ([\d.]+)', txt); period = float(m.group(1)) if m else None
cells = re.search(r'FLOORK cells (\d+)', txt); cells = int(cells.group(1)) if cells else None
rows = {}
for sec in re.split(r'FLOORK === ', txt)[1:]:
head, body = sec.split(' ===', 1)
tag = head.strip().replace('POWER_VCD ', 'vcd:').replace('POWER_PROPAGATED_0.5', 'prop')
# OpenSTA report_power: "Total <int> <sw> <leak> <total> 100.0%" in watts
tm = re.search(r'^Total\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)', body, re.M)
ann = re.search(r'(\d+)\s*\(\s*([\d.]+)%\)\s*annotated', body)
if tm:
rows[tag] = dict(internal=float(tm.group(1)), switching=float(tm.group(2)), leakage=float(tm.group(3)),
total=float(tm.group(4)), annotated=ann.group(2) if ann else '')
return period, cells, rows
out = []
for d in OPS:
log = os.path.join(work, 'out', d, 'logs', 'asap7', d, 'base', 'power.log')
if not os.path.exists(log):
continue
period, cells, rows = parse_log(log)
for tag, r in rows.items():
sub = tag.split(':')[1] if ':' in tag else 'prop'
key = f'{d}:{sub}' if sub != 'prop' else f'{d}:mix'
gpu = GPU.get(key, (None, None))
pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ
pj_dyn = (r['internal'] + r['switching']) * period / OPS[d]
row = dict(family=d, sim=tag, period_ps=period, cells=cells, ops_per_cycle=OPS[d],
total_W=r['total'], leak_W=r['leakage'], annotated_pct=r['annotated'],
pJ_asap7=pj, pJ_dyn_asap7=pj_dyn)
for node, s in SCALE.items():
row[f'pJ_{node}'] = pj * s
row[f'k_{node}_lock'] = (pj * s / gpu[1]) if gpu[1] else None
row[f'k_{node}_unlocked'] = (pj * s / gpu[0]) if gpu[0] else None
row['gpu_pJ_unlocked'] = gpu[0]; row['gpu_pJ_lock'] = gpu[1]
out.append(row)
if out:
with open(os.path.join(work, 'table.csv'), 'w', newline='') as f:
w = csv.DictWriter(f, fieldnames=list(out[0].keys())); w.writeheader(); w.writerows(out)
print('| Family | Sim | Period ps | Cells | Ops/cycle | Total W | Leak W | VCD annotated | pJ/op ASAP7 | pJ/op N5 | pJ/op N3 | pJ/op N2 | 5090 pJ/op unlocked / lock | k at N3 (lock) | k at N2 (lock) | k at N3 (unlocked) |')
print('|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|')
for r in out:
f = lambda v, n=3: ('' if v is None else (f'{v:.{n}g}' if isinstance(v, float) else str(v)))
print(f"| {r['family']} | {r['sim']} | {f(r['period_ps'])} | {r['cells']} | {r['ops_per_cycle']} | {f(r['total_W'])} | {f(r['leak_W'])} | {r['annotated_pct']} | {f(r['pJ_asap7'])} | {f(r['pJ_N5'])} | {f(r['pJ_N3'])} | {f(r['pJ_N2'])} | {f(r['gpu_pJ_unlocked'])} / {f(r['gpu_pJ_lock'])} | {f(r['k_N3_lock'])} | {f(r['k_N2_lock'])} | {f(r['k_N3_unlocked'])} |")

View file

@ -0,0 +1,9 @@
arx lane_arx 1000
mul lane_mul 1500
prmt lane_prmt 1000
lop3 lane_lop3 1000
fold lane_fold 1500
shfl shfl32 1000
xbar xbar32 1000
scratch scratch8k 1500
tile tile8 2000

View file

@ -0,0 +1,15 @@
# ORFS design config for the fold shadow-core family (top lane_fold), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_fold
export DESIGN_NICKNAME = fold
export VERILOG_FILES = /work/rtl/lane_fold.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/fold.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/fold

View file

@ -0,0 +1,10 @@
current_design lane_fold
set clk_name core_clock
set clk_port_name clk
set clk_period 1500
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,21 @@
#!/usr/bin/env bash
# gl2sim.sh <design-nickname> <top> : netlist -> sim netlist with cell bodies from the liberty.
set -euo pipefail
name=$1; top=$2
net=/work/out/$name/results/asap7/$name/base/6_final.v
lib=/OpenROAD-flow-scripts/flow/platforms/asap7/lib/NLDM
out=/work/sim/$name; mkdir -p $out
cat > $out/gl2sim.ys <<YS
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_AO_RVT_TT_nldm_211120.lib.gz
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_INVBUF_RVT_TT_nldm_220122.lib.gz
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_OA_RVT_TT_nldm_211120.lib.gz
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_SIMPLE_RVT_TT_nldm_211120.lib.gz
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_SEQ_RVT_TT_nldm_220123.lib
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_DFFHQNH2V2X_RVT_TT_nldm_FAKE.lib
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_DFFHQNV2X_RVT_TT_nldm_FAKE.lib
read_verilog $net
hierarchy -top $top
write_verilog -noattr $out/sim_net.v
YS
yosys -q -l $out/gl2sim.log $out/gl2sim.ys
echo "sim netlist: $out/sim_net.v"

View file

@ -0,0 +1,15 @@
# ORFS design config for the lop3 shadow-core family (top lane_lop3), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_lop3
export DESIGN_NICKNAME = lop3
export VERILOG_FILES = /work/rtl/lane_lop3.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/lop3.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/lop3

View file

@ -0,0 +1,10 @@
current_design lane_lop3
set clk_name core_clock
set clk_port_name clk
set clk_period 1000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,15 @@
# ORFS design config for the mul shadow-core family (top lane_mul), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_mul
export DESIGN_NICKNAME = mul
export VERILOG_FILES = /work/rtl/lane_mul.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/mul.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/mul

View file

@ -0,0 +1,10 @@
current_design lane_mul
set clk_name core_clock
set clk_port_name clk
set clk_period 1500
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,23 @@
# Power of the routed design under (a) the VCD of the random-input gate-level simulation and
# (b) a propagated 0.5 input activity, both at the TC corner (0.70 V, 0 C on ASAP7).
# Run through ORFS: make DESIGN_CONFIG=/work/flow/<name>.mk RUN_SCRIPT=/work/flow/power.tcl run
source $::env(SCRIPTS_DIR)/load.tcl
load_design 6_final.odb 6_final.sdc
set spef $::env(RESULTS_DIR)/6_final.spef
if { [file exists $spef] } { read_spef $spef } else { estimate_parasitics -global_routing }
puts "FLOORK clock_period_ps [expr [get_property [lindex [all_clocks] 0] period]]"
puts "FLOORK cells [llength [get_cells *]]"
report_tns
report_wns
puts "FLOORK === POWER_PROPAGATED_0.5 ==="
set_power_activity -input -activity 0.5 -duty 0.5
set_power_activity -input_port rst -activity 0 -duty 0
report_power
foreach vcd [glob -nocomplain /work/sim/$::env(DESIGN_NICKNAME)/*.vcd] {
set tag [file rootname [file tail $vcd]]
puts "FLOORK === POWER_VCD $tag ==="
read_vcd -scope tb/dut $vcd
if { [info commands report_activity_annotation] != "" } { report_activity_annotation }
report_power
}
puts "FLOORK done"

View file

@ -0,0 +1,15 @@
# ORFS design config for the prmt shadow-core family (top lane_prmt), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = lane_prmt
export DESIGN_NICKNAME = prmt
export VERILOG_FILES = /work/rtl/lane_prmt.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/prmt.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/prmt

View file

@ -0,0 +1,10 @@
current_design lane_prmt
set clk_name core_clock
set clk_port_name clk
set clk_period 1000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,15 @@
# ORFS design config for the scratch shadow-core family (top scratch8k), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = scratch8k
export DESIGN_NICKNAME = scratch
export VERILOG_FILES = /work/rtl/scratch8k.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/scratch.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/scratch

View file

@ -0,0 +1,10 @@
current_design scratch8k
set clk_name core_clock
set clk_port_name clk
set clk_period 1500
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,15 @@
# ORFS design config for the shfl shadow-core family (top shfl32), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = shfl32
export DESIGN_NICKNAME = shfl
export VERILOG_FILES = /work/rtl/shfl32.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/shfl.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/shfl

View file

@ -0,0 +1,10 @@
current_design shfl32
set clk_name core_clock
set clk_port_name clk
set clk_period 1000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,9 @@
#!/usr/bin/env bash
# sim.sh <design-nickname> <tag> [plusargs...] : gate-level random-input simulation -> /work/sim/<name>/<tag>.vcd
set -euo pipefail
name=$1; tag=$2; shift 2
out=/work/sim/$name
simcells=$(yosys-config --datdir)/simcells.v
iverilog -g2005 -o $out/sim_$tag $out/sim_net.v /work/tb/tb_$(sed -n "s/^$name \([^ ]*\) .*/\1/p" /work/flow/designs.txt).v $simcells
( cd $out && vvp -n sim_$tag "$@" | tee sim_$tag.log && mv dump.vcd $tag.vcd )
ls -la $out/$tag.vcd

View file

@ -0,0 +1,15 @@
# ORFS design config for the tile shadow-core family (top tile8), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = tile8
export DESIGN_NICKNAME = tile
export VERILOG_FILES = /work/rtl/tile8.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/tile.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/tile

View file

@ -0,0 +1,10 @@
current_design tile8
set clk_name core_clock
set clk_port_name clk
set clk_period 2000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,15 @@
# ORFS design config for the xbar shadow-core family (top xbar32), ASAP7.
export PLATFORM = asap7
export DESIGN_NAME = xbar32
export DESIGN_NICKNAME = xbar
export VERILOG_FILES = /work/rtl/xbar32.v
export VERILOG_INCLUDE_DIRS = /work/rtl
export SDC_FILE = /work/flow/xbar.sdc
export CORE_UTILIZATION = 40
export CORE_ASPECT_RATIO = 1
export CORE_MARGIN = 0.5
export PLACE_DENSITY = 0.55
export CORNER = TC
export SKIP_LAST_GASP = 1
export WORK_HOME = /work/out/xbar

View file

@ -0,0 +1,10 @@
current_design xbar32
set clk_name core_clock
set clk_port_name clk
set clk_period 1000
set clk_io_pct 0.2
set clk_port [get_ports $clk_port_name]
create_clock -name $clk_name -period $clk_period $clk_port
set non_clock_inputs [all_inputs -no_clocks]
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]

View file

@ -0,0 +1,38 @@
// 32-bit int ARX lane: add, sub, xor, or, rotl by immediate, rotr by register (the class v4/v6 families
// add, sub, xor, or, rotl, rotr). op: 0 add 1 sub 2 xor 3 or 4 rotl-imm 5 rotr-var 6 add 7 xor.
`include "lane_common.vh"
module lane_arx(
input clk, input rst,
input [2:0] op, input [2:0] dst, input [2:0] src, input [4:0] rot,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [2:0] op_q, dst_q, src_q; reg [4:0] rot_q; reg ld_q; reg [31:0] ld_val_q;
integer i;
always @(posedge clk) begin
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; rot_q <= 0; ld_q <= 0; ld_val_q <= 0; end
else begin op_q <= op; dst_q <= dst; src_q <= src; rot_q <= rot; ld_q <= ld_en; ld_val_q <= ld_val; end
end
wire [31:0] d = rf[dst_q];
wire [31:0] s = rf[src_q];
wire [4:0] rn = (rot_q == 5'd0) ? 5'd1 : rot_q; // rotate by 1..31
wire [4:0] sn = (s[4:0] == 5'd0) ? 5'd1 : s[4:0];
reg [31:0] res;
always @* begin
case (op_q)
3'd0: res = d + s;
3'd1: res = d - s;
3'd2: res = d ^ s;
3'd3: res = d | s;
3'd4: res = `ROTL32(d, rn);
3'd5: res = `ROTR32(d, sn);
3'd6: res = d + s;
default: res = d ^ s;
endcase
end
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end
else rf[dst_q] <= ld_q ? ld_val_q : res;
end
assign out = res;
endmodule

View file

@ -0,0 +1,5 @@
// Shared helpers for the shadow-core lanes (class v5/v6 mixer draw space).
// A lane = an instruction register, an 8 x 32-bit register window (flops), two or three
// read ports through muxes, one functional unit, one write port. One op per cycle.
`define ROTL32(x, n) (((x) << (n)) | ((x) >> (32 - (n))))
`define ROTR32(x, n) (((x) >> (n)) | ((x) << (32 - (n))))

View file

@ -0,0 +1,32 @@
// The index fold (class v6 layer 1, the ring-A rule): idx = ((rotl(x * M, R) & WM) | OFF) & MASK.
// M, R, WM, OFF and MASK are the era's constants, held in registers (loaded by cfg_en, then static).
// One fold per cycle on a register read; the index is written back to the lane's address register.
`include "lane_common.vh"
module lane_fold(
input clk, input rst,
input [2:0] dst, input [2:0] src,
input cfg_en, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [2:0] dst_q, src_q; reg ld_q; reg [31:0] ld_val_q;
reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q;
integer i;
always @(posedge clk) begin
if (rst) begin dst_q <= 0; src_q <= 0; ld_q <= 0; ld_val_q <= 0; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 32'h3; mask_q <= 32'h0fffffff; end
else begin
dst_q <= dst; src_q <= src; ld_q <= ld_en; ld_val_q <= ld_val;
if (cfg_en) begin m_q <= cfg_m | 32'h1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end
end
end
wire [31:0] x = rf[src_q];
wire [31:0] prod = x * m_q;
wire [4:0] rn = (r_q == 5'd0) ? 5'd1 : r_q;
wire [31:0] y = `ROTL32(prod, rn);
wire [31:0] res = ((y & wm_q) | off_q) & mask_q;
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 4; end
else rf[dst_q] <= ld_q ? ld_val_q : (res ^ rf[dst_q]);
end
assign out = res;
endmodule

View file

@ -0,0 +1,25 @@
// Three-input logic lane (LOP3): an 8-bit truth table over (d, s, s2), bitwise.
`include "lane_common.vh"
module lane_lop3(
input clk, input rst,
input [2:0] dst, input [2:0] src, input [2:0] src2, input [7:0] lut,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [2:0] dst_q, src_q, src2_q; reg [7:0] lut_q; reg ld_q; reg [31:0] ld_val_q;
integer i, b;
always @(posedge clk) begin
if (rst) begin dst_q <= 0; src_q <= 0; src2_q <= 0; lut_q <= 0; ld_q <= 0; ld_val_q <= 0; end
else begin dst_q <= dst; src_q <= src; src2_q <= src2; lut_q <= lut; ld_q <= ld_en; ld_val_q <= ld_val; end
end
wire [31:0] d = rf[dst_q];
wire [31:0] s = rf[src_q];
wire [31:0] s2 = rf[src2_q];
reg [31:0] res;
always @* for (b = 0; b < 32; b = b + 1) res[b] = lut_q[{d[b], s[b], s2[b]}];
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 3; end
else rf[dst_q] <= ld_q ? ld_val_q : res;
end
assign out = res;
endmodule

View file

@ -0,0 +1,36 @@
// 32 x 32 multiplier lane: mul (low 32), mulhi (high 32 of the 64-bit product), mad (src*src2 + dst, low 32).
// op: 0 mul 1 mulhi 2 mad 3 mul. The mad reads three registers.
`include "lane_common.vh"
module lane_mul(
input clk, input rst,
input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q;
integer i;
always @(posedge clk) begin
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end
else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end
end
wire [31:0] d = rf[dst_q];
wire [31:0] s = rf[src_q];
wire [31:0] s2 = rf[src2_q];
wire mad = (op_q == 2'd2);
wire [31:0] ma = mad ? s : d;
wire [31:0] mb = mad ? s2 : s;
wire [63:0] p = ma * mb;
reg [31:0] res;
always @* begin
case (op_q)
2'd1: res = p[63:32];
2'd2: res = p[31:0] + d;
default: res = p[31:0];
endcase
end
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 1; end
else rf[dst_q] <= ld_q ? ld_val_q : res;
end
assign out = res;
endmodule

View file

@ -0,0 +1,29 @@
// Byte-permute lane (PRMT): four output bytes, each one of the eight bytes of {s, d}, by a 16-bit selector
// (4 bits per output byte; bit 3 = sign-replicate as on the card).
`include "lane_common.vh"
module lane_prmt(
input clk, input rst,
input [2:0] dst, input [2:0] src, input [15:0] sel,
input ld_en, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:7];
reg [2:0] dst_q, src_q; reg [15:0] sel_q; reg ld_q; reg [31:0] ld_val_q;
integer i;
always @(posedge clk) begin
if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_val_q <= 0; end
else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_val_q <= ld_val; end
end
wire [31:0] d = rf[dst_q];
wire [31:0] s = rf[src_q];
wire [63:0] bytes = {s, d};
function [7:0] pick; input [63:0] b; input [3:0] k;
reg [7:0] v;
begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end
endfunction
wire [31:0] res = {pick(bytes, sel_q[15:12]), pick(bytes, sel_q[11:8]), pick(bytes, sel_q[7:4]), pick(bytes, sel_q[3:0])};
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 2; end
else rf[dst_q] <= ld_q ? ld_val_q : res;
end
assign out = res;
endmodule

View file

@ -0,0 +1,19 @@
// 8 KB random-read scratch (2,048 x 32-bit) as a flop array: one random read per cycle (the op), a write on
// one cycle in eight. A flop array is the pessimistic form of the chip's L1; an SRAM macro reads lower.
module scratch8k(
input clk, input rst,
input [10:0] raddr, input we, input [10:0] waddr, input [31:0] wdata,
output reg [31:0] rdata);
reg [31:0] mem [0:2047];
reg [10:0] raddr_q, waddr_q; reg we_q; reg [31:0] wdata_q;
integer i;
always @(posedge clk) begin
if (rst) begin raddr_q <= 0; waddr_q <= 0; we_q <= 0; wdata_q <= 0; end
else begin raddr_q <= raddr; waddr_q <= waddr; we_q <= we; wdata_q <= wdata; end
end
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 2048; i = i + 1) mem[i] <= 32'h9e3779b9 * (i + 1); end
else if (we_q) mem[waddr_q] <= wdata_q;
end
always @(posedge clk) rdata <= rst ? 32'd0 : mem[raddr_q];
endmodule

View file

@ -0,0 +1,36 @@
// 32-lane xor-mask shuffle over a 1 KB register window (32 lanes x 8 x 32 bits):
// r[dst][lane] ^= r[src][lane ^ m] for every lane, one instruction per cycle (32 lane-ops).
// The network is a 5-stage butterfly (one 2:1 mux per bit per stage).
`include "lane_common.vh"
module shfl32(
input clk, input rst,
input [2:0] dst, input [2:0] src, input [4:0] m,
input ld_en, input [4:0] ld_lane, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:255]; // rf[lane*8 + reg]
reg [2:0] dst_q, src_q; reg [4:0] m_q, ld_lane_q; reg ld_q; reg [31:0] ld_val_q;
integer i, l;
always @(posedge clk) begin
if (rst) begin dst_q <= 0; src_q <= 0; m_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end
else begin dst_q <= dst; src_q <= src; m_q <= m; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end
end
// explicit butterfly
wire [31:0] b0 [0:31]; wire [31:0] b1 [0:31]; wire [31:0] b2 [0:31]; wire [31:0] b3 [0:31]; wire [31:0] b4 [0:31]; wire [31:0] b5 [0:31];
genvar g;
generate for (g = 0; g < 32; g = g + 1) begin : bf
assign b0[g] = rf[g*8 + src_q];
assign b1[g] = m_q[0] ? b0[g ^ 1] : b0[g];
assign b2[g] = m_q[1] ? b1[g ^ 2] : b1[g];
assign b3[g] = m_q[2] ? b2[g ^ 4] : b2[g];
assign b4[g] = m_q[3] ? b3[g ^ 8] : b3[g];
assign b5[g] = m_q[4] ? b4[g ^ 16] : b4[g];
end endgenerate
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 5; end
else begin
for (l = 0; l < 32; l = l + 1)
rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ b5[l]);
end
end
assign out = b5[0] ^ b5[17];
endmodule

View file

@ -0,0 +1,33 @@
// int8 8x8x8 tile multiply: C[8][8] (int32) += A[8][8] (int8) x B[8][8] (int8): 512 MACs per cycle.
// A and B are loaded from the inputs every cycle; the accumulators clear on clr.
module tile8(
input clk, input rst, input clr,
input [511:0] a_in, input [511:0] b_in,
input [2:0] sel_r, input [2:0] sel_c,
output [31:0] out);
reg [511:0] a_q, b_q; reg clr_q; reg [2:0] sr_q, sc_q;
reg [31:0] c [0:63];
integer i;
always @(posedge clk) begin
if (rst) begin a_q <= 0; b_q <= 0; clr_q <= 1; sr_q <= 0; sc_q <= 0; end
else begin a_q <= a_in; b_q <= b_in; clr_q <= clr; sr_q <= sel_r; sc_q <= sel_c; end
end
genvar r, cc, k;
generate for (r = 0; r < 8; r = r + 1) begin : row
for (cc = 0; cc < 8; cc = cc + 1) begin : col
wire signed [19:0] dot;
wire signed [15:0] p [0:7];
for (k = 0; k < 8; k = k + 1) begin : mk
wire signed [7:0] av = a_q[(r*8 + k)*8 +: 8];
wire signed [7:0] bv = b_q[(k*8 + cc)*8 +: 8];
assign p[k] = av * bv;
end
assign dot = p[0] + p[1] + p[2] + p[3] + p[4] + p[5] + p[6] + p[7];
always @(posedge clk) begin
if (rst || clr_q) c[r*8 + cc] <= 32'd0;
else c[r*8 + cc] <= c[r*8 + cc] + {{12{dot[19]}}, dot};
end
end
end endgenerate
assign out = c[{sr_q, sc_q}];
endmodule

View file

@ -0,0 +1,29 @@
// 32-lane general crossbar over the same 1 KB window: each lane picks any source lane (a 5-bit select per
// lane), the upper bound on a shuffle's network cost. r[dst][lane] ^= r[src][sel[lane]].
`include "lane_common.vh"
module xbar32(
input clk, input rst,
input [2:0] dst, input [2:0] src, input [159:0] sel,
input ld_en, input [4:0] ld_lane, input [31:0] ld_val,
output [31:0] out);
reg [31:0] rf [0:255];
reg [2:0] dst_q, src_q; reg [159:0] sel_q; reg [4:0] ld_lane_q; reg ld_q; reg [31:0] ld_val_q;
integer i, l;
always @(posedge clk) begin
if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end
else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end
end
wire [31:0] s [0:31];
wire [31:0] x [0:31];
genvar g;
generate for (g = 0; g < 32; g = g + 1) begin : xb
assign s[g] = rf[g*8 + src_q];
assign x[g] = s[sel_q[5*g +: 5]];
end endgenerate
always @(posedge clk) begin
if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 6; end
else for (l = 0; l < 32; l = l + 1)
rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ x[l]);
end
assign out = x[0] ^ x[17];
endmodule

View file

@ -0,0 +1,20 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] op = 0, dst = 0, src = 0; reg [4:0] rot = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
lane_arx dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .rot(rot), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, fixed_op, cycles; reg [31:0] acc = 0;
always #500 clk = ~clk;
initial begin
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; rot = $random;
ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,21 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg cfg_en = 0; reg [31:0] cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0; reg [4:0] cfg_r = 0;
reg ld_en = 0; reg [31:0] ld_val = 0; wire [31:0] out;
lane_fold dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .cfg_en(cfg_en), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, cycles; reg [31:0] acc = 0;
always #750 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
// one era draw: the constants are written once and then static, as on the chip
@(negedge clk); cfg_en = 1; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 32'h3; cfg_mask = 32'h0fffffff;
@(negedge clk); cfg_en = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
dst = $random; src = $random; ld_en = (($random & 3) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,18 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0, src2 = 0; reg [7:0] lut = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
lane_lop3 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .src2(src2), .lut(lut), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, cycles; reg [31:0] acc = 0;
always #500 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
dst = $random; src = $random; src2 = $random; lut = $random; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,20 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
lane_mul dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, fixed_op, cycles; reg [31:0] acc = 0;
always #750 clk = ~clk;
initial begin
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random;
ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,18 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [15:0] sel = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
lane_prmt dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_val(ld_val), .out(out));
integer n, cycles; reg [31:0] acc = 0;
always #500 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
dst = $random; src = $random; sel = $random & 16'h7777; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,17 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [10:0] raddr = 0, waddr = 0; reg we = 0; reg [31:0] wdata = 0; wire [31:0] rdata;
scratch8k dut(.clk(clk), .rst(rst), .raddr(raddr), .we(we), .waddr(waddr), .wdata(wdata), .rdata(rdata));
integer n, cycles; reg [31:0] acc = 0;
always #750 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
raddr = $random; waddr = $random; we = (($random & 7) == 0); wdata = $random; acc = acc ^ rdata;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,18 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [4:0] m = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
shfl32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .m(m), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out));
integer n, cycles; reg [31:0] acc = 0;
always #500 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
dst = $random; src = $random; m = $random; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,18 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1, clr = 0; reg [511:0] a_in = 0, b_in = 0; reg [2:0] sel_r = 0, sel_c = 0; wire [31:0] out;
tile8 dut(.clk(clk), .rst(rst), .clr(clr), .a_in(a_in), .b_in(b_in), .sel_r(sel_r), .sel_c(sel_c), .out(out));
integer n, j, cycles; reg [31:0] acc = 0;
always #1000 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
for (j = 0; j < 16; j = j + 1) begin a_in[j*32 +: 32] = $random; b_in[j*32 +: 32] = $random; end
clr = (($random & 63) == 0); sel_r = $random; sel_c = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule

View file

@ -0,0 +1,18 @@
`timescale 1ps/1ps
module tb;
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [159:0] sel = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0;
wire [31:0] out;
xbar32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out));
integer n, cycles; reg [31:0] acc = 0;
always #500 clk = ~clk;
initial begin
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
repeat (4) @(negedge clk); rst = 0;
for (n = 0; n < cycles; n = n + 1) begin
@(negedge clk);
dst = $random; src = $random; sel = {$random, $random, $random, $random, $random}; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out;
end
$display("CHECKSUM %08x", acc); $finish;
end
endmodule