chip-model: the shadow-k floor lane's RTL, testbenches and ORFS flow (ASAP7) for every class v6 mixer family
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
7618e72981
commit
9009ed0f40
43 changed files with 869 additions and 0 deletions
60
tools/chip-model/rtl/Makefile
Normal file
60
tools/chip-model/rtl/Makefile
Normal file
|
|
@ -0,0 +1,60 @@
|
|||
# Shadow-k floor lane: RTL -> ASAP7 (Yosys + OpenROAD, the ORFS docker image) -> pJ per op from a
|
||||
# random-input gate-level simulation. Reproduces every row of docs/analysis/class-v6/floor/shadow-k.md.
|
||||
#
|
||||
# make row-arx synthesise, place, route, simulate and report one family
|
||||
# make rows every family (serially; use -j for parallel on the box)
|
||||
# make table collect every power log into table.md and table.csv
|
||||
#
|
||||
# Runs on build-4 or build-3 (docker, the build user in the docker group). Never on the Mac.
|
||||
# Images: openroad/orfs:latest (yosys 0.68, OpenROAD) and orfs-sim:latest (the same plus iverilog,
|
||||
# built once with: docker run --name b -u root openroad/orfs:latest bash -c 'apt-get update && apt-get install -y iverilog' && docker commit b orfs-sim:latest).
|
||||
|
||||
WORK ?= $(abspath .)
|
||||
ORFS_IMG ?= openroad/orfs:latest
|
||||
SIM_IMG ?= orfs-sim:latest
|
||||
UID_GID := $(shell id -u):$(shell id -g)
|
||||
DOCKER := docker run --rm -u $(UID_GID) -e HOME=/tmp -v $(WORK):/work
|
||||
ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG)
|
||||
SIM := $(DOCKER) -w /work $(SIM_IMG)
|
||||
|
||||
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile
|
||||
top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt)
|
||||
|
||||
# per-family simulation tags (the op field fixed per row where the family has several ops)
|
||||
SIMS_arx := mix add:+op=0 sub:+op=1 xor:+op=2 or:+op=3 rotl:+op=4 rotr:+op=5
|
||||
SIMS_mul := mix mul:+op=0 mulhi:+op=1 mad:+op=2
|
||||
SIMS_prmt := mix
|
||||
SIMS_lop3 := mix
|
||||
SIMS_fold := mix
|
||||
SIMS_shfl := mix
|
||||
SIMS_xbar := mix
|
||||
SIMS_scratch := mix
|
||||
SIMS_tile := mix
|
||||
|
||||
.PHONY: rows table clean
|
||||
|
||||
rows: $(addprefix row-,$(DESIGNS))
|
||||
|
||||
row-%: power-%
|
||||
@echo "row $* done: $(WORK)/out/$*/logs/asap7/$*/base/power.log"
|
||||
|
||||
flow-%:
|
||||
mkdir -p out/$* logs
|
||||
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk 2>&1 | tee logs/flow-$*.log
|
||||
test -f out/$*/results/asap7/$*/base/6_final.v
|
||||
|
||||
sim-%: flow-%
|
||||
mkdir -p sim/$*
|
||||
$(SIM) bash /work/flow/gl2sim.sh $* $(call top,$*) 2>&1 | tee logs/gl2sim-$*.log
|
||||
for t in $(SIMS_$*); do tag=$${t%%:*}; args=$${t#*:}; [ "$$args" = "$$t" ] && args=""; \
|
||||
$(SIM) bash /work/flow/sim.sh $* $$tag $$args 2>&1 | tee logs/sim-$*-$$tag.log; done
|
||||
|
||||
power-%: sim-%
|
||||
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power run 2>&1 | tee logs/power-$*.log
|
||||
|
||||
table:
|
||||
python3 flow/collect.py $(WORK) > table.md
|
||||
@echo wrote table.md and table.csv
|
||||
|
||||
clean:
|
||||
rm -rf out sim logs table.md table.csv
|
||||
15
tools/chip-model/rtl/flow/arx.mk
Normal file
15
tools/chip-model/rtl/flow/arx.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the arx shadow-core family (top lane_arx), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_arx
|
||||
export DESIGN_NICKNAME = arx
|
||||
export VERILOG_FILES = /work/rtl/lane_arx.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/arx.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/arx
|
||||
|
||||
10
tools/chip-model/rtl/flow/arx.sdc
Normal file
10
tools/chip-model/rtl/flow/arx.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_arx
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
72
tools/chip-model/rtl/flow/collect.py
Normal file
72
tools/chip-model/rtl/flow/collect.py
Normal file
|
|
@ -0,0 +1,72 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Collect the ORFS power logs into the shadow-k table: pJ per op per family at ASAP7 (TC corner),
|
||||
scaled to N5, N3 and N2, against the 5090's measured pJ per counted op (counter-asic-4-research.md 15.1a).
|
||||
Usage: collect.py <work dir> (writes table.csv beside, prints table.md)."""
|
||||
import re, sys, os, csv
|
||||
|
||||
work = sys.argv[1] if len(sys.argv) > 1 else '.'
|
||||
# ops per cycle per design (the per-op divisor) and the GPU row each family is read against
|
||||
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512}
|
||||
# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a
|
||||
GPU = {
|
||||
'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2),
|
||||
'arx:rotl': (11.3, 6.2), 'arx:rotr': (11.3, 6.2),
|
||||
'mul:mix': (13.9, 8.3), 'mul:mul': (13.9, 8.3), 'mul:mad': (13.9, 8.3), 'mul:mulhi': (39.6, 21.0),
|
||||
'prmt:mix': (22.3, 11.5), 'lop3:mix': (24.1, 13.0),
|
||||
'fold:mix': (13.9, 8.3), # the fold is a multiply plus a rotate and masks: read against int_mul
|
||||
'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4),
|
||||
'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed)
|
||||
'tile:mix': (4.1, 2.2), # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
|
||||
}
|
||||
# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed:
|
||||
# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent),
|
||||
# N3E -> N2 x0.72 (TSMC: 25 to 30 percent). Sources in shadow-k.md section 3.
|
||||
SCALE = {'ASAP7': 1.0, 'N5': 0.70, 'N3': 0.70 * 0.72, 'N2': 0.70 * 0.72 * 0.72}
|
||||
|
||||
def parse_log(path):
|
||||
txt = open(path, errors='replace').read()
|
||||
period = None
|
||||
m = re.search(r'FLOORK clock_period_ps ([\d.]+)', txt); period = float(m.group(1)) if m else None
|
||||
cells = re.search(r'FLOORK cells (\d+)', txt); cells = int(cells.group(1)) if cells else None
|
||||
rows = {}
|
||||
for sec in re.split(r'FLOORK === ', txt)[1:]:
|
||||
head, body = sec.split(' ===', 1)
|
||||
tag = head.strip().replace('POWER_VCD ', 'vcd:').replace('POWER_PROPAGATED_0.5', 'prop')
|
||||
# OpenSTA report_power: "Total <int> <sw> <leak> <total> 100.0%" in watts
|
||||
tm = re.search(r'^Total\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)', body, re.M)
|
||||
ann = re.search(r'(\d+)\s*\(\s*([\d.]+)%\)\s*annotated', body)
|
||||
if tm:
|
||||
rows[tag] = dict(internal=float(tm.group(1)), switching=float(tm.group(2)), leakage=float(tm.group(3)),
|
||||
total=float(tm.group(4)), annotated=ann.group(2) if ann else '')
|
||||
return period, cells, rows
|
||||
|
||||
out = []
|
||||
for d in OPS:
|
||||
log = os.path.join(work, 'out', d, 'logs', 'asap7', d, 'base', 'power.log')
|
||||
if not os.path.exists(log):
|
||||
continue
|
||||
period, cells, rows = parse_log(log)
|
||||
for tag, r in rows.items():
|
||||
sub = tag.split(':')[1] if ':' in tag else 'prop'
|
||||
key = f'{d}:{sub}' if sub != 'prop' else f'{d}:mix'
|
||||
gpu = GPU.get(key, (None, None))
|
||||
pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ
|
||||
pj_dyn = (r['internal'] + r['switching']) * period / OPS[d]
|
||||
row = dict(family=d, sim=tag, period_ps=period, cells=cells, ops_per_cycle=OPS[d],
|
||||
total_W=r['total'], leak_W=r['leakage'], annotated_pct=r['annotated'],
|
||||
pJ_asap7=pj, pJ_dyn_asap7=pj_dyn)
|
||||
for node, s in SCALE.items():
|
||||
row[f'pJ_{node}'] = pj * s
|
||||
row[f'k_{node}_lock'] = (pj * s / gpu[1]) if gpu[1] else None
|
||||
row[f'k_{node}_unlocked'] = (pj * s / gpu[0]) if gpu[0] else None
|
||||
row['gpu_pJ_unlocked'] = gpu[0]; row['gpu_pJ_lock'] = gpu[1]
|
||||
out.append(row)
|
||||
|
||||
if out:
|
||||
with open(os.path.join(work, 'table.csv'), 'w', newline='') as f:
|
||||
w = csv.DictWriter(f, fieldnames=list(out[0].keys())); w.writeheader(); w.writerows(out)
|
||||
print('| Family | Sim | Period ps | Cells | Ops/cycle | Total W | Leak W | VCD annotated | pJ/op ASAP7 | pJ/op N5 | pJ/op N3 | pJ/op N2 | 5090 pJ/op unlocked / lock | k at N3 (lock) | k at N2 (lock) | k at N3 (unlocked) |')
|
||||
print('|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|')
|
||||
for r in out:
|
||||
f = lambda v, n=3: ('' if v is None else (f'{v:.{n}g}' if isinstance(v, float) else str(v)))
|
||||
print(f"| {r['family']} | {r['sim']} | {f(r['period_ps'])} | {r['cells']} | {r['ops_per_cycle']} | {f(r['total_W'])} | {f(r['leak_W'])} | {r['annotated_pct']} | {f(r['pJ_asap7'])} | {f(r['pJ_N5'])} | {f(r['pJ_N3'])} | {f(r['pJ_N2'])} | {f(r['gpu_pJ_unlocked'])} / {f(r['gpu_pJ_lock'])} | {f(r['k_N3_lock'])} | {f(r['k_N2_lock'])} | {f(r['k_N3_unlocked'])} |")
|
||||
9
tools/chip-model/rtl/flow/designs.txt
Normal file
9
tools/chip-model/rtl/flow/designs.txt
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
arx lane_arx 1000
|
||||
mul lane_mul 1500
|
||||
prmt lane_prmt 1000
|
||||
lop3 lane_lop3 1000
|
||||
fold lane_fold 1500
|
||||
shfl shfl32 1000
|
||||
xbar xbar32 1000
|
||||
scratch scratch8k 1500
|
||||
tile tile8 2000
|
||||
15
tools/chip-model/rtl/flow/fold.mk
Normal file
15
tools/chip-model/rtl/flow/fold.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the fold shadow-core family (top lane_fold), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_fold
|
||||
export DESIGN_NICKNAME = fold
|
||||
export VERILOG_FILES = /work/rtl/lane_fold.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/fold.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/fold
|
||||
|
||||
10
tools/chip-model/rtl/flow/fold.sdc
Normal file
10
tools/chip-model/rtl/flow/fold.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_fold
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
21
tools/chip-model/rtl/flow/gl2sim.sh
Executable file
21
tools/chip-model/rtl/flow/gl2sim.sh
Executable file
|
|
@ -0,0 +1,21 @@
|
|||
#!/usr/bin/env bash
|
||||
# gl2sim.sh <design-nickname> <top> : netlist -> sim netlist with cell bodies from the liberty.
|
||||
set -euo pipefail
|
||||
name=$1; top=$2
|
||||
net=/work/out/$name/results/asap7/$name/base/6_final.v
|
||||
lib=/OpenROAD-flow-scripts/flow/platforms/asap7/lib/NLDM
|
||||
out=/work/sim/$name; mkdir -p $out
|
||||
cat > $out/gl2sim.ys <<YS
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_AO_RVT_TT_nldm_211120.lib.gz
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_INVBUF_RVT_TT_nldm_220122.lib.gz
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_OA_RVT_TT_nldm_211120.lib.gz
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_SIMPLE_RVT_TT_nldm_211120.lib.gz
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_SEQ_RVT_TT_nldm_220123.lib
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_DFFHQNH2V2X_RVT_TT_nldm_FAKE.lib
|
||||
read_liberty -ignore_miss_func -ignore_miss_dir -ignore_miss_data_latch $lib/asap7sc7p5t_DFFHQNV2X_RVT_TT_nldm_FAKE.lib
|
||||
read_verilog $net
|
||||
hierarchy -top $top
|
||||
write_verilog -noattr $out/sim_net.v
|
||||
YS
|
||||
yosys -q -l $out/gl2sim.log $out/gl2sim.ys
|
||||
echo "sim netlist: $out/sim_net.v"
|
||||
15
tools/chip-model/rtl/flow/lop3.mk
Normal file
15
tools/chip-model/rtl/flow/lop3.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the lop3 shadow-core family (top lane_lop3), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_lop3
|
||||
export DESIGN_NICKNAME = lop3
|
||||
export VERILOG_FILES = /work/rtl/lane_lop3.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/lop3.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/lop3
|
||||
|
||||
10
tools/chip-model/rtl/flow/lop3.sdc
Normal file
10
tools/chip-model/rtl/flow/lop3.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_lop3
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/mul.mk
Normal file
15
tools/chip-model/rtl/flow/mul.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the mul shadow-core family (top lane_mul), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_mul
|
||||
export DESIGN_NICKNAME = mul
|
||||
export VERILOG_FILES = /work/rtl/lane_mul.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/mul.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/mul
|
||||
|
||||
10
tools/chip-model/rtl/flow/mul.sdc
Normal file
10
tools/chip-model/rtl/flow/mul.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_mul
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
23
tools/chip-model/rtl/flow/power.tcl
Normal file
23
tools/chip-model/rtl/flow/power.tcl
Normal file
|
|
@ -0,0 +1,23 @@
|
|||
# Power of the routed design under (a) the VCD of the random-input gate-level simulation and
|
||||
# (b) a propagated 0.5 input activity, both at the TC corner (0.70 V, 0 C on ASAP7).
|
||||
# Run through ORFS: make DESIGN_CONFIG=/work/flow/<name>.mk RUN_SCRIPT=/work/flow/power.tcl run
|
||||
source $::env(SCRIPTS_DIR)/load.tcl
|
||||
load_design 6_final.odb 6_final.sdc
|
||||
set spef $::env(RESULTS_DIR)/6_final.spef
|
||||
if { [file exists $spef] } { read_spef $spef } else { estimate_parasitics -global_routing }
|
||||
puts "FLOORK clock_period_ps [expr [get_property [lindex [all_clocks] 0] period]]"
|
||||
puts "FLOORK cells [llength [get_cells *]]"
|
||||
report_tns
|
||||
report_wns
|
||||
puts "FLOORK === POWER_PROPAGATED_0.5 ==="
|
||||
set_power_activity -input -activity 0.5 -duty 0.5
|
||||
set_power_activity -input_port rst -activity 0 -duty 0
|
||||
report_power
|
||||
foreach vcd [glob -nocomplain /work/sim/$::env(DESIGN_NICKNAME)/*.vcd] {
|
||||
set tag [file rootname [file tail $vcd]]
|
||||
puts "FLOORK === POWER_VCD $tag ==="
|
||||
read_vcd -scope tb/dut $vcd
|
||||
if { [info commands report_activity_annotation] != "" } { report_activity_annotation }
|
||||
report_power
|
||||
}
|
||||
puts "FLOORK done"
|
||||
15
tools/chip-model/rtl/flow/prmt.mk
Normal file
15
tools/chip-model/rtl/flow/prmt.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the prmt shadow-core family (top lane_prmt), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = lane_prmt
|
||||
export DESIGN_NICKNAME = prmt
|
||||
export VERILOG_FILES = /work/rtl/lane_prmt.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/prmt.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/prmt
|
||||
|
||||
10
tools/chip-model/rtl/flow/prmt.sdc
Normal file
10
tools/chip-model/rtl/flow/prmt.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design lane_prmt
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/scratch.mk
Normal file
15
tools/chip-model/rtl/flow/scratch.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the scratch shadow-core family (top scratch8k), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = scratch8k
|
||||
export DESIGN_NICKNAME = scratch
|
||||
export VERILOG_FILES = /work/rtl/scratch8k.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/scratch.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/scratch
|
||||
|
||||
10
tools/chip-model/rtl/flow/scratch.sdc
Normal file
10
tools/chip-model/rtl/flow/scratch.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design scratch8k
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/shfl.mk
Normal file
15
tools/chip-model/rtl/flow/shfl.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the shfl shadow-core family (top shfl32), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = shfl32
|
||||
export DESIGN_NICKNAME = shfl
|
||||
export VERILOG_FILES = /work/rtl/shfl32.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/shfl.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/shfl
|
||||
|
||||
10
tools/chip-model/rtl/flow/shfl.sdc
Normal file
10
tools/chip-model/rtl/flow/shfl.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design shfl32
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
9
tools/chip-model/rtl/flow/sim.sh
Executable file
9
tools/chip-model/rtl/flow/sim.sh
Executable file
|
|
@ -0,0 +1,9 @@
|
|||
#!/usr/bin/env bash
|
||||
# sim.sh <design-nickname> <tag> [plusargs...] : gate-level random-input simulation -> /work/sim/<name>/<tag>.vcd
|
||||
set -euo pipefail
|
||||
name=$1; tag=$2; shift 2
|
||||
out=/work/sim/$name
|
||||
simcells=$(yosys-config --datdir)/simcells.v
|
||||
iverilog -g2005 -o $out/sim_$tag $out/sim_net.v /work/tb/tb_$(sed -n "s/^$name \([^ ]*\) .*/\1/p" /work/flow/designs.txt).v $simcells
|
||||
( cd $out && vvp -n sim_$tag "$@" | tee sim_$tag.log && mv dump.vcd $tag.vcd )
|
||||
ls -la $out/$tag.vcd
|
||||
15
tools/chip-model/rtl/flow/tile.mk
Normal file
15
tools/chip-model/rtl/flow/tile.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the tile shadow-core family (top tile8), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = tile8
|
||||
export DESIGN_NICKNAME = tile
|
||||
export VERILOG_FILES = /work/rtl/tile8.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/tile.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/tile
|
||||
|
||||
10
tools/chip-model/rtl/flow/tile.sdc
Normal file
10
tools/chip-model/rtl/flow/tile.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design tile8
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 2000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/xbar.mk
Normal file
15
tools/chip-model/rtl/flow/xbar.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the xbar shadow-core family (top xbar32), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = xbar32
|
||||
export DESIGN_NICKNAME = xbar
|
||||
export VERILOG_FILES = /work/rtl/xbar32.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/xbar.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/xbar
|
||||
|
||||
10
tools/chip-model/rtl/flow/xbar.sdc
Normal file
10
tools/chip-model/rtl/flow/xbar.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design xbar32
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1000
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
38
tools/chip-model/rtl/rtl/lane_arx.v
Normal file
38
tools/chip-model/rtl/rtl/lane_arx.v
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
// 32-bit int ARX lane: add, sub, xor, or, rotl by immediate, rotr by register (the class v4/v6 families
|
||||
// add, sub, xor, or, rotl, rotr). op: 0 add 1 sub 2 xor 3 or 4 rotl-imm 5 rotr-var 6 add 7 xor.
|
||||
`include "lane_common.vh"
|
||||
module lane_arx(
|
||||
input clk, input rst,
|
||||
input [2:0] op, input [2:0] dst, input [2:0] src, input [4:0] rot,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [2:0] op_q, dst_q, src_q; reg [4:0] rot_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; rot_q <= 0; ld_q <= 0; ld_val_q <= 0; end
|
||||
else begin op_q <= op; dst_q <= dst; src_q <= src; rot_q <= rot; ld_q <= ld_en; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] d = rf[dst_q];
|
||||
wire [31:0] s = rf[src_q];
|
||||
wire [4:0] rn = (rot_q == 5'd0) ? 5'd1 : rot_q; // rotate by 1..31
|
||||
wire [4:0] sn = (s[4:0] == 5'd0) ? 5'd1 : s[4:0];
|
||||
reg [31:0] res;
|
||||
always @* begin
|
||||
case (op_q)
|
||||
3'd0: res = d + s;
|
||||
3'd1: res = d - s;
|
||||
3'd2: res = d ^ s;
|
||||
3'd3: res = d | s;
|
||||
3'd4: res = `ROTL32(d, rn);
|
||||
3'd5: res = `ROTR32(d, sn);
|
||||
3'd6: res = d + s;
|
||||
default: res = d ^ s;
|
||||
endcase
|
||||
end
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : res;
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
5
tools/chip-model/rtl/rtl/lane_common.vh
Normal file
5
tools/chip-model/rtl/rtl/lane_common.vh
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
// Shared helpers for the shadow-core lanes (class v5/v6 mixer draw space).
|
||||
// A lane = an instruction register, an 8 x 32-bit register window (flops), two or three
|
||||
// read ports through muxes, one functional unit, one write port. One op per cycle.
|
||||
`define ROTL32(x, n) (((x) << (n)) | ((x) >> (32 - (n))))
|
||||
`define ROTR32(x, n) (((x) >> (n)) | ((x) << (32 - (n))))
|
||||
32
tools/chip-model/rtl/rtl/lane_fold.v
Normal file
32
tools/chip-model/rtl/rtl/lane_fold.v
Normal file
|
|
@ -0,0 +1,32 @@
|
|||
// The index fold (class v6 layer 1, the ring-A rule): idx = ((rotl(x * M, R) & WM) | OFF) & MASK.
|
||||
// M, R, WM, OFF and MASK are the era's constants, held in registers (loaded by cfg_en, then static).
|
||||
// One fold per cycle on a register read; the index is written back to the lane's address register.
|
||||
`include "lane_common.vh"
|
||||
module lane_fold(
|
||||
input clk, input rst,
|
||||
input [2:0] dst, input [2:0] src,
|
||||
input cfg_en, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [2:0] dst_q, src_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin dst_q <= 0; src_q <= 0; ld_q <= 0; ld_val_q <= 0; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 32'h3; mask_q <= 32'h0fffffff; end
|
||||
else begin
|
||||
dst_q <= dst; src_q <= src; ld_q <= ld_en; ld_val_q <= ld_val;
|
||||
if (cfg_en) begin m_q <= cfg_m | 32'h1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end
|
||||
end
|
||||
end
|
||||
wire [31:0] x = rf[src_q];
|
||||
wire [31:0] prod = x * m_q;
|
||||
wire [4:0] rn = (r_q == 5'd0) ? 5'd1 : r_q;
|
||||
wire [31:0] y = `ROTL32(prod, rn);
|
||||
wire [31:0] res = ((y & wm_q) | off_q) & mask_q;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 4; end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : (res ^ rf[dst_q]);
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
25
tools/chip-model/rtl/rtl/lane_lop3.v
Normal file
25
tools/chip-model/rtl/rtl/lane_lop3.v
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
// Three-input logic lane (LOP3): an 8-bit truth table over (d, s, s2), bitwise.
|
||||
`include "lane_common.vh"
|
||||
module lane_lop3(
|
||||
input clk, input rst,
|
||||
input [2:0] dst, input [2:0] src, input [2:0] src2, input [7:0] lut,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [2:0] dst_q, src_q, src2_q; reg [7:0] lut_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i, b;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin dst_q <= 0; src_q <= 0; src2_q <= 0; lut_q <= 0; ld_q <= 0; ld_val_q <= 0; end
|
||||
else begin dst_q <= dst; src_q <= src; src2_q <= src2; lut_q <= lut; ld_q <= ld_en; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] d = rf[dst_q];
|
||||
wire [31:0] s = rf[src_q];
|
||||
wire [31:0] s2 = rf[src2_q];
|
||||
reg [31:0] res;
|
||||
always @* for (b = 0; b < 32; b = b + 1) res[b] = lut_q[{d[b], s[b], s2[b]}];
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 3; end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : res;
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
36
tools/chip-model/rtl/rtl/lane_mul.v
Normal file
36
tools/chip-model/rtl/rtl/lane_mul.v
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
// 32 x 32 multiplier lane: mul (low 32), mulhi (high 32 of the 64-bit product), mad (src*src2 + dst, low 32).
|
||||
// op: 0 mul 1 mulhi 2 mad 3 mul. The mad reads three registers.
|
||||
`include "lane_common.vh"
|
||||
module lane_mul(
|
||||
input clk, input rst,
|
||||
input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end
|
||||
else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] d = rf[dst_q];
|
||||
wire [31:0] s = rf[src_q];
|
||||
wire [31:0] s2 = rf[src2_q];
|
||||
wire mad = (op_q == 2'd2);
|
||||
wire [31:0] ma = mad ? s : d;
|
||||
wire [31:0] mb = mad ? s2 : s;
|
||||
wire [63:0] p = ma * mb;
|
||||
reg [31:0] res;
|
||||
always @* begin
|
||||
case (op_q)
|
||||
2'd1: res = p[63:32];
|
||||
2'd2: res = p[31:0] + d;
|
||||
default: res = p[31:0];
|
||||
endcase
|
||||
end
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 1; end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : res;
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
29
tools/chip-model/rtl/rtl/lane_prmt.v
Normal file
29
tools/chip-model/rtl/rtl/lane_prmt.v
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
// Byte-permute lane (PRMT): four output bytes, each one of the eight bytes of {s, d}, by a 16-bit selector
|
||||
// (4 bits per output byte; bit 3 = sign-replicate as on the card).
|
||||
`include "lane_common.vh"
|
||||
module lane_prmt(
|
||||
input clk, input rst,
|
||||
input [2:0] dst, input [2:0] src, input [15:0] sel,
|
||||
input ld_en, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:7];
|
||||
reg [2:0] dst_q, src_q; reg [15:0] sel_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_val_q <= 0; end
|
||||
else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] d = rf[dst_q];
|
||||
wire [31:0] s = rf[src_q];
|
||||
wire [63:0] bytes = {s, d};
|
||||
function [7:0] pick; input [63:0] b; input [3:0] k;
|
||||
reg [7:0] v;
|
||||
begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end
|
||||
endfunction
|
||||
wire [31:0] res = {pick(bytes, sel_q[15:12]), pick(bytes, sel_q[11:8]), pick(bytes, sel_q[7:4]), pick(bytes, sel_q[3:0])};
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 2; end
|
||||
else rf[dst_q] <= ld_q ? ld_val_q : res;
|
||||
end
|
||||
assign out = res;
|
||||
endmodule
|
||||
19
tools/chip-model/rtl/rtl/scratch8k.v
Normal file
19
tools/chip-model/rtl/rtl/scratch8k.v
Normal file
|
|
@ -0,0 +1,19 @@
|
|||
// 8 KB random-read scratch (2,048 x 32-bit) as a flop array: one random read per cycle (the op), a write on
|
||||
// one cycle in eight. A flop array is the pessimistic form of the chip's L1; an SRAM macro reads lower.
|
||||
module scratch8k(
|
||||
input clk, input rst,
|
||||
input [10:0] raddr, input we, input [10:0] waddr, input [31:0] wdata,
|
||||
output reg [31:0] rdata);
|
||||
reg [31:0] mem [0:2047];
|
||||
reg [10:0] raddr_q, waddr_q; reg we_q; reg [31:0] wdata_q;
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin raddr_q <= 0; waddr_q <= 0; we_q <= 0; wdata_q <= 0; end
|
||||
else begin raddr_q <= raddr; waddr_q <= waddr; we_q <= we; wdata_q <= wdata; end
|
||||
end
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 2048; i = i + 1) mem[i] <= 32'h9e3779b9 * (i + 1); end
|
||||
else if (we_q) mem[waddr_q] <= wdata_q;
|
||||
end
|
||||
always @(posedge clk) rdata <= rst ? 32'd0 : mem[raddr_q];
|
||||
endmodule
|
||||
36
tools/chip-model/rtl/rtl/shfl32.v
Normal file
36
tools/chip-model/rtl/rtl/shfl32.v
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
// 32-lane xor-mask shuffle over a 1 KB register window (32 lanes x 8 x 32 bits):
|
||||
// r[dst][lane] ^= r[src][lane ^ m] for every lane, one instruction per cycle (32 lane-ops).
|
||||
// The network is a 5-stage butterfly (one 2:1 mux per bit per stage).
|
||||
`include "lane_common.vh"
|
||||
module shfl32(
|
||||
input clk, input rst,
|
||||
input [2:0] dst, input [2:0] src, input [4:0] m,
|
||||
input ld_en, input [4:0] ld_lane, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:255]; // rf[lane*8 + reg]
|
||||
reg [2:0] dst_q, src_q; reg [4:0] m_q, ld_lane_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i, l;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin dst_q <= 0; src_q <= 0; m_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end
|
||||
else begin dst_q <= dst; src_q <= src; m_q <= m; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end
|
||||
end
|
||||
// explicit butterfly
|
||||
wire [31:0] b0 [0:31]; wire [31:0] b1 [0:31]; wire [31:0] b2 [0:31]; wire [31:0] b3 [0:31]; wire [31:0] b4 [0:31]; wire [31:0] b5 [0:31];
|
||||
genvar g;
|
||||
generate for (g = 0; g < 32; g = g + 1) begin : bf
|
||||
assign b0[g] = rf[g*8 + src_q];
|
||||
assign b1[g] = m_q[0] ? b0[g ^ 1] : b0[g];
|
||||
assign b2[g] = m_q[1] ? b1[g ^ 2] : b1[g];
|
||||
assign b3[g] = m_q[2] ? b2[g ^ 4] : b2[g];
|
||||
assign b4[g] = m_q[3] ? b3[g ^ 8] : b3[g];
|
||||
assign b5[g] = m_q[4] ? b4[g ^ 16] : b4[g];
|
||||
end endgenerate
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 5; end
|
||||
else begin
|
||||
for (l = 0; l < 32; l = l + 1)
|
||||
rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ b5[l]);
|
||||
end
|
||||
end
|
||||
assign out = b5[0] ^ b5[17];
|
||||
endmodule
|
||||
33
tools/chip-model/rtl/rtl/tile8.v
Normal file
33
tools/chip-model/rtl/rtl/tile8.v
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
// int8 8x8x8 tile multiply: C[8][8] (int32) += A[8][8] (int8) x B[8][8] (int8): 512 MACs per cycle.
|
||||
// A and B are loaded from the inputs every cycle; the accumulators clear on clr.
|
||||
module tile8(
|
||||
input clk, input rst, input clr,
|
||||
input [511:0] a_in, input [511:0] b_in,
|
||||
input [2:0] sel_r, input [2:0] sel_c,
|
||||
output [31:0] out);
|
||||
reg [511:0] a_q, b_q; reg clr_q; reg [2:0] sr_q, sc_q;
|
||||
reg [31:0] c [0:63];
|
||||
integer i;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin a_q <= 0; b_q <= 0; clr_q <= 1; sr_q <= 0; sc_q <= 0; end
|
||||
else begin a_q <= a_in; b_q <= b_in; clr_q <= clr; sr_q <= sel_r; sc_q <= sel_c; end
|
||||
end
|
||||
genvar r, cc, k;
|
||||
generate for (r = 0; r < 8; r = r + 1) begin : row
|
||||
for (cc = 0; cc < 8; cc = cc + 1) begin : col
|
||||
wire signed [19:0] dot;
|
||||
wire signed [15:0] p [0:7];
|
||||
for (k = 0; k < 8; k = k + 1) begin : mk
|
||||
wire signed [7:0] av = a_q[(r*8 + k)*8 +: 8];
|
||||
wire signed [7:0] bv = b_q[(k*8 + cc)*8 +: 8];
|
||||
assign p[k] = av * bv;
|
||||
end
|
||||
assign dot = p[0] + p[1] + p[2] + p[3] + p[4] + p[5] + p[6] + p[7];
|
||||
always @(posedge clk) begin
|
||||
if (rst || clr_q) c[r*8 + cc] <= 32'd0;
|
||||
else c[r*8 + cc] <= c[r*8 + cc] + {{12{dot[19]}}, dot};
|
||||
end
|
||||
end
|
||||
end endgenerate
|
||||
assign out = c[{sr_q, sc_q}];
|
||||
endmodule
|
||||
29
tools/chip-model/rtl/rtl/xbar32.v
Normal file
29
tools/chip-model/rtl/rtl/xbar32.v
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
// 32-lane general crossbar over the same 1 KB window: each lane picks any source lane (a 5-bit select per
|
||||
// lane), the upper bound on a shuffle's network cost. r[dst][lane] ^= r[src][sel[lane]].
|
||||
`include "lane_common.vh"
|
||||
module xbar32(
|
||||
input clk, input rst,
|
||||
input [2:0] dst, input [2:0] src, input [159:0] sel,
|
||||
input ld_en, input [4:0] ld_lane, input [31:0] ld_val,
|
||||
output [31:0] out);
|
||||
reg [31:0] rf [0:255];
|
||||
reg [2:0] dst_q, src_q; reg [159:0] sel_q; reg [4:0] ld_lane_q; reg ld_q; reg [31:0] ld_val_q;
|
||||
integer i, l;
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end
|
||||
else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end
|
||||
end
|
||||
wire [31:0] s [0:31];
|
||||
wire [31:0] x [0:31];
|
||||
genvar g;
|
||||
generate for (g = 0; g < 32; g = g + 1) begin : xb
|
||||
assign s[g] = rf[g*8 + src_q];
|
||||
assign x[g] = s[sel_q[5*g +: 5]];
|
||||
end endgenerate
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 6; end
|
||||
else for (l = 0; l < 32; l = l + 1)
|
||||
rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ x[l]);
|
||||
end
|
||||
assign out = x[0] ^ x[17];
|
||||
endmodule
|
||||
20
tools/chip-model/rtl/tb/tb_lane_arx.v
Normal file
20
tools/chip-model/rtl/tb/tb_lane_arx.v
Normal file
|
|
@ -0,0 +1,20 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] op = 0, dst = 0, src = 0; reg [4:0] rot = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
lane_arx dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .rot(rot), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, fixed_op, cycles; reg [31:0] acc = 0;
|
||||
always #500 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; rot = $random;
|
||||
ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
21
tools/chip-model/rtl/tb/tb_lane_fold.v
Normal file
21
tools/chip-model/rtl/tb/tb_lane_fold.v
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg cfg_en = 0; reg [31:0] cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0; reg [4:0] cfg_r = 0;
|
||||
reg ld_en = 0; reg [31:0] ld_val = 0; wire [31:0] out;
|
||||
lane_fold dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .cfg_en(cfg_en), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #750 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
// one era draw: the constants are written once and then static, as on the chip
|
||||
@(negedge clk); cfg_en = 1; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 32'h3; cfg_mask = 32'h0fffffff;
|
||||
@(negedge clk); cfg_en = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
dst = $random; src = $random; ld_en = (($random & 3) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
18
tools/chip-model/rtl/tb/tb_lane_lop3.v
Normal file
18
tools/chip-model/rtl/tb/tb_lane_lop3.v
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0, src2 = 0; reg [7:0] lut = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
lane_lop3 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .src2(src2), .lut(lut), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #500 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
dst = $random; src = $random; src2 = $random; lut = $random; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
20
tools/chip-model/rtl/tb/tb_lane_mul.v
Normal file
20
tools/chip-model/rtl/tb/tb_lane_mul.v
Normal file
|
|
@ -0,0 +1,20 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
lane_mul dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, fixed_op, cycles; reg [31:0] acc = 0;
|
||||
always #750 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1;
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random;
|
||||
ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
18
tools/chip-model/rtl/tb/tb_lane_prmt.v
Normal file
18
tools/chip-model/rtl/tb/tb_lane_prmt.v
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [15:0] sel = 0; reg ld_en = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
lane_prmt dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_val(ld_val), .out(out));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #500 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
dst = $random; src = $random; sel = $random & 16'h7777; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
17
tools/chip-model/rtl/tb/tb_scratch8k.v
Normal file
17
tools/chip-model/rtl/tb/tb_scratch8k.v
Normal file
|
|
@ -0,0 +1,17 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [10:0] raddr = 0, waddr = 0; reg we = 0; reg [31:0] wdata = 0; wire [31:0] rdata;
|
||||
scratch8k dut(.clk(clk), .rst(rst), .raddr(raddr), .we(we), .waddr(waddr), .wdata(wdata), .rdata(rdata));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #750 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
raddr = $random; waddr = $random; we = (($random & 7) == 0); wdata = $random; acc = acc ^ rdata;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
18
tools/chip-model/rtl/tb/tb_shfl32.v
Normal file
18
tools/chip-model/rtl/tb/tb_shfl32.v
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [4:0] m = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
shfl32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .m(m), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #500 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
dst = $random; src = $random; m = $random; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
18
tools/chip-model/rtl/tb/tb_tile8.v
Normal file
18
tools/chip-model/rtl/tb/tb_tile8.v
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1, clr = 0; reg [511:0] a_in = 0, b_in = 0; reg [2:0] sel_r = 0, sel_c = 0; wire [31:0] out;
|
||||
tile8 dut(.clk(clk), .rst(rst), .clr(clr), .a_in(a_in), .b_in(b_in), .sel_r(sel_r), .sel_c(sel_c), .out(out));
|
||||
integer n, j, cycles; reg [31:0] acc = 0;
|
||||
always #1000 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
for (j = 0; j < 16; j = j + 1) begin a_in[j*32 +: 32] = $random; b_in[j*32 +: 32] = $random; end
|
||||
clr = (($random & 63) == 0); sel_r = $random; sel_c = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
18
tools/chip-model/rtl/tb/tb_xbar32.v
Normal file
18
tools/chip-model/rtl/tb/tb_xbar32.v
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [159:0] sel = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0;
|
||||
wire [31:0] out;
|
||||
xbar32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out));
|
||||
integer n, cycles; reg [31:0] acc = 0;
|
||||
always #500 clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000;
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk);
|
||||
dst = $random; src = $random; sel = {$random, $random, $random, $random, $random}; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
Loading…
Reference in a new issue