diff --git a/tools/chip-model/rtl/Makefile b/tools/chip-model/rtl/Makefile new file mode 100644 index 000000000..05b786486 --- /dev/null +++ b/tools/chip-model/rtl/Makefile @@ -0,0 +1,60 @@ +# Shadow-k floor lane: RTL -> ASAP7 (Yosys + OpenROAD, the ORFS docker image) -> pJ per op from a +# random-input gate-level simulation. Reproduces every row of docs/analysis/class-v6/floor/shadow-k.md. +# +# make row-arx synthesise, place, route, simulate and report one family +# make rows every family (serially; use -j for parallel on the box) +# make table collect every power log into table.md and table.csv +# +# Runs on build-4 or build-3 (docker, the build user in the docker group). Never on the Mac. +# Images: openroad/orfs:latest (yosys 0.68, OpenROAD) and orfs-sim:latest (the same plus iverilog, +# built once with: docker run --name b -u root openroad/orfs:latest bash -c 'apt-get update && apt-get install -y iverilog' && docker commit b orfs-sim:latest). + +WORK ?= $(abspath .) +ORFS_IMG ?= openroad/orfs:latest +SIM_IMG ?= orfs-sim:latest +UID_GID := $(shell id -u):$(shell id -g) +DOCKER := docker run --rm -u $(UID_GID) -e HOME=/tmp -v $(WORK):/work +ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG) +SIM := $(DOCKER) -w /work $(SIM_IMG) + +DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile +top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt) + +# per-family simulation tags (the op field fixed per row where the family has several ops) +SIMS_arx := mix add:+op=0 sub:+op=1 xor:+op=2 or:+op=3 rotl:+op=4 rotr:+op=5 +SIMS_mul := mix mul:+op=0 mulhi:+op=1 mad:+op=2 +SIMS_prmt := mix +SIMS_lop3 := mix +SIMS_fold := mix +SIMS_shfl := mix +SIMS_xbar := mix +SIMS_scratch := mix +SIMS_tile := mix + +.PHONY: rows table clean + +rows: $(addprefix row-,$(DESIGNS)) + +row-%: power-% + @echo "row $* done: $(WORK)/out/$*/logs/asap7/$*/base/power.log" + +flow-%: + mkdir -p out/$* logs + $(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk 2>&1 | tee logs/flow-$*.log + test -f out/$*/results/asap7/$*/base/6_final.v + +sim-%: flow-% + mkdir -p sim/$* + $(SIM) bash /work/flow/gl2sim.sh $* $(call top,$*) 2>&1 | tee logs/gl2sim-$*.log + for t in $(SIMS_$*); do tag=$${t%%:*}; args=$${t#*:}; [ "$$args" = "$$t" ] && args=""; \ + $(SIM) bash /work/flow/sim.sh $* $$tag $$args 2>&1 | tee logs/sim-$*-$$tag.log; done + +power-%: sim-% + $(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power run 2>&1 | tee logs/power-$*.log + +table: + python3 flow/collect.py $(WORK) > table.md + @echo wrote table.md and table.csv + +clean: + rm -rf out sim logs table.md table.csv diff --git a/tools/chip-model/rtl/flow/arx.mk b/tools/chip-model/rtl/flow/arx.mk new file mode 100644 index 000000000..ca1b61a86 --- /dev/null +++ b/tools/chip-model/rtl/flow/arx.mk @@ -0,0 +1,15 @@ +# ORFS design config for the arx shadow-core family (top lane_arx), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = lane_arx +export DESIGN_NICKNAME = arx +export VERILOG_FILES = /work/rtl/lane_arx.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/arx.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/arx + diff --git a/tools/chip-model/rtl/flow/arx.sdc b/tools/chip-model/rtl/flow/arx.sdc new file mode 100644 index 000000000..81f424f33 --- /dev/null +++ b/tools/chip-model/rtl/flow/arx.sdc @@ -0,0 +1,10 @@ +current_design lane_arx +set clk_name core_clock +set clk_port_name clk +set clk_period 1000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/collect.py b/tools/chip-model/rtl/flow/collect.py new file mode 100644 index 000000000..6ed151f21 --- /dev/null +++ b/tools/chip-model/rtl/flow/collect.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Collect the ORFS power logs into the shadow-k table: pJ per op per family at ASAP7 (TC corner), +scaled to N5, N3 and N2, against the 5090's measured pJ per counted op (counter-asic-4-research.md 15.1a). +Usage: collect.py (writes table.csv beside, prints table.md).""" +import re, sys, os, csv + +work = sys.argv[1] if len(sys.argv) > 1 else '.' +# ops per cycle per design (the per-op divisor) and the GPU row each family is read against +OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512} +# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a +GPU = { + 'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2), + 'arx:rotl': (11.3, 6.2), 'arx:rotr': (11.3, 6.2), + 'mul:mix': (13.9, 8.3), 'mul:mul': (13.9, 8.3), 'mul:mad': (13.9, 8.3), 'mul:mulhi': (39.6, 21.0), + 'prmt:mix': (22.3, 11.5), 'lop3:mix': (24.1, 13.0), + 'fold:mix': (13.9, 8.3), # the fold is a multiply plus a rotate and masks: read against int_mul + 'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4), + 'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed) + 'tile:mix': (4.1, 2.2), # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 +} +# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed: +# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent), +# N3E -> N2 x0.72 (TSMC: 25 to 30 percent). Sources in shadow-k.md section 3. +SCALE = {'ASAP7': 1.0, 'N5': 0.70, 'N3': 0.70 * 0.72, 'N2': 0.70 * 0.72 * 0.72} + +def parse_log(path): + txt = open(path, errors='replace').read() + period = None + m = re.search(r'FLOORK clock_period_ps ([\d.]+)', txt); period = float(m.group(1)) if m else None + cells = re.search(r'FLOORK cells (\d+)', txt); cells = int(cells.group(1)) if cells else None + rows = {} + for sec in re.split(r'FLOORK === ', txt)[1:]: + head, body = sec.split(' ===', 1) + tag = head.strip().replace('POWER_VCD ', 'vcd:').replace('POWER_PROPAGATED_0.5', 'prop') + # OpenSTA report_power: "Total 100.0%" in watts + tm = re.search(r'^Total\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)\s+([\d.eE+-]+)', body, re.M) + ann = re.search(r'(\d+)\s*\(\s*([\d.]+)%\)\s*annotated', body) + if tm: + rows[tag] = dict(internal=float(tm.group(1)), switching=float(tm.group(2)), leakage=float(tm.group(3)), + total=float(tm.group(4)), annotated=ann.group(2) if ann else '') + return period, cells, rows + +out = [] +for d in OPS: + log = os.path.join(work, 'out', d, 'logs', 'asap7', d, 'base', 'power.log') + if not os.path.exists(log): + continue + period, cells, rows = parse_log(log) + for tag, r in rows.items(): + sub = tag.split(':')[1] if ':' in tag else 'prop' + key = f'{d}:{sub}' if sub != 'prop' else f'{d}:mix' + gpu = GPU.get(key, (None, None)) + pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ + pj_dyn = (r['internal'] + r['switching']) * period / OPS[d] + row = dict(family=d, sim=tag, period_ps=period, cells=cells, ops_per_cycle=OPS[d], + total_W=r['total'], leak_W=r['leakage'], annotated_pct=r['annotated'], + pJ_asap7=pj, pJ_dyn_asap7=pj_dyn) + for node, s in SCALE.items(): + row[f'pJ_{node}'] = pj * s + row[f'k_{node}_lock'] = (pj * s / gpu[1]) if gpu[1] else None + row[f'k_{node}_unlocked'] = (pj * s / gpu[0]) if gpu[0] else None + row['gpu_pJ_unlocked'] = gpu[0]; row['gpu_pJ_lock'] = gpu[1] + out.append(row) + +if out: + with open(os.path.join(work, 'table.csv'), 'w', newline='') as f: + w = csv.DictWriter(f, fieldnames=list(out[0].keys())); w.writeheader(); w.writerows(out) +print('| Family | Sim | Period ps | Cells | Ops/cycle | Total W | Leak W | VCD annotated | pJ/op ASAP7 | pJ/op N5 | pJ/op N3 | pJ/op N2 | 5090 pJ/op unlocked / lock | k at N3 (lock) | k at N2 (lock) | k at N3 (unlocked) |') +print('|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|') +for r in out: + f = lambda v, n=3: ('' if v is None else (f'{v:.{n}g}' if isinstance(v, float) else str(v))) + print(f"| {r['family']} | {r['sim']} | {f(r['period_ps'])} | {r['cells']} | {r['ops_per_cycle']} | {f(r['total_W'])} | {f(r['leak_W'])} | {r['annotated_pct']} | {f(r['pJ_asap7'])} | {f(r['pJ_N5'])} | {f(r['pJ_N3'])} | {f(r['pJ_N2'])} | {f(r['gpu_pJ_unlocked'])} / {f(r['gpu_pJ_lock'])} | {f(r['k_N3_lock'])} | {f(r['k_N2_lock'])} | {f(r['k_N3_unlocked'])} |") diff --git a/tools/chip-model/rtl/flow/designs.txt b/tools/chip-model/rtl/flow/designs.txt new file mode 100644 index 000000000..133bafae3 --- /dev/null +++ b/tools/chip-model/rtl/flow/designs.txt @@ -0,0 +1,9 @@ +arx lane_arx 1000 +mul lane_mul 1500 +prmt lane_prmt 1000 +lop3 lane_lop3 1000 +fold lane_fold 1500 +shfl shfl32 1000 +xbar xbar32 1000 +scratch scratch8k 1500 +tile tile8 2000 diff --git a/tools/chip-model/rtl/flow/fold.mk b/tools/chip-model/rtl/flow/fold.mk new file mode 100644 index 000000000..d5fe9d932 --- /dev/null +++ b/tools/chip-model/rtl/flow/fold.mk @@ -0,0 +1,15 @@ +# ORFS design config for the fold shadow-core family (top lane_fold), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = lane_fold +export DESIGN_NICKNAME = fold +export VERILOG_FILES = /work/rtl/lane_fold.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/fold.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/fold + diff --git a/tools/chip-model/rtl/flow/fold.sdc b/tools/chip-model/rtl/flow/fold.sdc new file mode 100644 index 000000000..9dad665a3 --- /dev/null +++ b/tools/chip-model/rtl/flow/fold.sdc @@ -0,0 +1,10 @@ +current_design lane_fold +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/gl2sim.sh b/tools/chip-model/rtl/flow/gl2sim.sh new file mode 100755 index 000000000..17c94928f --- /dev/null +++ b/tools/chip-model/rtl/flow/gl2sim.sh @@ -0,0 +1,21 @@ +#!/usr/bin/env bash +# gl2sim.sh : netlist -> sim netlist with cell bodies from the liberty. +set -euo pipefail +name=$1; top=$2 +net=/work/out/$name/results/asap7/$name/base/6_final.v +lib=/OpenROAD-flow-scripts/flow/platforms/asap7/lib/NLDM +out=/work/sim/$name; mkdir -p $out +cat > $out/gl2sim.ys <.mk RUN_SCRIPT=/work/flow/power.tcl run +source $::env(SCRIPTS_DIR)/load.tcl +load_design 6_final.odb 6_final.sdc +set spef $::env(RESULTS_DIR)/6_final.spef +if { [file exists $spef] } { read_spef $spef } else { estimate_parasitics -global_routing } +puts "FLOORK clock_period_ps [expr [get_property [lindex [all_clocks] 0] period]]" +puts "FLOORK cells [llength [get_cells *]]" +report_tns +report_wns +puts "FLOORK === POWER_PROPAGATED_0.5 ===" +set_power_activity -input -activity 0.5 -duty 0.5 +set_power_activity -input_port rst -activity 0 -duty 0 +report_power +foreach vcd [glob -nocomplain /work/sim/$::env(DESIGN_NICKNAME)/*.vcd] { + set tag [file rootname [file tail $vcd]] + puts "FLOORK === POWER_VCD $tag ===" + read_vcd -scope tb/dut $vcd + if { [info commands report_activity_annotation] != "" } { report_activity_annotation } + report_power +} +puts "FLOORK done" diff --git a/tools/chip-model/rtl/flow/prmt.mk b/tools/chip-model/rtl/flow/prmt.mk new file mode 100644 index 000000000..e17ad1bde --- /dev/null +++ b/tools/chip-model/rtl/flow/prmt.mk @@ -0,0 +1,15 @@ +# ORFS design config for the prmt shadow-core family (top lane_prmt), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = lane_prmt +export DESIGN_NICKNAME = prmt +export VERILOG_FILES = /work/rtl/lane_prmt.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/prmt.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/prmt + diff --git a/tools/chip-model/rtl/flow/prmt.sdc b/tools/chip-model/rtl/flow/prmt.sdc new file mode 100644 index 000000000..dabb0c22f --- /dev/null +++ b/tools/chip-model/rtl/flow/prmt.sdc @@ -0,0 +1,10 @@ +current_design lane_prmt +set clk_name core_clock +set clk_port_name clk +set clk_period 1000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/scratch.mk b/tools/chip-model/rtl/flow/scratch.mk new file mode 100644 index 000000000..a8fdb0197 --- /dev/null +++ b/tools/chip-model/rtl/flow/scratch.mk @@ -0,0 +1,15 @@ +# ORFS design config for the scratch shadow-core family (top scratch8k), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = scratch8k +export DESIGN_NICKNAME = scratch +export VERILOG_FILES = /work/rtl/scratch8k.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/scratch.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/scratch + diff --git a/tools/chip-model/rtl/flow/scratch.sdc b/tools/chip-model/rtl/flow/scratch.sdc new file mode 100644 index 000000000..822efb6cb --- /dev/null +++ b/tools/chip-model/rtl/flow/scratch.sdc @@ -0,0 +1,10 @@ +current_design scratch8k +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/shfl.mk b/tools/chip-model/rtl/flow/shfl.mk new file mode 100644 index 000000000..a73d97541 --- /dev/null +++ b/tools/chip-model/rtl/flow/shfl.mk @@ -0,0 +1,15 @@ +# ORFS design config for the shfl shadow-core family (top shfl32), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = shfl32 +export DESIGN_NICKNAME = shfl +export VERILOG_FILES = /work/rtl/shfl32.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/shfl.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/shfl + diff --git a/tools/chip-model/rtl/flow/shfl.sdc b/tools/chip-model/rtl/flow/shfl.sdc new file mode 100644 index 000000000..530484342 --- /dev/null +++ b/tools/chip-model/rtl/flow/shfl.sdc @@ -0,0 +1,10 @@ +current_design shfl32 +set clk_name core_clock +set clk_port_name clk +set clk_period 1000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/sim.sh b/tools/chip-model/rtl/flow/sim.sh new file mode 100755 index 000000000..f7c7cbdd6 --- /dev/null +++ b/tools/chip-model/rtl/flow/sim.sh @@ -0,0 +1,9 @@ +#!/usr/bin/env bash +# sim.sh [plusargs...] : gate-level random-input simulation -> /work/sim//.vcd +set -euo pipefail +name=$1; tag=$2; shift 2 +out=/work/sim/$name +simcells=$(yosys-config --datdir)/simcells.v +iverilog -g2005 -o $out/sim_$tag $out/sim_net.v /work/tb/tb_$(sed -n "s/^$name \([^ ]*\) .*/\1/p" /work/flow/designs.txt).v $simcells +( cd $out && vvp -n sim_$tag "$@" | tee sim_$tag.log && mv dump.vcd $tag.vcd ) +ls -la $out/$tag.vcd diff --git a/tools/chip-model/rtl/flow/tile.mk b/tools/chip-model/rtl/flow/tile.mk new file mode 100644 index 000000000..69f85f03c --- /dev/null +++ b/tools/chip-model/rtl/flow/tile.mk @@ -0,0 +1,15 @@ +# ORFS design config for the tile shadow-core family (top tile8), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = tile8 +export DESIGN_NICKNAME = tile +export VERILOG_FILES = /work/rtl/tile8.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/tile.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/tile + diff --git a/tools/chip-model/rtl/flow/tile.sdc b/tools/chip-model/rtl/flow/tile.sdc new file mode 100644 index 000000000..2bac92473 --- /dev/null +++ b/tools/chip-model/rtl/flow/tile.sdc @@ -0,0 +1,10 @@ +current_design tile8 +set clk_name core_clock +set clk_port_name clk +set clk_period 2000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/xbar.mk b/tools/chip-model/rtl/flow/xbar.mk new file mode 100644 index 000000000..01026b194 --- /dev/null +++ b/tools/chip-model/rtl/flow/xbar.mk @@ -0,0 +1,15 @@ +# ORFS design config for the xbar shadow-core family (top xbar32), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = xbar32 +export DESIGN_NICKNAME = xbar +export VERILOG_FILES = /work/rtl/xbar32.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/xbar.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/xbar + diff --git a/tools/chip-model/rtl/flow/xbar.sdc b/tools/chip-model/rtl/flow/xbar.sdc new file mode 100644 index 000000000..509dd19ff --- /dev/null +++ b/tools/chip-model/rtl/flow/xbar.sdc @@ -0,0 +1,10 @@ +current_design xbar32 +set clk_name core_clock +set clk_port_name clk +set clk_period 1000 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/rtl/lane_arx.v b/tools/chip-model/rtl/rtl/lane_arx.v new file mode 100644 index 000000000..be590f9a1 --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_arx.v @@ -0,0 +1,38 @@ +// 32-bit int ARX lane: add, sub, xor, or, rotl by immediate, rotr by register (the class v4/v6 families +// add, sub, xor, or, rotl, rotr). op: 0 add 1 sub 2 xor 3 or 4 rotl-imm 5 rotr-var 6 add 7 xor. +`include "lane_common.vh" +module lane_arx( + input clk, input rst, + input [2:0] op, input [2:0] dst, input [2:0] src, input [4:0] rot, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [2:0] op_q, dst_q, src_q; reg [4:0] rot_q; reg ld_q; reg [31:0] ld_val_q; + integer i; + always @(posedge clk) begin + if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; rot_q <= 0; ld_q <= 0; ld_val_q <= 0; end + else begin op_q <= op; dst_q <= dst; src_q <= src; rot_q <= rot; ld_q <= ld_en; ld_val_q <= ld_val; end + end + wire [31:0] d = rf[dst_q]; + wire [31:0] s = rf[src_q]; + wire [4:0] rn = (rot_q == 5'd0) ? 5'd1 : rot_q; // rotate by 1..31 + wire [4:0] sn = (s[4:0] == 5'd0) ? 5'd1 : s[4:0]; + reg [31:0] res; + always @* begin + case (op_q) + 3'd0: res = d + s; + 3'd1: res = d - s; + 3'd2: res = d ^ s; + 3'd3: res = d | s; + 3'd4: res = `ROTL32(d, rn); + 3'd5: res = `ROTR32(d, sn); + 3'd6: res = d + s; + default: res = d ^ s; + endcase + end + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end + else rf[dst_q] <= ld_q ? ld_val_q : res; + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/rtl/lane_common.vh b/tools/chip-model/rtl/rtl/lane_common.vh new file mode 100644 index 000000000..c977a0d4c --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_common.vh @@ -0,0 +1,5 @@ +// Shared helpers for the shadow-core lanes (class v5/v6 mixer draw space). +// A lane = an instruction register, an 8 x 32-bit register window (flops), two or three +// read ports through muxes, one functional unit, one write port. One op per cycle. +`define ROTL32(x, n) (((x) << (n)) | ((x) >> (32 - (n)))) +`define ROTR32(x, n) (((x) >> (n)) | ((x) << (32 - (n)))) diff --git a/tools/chip-model/rtl/rtl/lane_fold.v b/tools/chip-model/rtl/rtl/lane_fold.v new file mode 100644 index 000000000..5b1fbc890 --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_fold.v @@ -0,0 +1,32 @@ +// The index fold (class v6 layer 1, the ring-A rule): idx = ((rotl(x * M, R) & WM) | OFF) & MASK. +// M, R, WM, OFF and MASK are the era's constants, held in registers (loaded by cfg_en, then static). +// One fold per cycle on a register read; the index is written back to the lane's address register. +`include "lane_common.vh" +module lane_fold( + input clk, input rst, + input [2:0] dst, input [2:0] src, + input cfg_en, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [2:0] dst_q, src_q; reg ld_q; reg [31:0] ld_val_q; + reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q; + integer i; + always @(posedge clk) begin + if (rst) begin dst_q <= 0; src_q <= 0; ld_q <= 0; ld_val_q <= 0; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 32'h3; mask_q <= 32'h0fffffff; end + else begin + dst_q <= dst; src_q <= src; ld_q <= ld_en; ld_val_q <= ld_val; + if (cfg_en) begin m_q <= cfg_m | 32'h1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end + end + end + wire [31:0] x = rf[src_q]; + wire [31:0] prod = x * m_q; + wire [4:0] rn = (r_q == 5'd0) ? 5'd1 : r_q; + wire [31:0] y = `ROTL32(prod, rn); + wire [31:0] res = ((y & wm_q) | off_q) & mask_q; + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 4; end + else rf[dst_q] <= ld_q ? ld_val_q : (res ^ rf[dst_q]); + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/rtl/lane_lop3.v b/tools/chip-model/rtl/rtl/lane_lop3.v new file mode 100644 index 000000000..68c1c8690 --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_lop3.v @@ -0,0 +1,25 @@ +// Three-input logic lane (LOP3): an 8-bit truth table over (d, s, s2), bitwise. +`include "lane_common.vh" +module lane_lop3( + input clk, input rst, + input [2:0] dst, input [2:0] src, input [2:0] src2, input [7:0] lut, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [2:0] dst_q, src_q, src2_q; reg [7:0] lut_q; reg ld_q; reg [31:0] ld_val_q; + integer i, b; + always @(posedge clk) begin + if (rst) begin dst_q <= 0; src_q <= 0; src2_q <= 0; lut_q <= 0; ld_q <= 0; ld_val_q <= 0; end + else begin dst_q <= dst; src_q <= src; src2_q <= src2; lut_q <= lut; ld_q <= ld_en; ld_val_q <= ld_val; end + end + wire [31:0] d = rf[dst_q]; + wire [31:0] s = rf[src_q]; + wire [31:0] s2 = rf[src2_q]; + reg [31:0] res; + always @* for (b = 0; b < 32; b = b + 1) res[b] = lut_q[{d[b], s[b], s2[b]}]; + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 3; end + else rf[dst_q] <= ld_q ? ld_val_q : res; + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/rtl/lane_mul.v b/tools/chip-model/rtl/rtl/lane_mul.v new file mode 100644 index 000000000..0b8b0736c --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_mul.v @@ -0,0 +1,36 @@ +// 32 x 32 multiplier lane: mul (low 32), mulhi (high 32 of the 64-bit product), mad (src*src2 + dst, low 32). +// op: 0 mul 1 mulhi 2 mad 3 mul. The mad reads three registers. +`include "lane_common.vh" +module lane_mul( + input clk, input rst, + input [1:0] op, input [2:0] dst, input [2:0] src, input [2:0] src2, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [1:0] op_q; reg [2:0] dst_q, src_q, src2_q; reg ld_q; reg [31:0] ld_val_q; + integer i; + always @(posedge clk) begin + if (rst) begin op_q <= 0; dst_q <= 0; src_q <= 0; src2_q <= 0; ld_q <= 0; ld_val_q <= 0; end + else begin op_q <= op; dst_q <= dst; src_q <= src; src2_q <= src2; ld_q <= ld_en; ld_val_q <= ld_val; end + end + wire [31:0] d = rf[dst_q]; + wire [31:0] s = rf[src_q]; + wire [31:0] s2 = rf[src2_q]; + wire mad = (op_q == 2'd2); + wire [31:0] ma = mad ? s : d; + wire [31:0] mb = mad ? s2 : s; + wire [63:0] p = ma * mb; + reg [31:0] res; + always @* begin + case (op_q) + 2'd1: res = p[63:32]; + 2'd2: res = p[31:0] + d; + default: res = p[31:0]; + endcase + end + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 1; end + else rf[dst_q] <= ld_q ? ld_val_q : res; + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/rtl/lane_prmt.v b/tools/chip-model/rtl/rtl/lane_prmt.v new file mode 100644 index 000000000..d8d472f28 --- /dev/null +++ b/tools/chip-model/rtl/rtl/lane_prmt.v @@ -0,0 +1,29 @@ +// Byte-permute lane (PRMT): four output bytes, each one of the eight bytes of {s, d}, by a 16-bit selector +// (4 bits per output byte; bit 3 = sign-replicate as on the card). +`include "lane_common.vh" +module lane_prmt( + input clk, input rst, + input [2:0] dst, input [2:0] src, input [15:0] sel, + input ld_en, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:7]; + reg [2:0] dst_q, src_q; reg [15:0] sel_q; reg ld_q; reg [31:0] ld_val_q; + integer i; + always @(posedge clk) begin + if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_val_q <= 0; end + else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_val_q <= ld_val; end + end + wire [31:0] d = rf[dst_q]; + wire [31:0] s = rf[src_q]; + wire [63:0] bytes = {s, d}; + function [7:0] pick; input [63:0] b; input [3:0] k; + reg [7:0] v; + begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end + endfunction + wire [31:0] res = {pick(bytes, sel_q[15:12]), pick(bytes, sel_q[11:8]), pick(bytes, sel_q[7:4]), pick(bytes, sel_q[3:0])}; + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 8; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 2; end + else rf[dst_q] <= ld_q ? ld_val_q : res; + end + assign out = res; +endmodule diff --git a/tools/chip-model/rtl/rtl/scratch8k.v b/tools/chip-model/rtl/rtl/scratch8k.v new file mode 100644 index 000000000..8df1c1346 --- /dev/null +++ b/tools/chip-model/rtl/rtl/scratch8k.v @@ -0,0 +1,19 @@ +// 8 KB random-read scratch (2,048 x 32-bit) as a flop array: one random read per cycle (the op), a write on +// one cycle in eight. A flop array is the pessimistic form of the chip's L1; an SRAM macro reads lower. +module scratch8k( + input clk, input rst, + input [10:0] raddr, input we, input [10:0] waddr, input [31:0] wdata, + output reg [31:0] rdata); + reg [31:0] mem [0:2047]; + reg [10:0] raddr_q, waddr_q; reg we_q; reg [31:0] wdata_q; + integer i; + always @(posedge clk) begin + if (rst) begin raddr_q <= 0; waddr_q <= 0; we_q <= 0; wdata_q <= 0; end + else begin raddr_q <= raddr; waddr_q <= waddr; we_q <= we; wdata_q <= wdata; end + end + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 2048; i = i + 1) mem[i] <= 32'h9e3779b9 * (i + 1); end + else if (we_q) mem[waddr_q] <= wdata_q; + end + always @(posedge clk) rdata <= rst ? 32'd0 : mem[raddr_q]; +endmodule diff --git a/tools/chip-model/rtl/rtl/shfl32.v b/tools/chip-model/rtl/rtl/shfl32.v new file mode 100644 index 000000000..1c88fbf45 --- /dev/null +++ b/tools/chip-model/rtl/rtl/shfl32.v @@ -0,0 +1,36 @@ +// 32-lane xor-mask shuffle over a 1 KB register window (32 lanes x 8 x 32 bits): +// r[dst][lane] ^= r[src][lane ^ m] for every lane, one instruction per cycle (32 lane-ops). +// The network is a 5-stage butterfly (one 2:1 mux per bit per stage). +`include "lane_common.vh" +module shfl32( + input clk, input rst, + input [2:0] dst, input [2:0] src, input [4:0] m, + input ld_en, input [4:0] ld_lane, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:255]; // rf[lane*8 + reg] + reg [2:0] dst_q, src_q; reg [4:0] m_q, ld_lane_q; reg ld_q; reg [31:0] ld_val_q; + integer i, l; + always @(posedge clk) begin + if (rst) begin dst_q <= 0; src_q <= 0; m_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end + else begin dst_q <= dst; src_q <= src; m_q <= m; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end + end + // explicit butterfly + wire [31:0] b0 [0:31]; wire [31:0] b1 [0:31]; wire [31:0] b2 [0:31]; wire [31:0] b3 [0:31]; wire [31:0] b4 [0:31]; wire [31:0] b5 [0:31]; + genvar g; + generate for (g = 0; g < 32; g = g + 1) begin : bf + assign b0[g] = rf[g*8 + src_q]; + assign b1[g] = m_q[0] ? b0[g ^ 1] : b0[g]; + assign b2[g] = m_q[1] ? b1[g ^ 2] : b1[g]; + assign b3[g] = m_q[2] ? b2[g ^ 4] : b2[g]; + assign b4[g] = m_q[3] ? b3[g ^ 8] : b3[g]; + assign b5[g] = m_q[4] ? b4[g ^ 16] : b4[g]; + end endgenerate + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 5; end + else begin + for (l = 0; l < 32; l = l + 1) + rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ b5[l]); + end + end + assign out = b5[0] ^ b5[17]; +endmodule diff --git a/tools/chip-model/rtl/rtl/tile8.v b/tools/chip-model/rtl/rtl/tile8.v new file mode 100644 index 000000000..5c6b2e94a --- /dev/null +++ b/tools/chip-model/rtl/rtl/tile8.v @@ -0,0 +1,33 @@ +// int8 8x8x8 tile multiply: C[8][8] (int32) += A[8][8] (int8) x B[8][8] (int8): 512 MACs per cycle. +// A and B are loaded from the inputs every cycle; the accumulators clear on clr. +module tile8( + input clk, input rst, input clr, + input [511:0] a_in, input [511:0] b_in, + input [2:0] sel_r, input [2:0] sel_c, + output [31:0] out); + reg [511:0] a_q, b_q; reg clr_q; reg [2:0] sr_q, sc_q; + reg [31:0] c [0:63]; + integer i; + always @(posedge clk) begin + if (rst) begin a_q <= 0; b_q <= 0; clr_q <= 1; sr_q <= 0; sc_q <= 0; end + else begin a_q <= a_in; b_q <= b_in; clr_q <= clr; sr_q <= sel_r; sc_q <= sel_c; end + end + genvar r, cc, k; + generate for (r = 0; r < 8; r = r + 1) begin : row + for (cc = 0; cc < 8; cc = cc + 1) begin : col + wire signed [19:0] dot; + wire signed [15:0] p [0:7]; + for (k = 0; k < 8; k = k + 1) begin : mk + wire signed [7:0] av = a_q[(r*8 + k)*8 +: 8]; + wire signed [7:0] bv = b_q[(k*8 + cc)*8 +: 8]; + assign p[k] = av * bv; + end + assign dot = p[0] + p[1] + p[2] + p[3] + p[4] + p[5] + p[6] + p[7]; + always @(posedge clk) begin + if (rst || clr_q) c[r*8 + cc] <= 32'd0; + else c[r*8 + cc] <= c[r*8 + cc] + {{12{dot[19]}}, dot}; + end + end + end endgenerate + assign out = c[{sr_q, sc_q}]; +endmodule diff --git a/tools/chip-model/rtl/rtl/xbar32.v b/tools/chip-model/rtl/rtl/xbar32.v new file mode 100644 index 000000000..52b7f7398 --- /dev/null +++ b/tools/chip-model/rtl/rtl/xbar32.v @@ -0,0 +1,29 @@ +// 32-lane general crossbar over the same 1 KB window: each lane picks any source lane (a 5-bit select per +// lane), the upper bound on a shuffle's network cost. r[dst][lane] ^= r[src][sel[lane]]. +`include "lane_common.vh" +module xbar32( + input clk, input rst, + input [2:0] dst, input [2:0] src, input [159:0] sel, + input ld_en, input [4:0] ld_lane, input [31:0] ld_val, + output [31:0] out); + reg [31:0] rf [0:255]; + reg [2:0] dst_q, src_q; reg [159:0] sel_q; reg [4:0] ld_lane_q; reg ld_q; reg [31:0] ld_val_q; + integer i, l; + always @(posedge clk) begin + if (rst) begin dst_q <= 0; src_q <= 0; sel_q <= 0; ld_q <= 0; ld_lane_q <= 0; ld_val_q <= 0; end + else begin dst_q <= dst; src_q <= src; sel_q <= sel; ld_q <= ld_en; ld_lane_q <= ld_lane; ld_val_q <= ld_val; end + end + wire [31:0] s [0:31]; + wire [31:0] x [0:31]; + genvar g; + generate for (g = 0; g < 32; g = g + 1) begin : xb + assign s[g] = rf[g*8 + src_q]; + assign x[g] = s[sel_q[5*g +: 5]]; + end endgenerate + always @(posedge clk) begin + if (rst) begin for (i = 0; i < 256; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1) + 6; end + else for (l = 0; l < 32; l = l + 1) + rf[l*8 + dst_q] <= (ld_q && ld_lane_q == l) ? ld_val_q : (rf[l*8 + dst_q] ^ x[l]); + end + assign out = x[0] ^ x[17]; +endmodule diff --git a/tools/chip-model/rtl/tb/tb_lane_arx.v b/tools/chip-model/rtl/tb/tb_lane_arx.v new file mode 100644 index 000000000..b4b618ab7 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_arx.v @@ -0,0 +1,20 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] op = 0, dst = 0, src = 0; reg [4:0] rot = 0; reg ld_en = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + lane_arx dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .rot(rot), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, fixed_op, cycles; reg [31:0] acc = 0; + always #500 clk = ~clk; + initial begin + if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1; + if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; rot = $random; + ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_lane_fold.v b/tools/chip-model/rtl/tb/tb_lane_fold.v new file mode 100644 index 000000000..644de5980 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_fold.v @@ -0,0 +1,21 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg cfg_en = 0; reg [31:0] cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0; reg [4:0] cfg_r = 0; + reg ld_en = 0; reg [31:0] ld_val = 0; wire [31:0] out; + lane_fold dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .cfg_en(cfg_en), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, cycles; reg [31:0] acc = 0; + always #750 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + // one era draw: the constants are written once and then static, as on the chip + @(negedge clk); cfg_en = 1; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 32'h3; cfg_mask = 32'h0fffffff; + @(negedge clk); cfg_en = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + dst = $random; src = $random; ld_en = (($random & 3) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_lane_lop3.v b/tools/chip-model/rtl/tb/tb_lane_lop3.v new file mode 100644 index 000000000..3bccc1670 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_lop3.v @@ -0,0 +1,18 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0, src2 = 0; reg [7:0] lut = 0; reg ld_en = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + lane_lop3 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .src2(src2), .lut(lut), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, cycles; reg [31:0] acc = 0; + always #500 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + dst = $random; src = $random; src2 = $random; lut = $random; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_lane_mul.v b/tools/chip-model/rtl/tb/tb_lane_mul.v new file mode 100644 index 000000000..f5f37f7a0 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_mul.v @@ -0,0 +1,20 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [1:0] op = 0; reg [2:0] dst = 0, src = 0, src2 = 0; reg ld_en = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + lane_mul dut(.clk(clk), .rst(rst), .op(op), .dst(dst), .src(src), .src2(src2), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, fixed_op, cycles; reg [31:0] acc = 0; + always #750 clk = ~clk; + initial begin + if (!$value$plusargs("op=%d", fixed_op)) fixed_op = -1; + if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + op = (fixed_op < 0) ? $random : fixed_op; dst = $random; src = $random; src2 = $random; + ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_lane_prmt.v b/tools/chip-model/rtl/tb/tb_lane_prmt.v new file mode 100644 index 000000000..30764b26c --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_lane_prmt.v @@ -0,0 +1,18 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [15:0] sel = 0; reg ld_en = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + lane_prmt dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_val(ld_val), .out(out)); + integer n, cycles; reg [31:0] acc = 0; + always #500 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + dst = $random; src = $random; sel = $random & 16'h7777; ld_en = (($random & 15) == 0); ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_scratch8k.v b/tools/chip-model/rtl/tb/tb_scratch8k.v new file mode 100644 index 000000000..ed118a378 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_scratch8k.v @@ -0,0 +1,17 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [10:0] raddr = 0, waddr = 0; reg we = 0; reg [31:0] wdata = 0; wire [31:0] rdata; + scratch8k dut(.clk(clk), .rst(rst), .raddr(raddr), .we(we), .waddr(waddr), .wdata(wdata), .rdata(rdata)); + integer n, cycles; reg [31:0] acc = 0; + always #750 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + raddr = $random; waddr = $random; we = (($random & 7) == 0); wdata = $random; acc = acc ^ rdata; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_shfl32.v b/tools/chip-model/rtl/tb/tb_shfl32.v new file mode 100644 index 000000000..2cf58a69f --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_shfl32.v @@ -0,0 +1,18 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [4:0] m = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + shfl32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .m(m), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out)); + integer n, cycles; reg [31:0] acc = 0; + always #500 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + dst = $random; src = $random; m = $random; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_tile8.v b/tools/chip-model/rtl/tb/tb_tile8.v new file mode 100644 index 000000000..f9b66a2dc --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_tile8.v @@ -0,0 +1,18 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1, clr = 0; reg [511:0] a_in = 0, b_in = 0; reg [2:0] sel_r = 0, sel_c = 0; wire [31:0] out; + tile8 dut(.clk(clk), .rst(rst), .clr(clr), .a_in(a_in), .b_in(b_in), .sel_r(sel_r), .sel_c(sel_c), .out(out)); + integer n, j, cycles; reg [31:0] acc = 0; + always #1000 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + for (j = 0; j < 16; j = j + 1) begin a_in[j*32 +: 32] = $random; b_in[j*32 +: 32] = $random; end + clr = (($random & 63) == 0); sel_r = $random; sel_c = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_xbar32.v b/tools/chip-model/rtl/tb/tb_xbar32.v new file mode 100644 index 000000000..6440b2340 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_xbar32.v @@ -0,0 +1,18 @@ +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1; reg [2:0] dst = 0, src = 0; reg [159:0] sel = 0; reg ld_en = 0; reg [4:0] ld_lane = 0; reg [31:0] ld_val = 0; + wire [31:0] out; + xbar32 dut(.clk(clk), .rst(rst), .dst(dst), .src(src), .sel(sel), .ld_en(ld_en), .ld_lane(ld_lane), .ld_val(ld_val), .out(out)); + integer n, cycles; reg [31:0] acc = 0; + always #500 clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 3000; + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); + dst = $random; src = $random; sel = {$random, $random, $random, $random, $random}; ld_en = 1; ld_lane = $random; ld_val = $random; acc = acc ^ out; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule