diff --git a/docs/analysis/class-v6/floor/shadow-k.md b/docs/analysis/class-v6/floor/shadow-k.md index bf294bc95..966f3df64 100644 --- a/docs/analysis/class-v6/floor/shadow-k.md +++ b/docs/analysis/class-v6/floor/shadow-k.md @@ -152,6 +152,39 @@ term, approximate), so the net figure for the re-fold is 7.0 / 4.9 / 3.5 / 2.5 p N2 (the synthesis-only figure within the rounding). The imem is amortised over 8 lanes in this row and over 32 in the core32 row. +### 4a. The design sweep: what a class could add to the core's cost (the coordinator's order, 14:3x UK; rows 14:5x) + +Synthesis-only (no wires, no clock tree), 8 lanes unless stated, the same corner and scaling; the steady-state run +power solved from two run lengths (150 and 600 cycles after the program load). The GPU side per knob is the hash +lane's: knob 3 measured on a rented 5090 and 4090 (RunPod, 14:28 to 14:39 UK), knobs 1 and 4 modelled until a +generator line exists (a 64-entry window is a new ISA: an init rule and a fold rule for the extra registers, about +half a day; the select tree is a new instruction kind), knob 2 has no GPU side (the warp's shuffle already spans +32 lanes). + +| Variant | Cells | pJ per lane-op ASAP7 | of which clocking (RF, imem, IR; no gating) | N5 | N3 | N2 | k at N3 vs 5090 stock / lock / M5 Max | GPU side | +|---|---|---|---|---|---|---|---|---| +| base: 32 registers, 256-entry imem (the headline row) | 186,443 | 6.9 | 2.4 | 4.8 | 3.5 | 2.5 | 0.31 / 0.56 / 0.50 | measured (class v4) | +| (1) 64-register file (a 40-bit instruction word) | 267,731 | 9.7 | 3.75 | 6.8 | 4.9 | 3.5 | 0.43 / 0.78 / 0.70 | new ISA; 64 live registers takes a 5090 or 4090 thread to about 110 of 255, occupancy to about half; under the latency-bound chain the rate is expected to hold and the energy to move little (the hash lane, modelled); the unmeasured term is the per-lane register traffic | +| (3) 1,024-entry imem, the program drawn at 1,024, as built (a flop array) | 324,543 | 12.0 | 6.0 | 8.4 | 6.1 | 4.4 | 0.53 / 0.97 / 0.87 | measured: the 1,024 block at 27 passes costs the 5090 1.58x the energy per hash (4.96 against 3.14 microjoules at stock, 111 against 141 MH/s at the 575 W cap) and the 4090 1.57x (7.01 against 4.48, the rate held at 62.5 MH/s), for 4x the shadow instructions: 0.40x per instruction | +| (3) the same with the imem as a 4 KB SRAM macro shared by the lanes (2 to 4 pJ per 32-bit read, approximate) | | about 7.2 | about 1.9 | 5.0 | 3.6 | 2.6 | about 0.32 / 0.58 / 0.52 | the same | +| (4) the drawn select tree (the era's 16-entry op permutation ahead of decode; every unit evaluated every cycle, as in the base) | 186,870 | 6.85 | 2.4 | 4.8 | 3.4 | 2.5 | 0.30 / 0.55 / 0.50 | the units' microbench sum (approximate) | +| (2) 32 lanes, 32 registers (the butterfly across 32; the imem amortised over 32) | ROW_CORE32 | +| (2') 32 lanes, 16 registers (the register-file sensitivity the other way) | ROW_CORE32R16 | +| (5) all four together (32 lanes, 64 registers, 1,024 imem, the select tree) | ROW_CORE32ALL | + +Reading, for the founder's "under 2x at the lock" (which needs k near 0.9 on the GDDR7 board): the 64-register +window is the one robust knob, because its cost is per lane and a chip cannot share it (+2.8 pJ per lane-op at +ASAP7, +0.22 of k at the lock). The long block's cost is instruction memory, which a chip shares across its lanes +as SRAM, so most of its 0.97 as built is the flop array's clock and the honest figure is about 0.6; the card +meanwhile pays 1.58x the energy per hash for it (measured), so on the GDDR7 board at stock the chip reads +4.96 / (0.466 + 408,400 x 3.6 pJ) = 2.6x (1.7x on the flop-array row, which is not a chip anyone builds). The +select tree costs the chip nothing because every unit already evaluates every cycle in the base core. (1) + (3) +together reach about 10 pJ per lane-op at ASAP7 with the SRAM imem (5.0 at N3, k about 0.81 at the lock, 0.45 at +stock), 14.8 as built (k about 1.2); so k 0.85 is reached only on the flop-array reading, and the DRAM board +under 2x at the lock needs the window, the long block and the knee together and holds only while the chip's imem +stays unamortised, which it does not. The GPU pays nothing for the window until occupancy binds, 1.58x per hash for +the long block (0.40x per forcing instruction), and nothing for the select tree. + ## 5. The chip edge at the measured k `E_chip = E_mem + N_ops x e_chip` (absolute: the chip's shadow cost is 102,100 x 3.5 pJ = 0.36 microjoules per hash diff --git a/tools/chip-model/rtl/Makefile b/tools/chip-model/rtl/Makefile index 974a433bd..dea28cc08 100644 --- a/tools/chip-model/rtl/Makefile +++ b/tools/chip-model/rtl/Makefile @@ -24,7 +24,7 @@ DOCKER := $(LEASEPFX) docker run --rm -u $(UID_GID) -e HOME=/tmp -e NUM_CORES= ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG) SIM := $(DOCKER) -w /work $(SIM_IMG) -DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 +DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16 core8r64 core8i1k core8sel core32all top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt) # per-family simulation tags (the op field fixed per row where the family has several ops) @@ -40,6 +40,10 @@ SIMS_tile := mix SIMS_core8 := mix mixld:+loads=1 SIMS_core32 := mix mixld:+loads=1 SIMS_core32r16 := mix mixld:+loads=1 +SIMS_core8r64 := mix +SIMS_core8i1k := mix +SIMS_core8sel := mix +SIMS_core32all := mix .PHONY: rows table clean @@ -67,7 +71,8 @@ synth-%: mkdir -p out/$* logs sim/$*-synth $(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk synth 2>&1 | tee logs/synth-$*.log NET=1_2_yosys.v $(SIM) bash -c 'NET=1_2_yosys.v bash /work/flow/gl2sim.sh $* $(call top,$*)' 2>&1 | tee logs/gl2sim-$*-synth.log - $(SIM) bash /work/flow/sim.sh $* synth 2>&1 | tee logs/sim-$*-synth.log + $(SIM) bash /work/flow/sim.sh $* s150 +cycles=150 2>&1 | tee logs/sim-$*-s150.log + $(SIM) bash /work/flow/sim.sh $* s600 +cycles=600 2>&1 | tee logs/sim-$*-s600.log $(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power_synth FLOORK_ODB=1_synth.odb FLOORK_SDC=1_synth.sdc run 2>&1 | tee logs/power-$*-synth.log table: diff --git a/tools/chip-model/rtl/flow/collect.py b/tools/chip-model/rtl/flow/collect.py index 01d9fa21b..842842019 100644 --- a/tools/chip-model/rtl/flow/collect.py +++ b/tools/chip-model/rtl/flow/collect.py @@ -6,7 +6,7 @@ import re, sys, os, csv work = sys.argv[1] if len(sys.argv) > 1 else '.' # ops per cycle per design (the per-op divisor) and the GPU row each family is read against -OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32} +OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32, 'core32r16': 32, 'core8r64': 8, 'core8i1k': 8, 'core8sel': 8, 'core32all': 32} # 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a GPU = { 'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2), @@ -17,7 +17,7 @@ GPU = { 'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4), 'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed) 'tile:mix': (4.1, 2.2), - 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 + 'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), 'core32r16:mix': (11.3, 6.2), 'core8r64:mix': (11.3, 6.2), 'core8i1k:mix': (11.3, 6.2), 'core8sel:mix': (11.3, 6.2), 'core32all:mix': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83 } # per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed: # N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent), @@ -42,15 +42,35 @@ def parse_log(path): total=float(tm.group(4)), annotated=(f"{ann.group(1)} pins, {unann.group(1) if unann else '?'} unannotated") if ann else '') return period, cells, rows +LOAD_CYCLES = 260 # the core testbenches: 4 reset + 2 config + 256 program-load cycles before the run (1,028 at NPROG 1024) +def steady(rows, period, short, long_, cs, cl, loadc): + """Solve the run-phase power from two run lengths: P_i = (loadc x b + c_i x a) / (loadc + c_i).""" + out = {} + for key in ('internal', 'switching', 'leakage', 'total'): + Ps, Pl = rows[short][key], rows[long_][key] + A = Ps * (loadc + cs); B = Pl * (loadc + cl) + a = (B - A) / (cl - cs) + out[key] = a + out['annotated'] = rows[long_]['annotated'] + '; steady state solved from the two run lengths' + return out out = [] for d in OPS: - log = os.path.join(work, 'out', d, 'logs', 'asap7', d, 'base', 'power.log') - if not os.path.exists(log): + found = None + for stem in ('power_synth', 'power_synth_short', 'power'): + log = os.path.join(work, 'out', d, 'logs', 'asap7', d, 'base', stem + '.log') + if os.path.exists(log): + found = log + if stem == 'power': break + if not found: continue - period, cells, rows = parse_log(log) + period, cells, rows = parse_log(found) + pairs = [('vcd:s150', 'vcd:s600', 150, 600), ('vcd:short', 'vcd:synth', 150, 800), ('vcd:s150', 'vcd:s400', 150, 400)] + for sh, lg, cs, cl in pairs: + if sh in rows and lg in rows: + rows['vcd:steady'] = steady(rows, period, sh, lg, cs, cl, 1028 if d in ('core8i1k', 'core32all') else LOAD_CYCLES) for tag, r in rows.items(): sub = tag.split(':')[1] if ':' in tag else 'prop' - key = f'{d}:{sub}' if sub != 'prop' else f'{d}:mix' + key = f'{d}:{sub}' if sub in ('add','sub','xor','or','rotl','rotr','mul','mulhi','mad','mixld') else f'{d}:mix' gpu = GPU.get(key, (None, None)) pj = r['total'] * period * 1e-12 / OPS[d] * 1e12 # W * s / ops -> pJ pj_dyn = (r['internal'] + r['switching']) * period / OPS[d] @@ -64,7 +84,7 @@ for d in OPS: row['gpu_pJ_unlocked'] = gpu[0]; row['gpu_pJ_lock'] = gpu[1] # the Apple M5 Max: 6.9 pJ per counted op measured on the class v4 shadow (the honest tier's top); only the # ARX-class and the core rows have a measured M5 Max figure - m5 = 6.9 if (d in ('arx', 'core8', 'core32', 'core32r16')) else None + m5 = 6.9 if (d == 'arx' or d.startswith('core')) else None row['m5_pJ'] = m5 for node, sc in SCALE.items(): row[f'k_{node}_m5'] = (pj * sc / m5) if m5 else None diff --git a/tools/chip-model/rtl/flow/core32all.mk b/tools/chip-model/rtl/flow/core32all.mk new file mode 100644 index 000000000..c027bf40e --- /dev/null +++ b/tools/chip-model/rtl/flow/core32all.mk @@ -0,0 +1,15 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_32all), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_32all +export DESIGN_NICKNAME = core32all +export VERILOG_FILES = /work/rtl/core_v6_32all.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core32all.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core32all +export SYNTH_MEMORY_MAX_BITS = 2000000 diff --git a/tools/chip-model/rtl/flow/core32all.sdc b/tools/chip-model/rtl/flow/core32all.sdc new file mode 100644 index 000000000..13993d335 --- /dev/null +++ b/tools/chip-model/rtl/flow/core32all.sdc @@ -0,0 +1,10 @@ +current_design core_v6_32all +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/core8i1k.mk b/tools/chip-model/rtl/flow/core8i1k.mk new file mode 100644 index 000000000..bbfe890c8 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8i1k.mk @@ -0,0 +1,15 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8i1k), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8i1k +export DESIGN_NICKNAME = core8i1k +export VERILOG_FILES = /work/rtl/core_v6_8i1k.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core8i1k.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core8i1k +export SYNTH_MEMORY_MAX_BITS = 2000000 diff --git a/tools/chip-model/rtl/flow/core8i1k.sdc b/tools/chip-model/rtl/flow/core8i1k.sdc new file mode 100644 index 000000000..fa6e34e57 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8i1k.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8i1k +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/core8r64.mk b/tools/chip-model/rtl/flow/core8r64.mk new file mode 100644 index 000000000..ff5d73224 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8r64.mk @@ -0,0 +1,15 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8r64), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8r64 +export DESIGN_NICKNAME = core8r64 +export VERILOG_FILES = /work/rtl/core_v6_8r64.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core8r64.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core8r64 +export SYNTH_MEMORY_MAX_BITS = 2000000 diff --git a/tools/chip-model/rtl/flow/core8r64.sdc b/tools/chip-model/rtl/flow/core8r64.sdc new file mode 100644 index 000000000..ffb92a9cd --- /dev/null +++ b/tools/chip-model/rtl/flow/core8r64.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8r64 +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/core8sel.mk b/tools/chip-model/rtl/flow/core8sel.mk new file mode 100644 index 000000000..07f9f6ebb --- /dev/null +++ b/tools/chip-model/rtl/flow/core8sel.mk @@ -0,0 +1,15 @@ +# ORFS design config for the programmable shadow core (core8: core_v6_8sel), ASAP7. +export PLATFORM = asap7 +export DESIGN_NAME = core_v6_8sel +export DESIGN_NICKNAME = core8sel +export VERILOG_FILES = /work/rtl/core_v6_8sel.v +export VERILOG_INCLUDE_DIRS = /work/rtl +export SDC_FILE = /work/flow/core8sel.sdc +export CORE_UTILIZATION = 40 +export CORE_ASPECT_RATIO = 1 +export CORE_MARGIN = 0.5 +export PLACE_DENSITY = 0.55 +export CORNER = TC +export SKIP_LAST_GASP = 1 +export WORK_HOME = /work/out/core8sel +export SYNTH_MEMORY_MAX_BITS = 2000000 diff --git a/tools/chip-model/rtl/flow/core8sel.sdc b/tools/chip-model/rtl/flow/core8sel.sdc new file mode 100644 index 000000000..36f995570 --- /dev/null +++ b/tools/chip-model/rtl/flow/core8sel.sdc @@ -0,0 +1,10 @@ +current_design core_v6_8sel +set clk_name core_clock +set clk_port_name clk +set clk_period 1500 +set clk_io_pct 0.2 +set clk_port [get_ports $clk_port_name] +create_clock -name $clk_name -period $clk_period $clk_port +set non_clock_inputs [all_inputs -no_clocks] +set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs +set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs] diff --git a/tools/chip-model/rtl/flow/designs.txt b/tools/chip-model/rtl/flow/designs.txt index 329ec661e..af4f9566e 100644 --- a/tools/chip-model/rtl/flow/designs.txt +++ b/tools/chip-model/rtl/flow/designs.txt @@ -10,3 +10,7 @@ tile tile8 2000 core8 core_v6_8 1500 core32 core_v6_32 1500 core32r16 core_v6_32r16 1500 +core8r64 core_v6_8r64 1500 +core8i1k core_v6_8i1k 1500 +core8sel core_v6_8sel 1500 +core32all core_v6_32all 1500 diff --git a/tools/chip-model/rtl/flow/synthrow.sh b/tools/chip-model/rtl/flow/synthrow.sh new file mode 100755 index 000000000..1b53a543f --- /dev/null +++ b/tools/chip-model/rtl/flow/synthrow.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +# synthrow.sh [cycles-short] [cycles-long] : the synthesis-only power row for a design whose +# ORFS synthesis has already run (1_2_yosys.v and 1_synth.odb present): gate-level sim at two run lengths, then +# OpenSTA power on the synthesised netlist (no wires). Every step under the lease tool when present. +set -euo pipefail +name=$1; cs=${2:-150}; cl=${3:-600} +cd "$(dirname "$0")/.." +WORK=$(pwd); top=$(sed -n "s/^$name \([^ ]*\) .*/\1/p" flow/designs.txt) +UG=$(id -u):$(id -g) +if [ -x /srv/builds/_bin/lease ]; then L="/srv/builds/_bin/lease pool 8 --label floor-k-synthrow-$name --owner floor-k --nice 19 --"; else L=""; fi +$L docker run --rm -u $UG -e HOME=/tmp -e NET=1_2_yosys.v -v $WORK:/work -w /work orfs-sim:latest bash /work/flow/gl2sim.sh $name $top +$L docker run --rm -u $UG -e HOME=/tmp -v $WORK:/work -w /work orfs-sim:latest bash /work/flow/sim.sh $name s$cs +cycles=$cs +$L docker run --rm -u $UG -e HOME=/tmp -v $WORK:/work -w /work orfs-sim:latest bash /work/flow/sim.sh $name s$cl +cycles=$cl +$L docker run --rm -u $UG -e HOME=/tmp -e NUM_CORES=8 -v $WORK:/work -w /OpenROAD-flow-scripts/flow openroad/orfs:latest make DESIGN_CONFIG=/work/flow/$name.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power_synth FLOORK_ODB=1_synth.odb FLOORK_SDC=1_synth.sdc run diff --git a/tools/chip-model/rtl/rtl/core_v6.v b/tools/chip-model/rtl/rtl/core_v6.v index 618df15bf..da57559ca 100644 --- a/tools/chip-model/rtl/rtl/core_v6.v +++ b/tools/chip-model/rtl/rtl/core_v6.v @@ -11,28 +11,35 @@ // 0 add 1 sub 2 xor 3 or 4 rotl(imm) 5 rotr(src) 6 mul 7 mulhi 8 mad 9 shfl(imm mask) 10 prmt(aux,aux) // 11 lop3(aux lut) 12 load(fold(src) -> addr; dst <= returned word) 13 add 14 xor 15 sub `include "lane_common.vh" -module core_v6 #(parameter LANES = 32, parameter LOG_LANES = 5, parameter REGS = 32, parameter LOG_REGS = 5) ( +module core_v6 #(parameter LANES = 32, parameter LOG_LANES = 5, parameter REGS = 32, parameter LOG_REGS = 5, + parameter IW = 32, parameter IMEM_LOG = 8, parameter SELTREE = 0) ( input clk, input rst, input run, - input prog_we, input [7:0] prog_addr, input [31:0] prog_data, - input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input prog_we, input [9:0] prog_addr, input [IW-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [31:0] ld_val, output [31:0] addr, output [31:0] out); // instruction memory and the sequencer - reg [31:0] imem [0:255]; - reg [7:0] pc; reg [7:0] n_q; reg [31:0] ir; + localparam IMEM = 1 << IMEM_LOG; + reg [IW-1:0] imem [0:IMEM-1]; + reg [IMEM_LOG-1:0] pc; reg [IMEM_LOG-1:0] n_q; reg [IW-1:0] ir; + reg [63:0] sel_q; // the era's drawn op permutation (SELTREE = 1): 16 x 4-bit op codes reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q; integer i, l; always @(posedge clk) begin - if (prog_we) imem[prog_addr] <= prog_data; - if (rst) begin pc <= 0; ir <= 0; n_q <= 8'd255; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end + if (prog_we) imem[prog_addr[IMEM_LOG-1:0]] <= prog_data; + if (rst) begin pc <= 0; ir <= 0; n_q <= {IMEM_LOG{1'b1}}; sel_q <= 64'hfedcba9876543210; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end else begin - if (cfg_en) begin n_q <= cfg_n; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end - if (run) begin ir <= imem[pc]; pc <= (pc == n_q) ? 8'd0 : pc + 8'd1; end + if (cfg_en) begin n_q <= cfg_n[IMEM_LOG-1:0]; sel_q <= cfg_sel; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end + if (run) begin ir <= imem[pc]; pc <= (pc == n_q) ? {IMEM_LOG{1'b0}} : pc + 1'b1; end end end // decode - wire [3:0] op = ir[3:0]; wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[9 +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[14 +: LOG_REGS]; - wire [4:0] imm = ir[23:19]; wire [7:0] aux = ir[31:24]; + // the drawn select tree (SELTREE = 1): the op code is remapped through the era's 16-entry permutation before + // decode, so the datapath's select structure is the era's draw, not a fixed table a chip could hard-wire + wire [3:0] op_raw = ir[3:0]; + wire [3:0] op = SELTREE ? sel_q[op_raw*4 +: 4] : op_raw; + wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[4+LOG_REGS +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[4+2*LOG_REGS +: LOG_REGS]; + wire [4:0] imm = ir[4+3*LOG_REGS +: 5]; wire [7:0] aux = ir[9+3*LOG_REGS +: 8]; wire is_load = (op == 4'd12); wire [4:0] rn = (imm == 0) ? 5'd1 : imm; wire [LOG_LANES-1:0] smask = imm[LOG_LANES-1:0]; diff --git a/tools/chip-model/rtl/rtl/core_v6_32.v b/tools/chip-model/rtl/rtl/core_v6_32.v index 4b4f0e4a7..a12f2b262 100644 --- a/tools/chip-model/rtl/rtl/core_v6_32.v +++ b/tools/chip-model/rtl/rtl/core_v6_32.v @@ -1,7 +1,7 @@ `include "core_v6.v" -module core_v6_32(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data, - input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, +module core_v6_32(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [32-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [31:0] ld_val, output [31:0] addr, output [31:0] out); core_v6 #(.LANES(32), .LOG_LANES(5)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), - .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_32all.v b/tools/chip-model/rtl/rtl/core_v6_32all.v new file mode 100644 index 000000000..918bdb392 --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_v6_32all.v @@ -0,0 +1,7 @@ +`include "core_v6.v" +module core_v6_32all(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [40-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_v6 #(.LANES(32), .LOG_LANES(5), .REGS(64), .LOG_REGS(6), .IW(40), .IMEM_LOG(10), .SELTREE(1)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_32r16.v b/tools/chip-model/rtl/rtl/core_v6_32r16.v index 96fe03d67..43528b44c 100644 --- a/tools/chip-model/rtl/rtl/core_v6_32r16.v +++ b/tools/chip-model/rtl/rtl/core_v6_32r16.v @@ -1,7 +1,7 @@ `include "core_v6.v" -module core_v6_32r16(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data, - input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, +module core_v6_32r16(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [32-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [31:0] ld_val, output [31:0] addr, output [31:0] out); core_v6 #(.LANES(32), .LOG_LANES(5), .REGS(16), .LOG_REGS(4)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), - .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_8.v b/tools/chip-model/rtl/rtl/core_v6_8.v index c9e21b232..77c025d50 100644 --- a/tools/chip-model/rtl/rtl/core_v6_8.v +++ b/tools/chip-model/rtl/rtl/core_v6_8.v @@ -1,7 +1,7 @@ `include "core_v6.v" -module core_v6_8(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data, - input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, +module core_v6_8(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [32-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, input [31:0] ld_val, output [31:0] addr, output [31:0] out); core_v6 #(.LANES(8), .LOG_LANES(3)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), - .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_8i1k.v b/tools/chip-model/rtl/rtl/core_v6_8i1k.v new file mode 100644 index 000000000..61758a7b9 --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_v6_8i1k.v @@ -0,0 +1,7 @@ +`include "core_v6.v" +module core_v6_8i1k(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [32-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_v6 #(.LANES(8), .LOG_LANES(3), .IMEM_LOG(10)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_8r64.v b/tools/chip-model/rtl/rtl/core_v6_8r64.v new file mode 100644 index 000000000..134032fce --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_v6_8r64.v @@ -0,0 +1,7 @@ +`include "core_v6.v" +module core_v6_8r64(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [40-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_v6 #(.LANES(8), .LOG_LANES(3), .REGS(64), .LOG_REGS(6), .IW(40)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/rtl/core_v6_8sel.v b/tools/chip-model/rtl/rtl/core_v6_8sel.v new file mode 100644 index 000000000..b769156c7 --- /dev/null +++ b/tools/chip-model/rtl/rtl/core_v6_8sel.v @@ -0,0 +1,7 @@ +`include "core_v6.v" +module core_v6_8sel(input clk, input rst, input run, input prog_we, input [9:0] prog_addr, input [32-1:0] prog_data, + input cfg_en, input [9:0] cfg_n, input [63:0] cfg_sel, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask, + input [31:0] ld_val, output [31:0] addr, output [31:0] out); + core_v6 #(.LANES(8), .LOG_LANES(3), .SELTREE(1)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), + .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); +endmodule diff --git a/tools/chip-model/rtl/tb/tb_core_common.vh b/tools/chip-model/rtl/tb/tb_core_common.vh index c1dac7f2c..38e969897 100644 --- a/tools/chip-model/rtl/tb/tb_core_common.vh +++ b/tools/chip-model/rtl/tb/tb_core_common.vh @@ -1,10 +1,10 @@ // shared body for the core testbenches: `TOP and `LANES set by the including file `timescale 1ps/1ps module tb; - reg clk = 0, rst = 1, run = 0, prog_we = 0, cfg_en = 0; reg [7:0] prog_addr = 0, cfg_n = 0; reg [31:0] prog_data = 0, cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0, ld_val = 0; reg [4:0] cfg_r = 0; + reg clk = 0, rst = 1, run = 0, prog_we = 0, cfg_en = 0; reg [9:0] prog_addr = 0, cfg_n = 0; reg [`IW-1:0] prog_data = 0; reg [63:0] cfg_sel = 0; reg [31:0] cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0, ld_val = 0; reg [4:0] cfg_r = 0; wire [31:0] addr, out; - `TOP dut(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); - integer n, cycles, loads, k, roll; reg [31:0] acc = 0; reg [3:0] opc; reg [31:0] w; + `TOP dut(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_sel(cfg_sel), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); + integer n, cycles, loads, k, roll; reg [31:0] acc = 0; reg [3:0] opc; reg [63:0] w; // the class v4 draw: add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4 (sum 75) function [3:0] draw_op; input integer r; integer x; begin x = r % 75; if (x < 0) x = -x; @@ -18,12 +18,12 @@ module tb; $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); repeat (4) @(negedge clk); rst = 0; // the era draw: constants and the program length - @(negedge clk); cfg_en = 1; cfg_n = 8'd255; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff; + @(negedge clk); cfg_en = 1; cfg_n = `NPROG - 1; cfg_sel = `SEL; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff; @(negedge clk); cfg_en = 0; - // the program: 256 instructions drawn with the class v4 weights - for (k = 0; k < 256; k = k + 1) begin - @(negedge clk); w = $random; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random); - prog_we = 1; prog_addr = k; prog_data = {w[31:4], opc}; + // the program: `NPROG instructions drawn with the class v4 weights + for (k = 0; k < `NPROG; k = k + 1) begin + @(negedge clk); w = {$random, $random}; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random); + prog_we = 1; prog_addr = k; prog_data = {w[`IW-1:4], opc}; end @(negedge clk); prog_we = 0; run = 1; for (n = 0; n < cycles; n = n + 1) begin diff --git a/tools/chip-model/rtl/tb/tb_core_legacy.vh b/tools/chip-model/rtl/tb/tb_core_legacy.vh new file mode 100644 index 000000000..4fecabfee --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_legacy.vh @@ -0,0 +1,34 @@ +// shared body for the core testbenches: `TOP and `LANES set by the including file +`timescale 1ps/1ps +module tb; + reg clk = 0, rst = 1, run = 0, prog_we = 0, cfg_en = 0; reg [7:0] prog_addr = 0, cfg_n = 0; reg [31:0] prog_data = 0, cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0, ld_val = 0; reg [4:0] cfg_r = 0; + wire [31:0] addr, out; + `TOP dut(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out)); + integer n, cycles, loads, k, roll; reg [31:0] acc = 0; reg [3:0] opc; reg [31:0] w; + // the class v4 draw: add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4 (sum 75) + function [3:0] draw_op; input integer r; integer x; + begin x = r % 75; if (x < 0) x = -x; + draw_op = (x < 12) ? 4'd0 : (x < 22) ? 4'd2 : (x < 30) ? 4'd6 : (x < 38) ? 4'd8 : (x < 46) ? 4'd9 : (x < 53) ? 4'd4 : (x < 59) ? 4'd1 : (x < 65) ? 4'd7 : (x < 71) ? 4'd5 : 4'd3; + end + endfunction + always #`HALF clk = ~clk; + initial begin + if (!$value$plusargs("cycles=%d", cycles)) cycles = 300; + if (!$value$plusargs("loads=%d", loads)) loads = 0; // 1: one load in 16 instructions + $dumpfile("dump.vcd"); $dumpvars(0, tb.dut); + repeat (4) @(negedge clk); rst = 0; + // the era draw: constants and the program length + @(negedge clk); cfg_en = 1; cfg_n = 8'd255; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff; + @(negedge clk); cfg_en = 0; + // the program: 256 instructions drawn with the class v4 weights + for (k = 0; k < 256; k = k + 1) begin + @(negedge clk); w = $random; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random); + prog_we = 1; prog_addr = k; prog_data = {w[31:4], opc}; + end + @(negedge clk); prog_we = 0; run = 1; + for (n = 0; n < cycles; n = n + 1) begin + @(negedge clk); ld_val = $random; acc = acc ^ out ^ addr; + end + $display("CHECKSUM %08x", acc); $finish; + end +endmodule diff --git a/tools/chip-model/rtl/tb/tb_core_v6_32.v b/tools/chip-model/rtl/tb/tb_core_v6_32.v index 032d82423..2c63f5096 100644 --- a/tools/chip-model/rtl/tb/tb_core_v6_32.v +++ b/tools/chip-model/rtl/tb/tb_core_v6_32.v @@ -1,3 +1,3 @@ `define TOP core_v6_32 `define HALF 750 -`include "tb_core_common.vh" +`include "tb_core_legacy.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_32all.v b/tools/chip-model/rtl/tb/tb_core_v6_32all.v new file mode 100644 index 000000000..9509ee182 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_v6_32all.v @@ -0,0 +1,6 @@ +`define TOP core_v6_32all +`define HALF 750 +`define IW 40 +`define NPROG 1024 +`define SEL 64'hc5e7092b4d6f81a3 +`include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_32r16.v b/tools/chip-model/rtl/tb/tb_core_v6_32r16.v index f2084f017..696d54346 100644 --- a/tools/chip-model/rtl/tb/tb_core_v6_32r16.v +++ b/tools/chip-model/rtl/tb/tb_core_v6_32r16.v @@ -1,3 +1,6 @@ `define TOP core_v6_32r16 `define HALF 750 +`define IW 32 +`define NPROG 256 +`define SEL 64'hfedcba9876543210 `include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_8.v b/tools/chip-model/rtl/tb/tb_core_v6_8.v index 367b83e59..b45dfef48 100644 --- a/tools/chip-model/rtl/tb/tb_core_v6_8.v +++ b/tools/chip-model/rtl/tb/tb_core_v6_8.v @@ -1,3 +1,6 @@ `define TOP core_v6_8 `define HALF 750 +`define IW 32 +`define NPROG 256 +`define SEL 64'hfedcba9876543210 `include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_8i1k.v b/tools/chip-model/rtl/tb/tb_core_v6_8i1k.v new file mode 100644 index 000000000..738a4815b --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_v6_8i1k.v @@ -0,0 +1,6 @@ +`define TOP core_v6_8i1k +`define HALF 750 +`define IW 32 +`define NPROG 1024 +`define SEL 64'hfedcba9876543210 +`include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_8r64.v b/tools/chip-model/rtl/tb/tb_core_v6_8r64.v new file mode 100644 index 000000000..5efbee6f6 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_v6_8r64.v @@ -0,0 +1,6 @@ +`define TOP core_v6_8r64 +`define HALF 750 +`define IW 40 +`define NPROG 256 +`define SEL 64'hfedcba9876543210 +`include "tb_core_common.vh" diff --git a/tools/chip-model/rtl/tb/tb_core_v6_8sel.v b/tools/chip-model/rtl/tb/tb_core_v6_8sel.v new file mode 100644 index 000000000..65094edc8 --- /dev/null +++ b/tools/chip-model/rtl/tb/tb_core_v6_8sel.v @@ -0,0 +1,6 @@ +`define TOP core_v6_8sel +`define HALF 750 +`define IW 32 +`define NPROG 256 +`define SEL 64'hc5e7092b4d6f81a3 +`include "tb_core_common.vh"