chip-model: the programmable shadow core (core_v6: imem, fetch, decode, per-lane register file, the class units, era registers), 8, 32 and 32-lane 16-register builds, synthesis-only power path
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
6fab785c51
commit
49c49c340d
20 changed files with 277 additions and 7 deletions
|
|
@ -17,7 +17,7 @@ DOCKER := docker run --rm -u $(UID_GID) -e HOME=/tmp -v $(WORK):/work
|
|||
ORFS := $(DOCKER) -w /OpenROAD-flow-scripts/flow $(ORFS_IMG)
|
||||
SIM := $(DOCKER) -w /work $(SIM_IMG)
|
||||
|
||||
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile
|
||||
DESIGNS := arx mul prmt lop3 fold shfl xbar scratch tile core8 core32 core32r16
|
||||
top = $(shell sed -n 's/^$(1) \([^ ]*\) .*/\1/p' flow/designs.txt)
|
||||
|
||||
# per-family simulation tags (the op field fixed per row where the family has several ops)
|
||||
|
|
@ -30,6 +30,9 @@ SIMS_shfl := mix
|
|||
SIMS_xbar := mix
|
||||
SIMS_scratch := mix
|
||||
SIMS_tile := mix
|
||||
SIMS_core8 := mix mixld:+loads=1
|
||||
SIMS_core32 := mix mixld:+loads=1
|
||||
SIMS_core32r16 := mix mixld:+loads=1
|
||||
|
||||
.PHONY: rows table clean
|
||||
|
||||
|
|
@ -52,6 +55,14 @@ sim-%: flow-%
|
|||
power-%: sim-%
|
||||
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power run 2>&1 | tee logs/power-$*.log
|
||||
|
||||
# synthesis-only row (no placement, no parasitics): the 14:45 fallback
|
||||
synth-%:
|
||||
mkdir -p out/$* logs sim/$*-synth
|
||||
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk synth 2>&1 | tee logs/synth-$*.log
|
||||
NET=1_2_yosys.v $(SIM) bash -c 'NET=1_2_yosys.v bash /work/flow/gl2sim.sh $* $(call top,$*)' 2>&1 | tee logs/gl2sim-$*-synth.log
|
||||
$(SIM) bash /work/flow/sim.sh $* synth 2>&1 | tee logs/sim-$*-synth.log
|
||||
$(ORFS) make DESIGN_CONFIG=/work/flow/$*.mk RUN_SCRIPT=/work/flow/power.tcl RUN_LOG_NAME_STEM=power_synth FLOORK_ODB=1_synth.odb FLOORK_SDC=1_synth.sdc run 2>&1 | tee logs/power-$*-synth.log
|
||||
|
||||
table:
|
||||
python3 flow/collect.py $(WORK) > table.md
|
||||
@echo wrote table.md and table.csv
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ import re, sys, os, csv
|
|||
|
||||
work = sys.argv[1] if len(sys.argv) > 1 else '.'
|
||||
# ops per cycle per design (the per-op divisor) and the GPU row each family is read against
|
||||
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512}
|
||||
OPS = {'arx': 1, 'mul': 1, 'prmt': 1, 'lop3': 1, 'fold': 1, 'shfl': 32, 'xbar': 32, 'scratch': 1, 'tile': 512, 'core8': 8, 'core32': 32}
|
||||
# 5090 measured pJ per counted op: (unlocked, at the 1,300 MHz lock); 15.1a
|
||||
GPU = {
|
||||
'arx:mix': (11.3, 6.2), 'arx:add': (11.3, 6.2), 'arx:sub': (11.3, 6.2), 'arx:xor': (11.3, 6.2), 'arx:or': (11.3, 6.2),
|
||||
|
|
@ -16,7 +16,8 @@ GPU = {
|
|||
'fold:mix': (13.9, 8.3), # the fold is a multiply plus a rotate and masks: read against int_mul
|
||||
'shfl:mix': (55.8, 29.4), 'xbar:mix': (55.8, 29.4),
|
||||
'scratch:mix': (2400.0, 1400.0), # the card's L2 hit (no shared-memory probe measured: owed)
|
||||
'tile:mix': (4.1, 2.2), # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
|
||||
'tile:mix': (4.1, 2.2),
|
||||
'core8:mix': (11.3, 6.2), 'core8:mixld': (11.3, 6.2), 'core32:mix': (11.3, 6.2), 'core32:mixld': (11.3, 6.2), # the class v4 draw: read against int_arx (the packs job read the whole mix at 10.8 / 6.4) # dependent u8 m8n8k16 per MAC; the wide s8 tile reads 1.36 / 0.83
|
||||
}
|
||||
# per-node energy scaling from ASAP7 (a 7 nm-class predictive PDK at 0.70 V), approximate and claimed:
|
||||
# N7 -> N5 x0.70 (TSMC: "30 percent lower power at the same speed"), N5 -> N3E x0.72 (TSMC: 25 to 30 percent),
|
||||
|
|
|
|||
15
tools/chip-model/rtl/flow/core32.mk
Normal file
15
tools/chip-model/rtl/flow/core32.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the programmable shadow core (core32: core_v6_32), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = core_v6_32
|
||||
export DESIGN_NICKNAME = core32
|
||||
export VERILOG_FILES = /work/rtl/core_v6_32.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/core32.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/core32
|
||||
export SYNTH_MEMORY_MAX_BITS = 2000000
|
||||
10
tools/chip-model/rtl/flow/core32.sdc
Normal file
10
tools/chip-model/rtl/flow/core32.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design core_v6_32
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/core32r16.mk
Normal file
15
tools/chip-model/rtl/flow/core32r16.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the programmable shadow core (core32r16: core_v6_32r16), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = core_v6_32r16
|
||||
export DESIGN_NICKNAME = core32r16
|
||||
export VERILOG_FILES = /work/rtl/core_v6_32r16.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/core32r16.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/core32r16
|
||||
export SYNTH_MEMORY_MAX_BITS = 2000000
|
||||
10
tools/chip-model/rtl/flow/core32r16.sdc
Normal file
10
tools/chip-model/rtl/flow/core32r16.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design core_v6_32r16
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
15
tools/chip-model/rtl/flow/core8.mk
Normal file
15
tools/chip-model/rtl/flow/core8.mk
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
# ORFS design config for the programmable shadow core (core8: core_v6_8), ASAP7.
|
||||
export PLATFORM = asap7
|
||||
export DESIGN_NAME = core_v6_8
|
||||
export DESIGN_NICKNAME = core8
|
||||
export VERILOG_FILES = /work/rtl/core_v6_8.v
|
||||
export VERILOG_INCLUDE_DIRS = /work/rtl
|
||||
export SDC_FILE = /work/flow/core8.sdc
|
||||
export CORE_UTILIZATION = 40
|
||||
export CORE_ASPECT_RATIO = 1
|
||||
export CORE_MARGIN = 0.5
|
||||
export PLACE_DENSITY = 0.55
|
||||
export CORNER = TC
|
||||
export SKIP_LAST_GASP = 1
|
||||
export WORK_HOME = /work/out/core8
|
||||
export SYNTH_MEMORY_MAX_BITS = 2000000
|
||||
10
tools/chip-model/rtl/flow/core8.sdc
Normal file
10
tools/chip-model/rtl/flow/core8.sdc
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
current_design core_v6_8
|
||||
set clk_name core_clock
|
||||
set clk_port_name clk
|
||||
set clk_period 1500
|
||||
set clk_io_pct 0.2
|
||||
set clk_port [get_ports $clk_port_name]
|
||||
create_clock -name $clk_name -period $clk_period $clk_port
|
||||
set non_clock_inputs [all_inputs -no_clocks]
|
||||
set_input_delay [expr $clk_period * $clk_io_pct] -clock $clk_name $non_clock_inputs
|
||||
set_output_delay [expr $clk_period * $clk_io_pct] -clock $clk_name [all_outputs]
|
||||
|
|
@ -7,3 +7,6 @@ shfl shfl32 1000
|
|||
xbar xbar32 1000
|
||||
scratch scratch8k 1500
|
||||
tile tile8 2000
|
||||
core8 core_v6_8 1500
|
||||
core32 core_v6_32 1500
|
||||
core32r16 core_v6_32r16 1500
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
# gl2sim.sh <design-nickname> <top> : netlist -> sim netlist with cell bodies from the liberty.
|
||||
set -euo pipefail
|
||||
name=$1; top=$2
|
||||
net=/work/out/$name/results/asap7/$name/base/6_final.v
|
||||
net=/work/out/$name/results/asap7/$name/base/${NET:-6_final.v}
|
||||
lib=/OpenROAD-flow-scripts/flow/platforms/asap7/lib/NLDM
|
||||
out=/work/sim/$name; mkdir -p $out
|
||||
cat > $out/gl2sim.ys <<YS
|
||||
|
|
|
|||
|
|
@ -2,9 +2,12 @@
|
|||
# (b) a propagated 0.5 input activity, both at the TC corner (0.70 V, 0 C on ASAP7).
|
||||
# Run through ORFS: make DESIGN_CONFIG=/work/flow/<name>.mk RUN_SCRIPT=/work/flow/power.tcl run
|
||||
source $::env(SCRIPTS_DIR)/load.tcl
|
||||
load_design 6_final.odb 6_final.sdc
|
||||
set odb [expr {[info exists ::env(FLOORK_ODB)] ? $::env(FLOORK_ODB) : "6_final.odb"}]
|
||||
set sdc [expr {[info exists ::env(FLOORK_SDC)] ? $::env(FLOORK_SDC) : "6_final.sdc"}]
|
||||
load_design $odb $sdc
|
||||
puts "FLOORK stage $odb"
|
||||
set spef $::env(RESULTS_DIR)/6_final.spef
|
||||
if { [file exists $spef] } { read_spef $spef } else { estimate_parasitics -global_routing }
|
||||
if { $odb == "6_final.odb" && [file exists $spef] } { read_spef $spef } elseif { $odb == "6_final.odb" } { estimate_parasitics -global_routing } elseif { [string match "3_*" $odb] || [string match "4_*" $odb] } { estimate_parasitics -placement } else { puts "FLOORK no parasitics (synthesis only)" }
|
||||
puts "FLOORK clock_period_ps [expr [get_property [lindex [all_clocks] 0] period]]"
|
||||
puts "FLOORK cells [llength [get_cells *]]"
|
||||
report_tns
|
||||
|
|
|
|||
|
|
@ -4,6 +4,6 @@ set -euo pipefail
|
|||
name=$1; tag=$2; shift 2
|
||||
out=/work/sim/$name
|
||||
simcells=$(yosys-config --datdir)/simcells.v
|
||||
iverilog -g2005 -o $out/sim_$tag $out/sim_net.v /work/tb/tb_$(sed -n "s/^$name \([^ ]*\) .*/\1/p" /work/flow/designs.txt).v $simcells
|
||||
iverilog -g2005 -I /work/tb -o $out/sim_$tag $out/sim_net.v /work/tb/tb_$(sed -n "s/^$name \([^ ]*\) .*/\1/p" /work/flow/designs.txt).v $simcells
|
||||
( cd $out && vvp -n sim_$tag "$@" | tee sim_$tag.log && mv dump.vcd $tag.vcd )
|
||||
ls -la $out/$tag.vcd
|
||||
|
|
|
|||
113
tools/chip-model/rtl/rtl/core_v6.v
Normal file
113
tools/chip-model/rtl/rtl/core_v6.v
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
// The programmable shadow core: the minimal in-order SIMD core that executes a class v6 shadow program as
|
||||
// drawn. Per core: an instruction memory sized to the drawn program length (256 x 32-bit, a flop array), a
|
||||
// program counter that wraps at the era's drawn length, fetch into an instruction register, decode, the era's
|
||||
// parameter registers (fold constants M, R, WM, OFF, MASK; program length N). Per lane: a 32 x 32-bit register
|
||||
// file (flops) with two or three read ports (mad reads three) and one write port, and the class's units: add, sub,
|
||||
// xor, or, rotl by immediate, rotr by register, mul, mulhi, mad, prmt, lop3, the xor-mask shuffle across the
|
||||
// lanes (a log2(LANES)-stage butterfly), and the load (the index fold on the address path, the returned word
|
||||
// written on the next cycle). One instruction per cycle for every lane: LANES lane-ops per cycle.
|
||||
//
|
||||
// Instruction word: op[3:0] dst[8:4] src[13:9] src2[18:14] imm[23:19] aux[31:24].
|
||||
// 0 add 1 sub 2 xor 3 or 4 rotl(imm) 5 rotr(src) 6 mul 7 mulhi 8 mad 9 shfl(imm mask) 10 prmt(aux,aux)
|
||||
// 11 lop3(aux lut) 12 load(fold(src) -> addr; dst <= returned word) 13 add 14 xor 15 sub
|
||||
`include "lane_common.vh"
|
||||
module core_v6 #(parameter LANES = 32, parameter LOG_LANES = 5, parameter REGS = 32, parameter LOG_REGS = 5) (
|
||||
input clk, input rst, input run,
|
||||
input prog_we, input [7:0] prog_addr, input [31:0] prog_data,
|
||||
input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input [31:0] ld_val,
|
||||
output [31:0] addr, output [31:0] out);
|
||||
// instruction memory and the sequencer
|
||||
reg [31:0] imem [0:255];
|
||||
reg [7:0] pc; reg [7:0] n_q; reg [31:0] ir;
|
||||
reg [31:0] m_q, wm_q, off_q, mask_q; reg [4:0] r_q;
|
||||
integer i, l;
|
||||
always @(posedge clk) begin
|
||||
if (prog_we) imem[prog_addr] <= prog_data;
|
||||
if (rst) begin pc <= 0; ir <= 0; n_q <= 8'd255; m_q <= 32'h9e3779b1; r_q <= 5'd13; wm_q <= 32'h0fffffc0; off_q <= 3; mask_q <= 32'h0fffffff; end
|
||||
else begin
|
||||
if (cfg_en) begin n_q <= cfg_n; m_q <= cfg_m | 1; r_q <= cfg_r; wm_q <= cfg_wm; off_q <= cfg_off; mask_q <= cfg_mask; end
|
||||
if (run) begin ir <= imem[pc]; pc <= (pc == n_q) ? 8'd0 : pc + 8'd1; end
|
||||
end
|
||||
end
|
||||
// decode
|
||||
wire [3:0] op = ir[3:0]; wire [LOG_REGS-1:0] dst = ir[4 +: LOG_REGS]; wire [LOG_REGS-1:0] src = ir[9 +: LOG_REGS]; wire [LOG_REGS-1:0] src2 = ir[14 +: LOG_REGS];
|
||||
wire [4:0] imm = ir[23:19]; wire [7:0] aux = ir[31:24];
|
||||
wire is_load = (op == 4'd12);
|
||||
wire [4:0] rn = (imm == 0) ? 5'd1 : imm;
|
||||
wire [LOG_LANES-1:0] smask = imm[LOG_LANES-1:0];
|
||||
// lanes
|
||||
reg [31:0] rf [0:LANES*REGS-1];
|
||||
wire [31:0] d_v [0:LANES-1]; wire [31:0] s_v [0:LANES-1]; wire [31:0] s2_v [0:LANES-1];
|
||||
wire [31:0] shin [0:LANES-1]; wire [31:0] shout [0:LANES-1];
|
||||
wire [31:0] res [0:LANES-1];
|
||||
genvar g, st;
|
||||
generate for (g = 0; g < LANES; g = g + 1) begin : ln
|
||||
assign d_v[g] = rf[g*REGS + dst];
|
||||
assign s_v[g] = rf[g*REGS + src];
|
||||
assign s2_v[g] = rf[g*REGS + src2];
|
||||
assign shin[g] = s_v[g];
|
||||
end endgenerate
|
||||
// the butterfly shuffle network across the lanes
|
||||
wire [31:0] bf [0:LOG_LANES][0:LANES-1];
|
||||
generate
|
||||
for (g = 0; g < LANES; g = g + 1) begin : bf0
|
||||
assign bf[0][g] = shin[g];
|
||||
end
|
||||
for (st = 0; st < LOG_LANES; st = st + 1) begin : bfs
|
||||
for (g = 0; g < LANES; g = g + 1) begin : bfl
|
||||
assign bf[st+1][g] = smask[st] ? bf[st][g ^ (1 << st)] : bf[st][g];
|
||||
end
|
||||
end
|
||||
for (g = 0; g < LANES; g = g + 1) begin : bfo
|
||||
assign shout[g] = bf[LOG_LANES][g];
|
||||
end
|
||||
endgenerate
|
||||
// the units per lane
|
||||
function [7:0] pick; input [63:0] b; input [3:0] k; reg [7:0] v;
|
||||
begin v = b[8*k[2:0] +: 8]; pick = k[3] ? {8{v[7]}} : v; end
|
||||
endfunction
|
||||
generate for (g = 0; g < LANES; g = g + 1) begin : un
|
||||
wire [31:0] d = d_v[g]; wire [31:0] s = s_v[g]; wire [31:0] s2 = s2_v[g];
|
||||
wire [4:0] sn = (s[4:0] == 0) ? 5'd1 : s[4:0];
|
||||
wire mad = (op == 4'd8);
|
||||
wire [63:0] p = (mad ? s : d) * (mad ? s2 : s);
|
||||
wire [63:0] bytes = {s, d}; wire [15:0] sel = {aux, aux};
|
||||
wire [31:0] prm = {pick(bytes, sel[15:12]), pick(bytes, sel[11:8]), pick(bytes, sel[7:4]), pick(bytes, sel[3:0])};
|
||||
reg [31:0] lp; integer b;
|
||||
always @* for (b = 0; b < 32; b = b + 1) lp[b] = aux[{d[b], s[b], s2[b]}];
|
||||
reg [31:0] r;
|
||||
always @* begin
|
||||
case (op)
|
||||
4'd0, 4'd13: r = d + s;
|
||||
4'd1, 4'd15: r = d - s;
|
||||
4'd2, 4'd14: r = d ^ s;
|
||||
4'd3: r = d | s;
|
||||
4'd4: r = `ROTL32(d, rn);
|
||||
4'd5: r = `ROTR32(d, sn);
|
||||
4'd6: r = p[31:0];
|
||||
4'd7: r = p[63:32];
|
||||
4'd8: r = p[31:0] + d;
|
||||
4'd9: r = d ^ shout[g];
|
||||
4'd10: r = prm;
|
||||
4'd11: r = lp;
|
||||
default: r = ld_val ^ (32'h9e3779b9 * (g + 1)); // the returned word (lane-salted by the testbench's bus)
|
||||
endcase
|
||||
end
|
||||
assign res[g] = r;
|
||||
end endgenerate
|
||||
// the address path: the fold of lane 0's source (one address per lane on a real part; lane 0 drives the port)
|
||||
wire [31:0] fx = s_v[0] * m_q;
|
||||
wire [4:0] frn = (r_q == 0) ? 5'd1 : r_q;
|
||||
wire [31:0] fy = `ROTL32(fx, frn);
|
||||
assign addr = is_load ? (((fy & wm_q) | off_q) & mask_q) : 32'd0;
|
||||
// writeback
|
||||
always @(posedge clk) begin
|
||||
if (rst) begin for (i = 0; i < LANES*REGS; i = i + 1) rf[i] <= 32'h9e3779b9 * (i + 1); end
|
||||
else if (run) for (l = 0; l < LANES; l = l + 1) rf[l*REGS + dst] <= res[l];
|
||||
end
|
||||
// keep every lane alive: the xor over the lanes' results
|
||||
reg [31:0] red; integer q;
|
||||
always @* begin red = 0; for (q = 0; q < LANES; q = q + 1) red = red ^ res[q]; end
|
||||
assign out = red;
|
||||
endmodule
|
||||
7
tools/chip-model/rtl/rtl/core_v6_32.v
Normal file
7
tools/chip-model/rtl/rtl/core_v6_32.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`include "core_v6.v"
|
||||
module core_v6_32(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data,
|
||||
input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
|
||||
core_v6 #(.LANES(32), .LOG_LANES(5)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
|
||||
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
endmodule
|
||||
7
tools/chip-model/rtl/rtl/core_v6_32r16.v
Normal file
7
tools/chip-model/rtl/rtl/core_v6_32r16.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`include "core_v6.v"
|
||||
module core_v6_32r16(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data,
|
||||
input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
|
||||
core_v6 #(.LANES(32), .LOG_LANES(5), .REGS(16), .LOG_REGS(4)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
|
||||
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
endmodule
|
||||
7
tools/chip-model/rtl/rtl/core_v6_8.v
Normal file
7
tools/chip-model/rtl/rtl/core_v6_8.v
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
`include "core_v6.v"
|
||||
module core_v6_8(input clk, input rst, input run, input prog_we, input [7:0] prog_addr, input [31:0] prog_data,
|
||||
input cfg_en, input [7:0] cfg_n, input [31:0] cfg_m, input [4:0] cfg_r, input [31:0] cfg_wm, input [31:0] cfg_off, input [31:0] cfg_mask,
|
||||
input [31:0] ld_val, output [31:0] addr, output [31:0] out);
|
||||
core_v6 #(.LANES(8), .LOG_LANES(3)) c(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data),
|
||||
.cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
endmodule
|
||||
34
tools/chip-model/rtl/tb/tb_core_common.vh
Normal file
34
tools/chip-model/rtl/tb/tb_core_common.vh
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
// shared body for the core testbenches: `TOP and `LANES set by the including file
|
||||
`timescale 1ps/1ps
|
||||
module tb;
|
||||
reg clk = 0, rst = 1, run = 0, prog_we = 0, cfg_en = 0; reg [7:0] prog_addr = 0, cfg_n = 0; reg [31:0] prog_data = 0, cfg_m = 0, cfg_wm = 0, cfg_off = 0, cfg_mask = 0, ld_val = 0; reg [4:0] cfg_r = 0;
|
||||
wire [31:0] addr, out;
|
||||
`TOP dut(.clk(clk), .rst(rst), .run(run), .prog_we(prog_we), .prog_addr(prog_addr), .prog_data(prog_data), .cfg_en(cfg_en), .cfg_n(cfg_n), .cfg_m(cfg_m), .cfg_r(cfg_r), .cfg_wm(cfg_wm), .cfg_off(cfg_off), .cfg_mask(cfg_mask), .ld_val(ld_val), .addr(addr), .out(out));
|
||||
integer n, cycles, loads, k, roll; reg [31:0] acc = 0; reg [3:0] opc; reg [31:0] w;
|
||||
// the class v4 draw: add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4 (sum 75)
|
||||
function [3:0] draw_op; input integer r; integer x;
|
||||
begin x = r % 75; if (x < 0) x = -x;
|
||||
draw_op = (x < 12) ? 4'd0 : (x < 22) ? 4'd2 : (x < 30) ? 4'd6 : (x < 38) ? 4'd8 : (x < 46) ? 4'd9 : (x < 53) ? 4'd4 : (x < 59) ? 4'd1 : (x < 65) ? 4'd7 : (x < 71) ? 4'd5 : 4'd3;
|
||||
end
|
||||
endfunction
|
||||
always #`HALF clk = ~clk;
|
||||
initial begin
|
||||
if (!$value$plusargs("cycles=%d", cycles)) cycles = 4000;
|
||||
if (!$value$plusargs("loads=%d", loads)) loads = 0; // 1: one load in 16 instructions
|
||||
$dumpfile("dump.vcd"); $dumpvars(0, tb.dut);
|
||||
repeat (4) @(negedge clk); rst = 0;
|
||||
// the era draw: constants and the program length
|
||||
@(negedge clk); cfg_en = 1; cfg_n = 8'd255; cfg_m = 32'h9e3779b1; cfg_r = 5'd13; cfg_wm = 32'h0fffffc0; cfg_off = 3; cfg_mask = 32'h0fffffff;
|
||||
@(negedge clk); cfg_en = 0;
|
||||
// the program: 256 instructions drawn with the class v4 weights
|
||||
for (k = 0; k < 256; k = k + 1) begin
|
||||
@(negedge clk); w = $random; opc = (loads && (k % 16 == 15)) ? 4'd12 : draw_op($random);
|
||||
prog_we = 1; prog_addr = k; prog_data = {w[31:4], opc};
|
||||
end
|
||||
@(negedge clk); prog_we = 0; run = 1;
|
||||
for (n = 0; n < cycles; n = n + 1) begin
|
||||
@(negedge clk); ld_val = $random; acc = acc ^ out ^ addr;
|
||||
end
|
||||
$display("CHECKSUM %08x", acc); $finish;
|
||||
end
|
||||
endmodule
|
||||
3
tools/chip-model/rtl/tb/tb_core_v6_32.v
Normal file
3
tools/chip-model/rtl/tb/tb_core_v6_32.v
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
`define TOP core_v6_32
|
||||
`define HALF 750
|
||||
`include "tb_core_common.vh"
|
||||
3
tools/chip-model/rtl/tb/tb_core_v6_32r16.v
Normal file
3
tools/chip-model/rtl/tb/tb_core_v6_32r16.v
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
`define TOP core_v6_32r16
|
||||
`define HALF 750
|
||||
`include "tb_core_common.vh"
|
||||
3
tools/chip-model/rtl/tb/tb_core_v6_8.v
Normal file
3
tools/chip-model/rtl/tb/tb_core_v6_8.v
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
`define TOP core_v6_8
|
||||
`define HALF 750
|
||||
`include "tb_core_common.vh"
|
||||
Loading…
Reference in a new issue