Files
FPGA-Neural/hardware/v1/sim/spi_slave_tb.v
T
micheleandClaude Sonnet 5 dc0b331d3e feat(v2): scaffold hardware/v1 frozen baseline + M1 Neural Processor
Begins the V2 Neural Multiprocessor / Dataflow architecture per
docs/v2-description.md, per explicit user request to freeze V1 and
start V2 development, copying from V1 what's needed.

Scaffold:
- hardware/v1/: byte-exact, read-only copy of the current V1 codebase
  (rtl, testbenches, tools, constraints, a representative subset of
  synthesis results, and reference docs) -- verified identical via
  diff/cmp against the live top-level tree before being made
  filesystem-read-only. The live top-level tree is untouched and
  remains the project's "production" V1 (see hardware/v1/README.md
  and hardware/v2/logs/decisions.log DEC-0001 for why copy-not-move).
- hardware/v2/: mandatory structure (rtl/sim/constraints/synthesis/
  reports/scripts/logs/docs) plus the full logging system required by
  the spec (development/architecture/simulation/synthesis/timing/
  benchmark/decisions/experiments/errors.log).

M1 -- Neural Processor (hardware/v2/rtl/neural_processor.v):
- 8-stage pipelined perceptron unit (P_IN=8): input align, 8
  multipliers, 3-level adder tree, accumulator, bias+activation, INT8
  saturation. Genuine 1-tile/cycle throughput, not just a wider
  combinational datapath.
- 7-state FSM (NP_IDLE..NP_ERROR per docs/v2-description.md §6, with
  4 baseline states merged into NP_WAIT_OPERANDS -- see
  decisions.log DEC-0002); valid/ready/data/last stream interfaces
  per §7.
- Bit-exact vs the frozen hardware/v1/rtl/neuron_parallel.v + mac8.v
  + mac_unit.v: 7/7 tests pass (hardware/v2/sim/tb_neural_processor.v),
  covering regular/mixed-sign/extreme-INT8 vectors, both activations,
  a zero-idle-gap back-to-back-tiles throughput check, and an 8-tile
  job -- verified with Verilator (see below for why).
- Real synthesis + place&route (Yosys + nextpnr-ecp5): 0 CHECK
  problems, Fmax 183.12 MHz at ACC_WIDTH=32 (PASS at 80MHz, ~3x V1's
  isolated PARALLEL=8 Fmax of 61.71 MHz) and 176.21 MHz at ACC_WIDTH=24
  (a user-requested comparison experiment, also bit-exact-verified;
  see experiments.log EXP-0001/EXP-0002 and benchmark.log).

Three real bugs found and resolved during M1 development (full
diagnostic record in errors.log):
- Two independent, reproducible Icarus Verilog v13.0 scheduling
  defects (ERR-0001, ERR-0002) that silently produced wrong simulation
  results for standard sequential Verilog -- confirmed via Verilator
  5.050 giving correct results on the same minimal repros. Verilator
  is now the trusted simulator for hardware/v2/ (decisions.log
  DEC-0004); Icarus's affected protocol-violation check was removed
  from the RTL and deferred architecturally to the Neural Director
  (DEC-0003) rather than chased further.
- One real RTL bug (ERR-0003): last0 wasn't gated like valid0,
  letting a "last tile" tag leak into the pipeline ahead of its
  actual valid tile on back-to-back jobs. Fixed and verified.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_013xXuuRUWZScuo1DeYJxs3v
2026-09-05 14:06:53 +02:00

381 lines
13 KiB
Verilog

`timescale 1ns/1ps
// ================================================================
// SPI_SLAVE PHYSICAL LAYER TESTBENCH
//
// Bit-bangs a simulated SPI master (Mode 0, MSB-first) against
// rtl/spi_slave.v and checks:
// TEST 1: single-byte transaction (rx_byte/rx_valid, MISO readback)
// TEST 2: multi-byte transaction within one CS-low period
// TEST 3: back-to-back separate transactions (state resets cleanly)
// TEST 4: a slower SPI clock (stresses nothing new, but confirms
// the module isn't implicitly tied to one SCLK/clk ratio)
// ================================================================
module tb;
localparam CLK_PERIOD = 12.5; // 80 MHz system clock
reg clk;
reg rst;
initial begin
clk = 1'b0;
forever #(CLK_PERIOD / 2.0) clk = ~clk;
end
reg sclk;
reg mosi;
wire miso;
reg cs_n;
wire [7:0] rx_byte;
wire rx_valid;
reg [7:0] tx_byte;
wire tx_byte_req;
wire cs_active;
wire cs_start;
wire cs_end;
spi_slave dut (
.clk(clk),
.rst(rst),
.sclk(sclk),
.mosi(mosi),
.miso(miso),
.cs_n(cs_n),
.rx_byte(rx_byte),
.rx_valid(rx_valid),
.tx_byte(tx_byte),
.tx_byte_req(tx_byte_req),
.cs_active(cs_active),
.cs_start(cs_start),
.cs_end(cs_end)
);
// ============================================================
// tx_byte queue: serves tx_queue[tx_queue_idx] combinationally
// at all times (spi_slave.v prefetches it via tx_byte_req).
//
// The index advances on `rx_valid`, NOT on `tx_byte_req`:
// tx_byte_req fires one extra ("phantom") time after the last
// byte of every transaction (see the contract note in
// rtl/spi_slave.v), while rx_valid fires exactly once per REAL
// byte transferred, in both directions (SPI is full-duplex) --
// the correct signal to retire one queue entry.
// ============================================================
reg [7:0] tx_queue [0:7];
integer tx_queue_len;
integer tx_queue_idx;
always @(posedge clk) begin
if (rst) begin
tx_queue_idx <= 0;
end else if (rx_valid) begin
if (tx_queue_idx < tx_queue_len)
tx_queue_idx <= tx_queue_idx + 1;
end
end
always @(*) begin
tx_byte = (tx_queue_idx < tx_queue_len) ? tx_queue[tx_queue_idx] : 8'h00;
end
// ============================================================
// rx capture: record every received byte in order
// ============================================================
reg [7:0] rx_log [0:7];
integer rx_log_len;
always @(posedge clk) begin
if (rst) begin
rx_log_len <= 0;
end else if (rx_valid) begin
rx_log[rx_log_len] <= rx_byte;
rx_log_len <= rx_log_len + 1;
end
end
// ============================================================
// cs_start / cs_end pulse counters
// ============================================================
integer cs_start_count;
integer cs_end_count;
always @(posedge clk) begin
if (rst) begin
cs_start_count <= 0;
cs_end_count <= 0;
end else begin
if (cs_start) cs_start_count <= cs_start_count + 1;
if (cs_end) cs_end_count <= cs_end_count + 1;
end
end
// ============================================================
// SPI MASTER BFM (Mode 0, MSB-first, bit-banged)
//
// half_period is in ns; must stay large enough relative to
// CLK_PERIOD for the 2-flop CDC synchronizer in spi_slave.v to
// reliably catch every edge (>= ~3 system clocks per SCLK
// half-period is a safe margin).
// ============================================================
reg [7:0] miso_capture [0:7];
integer miso_capture_len;
// clk-cycle-counted wait: every SPI edge in this BFM is placed a
// fixed number of `clk` cycles apart, instead of a raw `#ns`
// delay. This keeps the master deterministically phase-aligned
// to the system clock, so the fixed CDC latency of spi_slave.v
// (3-stage synchronizer + 1 cycle for edge detect, ~4 clk
// cycles) always falls comfortably inside the margin instead of
// drifting against it run to run.
task clk_wait;
input integer n;
integer k;
begin
for (k = 0; k < n; k = k + 1)
@(posedge clk);
end
endtask
task spi_begin;
input integer half_bit_cycles;
begin
cs_n = 1'b1;
sclk = 1'b0;
mosi = 1'b0;
clk_wait(half_bit_cycles * 2);
cs_n = 1'b0;
clk_wait(half_bit_cycles * 2);
end
endtask
task spi_end;
input integer half_bit_cycles;
begin
clk_wait(half_bit_cycles * 2);
cs_n = 1'b1;
clk_wait(half_bit_cycles * 2);
end
endtask
task spi_xfer_byte;
input [7:0] tx;
input integer half_bit_cycles;
output [7:0] rx;
integer i;
reg [7:0] rx_acc;
begin
rx_acc = 8'h00;
for (i = 7; i >= 0; i = i - 1) begin
mosi = tx[i];
clk_wait(half_bit_cycles);
sclk = 1'b1; // rising edge: slave samples MOSI
rx_acc[i] = miso; // master samples MISO (stable since the prior falling edge)
clk_wait(half_bit_cycles);
sclk = 1'b0; // falling edge: slave updates MISO
clk_wait(half_bit_cycles);
end
rx = rx_acc;
end
endtask
reg [7:0] rx_tmp;
integer errors;
integer errors_before;
// ============================================================
// MAIN
// ============================================================
initial begin
$dumpfile("sim/spi_slave.vcd");
$dumpvars(0, tb);
rst = 1'b1;
cs_n = 1'b1;
sclk = 1'b0;
mosi = 1'b0;
errors = 0;
tx_queue_len = 0;
rx_log_len = 0;
repeat (5) @(posedge clk);
rst = 1'b0;
repeat (5) @(posedge clk);
$display("");
$display("========================================");
$display("SPI_SLAVE PHYSICAL LAYER TEST");
$display("========================================");
// --------------------------------------------------------
// TEST 1: single-byte transaction
// Master sends 0xA5, slave echoes back queued 0x3C.
// --------------------------------------------------------
errors_before = errors;
tx_queue[0] = 8'h3C;
tx_queue_len = 1;
spi_begin(8);
spi_xfer_byte(8'hA5, 8, rx_tmp);
spi_end(8);
@(posedge clk); @(posedge clk);
$display("");
$display("TEST 1: single byte");
$display(" MOSI sent = 0xA5, slave rx_byte = 0x%02x (expect 0xA5)", rx_log[0]);
$display(" MISO sent = 0x3C, master received = 0x%02x (expect 0x3C)", rx_tmp);
$display(" cs_start pulses = %0d (expect 1), cs_end pulses = %0d (expect 1)",
cs_start_count, cs_end_count);
if (rx_log[0] !== 8'hA5) begin $display(" FAIL: rx_byte mismatch"); errors = errors + 1; end
if (rx_tmp !== 8'h3C) begin $display(" FAIL: MISO readback mismatch"); errors = errors + 1; end
if (cs_start_count !== 1) begin $display(" FAIL: cs_start count"); errors = errors + 1; end
if (cs_end_count !== 1) begin $display(" FAIL: cs_end count"); errors = errors + 1; end
if (errors == errors_before) $display(" PASS");
// --------------------------------------------------------
// TEST 2: multi-byte transaction, single CS-low period
// Master sends 0x11, 0x22, 0x33, 0x44.
// Slave echoes back 0xDE, 0xAD, 0xBE, 0xEF.
// --------------------------------------------------------
@(negedge clk); rst = 1'b1; @(negedge clk); rst = 1'b0; @(posedge clk);
rx_log_len = 0; cs_start_count = 0; cs_end_count = 0;
errors_before = errors;
tx_queue[0] = 8'hDE;
tx_queue[1] = 8'hAD;
tx_queue[2] = 8'hBE;
tx_queue[3] = 8'hEF;
tx_queue_len = 4;
spi_begin(8);
spi_xfer_byte(8'h11, 8, rx_tmp); miso_capture[0] = rx_tmp;
spi_xfer_byte(8'h22, 8, rx_tmp); miso_capture[1] = rx_tmp;
spi_xfer_byte(8'h33, 8, rx_tmp); miso_capture[2] = rx_tmp;
spi_xfer_byte(8'h44, 8, rx_tmp); miso_capture[3] = rx_tmp;
spi_end(8);
@(posedge clk); @(posedge clk);
$display("");
$display("TEST 2: multi-byte, one CS period");
$display(" rx_log = %02x %02x %02x %02x (expect 11 22 33 44)",
rx_log[0], rx_log[1], rx_log[2], rx_log[3]);
$display(" miso = %02x %02x %02x %02x (expect de ad be ef)",
miso_capture[0], miso_capture[1], miso_capture[2], miso_capture[3]);
$display(" cs_start pulses = %0d (expect 1), cs_end pulses = %0d (expect 1)",
cs_start_count, cs_end_count);
if (rx_log[0] !== 8'h11 || rx_log[1] !== 8'h22 ||
rx_log[2] !== 8'h33 || rx_log[3] !== 8'h44) begin
$display(" FAIL: rx sequence mismatch");
errors = errors + 1;
end
if (miso_capture[0] !== 8'hDE || miso_capture[1] !== 8'hAD ||
miso_capture[2] !== 8'hBE || miso_capture[3] !== 8'hEF) begin
$display(" FAIL: MISO sequence mismatch");
errors = errors + 1;
end
if (cs_start_count !== 1) begin $display(" FAIL: cs_start count"); errors = errors + 1; end
if (cs_end_count !== 1) begin $display(" FAIL: cs_end count"); errors = errors + 1; end
if (errors == errors_before) $display(" PASS");
// --------------------------------------------------------
// TEST 3: back-to-back separate transactions
// Two independent single-byte transactions; state must
// reset cleanly between them (no leftover bit_count/shift).
// --------------------------------------------------------
@(negedge clk); rst = 1'b1; @(negedge clk); rst = 1'b0; @(posedge clk);
rx_log_len = 0; cs_start_count = 0; cs_end_count = 0;
errors_before = errors;
tx_queue[0] = 8'h01;
tx_queue_len = 1;
spi_begin(8);
spi_xfer_byte(8'h7E, 8, rx_tmp);
spi_end(8);
repeat (10) @(posedge clk);
tx_queue[0] = 8'h02;
tx_queue_len = 1;
spi_begin(8);
spi_xfer_byte(8'h81, 8, rx_tmp);
spi_end(8);
@(posedge clk); @(posedge clk);
$display("");
$display("TEST 3: back-to-back transactions");
$display(" rx_log = %02x %02x (expect 7e 81)", rx_log[0], rx_log[1]);
$display(" cs_start pulses = %0d (expect 2), cs_end pulses = %0d (expect 2)",
cs_start_count, cs_end_count);
if (rx_log[0] !== 8'h7E || rx_log[1] !== 8'h81) begin
$display(" FAIL: rx sequence mismatch");
errors = errors + 1;
end
if (cs_start_count !== 2) begin $display(" FAIL: cs_start count"); errors = errors + 1; end
if (cs_end_count !== 2) begin $display(" FAIL: cs_end count"); errors = errors + 1; end
if (errors == errors_before) $display(" PASS");
// --------------------------------------------------------
// TEST 4: slower SPI clock (larger half_period), same
// single-byte check, confirms no hidden dependency on a
// specific SCLK/clk ratio (as long as the CDC margin holds).
// --------------------------------------------------------
@(negedge clk); rst = 1'b1; @(negedge clk); rst = 1'b0; @(posedge clk);
rx_log_len = 0; cs_start_count = 0; cs_end_count = 0;
errors_before = errors;
tx_queue[0] = 8'h5A;
tx_queue_len = 1;
spi_begin(20);
spi_xfer_byte(8'h96, 20, rx_tmp);
spi_end(20);
@(posedge clk); @(posedge clk);
$display("");
$display("TEST 4: slower SCLK (200ns half-period)");
$display(" rx_byte = 0x%02x (expect 0x96), MISO = 0x%02x (expect 0x5a)",
rx_log[0], rx_tmp);
if (rx_log[0] !== 8'h96) begin $display(" FAIL: rx_byte mismatch"); errors = errors + 1; end
if (rx_tmp !== 8'h5A) begin $display(" FAIL: MISO readback mismatch"); errors = errors + 1; end
if (errors == errors_before) $display(" PASS");
$display("");
$display("========================================");
if (errors == 0)
$display("SPI_SLAVE TEST PASSED");
else
$display("SPI_SLAVE TEST FAILED: %0d errors", errors);
$display("========================================");
$display("");
$finish;
end
endmodule