Files
FPGA-Neural/hardware/v3/sim/tb_spi_host_bridge_v3.v
T
micheleandClaude Sonnet 5 a4c080da83 feat: config-flash passthrough bridge via STARTUPE2, real board-exclusive flash access (EXP-0077)
Implements the user's board architecture: config flash wired
exclusively to the FPGA, host (ESP32) reaches it only through the
FPGA. flash_spi_master.v is a plain byte-wide SPI master using
STARTUPE2 to reclaim CCLK after configuration (the real, Xilinx-
documented "indirect SPI flash programming" technique, UG470 p94-96).
New opcode 0x40 FLASH_XFER in spi_host_bridge_v3.v relays bytes
byte-for-byte between host and the physical flash bus -- the host
decides the exact SPI NOR command sequence (verified against the real
W25Q32JV datasheet), this RTL knows nothing about flash semantics.

Found and fixed two real bugs during verification: a byte-assembly
off-by-one in flash_spi_master.v, and a genuine protocol-latency bug
in the FLASH_XFER opcode's response timing (needed 2 trailing margin
bytes, not 1 -- the internal flash transfer doesn't start until the
triggering byte finishes, so 1 byte of margin isn't enough). 39/39
tests pass end to end (host SPI -> bridge -> flash_spi_master ->
behavioral flash model).

Wired into n2_system_ddr3_top.v with real pin constraints (flash_mosi
=K17/flash_miso=K18/flash_cs_n=L13, the same pins reserved-but-unused
in EXP-0075) and BITSTREAM.CONFIG.PERSIST=FALSE made explicit.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MUG92aM9m68TRc4rG55BcC
2026-09-19 21:45:17 +02:00

458 lines
21 KiB
Verilog

`timescale 1ns/1ps
// ================================================================
// Isolated unit regression for spi_host_bridge_v3.v (V3 SPI opcode
// re-audit, this session). Mirrors hardware/v2/sim/tb_spi_host_
// bridge.v's own proven BFM/latency-model structure exactly, adapted
// for the new job_in_*/mem_* port shapes (16-byte WRITE_JOB, no
// required/producer_ids fields; 4-byte WRITE_MEM/READ_MEM address).
//
// Emulates: (1) neural_director_packed.v's job_in_ready contract (a
// level, deliberately delayed for a few cycles on the first job to
// prove job_in_valid is HELD, not pulsed blind); (2) host_mem_
// bridge.v's mem_ready contract (one clean req/ready handshake, fixed
// latency, backed by a simple model array standing in for real DDR3
// content -- host_mem_bridge.v itself is already independently
// verified in EXP-0071, so this test only needs to prove
// spi_host_bridge_v3.v drives ITS OWN side of that same word-
// granularity contract correctly).
// ================================================================
module tb_spi_host_bridge_v3;
localparam JOB_ADDR_WIDTH = 26;
localparam MEM_ADDR_WIDTH = 25;
reg clk = 0, rst = 1;
always #5 clk = ~clk; // 100MHz sim clock
reg sclk = 0, mosi = 0, cs_n = 1;
wire miso;
reg job_in_ready_model = 0;
wire job_in_valid;
wire [JOB_ADDR_WIDTH-1:0] job_in_x_base, job_in_w_base, job_in_result_addr;
wire [15:0] job_in_n_tiles, job_in_node_id;
wire mem_req, mem_wr, mem_lb_n, mem_ub_n;
wire [MEM_ADDR_WIDTH-1:0] mem_addr;
wire [15:0] mem_wdata;
reg [15:0] mem_rdata_model;
reg mem_ready_model = 0;
wire soft_rst_pulse;
reg init_calib_complete_model = 0;
reg dir_error_model = 0;
// ---- config-flash passthrough path: real flash_spi_master.v +
// the same behavioral W25Q32JV-like model used standalone in
// tb_flash_spi_master.v (EXP-0077), wired end to end through
// spi_host_bridge_v3.v's own new FLASH_XFER opcode ----
wire flash_xfer_active, flash_byte_req, flash_byte_done;
wire [7:0] flash_byte_wdata, flash_byte_rdata;
wire flash_cs_n, flash_mosi, flash_miso;
flash_spi_master u_flash (
.clk(clk), .rst(rst),
.xfer_active(flash_xfer_active), .byte_req(flash_byte_req),
.byte_wdata(flash_byte_wdata), .byte_rdata(flash_byte_rdata), .byte_done(flash_byte_done), .busy(),
.flash_cs_n(flash_cs_n), .flash_mosi(flash_mosi), .flash_miso(flash_miso)
);
flash_model_w25q32 u_flash_model (
.flash_cs_n(flash_cs_n), .flash_mosi(flash_mosi), .flash_miso(flash_miso),
.flash_sclk(u_flash.cclk_r)
);
spi_host_bridge_v3 #(
.JOB_ADDR_WIDTH(JOB_ADDR_WIDTH), .MEM_ADDR_WIDTH(MEM_ADDR_WIDTH), .N_SLOTS(2)
) dut (
.clk(clk), .rst(rst),
.flash_xfer_active(flash_xfer_active), .flash_byte_req(flash_byte_req),
.flash_byte_wdata(flash_byte_wdata), .flash_byte_rdata(flash_byte_rdata), .flash_byte_done(flash_byte_done),
.sclk(sclk), .mosi(mosi), .miso(miso), .cs_n(cs_n),
.init_calib_complete(init_calib_complete_model), .dir_error(dir_error_model),
.job_in_valid(job_in_valid), .job_in_ready(job_in_ready_model),
.job_in_x_base(job_in_x_base), .job_in_w_base(job_in_w_base),
.job_in_n_tiles(job_in_n_tiles), .job_in_result_addr(job_in_result_addr),
.job_in_node_id(job_in_node_id),
.mem_req(mem_req), .mem_wr(mem_wr), .mem_addr(mem_addr),
.mem_wdata(mem_wdata), .mem_lb_n(mem_lb_n), .mem_ub_n(mem_ub_n),
.mem_rdata(mem_rdata_model), .mem_ready(mem_ready_model),
.soft_rst_pulse(soft_rst_pulse)
);
// ---- simple backing memory model: fixed 6-cycle mem_ready latency ----
reg [15:0] mem_model [0:1023];
integer mem_latency_cnt;
reg mem_pending;
always @(posedge clk) begin
if (rst) begin
mem_ready_model <= 1'b0; mem_pending <= 1'b0; mem_latency_cnt <= 0;
end else begin
mem_ready_model <= 1'b0;
if (mem_req && !mem_pending) begin
mem_pending <= 1'b1;
mem_latency_cnt <= 6;
end else if (mem_pending) begin
if (mem_latency_cnt == 0) begin
mem_pending <= 1'b0;
mem_ready_model <= 1'b1;
if (mem_wr) mem_model[mem_addr[9:0]] <= mem_wdata;
else mem_rdata_model <= mem_model[mem_addr[9:0]];
end else begin
mem_latency_cnt <= mem_latency_cnt - 1;
end
end
end
end
// ---- SPI master BFM: mode 0, MSB-first (same timing as tb_spi_host_bridge.v) ----
task spi_byte(input [7:0] tx, output [7:0] rx);
integer i;
begin
rx = 8'h00;
for (i = 7; i >= 0; i = i - 1) begin
mosi = tx[i];
#200; sclk = 1; #50; rx = {rx[6:0], miso}; #50; sclk = 0; #200;
end
end
endtask
integer errors = 0, tests = 0;
task check(input cond, input [255:0] name);
begin
tests = tests + 1;
if (!cond) begin errors = errors + 1; $display("FAIL: %0s", name); end
else $display("PASS: %0s", name);
end
endtask
reg [7:0] rxb;
initial begin
rst = 1; cs_n = 1; sclk = 0; mosi = 0;
repeat (10) @(posedge clk);
rst = 0;
repeat (5) @(posedge clk);
// ================= Test A: WRITE_JOB (16 bytes), delayed job_in_ready =====
job_in_ready_model = 0;
cs_n = 0; #20;
spi_byte(8'h10, rxb); // opcode WRITE_JOB
spi_byte(8'h00, rxb); // node_id[15:8]
spi_byte(8'h05, rxb); // node_id[7:0] -> node_id=5
spi_byte(8'h00, rxb); // x_base[25:24]
spi_byte(8'h00, rxb); // x_base[23:16]
spi_byte(8'h10, rxb); // x_base[15:8]
spi_byte(8'h00, rxb); // x_base[7:0] -> x_base=0x001000
spi_byte(8'h00, rxb); // w_base[25:24]
spi_byte(8'h00, rxb); // w_base[23:16]
spi_byte(8'h20, rxb); // w_base[15:8]
spi_byte(8'h00, rxb); // w_base[7:0] -> w_base=0x002000
spi_byte(8'h00, rxb); // n_tiles[15:8]
spi_byte(8'h04, rxb); // n_tiles[7:0] -> n_tiles=4
spi_byte(8'h00, rxb); // result_addr[25:24]
spi_byte(8'h00, rxb); // result_addr[23:16]
spi_byte(8'h30, rxb); // result_addr[15:8]
spi_byte(8'h00, rxb); // result_addr[7:0] -> result_addr=0x003000
repeat (8) @(posedge clk);
check(job_in_valid == 1'b1, "A: job_in_valid asserted after 16th payload byte");
check(job_in_node_id == 16'h0005, "A: job_in_node_id");
check(job_in_x_base == 26'h001000, "A: job_in_x_base");
check(job_in_w_base == 26'h002000, "A: job_in_w_base");
check(job_in_n_tiles == 16'h0004, "A: job_in_n_tiles");
check(job_in_result_addr == 26'h003000, "A: job_in_result_addr");
repeat (3) begin
@(posedge clk);
check(job_in_valid == 1'b1, "A: job_in_valid still held while job_in_ready=0");
end
job_in_ready_model = 1;
@(posedge clk);
#1;
check(job_in_valid == 1'b0, "A: job_in_valid drops the cycle after job_in_ready seen");
job_in_ready_model = 0;
cs_n = 1; #40;
// ================= Test B: STATUS after accepted job ========
cs_n = 0; #20;
spi_byte(8'h20, rxb); // opcode STATUS
spi_byte(8'h00, rxb); // clocks out status byte
check(rxb[2] == 1'b1, "B: STATUS last_job_accepted=1");
check(rxb[0] == 1'b0, "B: STATUS job_busy=0 (already accepted)");
cs_n = 1; #40;
// ================= Test C: WRITE_MEM, single word (4-byte addr) =====
cs_n = 0; #20;
spi_byte(8'h01, rxb); // opcode WRITE_MEM
spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h55, rxb); // addr=0x000055
spi_byte(8'h00, rxb); spi_byte(8'h01, rxb); // len_words=1
spi_byte(8'h12, rxb); spi_byte(8'h34, rxb); // data=0x1234
#200;
cs_n = 1; #40;
check(mem_model[16'h0055] == 16'h1234, "C: WRITE_MEM wrote 0x1234 @ 0x000055");
// ================= Test D: READ_MEM, single word =============
cs_n = 0; #20;
spi_byte(8'h02, rxb); // opcode READ_MEM
spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h55, rxb); // addr=0x000055
spi_byte(8'h00, rxb); spi_byte(8'h01, rxb); // len_words=1
#200;
spi_byte(8'h00, rxb);
check(rxb == 8'h12, "D: READ_MEM MSB byte == 0x12");
spi_byte(8'h00, rxb);
check(rxb == 8'h34, "D: READ_MEM LSB byte == 0x34");
cs_n = 1; #40;
// ================= Test E: multi-word WRITE_MEM/READ_MEM, exercising
// the 25-bit MEM_ADDR_WIDTH's own top bit (addr near 2^24) =========
cs_n = 0; #20;
spi_byte(8'h01, rxb); // opcode WRITE_MEM
spi_byte(8'h01, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); // addr=0x1000000 (bit24=1)
spi_byte(8'h00, rxb); spi_byte(8'h02, rxb); // len_words=2
spi_byte(8'hAA, rxb); spi_byte(8'hBB, rxb); // word0=0xAABB
spi_byte(8'hCC, rxb); spi_byte(8'hDD, rxb); // word1=0xCCDD
#400;
cs_n = 1; #40;
check(mem_model[(25'h1000000) & 10'h3FF] == 16'hAABB, "E: WRITE_MEM word0 @ addr bit24 set");
check(mem_model[((25'h1000000)+1) & 10'h3FF] == 16'hCCDD, "E: WRITE_MEM word1 @ addr bit24 set");
// ================= Test G: REG_READ, DEVICE_ID (0x00) ========
cs_n = 0; #20;
spi_byte(8'h31, rxb); // opcode REG_READ
spi_byte(8'h00, rxb); // reg_addr=0x00 DEVICE_ID
spi_byte(8'h00, rxb); check(rxb == 8'h4E, "G: DEVICE_ID byte0 == 'N'");
spi_byte(8'h00, rxb); check(rxb == 8'h50, "G: DEVICE_ID byte1 == 'P'");
spi_byte(8'h00, rxb); check(rxb == 8'h56, "G: DEVICE_ID byte2 == 'V'");
spi_byte(8'h00, rxb); check(rxb == 8'h01, "G: DEVICE_ID byte3 == version 1");
cs_n = 1; #40;
// ================= Test H: REG_READ, N_SLOTS (0x03) ==========
cs_n = 0; #20;
spi_byte(8'h31, rxb);
spi_byte(8'h03, rxb);
spi_byte(8'h00, rxb); check(rxb == 8'h00, "H: N_SLOTS byte0 == 0");
spi_byte(8'h00, rxb); check(rxb == 8'h00, "H: N_SLOTS byte1 == 0");
spi_byte(8'h00, rxb); check(rxb == 8'h00, "H: N_SLOTS byte2 == 0");
spi_byte(8'h00, rxb); check(rxb == 8'h02, "H: N_SLOTS byte3 == 2 (matches N_SLOTS param)");
cs_n = 1; #40;
// ================= Test I: REG_READ, STATUS (0x02), with
// init_calib_complete and dir_error both driven high by the
// model, confirming they land in the right bits ============
init_calib_complete_model = 1'b1;
dir_error_model = 1'b1;
cs_n = 0; #20;
spi_byte(8'h31, rxb);
spi_byte(8'h02, rxb);
spi_byte(8'h00, rxb); check(rxb == 8'h00, "I: STATUS byte0 == 0 (bits[31:8] reserved)");
spi_byte(8'h00, rxb); check(rxb == 8'h00, "I: STATUS byte1 == 0");
spi_byte(8'h00, rxb); check(rxb == 8'h00, "I: STATUS byte2 == 0");
spi_byte(8'h00, rxb); check(rxb[3] == 1'b1, "I: STATUS bit3 == init_calib_complete");
check(rxb[4] == 1'b1, "I: STATUS bit4 == dir_error");
cs_n = 1; #40;
init_calib_complete_model = 1'b0;
dir_error_model = 1'b0;
// ================= Test J: REG_READ, unknown address =========
cs_n = 0; #20;
spi_byte(8'h31, rxb);
spi_byte(8'hEE, rxb); // unmapped register
spi_byte(8'h00, rxb); check(rxb == 8'hFF, "J: unmapped reg byte0 == 0xFF");
spi_byte(8'h00, rxb); check(rxb == 8'hFF, "J: unmapped reg byte1 == 0xFF");
spi_byte(8'h00, rxb); check(rxb == 8'hFF, "J: unmapped reg byte2 == 0xFF");
spi_byte(8'h00, rxb); check(rxb == 8'hFF, "J: unmapped reg byte3 == 0xFF (distinct from a real 0)");
cs_n = 1; #40;
// ================= Test K: REG_WRITE to CONTROL (0x01) bit0
// pulses soft_rst_pulse, same physical effect as RESET.
// UNLIKE the RESET opcode (which pulses only after CS rises),
// REG_WRITE applies immediately when its last data byte
// lands -- no backend handshake to wait on (see this module's
// own header). The watchdog must therefore run CONCURRENTLY
// with the last data byte's own spi_byte() call (a `fork`,
// same technique as tb_sdram_arbiter_n.v's own one-shot-pulse
// watchers), not after CS has already risen -- a first draft
// of this test watched only after CS rose and missed the
// pulse entirely (a testbench-timing bug, not an RTL one,
// confirmed via a DUT-internal trace before writing this). ==
cs_n = 0; #20;
spi_byte(8'h30, rxb); // opcode REG_WRITE
spi_byte(8'h01, rxb); // reg_addr=0x01 CONTROL
spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); // value bytes 31:8 = 0
begin : wait_reg_soft_rst
reg seen;
seen = 1'b0;
fork
spi_byte(8'h01, rxb); // value byte 7:0 = 1 (bit0 set) -- triggers the pulse
begin : watcher
integer wi;
for (wi = 0; wi < 410; wi = wi + 1) begin // covers spi_byte's own ~400-clk duration plus margin
@(posedge clk);
if (soft_rst_pulse) seen = 1'b1;
end
end
join
check(seen, "K: REG_WRITE CONTROL bit0 pulses soft_rst_pulse");
end
cs_n = 1; #40;
// ================= Test L: RESET opcode ======================
cs_n = 0; #20;
spi_byte(8'h0F, rxb); // opcode RESET
cs_n = 1;
begin : wait_soft_rst
integer wi; reg seen;
seen = 1'b0;
for (wi = 0; wi < 10; wi = wi + 1) begin
@(posedge clk);
if (soft_rst_pulse) seen = 1'b1;
end
check(seen, "L: soft_rst_pulse asserted after CS rises (within CDC latency)");
end
// ================= Test M: READ_MEM regression for the ROUT-
// exit bit_count==0 corruption (found via REG_READ this
// session, see spi_host_bridge_v3.v's own header note) --
// Test C/D's word 0x1234 has LSB byte 0x34 (bit0=0), which
// coincidentally matched the corrupted substitute's bit7=0
// and masked the bug. Use 0x5679 instead: LSB byte 0x79 =
// 0111_1001, bit0=1, which the (now-fixed) bug would have
// flipped to 0 (reading back 0x78 instead of 0x79). =========
cs_n = 0; #20;
spi_byte(8'h01, rxb);
spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h60, rxb); // addr=0x60
spi_byte(8'h00, rxb); spi_byte(8'h01, rxb); // len_words=1
spi_byte(8'h56, rxb); spi_byte(8'h79, rxb); // data=0x5679
#200;
cs_n = 1; #40;
cs_n = 0; #20;
spi_byte(8'h02, rxb);
spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h00, rxb); spi_byte(8'h60, rxb);
spi_byte(8'h00, rxb); spi_byte(8'h01, rxb);
#200;
spi_byte(8'h00, rxb); check(rxb == 8'h56, "M: READ_MEM MSB byte == 0x56");
spi_byte(8'h00, rxb); check(rxb == 8'h79, "M: READ_MEM LSB byte == 0x79 (bit0=1, catches the ROUT-exit bug)");
cs_n = 1; #40;
// ================= Test N: FLASH_XFER passthrough end to end
// -- real flash_spi_master.v + a real Winbond-command-set
// behavioral flash model behind it. Write Enable + Page
// Program + Read Data, entirely through spi_host_bridge_v3.v's
// own opcode 0x40, proving the WHOLE relay chain (host SPI ->
// this bridge -> flash_spi_master.v -> physical flash bus) is
// bit-exact, not just each half in isolation. ===============
cs_n = 0; #20;
spi_byte(8'h40, rxb); // opcode FLASH_XFER
spi_byte(8'h06, rxb); // relay: Write Enable
cs_n = 1; #40;
cs_n = 0; #20;
spi_byte(8'h40, rxb);
spi_byte(8'h02, rxb); // relay: Page Program
spi_byte(8'h30, rxb); // relay: addr=0x30
spi_byte(8'h5A, rxb); // relay: data=0x5A
spi_byte(8'h00, rxb); // trailing margin byte 1 of 2 -- see header's own real latency note
spi_byte(8'h00, rxb); // trailing margin byte 2 of 2
cs_n = 1; #40;
cs_n = 0; #20;
spi_byte(8'h40, rxb);
spi_byte(8'h03, rxb); // relay: Read Data
spi_byte(8'h30, rxb); // relay: addr=0x30
spi_byte(8'h00, rxb); // relay: dummy clock for the data byte
spi_byte(8'h00, rxb); // trailing margin byte 1 of 2
spi_byte(8'h00, rxb); // trailing margin byte 2 of 2 -- response is safely stable here
check(rxb == 8'h5A, "N: FLASH_XFER end-to-end round trip through the real flash model, bit-exact");
cs_n = 1; #40;
$display("=== tb_spi_host_bridge_v3: %0d/%0d PASS ===", tests-errors, tests);
if (errors != 0) $display("*** %0d FAILURES ***", errors);
$finish;
end
endmodule
// STARTUPE2 simulation-only stub -- see tb_flash_spi_master.v's own
// header for why real UNISIM verification is a disclosed follow-up,
// not done here (this test verifies the protocol/relay logic, which
// is independent of STARTUPE2's own real behavior).
module STARTUPE2 #(
parameter PROG_USR = "FALSE",
parameter real SIM_CCLK_FREQ = 0.0
)(
output wire CFGCLK, output wire CFGMCLK, output wire EOS, output wire PREQ,
input wire CLK, input wire GSR, input wire GTS, input wire KEYCLEARB, input wire PACK,
input wire USRCCLKO, input wire USRCCLKTS,
input wire USRDONEO, input wire USRDONETS
);
endmodule
// Same behavioral W25Q32JV-like flash model as tb_flash_spi_master.v
// (EXP-0077) -- kept independent (not shared via `include) since each
// testbench owns its own self-contained model, matching this
// project's existing convention (e.g. sdram_model.v is the one real
// exception, shared because it stands in for real vendor-supplied
// silicon behavior, not a test-specific convenience model).
module flash_model_w25q32 (
input wire flash_cs_n,
input wire flash_mosi,
output reg flash_miso,
input wire flash_sclk
);
reg [7:0] flash_mem [0:255];
reg [7:0] flash_cmd;
reg [7:0] flash_addr;
reg flash_wel;
reg [7:0] model_shift;
reg [2:0] model_bitcnt;
reg [2:0] model_bytecnt;
reg [7:0] model_rdata_byte;
initial begin flash_wel = 0; model_bytecnt = 0; model_bitcnt = 0; flash_miso = 0; end
always @(posedge flash_sclk) begin
if (!flash_cs_n) begin
model_shift <= {model_shift[6:0], flash_mosi};
if (model_bitcnt == 3'd7) begin
model_bitcnt <= 3'd0;
case (model_bytecnt)
3'd0: begin
flash_cmd <= {model_shift[6:0], flash_mosi};
if ({model_shift[6:0], flash_mosi} == 8'h06) flash_wel <= 1'b1;
model_bytecnt <= model_bytecnt + 1'b1;
end
3'd1: begin
if (flash_cmd == 8'h02 || flash_cmd == 8'h03) begin
flash_addr <= {model_shift[6:0], flash_mosi};
model_bytecnt <= model_bytecnt + 1'b1;
end
end
3'd2: begin
if (flash_cmd == 8'h02) flash_mem[flash_addr] <= {model_shift[6:0], flash_mosi};
model_bytecnt <= model_bytecnt + 1'b1;
end
default: ;
endcase
end else begin
model_bitcnt <= model_bitcnt + 1'b1;
end
end
end
always @(*) begin
if (flash_cmd == 8'h05) model_rdata_byte = {6'b0, flash_wel, 1'b0};
else if (flash_cmd == 8'h03) model_rdata_byte = flash_mem[flash_addr];
else model_rdata_byte = 8'h00;
end
always @(negedge flash_sclk) begin
if (!flash_cs_n && model_bytecnt >= (flash_cmd==8'h05 ? 3'd1 : 3'd2))
flash_miso <= model_rdata_byte[3'd7 - model_bitcnt];
end
always @(posedge flash_cs_n) begin
model_bytecnt <= 3'd0;
model_bitcnt <= 3'd0;
end
endmodule