feat: SDRAM 8MB->64MB upgrade (AS4C32M16SB-7BIN) + N_SLOTS=8 support

Memory upgrade, at the user's own explicit request: Alliance Memory
AS4C4M16SA-6TIN (64Mbit/8MB) -> AS4C32M16SB-7BIN (512Mbit/64MB, 54-ball
TFBGA), the largest same-family SDR SDRAM Alliance Memory offers.
Real-datasheet-driven (whole AS4C4M16SA/AS4C8M16SA/AS4C16M16SA/
AS4C32M16SA family investigated): 13 row bits (was 12, one new FPGA
pin sdram_a[12]/ball F1), 10 column bits (was 8), real -7-grade AC
timing (tRCD/tRP improved to 15ns, tREFI halved to 7.8us for the
doubled row count). sdram_controller.v and sdram_model.v gained real
ROW_BITS/COL_BITS/BANK_BITS parameters (was hardcoded 12/8/2).

ADDR_WIDTH widened 23->26 bits across the live instantiation tree.
This required a real SPI protocol change (spi_host_bridge.v): a 26-bit
byte address no longer fits in 3 bytes -- every address field widened
3->4 bytes (WRITE_JOB 15->18 payload bytes, WRITE_MEM/READ_MEM header
5->6 bytes).

Found and fixed two real timing regressions via nextpnr-ecp5 P&R
(not assumed): neural_director.v's own runtime-indexed demux write
(ERR-0027, was silently synthesizing an extra MULT18X18D) and
nms_activation_fill_ctrl_v3.v's own linear N_SLOTS-wide max-scan
(ERR-0028, became dominant at N_SLOTS=8) -- both replaced with
constant-indexed/tree-based equivalents, bit-exact same behavior,
confirmed via full D-Stress N=2/4/8 regression (identical cycle
counts). N_SLOTS=4 now fully closes timing at 64MHz (8/8 seeds);
N_SLOTS=8 significantly improved but not yet fully reliable (5/8
seeds) -- honestly disclosed, not claimed complete.

Full regression re-verified: sdram_controller (461/461, 18 configs),
tb_sdram_boundary (21/21), D-Stress N=2/4/8 (bit-exact), spi_host_bridge
(18/18), board-level SPI smoke test (11/11), unified backend (40/40).

See hardware/v2/docs/MEMORY_UPGRADE_64MB_N8.md for the full
investigation, and errors.log/decisions.log (ERR-0027, ERR-0028,
DEC-0039) for the complete root-cause writeups.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_013xXuuRUWZScuo1DeYJxs3v
This commit is contained in:
2026-09-07 00:20:53 +02:00
co-authored by Claude Sonnet 5
parent 9b5d1055b8
commit 8d83d97bde
21 changed files with 558 additions and 295 deletions
+75 -22
View File
@@ -30,7 +30,7 @@
// ================================================================
module neural_director #(
parameter ADDR_WIDTH = 23,
parameter ADDR_WIDTH = 26,
parameter N_SLOTS = 4,
parameter QUEUE_DEPTH = 8
)(
@@ -48,11 +48,16 @@ module neural_director #(
input wire [15:0] job_in_node_id,
// ---- per-slot memory_manager job control (arrayed, §9) ----
output reg [N_SLOTS-1:0] slot_job_start,
output reg [ADDR_WIDTH*N_SLOTS-1:0] slot_x_base,
output reg [ADDR_WIDTH*N_SLOTS-1:0] slot_w_base,
output reg [16*N_SLOTS-1:0] slot_n_tiles,
output reg [ADDR_WIDTH*N_SLOTS-1:0] slot_result_addr,
// Ports are `wire`, driven by the GEN_SLOT_OUT generate block below
// from internal unpacked-array registers (see that block's own
// comment for why -- a real, measured timing regression found this
// session when ADDR_WIDTH grew from 23 to 26 bits, section
// "post-PRE-PCB-FREEZE memory upgrade").
output wire [N_SLOTS-1:0] slot_job_start,
output wire [ADDR_WIDTH*N_SLOTS-1:0] slot_x_base,
output wire [ADDR_WIDTH*N_SLOTS-1:0] slot_w_base,
output wire [16*N_SLOTS-1:0] slot_n_tiles,
output wire [ADDR_WIDTH*N_SLOTS-1:0] slot_result_addr,
// slot_node_id: which node_id is currently occupying each slot --
// not needed by memory_manager itself (it has no notion of node
// ids), but needed by a caller (dataflow_core.v, M7) that must
@@ -60,7 +65,7 @@ module neural_director #(
// to notify the Dependency Manager (M6). Purely additive: existing
// callers (hardware/v2/sim/tb_neural_director.v, M5) that don't
// connect it are unaffected.
output reg [16*N_SLOTS-1:0] slot_node_id,
output wire [16*N_SLOTS-1:0] slot_node_id,
input wire [N_SLOTS-1:0] slot_job_done,
// ---- completion notification (§9 "rilevamento dei completamenti") ----
@@ -121,6 +126,45 @@ module neural_director #(
end
end
// ---- per-slot output storage (unpacked arrays, one real register
// set per slot) + constant-indexed generate wiring out to the
// packed ports above. Found and fixed this session (post-PRE-PCB-
// FREEZE memory upgrade, ADDR_WIDTH 23->26): the PREVIOUS design
// used one wide packed `output reg` per field and wrote it with a
// RUNTIME-computed part-select (`slot_x_base[free_slot_idx*
// ADDR_WIDTH +: ADDR_WIDTH] <= ...`). A variable-indexed write into
// a wide packed register is not free logic -- Yosys/synth_ecp5
// synthesized the index computation (`free_slot_idx*ADDR_WIDTH`)
// as an actual MULT18X18D hard multiplier feeding a wide demux/
// crossbar into the destination slot, and this got measurably
// worse as ADDR_WIDTH grew (real nextpnr-ecp5 P&R: worst-seed Fmax
// collapsed from 68.51MHz at ADDR_WIDTH=23 to ~40-47MHz at
// ADDR_WIDTH=26, confirmed across 8 seeds, all failing the 64MHz
// target). The fix below replaces the runtime-indexed demux write
// with N_SLOTS parallel CONSTANT-indexed comparisons (`fi ==
// free_slot_idx`, each a cheap few-bit compare, no multiply) each
// gating its own slot's own narrow register -- functionally
// IDENTICAL behavior, bit-exact same external port semantics, only
// the internal implementation changed.
reg slot_job_start_r [0:N_SLOTS-1];
reg [ADDR_WIDTH-1:0] slot_x_base_r [0:N_SLOTS-1];
reg [ADDR_WIDTH-1:0] slot_w_base_r [0:N_SLOTS-1];
reg [15:0] slot_n_tiles_r [0:N_SLOTS-1];
reg [ADDR_WIDTH-1:0] slot_result_addr_r [0:N_SLOTS-1];
reg [15:0] slot_node_id_r [0:N_SLOTS-1];
genvar gs;
generate
for (gs = 0; gs < N_SLOTS; gs = gs + 1) begin : GEN_SLOT_OUT
assign slot_job_start[gs] = slot_job_start_r[gs];
assign slot_x_base[gs*ADDR_WIDTH +: ADDR_WIDTH] = slot_x_base_r[gs];
assign slot_w_base[gs*ADDR_WIDTH +: ADDR_WIDTH] = slot_w_base_r[gs];
assign slot_n_tiles[gs*16 +: 16] = slot_n_tiles_r[gs];
assign slot_result_addr[gs*ADDR_WIDTH +: ADDR_WIDTH] = slot_result_addr_r[gs];
assign slot_node_id[gs*16 +: 16] = slot_node_id_r[gs];
end
endgenerate
// Priority-encoded lowest-indexed slot reporting job_done this
// cycle (combinational, so it reflects THIS cycle's slot_job_done
// bus directly -- a register-based "already reported one" flag
@@ -143,16 +187,18 @@ module neural_director #(
q_tail <= {Q_ADDR_WIDTH{1'b0}};
q_count <= {(Q_ADDR_WIDTH+1){1'b0}};
slot_busy <= {N_SLOTS{1'b0}};
slot_job_start <= {N_SLOTS{1'b0}};
slot_x_base <= {(ADDR_WIDTH*N_SLOTS){1'b0}};
slot_w_base <= {(ADDR_WIDTH*N_SLOTS){1'b0}};
slot_n_tiles <= {(16*N_SLOTS){1'b0}};
slot_result_addr <= {(ADDR_WIDTH*N_SLOTS){1'b0}};
slot_node_id <= {(16*N_SLOTS){1'b0}};
for (fi = 0; fi < N_SLOTS; fi = fi + 1) begin
slot_job_start_r[fi] <= 1'b0;
slot_x_base_r[fi] <= {ADDR_WIDTH{1'b0}};
slot_w_base_r[fi] <= {ADDR_WIDTH{1'b0}};
slot_n_tiles_r[fi] <= 16'b0;
slot_result_addr_r[fi] <= {ADDR_WIDTH{1'b0}};
slot_node_id_r[fi] <= 16'b0;
end
job_out_done <= 1'b0;
job_out_slot <= '0;
end else begin
slot_job_start <= {N_SLOTS{1'b0}};
for (fi = 0; fi < N_SLOTS; fi = fi + 1) slot_job_start_r[fi] <= 1'b0;
job_out_done <= 1'b0;
// ---- Accept a new job into the ready queue (independent
@@ -201,14 +247,21 @@ module neural_director #(
end
DIR_ALLOCATE: begin
// Dispatch the head of the queue to the first
// free slot found this cycle.
slot_job_start[free_slot_idx] <= 1'b1;
slot_x_base[free_slot_idx*ADDR_WIDTH +: ADDR_WIDTH] <= q_x_base[q_head];
slot_w_base[free_slot_idx*ADDR_WIDTH +: ADDR_WIDTH] <= q_w_base[q_head];
slot_n_tiles[free_slot_idx*16 +: 16] <= q_n_tiles[q_head];
slot_result_addr[free_slot_idx*ADDR_WIDTH +: ADDR_WIDTH] <= q_result_addr[q_head];
slot_node_id[free_slot_idx*16 +: 16] <= q_node_id[q_head];
// Dispatch the head of the queue to the first free
// slot found this cycle -- N_SLOTS parallel
// constant-indexed compares (cheap) instead of one
// runtime-indexed wide demux write (see
// slot_x_base_r's own declaration comment for why).
for (fi = 0; fi < N_SLOTS; fi = fi + 1) begin
if (fi[$clog2(N_SLOTS)-1:0] == free_slot_idx) begin
slot_job_start_r[fi] <= 1'b1;
slot_x_base_r[fi] <= q_x_base[q_head];
slot_w_base_r[fi] <= q_w_base[q_head];
slot_n_tiles_r[fi] <= q_n_tiles[q_head];
slot_result_addr_r[fi] <= q_result_addr[q_head];
slot_node_id_r[fi] <= q_node_id[q_head];
end
end
slot_busy[free_slot_idx] <= 1'b1;
q_head <= (q_head == QUEUE_DEPTH[Q_ADDR_WIDTH-1:0]-1'b1) ? {Q_ADDR_WIDTH{1'b0}} : q_head + 1'b1;
dir_state <= DIR_SCAN_READY;