`timescale 1ns/1ps // ============================================================ // V3 -- result_writeback.v: real result-writeback engine, closing the // gap disclosed since packed_slot.v's own original header ("no // result-writeback engine exists yet either -- result_addr_a/b are // passed through unused") and flagged as a hard scaling blocker // (docs/ARCHITECTURE_ANALYSIS.md S4.6/S5.3): literal top-level // result_data_a/b pins do not scale past a handful of cores (8 bits x // 2 lanes x N cores -- at N=16 that's 256 pins on this port alone). // // REAL FIX: write each completed job's result INTO DDR3 at the job's // own result_addr_a/result_addr_b (already carried through packed_ // slot.v's own interface, previously unused), reusing the SAME shared // ctrl port packed_slot.v already time-multiplexes among its other // sub-engines (layer_prefetch_ctrl.v / ddr_prefetch_mgr.v) -- same // architectural shape as the weight-fetch path, in reverse. The host // reads results back via the ALREADY-EXISTING READ_MEM (0x02) opcode // -- no new SPI protocol needed. // // REAL ADDRESSING (verified against act_tile_fetch.v's/layer_ // prefetch_ctrl.v's own real address-computation code, not guessed): // result_addr_a/b arrive in packed_slot.v's own JOB_ADDR_WIDTH=26-bit // convention. Exactly like x_base_a/w_base already do, the LOW // ADDR_WIDTH=25 bits (dropping the unused top/MSB headroom bit) are // used DIRECTLY as a ctrl-bus-native 32-bit-word address -- the SAME // address space act_tile_fetch.v's own ctrl_addr already lives in. // ONE full 32-bit ctrl-word is written per lane: // {node_id[15:0], 8'h00, result_data[7:0]} (low 16 bits = // zero-extended 8-bit result value, high 16 bits = node_id). // // REAL, DISCLOSED HOST-FIRMWARE IMPLICATION (not yet built, same as // this project's other disclosed host-firmware gaps, e.g. JTAG // bit-banging): reading a written result back via the EXISTING // READ_MEM (16-bit-word-addressed) opcode needs // `mem_addr = result_addr[24:0]*2` for the value and // `mem_addr = result_addr[24:0]*2 + 1` for node_id (2 host reads per // lane, since READ_MEM's own mem_addr is 16-bit-word-granular while // this engine writes a native 32-bit ctrl-word -- see host_mem_ // bridge.v's own header for the real reason that halving exists). // // WMASK CONVENTION (matches host_mem_bridge.v's own real, already- // working pattern exactly, not reinvented): 0 = write this byte, 1 = // masked -- the same DQM-style polarity this project's whole memory // stack already uses end to end. // // TWO LANES, ONE TRANSACTION EACH, SEQUENTIAL: lane A's write // completes fully (through its own ctrl_ready) before lane B's own // starts -- mirrors act_tile_fetch.v's own "lane A then lane B" // sequencing for its two burst reads, the same discipline already // proven safe on this shared bus. // ============================================================ module result_writeback #( parameter BURST_LEN = 8, parameter DATA_WIDTH = 8, parameter JOB_ADDR_WIDTH = 26, parameter ADDR_WIDTH = 25 // ctrl-bus-native word address, matches sdram_arbiter_n.v's own convention )( input wire clk, input wire rst, // one-shot request: pulse `start` with all fields valid the same // cycle (matches this project's own established one-shot-pulse- // requester discipline, EXP-0066). input wire start, input wire [JOB_ADDR_WIDTH-1:0] result_addr_a, input wire [JOB_ADDR_WIDTH-1:0] result_addr_b, input wire signed [DATA_WIDTH-1:0] result_data_a, input wire signed [DATA_WIDTH-1:0] result_data_b, input wire [15:0] result_node_id_a, input wire [15:0] result_node_id_b, output wire busy, output reg done, // one-cycle pulse // ---- shared ctrl port (packed_slot.v's own local mux gates this // the same way it already gates pf_ctrl_*/act_ctrl_*) ---- output wire mem_active, input wire mem_grant, output reg ctrl_req, output reg ctrl_wr, output reg [ADDR_WIDTH-1:0] ctrl_addr, output reg [32*BURST_LEN-1:0] ctrl_wdata, output reg [4*BURST_LEN-1:0] ctrl_wmask, input wire [32*BURST_LEN-1:0] ctrl_rdata, input wire ctrl_ready, input wire ctrl_busy ); localparam ALIGN_BITS = $clog2(BURST_LEN); // 3: which of the BURST_LEN 32-bit words in the burst localparam S_IDLE = 3'd0, S_MEMWAIT = 3'd1, S_XFER_A = 3'd2, S_GAP = 3'd3, // wait for ctrl_busy to clear before firing lane B's request S_XFER_B = 3'd4, S_DONE = 3'd5; reg [2:0] state; reg [DATA_WIDTH-1:0] data_a_lat, data_b_lat; reg [15:0] nid_a_lat, nid_b_lat; reg [ADDR_WIDTH-1:0] word_addr_a_lat, word_addr_b_lat; assign busy = (state != S_IDLE); // real, established discipline (EXP-0066): mem_active must be // visible to the arbiter the SAME cycle this module first wants // the bus, i.e. as soon as it leaves S_IDLE -- not only once a // transaction is actually in flight. assign mem_active = (state != S_IDLE); // real ctrl-bus-native word address: low ADDR_WIDTH bits of the // JOB_ADDR_WIDTH job address -- the exact same truncation act_ // tile_fetch.v/layer_prefetch_ctrl.v already apply to x_base_a/ // w_base (verified against their own real code, not guessed). wire [ADDR_WIDTH-1:0] word_addr_a = result_addr_a[ADDR_WIDTH-1:0]; wire [ADDR_WIDTH-1:0] word_addr_b = result_addr_b[ADDR_WIDTH-1:0]; always @(posedge clk) begin if (rst) begin state <= S_IDLE; ctrl_req <= 1'b0; ctrl_wr <= 1'b0; done <= 1'b0; end else begin ctrl_req <= 1'b0; done <= 1'b0; case (state) S_IDLE: begin if (start) begin data_a_lat <= result_data_a; data_b_lat <= result_data_b; nid_a_lat <= result_node_id_a; nid_b_lat <= result_node_id_b; word_addr_a_lat <= word_addr_a; word_addr_b_lat <= word_addr_b; state <= S_MEMWAIT; end end // real, established discipline (EXP-0066): never issue // ctrl_req before mem_grant is actually observed -- a // blind/early ctrl_req on a shared, arbitrated bus can // lose the request permanently. S_MEMWAIT: begin if (mem_grant) begin ctrl_req <= 1'b1; ctrl_wr <= 1'b1; ctrl_addr <= {word_addr_a_lat[ADDR_WIDTH-1:ALIGN_BITS], {ALIGN_BITS{1'b0}}}; ctrl_wdata <= {(BURST_LEN){nid_a_lat, 8'h00, data_a_lat}}; ctrl_wmask <= ~({{(4*BURST_LEN-4){1'b0}}, 4'hF} << (word_addr_a_lat[ALIGN_BITS-1:0] * 4)); state <= S_XFER_A; end end S_XFER_A: begin if (ctrl_ready) begin state <= S_GAP; end end // mig_native_adapter.v's own S_DONE state keeps `busy` // asserted one cycle past ctrl_ready (act_tile_fetch.v's // own header/code already established this) -- wait for // !ctrl_busy before firing lane B's write, instead of // assuming back-to-back is safe. S_GAP: begin if (!ctrl_busy) begin ctrl_req <= 1'b1; ctrl_wr <= 1'b1; ctrl_addr <= {word_addr_b_lat[ADDR_WIDTH-1:ALIGN_BITS], {ALIGN_BITS{1'b0}}}; ctrl_wdata <= {(BURST_LEN){nid_b_lat, 8'h00, data_b_lat}}; ctrl_wmask <= ~({{(4*BURST_LEN-4){1'b0}}, 4'hF} << (word_addr_b_lat[ALIGN_BITS-1:0] * 4)); state <= S_XFER_B; end end S_XFER_B: begin if (ctrl_ready) begin state <= S_DONE; end end S_DONE: begin done <= 1'b1; state <= S_IDLE; end default: state <= S_IDLE; endcase end end endmodule