diff --git a/Bender.yml b/Bender.yml index e5ff5e8d..a29c1eff 100644 --- a/Bender.yml +++ b/Bender.yml @@ -118,6 +118,7 @@ sources: # Level 1 - test/frontend/tb_idma_desc64_top.sv - test/frontend/tb_idma_desc64_bench.sv + - test/frontend/tb_idma_reg_frontend.sv - test/future/idma_tb_per2axi.sv - test/future/TLToAXI4.v - test/midend/tb_idma_nd_midend.sv diff --git a/idma.mk b/idma.mk index 8e6bb18f..1789fad9 100644 --- a/idma.mk +++ b/idma.mk @@ -386,6 +386,13 @@ idma_sim_tb_idma_nd_midend_b2b: $(IDMA_VSIM_DIR)/compile.tcl cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit" cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc tb_idma_nd_midend_b2b -do "run -all; quit" +.PHONY: idma_sim_tb_idma_reg_frontend +idma_sim_tb_idma_reg_frontend: $(IDMA_VSIM_DIR)/compile.tcl + cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit" + cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=1 tb_idma_reg_frontend -do "run -all; quit" + cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=2 tb_idma_reg_frontend -do "run -all; quit" + cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=2 -gNumRegs=2 tb_idma_reg_frontend -do "run -all; quit" + .PHONY: idma_sim_tb_idma_transpose_b2b idma_sim_tb_idma_transpose_b2b: $(IDMA_VSIM_DIR)/compile.tcl cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit" diff --git a/src/frontend/reg/idma_reg.rdl b/src/frontend/reg/idma_reg.rdl index a46b73c1..4c4145e0 100644 --- a/src/frontend/reg/idma_reg.rdl +++ b/src/frontend/reg/idma_reg.rdl @@ -9,6 +9,11 @@ `ifndef IDMA_REG_REG_RDL `define IDMA_REG_REG_RDL +property rd_swacc { + type = boolean; + component = field; +}; + addrmap idma_reg #( longint unsigned SysAddrWidth = 32, // Address width longint unsigned NumDims = 2, // Number of dimensions available @@ -68,9 +73,10 @@ addrmap idma_reg #( name = "next_id"; desc = "Next ID, launches transfer, returns 0 if transfer not set up properly."; default sw = r; - default hw = rw; + default hw = w; field { desc = "Next ID, launches transfer, returns 0 if transfer not set up properly."; + rd_swacc = true; } next_id [31:0] = 0; }; @@ -153,9 +159,9 @@ addrmap idma_reg #( }; conf conf; - external status status[16]; - external next_id next_id[16]; - external done_id done_id[16]; + status status[16]; + next_id next_id[16]; + done_id done_id[16]; dst_addr dst_addr[SysAddrWidth/32] @ 0xD0; src_addr src_addr[SysAddrWidth/32]; length length[SysAddrWidth/32]; diff --git a/src/frontend/reg/tpl/idma_reg.sv.tpl b/src/frontend/reg/tpl/idma_reg.sv.tpl index a1ea9175..54055909 100644 --- a/src/frontend/reg/tpl/idma_reg.sv.tpl +++ b/src/frontend/reg/tpl/idma_reg.sv.tpl @@ -95,20 +95,17 @@ module idma_${identifier} #( idma_${identifier}_reg_pkg::idma_reg__in_t dma_hw2reg [NumRegs-1:0]; // arbitration output - dma_req_t [NumRegs-1:0] arb_dma_req; + dma_req_t [NumRegs-1:0] arb_dma_req_q; logic [NumRegs-1:0] arb_valid; logic [NumRegs-1:0] arb_ready; + logic [cf_math_pkg::idx_width(NumRegs)-1:0] arb_idx; - always_comb begin - stream_idx_o = '0; - for (int r = 0; r < NumRegs; r++) begin - for (int c = 0; c < NumStreams; c++) begin - if (dma_reg2hw[r].next_id[c].req && !dma_reg2hw[r].next_id[c].req_is_wr) begin - stream_idx_o = c; - end - end - end - end + // per-port launch-pending latch + logic [NumRegs-1:0] launch_pending_q; + stream_t [NumRegs-1:0] held_stream_q; + + // stream of the arbitrated winner, not the last pending port + assign stream_idx_o = req_valid_o ? held_stream_q[arb_idx] : '0; // generate the registers for (genvar i = 0; i < NumRegs; i++) begin : gen_core_regs @@ -175,66 +172,89 @@ module idma_${identifier} #( .hwif_in ( dma_hw2reg [i] ) ); - logic read_happens; - // launch-stall: hold the reg read-ack until the arbiter accepts the request - // (protocol-agnostic — driven into hwif rd_ack below, see gen_hw2reg_connections) + // a next_id rd_swacc strobe launches a transfer; latched until the arbiter accepts + logic read_happens; + stream_t read_stream; + dma_req_t nxt_dma_req; always_comb begin : proc_launch read_happens = 1'b0; + read_stream = '0; for (int c = 0; c < NumStreams; c++) begin - read_happens |= dma_reg2hw[i].next_id[c].req & ~dma_reg2hw[i].next_id[c].req_is_wr; + if (dma_reg2hw[i].next_id[c].next_id.rd_swacc) begin + read_happens = 1'b1; + read_stream = c; + end end - arb_valid[i] = read_happens; end - // assign request struct + // set on the read strobe (or an accept-and-reload in the same cycle), clear on accept + always_ff @(posedge clk_i or negedge rst_ni) begin : proc_launch_pending + if (!rst_ni) begin + launch_pending_q[i] <= 1'b0; + held_stream_q [i] <= '0; + arb_dma_req_q [i] <= '0; + end else begin + if (read_happens && (!launch_pending_q[i] || arb_ready[i])) begin + launch_pending_q[i] <= 1'b1; + held_stream_q [i] <= read_stream; + arb_dma_req_q [i] <= nxt_dma_req; + end else if (launch_pending_q[i] && arb_ready[i]) begin + launch_pending_q[i] <= 1'b0; + end + end + end + + assign arb_valid[i] = launch_pending_q[i]; + + // combinational request struct, captured into arb_dma_req_q at launch time always_comb begin : proc_hw_req_conv // all fields are zero per default - arb_dma_req[i] = '0; + nxt_dma_req = '0; // address and length % if bit_width == '32': - arb_dma_req[i]${sep}length = dma_reg2hw[i].length[0].length.value; - arb_dma_req[i]${sep}src_addr = dma_reg2hw[i].src_addr[0].src_addr.value; - arb_dma_req[i]${sep}dst_addr = dma_reg2hw[i].dst_addr[0].dst_addr.value; + nxt_dma_req${sep}length = dma_reg2hw[i].length[0].length.value; + nxt_dma_req${sep}src_addr = dma_reg2hw[i].src_addr[0].src_addr.value; + nxt_dma_req${sep}dst_addr = dma_reg2hw[i].dst_addr[0].dst_addr.value; % else: - arb_dma_req[i]${sep}length = {dma_reg2hw[i].length[1].length.value, dma_reg2hw[i].length[0].length.value}; - arb_dma_req[i]${sep}src_addr = {dma_reg2hw[i].src_addr[1].src_addr.value, dma_reg2hw[i].src_addr[0].src_addr.value}; - arb_dma_req[i]${sep}dst_addr = {dma_reg2hw[i].dst_addr[1].dst_addr.value, dma_reg2hw[i].dst_addr[0].dst_addr.value}; + nxt_dma_req${sep}length = {dma_reg2hw[i].length[1].length.value, dma_reg2hw[i].length[0].length.value}; + nxt_dma_req${sep}src_addr = {dma_reg2hw[i].src_addr[1].src_addr.value, dma_reg2hw[i].src_addr[0].src_addr.value}; + nxt_dma_req${sep}dst_addr = {dma_reg2hw[i].dst_addr[1].dst_addr.value, dma_reg2hw[i].dst_addr[0].dst_addr.value}; % endif // Protocols - arb_dma_req[i]${sep}opt.src_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.src_protocol.value); - arb_dma_req[i]${sep}opt.dst_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.dst_protocol.value); + nxt_dma_req${sep}opt.src_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.src_protocol.value); + nxt_dma_req${sep}opt.dst_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.dst_protocol.value); // Current backend only supports incremental burst - arb_dma_req[i]${sep}opt.src.burst = axi_pkg::BURST_INCR; - arb_dma_req[i]${sep}opt.dst.burst = axi_pkg::BURST_INCR; + nxt_dma_req${sep}opt.src.burst = axi_pkg::BURST_INCR; + nxt_dma_req${sep}opt.dst.burst = axi_pkg::BURST_INCR; // this frontend currently does not support cache variations - arb_dma_req[i]${sep}opt.src.cache = axi_pkg::CACHE_MODIFIABLE; - arb_dma_req[i]${sep}opt.dst.cache = axi_pkg::CACHE_MODIFIABLE; + nxt_dma_req${sep}opt.src.cache = axi_pkg::CACHE_MODIFIABLE; + nxt_dma_req${sep}opt.dst.cache = axi_pkg::CACHE_MODIFIABLE; // Backend options - arb_dma_req[i]${sep}opt.beo.decouple_aw = dma_reg2hw[i].conf.decouple_aw.value; - arb_dma_req[i]${sep}opt.beo.decouple_rw = dma_reg2hw[i].conf.decouple_rw.value; - arb_dma_req[i]${sep}opt.beo.src_max_llen = dma_reg2hw[i].conf.src_max_llen.value; - arb_dma_req[i]${sep}opt.beo.dst_max_llen = dma_reg2hw[i].conf.dst_max_llen.value; - arb_dma_req[i]${sep}opt.beo.src_reduce_len = dma_reg2hw[i].conf.src_reduce_len.value; - arb_dma_req[i]${sep}opt.beo.dst_reduce_len = dma_reg2hw[i].conf.dst_reduce_len.value; + nxt_dma_req${sep}opt.beo.decouple_aw = dma_reg2hw[i].conf.decouple_aw.value; + nxt_dma_req${sep}opt.beo.decouple_rw = dma_reg2hw[i].conf.decouple_rw.value; + nxt_dma_req${sep}opt.beo.src_max_llen = dma_reg2hw[i].conf.src_max_llen.value; + nxt_dma_req${sep}opt.beo.dst_max_llen = dma_reg2hw[i].conf.dst_max_llen.value; + nxt_dma_req${sep}opt.beo.src_reduce_len = dma_reg2hw[i].conf.src_reduce_len.value; + nxt_dma_req${sep}opt.beo.dst_reduce_len = dma_reg2hw[i].conf.dst_reduce_len.value; % if num_dim != 1: // ND connections % for nd in range(0, num_dim-1): % if bit_width == '32': - arb_dma_req[i].d_req[${nd}].reps = dma_reg2hw[i].dim[${nd}].reps[0].reps.value; - arb_dma_req[i].d_req[${nd}].src_strides = dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value; - arb_dma_req[i].d_req[${nd}].dst_strides = dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value; + nxt_dma_req.d_req[${nd}].reps = dma_reg2hw[i].dim[${nd}].reps[0].reps.value; + nxt_dma_req.d_req[${nd}].src_strides = dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value; + nxt_dma_req.d_req[${nd}].dst_strides = dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value; % else: - arb_dma_req[i].d_req[${nd}].reps = {dma_reg2hw[i].dim[${nd}].reps[1].reps.value, + nxt_dma_req.d_req[${nd}].reps = {dma_reg2hw[i].dim[${nd}].reps[1].reps.value, dma_reg2hw[i].dim[${nd}].reps[0].reps.value }; - arb_dma_req[i].d_req[${nd}].src_strides = {dma_reg2hw[i].dim[${nd}].src_stride[1].src_stride.value, + nxt_dma_req.d_req[${nd}].src_strides = {dma_reg2hw[i].dim[${nd}].src_stride[1].src_stride.value, dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value}; - arb_dma_req[i].d_req[${nd}].dst_strides = {dma_reg2hw[i].dim[${nd}].dst_stride[1].dst_stride.value, + nxt_dma_req.d_req[${nd}].dst_strides = {dma_reg2hw[i].dim[${nd}].dst_stride[1].dst_stride.value, dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value}; % endif % endfor @@ -242,41 +262,31 @@ module idma_${identifier} #( // Disable higher dimensions if ( dma_reg2hw[i].conf.enable_nd.value == 0) begin % for nd in range(0, num_dim-1): - arb_dma_req[i].d_req[${nd}].reps = ${"'0" if nd != num_dim-2 else "'d1"}; + nxt_dma_req.d_req[${nd}].reps = ${"'0" if nd != num_dim-2 else "'d1"}; % endfor end % for nd in range(1, num_dim-1): else if ( dma_reg2hw[i].conf.enable_nd.value == ${nd}) begin % for snd in range(nd, num_dim-1): - arb_dma_req[i].d_req[${snd}].reps = 'd1; + nxt_dma_req.d_req[${snd}].reps = 'd1; % endfor end % endfor % endif end - // observational registers + // observational registers: drive .next (read-side launch is the rd_swacc strobe above) for (genvar c = 0; c < NumStreams; c++) begin : gen_hw2reg_connections - assign dma_hw2reg[i].status[c].rd_data.busy = {midend_busy_i[c], busy_i[c]}; - assign dma_hw2reg[i].status[c].rd_ack = dma_reg2hw[i].status[c].req - & ~dma_reg2hw[i].status[c].req_is_wr; - assign dma_hw2reg[i].next_id[c].rd_data.next_id = next_id_i; - assign dma_hw2reg[i].next_id[c].rd_ack = dma_reg2hw[i].next_id[c].req - & ~dma_reg2hw[i].next_id[c].req_is_wr - & arb_ready[i]; - assign dma_hw2reg[i].done_id[c].rd_data.done_id = done_id_i[c]; - assign dma_hw2reg[i].done_id[c].rd_ack = dma_reg2hw[i].done_id[c].req - & ~dma_reg2hw[i].done_id[c].req_is_wr; + assign dma_hw2reg[i].status[c].busy.next = {midend_busy_i[c], busy_i[c]}; + assign dma_hw2reg[i].next_id[c].next_id.next = next_id_i; + assign dma_hw2reg[i].done_id[c].done_id.next = done_id_i[c]; end // tie-off unused channels for (genvar c = NumStreams; c < MaxNumStreams; c++) begin : gen_hw2reg_unused - assign dma_hw2reg[i].status[c].rd_data = '0; - assign dma_hw2reg[i].status[c].rd_ack = '0; - assign dma_hw2reg[i].next_id[c].rd_data.next_id = '0; - assign dma_hw2reg[i].next_id[c].rd_ack = '0; - assign dma_hw2reg[i].done_id[c].rd_data.done_id = '0; - assign dma_hw2reg[i].done_id[c].rd_ack = '0; + assign dma_hw2reg[i].status[c].busy.next = '0; + assign dma_hw2reg[i].next_id[c].next_id.next = '0; + assign dma_hw2reg[i].done_id[c].done_id.next = '0; end end @@ -295,11 +305,11 @@ module idma_${identifier} #( .rr_i ( '0 ), .req_i ( arb_valid ), .gnt_o ( arb_ready ), - .data_i ( arb_dma_req ), + .data_i ( arb_dma_req_q ), .gnt_i ( req_ready_i ), .req_o ( req_valid_o ), .data_o ( dma_req_o ), - .idx_o ( /* NC */ ) + .idx_o ( arb_idx ) ); endmodule diff --git a/test/frontend/tb_idma_reg_frontend.sv b/test/frontend/tb_idma_reg_frontend.sv new file mode 100644 index 00000000..7132d1b3 --- /dev/null +++ b/test/frontend/tb_idma_reg_frontend.sv @@ -0,0 +1,727 @@ +// Copyright 2025 ETH Zurich and University of Bologna. +// Solderpad Hardware License, Version 0.51, see LICENSE for details. +// SPDX-License-Identifier: SHL-0.51 + +// Authors: +// - Daniel Keller + +// Self-checking testbench for the iDMA register frontend (idma_reg32_3d, apb4-flat). +// Drives the APB config slave with the standard apb_test::apb_driver against a +// controllable backend stub and checks the non-blocking next_id launch contract: +// the config read completes promptly (even under backend backpressure) and the +// launch fires exactly once when the arbiter grants. A per-read watchdog guards +// against any read that hangs. + +`include "apb/typedef.svh" +`include "apb/assign.svh" +`include "idma/typedef.svh" + +module tb_idma_reg_frontend import idma_pkg::*; import apb_test::apb_driver; #( + // number of streams the elaborated DUT exposes (checked at instantiation) + parameter int unsigned NumStreams = 32'd1, + // number of config-bus ports (arbitrated by the reg frontend's rr_arb_tree) + parameter int unsigned NumRegs = 32'd1 +); + + // -------------------------------------------------------------------------- + // Parameters + // -------------------------------------------------------------------------- + localparam time TCK = 10ns; + localparam time TA = TCK * 1 / 4; // driver application time + localparam time TT = TCK * 3 / 4; // driver test (sample) time + localparam int unsigned CfgAddrWidth = 32'd32; + localparam int unsigned CfgDataWidth = 32'd32; + localparam int unsigned CfgStrbWidth = CfgDataWidth / 32'd8; + localparam int unsigned IdCounterWidth = 32'd32; + // idma data-path (reg32_3d: 32-bit data, 3 ND dims) + localparam int unsigned AddrWidth = 32'd32; + localparam int unsigned DataWidth = 32'd32; + localparam int unsigned NumDim = 32'd3; + localparam int unsigned RepWidth = 32'd32; + // apb_driver framing: the blocking driver.read() spans SETUP + first-ACCESS-check + + // trailing edge before returning, so a same-cycle (non-blocking) read takes this many + // config clocks end-to-end; each extra ACCESS wait state adds one more clock. + localparam int unsigned DrvFraming = 32'd2; + // bounded-latency bound for a non-blocking next_id read: measured ACCESS-phase latency + // (raw driver.read span minus DrvFraming) must be 0 config-clock cycles — the read must + // complete in its first ACCESS check even while req_ready_i is low. + localparam int unsigned MaxReadLatency = 32'd0; + // watchdog bound: any next_id APB read that does not complete within this many + // config-clock cycles is a hang (the non-blocking read completes immediately). + localparam int unsigned DeadlockCycles = 32'd2000; + + // register map (idma_reg32_3d_addrmap_pkg): base + per-stream stride 0x4 + localparam logic [31:0] REG_CONF = 32'h0000_0000; + localparam logic [31:0] REG_STATUS0 = 32'h0000_0004; + localparam logic [31:0] REG_NEXT_ID0 = 32'h0000_0044; + localparam logic [31:0] REG_DONE_ID0 = 32'h0000_0084; + localparam logic [31:0] REG_DST_ADDR = 32'h0000_00D0; + localparam logic [31:0] REG_SRC_ADDR = 32'h0000_00D4; + localparam logic [31:0] REG_LENGTH = 32'h0000_00D8; + + function automatic logic [31:0] reg_next_id(input int unsigned s); + return REG_NEXT_ID0 + 32'(s) * 32'h4; + endfunction + function automatic logic [31:0] reg_done_id(input int unsigned s); + return REG_DONE_ID0 + 32'(s) * 32'h4; + endfunction + + // -------------------------------------------------------------------------- + // Types + // -------------------------------------------------------------------------- + typedef logic [CfgAddrWidth-1:0] cfg_addr_t; + typedef logic [CfgDataWidth-1:0] cfg_data_t; + typedef logic [CfgStrbWidth-1:0] cfg_strb_t; + + `APB_TYPEDEF_REQ_T(cfg_apb_req_t, cfg_addr_t, cfg_data_t, cfg_strb_t) + `APB_TYPEDEF_RESP_T(cfg_apb_rsp_t, cfg_data_t) + + localparam int unsigned StrbWidth = DataWidth / 32'd8; + localparam int unsigned OffsetWidth = $clog2(StrbWidth); + typedef logic [AddrWidth-1:0] addr_t; + typedef logic [StrbWidth-1:0] strb_t; + typedef logic [OffsetWidth-1:0] offset_t; + typedef logic [RepWidth-1:0] strides_t; + typedef logic [RepWidth-1:0] reps_t; + typedef logic [AddrWidth-1:0] tf_len_t; + typedef logic idma_user_t; + + `IDMA_TYPEDEF_OPTIONS_T(options_t, logic) + `IDMA_TYPEDEF_REQ_T(idma_req_t, tf_len_t, addr_t, options_t, idma_user_t) + `IDMA_TYPEDEF_D_REQ_T(idma_d_req_t, reps_t, strides_t) + `IDMA_TYPEDEF_ND_REQ_T(idma_nd_req_t, idma_req_t, idma_d_req_t) + + typedef logic [IdCounterWidth-1:0] cnt_width_t; + + typedef apb_driver #( + .ADDR_WIDTH ( CfgAddrWidth ), + .DATA_WIDTH ( CfgDataWidth ), + .TA ( TA ), + .TT ( TT ) + ) apb_driver_t; + + // -------------------------------------------------------------------------- + // Clock / reset + // -------------------------------------------------------------------------- + logic clk; + logic rst_n; + + initial begin + clk = 1'b0; + forever #(TCK/2) clk = ~clk; + end + + // -------------------------------------------------------------------------- + // DUT nets + // -------------------------------------------------------------------------- + cfg_apb_req_t [NumRegs-1:0] apb_req; + cfg_apb_rsp_t [NumRegs-1:0] apb_rsp; + + idma_nd_req_t dma_req; + logic req_valid; + logic req_ready; // driven by the backend stub + cnt_width_t next_id; // from the id gen + logic [(NumStreams>1?$clog2(NumStreams):1)-1:0] stream_idx; + cnt_width_t [NumStreams-1:0] done_id; + idma_busy_t [NumStreams-1:0] busy; + logic [NumStreams-1:0] midend_busy; + + // -------------------------------------------------------------------------- + // APB DV interfaces + drivers: one per config port. Each interface is bridged + // to the DUT's packed dma_ctrl_req_i[i]/dma_ctrl_rsp_o[i] via the apb assign + // macros; one apb_driver per port lets Test 5 drive two ports concurrently. + // -------------------------------------------------------------------------- + // virtual-interface handles (interface arrays can only be indexed by a constant, + // so each element is captured into this array from the generate loop below) + typedef virtual APB_DV #(.ADDR_WIDTH(CfgAddrWidth), .DATA_WIDTH(CfgDataWidth)) apb_dv_t; + apb_dv_t apb_vif[NumRegs]; + apb_driver_t drv[NumRegs]; + + for (genvar i = 0; i < NumRegs; i++) begin : gen_apb_bridge + APB_DV #( + .ADDR_WIDTH ( CfgAddrWidth ), + .DATA_WIDTH ( CfgDataWidth ) + ) apb_dv (clk); + // master interface -> DUT req struct, DUT rsp struct -> master interface + `APB_ASSIGN_TO_REQ(apb_req[i], apb_dv) + assign apb_dv.pready = apb_rsp[i].pready; + assign apb_dv.prdata = apb_rsp[i].prdata; + assign apb_dv.pslverr = apb_rsp[i].pslverr; + initial apb_vif[i] = apb_dv; // publish the vif handle for the driver + end + + // -------------------------------------------------------------------------- + // Transfer-id generator (owns the next/completed counters). Reset next=2. + // issue on an accepted launch, retire on a modeled backend completion. + // -------------------------------------------------------------------------- + logic issue; + logic retire; + + idma_transfer_id_gen #( + .IdWidth ( IdCounterWidth ) + ) i_id_gen ( + .clk_i ( clk ), + .rst_ni ( rst_n ), + .issue_i ( issue ), + .retire_i ( retire ), + .next_o ( next_id ), + .completed_o ( done_id[0] ) + ); + // multi-stream: id gen models stream 0; other streams share the same completed + // counter here (the DUT only compares per-stream done_id on read-back). + for (genvar s = 1; s < NumStreams; s++) begin : gen_done_other + assign done_id[s] = done_id[0]; + end + + // an accepted launch is the arbiter handshake; also the SW "id-advance" event + assign issue = req_valid & req_ready; + + // -------------------------------------------------------------------------- + // DUT + // -------------------------------------------------------------------------- + idma_reg32_3d #( + .NumRegs ( NumRegs ), + .NumStreams ( NumStreams ), + .IdCounterWidth ( IdCounterWidth ), + .apb_req_t ( cfg_apb_req_t ), + .apb_rsp_t ( cfg_apb_rsp_t ), + .dma_req_t ( idma_nd_req_t ) + ) i_dut ( + .clk_i ( clk ), + .rst_ni ( rst_n ), + .dma_ctrl_req_i ( apb_req ), + .dma_ctrl_rsp_o ( apb_rsp ), + .dma_req_o ( dma_req ), + .req_valid_o ( req_valid ), + .req_ready_i ( req_ready ), + .next_id_i ( next_id ), + .stream_idx_o ( stream_idx ), + .done_id_i ( done_id ), + .busy_i ( busy ), + .midend_busy_i ( midend_busy ) + ); + + // -------------------------------------------------------------------------- + // Backend stub. `req_ready` is directly controllable by the tests. On each + // accepted launch (req_valid & req_ready) the emitted ND request is captured + // and enqueued with a retire-deadline; when its timer expires it is retired in + // order via the id gen so `done_id` advances. `busy`/`midend_busy` follow the + // outstanding count. A single-entry timer is enough — launches are retired FIFO + // and only re-armed once the previous one drains, which keeps ordering exact. + // -------------------------------------------------------------------------- + idma_nd_req_t captured_q[$]; // every accepted launch, for self-check + int unsigned outstanding; // in-flight (not yet retired) launches + int unsigned retire_delay; // clocks a launch stays in flight + logic backend_auto_retire; // if 0, retirement is suppressed + int retire_timer; // -1 == no launch currently timing out + + assign busy[0] = (outstanding != 0) ? '1 : '0; + assign midend_busy[0] = (outstanding != 0) ? 1'b1 : 1'b0; + for (genvar s = 1; s < NumStreams; s++) begin : gen_busy_other + assign busy[s] = busy[0]; + assign midend_busy[s] = midend_busy[0]; + end + + initial begin + outstanding = 0; + retire_timer = -1; + retire_delay = 3; + backend_auto_retire = 1'b1; + retire = 1'b0; + end + + // Unified backend model: capture launches, count outstanding, and retire FIFO. + always @(posedge clk) begin + automatic bit accept = rst_n && req_valid && req_ready; + automatic bit do_retire = 1'b0; + + retire <= 1'b0; + if (!rst_n) begin + outstanding <= 0; + retire_timer <= -1; + end else begin + // 1) capture an accepted launch + if (accept) + captured_q.push_back(dma_req); + + // 2) advance / fire the retire timer + if (backend_auto_retire && retire_timer == 0) begin + retire <= 1'b1; + do_retire = 1'b1; + retire_timer <= -1; // re-armed below if launches remain + end else if (retire_timer > 0) begin + retire_timer <= retire_timer - 1; + end + + // 3) update outstanding count (+accept, -retire) + outstanding <= outstanding + (accept ? 1 : 0) - (do_retire ? 1 : 0); + + // 4) arm the timer whenever a launch is waiting and none is timing out + if (backend_auto_retire) begin + automatic int unsigned next_out = + outstanding + (accept ? 1 : 0) - (do_retire ? 1 : 0); + if ((retire_timer < 0 || do_retire) && next_out > 0) + retire_timer <= retire_delay; + end + end + end + + // -------------------------------------------------------------------------- + // Watchdog: fatal if a next_id APB read stays outstanding too long. + // `nxt_read_active` is raised by launch() around the blocking driver.read and + // cleared when it returns — a driver.read that never returns is caught here. + // -------------------------------------------------------------------------- + logic nxt_read_active; + int unsigned nxt_read_watchdog; + initial begin + nxt_read_active = 1'b0; + nxt_read_watchdog = 0; + end + always @(posedge clk) begin + if (!rst_n) begin + nxt_read_watchdog <= 0; + end else if (nxt_read_active) begin + nxt_read_watchdog <= nxt_read_watchdog + 1; + if (nxt_read_watchdog > DeadlockCycles) begin + $fatal(1, "DEADLOCK: next_id read did not complete within %0d cycles", + DeadlockCycles); + end + end else begin + nxt_read_watchdog <= 0; + end + end + + // -------------------------------------------------------------------------- + // Global "launch accepted" pulse counter: an accepted launch is one arbiter + // handshake (issue). Used by the launch-integrity scoreboard and by the + // driver to wait for id-advance before re-launching. + // -------------------------------------------------------------------------- + int unsigned launch_accept_count; + int unsigned launch_acc_base; // accept-count snapshot taken at a launch read + always @(posedge clk) begin + if (!rst_n) launch_accept_count <= 0; + else if (issue) launch_accept_count <= launch_accept_count + 1; + end + + // Test 5 (multi-port arbitration) scoreboard state. Each config port programs a + // unique src_addr that encodes its target stream, so the arbiter's presented + // winner (dma_req_o) can be mapped back to a stream and compared to stream_idx_o. + logic [31:0] sb_addr_stream0; // src_addr programmed for stream 0's port + logic [31:0] sb_addr_stream1; // src_addr programmed for stream 1's port + int unsigned sb_mismatch; // times stream_idx != the winner's stream + initial sb_mismatch = 0; + + // -------------------------------------------------------------------------- + // APB stimulus via apb_test::apb_driver (per-port drivers built in init). + // -------------------------------------------------------------------------- + task automatic apb_write(input logic [31:0] addr, input logic [31:0] data, + input int unsigned port = 0); + logic err; + drv[port].write(addr, data, '1, err); + endtask + + // next_id read via the driver, TIMED to recover the ACCESS-phase latency the + // non-blocking contract bounds. driver.read blocks until pready; measure the + // elapsed config clocks and subtract the fixed driver framing. The watchdog + // flag is held across the (blocking) call so a read that never returns fatals. + task automatic apb_read_next(input logic [31:0] addr, + output logic [31:0] data, + output int unsigned cyc, + input int unsigned port = 0); + logic err; + time t0; + nxt_read_active = 1'b1; + t0 = $time; + drv[port].read(addr, data, err); + // raw span in config clocks, minus the driver's fixed SETUP+trailing framing + cyc = (($time - t0) / TCK) - DrvFraming; + nxt_read_active = 1'b0; + endtask + + // -------------------------------------------------------------------------- + // High-level helpers + // -------------------------------------------------------------------------- + task automatic program_transfer(input logic [31:0] src, + input logic [31:0] dst, + input logic [31:0] len, + input int unsigned port = 0); + // conf: plain 1D incremental copy, ND disabled + apb_write(REG_CONF, 32'h0, port); + apb_write(REG_SRC_ADDR, src, port); + apb_write(REG_DST_ADDR, dst, port); + apb_write(REG_LENGTH, len, port); + endtask + + // launch: read next_id (the transfer trigger, non-blocking); returns id and the + // measured ACCESS-phase latency in `cyc`. Snapshots the accept count before the + // read so an accept coinciding with the (non-blocking) read is still observed. + task automatic launch(output logic [31:0] id, output int unsigned cyc, + input int unsigned s = 0, input int unsigned port = 0); + launch_acc_base = launch_accept_count; + apb_read_next(reg_next_id(s), id, cyc, port); + endtask + + // wait until the launch read since the last launch() has been accepted (the + // arbiter grant / id-advance). SW confirms acceptance before re-launching. + task automatic wait_launch_accepted(); + int unsigned tries; + tries = 0; + while (launch_accept_count == launch_acc_base) begin + @(posedge clk); + tries++; + if (tries > 1000) + $fatal(1, "wait_launch_accepted: launch never accepted"); + end + endtask + + task automatic read_done(output logic [31:0] id, input int unsigned s = 0); + logic err; + drv[0].read(reg_done_id(s), id, err); + endtask + + // poll done_id until it reaches `id` (or a bounded number of tries) + task automatic poll_done(input logic [31:0] id, input int unsigned s = 0); + logic [31:0] d; + int unsigned tries; + tries = 0; + do begin + read_done(d, s); + tries++; + if (tries > 1000) + $fatal(1, "poll_done: done_id never reached %0d (last %0d)", id, d); + end while (d != id); + endtask + + // -------------------------------------------------------------------------- + // Bookkeeping for the test program + // -------------------------------------------------------------------------- + int unsigned errors; + int unsigned checks; + + task automatic check_eq(input logic [63:0] got, input logic [63:0] exp, + input string msg); + checks++; + if (got !== exp) begin + errors++; + $error("[FAIL] %s: got 0x%0h, expected 0x%0h", msg, got, exp); + end else begin + $display("[ ok ] %s = 0x%0h", msg, got); + end + endtask + + // -------------------------------------------------------------------------- + // Test program + // -------------------------------------------------------------------------- + logic [31:0] id0, id1, id2, id3; + logic [31:0] exp_id; // id gen next_o snapshot before a launch + logic [31:0] prev_id; // last returned id, for monotonic checks + int unsigned rcyc; // last read's ACCESS-phase cycle count + + initial begin : test + errors = 0; + checks = 0; + req_ready = 1'b1; + rst_n = 1'b0; + // let the generate-block initials publish their vif handles, then bind drivers + @(negedge clk); + for (int unsigned p = 0; p < NumRegs; p++) drv[p] = new (apb_vif[p]); + for (int unsigned p = 0; p < NumRegs; p++) drv[p].reset_master(); + + // reset + repeat (5) @(negedge clk); + rst_n = 1'b1; + repeat (2) @(negedge clk); + + $display("====================================================="); + $display(" tb_idma_reg_frontend (NumStreams=%0d)", NumStreams); + $display("====================================================="); + + // ------------------------------------------------------------------ + // Test 1 — Basic launch (backend ready), non-blocking read + // ------------------------------------------------------------------ + $display("\n--- Test 1: basic launch ---"); + backend_auto_retire = 1'b1; + req_ready = 1'b1; + captured_q.delete(); + program_transfer(32'h1000_0000, 32'h2000_0000, 32'h0000_0040); + // id gen resets next=2, so the very first launch must return id 2 + exp_id = next_id; + check_eq(exp_id, 32'd2, "Test1 id gen resets next=2"); + launch(id0, rcyc); + // the launch returns exactly the id that was pending + check_eq(id0, exp_id, "Test1 first launch id == next_id"); + // non-blocking: the read completed within the bounded latency + check_eq(rcyc <= MaxReadLatency, 1'b1, "Test1 read within bounded latency"); + wait_launch_accepted(); + prev_id = id0; + // exactly one launch captured, with the programmed geometry + check_eq(captured_q.size(), 32'd1, "Test1 launch count"); + if (captured_q.size() > 0) begin + check_eq(captured_q[0].burst_req.src_addr, 32'h1000_0000, "Test1 src_addr"); + check_eq(captured_q[0].burst_req.dst_addr, 32'h2000_0000, "Test1 dst_addr"); + check_eq(captured_q[0].burst_req.length, 32'h0000_0040, "Test1 length"); + end + poll_done(id0); + $display("[ ok ] Test1 done_id reached %0d", id0); + + // ------------------------------------------------------------------ + // Test 2 — Non-blocking read under backpressure (core gate) + // The next_id read MUST complete promptly (bounded latency) even while + // req_ready_i is held LOW. On a blocking template this FAILS. + // ------------------------------------------------------------------ + $display("\n--- Test 2: non-blocking read under backpressure ---"); + backend_auto_retire = 1'b0; // no auto retire while we hold the stall + captured_q.delete(); + // model a busy backend: hold req_ready LOW so the arbiter cannot grant + req_ready = 1'b0; + program_transfer(32'h3000_0000, 32'h4000_0000, 32'h0000_0080); + exp_id = next_id; // id that this launch returns + // the read completes despite req_ready low — the non-blocking property + launch(id1, rcyc); + check_eq(rcyc <= MaxReadLatency, 1'b1, "Test2 read within bounded latency (BP)"); + check_eq(id1, exp_id, "Test2 launch id == pre-stall next_id"); + check_eq(id1, prev_id + 32'd1, "Test2 id monotonic after Test1"); + prev_id = id1; + // the launch is held pending (not yet granted): id must not have advanced yet + check_eq(next_id, exp_id, "Test2 id held (no issue) while req_ready low"); + // release backpressure — the held launch now completes exactly once + req_ready = 1'b1; + wait_launch_accepted(); + check_eq(captured_q.size(), 32'd1, "Test2 launch accepted exactly once"); + if (captured_q.size() > 0) begin + check_eq(captured_q[0].burst_req.src_addr, 32'h3000_0000, "Test2 src_addr held"); + check_eq(captured_q[0].burst_req.length, 32'h0000_0080, "Test2 length held"); + end + // let it retire and confirm the reg block is still live afterwards + backend_auto_retire = 1'b1; + poll_done(id1); + $display("[ ok ] Test2 completed after backpressure, block still live"); + + // ------------------------------------------------------------------ + // Test 2b — Launch integrity: a launch is never dropped when the grant + // is late. Hold req_ready low for several cycles AFTER a next_id read, + // then release; the latch must fire the launch EXACTLY ONCE. + // ------------------------------------------------------------------ + $display("\n--- Test 2b: launch integrity (late grant, no drop) ---"); + backend_auto_retire = 1'b0; + captured_q.delete(); + req_ready = 1'b0; + program_transfer(32'h7000_0000, 32'h8000_0000, 32'h0000_00C0); + exp_id = next_id; + begin + int unsigned acc_before; + acc_before = launch_accept_count; + // single next_id read (one launch), completes non-blocking under BP + launch(id3, rcyc); + check_eq(rcyc <= MaxReadLatency, 1'b1, "Test2b read within bounded latency (BP)"); + check_eq(id3, exp_id, "Test2b launch id == next_id"); + // hold the grant off for several cycles: the launch must stay pending, not drop + repeat (12) @(posedge clk); + check_eq(launch_accept_count, acc_before, "Test2b no accept while req_ready low"); + check_eq(req_valid, 1'b1, "Test2b req_valid held high across late grant"); + check_eq(captured_q.size(), 32'd0, "Test2b nothing captured before grant"); + // release: exactly one accept, exactly one captured launch + req_ready = 1'b1; + wait_launch_accepted(); + check_eq(launch_accept_count, acc_before + 32'd1, "Test2b launch fired exactly once"); + end + // give the arbiter a settle cycle, then confirm no second spurious launch + repeat (4) @(posedge clk); + check_eq(captured_q.size(), 32'd1, "Test2b exactly one launch captured"); + if (captured_q.size() > 0) begin + check_eq(captured_q[0].burst_req.src_addr, 32'h7000_0000, "Test2b src_addr held"); + check_eq(captured_q[0].burst_req.length, 32'h0000_00C0, "Test2b length held"); + end + check_eq(id3, prev_id + 32'd1, "Test2b id monotonic"); + prev_id = id3; + backend_auto_retire = 1'b1; + poll_done(id3); + $display("[ ok ] Test2b launch fired exactly once after late grant"); + + // ------------------------------------------------------------------ + // Test 3 — Multi-stream (only meaningful when NumStreams > 1) + // stream_idx_o must point at the launching stream until the grant lands. + // ------------------------------------------------------------------ + if (NumStreams > 1) begin + int unsigned held1_cnt; // cycles stream_idx==1 while pending + logic bad_idx; // stream_idx ever pointed at a wrong stream + int unsigned acc_before; + $display("\n--- Test 3: multi-stream stream_idx held until grant ---"); + backend_auto_retire = 1'b0; + captured_q.delete(); + req_ready = 1'b0; + program_transfer(32'h5000_0000, 32'h6000_0000, 32'h0000_0100); + exp_id = next_id; + held1_cnt = 0; + bad_idx = 1'b0; + acc_before = launch_accept_count; + launch(id2, rcyc, 1); // launch on stream 1 (non-blocking read) + check_eq(rcyc <= MaxReadLatency, 1'b1, "Test3 read within bounded latency (BP)"); + // while the launch is pending (req_valid high, grant withheld) stream_idx==1 + repeat (12) begin + @(posedge clk); + if (req_valid && !req_ready) begin + if (stream_idx == 1) held1_cnt++; + else bad_idx = 1'b1; // wrong / dropped stream index + end + end + check_eq(held1_cnt >= 32'd8, 1'b1, "Test3 stream_idx held == 1 across stall"); + check_eq(bad_idx, 1'b0, "Test3 stream_idx never pointed at wrong stream"); + // release: exactly one accept on stream 1 + req_ready = 1'b1; + wait_launch_accepted(); + check_eq(launch_accept_count, acc_before + 32'd1, "Test3 stream1 accepted once"); + check_eq(id2, exp_id, "Test3 stream1 launch id == next_id"); + check_eq(id2, prev_id + 32'd1, "Test3 id monotonic"); + prev_id = id2; + check_eq(captured_q.size(), 32'd1, "Test3 stream1 captured once"); + backend_auto_retire = 1'b1; + poll_done(id2, 1); + $display("[ ok ] Test3 multi-stream launch completed"); + end else begin + $display("\n--- Test 3: skipped (NumStreams == 1) ---"); + end + + // ------------------------------------------------------------------ + // Test 4 — Back-to-back launches, monotonic ids, in-order done. + // Non-blocking: after each launch the driver waits for the id-advance + // (accept) before re-programming and re-launching. + // ------------------------------------------------------------------ + $display("\n--- Test 4: back-to-back launches ---"); + backend_auto_retire = 1'b1; + req_ready = 1'b1; + captured_q.delete(); + begin + logic [31:0] ids[4]; + for (int unsigned k = 0; k < 4; k++) begin + program_transfer(32'h1000 + k*32'h100, 32'h9000 + k*32'h100, 32'h40 + k*32'h10); + launch(ids[k], rcyc); + check_eq(rcyc <= MaxReadLatency, 1'b1, $sformatf("Test4 read[%0d] bounded latency", k)); + wait_launch_accepted(); // model SW confirming id-advance before re-launch + end + // exactly four accepted launches, each id one more than the last + check_eq(captured_q.size(), 32'd4, "Test4 four launches captured"); + for (int unsigned k = 0; k < 4; k++) begin + check_eq(ids[k], prev_id + 32'd1 + k, $sformatf("Test4 id[%0d] monotonic", k)); + if (k < captured_q.size()) begin + check_eq(captured_q[k].burst_req.src_addr, 32'h1000 + k*32'h100, + $sformatf("Test4 src[%0d]", k)); + check_eq(captured_q[k].burst_req.length, 32'h40 + k*32'h10, + $sformatf("Test4 len[%0d]", k)); + end + end + prev_id = ids[3]; + // done_id must advance in order to the last id + poll_done(ids[3]); + $display("[ ok ] Test4 done_id advanced in order to %0d", ids[3]); + end + + // ------------------------------------------------------------------ + // Test 5 — Concurrent multi-port launch: stream_idx must ride the + // arbitration (only meaningful when NumRegs>1 and NumStreams>1). + // Two config ports launch on *different* streams in the same window; on + // every grant stream_idx must match the ARBITRATED port's stream, not the + // last-pending port. This FAILS on the pre-fix RTL and PASSES after it. + // ------------------------------------------------------------------ + if (NumRegs > 1 && NumStreams > 1) begin + logic [31:0] id_p0, id_p1; + int unsigned cyc0, cyc1; + $display("\n--- Test 5: concurrent multi-port arbitration (stream_idx) ---"); + backend_auto_retire = 1'b0; + captured_q.delete(); + req_ready = 1'b0; + // port 0 -> stream 0, port 1 -> stream 1, each with a unique src_addr + sb_addr_stream0 = 32'hAAAA_0000; + sb_addr_stream1 = 32'hBBBB_0000; + program_transfer(sb_addr_stream0, 32'hCCCC_0000, 32'h0000_0040, 0); + program_transfer(sb_addr_stream1, 32'hDDDD_0000, 32'h0000_0080, 1); + sb_mismatch = 0; + // launch both ports concurrently on different streams while the grant is held off, + // so both launch_pending latches are set at once (the arbiter must pick one). + fork + launch(id_p0, cyc0, 0, 0); // port 0, stream 0 + launch(id_p1, cyc1, 1, 1); // port 1, stream 1 + join + // Both launches are now pending with req_ready still low. rr_arb_tree (AxiVldRdy=1) + // *presents* its chosen winner on dma_req_o / idx_o even while the grant is withheld, + // so stream_idx_o must equal the winner's stream. On the pre-fix RTL stream_idx_o is + // the last-pending port's stream (held_stream[NumRegs-1]) regardless of the winner, + // so it disagrees with dma_req_o whenever the winner is not the last port. Check the + // presented (winner, stream_idx) pair across the whole held window. + begin + int unsigned held_checks; + held_checks = 0; + repeat (12) begin + @(negedge clk); + #(TCK/10); // let combinational DUT outputs settle + if (req_valid && !req_ready) begin + automatic int unsigned won_stream = 32'hFFFF_FFFF; + if (dma_req.burst_req.src_addr == sb_addr_stream0) won_stream = 0; + else if (dma_req.burst_req.src_addr == sb_addr_stream1) won_stream = 1; + if (won_stream != 32'hFFFF_FFFF) begin + held_checks++; + if (stream_idx != won_stream[$bits(stream_idx)-1:0]) begin + sb_mismatch++; + $display("[Test5] MISMATCH: dma_req_o=port for stream %0d but stream_idx=%0d", + won_stream, stream_idx); + end + end + end + end + check_eq(held_checks > 0, 1'b1, "Test5 winner presented while grant withheld"); + check_eq(sb_mismatch, 32'd0, "Test5 stream_idx matches arbitrated port (winner)"); + end + // now let both launches drain and confirm both transfers are captured correctly + req_ready = 1'b1; + backend_auto_retire = 1'b1; + begin + int unsigned tries; + tries = 0; + while (captured_q.size() < 2) begin + @(posedge clk); + tries++; + if (tries > 1000) $fatal(1, "Test5: both launches never drained (got %0d)", + captured_q.size()); + end + end + check_eq(captured_q.size(), 32'd2, "Test5 both launches captured"); + begin + logic saw_a, saw_b; + saw_a = 1'b0; saw_b = 1'b0; + foreach (captured_q[k]) begin + if (captured_q[k].burst_req.src_addr == sb_addr_stream0) saw_a = 1'b1; + if (captured_q[k].burst_req.src_addr == sb_addr_stream1) saw_b = 1'b1; + end + check_eq(saw_a, 1'b1, "Test5 port0/stream0 transfer captured"); + check_eq(saw_b, 1'b1, "Test5 port1/stream1 transfer captured"); + end + $display("[ ok ] Test5 concurrent arbitration: %0d mismatches", sb_mismatch); + end else begin + $display("\n--- Test 5: skipped (needs NumRegs>1 and NumStreams>1) ---"); + end + + // ------------------------------------------------------------------ + // Summary + // ------------------------------------------------------------------ + repeat (5) @(negedge clk); + $display("\n====================================================="); + $display(" checks run : %0d", checks); + $display(" errors : %0d", errors); + if (errors == 0) + $display(" RESULT : PASS"); + else + $display(" RESULT : FAIL"); + $display("====================================================="); + if (errors != 0) + $fatal(1, "tb_idma_reg_frontend FAILED with %0d error(s)", errors); + $finish; + end + + // global safety net: kill a run that hangs entirely (belt-and-braces with the + // per-read watchdog, in case a hang happens outside a tracked next_id read). + initial begin + #(TCK * 200000); + $fatal(1, "GLOBAL TIMEOUT: testbench did not finish"); + end + +endmodule