Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Bender.yml
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,7 @@ sources:
# Level 1
- test/frontend/tb_idma_desc64_top.sv
- test/frontend/tb_idma_desc64_bench.sv
- test/frontend/tb_idma_reg_frontend.sv
- test/future/idma_tb_per2axi.sv
- test/future/TLToAXI4.v
- test/midend/tb_idma_nd_midend.sv
Expand Down
7 changes: 7 additions & 0 deletions idma.mk
Original file line number Diff line number Diff line change
Expand Up @@ -386,6 +386,13 @@ idma_sim_tb_idma_nd_midend_b2b: $(IDMA_VSIM_DIR)/compile.tcl
cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit"
cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc tb_idma_nd_midend_b2b -do "run -all; quit"

.PHONY: idma_sim_tb_idma_reg_frontend
idma_sim_tb_idma_reg_frontend: $(IDMA_VSIM_DIR)/compile.tcl
cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit"
cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=1 tb_idma_reg_frontend -do "run -all; quit"
cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=2 tb_idma_reg_frontend -do "run -all; quit"
cd $(IDMA_VSIM_DIR); $(VSIM) -c -t 1ps -voptargs=+acc -gNumStreams=2 -gNumRegs=2 tb_idma_reg_frontend -do "run -all; quit"
Comment thread
DanielKellerM marked this conversation as resolved.

.PHONY: idma_sim_tb_idma_transpose_b2b
idma_sim_tb_idma_transpose_b2b: $(IDMA_VSIM_DIR)/compile.tcl
cd $(IDMA_VSIM_DIR); $(VSIM) -c -do "source compile.tcl; quit"
Expand Down
14 changes: 10 additions & 4 deletions src/frontend/reg/idma_reg.rdl
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,11 @@
`ifndef IDMA_REG_REG_RDL
`define IDMA_REG_REG_RDL

property rd_swacc {
type = boolean;
component = field;
};

addrmap idma_reg #(
longint unsigned SysAddrWidth = 32, // Address width
longint unsigned NumDims = 2, // Number of dimensions available
Expand Down Expand Up @@ -68,9 +73,10 @@ addrmap idma_reg #(
name = "next_id";
desc = "Next ID, launches transfer, returns 0 if transfer not set up properly.";
default sw = r;
default hw = rw;
default hw = w;
field {
desc = "Next ID, launches transfer, returns 0 if transfer not set up properly.";
rd_swacc = true;
} next_id [31:0] = 0;
};

Expand Down Expand Up @@ -153,9 +159,9 @@ addrmap idma_reg #(
};

conf conf;
external status status[16];
external next_id next_id[16];
external done_id done_id[16];
status status[16];
next_id next_id[16];
done_id done_id[16];
dst_addr dst_addr[SysAddrWidth/32] @ 0xD0;
src_addr src_addr[SysAddrWidth/32];
length length[SysAddrWidth/32];
Expand Down
136 changes: 73 additions & 63 deletions src/frontend/reg/tpl/idma_reg.sv.tpl
Original file line number Diff line number Diff line change
Expand Up @@ -95,20 +95,17 @@ module idma_${identifier} #(
idma_${identifier}_reg_pkg::idma_reg__in_t dma_hw2reg [NumRegs-1:0];

// arbitration output
dma_req_t [NumRegs-1:0] arb_dma_req;
dma_req_t [NumRegs-1:0] arb_dma_req_q;
logic [NumRegs-1:0] arb_valid;
logic [NumRegs-1:0] arb_ready;
logic [cf_math_pkg::idx_width(NumRegs)-1:0] arb_idx;

always_comb begin
stream_idx_o = '0;
for (int r = 0; r < NumRegs; r++) begin
for (int c = 0; c < NumStreams; c++) begin
if (dma_reg2hw[r].next_id[c].req && !dma_reg2hw[r].next_id[c].req_is_wr) begin
stream_idx_o = c;
end
end
end
end
// per-port launch-pending latch
logic [NumRegs-1:0] launch_pending_q;
stream_t [NumRegs-1:0] held_stream_q;

// stream of the arbitrated winner, not the last pending port
assign stream_idx_o = req_valid_o ? held_stream_q[arb_idx] : '0;

// generate the registers
for (genvar i = 0; i < NumRegs; i++) begin : gen_core_regs
Expand Down Expand Up @@ -175,108 +172,121 @@ module idma_${identifier} #(
.hwif_in ( dma_hw2reg [i] )
);

logic read_happens;
// launch-stall: hold the reg read-ack until the arbiter accepts the request
// (protocol-agnostic — driven into hwif rd_ack below, see gen_hw2reg_connections)
// a next_id rd_swacc strobe launches a transfer; latched until the arbiter accepts
logic read_happens;
stream_t read_stream;
dma_req_t nxt_dma_req;

always_comb begin : proc_launch
read_happens = 1'b0;
read_stream = '0;
for (int c = 0; c < NumStreams; c++) begin
read_happens |= dma_reg2hw[i].next_id[c].req & ~dma_reg2hw[i].next_id[c].req_is_wr;
if (dma_reg2hw[i].next_id[c].next_id.rd_swacc) begin
read_happens = 1'b1;
read_stream = c;
end
end
arb_valid[i] = read_happens;
end

// assign request struct
// set on the read strobe (or an accept-and-reload in the same cycle), clear on accept
always_ff @(posedge clk_i or negedge rst_ni) begin : proc_launch_pending
if (!rst_ni) begin
launch_pending_q[i] <= 1'b0;
held_stream_q [i] <= '0;
arb_dma_req_q [i] <= '0;
end else begin
if (read_happens && (!launch_pending_q[i] || arb_ready[i])) begin
launch_pending_q[i] <= 1'b1;
held_stream_q [i] <= read_stream;
arb_dma_req_q [i] <= nxt_dma_req;
end else if (launch_pending_q[i] && arb_ready[i]) begin
launch_pending_q[i] <= 1'b0;
end
end
end

assign arb_valid[i] = launch_pending_q[i];

// combinational request struct, captured into arb_dma_req_q at launch time
always_comb begin : proc_hw_req_conv
// all fields are zero per default
arb_dma_req[i] = '0;
nxt_dma_req = '0;

// address and length
% if bit_width == '32':
arb_dma_req[i]${sep}length = dma_reg2hw[i].length[0].length.value;
arb_dma_req[i]${sep}src_addr = dma_reg2hw[i].src_addr[0].src_addr.value;
arb_dma_req[i]${sep}dst_addr = dma_reg2hw[i].dst_addr[0].dst_addr.value;
nxt_dma_req${sep}length = dma_reg2hw[i].length[0].length.value;
nxt_dma_req${sep}src_addr = dma_reg2hw[i].src_addr[0].src_addr.value;
nxt_dma_req${sep}dst_addr = dma_reg2hw[i].dst_addr[0].dst_addr.value;
% else:
arb_dma_req[i]${sep}length = {dma_reg2hw[i].length[1].length.value, dma_reg2hw[i].length[0].length.value};
arb_dma_req[i]${sep}src_addr = {dma_reg2hw[i].src_addr[1].src_addr.value, dma_reg2hw[i].src_addr[0].src_addr.value};
arb_dma_req[i]${sep}dst_addr = {dma_reg2hw[i].dst_addr[1].dst_addr.value, dma_reg2hw[i].dst_addr[0].dst_addr.value};
nxt_dma_req${sep}length = {dma_reg2hw[i].length[1].length.value, dma_reg2hw[i].length[0].length.value};
nxt_dma_req${sep}src_addr = {dma_reg2hw[i].src_addr[1].src_addr.value, dma_reg2hw[i].src_addr[0].src_addr.value};
nxt_dma_req${sep}dst_addr = {dma_reg2hw[i].dst_addr[1].dst_addr.value, dma_reg2hw[i].dst_addr[0].dst_addr.value};
% endif

// Protocols
arb_dma_req[i]${sep}opt.src_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.src_protocol.value);
arb_dma_req[i]${sep}opt.dst_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.dst_protocol.value);
nxt_dma_req${sep}opt.src_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.src_protocol.value);
nxt_dma_req${sep}opt.dst_protocol = idma_pkg::protocol_e'(dma_reg2hw[i].conf.dst_protocol.value);

// Current backend only supports incremental burst
arb_dma_req[i]${sep}opt.src.burst = axi_pkg::BURST_INCR;
arb_dma_req[i]${sep}opt.dst.burst = axi_pkg::BURST_INCR;
nxt_dma_req${sep}opt.src.burst = axi_pkg::BURST_INCR;
nxt_dma_req${sep}opt.dst.burst = axi_pkg::BURST_INCR;
// this frontend currently does not support cache variations
arb_dma_req[i]${sep}opt.src.cache = axi_pkg::CACHE_MODIFIABLE;
arb_dma_req[i]${sep}opt.dst.cache = axi_pkg::CACHE_MODIFIABLE;
nxt_dma_req${sep}opt.src.cache = axi_pkg::CACHE_MODIFIABLE;
nxt_dma_req${sep}opt.dst.cache = axi_pkg::CACHE_MODIFIABLE;

// Backend options
arb_dma_req[i]${sep}opt.beo.decouple_aw = dma_reg2hw[i].conf.decouple_aw.value;
arb_dma_req[i]${sep}opt.beo.decouple_rw = dma_reg2hw[i].conf.decouple_rw.value;
arb_dma_req[i]${sep}opt.beo.src_max_llen = dma_reg2hw[i].conf.src_max_llen.value;
arb_dma_req[i]${sep}opt.beo.dst_max_llen = dma_reg2hw[i].conf.dst_max_llen.value;
arb_dma_req[i]${sep}opt.beo.src_reduce_len = dma_reg2hw[i].conf.src_reduce_len.value;
arb_dma_req[i]${sep}opt.beo.dst_reduce_len = dma_reg2hw[i].conf.dst_reduce_len.value;
nxt_dma_req${sep}opt.beo.decouple_aw = dma_reg2hw[i].conf.decouple_aw.value;
nxt_dma_req${sep}opt.beo.decouple_rw = dma_reg2hw[i].conf.decouple_rw.value;
nxt_dma_req${sep}opt.beo.src_max_llen = dma_reg2hw[i].conf.src_max_llen.value;
nxt_dma_req${sep}opt.beo.dst_max_llen = dma_reg2hw[i].conf.dst_max_llen.value;
nxt_dma_req${sep}opt.beo.src_reduce_len = dma_reg2hw[i].conf.src_reduce_len.value;
nxt_dma_req${sep}opt.beo.dst_reduce_len = dma_reg2hw[i].conf.dst_reduce_len.value;

% if num_dim != 1:
// ND connections
% for nd in range(0, num_dim-1):
% if bit_width == '32':
arb_dma_req[i].d_req[${nd}].reps = dma_reg2hw[i].dim[${nd}].reps[0].reps.value;
arb_dma_req[i].d_req[${nd}].src_strides = dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value;
arb_dma_req[i].d_req[${nd}].dst_strides = dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value;
nxt_dma_req.d_req[${nd}].reps = dma_reg2hw[i].dim[${nd}].reps[0].reps.value;
nxt_dma_req.d_req[${nd}].src_strides = dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value;
nxt_dma_req.d_req[${nd}].dst_strides = dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value;
% else:
arb_dma_req[i].d_req[${nd}].reps = {dma_reg2hw[i].dim[${nd}].reps[1].reps.value,
nxt_dma_req.d_req[${nd}].reps = {dma_reg2hw[i].dim[${nd}].reps[1].reps.value,
dma_reg2hw[i].dim[${nd}].reps[0].reps.value };
arb_dma_req[i].d_req[${nd}].src_strides = {dma_reg2hw[i].dim[${nd}].src_stride[1].src_stride.value,
nxt_dma_req.d_req[${nd}].src_strides = {dma_reg2hw[i].dim[${nd}].src_stride[1].src_stride.value,
dma_reg2hw[i].dim[${nd}].src_stride[0].src_stride.value};
arb_dma_req[i].d_req[${nd}].dst_strides = {dma_reg2hw[i].dim[${nd}].dst_stride[1].dst_stride.value,
nxt_dma_req.d_req[${nd}].dst_strides = {dma_reg2hw[i].dim[${nd}].dst_stride[1].dst_stride.value,
dma_reg2hw[i].dim[${nd}].dst_stride[0].dst_stride.value};
% endif
% endfor

// Disable higher dimensions
if ( dma_reg2hw[i].conf.enable_nd.value == 0) begin
% for nd in range(0, num_dim-1):
arb_dma_req[i].d_req[${nd}].reps = ${"'0" if nd != num_dim-2 else "'d1"};
nxt_dma_req.d_req[${nd}].reps = ${"'0" if nd != num_dim-2 else "'d1"};
% endfor
end
% for nd in range(1, num_dim-1):
else if ( dma_reg2hw[i].conf.enable_nd.value == ${nd}) begin
% for snd in range(nd, num_dim-1):
arb_dma_req[i].d_req[${snd}].reps = 'd1;
nxt_dma_req.d_req[${snd}].reps = 'd1;
% endfor
end
% endfor
% endif
end

// observational registers
// observational registers: drive .next (read-side launch is the rd_swacc strobe above)
for (genvar c = 0; c < NumStreams; c++) begin : gen_hw2reg_connections
assign dma_hw2reg[i].status[c].rd_data.busy = {midend_busy_i[c], busy_i[c]};
assign dma_hw2reg[i].status[c].rd_ack = dma_reg2hw[i].status[c].req
& ~dma_reg2hw[i].status[c].req_is_wr;
assign dma_hw2reg[i].next_id[c].rd_data.next_id = next_id_i;
assign dma_hw2reg[i].next_id[c].rd_ack = dma_reg2hw[i].next_id[c].req
& ~dma_reg2hw[i].next_id[c].req_is_wr
& arb_ready[i];
assign dma_hw2reg[i].done_id[c].rd_data.done_id = done_id_i[c];
assign dma_hw2reg[i].done_id[c].rd_ack = dma_reg2hw[i].done_id[c].req
& ~dma_reg2hw[i].done_id[c].req_is_wr;
assign dma_hw2reg[i].status[c].busy.next = {midend_busy_i[c], busy_i[c]};
assign dma_hw2reg[i].next_id[c].next_id.next = next_id_i;
assign dma_hw2reg[i].done_id[c].done_id.next = done_id_i[c];
end

// tie-off unused channels
for (genvar c = NumStreams; c < MaxNumStreams; c++) begin : gen_hw2reg_unused
assign dma_hw2reg[i].status[c].rd_data = '0;
assign dma_hw2reg[i].status[c].rd_ack = '0;
assign dma_hw2reg[i].next_id[c].rd_data.next_id = '0;
assign dma_hw2reg[i].next_id[c].rd_ack = '0;
assign dma_hw2reg[i].done_id[c].rd_data.done_id = '0;
assign dma_hw2reg[i].done_id[c].rd_ack = '0;
assign dma_hw2reg[i].status[c].busy.next = '0;
assign dma_hw2reg[i].next_id[c].next_id.next = '0;
assign dma_hw2reg[i].done_id[c].done_id.next = '0;
end

end
Expand All @@ -295,11 +305,11 @@ module idma_${identifier} #(
.rr_i ( '0 ),
.req_i ( arb_valid ),
.gnt_o ( arb_ready ),
.data_i ( arb_dma_req ),
.data_i ( arb_dma_req_q ),
.gnt_i ( req_ready_i ),
.req_o ( req_valid_o ),
.data_o ( dma_req_o ),
.idx_o ( /* NC */ )
.idx_o ( arb_idx )
);

endmodule
Loading
Loading