diff --git a/lib/rfnoc/crossbar/chdr_crossbar_nxn.v b/lib/rfnoc/crossbar/chdr_crossbar_nxn.v index 79bb40e..8be3b30 100644 --- a/lib/rfnoc/crossbar/chdr_crossbar_nxn.v +++ b/lib/rfnoc/crossbar/chdr_crossbar_nxn.v @@ -1,68 +1,132 @@ // -// Copyright 2018 Ettus Research, A National Instruments Company +// Copyright 2023 Ettus Research, a National Instruments Brand // // SPDX-License-Identifier: LGPL-3.0-or-later // // Module: chdr_crossbar_nxn -// Description: -// This module implements a full-bandwidth NxN crossbar with N input and output ports -// for CHDR traffic. It supports multiple optimization strategies for performance, -// area and timing tradeoffs. It uses AXI-Stream for all of its links. The crossbar -// has a dynamic routing table based on a Content Addressable Memory (CAM). The SID -// is used to determine the destination of a packet and the routing table contains -// a re-programmable SID to crossbar port mapping. The table is programmed using -// special route config packets on the data input ports or using an optional -// management port. -// The topology, routing algorithms and the router architecture is -// described in README.md in this directory. +// +// Description: +// +// This module implements a full-bandwidth NxN crossbar with N input and +// output ports for CHDR traffic. It supports multiple optimization +// strategies for performance, area and timing trade-offs. It uses AXI-Stream +// for all of its links. The crossbar has a dynamic routing table based on a +// Content Addressable Memory (CAM). The SID is used to determine the +// destination of a packet and the routing table contains a re-programmable +// SID to crossbar port mapping. The table is programmed using special route +// config packets on the data input ports or using an optional management +// port. +// +// The topology, routing algorithms and the router architecture is described +// in README.pdf in this directory. +// +// This crossbar also supports multiple port sizes. By default, each port +// will be PORT_W bits wide. This can be changed using the CHDR_WIDTHS +// parameter. This parameter allows the CHDR width of each port to be +// specified. When using multiple CHDR widths, the PORT_W parameter should be +// the size of the widest port. The CHDR_W value reported by the management +// port will be the value specified for that port in CHDR_WIDTHS. +// // Parameters: -// - CHDR_W: Width of the AXI-Stream data bus -// - NPORTS: Number of ports to instantiate -// - DEFAULT_PORT: The failsafe port to forward a packet to is SID mapping is missing -// - MTU: log2 of max packet size (in words) -// - ROUTE_TBL_SIZE: log2 of the number of mappings that the routing table can hold -// at any time. Mapping values are maintained in a FIFO fashion. -// - MUX_ALLOC: Algorithm to allocate the egress MUX -// * PRIO: Priority based. Lower port numbers have a higher priority -// * ROUND-ROBIN: Round robin input port allocation -// - OPTIMIZE: Optimization strategy for performance vs area vs timing tradeoffs -// * AREA: Attempt to minimize area at the cost of performance (throughput) and/or timing -// * PERFORMANCE: Attempt to maximize performance at the cost of area and/or timing -// * TIMING: Attempt to maximize Fmax at the cost of area and/or performance -// - NPORTS_MGMT: Number of ports with management endpoint. The first NPORTS_MGMT ports will -// have the management port instantiated -// - EXT_RTCFG_PORT: Enable a side-channel AXI-Stream management port to configure the -// routing table -// Signals: -// - s_axis_*: Slave port for router (flattened) -// - m_axis_*: Master port for router (flattened) -// - s_axis_mgmt_*: Management slave port -// - device_id: The ID of the device that has instantiated this module +// +// PORT_W : Width of the AXI-Stream data buses s_axis and m_axis. If +// using multiple port widths, this should be set to the +// width of the widest port. +// NPORTS : Number of ports to instantiate. +// CHDR_WIDTHS : Descending array of NUM_PORT integers representing the +// width of each crossbar port. The width of port n is given +// by CHDR_WIDTHS[(N+1)*32-1 : N*32]. +// ROUTES : Descending array representing which crossbar routes to +// enable. This is an NPORTS*NPORTS-bit array where bit +// [NPORTS*A + B] corresponds to the path from input port A +// to output port B. A '1' indicates the logic for that route +// is included. All routes are enabled by default. +// EN_ROUTE_FIFO : Set to 1 to include a FIFO on all routes going from a wide +// port to a narrow port. This may improve performance when a +// single wide input port streams to multiple narrow output +// ports by buffering the input data while it's resized for +// the slower output port. This helps to avoid congestion on +// the input port. +// EN_ROUTE_GATE : Set to 1 to include a packet gate on all routes going from +// a narrow port to a wide port. This may improve performance +// when multiple narrow input ports stream to a single wide +// output port by removing idle transfer cycles caused by the +// slower rate of the narrow input port. This helps to avoid +// congestion on the output port. +// DEFAULT_PORT : The fail-safe port to forward a packet to if SID mapping +// is missing. +// BYTE_MTU : log2 of the max packet size in bytes. +// ROUTE_TBL_SIZE: log2 of the number of mappings that the routing table can +// hold at any time. Mapping values are maintained in a FIFO +// fashion. +// MUX_ALLOC : Algorithm to allocate the egress MUX. Possible values: +// * "PRIO": Priority based. Lower port numbers have a +// higher priority +// * "ROUND-ROBIN": Round robin input port allocation +// OPTIMIZE : Optimization strategy for performance vs area vs timing +// trade-offs. Possible values: +// * "AREA": Attempt to minimize area at the cost of +// performance (throughput) and/or timing. +// * "PERFORMANCE": Attempt to maximize performance at the +// cost of area and/or timing. +// * "TIMING": Attempt to maximize Fmax at the cost of area +// and/or performance. +// NPORTS_MGMT : Number of ports with management endpoint. The first +// NPORTS_MGMT ports will have the management port +// instantiated. +// EXT_RTCFG_PORT: Enable a side-channel AXI-Stream management port to +// configure the routing table. +// +// CHDR_WIDTHS Bit Mapping Example (4x4): +// +// Port #: 3 2 1 0 +// ↓ ↓ ↓ ↓ +// {32'd64, 32'd64, 32'd64, 32'd64} +// +// ROUTES Bit Mapping Example (4x4): +// +// Output Port: 3210 +// ↓↓↓↓ +// Input Port 3 → {4'b1111, +// Input Port 2 → 4'b1111, +// Input Port 1 → 4'b1111, +// Input Port 0 → 4'b1111} +// +// Ports: +// +// s_axis_* : Slave port for router (flattened) +// m_axis_* : Master port for router (flattened) +// s_axis_mgmt_*: Management slave port +// device_id : The ID of the device that has instantiated this module // module chdr_crossbar_nxn #( - parameter [15:0] PROTOVER = {8'd1, 8'd0}, - parameter CHDR_W = 64, - parameter [7:0] NPORTS = 8, - parameter [7:0] DEFAULT_PORT = 0, - parameter MTU = 10, - parameter ROUTE_TBL_SIZE = 6, - parameter MUX_ALLOC = "ROUND-ROBIN", - parameter OPTIMIZE = "AREA", - parameter [7:0] NPORTS_MGMT = NPORTS, - parameter [0:0] EXT_RTCFG_PORT = 0 + parameter [15:0] PROTOVER = {8'd1, 8'd0}, + parameter [31:0] PORT_W = 64, + parameter [7:0] NPORTS = 8, + parameter [NPORTS*32-1:0] CHDR_WIDTHS = {NPORTS{PORT_W}}, + parameter EN_ROUTE_FIFO = 0, + parameter EN_ROUTE_GATE = 0, + parameter [7:0] DEFAULT_PORT = 0, + parameter [NPORTS**2-1:0] ROUTES = {NPORTS*NPORTS{1'b1}}, + parameter BYTE_MTU = $clog2(8192), + parameter ROUTE_TBL_SIZE = 6, + parameter MUX_ALLOC = "ROUND-ROBIN", + parameter OPTIMIZE = "AREA", + parameter [7:0] NPORTS_MGMT = NPORTS, + parameter EXT_RTCFG_PORT = 0 ) ( input wire clk, input wire reset, // Device info input wire [15:0] device_id, // Inputs - input wire [(CHDR_W*NPORTS)-1:0] s_axis_tdata, + input wire [(PORT_W*NPORTS)-1:0] s_axis_tdata, input wire [NPORTS-1:0] s_axis_tlast, input wire [NPORTS-1:0] s_axis_tvalid, output wire [NPORTS-1:0] s_axis_tready, // Output - output wire [(CHDR_W*NPORTS)-1:0] m_axis_tdata, + output wire [(PORT_W*NPORTS)-1:0] m_axis_tdata, output wire [NPORTS-1:0] m_axis_tlast, output wire [NPORTS-1:0] m_axis_tvalid, input wire [NPORTS-1:0] m_axis_tready, @@ -72,12 +136,17 @@ module chdr_crossbar_nxn #( input wire [31:0] ext_rtcfg_data, output wire ext_rtcfg_ack ); - // --------------------------------------------------- + //--------------------------------------------------------------------------- // RFNoC Includes - // --------------------------------------------------- + //--------------------------------------------------------------------------- + `include "../core/rfnoc_chdr_utils.vh" `include "../core/rfnoc_chdr_internal_utils.vh" + //--------------------------------------------------------------------------- + // Parameters + //--------------------------------------------------------------------------- + localparam NPORTS_W = $clog2(NPORTS); localparam EPID_W = 16; localparam [17:0] EXT_INFO = {1'b0, EXT_RTCFG_PORT, NPORTS_MGMT, NPORTS}; @@ -85,28 +154,74 @@ module chdr_crossbar_nxn #( localparam [0:0] PKT_ST_HEAD = 1'b0; localparam [0:0] PKT_ST_BODY = 1'b1; - // The compute_mux_alloc function is the switch allocation function for the MUX - // i.e. it chooses which input port reserves the output MUX for packet transfer. - function [NPORTS_W-1:0] compute_mux_alloc; - input [NPORTS-1:0] pkt_waiting; - input [NPORTS_W-1:0] last_alloc; + //--------------------------------------------------------------------------- + // Functions + //--------------------------------------------------------------------------- + + // The compute_mux_alloc function is the switch allocation function for the + // MUX. That is, it chooses which input port reserves the output MUX for + // packet transfer. + function [NPORTS_W-1:0] compute_mux_alloc( + input [ NPORTS-1:0] pkt_waiting, + input [NPORTS_W-1:0] last_alloc + ); reg signed [NPORTS_W:0] i; - begin - compute_mux_alloc = last_alloc; - for (i = NPORTS-1; i >= 0; i=i-1) begin - if (MUX_ALLOC == "PRIO") begin - // Priority. Lower port index gets a higher priority. - if (pkt_waiting[i]) - compute_mux_alloc = i; - end else begin - // Round-robin - if (pkt_waiting[(last_alloc + i + 1) % NPORTS]) - compute_mux_alloc = (last_alloc + i + 1) % NPORTS; + begin + compute_mux_alloc = last_alloc; + for (i = NPORTS-1; i >= 0; i=i-1) begin + if (MUX_ALLOC == "PRIO") begin + // Priority. Lower port index gets a higher priority. + if (pkt_waiting[i]) begin + compute_mux_alloc = i; + end + end else begin + // Round-robin + if (pkt_waiting[(last_alloc + i + 1) % NPORTS]) begin + compute_mux_alloc = (last_alloc + i + 1) % NPORTS; + end + end end end - end endfunction + // Return the CHDR width of the given port. + function [31:0] CHDR_W(input integer n); + CHDR_W = CHDR_WIDTHS[32*n +: 32]; + endfunction + + // Return the MTU size for the given port in terms of its CHDR width. + function [31:0] WORD_MTU(input integer n); + WORD_MTU = BYTE_MTU - $clog2(CHDR_W(n)/8); + endfunction + + // Return bit indicating if the route between input port i and output port j + // is enabled. + function [0:0] ROUTE_ENABLED(input integer i, j); + ROUTE_ENABLED = ROUTES[NPORTS*i + j]; + endfunction + + // Return bit indicating if the given input port has any routes connected to + // it. + function [0:0] INPUT_HAS_ROUTES(input integer i); + INPUT_HAS_ROUTES = |ROUTES[NPORTS*i +: NPORTS]; + endfunction + + // Return bit indicating if the given output port has any routes connected to + // it. + function automatic [0:0] OUTPUT_HAS_ROUTES(input integer j); + integer i; + begin + OUTPUT_HAS_ROUTES = 1'b0; + for (i = 0; i < NPORTS; i = i+1) begin + OUTPUT_HAS_ROUTES = OUTPUT_HAS_ROUTES | ROUTES[NPORTS*i + j]; + end + end + endfunction + + //--------------------------------------------------------------------------- + // CHDR Routing Table + //--------------------------------------------------------------------------- + wire [NPORTS-1:0] rtcfg_req_wr; wire [(16*NPORTS)-1:0] rtcfg_req_addr; wire [(32*NPORTS)-1:0] rtcfg_req_data; @@ -119,14 +234,15 @@ module chdr_crossbar_nxn #( wire [NPORTS-1:0] result_tvalid; wire [NPORTS-1:0] result_tready; - // Instantiate a single CAM-based routing table that will be shared between all - // input ports. Configuration and lookup is performed using an AXI-Stream iface. - // If multiple packets arrive simultaneously, only the headers of those packets will - // be serialized in order to arbitrate this map. Selection is done round-robin. + // Instantiate a single CAM-based routing table that will be shared between + // all input ports. Configuration and lookup is performed using an AXI-Stream + // interface. If multiple packets arrive simultaneously, only the headers of + // those packets will be serialized in order to arbitrate this map. Selection + // is done round-robin. chdr_xb_routing_table #( .SIZE(ROUTE_TBL_SIZE), .NPORTS(NPORTS), .EXT_INS_PORT_EN(EXT_RTCFG_PORT) - ) routing_tbl_i ( + ) chdr_xb_routing_table_i ( .clk (clk ), .reset (reset ), .port_req_wr (rtcfg_req_wr ), @@ -146,233 +262,427 @@ module chdr_crossbar_nxn #( .axis_result_tready(result_tready ) ); - wire [CHDR_W-1:0] i_tdata [0:NPORTS-1]; + wire [PORT_W-1:0] i_tdata [0:NPORTS-1]; wire [9:0] i_tdest [0:NPORTS-1]; wire [1:0] i_tid [0:NPORTS-1]; wire i_tlast [0:NPORTS-1]; wire i_tvalid [0:NPORTS-1]; wire i_tready [0:NPORTS-1]; - wire [CHDR_W-1:0] buf_tdata [0:NPORTS-1]; + wire [PORT_W-1:0] buf_tdata [0:NPORTS-1]; wire [NPORTS_W-1:0] buf_tdest [0:NPORTS-1], buf_tdest_tmp[0:NPORTS-1]; wire buf_tkeep [0:NPORTS-1]; wire buf_tlast [0:NPORTS-1]; wire buf_tvalid[0:NPORTS-1]; wire buf_tready[0:NPORTS-1]; - wire [CHDR_W-1:0] swi_tdata [0:NPORTS-1]; + wire [PORT_W-1:0] swi_tdata [0:NPORTS-1]; wire [NPORTS_W-1:0] swi_tdest [0:NPORTS-1]; wire swi_tlast [0:NPORTS-1]; wire swi_tvalid[0:NPORTS-1]; wire swi_tready[0:NPORTS-1]; - wire [(CHDR_W*NPORTS)-1:0] swo_tdata [0:NPORTS-1], muxi_tdata [0:NPORTS-1]; + wire [(PORT_W*NPORTS)-1:0] swo_tdata [0:NPORTS-1], muxi_tdata [0:NPORTS-1]; wire [NPORTS-1:0] swo_tlast [0:NPORTS-1], muxi_tlast [0:NPORTS-1]; wire [NPORTS-1:0] swo_tvalid[0:NPORTS-1], muxi_tvalid[0:NPORTS-1]; wire [NPORTS-1:0] swo_tready[0:NPORTS-1], muxi_tready[0:NPORTS-1]; - genvar n, i, j; + //--------------------------------------------------------------------------- + // Port Generation + //--------------------------------------------------------------------------- + + genvar n, i, j, port; generate - for (n = 0; n < NPORTS; n = n + 1) begin: i_ports - // For each input port, first check if we have a management packet - // arriving. If it arrives, the top config commands are extrated, sent to the - // routing table for configuration, and the rest of the packet is forwarded - // down to the router. - // the router. - if (n < NPORTS_MGMT) begin - chdr_mgmt_pkt_handler #( - .PROTOVER(PROTOVER), .CHDR_W(CHDR_W), .MGMT_ONLY(0) - ) mgmt_ep_i ( - .clk (clk ), - .rst (reset ), - .node_info (chdr_mgmt_build_node_info(EXT_INFO, n, NODE_TYPE_XBAR, device_id)), - .s_axis_chdr_tdata (s_axis_tdata [(n*CHDR_W)+:CHDR_W] ), - .s_axis_chdr_tlast (s_axis_tlast [n] ), - .s_axis_chdr_tvalid (s_axis_tvalid[n] ), - .s_axis_chdr_tready (s_axis_tready[n] ), - .s_axis_chdr_tuser (1'd0 ), - .m_axis_chdr_tdata (i_tdata [n] ), - .m_axis_chdr_tdest (i_tdest [n] ), - .m_axis_chdr_tid (i_tid [n] ), - .m_axis_chdr_tlast (i_tlast [n] ), - .m_axis_chdr_tvalid (i_tvalid [n] ), - .m_axis_chdr_tready (i_tready [n] ), - .ctrlport_req_wr (rtcfg_req_wr [n] ), - .ctrlport_req_rd (/* unused */ ), - .ctrlport_req_addr (rtcfg_req_addr[(n*16)+:16] ), - .ctrlport_req_data (rtcfg_req_data[(n*32)+:32] ), - .ctrlport_resp_ack (rtcfg_resp_ack[n] ), - .ctrlport_resp_data (32'h0 /* unused */ ), - .op_stb (/* unused */ ), - .op_dst_epid (/* unused */ ), - .op_src_epid (/* unused */ ), - .op_data (/* unused */ ) + for (n = 0; n < NPORTS; n = n + 1) begin: gen_in_ports + // Only generate the input logic for this input port if it has routes + if (INPUT_HAS_ROUTES(n)) begin : gen_in_port + + //----------------------------------------------------------------------- + // Assertions + //----------------------------------------------------------------------- + + // Make sure the width of this port does not exceed the given maximum + // port width. + if (CHDR_W(n) > PORT_W) begin : gen_chdr_w_too_large + ERROR__CHDR_W_must_not_exceed_PORT_W_parameter(); + end + + // Make sure the port's CHDR width is a valid CHDR width (a power of 2 + // and at least 64 bits). + if (2**$clog2(CHDR_W(n)) != CHDR_W(n) || CHDR_W(n) < 64) begin : gen_invalid_chdr_w + ERROR__CHDR_W_is_not_a_valid_CHDR_width(); + end + + // Make sure the maximum port width is a valid CHDR width (a power of 2 + // and at least 64 bits). + if (2**$clog2(PORT_W) != PORT_W || PORT_W < 64) begin : gen_invalid_port_w + ERROR__PORT_W_is_not_a_valid_CHDR_width(); + end + + //----------------------------------------------------------------------- + // Management Ports + //----------------------------------------------------------------------- + + wire [47:0] node_info = + chdr_mgmt_build_node_info(EXT_INFO, n, NODE_TYPE_XBAR, device_id); + + // For each input port, first check if we have a management packet + // arriving. If it arrives, the top config commands are extracted, sent + // to the routing table for configuration, and the rest of the packet is + // forwarded down to the router. the router. + if (n < NPORTS_MGMT) begin : gen_mgmt + chdr_mgmt_pkt_handler #( + .PROTOVER (PROTOVER ), + .CHDR_W (CHDR_W(n)), + .MGMT_ONLY (0 ) + ) chdr_mgmt_pkt_handler_i ( + .clk (clk ), + .rst (reset ), + .node_info (node_info ), + .s_axis_chdr_tdata (s_axis_tdata [(n*PORT_W)+:CHDR_W(n)]), + .s_axis_chdr_tlast (s_axis_tlast [n] ), + .s_axis_chdr_tvalid (s_axis_tvalid[n] ), + .s_axis_chdr_tready (s_axis_tready[n] ), + .s_axis_chdr_tuser (1'd0 ), + .m_axis_chdr_tdata (i_tdata [n] ), + .m_axis_chdr_tdest (i_tdest [n] ), + .m_axis_chdr_tid (i_tid [n] ), + .m_axis_chdr_tlast (i_tlast [n] ), + .m_axis_chdr_tvalid (i_tvalid [n] ), + .m_axis_chdr_tready (i_tready [n] ), + .ctrlport_req_wr (rtcfg_req_wr [n] ), + .ctrlport_req_rd (/* unused */ ), + .ctrlport_req_addr (rtcfg_req_addr[(n*16)+:16] ), + .ctrlport_req_data (rtcfg_req_data[(n*32)+:32] ), + .ctrlport_resp_ack (rtcfg_resp_ack[n] ), + .ctrlport_resp_data (32'h0 /* unused */ ), + .op_stb (/* unused */ ), + .op_dst_epid (/* unused */ ), + .op_src_epid (/* unused */ ), + .op_data (/* unused */ ) + ); + end else begin : gen_no_mgmt + assign i_tdata [n] = s_axis_tdata [(n*PORT_W)+:CHDR_W(n)]; + assign i_tid [n] = CHDR_MGMT_ROUTE_EPID; + assign i_tdest [n] = 10'd0; // Unused + assign i_tlast [n] = s_axis_tlast [n]; + assign i_tvalid [n] = s_axis_tvalid[n]; + assign s_axis_tready[n] = i_tready [n]; + + assign rtcfg_req_wr [n] = 1'b0; + assign rtcfg_req_addr[(n*16)+:16] = 16'h0; + assign rtcfg_req_data[(n*32)+:32] = 32'h0; + end + + //----------------------------------------------------------------------- + // Port Ingress Buffer + //----------------------------------------------------------------------- + + // Ingress buffer module that does the following: + // - Stores and gates an incoming packet + // - Looks up destination in routing table and attaches a tdest for the packet + chdr_xb_ingress_buff #( + .WIDTH (CHDR_W(n) ), + .MTU (WORD_MTU(n)), + .DEST_W (NPORTS_W ), + .NODE_ID(n ) + ) chdr_xb_ingress_buff_i ( + .clk (clk ), + .reset (reset ), + .s_axis_chdr_tdata (i_tdata [n] ), + .s_axis_chdr_tdest (i_tdest [n][NPORTS_W-1:0] ), + .s_axis_chdr_tid (i_tid [n] ), + .s_axis_chdr_tlast (i_tlast [n] ), + .s_axis_chdr_tvalid (i_tvalid [n] ), + .s_axis_chdr_tready (i_tready [n] ), + .m_axis_chdr_tdata (buf_tdata [n] ), + .m_axis_chdr_tdest (buf_tdest_tmp[n] ), + .m_axis_chdr_tkeep (buf_tkeep [n] ), + .m_axis_chdr_tlast (buf_tlast [n] ), + .m_axis_chdr_tvalid (buf_tvalid [n] ), + .m_axis_chdr_tready (buf_tready [n] ), + .m_axis_find_tdata (find_tdata [(n*EPID_W)+:EPID_W] ), + .m_axis_find_tvalid (find_tvalid [n] ), + .m_axis_find_tready (find_tready [n] ), + .s_axis_result_tdata (result_tdata [(n*NPORTS_W)+:NPORTS_W]), + .s_axis_result_tkeep (result_tkeep [n] ), + .s_axis_result_tvalid(result_tvalid[n] ), + .s_axis_result_tready(result_tready[n] ) ); - end else begin - assign i_tdata [n] = s_axis_tdata [(n*CHDR_W)+:CHDR_W]; - assign i_tid [n] = CHDR_MGMT_ROUTE_EPID; - assign i_tdest [n] = 10'd0; // Unused - assign i_tlast [n] = s_axis_tlast [n]; - assign i_tvalid [n] = s_axis_tvalid[n]; - assign s_axis_tready[n] = i_tready [n]; + assign buf_tdest[n] = buf_tkeep[n] ? buf_tdest_tmp[n] : DEFAULT_PORT[NPORTS_W-1:0]; - assign rtcfg_req_wr [n] = 1'b0; - assign rtcfg_req_addr[(n*16)+:16] = 16'h0; - assign rtcfg_req_data[(n*32)+:32] = 32'h0; - end + // Pipeline stage + axi_fifo #( + .WIDTH(CHDR_W(n)+1+NPORTS_W), + .SIZE (1 ) + ) axi_fifo_i ( + .clk (clk ), + .reset (reset ), + .clear (1'b0 ), + .i_tdata ({buf_tlast[n], buf_tdest[n], buf_tdata[n][CHDR_W(n)-1:0]}), + .i_tvalid(buf_tvalid[n] ), + .i_tready(buf_tready[n] ), + .o_tdata ({swi_tlast[n], swi_tdest[n], swi_tdata[n][CHDR_W(n)-1:0]}), + .o_tvalid(swi_tvalid[n] ), + .o_tready(swi_tready[n] ), + .space (/* Unused */ ), + .occupied(/* Unused */ ) + ); - // Ingress buffer module that does the following: - // - Stores and gates an incoming packet - // - Looks up destination in routing table and attaches a tdest for the packet - chdr_xb_ingress_buff #( - .WIDTH(CHDR_W), .MTU(MTU), .DEST_W(NPORTS_W), .NODE_ID(n) - ) buf_i ( - .clk (clk ), - .reset (reset ), - .s_axis_chdr_tdata (i_tdata [n] ), - .s_axis_chdr_tdest (i_tdest [n][NPORTS_W-1:0] ), - .s_axis_chdr_tid (i_tid [n] ), - .s_axis_chdr_tlast (i_tlast [n] ), - .s_axis_chdr_tvalid (i_tvalid [n] ), - .s_axis_chdr_tready (i_tready [n] ), - .m_axis_chdr_tdata (buf_tdata [n] ), - .m_axis_chdr_tdest (buf_tdest_tmp[n] ), - .m_axis_chdr_tkeep (buf_tkeep [n] ), - .m_axis_chdr_tlast (buf_tlast [n] ), - .m_axis_chdr_tvalid (buf_tvalid [n] ), - .m_axis_chdr_tready (buf_tready [n] ), - .m_axis_find_tdata (find_tdata [(n*EPID_W)+:EPID_W] ), - .m_axis_find_tvalid (find_tvalid [n] ), - .m_axis_find_tready (find_tready [n] ), - .s_axis_result_tdata (result_tdata [(n*NPORTS_W)+:NPORTS_W]), - .s_axis_result_tkeep (result_tkeep [n] ), - .s_axis_result_tvalid(result_tvalid[n] ), - .s_axis_result_tready(result_tready[n] ) - ); - assign buf_tdest[n] = buf_tkeep[n] ? buf_tdest_tmp[n] : DEFAULT_PORT[NPORTS_W-1:0]; + //----------------------------------------------------------------------- + // Ingress Switch (De-multiplexers) + //----------------------------------------------------------------------- - // Pipeline state - axi_fifo #( - .WIDTH(CHDR_W+1+NPORTS_W), .SIZE(1) - ) pipe_i ( - .clk (clk ), - .reset (reset ), - .clear (1'b0 ), - .i_tdata ({buf_tlast[n], buf_tdest[n], buf_tdata[n]}), - .i_tvalid (buf_tvalid[n] ), - .i_tready (buf_tready[n] ), - .o_tdata ({swi_tlast[n], swi_tdest[n], swi_tdata[n]}), - .o_tvalid (swi_tvalid[n] ), - .o_tready (swi_tready[n] ), - .space (/* Unused */ ), - .occupied (/* Unused */ ) - ); + wire [CHDR_W(n)*NPORTS-1:0] swo_tdata_packed; - // Ingress demux. Use the tdest field to determine packet destination - axis_switch #( - .DATA_W(CHDR_W), .DEST_W(1), .IN_PORTS(1), .OUT_PORTS(NPORTS), .PIPELINE(1) - ) demux_i ( - .clk (clk ), - .reset (reset ), - .s_axis_tdata (swi_tdata [n] ), - .s_axis_tdest ({1'b0, swi_tdest [n]}), - .s_axis_tlast (swi_tlast [n] ), - .s_axis_tvalid (swi_tvalid[n] ), - .s_axis_tready (swi_tready[n] ), - .s_axis_alloc (1'b0 ), - .m_axis_tdata (swo_tdata [n] ), - .m_axis_tdest (/* Unused */ ), - .m_axis_tlast (swo_tlast [n] ), - .m_axis_tvalid (swo_tvalid[n] ), - .m_axis_tready (swo_tready[n] ) - ); - end - - for (i = 0; i < NPORTS; i = i + 1) begin - for (j = 0; j < NPORTS; j = j + 1) begin - assign muxi_tdata [i][j*CHDR_W+:CHDR_W] = swo_tdata [j][i*CHDR_W+:CHDR_W]; - assign muxi_tlast [i][j] = swo_tlast [j][i]; - assign muxi_tvalid[i][j] = swo_tvalid [j][i]; - assign swo_tready [i][j] = muxi_tready[j][i]; - end - end - - for (n = 0; n < NPORTS; n = n + 1) begin: o_ports - if (OPTIMIZE == "PERFORMANCE") begin - // Use the axis_switch module when optimizing for performance - // This logic has some extra levels of logic to ensure - // that the switch allocation happens in 0 clock cycles which - // means that Fmax for this implementation will be lower. - - wire mux_ready = |muxi_tready[n]; // Max 1 bit should be high - wire mux_valid = |muxi_tvalid[n]; - wire mux_last = |(muxi_tvalid[n] & muxi_tlast[n]); - - // Track the input packet state - reg [0:0] pkt_state = PKT_ST_HEAD; - always @(posedge clk) begin - if (reset) begin - pkt_state <= PKT_ST_HEAD; - end else if (mux_valid & mux_ready) begin - pkt_state <= mux_last ? PKT_ST_HEAD : PKT_ST_BODY; - end - end - - // The switch requires the allocation to stay valid until the - // end of the packet. We also might need to keep the previous - // packet's allocation to compute the current one - reg [NPORTS_W-1:0] prev_sw_alloc = {NPORTS_W{1'b0}}; - reg [NPORTS_W-1:0] pkt_sw_alloc = {NPORTS_W{1'b0}}; - wire [NPORTS_W-1:0] muxi_sw_alloc = (mux_valid && pkt_state == PKT_ST_HEAD) ? - compute_mux_alloc(muxi_tvalid[n], prev_sw_alloc) : pkt_sw_alloc; - - always @(posedge clk) begin - if (reset) begin - prev_sw_alloc <= {NPORTS_W{1'b0}}; - pkt_sw_alloc <= {NPORTS_W{1'b0}}; - end else if (mux_valid & mux_ready) begin - if (pkt_state == PKT_ST_HEAD) - pkt_sw_alloc <= muxi_sw_alloc; - if (mux_last) - prev_sw_alloc <= muxi_sw_alloc; - end - end - + // Ingress de-mux. Use the tdest field to determine packet destination. axis_switch #( - .DATA_W(CHDR_W), .DEST_W(1), .IN_PORTS(NPORTS), .OUT_PORTS(1), - .PIPELINE(0) - ) mux_i ( - .clk (clk ), - .reset (reset ), - .s_axis_tdata (muxi_tdata [n] ), - .s_axis_tdest ({NPORTS{1'b0}} /* Unused */ ), - .s_axis_tlast (muxi_tlast [n] ), - .s_axis_tvalid (muxi_tvalid[n] ), - .s_axis_tready (muxi_tready[n] ), - .s_axis_alloc (muxi_sw_alloc ), - .m_axis_tdata (m_axis_tdata [(n*CHDR_W)+:CHDR_W]), - .m_axis_tdest (/* Unused */ ), - .m_axis_tlast (m_axis_tlast [n] ), - .m_axis_tvalid (m_axis_tvalid[n] ), - .m_axis_tready (m_axis_tready[n] ) - ); - end else begin - // axi_mux has an additional bubble cycle but the logic - // to allocate an input port has fewer levels and takes - // up fewer resources. - axi_mux #( - .PRIO(MUX_ALLOC == "PRIO"), .WIDTH(CHDR_W), .SIZE(NPORTS), - .PRE_FIFO_SIZE(OPTIMIZE == "TIMING" ? 1 : 0), .POST_FIFO_SIZE(1) - ) mux_i ( - .clk (clk ), - .reset (reset ), - .clear (1'b0 ), - .i_tdata (muxi_tdata [n] ), - .i_tlast (muxi_tlast [n] ), - .i_tvalid (muxi_tvalid [n] ), - .i_tready (muxi_tready [n] ), - .o_tdata (m_axis_tdata [(n*CHDR_W)+:CHDR_W]), - .o_tlast (m_axis_tlast [n] ), - .o_tvalid (m_axis_tvalid[n] ), - .o_tready (m_axis_tready[n] ) + .DATA_W (CHDR_W(n)), + .DEST_W (1 ), + .IN_PORTS (1 ), + .OUT_PORTS(NPORTS ), + .PIPELINE (1 ) + ) axis_switch_demux ( + .clk (clk ), + .reset (reset ), + .s_axis_tdata (swi_tdata[n][CHDR_W(n)-1:0]), + .s_axis_tdest ({1'b0, swi_tdest[n]} ), + .s_axis_tlast (swi_tlast [n] ), + .s_axis_tvalid(swi_tvalid[n] ), + .s_axis_tready(swi_tready[n] ), + .s_axis_alloc (1'b0 ), + .m_axis_tdata (swo_tdata_packed ), + .m_axis_tdest (/* Unused */ ), + .m_axis_tlast (swo_tlast [n] ), + .m_axis_tvalid(swo_tvalid[n] ), + .m_axis_tready(swo_tready[n] ) ); + + // Unpack the switch output to handle the case where this port's CHDR_W + // is narrower than PORT_W. + for (port = 0; port < NPORTS; port = port+1) begin : gen_switch_output + assign swo_tdata[n][PORT_W*port +: CHDR_W(n)] = + swo_tdata_packed[CHDR_W(n)*port +: CHDR_W(n)]; + end + end // gen_in_port + end // gen_in_ports + + //------------------------------------------------------------------------- + // Crossbar Routing + //------------------------------------------------------------------------- + + // Generate the routing for a full NxN crossbar where i is the input port + // number and j is the output port number. Some paths are resized, + // depending on CHDR_WIDTHS, or excluded, depending on ROUTES. + for (i = 0; i < NPORTS; i = i + 1) begin : gen_for_i + for (j = 0; j < NPORTS; j = j + 1) begin : gen_for_j + if (ROUTE_ENABLED(i,j)) begin : gen_enabled_route + wire [CHDR_W(i)-1:0] rs_i_tdata; + wire rs_i_tlast; + wire rs_i_tvalid; + wire rs_i_tready; + + wire [CHDR_W(j)-1:0] rs_o_tdata; + wire rs_o_tlast; + wire rs_o_tvalid; + wire rs_o_tready; + + // Connect output j of ingress port i to input i of egress port j. + // Resize the bus if the ports have different widths, otherwise + // directly connect them. + if (CHDR_W(i) != CHDR_W(j)) begin : gen_port_resize + if (CHDR_W(i) > CHDR_W(j) && EN_ROUTE_FIFO) begin : gen_input_fifo + // If we're downsizing, we need a wide FIFO on the input to the + // resize block to buffer the fast incoming packet. + axi_fifo #( + .WIDTH(CHDR_W(i)+1), + .SIZE (WORD_MTU(i)) + ) axi_fifo_i ( + .clk (clk ), + .reset (reset ), + .clear (1'b0 ), + .i_tdata ({swo_tlast[i][j], swo_tdata[i][j*PORT_W+:CHDR_W(i)]}), + .i_tvalid(swo_tvalid[i][j] ), + .i_tready(swo_tready[i][j] ), + .o_tdata ({rs_i_tlast, rs_i_tdata} ), + .o_tvalid(rs_i_tvalid ), + .o_tready(rs_i_tready ), + .space ( ), + .occupied( ) + ); + end else begin : gen_no_input_fifo + assign rs_i_tdata = swo_tdata[i][j*PORT_W+:CHDR_W(i)]; + assign rs_i_tlast = swo_tlast[i][j]; + assign rs_i_tvalid = swo_tvalid[i][j]; + assign swo_tready[i][j] = rs_i_tready; + end + + chdr_resize #( + .I_CHDR_W(CHDR_W(i)), + .O_CHDR_W(CHDR_W(j)), + .I_DATA_W(CHDR_W(i)), + .O_DATA_W(CHDR_W(j)), + .USER_W (1 ), + .PIPELINE("OUT" ) + ) chdr_resize_i ( + .clk (clk ), + .rst (reset ), + .i_chdr_tdata (rs_i_tdata ), + .i_chdr_tuser (1'b0 ), + .i_chdr_tlast (rs_i_tlast ), + .i_chdr_tvalid(rs_i_tvalid), + .i_chdr_tready(rs_i_tready), + .o_chdr_tdata (rs_o_tdata ), + .o_chdr_tuser ( ), + .o_chdr_tlast (rs_o_tlast ), + .o_chdr_tvalid(rs_o_tvalid), + .o_chdr_tready(rs_o_tready) + ); + + if (CHDR_W(i) < CHDR_W(j) && EN_ROUTE_GATE) begin : gen_output_pkt_gate + // If we are up-sizing, then there will be idle cycles on the + // wider output bus that will waste time on the output mux. To + // maximize throughput on the output port, we gate packets here + // so that we can output a continuous stream of data without idle + // cycles. + axi_packet_gate #( + .WIDTH(CHDR_W(j) ), + .SIZE (WORD_MTU(j)) + ) axi_packet_gate_i ( + .clk (clk ), + .reset (reset ), + .clear (1'b0 ), + .i_tdata (rs_o_tdata ), + .i_tlast (rs_o_tlast ), + .i_terror(1'b0 ), + .i_tvalid(rs_o_tvalid ), + .i_tready(rs_o_tready ), + .o_tdata (muxi_tdata[j][i*PORT_W+:CHDR_W(j)]), + .o_tlast (muxi_tlast[j][i] ), + .o_tvalid(muxi_tvalid[j][i] ), + .o_tready(muxi_tready[j][i] ) + ); + end else begin : gen_no_output_pkt_gate + assign muxi_tdata[j][i*PORT_W+:CHDR_W(j)] = rs_o_tdata; + assign muxi_tlast[j][i] = rs_o_tlast; + assign muxi_tvalid[j][i] = rs_o_tvalid; + assign rs_o_tready = muxi_tready[j][i]; + end + + end else begin : gen_port_same_size + assign muxi_tdata[j][i*PORT_W+:CHDR_W(j)] = swo_tdata [i][j*PORT_W+:CHDR_W(i)]; + assign muxi_tlast[j][i] = swo_tlast [i][j]; + assign muxi_tvalid[j][i] = swo_tvalid [i][j]; + assign swo_tready[i][j] = muxi_tready[j][i]; + end + end else begin : gen_disabled_route + // Tie off these unused paths so they can be optimized out. + assign muxi_tdata[j][i*PORT_W+:PORT_W] = { PORT_W {1'b0} }; + assign muxi_tlast[j][i] = 1'b0; + assign muxi_tvalid[j][i] = 1'b0; + assign swo_tready[i][j] = 1'b1; + end + end + end + + //------------------------------------------------------------------------- + // Egress Switch (Multiplexers) + //------------------------------------------------------------------------- + + for (n = 0; n < NPORTS; n = n + 1) begin: gen_out_ports + // Only generate egress logic for this output port if it has routes + if (OUTPUT_HAS_ROUTES(n)) begin : gen_out_port + wire [CHDR_W(n)*NPORTS-1:0] muxi_tdata_repacked; + + // Repack the mux input to handle the case where this port's CHDR_W is + // narrower than PORT_W. + for (port = 0; port < NPORTS; port = port+1) begin : gen_mux_input + assign muxi_tdata_repacked[CHDR_W(n)*port +: CHDR_W(n)] = + muxi_tdata[n][PORT_W*port +: CHDR_W(n)]; + end + + if (OPTIMIZE == "PERFORMANCE") begin : gen_performance + // Use the axis_switch module when optimizing for performance + // This logic has some extra levels of logic to ensure + // that the switch allocation happens in 0 clock cycles which + // means that Fmax for this implementation will be lower. + + wire mux_ready = |muxi_tready[n]; // Max 1 bit should be high + wire mux_valid = |muxi_tvalid[n]; + wire mux_last = |(muxi_tvalid[n] & muxi_tlast[n]); + + // Track the input packet state + reg [0:0] pkt_state = PKT_ST_HEAD; + always @(posedge clk) begin + if (reset) begin + pkt_state <= PKT_ST_HEAD; + end else if (mux_valid & mux_ready) begin + pkt_state <= mux_last ? PKT_ST_HEAD : PKT_ST_BODY; + end + end + + // The switch requires the allocation to stay valid until the + // end of the packet. We also might need to keep the previous + // packet's allocation to compute the current one + reg [NPORTS_W-1:0] prev_sw_alloc = {NPORTS_W{1'b0}}; + reg [NPORTS_W-1:0] pkt_sw_alloc = {NPORTS_W{1'b0}}; + wire [NPORTS_W-1:0] muxi_sw_alloc = (mux_valid && pkt_state == PKT_ST_HEAD) ? + compute_mux_alloc(muxi_tvalid[n], prev_sw_alloc) : pkt_sw_alloc; + + always @(posedge clk) begin + if (reset) begin + prev_sw_alloc <= {NPORTS_W{1'b0}}; + pkt_sw_alloc <= {NPORTS_W{1'b0}}; + end else if (mux_valid & mux_ready) begin + if (pkt_state == PKT_ST_HEAD) + pkt_sw_alloc <= muxi_sw_alloc; + if (mux_last) + prev_sw_alloc <= muxi_sw_alloc; + end + end + + axis_switch #( + .DATA_W (CHDR_W(n)), + .DEST_W (1 ), + .IN_PORTS (NPORTS ), + .OUT_PORTS (1 ), + .PIPELINE (0 ) + ) axis_switch_mux ( + .clk (clk ), + .reset (reset ), + .s_axis_tdata (muxi_tdata_repacked ), + .s_axis_tdest ({NPORTS{1'b0}} /* Unused */ ), + .s_axis_tlast (muxi_tlast [n] ), + .s_axis_tvalid (muxi_tvalid[n] ), + .s_axis_tready (muxi_tready[n] ), + .s_axis_alloc (muxi_sw_alloc ), + .m_axis_tdata (m_axis_tdata [(n*PORT_W)+:CHDR_W(n)]), + .m_axis_tdest (/* Unused */ ), + .m_axis_tlast (m_axis_tlast [n] ), + .m_axis_tvalid (m_axis_tvalid[n] ), + .m_axis_tready (m_axis_tready[n] ) + ); + end else begin : gen_not_performance + // axi_mux has an additional bubble cycle but the logic + // to allocate an input port has fewer levels and takes + // up fewer resources. + axi_mux #( + .PRIO (MUX_ALLOC == "PRIO" ), + .WIDTH (CHDR_W(n) ), + .SIZE (NPORTS ), + .PRE_FIFO_SIZE (OPTIMIZE == "TIMING" ? 1 : 0), + .POST_FIFO_SIZE(1 ) + ) axi_mux_i ( + .clk (clk ), + .reset (reset ), + .clear (1'b0 ), + .i_tdata (muxi_tdata_repacked ), + .i_tlast (muxi_tlast [n] ), + .i_tvalid(muxi_tvalid [n] ), + .i_tready(muxi_tready [n] ), + .o_tdata (m_axis_tdata [(n*PORT_W)+:CHDR_W(n)]), + .o_tlast (m_axis_tlast [n] ), + .o_tvalid(m_axis_tvalid[n] ), + .o_tready(m_axis_tready[n] ) + ); + end end end endgenerate diff --git a/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/Makefile b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/Makefile new file mode 100644 index 0000000..f25987b --- /dev/null +++ b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/Makefile @@ -0,0 +1,54 @@ +# +# Copyright 2023 Ettus Research, a National Instruments Brand +# +# SPDX-License-Identifier: LGPL-3.0-or-later +# + +#------------------------------------------------- +# Top-of-Makefile +#------------------------------------------------- +# Define BASE_DIR to point to the "top" dir +BASE_DIR = $(abspath ../../../../top) +# Include viv_sim_preamble after defining BASE_DIR +include $(BASE_DIR)/../tools/make/viv_sim_preamble.mak + +#------------------------------------------------- +# Design Specific +#------------------------------------------------- +# Define part using PART_ID (//) +ARCH = kintex7 +PART_ID = xc7k410t/ffg900/-2 + +# Include makefiles and sources for the DUT and its dependencies +include $(BASE_DIR)/../lib/axi/Makefile.srcs +include $(BASE_DIR)/../lib/control/Makefile.srcs +include $(BASE_DIR)/../lib/fifo/Makefile.srcs +include $(BASE_DIR)/../lib/rfnoc/crossbar/Makefile.srcs +include $(BASE_DIR)/../lib/rfnoc/core/Makefile.srcs +include $(BASE_DIR)/../lib/rfnoc/utils/Makefile.srcs + +DESIGN_SRCS = $(abspath \ +$(AXI_SRCS) \ +$(FIFO_SRCS) \ +$(CONTROL_LIB_SRCS) \ +$(RFNOC_XBAR_SRCS) \ +$(RFNOC_CORE_SRCS) \ +$(RFNOC_UTIL_SRCS) \ +) + +#------------------------------------------------- +# Testbench Specific +#------------------------------------------------- +# Define only one toplevel module +SIM_TOP = chdr_crossbar_nxn_all_tb +SIM_SRCS = \ +$(abspath chdr_crossbar_nxn_tb.sv) \ +$(abspath chdr_crossbar_nxn_all_tb.sv) \ + +#------------------------------------------------- +# Bottom-of-Makefile +#------------------------------------------------- +# Include all simulator specific makefiles here +# Each should define a unique target to simulate +# e.g. xsim, vsim, etc and a common "clean" target +include $(BASE_DIR)/../tools/make/viv_simulator.mak diff --git a/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_all_tb.sv b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_all_tb.sv new file mode 100644 index 0000000..dbb0d0f --- /dev/null +++ b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_all_tb.sv @@ -0,0 +1,68 @@ +// +// Copyright 2023 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: chdr_crossbar_nxn +// +// Description: +// +// Top-level testbench for chdr_crossbar_nxn that instantiates multiple +// permutations of the DUT. +// + + +module chdr_crossbar_nxn_all_tb; + + localparam NUM_PKTS = 64; + + // 2x2, 64-bit, fully connected + chdr_crossbar_nxn_tb #( + .NUM_PORTS (2 ), + .CHDR_WIDTHS ('{64, 64}), + .ROUTES ('{'b11, + 'b11} ), + .TEST_BAD_ROUTES(1 ), + .NUM_PKTS (NUM_PKTS ), + .USE_MGMT_PORTS (0 ) + ) chdr_crossbar_nxn_tb_2x2 (); + + // 3x3, 128-bit, loopback disabled + chdr_crossbar_nxn_tb #( + .NUM_PORTS (3 ), + .CHDR_WIDTHS ('{128, 128, 128}), + .ROUTES ('{'b011, + 'b101, + 'b110} ), + .TEST_BAD_ROUTES(1 ), + .NUM_PKTS (NUM_PKTS ), + .USE_MGMT_PORTS (0 ) + ) chdr_crossbar_nxn_tb_3x3 (); + + // 4x4, variable port widths, paths disabled + chdr_crossbar_nxn_tb #( + .NUM_PORTS (4 ), + .CHDR_WIDTHS ('{64, 128, 256, 512}), + .ROUTES ('{'b0111, + 'b1011, + 'b1101, + 'b1010} ), + .TEST_BAD_ROUTES(1 ), + .NUM_PKTS (NUM_PKTS ), + .USE_MGMT_PORTS (0 ) + ) chdr_crossbar_nxn_tb_4x4_var (); + + // 3x3, variable port widths, paths disabled. This tests 2 64-bit ports + // saturating a single 128-bit port. + chdr_crossbar_nxn_tb #( + .NUM_PORTS (3 ), + .CHDR_WIDTHS ('{64, 64, 128}), + .ROUTES ('{'b001, + 'b001, + 'b000} ), + .TEST_BAD_ROUTES(0 ), + .NUM_PKTS (NUM_PKTS ), + .USE_MGMT_PORTS (0 ) + ) chdr_crossbar_nxn_tb_3x3_var (); + +endmodule diff --git a/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_tb.sv b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_tb.sv new file mode 100644 index 0000000..ba28fa1 --- /dev/null +++ b/lib/rfnoc/crossbar/chdr_crossbar_nxn_tb/chdr_crossbar_nxn_tb.sv @@ -0,0 +1,565 @@ +// +// Copyright 2023 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Description: Testbench for chdr_crossbar_nxn. +// +// Parameters: +// +// NUM_PORTS : Size of crossbar to test will be NUM_PORTS x NUM_PORTS. +// CHDR_WIDTHS : Descending array of NUM_PORT integers representing the +// width of each crossbar port. The width of port n is +// given by CHDR_WIDTHS[n]. +// ROUTES : Descending 2D array representing which crossbar routes +// to enable. This is an NPORTS x NPORTS array where bit +// ROUTES[A][B] corresponds to the path from input port A +// to output port B. A '1' indicates the logic for that +// route is included. +// TEST_BAD_ROUTES : When 1, test all routes, whether enabled or not. When 0, +// test only the enabled routes specified in the ROUTES +// parameter. +// NUM_PKTS : Number of packets to test on each port. This determines +// the length of the simulation. + +// USE_MGMT_PORTS : Indicates whether or not to use the management port for +// crossbar configuration. Set to 1 to use the CHDR +// management port. Set to 0 to use the CtrlPort instead. +// +// Note: The array parameters above are SystemVerilog arrays, but they are +// ordered to match the DUT, which uses flat Verilog arrays. +// +// CHDR_WIDTHS Bit Mapping Example (4x4): +// +// Port: 3 2 1 0 +// ↓ ↓ ↓ ↓ +// '{64, 64, 64, 64} +// +// ROUTES Bit Mapping Example (4x4): +// +// Output Port 3210 +// ↓↓↓↓ +// Input Port 3 → '{'b1111, +// Input Port 2 → 'b1111, +// Input Port 1 → 'b1111, +// Input Port 0 → 'b1111} +// + +`default_nettype none + + +module chdr_crossbar_nxn_tb #( + parameter int NUM_PORTS = 2, + parameter bit [NUM_PORTS-1:0][31:0] CHDR_WIDTHS = {NUM_PORTS{32'd64}}, + parameter bit [NUM_PORTS-1:0][NUM_PORTS-1:0] ROUTES = {NUM_PORTS**2{1'b1}}, + parameter bit TEST_BAD_ROUTES = 1, + parameter int NUM_PKTS = 64, + parameter bit USE_MGMT_PORTS = 0 +); + // Include macros and time declarations for use with PkgTestExec + `include "test_exec.svh" + + import PkgTestExec::*; + import PkgChdrBfm::*; + import PkgChdrUtils::*; + + + //--------------------------------------------------------------------------- + // Functions + //--------------------------------------------------------------------------- + + // Determine the largest port size, which will be used as the signal width + // for the crossbar ports. + function static int max_port_width(); + static int max = 0; + for (int i = 0; i < NUM_PORTS; i++) begin + if (CHDR_WIDTHS[i] > max) max = CHDR_WIDTHS[i]; + end + return max; + endfunction : max_port_width + + + //--------------------------------------------------------------------------- + // Local Parameters + //--------------------------------------------------------------------------- + + localparam real CLK_PERIOD = 10.0; + localparam bit DEBUG = 0; // Set to 1 to enable more log prints + localparam int MAX_PKT_BYTES = 512; // Max packet length in bytes to test + + localparam int BYTE_MTU = $clog2(MAX_PKT_BYTES); + localparam int PORT_W = max_port_width(); + // MAX_PYLD_BYTES is MTU minus two CHDR words for the header + localparam int MAX_PYLD_BYTES = MAX_PKT_BYTES - 2*(PORT_W/8); + + // DUT default parameters + localparam [15:0] PROTOVER = {8'd1, 8'd0}; + localparam [7:0] DEFAULT_PORT = 0; + localparam ROUTE_TBL_SIZE = NUM_PORTS**2; // One route for every port combination + localparam MUX_ALLOC = "ROUND-ROBIN"; + localparam OPTIMIZE = "AREA"; + localparam [7:0] NPORTS_MGMT = USE_MGMT_PORTS ? NUM_PORTS : 0; + localparam EXT_RTCFG_PORT = 1; + localparam DEVICE_ID = 16'hBEEF; + + + //--------------------------------------------------------------------------- + // Inter-process Communication + //--------------------------------------------------------------------------- + + // Event to indicate if the mailboxes have been initialized. + event start_consumer; + + // Mailbox to communicate how many packets to expect. These mailboxes are + // created prior to the start_consumer event. + mailbox #(int) mb_num_pkts [NUM_PORTS]; + + // Semaphore to track the number of output ports that have received their + // expected number of packets. + semaphore ports_done = new(); + + + //--------------------------------------------------------------------------- + // Clocks and Resets + //--------------------------------------------------------------------------- + + bit clk, rst; + + sim_clock_gen #(.PERIOD(CLK_PERIOD), .AUTOSTART(0)) clk_gen (.clk(clk), .rst(rst)); + + + //--------------------------------------------------------------------------- + // Bus Functional Models + //--------------------------------------------------------------------------- + + // Interfaces for AXI-Stream + AxiStreamIf #(PORT_W) chdr_to_dut [NUM_PORTS] (clk, rst); + AxiStreamIf #(PORT_W) chdr_from_dut [NUM_PORTS] (clk, rst); + + // Bus functional model for each of the CHDR ports + ChdrBfm #(PORT_W) chdr_bfm [NUM_PORTS]; + + // Create the BFM instances + for (genvar i = 0; i < NUM_PORTS; i++) begin : gen_bfm_creation + initial chdr_bfm[i] = new(chdr_to_dut[i], chdr_from_dut[i]); + end + + + //--------------------------------------------------------------------------- + // DUT + //--------------------------------------------------------------------------- + + logic [NUM_PORTS-1:0][PORT_W-1:0] dut_in_tdata; + logic [NUM_PORTS-1:0][ 0:0] dut_in_tlast; + logic [NUM_PORTS-1:0][ 0:0] dut_in_tvalid; + logic [NUM_PORTS-1:0][ 0:0] dut_in_tready; + logic [NUM_PORTS-1:0][PORT_W-1:0] dut_out_tdata; + logic [NUM_PORTS-1:0][ 0:0] dut_out_tlast; + logic [NUM_PORTS-1:0][ 0:0] dut_out_tvalid; + logic [NUM_PORTS-1:0][ 0:0] dut_out_tready; + + logic ext_rtcfg_stb = 1'b0; + logic [15:0] ext_rtcfg_addr = 'X; + logic [31:0] ext_rtcfg_data = 'X; + logic ext_rtcfg_ack; + + chdr_crossbar_nxn #( + .PROTOVER (PROTOVER ), + .PORT_W (PORT_W ), + .NPORTS (NUM_PORTS ), + .CHDR_WIDTHS (CHDR_WIDTHS ), + .DEFAULT_PORT (DEFAULT_PORT ), + .ROUTES (ROUTES ), + .BYTE_MTU (BYTE_MTU ), + .ROUTE_TBL_SIZE(ROUTE_TBL_SIZE), + .MUX_ALLOC (MUX_ALLOC ), + .OPTIMIZE (OPTIMIZE ), + .NPORTS_MGMT (NPORTS_MGMT ), + .EXT_RTCFG_PORT(EXT_RTCFG_PORT) + ) chdr_crossbar_nxn_i ( + .clk (clk ), + .reset (rst ), + .device_id (DEVICE_ID ), + .s_axis_tdata (dut_in_tdata ), + .s_axis_tlast (dut_in_tlast ), + .s_axis_tvalid (dut_in_tvalid ), + .s_axis_tready (dut_in_tready ), + .m_axis_tdata (dut_out_tdata ), + .m_axis_tlast (dut_out_tlast ), + .m_axis_tvalid (dut_out_tvalid), + .m_axis_tready (dut_out_tready), + .ext_rtcfg_stb (ext_rtcfg_stb ), + .ext_rtcfg_addr(ext_rtcfg_addr), + .ext_rtcfg_data(ext_rtcfg_data), + .ext_rtcfg_ack (ext_rtcfg_ack ) + ); + + + //--------------------------------------------------------------------------- + // Resize + //--------------------------------------------------------------------------- + // + // Resize the input and output to each port on the crossbar to the same size. + // This allows us to use BFMs of the same type to interface to each port + // without having to worry about the port-width. + // + //--------------------------------------------------------------------------- + + for (genvar port_index = 0; port_index < NUM_PORTS; port_index++) begin : gen_resize + chdr_resize #( + .I_CHDR_W(PORT_W ), + .O_CHDR_W(CHDR_WIDTHS[port_index]), + .PIPELINE("NONE" ) + ) chdr_resize_input ( + .clk (clk ), + .rst (rst ), + .i_chdr_tdata (chdr_to_dut[port_index].tdata ), + .i_chdr_tuser ('0 ), + .i_chdr_tlast (chdr_to_dut[port_index].tlast ), + .i_chdr_tvalid(chdr_to_dut[port_index].tvalid), + .i_chdr_tready(chdr_to_dut[port_index].tready), + .o_chdr_tdata (dut_in_tdata[port_index] ), + .o_chdr_tuser ( ), + .o_chdr_tlast (dut_in_tlast[port_index] ), + .o_chdr_tvalid(dut_in_tvalid[port_index] ), + .o_chdr_tready(dut_in_tready[port_index] ) + ); + + chdr_resize #( + .I_CHDR_W(CHDR_WIDTHS[port_index]), + .O_CHDR_W(PORT_W ), + .PIPELINE("NONE" ) + ) chdr_resize_output ( + .clk (clk ), + .rst (rst ), + .i_chdr_tdata (dut_out_tdata[port_index] ), + .i_chdr_tuser ('0 ), + .i_chdr_tlast (dut_out_tlast[port_index] ), + .i_chdr_tvalid(dut_out_tvalid[port_index] ), + .i_chdr_tready(dut_out_tready[port_index] ), + .o_chdr_tdata (chdr_from_dut[port_index].tdata ), + .o_chdr_tuser ( ), + .o_chdr_tlast (chdr_from_dut[port_index].tlast ), + .o_chdr_tvalid(chdr_from_dut[port_index].tvalid), + .o_chdr_tready(chdr_from_dut[port_index].tready) + ); + end + + + //--------------------------------------------------------------------------- + // Crossbar Configuration + //--------------------------------------------------------------------------- + + task cfg_write(shortint unsigned addr, int unsigned data); + $display("Writing route 0x%X to addr 0x%X", data, addr); + @(posedge clk); + ext_rtcfg_stb <= 1'b1; + ext_rtcfg_addr <= addr; + ext_rtcfg_data <= data; + @(posedge clk); + ext_rtcfg_stb <= 1'b0; + ext_rtcfg_addr <= 'X; + ext_rtcfg_data <= 'X; + @(negedge ext_rtcfg_ack); + endtask : cfg_write + + + task automatic configure_crossbar(); + bit [15:0] epid; + + // Configure an EPID for every possible crossbar route. This will allow us + // to use the EPID to see indicate the intended source port and intended + // destination port. The left 8 bits of EPID will be the source port and + // the right 8 bits will be the destination port. + $display("Initializing crossbar"); + + if (USE_MGMT_PORTS) begin + // Configure using the management ports + $fatal(1, "Configuration of crossbar via management ports is NOT supported yet"); + end else begin + // Configure using the external configuration port + for (int out_port = 0; out_port < NUM_PORTS; out_port++) begin + for (int in_port = 0; in_port < NUM_PORTS; in_port++) begin + // To add an entry to the routing table, put the desired output port + // number in the data field (the width of the data field is + // clog2(NUM_PORTS)) and the corresponding 16-bit EPID in the address + // field. + epid = { in_port[7:0], out_port[7:0] }; + cfg_write(epid, out_port); + end + end + end + + // The routing table can buffer a few write requests, so it takes a little + // bit of extra time for the last write to update the KV map before we can + // start routing packets. + #(1ns * 100*CLK_PERIOD); + endtask : configure_crossbar + + + //--------------------------------------------------------------------------- + // Traffic Consumers + //--------------------------------------------------------------------------- + // + // Each output port has its own BFM. Here we generate a consumer for each + // output port. We use a for-generate statement to avoid having to manage a + // bunch of threads for the consumers, although we certainly could have done + // that. + // + //--------------------------------------------------------------------------- + + for (genvar port_num = 0; port_num < NUM_PORTS; port_num++) begin : gen_consumers + initial begin + int expected_pkts; + forever begin + @(start_consumer); + $display("Consumer started for port %0d", port_num); + mb_num_pkts[port_num].get(expected_pkts); + $display("Consumer %0d: Expecting %0d packets", port_num, expected_pkts); + + repeat(expected_pkts) begin + shortint epid; + int data_length; + int start_val; + + // Get the next packet + ChdrPacket #(PORT_W) pkt; + chdr_bfm[port_num].get_chdr(pkt); + epid = pkt.header.dst_epid; + { data_length, start_val } = pkt.metadata[0]; + if (DEBUG) begin + $display( + "Consumer %0d: Received packet %0d -> %0d, EPID: %X, StartVal: 0x%02X, Length: %0d (%0d)", + port_num, epid[15:8], epid[7:0], epid, start_val, data_length, pkt.data.size() + ); + end + + // Check the EPID + `ASSERT_ERROR( + epid[7:0] == port_num, + $sformatf( + "Consumer %0d: Received EPID %X. Expected EPID ending in **%X.", + port_num, epid, byte'(port_num) + ) + ); + + // Check the payload + begin + ChdrData #(PORT_W, 8)::item_queue_t data_bytes; + byte expected; + data_bytes = ChdrData #(PORT_W, 8)::chdr_to_item(pkt.data, data_length); + foreach (data_bytes[i]) begin + expected = start_val + i; + `ASSERT_ERROR( + data_bytes[i] == expected, + $sformatf( + "Consumer %0d: Byte %0d of packet is incorrect. Expected 0x%X, found 0x%X.", + port_num, i, expected, data_bytes[i] + ) + ); + end + end + + // Check that the payload was the expected size + begin + int exp_num_words; + exp_num_words = $ceil(data_length / (PORT_W/8.0)); + `ASSERT_ERROR( + pkt.data.size() == exp_num_words, + $sformatf( + "Consumer %0d: Received %0d words in payload, expected %0d words", + port_num, pkt.data.size(), exp_num_words + ) + ); + end + + end + + ports_done.put(); + end + end + end : gen_consumers + + + //--------------------------------------------------------------------------- + // Traffic Producer + //--------------------------------------------------------------------------- + // + // Each input port has its own input BFM. This task will generate the + // indicated number of random packets for each input port and enqueue them in + // the associated BFM for each port. The enqueuing is non-blocking, so all + // packets get enqueued at the same time and will be transmitted by the BFMs + // in the order provided. The destination output port is randomly selected. + // All input ports will be receiving in parallel. + // + //--------------------------------------------------------------------------- + + task automatic gen_traffic(int num_packets); + int num_pkts [NUM_PORTS]; + int src_port; + int dst_port; + + // Because the BFM calls are non-blocking, we can enqueue all the packets + // we want to send, and they will all start sending at time zero. + for (src_port = 0; src_port < NUM_PORTS; src_port++) begin + // Skips this input port if it doesn't have any routes + if (!TEST_BAD_ROUTES && !ROUTES[src_port]) continue; + + // Send num_packets packets on each input port with a random output + // port as the destination. + for (int pkt_count = 0; pkt_count < num_packets; pkt_count++) begin + ChdrPacket #(PORT_W) packet = new(); + chdr_header_t hdr = '0; + ChdrData #(PORT_W, 8)::item_queue_t data_bytes; + ChdrData #(PORT_W, 8)::chdr_word_queue_t data_words; + ChdrData #(PORT_W, 8)::chdr_word_queue_t mdata_words = '{ 0 }; + int data_length; + int start_val; + + // Generate a random byte payload + data_length = $urandom_range(1, MAX_PYLD_BYTES); + start_val = $urandom_range(0, 255); + for(int i = 0; i < data_length; i++) begin + data_bytes.push_back(start_val + i); + end + data_words = ChdrData #(PORT_W, 8)::item_to_chdr(data_bytes); + + // Choose a random destination port. We allow paths that are disabled + // and expect these to be ignored. + do begin + dst_port = $urandom_range(0, NUM_PORTS-1); + end while (!TEST_BAD_ROUTES && !ROUTES[src_port][dst_port]); + + + // Generate packet. The EPID holds the expected route and the metadata + // holds the expected start value and length for the data payload. + hdr.pkt_type = CHDR_DATA_NO_TS; + hdr.dst_epid = { src_port[7:0], dst_port[7:0] }; + mdata_words[0] = { data_length, start_val }; + packet.write_raw(hdr, data_words, mdata_words); + + // Enqueue the packet + if (DEBUG) begin + $display( + "Producer %0d: Sending packet %0d -> %0d, EPID: %X, StartVal: 0x%02X, Length: %0d (%0d)", + src_port, src_port, dst_port, packet.header.dst_epid, start_val, + data_length, packet.data.size() + ); + end + chdr_bfm[src_port].put_chdr(packet); + + // Updated the number of expected packets for the selected port. + if (ROUTES[src_port][dst_port]) begin + num_pkts[dst_port]++; + end + end + end + + // Let the consumer know how many packets to expect + for (dst_port = 0; dst_port < NUM_PORTS; dst_port++) begin + mb_num_pkts[dst_port].put(num_pkts[dst_port]); + end + + // Signal that the mailboxes are ready to be read + -> start_consumer; + endtask : gen_traffic + + + //--------------------------------------------------------------------------- + // Test Executor + //--------------------------------------------------------------------------- + + // Tests num_packets on each port using the provided stall rates. + task automatic test_traffic_pattern( + int num_packets, + int input_stall_prob, + int output_stall_prob + ); + // Configure the BFM rates + for (int i = 0; i < NUM_PORTS; i++) begin + chdr_bfm[i].set_master_stall_prob(input_stall_prob); + chdr_bfm[i].set_slave_stall_prob(output_stall_prob); + end + + // Generate the random traffic to transmit + gen_traffic(num_packets); + + // Wait until all ports have finished receiving + ports_done.get(NUM_PORTS); + endtask : test_traffic_pattern + + + //--------------------------------------------------------------------------- + // Main + //--------------------------------------------------------------------------- + + initial begin : tb_main + string tb_name; + tb_name = $sformatf( + "chdr_crossbar_nxn\nNUM_PORTS = %0D\nCHDR_WIDTHS = %p\nROUTES = %b\nNUM_PKTS = %0D", + NUM_PORTS, CHDR_WIDTHS, ROUTES, NUM_PKTS + ); + + test.start_tb(tb_name); + + // Initialize mailboxes + for (int port = 0; port < NUM_PORTS; port++) begin + mb_num_pkts[port] = new(); + end + + // Start the BFMs. Wait a delta cycle to ensure that the BFMs get created + // before we use them. + #0ns; + for (int port = 0; port < NUM_PORTS; port++) begin + chdr_bfm[port].run(); + end + + // Start the clocks + clk_gen.start(); + + // Reset + clk_gen.reset(); + @(negedge rst); + + //------------------------------------------------------------------------- + // Initialize the Crossbar Routing + //------------------------------------------------------------------------- + + configure_crossbar(); + + //------------------------------------------------------------------------- + // Run Tests + //------------------------------------------------------------------------- + + test.start_test("Full rate", 10ms); + test_traffic_pattern(NUM_PKTS, 0, 0); + test.end_test(); + + test.start_test("Three-quarter rate", 10ms); + test_traffic_pattern(NUM_PKTS, 25, 25); + test.end_test(); + + test.start_test("Back pressure", 10ms); + test_traffic_pattern(NUM_PKTS, 25, 50); + test.end_test(); + + test.start_test("Underflow", 10ms); + test_traffic_pattern(NUM_PKTS, 50, 25); + test.end_test(); + + //------------------------------------------------------------------------- + // Clean Up + //------------------------------------------------------------------------- + + // End the TB, but don't $finish, since we don't want to kill other + // instances of this testbench that may be running. + test.end_tb(0); + + // Kill the clocks to end this instance of the testbench + clk_gen.kill(); + end : tb_main + +endmodule : chdr_crossbar_nxn_tb + + +`default_nettype wire diff --git a/lib/rfnoc/crossbar/crossbar_tb/crossbar_tb.sv b/lib/rfnoc/crossbar/crossbar_tb/crossbar_tb.sv index 33b09df..4b48800 100644 --- a/lib/rfnoc/crossbar/crossbar_tb/crossbar_tb.sv +++ b/lib/rfnoc/crossbar/crossbar_tb/crossbar_tb.sv @@ -207,10 +207,10 @@ module crossbar_tb #( ); end else if (ROUTER_IMPL == "chdr_crossbar_nxn") begin chdr_crossbar_nxn #( - .CHDR_W (ROUTER_DWIDTH), + .PORT_W (ROUTER_DWIDTH), .NPORTS (ROUTER_PORTS), .DEFAULT_PORT (0), - .MTU (MTU_LOG2), + .BYTE_MTU (MTU_LOG2 + $clog2(ROUTER_DWIDTH/8)), .ROUTE_TBL_SIZE (6), .MUX_ALLOC ("ROUND-ROBIN"), .OPTIMIZE ("AREA"), diff --git a/lib/rfnoc/sim/chdr_stream_endpoint_tb/chdr_stream_endpoint_tb.sv b/lib/rfnoc/sim/chdr_stream_endpoint_tb/chdr_stream_endpoint_tb.sv index f9dde3c..95ea7ea 100644 --- a/lib/rfnoc/sim/chdr_stream_endpoint_tb/chdr_stream_endpoint_tb.sv +++ b/lib/rfnoc/sim/chdr_stream_endpoint_tb/chdr_stream_endpoint_tb.sv @@ -187,10 +187,10 @@ module chdr_stream_endpoint_tb#( ); chdr_crossbar_nxn #( - .CHDR_W (CHDR_W), + .PORT_W (CHDR_W), .NPORTS (3), .DEFAULT_PORT (0), - .MTU (MTU), + .BYTE_MTU (MTU + $clog2(CHDR_W/8)), .ROUTE_TBL_SIZE (6), .MUX_ALLOC ("ROUND-ROBIN"), .OPTIMIZE ("AREA"),