diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/Makefile b/lib/rfnoc/blocks/rfnoc_block_fft/Makefile index 14869d7..9c07521 100644 --- a/lib/rfnoc/blocks/rfnoc_block_fft/Makefile +++ b/lib/rfnoc/blocks/rfnoc_block_fft/Makefile @@ -1,5 +1,5 @@ # -# Copyright 2019 Ettus Research, a National Instruments Brand +# Copyright 2024 Ettus Research, a National Instruments Brand # # SPDX-License-Identifier: LGPL-3.0-or-later # @@ -21,17 +21,32 @@ LIB_IP_DIR = $(BASE_DIR)/../lib/ip # Include makefiles and sources for all IP components # *after* defining the LIB_IP_DIR -include $(LIB_IP_DIR)/axi_fft/Makefile.inc -include $(LIB_IP_DIR)/complex_to_magphase/Makefile.inc +include $(LIB_IP_DIR)/xfft_64k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_32k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_16k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_8k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_4k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_2k_16b/Makefile.inc +include $(LIB_IP_DIR)/xfft_1k_16b/Makefile.inc +include $(LIB_IP_DIR)/complex_to_magphase_int17/Makefile.inc DESIGN_SRCS += $(abspath \ -$(LIB_IP_AXI_FFT_OUTS) \ +$(LIB_IP_XFFT_64K_16B_OUTS) \ +$(LIB_IP_XFFT_32K_16B_OUTS) \ +$(LIB_IP_XFFT_16K_16B_OUTS) \ +$(LIB_IP_XFFT_8K_16B_OUTS) \ +$(LIB_IP_XFFT_4K_16B_OUTS) \ +$(LIB_IP_XFFT_3K_16B_OUTS) \ +$(LIB_IP_XFFT_2K_16B_OUTS) \ +$(LIB_IP_XFFT_1K_16B_OUTS) \ +$(LIB_IP_COMPLEX_TO_MAGPHASE_INT17_OUTS) \ ) #------------------------------------------------- # Design Specific #------------------------------------------------- -# Include makefiles and sources for the DUT and its dependencies +# Include makefiles and sources for the DUT and its +# dependencies. include $(BASE_DIR)/../lib/rfnoc/core/Makefile.srcs include $(BASE_DIR)/../lib/rfnoc/utils/Makefile.srcs include Makefile.srcs @@ -45,10 +60,12 @@ $(RFNOC_OOT_SRCS) \ #------------------------------------------------- # Testbench Specific #------------------------------------------------- -SIM_TOP = rfnoc_block_fft_tb glbl -SIM_SRCS = \ -$(abspath rfnoc_block_fft_tb.sv) \ +SIM_TOP = rfnoc_block_fft_all_tb glbl +SIM_SRCS = $(abspath \ +rfnoc_block_fft_tb.sv \ +rfnoc_block_fft_all_tb.sv \ $(VIVADO_PATH)/data/verilog/src/glbl.v \ +) #------------------------------------------------- # Bottom-of-Makefile diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/Makefile.srcs b/lib/rfnoc/blocks/rfnoc_block_fft/Makefile.srcs index b2d8234..53bdea8 100644 --- a/lib/rfnoc/blocks/rfnoc_block_fft/Makefile.srcs +++ b/lib/rfnoc/blocks/rfnoc_block_fft/Makefile.srcs @@ -1,10 +1,19 @@ # -# Copyright 2019 Ettus Research, a National Instruments Brand +# Copyright 2024 Ettus Research, a National Instruments Brand # # SPDX-License-Identifier: LGPL-3.0-or-later # RFNOC_OOT_SRCS += $(abspath $(addprefix $(BASE_DIR)/../lib/rfnoc/blocks/rfnoc_block_fft/, \ +fft_reorder_pkg.sv \ +fft_reorder.sv \ +fft_post_processing.sv \ +cp_removal.sv \ +axis_cp_list.sv \ noc_shell_fft.v \ -rfnoc_block_fft.v \ +xfft_config_pkg.sv \ +fft_core_regs_pkg.sv \ +xfft_wrapper.sv \ +fft_core.sv \ +rfnoc_block_fft.sv \ )) diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/axis_cp_list.sv b/lib/rfnoc/blocks/rfnoc_block_fft/axis_cp_list.sv new file mode 100644 index 0000000..0f584aa --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/axis_cp_list.sv @@ -0,0 +1,167 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: axis_cp_list +// +// Description: +// +// This module maintains a list that can be used to implement storing the +// cyclic prefix list. Items can be loaded into the list by writing them to +// the input port. As soon as items are in the list, they can be read out on +// the output port. Items in the list are read out in the order that they +// were added and the list automatically repeats in a circular manner. Once +// the list is full, it stops accepting inputs. +// +// The list can be cleared by asserting clear for one clock cycle, after +// which you can load and read out a new list. +// +// There's no arbitration between reading and writing. It's assumed the list +// is loaded in one step, then read out and used in another. You can do both +// at the same time, but there's no guarantee about the order in which things +// complete. +// +// The output is fairly slow at one output every three clock cycles. +// +// Parameters: +// +// ADDR_W : Sets the maximum length of the list, which will be 2**ADDR_W. +// DATA_W : Width of the cyclic-prefix length. +// REPEAT : 1: Cyclic prefix list repeats. 0: Cyclic prefix list +// does not repeat, and the default prefix length will +// be used once the list is completed. +// DEFAULT : The value that is output when there is no valid data to output. +// + +`default_nettype none + + +module axis_cp_list #( + int ADDR_W = 5, + int DATA_W = 32, + bit REPEAT = 1, + logic [DATA_W-1:0] DEFAULT = '0 +) ( + input wire clk, + input wire rst, + + input wire clear, + + input wire [DATA_W-1:0] i_tdata, + input wire i_tvalid, + output wire i_tready, + + output reg [DATA_W-1:0] o_tdata, + output reg o_tvalid, + input wire o_tready, + + output wire [ADDR_W:0] occupied +); + + // Make addresses one extra bit wide to double as fullness and to detect the + // full condition. + logic [ADDR_W:0] wr_addr; + logic [ADDR_W:0] rd_addr; + logic full; + + + //--------------------------------------------------------------------------- + // RAM + //--------------------------------------------------------------------------- + + logic [DATA_W-1:0] rd_data; + logic wr_en; + + assign wr_en = i_tvalid && i_tready; + + ram_2port #( + .DWIDTH (DATA_W), + .AWIDTH (ADDR_W), + .OUT_REG(0 ) + ) ram_2port_i ( + .clka (clk ), + .ena ('1 ), + .wea (wr_en ), + .addra(wr_addr[0+:ADDR_W]), + .dia (i_tdata ), + .doa ( ), + .clkb (clk ), + .enb ('1 ), + .web ('0 ), + .addrb(rd_addr[0+:ADDR_W]), + .dib ('0 ), + .dob (rd_data ) + ); + + + //--------------------------------------------------------------------------- + // Write Logic + //--------------------------------------------------------------------------- + + assign occupied = wr_addr; + assign full = wr_addr[ADDR_W]; + assign i_tready = !full; + + always_ff @(posedge clk) begin + if (i_tvalid && i_tready) begin + wr_addr <= wr_addr + 1; + end + + if (rst || clear) begin + wr_addr <= '0; + end + end + + + //--------------------------------------------------------------------------- + // Read Logic + //--------------------------------------------------------------------------- + + enum logic [1:0] { ST_IDLE, ST_LOAD, ST_OUTPUT } rd_state; + + always_ff @(posedge clk) begin + case (rd_state) + ST_IDLE : begin + // The read is started during this cycle, since rd_addr is valid + if (rd_addr < occupied) begin + rd_state <= ST_LOAD; + end + end + ST_LOAD : begin + // The read is available during this cycle + o_tvalid <= 1'b1; + o_tdata <= rd_data; + rd_state <= ST_OUTPUT; + end + ST_OUTPUT : begin + // Wait for the output to be captured during this cycle + if (o_tready) begin + // Always output default value when there's no data to output. This + // is important for cyclic-prefix insertion/removal where we want the + // default to be 0. + o_tdata <= DEFAULT; + o_tvalid <= '0; + rd_state <= ST_IDLE; + + if (REPEAT && rd_addr == occupied-1) begin + rd_addr <= '0; + end else begin + rd_addr <= rd_addr + 1; + end + end + end + endcase + + if (rst | clear) begin + rd_state <= ST_IDLE; + rd_addr <= '0; + o_tdata <= '0; + o_tvalid <= '0; + end + end + +endmodule : axis_cp_list + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/cp_removal.sv b/lib/rfnoc/blocks/rfnoc_block_fft/cp_removal.sv new file mode 100644 index 0000000..c66d84b --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/cp_removal.sv @@ -0,0 +1,235 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: cp_removal +// +// Description: +// +// Removes the cyclic prefix from OFDM symbols. A configuration list allows +// for queuing up multiple cyclic prefix lengths, and has an optional repeat +// mode that causes the same list of cyclic prefixes to be reused as new +// symbols arrive. This allows the block to execute a pattern for cases when +// CP lengths change symbol to symbol in a repeating pattern. +// +// There is a two-clock bubble cycle after every symbol due to returning to +// the idle state to load the next config, so this block must be clocked at +// least slightly faster than the sample rate. That is: +// +// Clock rate > Fs * (1 + 2/(CP length + FFT Size)) +// +// Parameters: +// +// DATA_W : Data/sample AXI-Stream bus width +// USER_W : Width of TUSER on the data/sample AXI-Stream bus +// SYM_LEN_W : Width of the maximum symbol length. The maximum +// supported symbol length is 2**SYM_LEN_W - 1. +// CP_LEN_W : Width of the maximum cyclic prefix length. The maximum +// supported CP length is 2**CP_LEN_W - 1. +// DEFAULT_CP_LEN : Default cyclic prefix length to output +// CP_REPEAT : 1: Cyclic prefix list repeats. 0: Cyclic prefix list +// does not repeat, and the last used prefix length will +// be used once the list is completed. +// MAX_LIST_LOG2 : Log base 2 of the size of the prefix length list +// SET_TLAST : 1: Always set tlast at the end of each symbol. 0: Pass +// through input tlast unchanged. +// +// Signals: +// +// clear_list : Clear the CP removal list +// symbol_len : Symbol/FFT size to use for generating TLAST +// cp_len_t* : AXI-Stream cyclic prefix length list input. Use this to +// write prefix lengths to the list in order. +// cp_list_occupied : Number of items in the cyclic prefix list +// i_t* : AXI-Stream data input on which to do cyclic prefix removal +// o_t* : AXI-Stream data output with cyclic prefix removed +// + +`default_nettype none + + +module cp_removal #( + parameter int DATA_W = 32, + parameter int USER_W = 1, + parameter int CP_LEN_W = 16, + parameter int SYM_LEN_W = 17, + parameter int DEFAULT_CP_LEN = 0, + parameter bit CP_REPEAT = 0, + parameter int MAX_LIST_LOG2 = 5, + parameter bit SET_TLAST = 1 +) ( + input wire clk, + input wire rst, + input wire clear_list, + + // Cyclic prefix length input port + input wire [SYM_LEN_W-1:0] symbol_len, + input wire [ CP_LEN_W-1:0] cp_len_tdata, + input wire cp_len_tvalid, + output wire cp_len_tready, + output wire [ 15:0] cp_list_occupied, + + // Symbol data stream input + input wire [ DATA_W-1:0] i_tdata, + input wire [ USER_W-1:0] i_tuser, + input wire i_tlast, + input wire i_tvalid, + output wire i_tready, + + // Symbol data stream output + output wire [ DATA_W-1:0] o_tdata, + output wire [ USER_W-1:0] o_tuser, + output wire o_tlast, + output wire o_tvalid, + input wire o_tready +); + `include "usrp_utils.svh" + + enum logic [2:0] { S_IDLE, S_CONFIG, S_PREFIX, S_SYMBOL, S_CLEAR } state; + + logic [CP_LEN_W-1:0] fifo_in_tdata, fifo_out_tdata; + logic fifo_in_tvalid, fifo_out_tvalid; + logic fifo_in_tready, fifo_out_tready; + logic fifo_clear; + + assign fifo_clear = (state == S_CLEAR); + + axi_fifo #( + .WIDTH(CP_LEN_W), + .SIZE (MAX_LIST_LOG2) + ) axi_fifo_config_inst ( + .clk (clk), + .reset (rst), + .clear (fifo_clear), + .i_tdata (fifo_in_tdata), + .i_tvalid(fifo_in_tvalid), + .i_tready(fifo_in_tready), + .o_tdata (fifo_out_tdata), + .o_tvalid(fifo_out_tvalid), + .o_tready(fifo_out_tready), + .space (), + .occupied(cp_list_occupied) + ); + + generate + if (CP_REPEAT == 0) begin + // No config list loopback. New configs can be written at any time. + assign fifo_in_tdata = cp_len_tdata; + assign fifo_in_tvalid = (state == S_CLEAR) ? 1'b0 : cp_len_tvalid; + assign cp_len_tready = (state == S_CLEAR) ? 1'b0 : fifo_in_tready; + assign fifo_out_tready = (state == S_CONFIG); + end else begin + // Config list loopback enabled. Write current config back into config + // FIFO in the S_CONFIG state. New configs can be written in any state + // but S_CONFIG & S_CLEAR. + assign fifo_in_tdata = (state == S_CONFIG) ? fifo_out_tdata : + cp_len_tdata; + assign fifo_in_tvalid = (state == S_CONFIG) ? fifo_out_tvalid : + (state == S_CLEAR) ? 1'b0 : + cp_len_tvalid; + assign cp_len_tready = (state == S_CONFIG) ? 1'b0 : + (state == S_CLEAR) ? 1'b0 : + fifo_in_tready; + assign fifo_out_tready = (state == S_CONFIG); + end + endgenerate + + localparam COUNT_W = `MAX(SYM_LEN_W, CP_LEN_W); + + logic [ CP_LEN_W-1:0] cp_len_reg = DEFAULT_CP_LEN; + logic [SYM_LEN_W-1:0] symbol_len_reg = '0; + logic [ COUNT_W-1:0] count = '0; + logic clear_fifo_hold = 1'b0; + + always @(posedge clk) begin + // Latch FIFO clear + if (clear_list) begin + clear_fifo_hold <= 1'b1; + end + + // State machine + case (state) + // Wait in idle state until either a configuration list clear is + // requested or we get a new data input. + S_IDLE : begin + count <= 1; + if (clear_fifo_hold) begin + state <= S_CLEAR; + end else if (i_tvalid) begin + // Only update the CP length being used if there's a valid one in the + // list. Otherwise, keep using the previous value. + if (fifo_out_tvalid) begin + cp_len_reg <= fifo_out_tdata; + end + symbol_len_reg <= symbol_len; + state <= S_CONFIG; + end + end + S_CONFIG : begin + if (cp_len_reg > 0) begin + state <= S_PREFIX; + end else if (symbol_len_reg > 0) begin + state <= S_SYMBOL; + end else begin + state <= S_IDLE; + end + end + S_PREFIX : begin + if (i_tvalid & i_tready) begin + count <= count + 1; + if (count >= cp_len_reg) begin + count <= 1; + if (symbol_len_reg > 0) begin + state <= S_SYMBOL; + end else begin + state <= S_IDLE; + end + end + end + end + S_SYMBOL : begin + if (i_tvalid & i_tready) begin + count <= count + 1; + if (count >= symbol_len_reg) begin + count <= 1; + state <= S_IDLE; + end + end + end + S_CLEAR : begin + clear_fifo_hold <= 1'b0; + cp_len_reg <= DEFAULT_CP_LEN; + state <= S_IDLE; + end + default : state <= S_IDLE; + endcase + + if (rst) begin + clear_fifo_hold <= 1'b0; + cp_len_reg <= DEFAULT_CP_LEN; + count <= 1; + state <= S_IDLE; + end + end + + logic new_tlast; + assign new_tlast = (state == S_SYMBOL) & (count >= symbol_len_reg); + + assign o_tdata = i_tdata; + assign o_tuser = i_tuser; + assign o_tlast = (SET_TLAST == 0) ? i_tlast : new_tlast; + assign o_tvalid = (state == S_IDLE) ? 1'b0 : + (state == S_PREFIX) ? 1'b0 : + (state == S_SYMBOL) ? i_tvalid : + (state == S_CLEAR) ? 1'b0 : + 1'b0; + assign i_tready = (state == S_IDLE) ? 1'b0 : + (state == S_PREFIX) ? 1'b1 : + (state == S_SYMBOL) ? o_tready : + (state == S_CLEAR) ? 1'b0 : + 1'b0; + +endmodule : cp_removal + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_core.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_core.sv new file mode 100644 index 0000000..55ef9cf --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_core.sv @@ -0,0 +1,846 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_core +// +// Description: +// +// This module encapsulates the core components that make up a single or +// multi-channel FFT with cyclic prefix insertion/removal. +// +// All channels going into this core must be used simultaneously and share +// the same register settings. In other words, you can't use one channel and +// leave the other idle. The used channel will stall while waiting for the +// other channel's data to arrive. This is because all channels share the +// same cyclic prefix removal logic. You also cannot have different settings +// per channel within a single fft_core. However, multiple instances of this +// fft_core can be instantiated to allow for independent channels. +// +// The maximum cyclic prefix insertion/removal length is always the maximum +// FFT size minus 1. The length of the list used to track a sequence of CP +// insertions/removals is set by parameters. +// +// Parameters: +// +// NUM_CHAN : Number of channels to instantiate on this +// fft_core instance. +// NUM_CORES : Total number of fft_core instances in the +// parent RFNoC block, including this one. +// MAX_FFT_SIZE_LOG2 : Log2 of maximum configurable FFT size. Actual +// max is 2**MAX_FFT_SIZE_LOG2. +// MAX_CP_LIST_LEN_INS_LOG2 : Log2 of max length of cyclic prefix insertion +// list. Actual max is 2**MAX_CP_LIST_LEN_INS_LOG2. +// MAX_CP_LIST_LEN_REM_LOG2 : Log2 of max length of cyclic prefix removal +// list. Actual max is 2**MAX_CP_LIST_LEN_REM_LOG2. +// CP_INSERTION_REPEAT : Enable repeating the CP insertion list. When 1, +// the list repeats. When 0, CP insertion will +// stop when the list is finished. +// CP_REMOVAL_REPEAT : Enable repeating the CP removal list. When 1, +// the list repeats. When 0, CP removal will +// stop when the list is finished. +// EN_FFT_BYPASS : Controls whether to include the FFT bypass logic. +// EN_FFT_ORDER : Controls whether to include the FFT reorder logic. +// EN_MAGNITUDE : Controls whether to include the magnitude +// output calculation logic. +// EN_MAGNITUDE_SQ : Controls whether to include the +// magnitude-squared output calculation logic. +// USE_APPROX_MAG : Controls whether to use the low-resource +// approximate calculation (1) or the more exact +// and more resource-intensive calculation (0) for +// the magnitude calculation. +// + +`default_nettype none + + +module fft_core + import rfnoc_chdr_utils_pkg::*; + import ctrlport_pkg::*; +#( + int NUM_CHAN = 1, + int NUM_CORES = 1, + int MAX_FFT_SIZE_LOG2 = 12, + int MAX_CP_LIST_LEN_INS_LOG2 = 5, + int MAX_CP_LIST_LEN_REM_LOG2 = 5, + bit CP_INSERTION_REPEAT = 1, + bit CP_REMOVAL_REPEAT = 1, + bit EN_FFT_BYPASS = 1, + bit EN_FFT_ORDER = 1, + bit EN_MAGNITUDE = 1, + bit EN_MAGNITUDE_SQ = 1, + bit USE_APPROX_MAG = 1, + + // Data width of each FFT channel + localparam int ITEM_W = 32, + localparam int DATA_W = ITEM_W, + localparam int KEEP_W = 1 +) ( + input wire ce_clk, + input wire ce_rst, + + // CtrlPort Register Interface + input wire s_ctrlport_req_wr, + input wire s_ctrlport_req_rd, + input wire [ CTRLPORT_ADDR_W-1:0] s_ctrlport_req_addr, + input wire [ CTRLPORT_DATA_W-1:0] s_ctrlport_req_data, + output logic s_ctrlport_resp_ack, + output logic [ CTRLPORT_DATA_W-1:0] s_ctrlport_resp_data, + + // Data Input Packets + input wire [ DATA_W*NUM_CHAN-1:0] s_in_axis_tdata, + input wire [ KEEP_W*NUM_CHAN-1:0] s_in_axis_tkeep, + input wire [ NUM_CHAN-1:0] s_in_axis_tlast, + input wire [ NUM_CHAN-1:0] s_in_axis_tvalid, + output logic [ NUM_CHAN-1:0] s_in_axis_tready, + input wire [CHDR_TIMESTAMP_W*NUM_CHAN-1:0] s_in_axis_ttimestamp, + input wire [ NUM_CHAN-1:0] s_in_axis_thas_time, + input wire [ CHDR_LENGTH_W*NUM_CHAN-1:0] s_in_axis_tlength, + input wire [ NUM_CHAN-1:0] s_in_axis_teov, + input wire [ NUM_CHAN-1:0] s_in_axis_teob, + + // Data Output Packets + output wire [ DATA_W*NUM_CHAN-1:0] m_out_axis_tdata, + output wire [ KEEP_W*NUM_CHAN-1:0] m_out_axis_tkeep, + output wire [ NUM_CHAN-1:0] m_out_axis_tlast, + output wire [ NUM_CHAN-1:0] m_out_axis_tvalid, + input wire [ NUM_CHAN-1:0] m_out_axis_tready, + output wire [CHDR_TIMESTAMP_W*NUM_CHAN-1:0] m_out_axis_ttimestamp, + output wire [ NUM_CHAN-1:0] m_out_axis_thas_time, + output wire [ CHDR_LENGTH_W*NUM_CHAN-1:0] m_out_axis_tlength, + output wire [ NUM_CHAN-1:0] m_out_axis_teov, + output wire [ NUM_CHAN-1:0] m_out_axis_teob +); + // Import utilities for working with Xilinx FFT block + import xfft_config_pkg::*; + + // Import register descriptions + import fft_core_regs_pkg::*; + + `include "usrp_utils.svh" + + + //--------------------------------------------------------------------------- + // FFT Configuration Interface Constants + //--------------------------------------------------------------------------- + + localparam int MAX_FFT_SIZE = 2**MAX_FFT_SIZE_LOG2; + localparam int MAX_CP_LEN_LOG2 = MAX_FFT_SIZE_LOG2; + + // Calculate the widths needed for the configuration settings of the Xilinx + // FFT core. The widths change depending on the configuration. + localparam int FFT_SCALE_W = fft_scale_w(MAX_FFT_SIZE_LOG2); + localparam int FFT_FWD_INV_W = fft_fwd_inv_w(MAX_FFT_SIZE_LOG2); + localparam int FFT_CP_LEN_W = fft_cp_len_w(MAX_FFT_SIZE_LOG2); + localparam int FFT_NFFT_W = fft_nfft_w(MAX_FFT_SIZE_LOG2); + localparam int FFT_CONFIG_W = fft_config_w(MAX_FFT_SIZE_LOG2); + + // Use conservative 1/N scaling by default + localparam [FFT_SCALE_W-1:0] DEFAULT_FFT_SCALING = fft_scale_default(MAX_FFT_SIZE_LOG2); + + + //--------------------------------------------------------------------------- + // Registers + //--------------------------------------------------------------------------- + + localparam FFT_SIZE_LOG2_W = $clog2(MAX_FFT_SIZE_LOG2+1); + localparam FFT_SIZE_W = MAX_FFT_SIZE_LOG2+1; + localparam CP_LEN_W = MAX_CP_LEN_LOG2; + + localparam int REG_LENGTH_LOG2_WIDTH = FFT_NFFT_W; + localparam int REG_SCALING_WIDTH = FFT_SCALE_W; + localparam int REG_CP_INS_LEN_WIDTH = CP_LEN_W; + localparam int REG_CP_REM_LEN_WIDTH = CP_LEN_W; + + localparam bit [REG_COMPAT_WIDTH-1:0] COMPAT = {16'h03, 16'h00}; + localparam bit [REG_CAPABILITIES_WIDTH-1:0] CAPABILITIES = { + 8'(MAX_CP_LIST_LEN_INS_LOG2), + 8'(MAX_CP_LIST_LEN_REM_LOG2), + 8'(MAX_CP_LEN_LOG2), + 8'(MAX_FFT_SIZE_LOG2) + }; + localparam bit [REG_CAPABILITIES2_WIDTH-1:0] CAPABILITIES2 = { + 1'(EN_MAGNITUDE_SQ), + 1'(EN_MAGNITUDE), + 1'(EN_FFT_ORDER), + 1'(EN_FFT_BYPASS) + }; + localparam bit [REG_PORT_CONFIG_WIDTH-1:0] PORT_CONFIG = { + REG_NUM_CORES_W'(NUM_CORES), + REG_NUM_CHAN_W'(NUM_CHAN) + }; + + localparam int DEFAULT_FFT_SIZE_LOG2 = MAX_FFT_SIZE_LOG2; + localparam int DEFAULT_FFT_DIRECTION = FFT_INVERSE; + localparam int DEFAULT_CP_LEN = 0; + + logic core_rst; + + logic [ REG_RESET_WIDTH-1:0] reg_user_reset = '0; + logic [ REG_LENGTH_LOG2_WIDTH-1:0] reg_fft_size_log2 = DEFAULT_FFT_SIZE_LOG2; + logic [ REG_SCALING_WIDTH-1:0] reg_fft_scaling = DEFAULT_FFT_SCALING; + logic [ REG_DIRECTION_WIDTH-1:0] reg_fft_direction = DEFAULT_FFT_DIRECTION; + logic [ REG_CP_INS_LEN_WIDTH-1:0] reg_cp_ins_length = DEFAULT_CP_LEN; + logic [ REG_CP_REM_LEN_WIDTH-1:0] reg_cp_rem_length = DEFAULT_CP_LEN; + logic [REG_CP_INS_LIST_LOAD_WIDTH-1:0] reg_cp_ins_list_load = '0; + logic [ REG_CP_INS_LIST_CLR_WIDTH-1:0] reg_cp_ins_list_clr = '0; + logic [REG_CP_REM_LIST_LOAD_WIDTH-1:0] reg_cp_rem_list_load = '0; + logic [ REG_CP_REM_LIST_CLR_WIDTH-1:0] reg_cp_rem_list_clr = '0; + logic [ REG_BYPASS_WIDTH-1:0] reg_fft_bypass = '0; + logic [ REG_ORDER_WIDTH-1:0] reg_fft_order = FFT_ORDER_NORMAL; + logic [ REG_MAGNITUDE_WIDTH-1:0] reg_magnitude = '0; + logic [ NUM_CHAN-1:0] reg_overflow = '0; + logic [ REG_CP_INS_LIST_OCC_WIDTH-1:0] reg_cp_ins_list_occupied; + logic [ REG_CP_REM_LIST_OCC_WIDTH-1:0] reg_cp_rem_list_occupied; + + logic [FFT_SIZE_LOG2_W-1:0] fft_size_log2; + assign fft_size_log2 = FFT_SIZE_LOG2_W'(reg_fft_size_log2); + + logic [FFT_SIZE_W-1:0] fft_size; + assign fft_size = 1 << reg_fft_size_log2[FFT_SIZE_LOG2_W-1:0]; + + logic [REG_ADDR_W-1:0] s_ctrlport_req_addr_aligned; + assign s_ctrlport_req_addr_aligned = REG_ADDR_W'({s_ctrlport_req_addr[19:2], 2'b0}); + + logic [NUM_CHAN-1:0] event_fft_overflow; + + always_ff @(posedge ce_clk) begin + // Default assignment + s_ctrlport_resp_ack <= 0; + + // Always clear these regs after being set + reg_user_reset <= 1'b0; + reg_cp_ins_list_load <= 1'b0; + reg_cp_ins_list_clr <= 1'b0; + reg_cp_rem_list_load <= 1'b0; + reg_cp_rem_list_clr <= 1'b0; + + // Read user registers + if (s_ctrlport_req_rd) begin // Read request + s_ctrlport_resp_ack <= 1; // Always immediately ack + s_ctrlport_resp_data <= 0; // Zero out by default + case (s_ctrlport_req_addr_aligned) + REG_COMPAT_ADDR: s_ctrlport_resp_data <= 32'(COMPAT); + REG_PORT_CONFIG_ADDR: s_ctrlport_resp_data <= 32'(PORT_CONFIG); + REG_CAPABILITIES_ADDR: s_ctrlport_resp_data <= 32'(CAPABILITIES); + REG_CAPABILITIES2_ADDR: s_ctrlport_resp_data <= 32'(CAPABILITIES2); + REG_OVERFLOW_ADDR: s_ctrlport_resp_data <= 32'(reg_overflow); + REG_LENGTH_LOG2_ADDR: s_ctrlport_resp_data <= 32'(reg_fft_size_log2); + REG_SCALING_ADDR: s_ctrlport_resp_data <= 32'(reg_fft_scaling); + REG_DIRECTION_ADDR: s_ctrlport_resp_data <= 32'(reg_fft_direction); + REG_CP_INS_LEN_ADDR: s_ctrlport_resp_data <= 32'(reg_cp_ins_length); + REG_CP_REM_LEN_ADDR: s_ctrlport_resp_data <= 32'(reg_cp_rem_length); + REG_CP_INS_LIST_OCC_ADDR: s_ctrlport_resp_data <= 32'(reg_cp_ins_list_occupied); + REG_CP_REM_LIST_OCC_ADDR: s_ctrlport_resp_data <= 32'(reg_cp_rem_list_occupied); + REG_BYPASS_ADDR: s_ctrlport_resp_data <= EN_FFT_BYPASS ? 32'(reg_fft_bypass) : '0; + REG_ORDER_ADDR: s_ctrlport_resp_data <= EN_FFT_ORDER ? 32'(reg_fft_order) : '0; + REG_MAGNITUDE_ADDR: s_ctrlport_resp_data <= (EN_MAGNITUDE || EN_MAGNITUDE_SQ) ? + 32'(reg_magnitude) : 0; + default: s_ctrlport_resp_data <= 32'h0BAD_C0DE; + endcase + end + + // Write user registers + if (s_ctrlport_req_wr) begin // Write request + s_ctrlport_resp_ack <= 1; // Always immediately ack + case (s_ctrlport_req_addr_aligned) + REG_RESET_ADDR: + reg_user_reset <= 1'b1; // Strobe + REG_LENGTH_LOG2_ADDR: + reg_fft_size_log2 <= s_ctrlport_req_data[REG_LENGTH_LOG2_WIDTH-1:0]; + REG_SCALING_ADDR: + reg_fft_scaling <= s_ctrlport_req_data[REG_SCALING_WIDTH-1:0]; + REG_DIRECTION_ADDR: + reg_fft_direction <= s_ctrlport_req_data[REG_DIRECTION_WIDTH-1:0]; + REG_CP_INS_LEN_ADDR: + reg_cp_ins_length <= s_ctrlport_req_data[REG_CP_INS_LEN_WIDTH-1:0]; + REG_CP_REM_LEN_ADDR: + reg_cp_rem_length <= s_ctrlport_req_data[REG_CP_REM_LEN_WIDTH-1:0]; + REG_CP_INS_LIST_LOAD_ADDR: + reg_cp_ins_list_load <= 1'b1; // Strobe + REG_CP_INS_LIST_CLR_ADDR: + reg_cp_ins_list_clr <= 1'b1; // Strobe + REG_CP_REM_LIST_LOAD_ADDR: + reg_cp_rem_list_load <= 1'b1; // Strobe + REG_CP_REM_LIST_CLR_ADDR: + reg_cp_rem_list_clr <= 1'b1; // Strobe + REG_BYPASS_ADDR: + reg_fft_bypass <= s_ctrlport_req_data[REG_BYPASS_WIDTH-1:0]; + REG_ORDER_ADDR: + reg_fft_order <= s_ctrlport_req_data[REG_ORDER_WIDTH-1:0]; + REG_MAGNITUDE_ADDR: + reg_magnitude <= s_ctrlport_req_data[REG_MAGNITUDE_WIDTH-1:0]; + endcase + end + + // Store whether or not we had overflow events on this clock cycle. These + // bits are sticky and clear on read. + if (s_ctrlport_req_rd && s_ctrlport_req_addr_aligned == REG_OVERFLOW_ADDR) begin + reg_overflow <= event_fft_overflow; + end else begin + reg_overflow <= reg_overflow | event_fft_overflow; + end + + if (core_rst) begin + s_ctrlport_resp_ack <= '0; + reg_user_reset <= '0; + reg_fft_size_log2 <= DEFAULT_FFT_SIZE_LOG2; + reg_fft_scaling <= DEFAULT_FFT_SCALING; + reg_fft_direction <= DEFAULT_FFT_DIRECTION; + reg_cp_ins_length <= DEFAULT_CP_LEN; + reg_cp_rem_length <= DEFAULT_CP_LEN; + reg_cp_ins_list_load <= '0; + reg_cp_ins_list_clr <= '0; + reg_cp_rem_list_load <= '0; + reg_cp_rem_list_clr <= '0; + reg_fft_bypass <= '0; + reg_fft_order <= FFT_ORDER_NORMAL; + reg_magnitude <= '0; + reg_overflow <= '0; + end + end + + + //--------------------------------------------------------------------------- + // Reset Logic + //--------------------------------------------------------------------------- + + localparam RESET_PULSE_LEN = 8; + localparam RESET_CNT_WIDTH = $clog2(RESET_PULSE_LEN); + + enum logic [0:0] { S_RESET_IDLE, S_RESET_ASSERT } reset_state = S_RESET_ASSERT; + + reg [RESET_CNT_WIDTH:0] user_reset_cnt = 'd0; + reg user_reset = 1'b1; + + always_ff @(posedge ce_clk) begin + core_rst <= user_reset | ce_rst; + case (reset_state) + S_RESET_IDLE : begin + user_reset_cnt <= 'd0; + user_reset <= 1'b0; + if (reg_user_reset) begin + user_reset <= 1'b1; + reset_state <= S_RESET_ASSERT; + end + end + S_RESET_ASSERT : begin + user_reset_cnt <= user_reset_cnt + 1; + if (user_reset_cnt == RESET_PULSE_LEN-1) begin + user_reset_cnt <= 'd0; + user_reset <= 1'b0; + reset_state <= S_RESET_IDLE; + end + end + endcase + if (ce_rst) begin + user_reset <= 1'b1; + user_reset_cnt <= 'd0; + reset_state <= S_RESET_ASSERT; + end + end + + + //--------------------------------------------------------------------------- + // Packetization + //--------------------------------------------------------------------------- + + wire [DATA_W*NUM_CHAN-1:0] user_in_tdata; + wire [ NUM_CHAN-1:0] user_in_teob; + wire [ NUM_CHAN-1:0] user_in_tlast; + wire [ NUM_CHAN-1:0] user_in_tvalid; + wire [ NUM_CHAN-1:0] user_in_tready; + wire [DATA_W*NUM_CHAN-1:0] user_out_tdata; + wire [ NUM_CHAN-1:0] user_out_teob; + wire [ NUM_CHAN-1:0] user_out_teov; + wire [ NUM_CHAN-1:0] user_out_tlast; + wire [ NUM_CHAN-1:0] user_out_tvalid; + wire [ NUM_CHAN-1:0] user_out_tready; + + for (genvar ch_i = 0; ch_i < NUM_CHAN; ch_i = ch_i + 1) begin : gen_packetize + axis_data_if_packetize #( + .NIPC (1 ), + .ITEM_W (ITEM_W), + .SIDEBAND_FWD_FIFO_SIZE_LOG2(1 ) + ) axis_data_if_packetize_i ( + .clk (ce_clk ), + .reset (core_rst ), + .spp ('0 ), + .s_axis_tdata (`BUS_I(s_in_axis_tdata, DATA_W, ch_i)), + .s_axis_tlast (`BUS_I(s_in_axis_tlast, 1, ch_i)), + .s_axis_tkeep (`BUS_I(s_in_axis_tkeep, KEEP_W, ch_i)), + .s_axis_tvalid (`BUS_I(s_in_axis_tvalid, 1, ch_i)), + .s_axis_tready (`BUS_I(s_in_axis_tready, 1, ch_i)), + .s_axis_ttimestamp (`BUS_I(s_in_axis_ttimestamp, CHDR_TIMESTAMP_W, ch_i)), + .s_axis_thas_time (`BUS_I(s_in_axis_thas_time, 1, ch_i)), + .s_axis_tlength (`BUS_I(s_in_axis_tlength, CHDR_LENGTH_W, ch_i)), + .s_axis_teov (`BUS_I(s_in_axis_teov, 1, ch_i)), + .s_axis_teob (`BUS_I(s_in_axis_teob, 1, ch_i)), + .m_axis_tdata (`BUS_I(m_out_axis_tdata, DATA_W, ch_i)), + .m_axis_tkeep (`BUS_I(m_out_axis_tkeep, KEEP_W, ch_i)), + .m_axis_tlast (`BUS_I(m_out_axis_tlast, 1, ch_i)), + .m_axis_tvalid (`BUS_I(m_out_axis_tvalid, 1, ch_i)), + .m_axis_tready (`BUS_I(m_out_axis_tready, 1, ch_i)), + .m_axis_ttimestamp (`BUS_I(m_out_axis_ttimestamp, CHDR_TIMESTAMP_W, ch_i)), + .m_axis_thas_time (`BUS_I(m_out_axis_thas_time, 1, ch_i)), + .m_axis_tlength (`BUS_I(m_out_axis_tlength, CHDR_LENGTH_W, ch_i)), + .m_axis_teov (`BUS_I(m_out_axis_teov, 1, ch_i)), + .m_axis_teob (`BUS_I(m_out_axis_teob, 1, ch_i)), + .m_axis_user_tdata (`BUS_I(user_in_tdata, DATA_W, ch_i)), + .m_axis_user_tkeep ( ), + .m_axis_user_teob (`BUS_I(user_in_teob, 1, ch_i)), + .m_axis_user_teov ( ), + .m_axis_user_tlast (`BUS_I(user_in_tlast, 1, ch_i)), + .m_axis_user_tvalid(`BUS_I(user_in_tvalid, 1, ch_i)), + .m_axis_user_tready(`BUS_I(user_in_tready, 1, ch_i)), + .s_axis_user_tdata (`BUS_I(user_out_tdata, DATA_W, ch_i)), + .s_axis_user_tkeep ('1 ), + .s_axis_user_teob (`BUS_I(user_out_teob, 1, ch_i)), + .s_axis_user_teov (`BUS_I(user_out_teov, 1, ch_i)), + .s_axis_user_tlast (`BUS_I(user_out_tlast, 1, ch_i)), + .s_axis_user_tvalid(`BUS_I(user_out_tvalid, 1, ch_i)), + .s_axis_user_tready(`BUS_I(user_out_tready, 1, ch_i)) + ); + end + + + //--------------------------------------------------------------------------- + // Combine Streams + //--------------------------------------------------------------------------- + // + // Combine input streams into one AXI-Stream bus. This lets us run the FFT + // instances (one per channel) in lock-step by sharing valid/ready signals. + // Also removes the need for multiple instances of FFT configuration logic. + // + //--------------------------------------------------------------------------- + + wire [DATA_W*NUM_CHAN-1:0] combine_out_tdata; + wire [ NUM_CHAN-1:0] combine_out_teob; + wire combine_out_tlast; + wire combine_out_tvalid; + wire combine_out_tready; + + axis_combine #( + .SIZE (NUM_CHAN), + .WIDTH (DATA_W), + .USER_WIDTH (1), + .FIFO_SIZE_LOG2 (0)) + axis_combine_inst ( + .clk (ce_clk), + .reset (core_rst), + .s_axis_tdata (user_in_tdata), + .s_axis_tuser (user_in_teob), + .s_axis_tlast (user_in_tlast), + .s_axis_tvalid (user_in_tvalid), + .s_axis_tready (user_in_tready), + .m_axis_tdata (combine_out_tdata), + .m_axis_tuser (combine_out_teob), + .m_axis_tlast (combine_out_tlast), + .m_axis_tvalid (combine_out_tvalid), + .m_axis_tready (combine_out_tready) + ); + + + //--------------------------------------------------------------------------- + // Cyclic Prefix Removal + //--------------------------------------------------------------------------- + // + // This block does the cyclic prefix removal and also sets tlast to ensure + // that the packets going into the FFT block have the expected length. + // + //--------------------------------------------------------------------------- + + logic [DATA_W*NUM_CHAN-1:0] cp_removal_out_tdata; + logic [ NUM_CHAN-1:0] cp_removal_out_teob; + logic [ NUM_CHAN-1:0] cp_removal_out_teov; + + logic cp_removal_out_tlast; + logic cp_removal_out_tvalid; + logic cp_removal_out_tready; + + // Also sets the packet size (i.e. tlast) to the FFT size + cp_removal #( + .DATA_W (NUM_CHAN*DATA_W ), + .USER_W (NUM_CHAN ), + .CP_LEN_W (CP_LEN_W ), + .SYM_LEN_W (FFT_SIZE_W ), + .DEFAULT_CP_LEN(DEFAULT_CP_LEN ), + .CP_REPEAT (CP_REMOVAL_REPEAT ), + .MAX_LIST_LOG2 (MAX_CP_LIST_LEN_REM_LOG2), + .SET_TLAST (1 ) + ) cp_removal_i ( + .clk (ce_clk ), + .rst (core_rst ), + .clear_list (reg_cp_rem_list_clr ), + .symbol_len (fft_size ), + .cp_len_tdata (reg_cp_rem_length ), + .cp_len_tvalid (reg_cp_rem_list_load ), + // No back-pressure needed since block controller checks + // cp_len_fifo_occupied to not overflow FIFO. + .cp_len_tready ( ), + .cp_list_occupied(reg_cp_rem_list_occupied), + .i_tdata (combine_out_tdata ), + .i_tuser (combine_out_teob ), + .i_tlast (combine_out_tlast ), + .i_tvalid (combine_out_tvalid ), + .i_tready (combine_out_tready ), + .o_tdata (cp_removal_out_tdata ), + .o_tuser (cp_removal_out_teob ), + .o_tlast (cp_removal_out_tlast ), + .o_tvalid (cp_removal_out_tvalid ), + .o_tready (cp_removal_out_tready ) + ); + + // We can create a teov from tlast because the packet size is the same as the FFT size + assign cp_removal_out_teov = {NUM_CHAN{cp_removal_out_tlast}}; + + + //--------------------------------------------------------------------------- + // Configuration State Machine + //--------------------------------------------------------------------------- + + logic [ITEM_W*NUM_CHAN-1:0] fft_data_in_tdata; + logic fft_data_in_tlast; + logic fft_data_in_tvalid; + logic [ NUM_CHAN-1:0] fft_data_in_tready; + + // Loads a new configuration at the start of every FFT + enum logic [0:0] { S_FFT_CONFIG, S_FFT_WAIT_FOR_TLAST } fft_config_state = S_FFT_CONFIG; + + always_ff @(posedge ce_clk) begin + case (fft_config_state) + S_FFT_CONFIG: begin + if (fft_data_in_tvalid & fft_data_in_tready[0]) begin + fft_config_state <= S_FFT_WAIT_FOR_TLAST; + end + end + S_FFT_WAIT_FOR_TLAST: begin + if (fft_data_in_tvalid & fft_data_in_tready[0] & fft_data_in_tlast) begin + fft_config_state <= S_FFT_CONFIG; + end + end + endcase + if (core_rst) begin + fft_config_state <= S_FFT_CONFIG; + end + end + + + //--------------------------------------------------------------------------- + // FFT Configuration + //--------------------------------------------------------------------------- + + logic [FFT_CONFIG_W-1:0] fft_config_tdata; + logic fft_config_tvalid; + logic [ NUM_CHAN-1:0] fft_config_tready; + + assign fft_config_tdata = build_fft_config( + MAX_FFT_SIZE_LOG2, + reg_fft_scaling, + reg_fft_direction, + reg_fft_size_log2 + ); + + assign fft_config_tvalid = (fft_config_state == S_FFT_CONFIG) ? + fft_data_in_tvalid && fft_data_in_tready : 1'b0; + + + //--------------------------------------------------------------------------- + // Sideband Info Bypass + //--------------------------------------------------------------------------- + // + // The Xilinx FFT IP lacks a TUSER signal so this adds one to pass through + // our EOB and EOV signals. + // + //--------------------------------------------------------------------------- + + logic [DATA_W*NUM_CHAN-1:0] fft_data_out_tdata; + logic [ NUM_CHAN-1:0] fft_data_out_tlast; + logic [ NUM_CHAN-1:0] fft_data_out_tvalid; + logic fft_data_out_tready; // One bit shared by all channels + + logic [DATA_W*NUM_CHAN-1:0] split_in_tdata; + logic [ NUM_CHAN-1:0] split_in_teob; + logic [ NUM_CHAN-1:0] split_in_teov; + logic split_in_tlast; + logic split_in_tvalid; + logic split_in_tready; + + // TUSER is EOB and static for the entire packet, so we can use + // PACKET_MODE=2, which is more efficient. + axis_sideband_tuser #( + .WIDTH (NUM_CHAN*DATA_W), + .USER_WIDTH (NUM_CHAN*2 ), + .FIFO_SIZE_LOG2(5 ), + .PACKET_MODE (2 ) + ) axis_sideband_tuser_i ( + .clk (ce_clk ), + .reset (core_rst ), + // Input bus with a TUSER signal + .s_axis_tdata (cp_removal_out_tdata ), + .s_axis_tuser ({cp_removal_out_teob, cp_removal_out_teov}), + .s_axis_tlast (cp_removal_out_tlast ), + .s_axis_tvalid (cp_removal_out_tvalid ), + .s_axis_tready (cp_removal_out_tready ), + // Input bus with TUSER removed, going to our FFT block + .m_axis_mod_tdata (fft_data_in_tdata ), + .m_axis_mod_tlast (fft_data_in_tlast ), + .m_axis_mod_tvalid(fft_data_in_tvalid ), + .m_axis_mod_tready(fft_data_in_tready[0] ), + // Output bus from FFT block + .s_axis_mod_tdata (fft_data_out_tdata ), + .s_axis_mod_tlast (fft_data_out_tlast[0] ), + .s_axis_mod_tvalid(fft_data_out_tvalid[0] ), + .s_axis_mod_tready(fft_data_out_tready ), + // Output bus from FFT block with TUSER added back on + .m_axis_tdata (split_in_tdata ), + .m_axis_tuser ({split_in_teob, split_in_teov} ), + .m_axis_tlast (split_in_tlast ), + .m_axis_tvalid (split_in_tvalid ), + .m_axis_tready (split_in_tready ) + ); + + + //--------------------------------------------------------------------------- + // Cyclic Prefix Insertion List + //--------------------------------------------------------------------------- + + // Output from CP list + logic [CP_LEN_W-1:0] cp_ins_list_tdata; + logic cp_ins_list_tvalid; + logic cp_ins_list_tready; + + logic [MAX_CP_LIST_LEN_INS_LOG2:0] cp_ins_list_occupied; + assign reg_cp_ins_list_occupied = REG_CP_INS_LIST_OCC_WIDTH'(cp_ins_list_occupied); + + axis_cp_list #( + .ADDR_W (MAX_CP_LIST_LEN_INS_LOG2), + .DATA_W (CP_LEN_W ), + .REPEAT (1 ), + .DEFAULT('0 ) + ) axis_cp_list_ins ( + .clk (ce_clk ), + .rst (core_rst ), + .clear (reg_cp_ins_list_clr ), + .i_tdata (reg_cp_ins_length ), + .i_tvalid(reg_cp_ins_list_load), + .i_tready( ), + .o_tdata (cp_ins_list_tdata ), + .o_tvalid(cp_ins_list_tvalid ), + .o_tready(cp_ins_list_tready ), + .occupied(cp_ins_list_occupied) + ); + + logic [DATA_W*NUM_CHAN-1:0] fft_mux_out_tdata; + logic [ NUM_CHAN-1:0] fft_mux_out_tlast; + logic [ NUM_CHAN-1:0] fft_mux_out_tvalid; + logic [ NUM_CHAN-1:0] fft_mux_out_tready; + + logic fft_mux_out_tstart = '1; // Indicates first word transfer of packet + + always_ff @(posedge ce_clk) begin + // Create a register that indicates when the first transfer of a packet + // occurs (analogous to TLAST). + if (fft_mux_out_tvalid[0] && fft_mux_out_tready[0]) begin + fft_mux_out_tstart <= fft_mux_out_tlast[0]; + end + + if (ce_rst) begin + fft_mux_out_tstart <= '1; + end + end + + // Pop off the next list item after the start of each packet. + assign cp_ins_list_tready = + fft_mux_out_tstart && fft_mux_out_tvalid[0] && fft_mux_out_tready[0]; + + + //--------------------------------------------------------------------------- + // FFT Bypass + //--------------------------------------------------------------------------- + + logic [DATA_W*NUM_CHAN-1:0] xfft_in_tdata; + logic [ NUM_CHAN-1:0] xfft_in_tlast; + logic [ NUM_CHAN-1:0] xfft_in_tvalid; + logic [ NUM_CHAN-1:0] xfft_in_tready; + + logic [DATA_W*NUM_CHAN-1:0] xfft_out_tdata; + logic [ NUM_CHAN-1:0] xfft_out_tlast; + logic [ NUM_CHAN-1:0] xfft_out_tvalid; + logic [ NUM_CHAN-1:0] xfft_out_tready; + + logic [DATA_W*NUM_CHAN-1:0] xfft_bypass_in_tdata; + logic [ NUM_CHAN-1:0] xfft_bypass_in_tlast; + logic [ NUM_CHAN-1:0] xfft_bypass_in_tvalid; + logic [ NUM_CHAN-1:0] xfft_bypass_in_tready; + + logic [DATA_W*NUM_CHAN-1:0] xfft_bypass_out_tdata; + logic [ NUM_CHAN-1:0] xfft_bypass_out_tlast; + logic [ NUM_CHAN-1:0] xfft_bypass_out_tvalid; + logic [ NUM_CHAN-1:0] xfft_bypass_out_tready; + + always_comb begin + if (reg_fft_bypass && EN_FFT_BYPASS) begin + // FIFO connections pass through + xfft_bypass_in_tdata = fft_data_in_tdata; + xfft_bypass_in_tlast = {NUM_CHAN{fft_data_in_tlast}}; + xfft_bypass_in_tvalid = {NUM_CHAN{fft_data_in_tvalid}}; + fft_data_in_tready = xfft_bypass_in_tready; + // + fft_mux_out_tdata = xfft_bypass_out_tdata; + fft_mux_out_tlast = xfft_bypass_out_tlast; + fft_mux_out_tvalid = xfft_bypass_out_tvalid; + xfft_bypass_out_tready = {NUM_CHAN{fft_mux_out_tready}}; + + // Tie off bypassed FFT connections + xfft_in_tdata = fft_data_in_tdata; + xfft_in_tlast = {NUM_CHAN{fft_data_in_tlast}}; + xfft_in_tvalid = {NUM_CHAN{1'b0}}; + xfft_out_tready = {NUM_CHAN{1'b1}}; + end else begin + // Tie off bypassed FIFO connections + xfft_bypass_in_tdata = fft_data_in_tdata; + xfft_bypass_in_tlast = {NUM_CHAN{fft_data_in_tlast}}; + xfft_bypass_in_tvalid = {NUM_CHAN{1'b0}}; + xfft_bypass_out_tready = {NUM_CHAN{1'b1}}; + + // FFT connections pass through like normal + xfft_in_tdata = fft_data_in_tdata; + xfft_in_tlast = {NUM_CHAN{fft_data_in_tlast}}; + xfft_in_tvalid = {NUM_CHAN{fft_data_in_tvalid}}; + fft_data_in_tready = xfft_in_tready; + // + fft_mux_out_tdata = xfft_out_tdata; + fft_mux_out_tlast = xfft_out_tlast; + fft_mux_out_tvalid = xfft_out_tvalid; + xfft_out_tready = fft_mux_out_tready; + end + end + + + //--------------------------------------------------------------------------- + // FFT Core + //--------------------------------------------------------------------------- + + logic [NUM_CHAN-1:0] event_frame_started; + logic [NUM_CHAN-1:0] event_frame_tlast_unexpected; + logic [NUM_CHAN-1:0] event_frame_tlast_missing; + + for (genvar fft_i = 0; fft_i < NUM_CHAN; fft_i = fft_i + 1) begin : gen_fft + if (EN_FFT_BYPASS) begin : gen_bypass_fifo + axi_fifo #( + .WIDTH(DATA_W+1 ), + .SIZE (MAX_FFT_SIZE_LOG2) + ) axi_fifo_bypass ( + .clk (ce_clk ), + .reset (core_rst ), + .clear (1'b0 ), + .i_tdata ({ xfft_bypass_in_tlast[fft_i], + xfft_bypass_in_tdata[DATA_W*fft_i +: DATA_W]} ), + .i_tvalid(xfft_bypass_in_tvalid[fft_i] ), + .i_tready(xfft_bypass_in_tready[fft_i] ), + .o_tdata ({ xfft_bypass_out_tlast[fft_i], + xfft_bypass_out_tdata[DATA_W*fft_i +: DATA_W]}), + .o_tvalid(xfft_bypass_out_tvalid[fft_i] ), + .o_tready(xfft_bypass_out_tready[fft_i] ), + .space ( ), + .occupied( ) + ); + end : gen_bypass_fifo + + xfft_wrapper #( + .MAX_FFT_SIZE_LOG2(MAX_FFT_SIZE_LOG2) + ) xfft_wrapper_i ( + .aclk (ce_clk ), + .aresetn (~core_rst ), + .s_axis_config_tdata (fft_config_tdata ), + .s_axis_config_tvalid (fft_config_tvalid ), + .s_axis_config_tready (fft_config_tready[fft_i] ), + .s_axis_data_tdata ({ xfft_in_tdata[32*fft_i +: 16], + xfft_in_tdata[32*fft_i+16 +: 16] } ), + .s_axis_data_tlast (xfft_in_tlast[fft_i] ), + .s_axis_data_tvalid (xfft_in_tvalid[fft_i] ), + .s_axis_data_tready (xfft_in_tready[fft_i] ), + .m_axis_data_tdata ({ xfft_out_tdata[32*fft_i +: 16], + xfft_out_tdata[32*fft_i+16 +: 16] }), + .m_axis_data_tuser ( ), + .m_axis_data_tlast (xfft_out_tlast[fft_i] ), + .m_axis_data_tvalid (xfft_out_tvalid[fft_i] ), + .m_axis_data_tready (xfft_out_tready[fft_i] ), + .m_axis_status_tdata ( ), + .m_axis_status_tvalid ( ), + .m_axis_status_tready (1'b1 ), + .event_frame_started (event_frame_started[fft_i] ), + .event_tlast_unexpected (event_frame_tlast_unexpected[fft_i] ), + .event_tlast_missing (event_frame_tlast_missing[fft_i] ), + .event_fft_overflow (event_fft_overflow[fft_i] ), + .event_status_channel_halt ( ), + .event_data_in_channel_halt ( ), + .event_data_out_channel_halt( ) + ); + + if (EN_FFT_ORDER || EN_MAGNITUDE || EN_MAGNITUDE_SQ) begin : gen_fft_post_processing + fft_post_processing #( + .EN_FFT_ORDER (EN_FFT_ORDER ), + .EN_MAGNITUDE (EN_MAGNITUDE ), + .EN_MAGNITUDE_SQ (EN_MAGNITUDE_SQ ), + .USE_APPROX_MAG (USE_APPROX_MAG ), + .MAX_FFT_SIZE_LOG2(MAX_FFT_SIZE_LOG2) + ) fft_post_processing_i ( + .clk (ce_clk ), + .rst (core_rst ), + .fft_order_sel(reg_fft_order ), + .magnitude_sel(reg_magnitude ), + .fft_size_log2(reg_fft_size_log2[FFT_SIZE_LOG2_W-1:0] ), + .s_axis_tdata (fft_mux_out_tdata [DATA_W*fft_i +: DATA_W] ), + .s_axis_tuser (cp_ins_list_tdata ), + .s_axis_tlast (fft_mux_out_tlast [fft_i] ), + .s_axis_tvalid(fft_mux_out_tvalid[fft_i] ), + .s_axis_tready(fft_mux_out_tready[fft_i] ), + .m_axis_tdata (fft_data_out_tdata [DATA_W*fft_i +: DATA_W]), + .m_axis_tlast (fft_data_out_tlast [fft_i] ), + .m_axis_tvalid(fft_data_out_tvalid[fft_i] ), + .m_axis_tready(fft_data_out_tready ) + ); + end else begin : gen_no_fft_post_processing + assign fft_data_out_tdata = fft_mux_out_tdata; + assign fft_data_out_tlast = fft_mux_out_tlast; + assign fft_data_out_tvalid = fft_mux_out_tvalid; + assign fft_mux_out_tready = {NUM_CHAN{fft_data_out_tready}}; + end + end + + + //--------------------------------------------------------------------------- + // Split Streams + //--------------------------------------------------------------------------- + // + // Take the time-aligned streams and make them independent AXI-Stream buses. + // + //--------------------------------------------------------------------------- + + // Split back into multiple streams + axis_split_bus #( + .WIDTH (DATA_W ), + .USER_WIDTH(2 ), + .NUM_PORTS (NUM_CHAN) + ) axis_split_bus_i ( + .clk (ce_clk ), + .reset (core_rst ), + .s_axis_tdata (split_in_tdata ), + .s_axis_tuser ({split_in_teob, split_in_teov}), + .s_axis_tlast (split_in_tlast ), + .s_axis_tvalid(split_in_tvalid ), + .s_axis_tready(split_in_tready ), + .m_axis_tdata (user_out_tdata ), + .m_axis_tuser ({user_out_teob, user_out_teov}), + .m_axis_tlast (user_out_tlast ), + .m_axis_tvalid(user_out_tvalid ), + .m_axis_tready(user_out_tready ) + ); + +endmodule : fft_core + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_core_regs_pkg.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_core_regs_pkg.sv new file mode 100644 index 0000000..8cfd1fa --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_core_regs_pkg.sv @@ -0,0 +1,268 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_core_regs_pkg +// +// Description: +// +// Package file for registers and register documentation for the fft_core +// module. +// +// WARNING: The FFT block's configuration registers should only be changed +// when the block is idle, otherwise data corruption will occur. +// + +package fft_core_regs_pkg; + + // Amount of address space allocated to each FFT core. + localparam int FFT_CORE_ADDR_W = 10; + + // Total address space required by the registers below. + localparam int REG_ADDR_W = 7; + + + //--------------------------------------------------------------------------- + // Register Descriptions + //--------------------------------------------------------------------------- + + // REG_COMPAT (Read-only) + // + // Compatibility register to indicate the version of the FPGA code. Returns a + // 16-bit value where the upper 8 bits are the major and the lower 8 bits are + // the minor compat number, with the current version being "major.minor". A + // "major" change indicates a change that breaks backwards compatibility. A + // "minor" change is one in which backwards compatibility has been preserved. + // + localparam int REG_COMPAT_ADDR = 'h00; + localparam int REG_COMPAT_WIDTH = 32; + + // REG_PORT_CONFIG (Read-only) + // + // Returns information about the the number of channels per FFT core. Each + // FFT core has its own register space. So, this effectively tells you which + // ports can be individually configured. + // + // [31:16] : NUM_CORES. The number of FFT cores in the parent RFNoC block. + // [15: 0] : NUM_CHAN. The number of channels per FFT core. + // + localparam int REG_PORT_CONFIG_ADDR = 'h04; + localparam int REG_PORT_CONFIG_WIDTH = 32; + // + localparam int REG_NUM_CORES_W = 16; + localparam int REG_NUM_CHAN_W = 16; + + // REG_CAPABILITIES (Read-only) + // + // Returns the capabilities of the core. + // + // [31:24] Log base 2 of the maximum cyclic prefix list length for insertion. + // The maximum CP list length is 2**this-1. Example: A value of 5 + // means 31 is the maximum list length. + // + // [23:16] Log base 2 of the maximum cyclic prefix list length for removal. + // The maximum CP list length is 2**this-1. Example: A value of 5 + // means 31 is the maximum list length. + // + // [15: 8] Log base 2 of the maximum cyclic prefix length (for both insertion + // and removal). The maximum cyclic prefix is 2**this-1. Example: A + // value of 12 means 4095 is the maximum cyclic prefix length. + // + // [ 7: 0] Log base 2 of the maximum supported FFT size. Example: A value of + // 12 means 4096 is the maximum FFT size. The minimum supported FFT + // size is always 8 for the Xilinx FFT core. + // + localparam int REG_CAPABILITIES_ADDR = 'h08; + localparam int REG_CAPABILITIES_WIDTH = 32; + + // REG_CAPABILITIES2 (Read-only) + // + // Returns information about the post-processing capabilities. + // + // [3] : MAGNITUDE_SQ. Indicates whether or not the magnitude-squared + // output capability is present in the core. + // [2] : MAGNITUDE. Indicates whether or not the magnitude output option + // is present in the core. + // [1] : FFT_ORDER. Indicates whether or not the FFT reorder capability is + // present in the core. + // [0] : FFT_BYPASS. Indicates whether or not the FFT bypass capability is + // present in the core. + // + localparam int REG_CAPABILITIES2_ADDR = 'h0C; + localparam int REG_CAPABILITIES2_WIDTH = 4; + + // REG_RESET (Write-only strobe) + // + // Any write to this register forces a reset of the block's internal logic. + // This register is self-clearing. + // + localparam int REG_RESET_ADDR = 'h10; + localparam int REG_RESET_WIDTH = 1; + + // REG_LENGTH (Read/Write) + // + // Log base 2 of FFT size. Use this register to configure the desired FFT + // size. Example: For 512 point FFT, set this register to 9. For a 4k FFT, + // set this register to 12. This should never exceed the maximum FFT size + // indicated by the REG_CAPABILITIES register. + // + localparam int REG_LENGTH_LOG2_ADDR = 'h14; + + // REG_SCALING (Read/Write) + // + // This is the FFT scaling word used by the Xilinx FFT core. This determines + // the scale of the output and can be adjusted to prevent overflow in the FFT + // computation. The value needed here depends on the FFT size, because that + // determines the number of stages in the FFT core. see the Xilinx FFT + // documentation PG901 for details. In general, it is a + // 2*ceil(length_log2)-bit number. To to achieve 1/N scaling, it should be + // 10...10 if the FFT size is a power of 4 (log2 of size is even) and it + // should be 0110...10 if the FFT size is not a power of 4 (log2 of size is + // odd). Example: A value of 0b_10_10_10_10_10_10 (2730) leads to 1/N scaling + // for the 4K FFT (6 stages). A value of 0b_01_10_10_10_10_10 (1706) leads to + // 1/N scaling for the 2K FFT (6 stages). + // + localparam int REG_SCALING_ADDR = 'h18; + + // REG_DIRECTION (Read/Write) + // + // Sets the FFT direction. Use 1 for forward and 0 for inverse FFT. + // + localparam int REG_DIRECTION_ADDR = 'h1C; + localparam int REG_DIRECTION_WIDTH = 1; + + // FFT direction constants + localparam bit FFT_INVERSE = 0; + localparam bit FFT_FORWARD = 1; + + // REG_CP_INS_LEN (Read/Write) + // + // Cyclic Prefix (CP) insertion length. This register holds the next CP + // insertion length to be loaded into the CP insertion list. Write the value + // to be loaded into this register then use the REG_CP_INS_LIST_LOAD register + // to load it into the list. + // + localparam int REG_CP_INS_LEN_ADDR = 'h20; + + // REG_CP_INS_LIST_LOAD (Write-only strobe) + // + // Cyclic prefix insertion list load. Any write to this register will load + // the value in the REG_CP_INS_LEN register into the cyclic prefix insertion + // list. + // + localparam int REG_CP_INS_LIST_LOAD_ADDR = 'h24; + localparam int REG_CP_INS_LIST_LOAD_WIDTH = 1; + + // REG_CP_INS_LIST_CLR (Write-only strobe) + // + // Cyclic prefix insertion list clear. Any write to this register will clear + // the cyclic prefix insertion list so that it becomes empty. + // + localparam int REG_CP_INS_LIST_CLR_ADDR = 'h28; + localparam int REG_CP_INS_LIST_CLR_WIDTH = 1; + + // REG_CP_INS_LIST_OCC (Read-only) + // + // Cyclic prefix insertion list occupied length. Returns the fullness of + // cyclic prefix insertion list. You must not overfill the insertion list, so + // this should never exceed the maximum list length indicated by the + // REG_CAPABILITIES register. + // + localparam int REG_CP_INS_LIST_OCC_ADDR = 'h2C; + localparam int REG_CP_INS_LIST_OCC_WIDTH = 16; + + // REG_CP_REM_LEN (Read/Write) + // + // Cyclic Prefix (CP) removal length. This register holds the next CP removal + // length to be loaded into the CP removal list. Write the value to be loaded + // into this register then use the REG_CP_REM_LIST_LOAD register to load it + // into the list. + // + localparam int REG_CP_REM_LEN_ADDR = 'h30; + + // REG_CP_REM_LIST_LOAD (Write-only strobe) + // + // Cyclic prefix removal list load. Any write to this register will load the + // value in the REG_CP_REM_LEN register into the cyclic prefix removal list. + // + localparam int REG_CP_REM_LIST_LOAD_ADDR = 'h34; + localparam int REG_CP_REM_LIST_LOAD_WIDTH = 1; + + // REG_CP_REM_LIST_CLR (Write-only strobe) + // + // Cyclic prefix removal list clear. Any write to this register will clear + // the cyclic prefix removal list so that it becomes empty. + // + localparam int REG_CP_REM_LIST_CLR_ADDR = 'h38; + localparam int REG_CP_REM_LIST_CLR_WIDTH = 1; + + // REG_CP_REM_LIST_OCC (Read-only) + // + // Cyclic prefix insertion list occupied length. Returns the fullness of + // cyclic prefix insertion list. You must not overfill the insertion list, so + // this should never exceed the maximum list length indicated by the + // REG_CAPABILITIES register. + // + localparam int REG_CP_REM_LIST_OCC_ADDR = 'h3C; + localparam int REG_CP_REM_LIST_OCC_WIDTH = 16; + + // REG_OVERFLOW (Read-only) + // + // Returns the overflow status of the currently addressed FFT core. Each bit + // position corresponds to a unique channel on the core. The least + // significant bit (bit 0) corresponds to the first channel, and so on. A + // value of 1 indicates that an overflow has occurred on the corresponding + // channel since the register was last read. Value of zero indicates that an + // overflow has not occurred since the register was last read. The register + // is reset back to zero whenever the register is read. + // + localparam int REG_OVERFLOW_ADDR = 'h40; + + // REG_BYPASS + // + // Enable FFT bypass. Set to 1 to enable, 0 to disable. When enabled, the + // data is passed through without performing the FFT/IFFT processing. Ensure + // that REG_CAPABILITIES2 reports this logic is present before using. + // + localparam int REG_BYPASS_ADDR = 'h44; + localparam int REG_BYPASS_WIDTH = 1; + + // REG_ORDER (Read/Write) + // + // Configures the FFT data order. The default is NORMAL. Use NATURAL for + // inverse FFT. The following values are allowed: + // + // 0 : NORMAL. Negative frequencies first, then positive frequencies. 0 Hz + // is in the center. + // 1 : REVERSE. Reverse order of NORMAL. Positive frequencies first, then + // negative frequencies. 0 Hz in the center. + // 2 : NATURAL. Positive frequencies are first, followed by negative + // frequencies. 0 Hz is on the left. + // + localparam int REG_ORDER_ADDR = 'h48; + localparam int REG_ORDER_WIDTH = 2; + // + localparam int FFT_ORDER_NORMAL = 0; + localparam int FFT_ORDER_REVERSE = 1; + localparam int FFT_ORDER_NATURAL = 2; + + // REG_MAGNITUDE (Read/Write) + // + // Configures the magnitude computation. Normal complex output is the + // default. Use normal mode for inverse FFT. Ensure that REG_CAPABILITIES2 + // reports the desired logic is present before using. The following values + // are allowed: + // + // 0 - Normal complex output (no magnitude calculation) + // 1 - Magnitude output + // 2 - Magnitude squared output + // + localparam int REG_MAGNITUDE_ADDR = 'h4C; + localparam int REG_MAGNITUDE_WIDTH = 2; + // + localparam int MAG_SEL_NONE = 0; + localparam int MAG_SEL_MAG = 1; + localparam int MAG_SEL_MAG_SQ = 2; + +endpackage : fft_core_regs_pkg diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_post_processing.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_post_processing.sv new file mode 100644 index 0000000..033140e --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_post_processing.sv @@ -0,0 +1,378 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_post_processing +// +// Description: +// +// This module contains the optional post-processing stages of an FFT, +// including FFT output reordering, magnitude, and magnitude-squared +// calculations. +// +// For the magnitude, the result is clipped to a signed 16-bit result in the +// range [0, 0x7FFF]. The result is placed in the real part of the sc16 +// output (the upper 16 bits) and the imaginary part (the lower 16 bits) is +// set to 0. +// +// For the magnitude squared, it computes (i^2 + q^2) / 0x8000, rounding and +// clipping the result to a signed 16-bit value in the range [0, 0x7FFF]. The +// division helps to avoid saturation and to put the result in a useful +// range. The result is placed in the real part of the sc16 output (the upper +// 16 bits) and the imaginary part (the lower 16 bits) is set to 0. +// +// Note that the I and Q may be swapped in the RFNoC transport adapter, so +// this order will likely be reversed by the time it makes it back to a host +// computer. +// +// Parameters: +// +// EN_FFT_ORDER : Set to 1 to add the optional FFT reorder core. Set to +// 0 to remove it and save resources. +// EN_CP_INSERTION : Controls whether to include the cyclic prefix +// insertion logic, which is a subset of EN_FFT_ORDER. +// EN_MAGNITUDE : Set to 1 to add the magnitude output calculation core. +// Set to 0 to remove it and save resources. +// EN_MAGNITUDE_SQ : Set to 1 to add the magnitude squared output +// calculation core. Set to 0 to remove it and save +// resources. +// USE_APPROX_MAG : Controls which magnitude calculation to use. Set to 1 +// to use a simpler circuit that gives pretty good +// results in order to save resources. Set to 0 to use +// the CORDIC IP to calculate the magnitude. +// MAX_FFT_SIZE_LOG2 : Set to the log base 2 of the maximum FFT size to be +// supported. For example, a value of 14 means the +// maximum FFT size is 2**14 = 4096. +// +// Signals: +// +// fft_order_sel : 0 - Normal (0 Hz in the center) +// 1 - Reverse (same as normal but in reverse) +// 2 - Natural (0 Hz on the left) +// magnitude_sel : 0 - Normal complex output (No magnitude calculation) +// 1 - Magnitude output +// 2 - Magnitude-squared output +// fft_size_log2 : Log base-2 of the FFT size. That is, the FFT size is +// exactly 2**fft_size_log2. The packet size must match. +// s_axis_* : AXI-Stream data input. s_axis_tuser contains the cyclic +// prefix length and must be valid on the first transfer of +// the packet. + +// m_axis_* : AXI-Stream data output + +`default_nettype none + + +module fft_post_processing #( + bit EN_FFT_ORDER = 1, + bit EN_CP_INSERTION = 1, + bit EN_MAGNITUDE = 1, + bit EN_MAGNITUDE_SQ = 1, + bit USE_APPROX_MAG = 1, + int MAX_FFT_SIZE_LOG2 = 12, + + localparam int FFT_SIZE_LOG2_W = $clog2(MAX_FFT_SIZE_LOG2+1), + localparam int CP_LEN_W = MAX_FFT_SIZE_LOG2 +) ( + input wire clk, + input wire rst, + + input wire [1:0] fft_order_sel, + input wire [1:0] magnitude_sel, + + input wire [FFT_SIZE_LOG2_W-1:0] fft_size_log2, + + input wire [ 31:0] s_axis_tdata, + input wire [CP_LEN_W-1:0] s_axis_tuser, + input wire s_axis_tlast, + input wire s_axis_tvalid, + output wire s_axis_tready, + + output wire [31:0] m_axis_tdata, + output wire m_axis_tlast, + output wire m_axis_tvalid, + input wire m_axis_tready +); + + //--------------------------------------------------------------------------- + // FFT Reorder + //--------------------------------------------------------------------------- + + import fft_reorder_pkg::*; + + wire [31:0] reorder_tdata; + wire reorder_tlast; + wire reorder_tvalid; + wire reorder_tready; + + if (EN_FFT_ORDER) begin : gen_fft_reorder + logic [1:0] old_fft_order_sel; + logic [FFT_SIZE_LOG2_W-1:0] old_fft_size_log2; + logic fft_cfg_wr; + + // Update the FFT config whenever it changes + always_ff @(posedge clk) begin + fft_cfg_wr <= 0; + if ( + (old_fft_order_sel != fft_order_sel) || + (old_fft_size_log2 != fft_size_log2) + ) begin + fft_cfg_wr <= 1; + end + old_fft_order_sel <= fft_order_sel; + old_fft_size_log2 <= fft_size_log2; + end + + fft_reorder #( + .INPUT_ORDER (BIT_REVERSE), + .MAX_FFT_LEN_LOG2(MAX_FFT_SIZE_LOG2), + .DATA_W (32), + .EN_CP_INSERTION (EN_CP_INSERTION) + ) fft_reorder_i ( + .clk (clk), + .rst (rst), + .fft_cfg_wr (fft_cfg_wr), + .fft_len_log2 (fft_size_log2), + .fft_out_order(fft_order_t'(fft_order_sel)), + .i_tdata (s_axis_tdata), + .i_tuser (s_axis_tuser), + .i_tlast (s_axis_tlast), + .i_tvalid (s_axis_tvalid), + .i_tready (s_axis_tready), + .o_tdata (reorder_tdata), + .o_tlast (reorder_tlast), + .o_tvalid (reorder_tvalid), + .o_tready (reorder_tready) + ); + end else begin : gen_no_fft_reorder + // Pass the data directly through when reordering is disabled. + assign reorder_tdata = s_axis_tdata; + assign reorder_tlast = s_axis_tlast; + assign reorder_tvalid = s_axis_tvalid; + assign s_axis_tready = reorder_tready; + end + + //--------------------------------------------------------------------------- + // Demultiplex Magnitude Options + //--------------------------------------------------------------------------- + + wire [31:0] mag_bypass_tdata; + wire mag_bypass_tlast; + wire mag_bypass_tvalid; + wire mag_bypass_tready; + + wire [31:0] mag_in_tdata; + wire mag_in_tlast; + wire mag_in_tvalid; + wire mag_in_tready; + + wire [31:0] mag_sq_in_tdata; + wire mag_sq_in_tlast; + wire mag_sq_in_tvalid; + wire mag_sq_in_tready; + + if (EN_MAGNITUDE || EN_MAGNITUDE_SQ) begin : gen_mag_demux + axi_demux #( + .WIDTH (32), + .SIZE (3), + .PRE_FIFO_SIZE (0), + .POST_FIFO_SIZE(0) + ) axi_demux_i ( + .clk (clk), + .reset (rst), + .clear (1'b0), + .header (), + .dest (magnitude_sel), + .i_tdata (reorder_tdata), + .i_tlast (reorder_tlast), + .i_tvalid(reorder_tvalid), + .i_tready(reorder_tready), + .o_tdata ({mag_sq_in_tdata , mag_in_tdata , mag_bypass_tdata }), + .o_tlast ({mag_sq_in_tlast , mag_in_tlast , mag_bypass_tlast }), + .o_tvalid({mag_sq_in_tvalid, mag_in_tvalid, mag_bypass_tvalid}), + .o_tready({mag_sq_in_tready, mag_in_tready, mag_bypass_tready}) + ); + end + + //--------------------------------------------------------------------------- + // Magnitude + //--------------------------------------------------------------------------- + + wire [31:0] mag_out_tdata; + wire mag_out_tlast; + wire mag_out_tvalid; + wire mag_out_tready; + + if (EN_MAGNITUDE) begin : gen_magnitude + wire [16:0] round_in_tdata; + wire [31:0] round_in_tdata_tmp; + wire round_in_tlast; + wire round_in_tvalid; + wire round_in_tready; + + wire [15:0] mag_out_tdata_tmp; + + if (!USE_APPROX_MAG) begin : gen_cordic + wire [47:0] m_axis_dout_tdata; + // The CORDIC IP below inputs/outputs its data as signed numbers having 2 + // whole bits and 15 fractional bits (17 total bits). To be compliant + // with AXI, each value is stuffed into a 24-bit vector. On the input, we + // resize our sc16 inputs to be 24 bits (the upper 7 bits will be ignored + // by the IP). On the output side, we only need the magnitude, which is + // in the lower 17 bits. The phase, in the upper bits, is left unused. + complex_to_magphase_int17 complex_to_magphase_int17_i ( + .aclk (clk), + .aresetn (~rst), + .s_axis_cartesian_tvalid(mag_in_tvalid), + .s_axis_cartesian_tlast (mag_in_tlast), + .s_axis_cartesian_tready(mag_in_tready), + .s_axis_cartesian_tdata ({ 24'(signed'(mag_in_tdata[31:16])), + 24'(signed'(mag_in_tdata[15:0])) }), + .m_axis_dout_tvalid (round_in_tvalid), + .m_axis_dout_tlast (round_in_tlast), + .m_axis_dout_tdata (m_axis_dout_tdata), + .m_axis_dout_tready (round_in_tready) + ); + assign round_in_tdata_tmp = 32'(m_axis_dout_tdata[16:0]); + end else if (USE_APPROX_MAG) begin : gen_approx + complex_to_mag_approx complex_to_mag_approx_i ( + .clk (clk), + .reset (rst), + .clear (1'b0), + .i_tvalid(mag_in_tvalid), + .i_tlast (mag_in_tlast), + .i_tready(mag_in_tready), + .i_tdata (mag_in_tdata), + .o_tvalid(round_in_tvalid), + .o_tlast (round_in_tlast), + .o_tready(round_in_tready), + .o_tdata (round_in_tdata_tmp[15:0]) + ); + assign round_in_tdata_tmp[31:16] = '0; + end + + // The magnitude is always positive, so we set the MSB to 0 then clip the + // result to a signed 16-bit value. + assign round_in_tdata = {1'b0, round_in_tdata_tmp[15:0]}; + + axi_round_and_clip #( + .WIDTH_IN (17), + .WIDTH_OUT(16), + .CLIP_BITS(1) + ) axi_round_and_clip_i ( + .clk (clk), + .reset (rst), + .i_tdata (round_in_tdata), + .i_tlast (round_in_tlast), + .i_tvalid(round_in_tvalid), + .i_tready(round_in_tready), + .o_tdata (mag_out_tdata_tmp), + .o_tlast (mag_out_tlast), + .o_tvalid(mag_out_tvalid), + .o_tready(mag_out_tready) + ); + + // Put the resulting magnitude in the "real" part of the output + assign mag_out_tdata = {mag_out_tdata_tmp, 16'd0}; + + end else begin : gen_no_magnitude + assign mag_out_tdata = '0; + assign mag_out_tlast = '0; + assign mag_out_tvalid = '0; + assign s_axis_tready = '1; + end + + //--------------------------------------------------------------------------- + // Magnitude Squared + //--------------------------------------------------------------------------- + + wire [31:0] mag_sq_out_tdata; + wire mag_sq_out_tlast; + wire mag_sq_out_tvalid; + wire mag_sq_out_tready; + + if (EN_MAGNITUDE_SQ) begin : gen_magnitude_squared + wire [31:0] round_in_tdata; + wire round_in_tlast; + wire round_in_tvalid; + wire round_in_tready; + + wire [15:0] mag_sq_out_tdata_tmp; + + complex_to_magsq complex_to_magsq_i ( + .clk (clk), + .reset (rst), + .clear (1'b0), + .i_tvalid(mag_sq_in_tvalid), + .i_tlast (mag_sq_in_tlast), + .i_tready(mag_sq_in_tready), + .i_tdata (mag_sq_in_tdata), + .o_tvalid(round_in_tvalid), + .o_tlast (round_in_tlast), + .o_tready(round_in_tready), + .o_tdata (round_in_tdata) + ); + + axi_round_and_clip #( + .WIDTH_IN (32), + .WIDTH_OUT(16), + .CLIP_BITS(1) + ) axi_round_and_clip_i ( + .clk (clk), + .reset (rst), + .i_tdata (round_in_tdata), + .i_tlast (round_in_tlast), + .i_tvalid(round_in_tvalid), + .i_tready(round_in_tready), + .o_tdata (mag_sq_out_tdata_tmp), + .o_tlast (mag_sq_out_tlast), + .o_tvalid(mag_sq_out_tvalid), + .o_tready(mag_sq_out_tready) + ); + + assign mag_sq_out_tdata = {mag_sq_out_tdata_tmp, 16'd0}; + + end else begin : gen_no_magnitude_squared + assign mag_sq_out_tdata = '0; + assign mag_sq_out_tlast = '0; + assign mag_sq_out_tvalid = '0; + assign mag_sq_in_tready = '1; + end + + //--------------------------------------------------------------------------- + // Combine Magnitude Options + //--------------------------------------------------------------------------- + + if (EN_MAGNITUDE || EN_MAGNITUDE_SQ) begin : gen_mag_mux + axi_mux #( + .PRIO (1), + .WIDTH (32), + .SIZE (3), + .PRE_FIFO_SIZE (0), + .POST_FIFO_SIZE(0) + ) axi_demux_i ( + .clk (clk), + .reset (rst), + .clear (1'b0), + .i_tdata ({mag_sq_out_tdata , mag_out_tdata , mag_bypass_tdata }), + .i_tlast ({mag_sq_out_tlast , mag_out_tlast , mag_bypass_tlast }), + .i_tvalid({mag_sq_out_tvalid, mag_out_tvalid, mag_bypass_tvalid}), + .i_tready({mag_sq_out_tready, mag_out_tready, mag_bypass_tready}), + .o_tdata (m_axis_tdata), + .o_tlast (m_axis_tlast), + .o_tvalid(m_axis_tvalid), + .o_tready(m_axis_tready) + ); + end else begin : gen_no_mag_mux + assign m_axis_tdata = reorder_tdata; + assign m_axis_tlast = reorder_tlast; + assign m_axis_tvalid = reorder_tvalid; + assign reorder_tready = m_axis_tready; + end + + +endmodule : fft_post_processing + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder.sv new file mode 100644 index 0000000..0053c57 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder.sv @@ -0,0 +1,623 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_reorder +// +// Description: +// +// This module optionally rearranges the order of FFT bins to put them in the +// desired order. It also supports cyclic prefix insertion. +// +// The input order that this module receives is a parameter that must be +// chosen at compile time. The following input orders are supported: +// +// NATURAL: Positive frequencies are input first, starting with 0 Hz, +// followed by negative frequencies. Frequencies are input in +// ascending order. +// BIT_REVERSE: Like natural, but the bits of the indices are in reverse +// order. For example, for a size 16 FFT, bin 0000 is input +// first, followed by bin 1000, 0100, 1100, 0010, etc. +// +// The output order can be chosen at run time. The following output orders +// are supported: +// +// NORMAL: Negative frequencies first, then positive frequencies. 0 Hz +// is in the center. Frequencies are output in ascending order. +// REVERSE: Reverse order of NORMAL. Positive frequencies first, then +// negative frequencies. 0 Hz in the center. Frequencies are +// output in descending order. +// NATURAL: Positive frequencies are first, starting with 0 Hz, +// followed by negative frequencies. Frequencies are output in +// ascending order. +// BIT_REVERSE: Like natural, but the bits of the indices are in reverse +// order. For example, for a size 16 FFT, bin 0000 is output +// first, followed by bin 1000, 0100, 1100, 0010, etc. +// +// Typically the FFT IP feeding this module will output data in BIT_REVERSE +// order. The FFT IP may have an option to rearrange the data into NATURAL +// order but enabling this feature causes a large memory to be added to the +// IP to do the reordering. Since we want to also be able to provide NORMAL +// order, and we don't want to add a second memory for that reordering, we do +// all the reordering here in one memory. +// +// If the FFT core is outputting in the order you want, then this module +// should probably be removed to save RAM and logic. +// +// The TLAST input/output corresponds to when the FFT input/output ends for a +// single FFT-sized sequence of data. i_tlast must be asserted during the +// last transfer of the input FFT to reset things for the next FFT input. +// +// For cyclic prefix insertion, the EN_CP_INSERTION parameter must be true +// and i_tuser contains the cyclic prefix size to insert. It must be valid +// during the first transfer of the packet. It can be any size from 0 to +// 2**MAX_FFT_LEN_LOG2-1. +// +// Parameters: +// +// IN_FIFO_LOG2 : Log base-2 of the input FIFO size. Set to -1 to remove +// the input FIFO. This FIFO is intended as a pipeline +// stage to cut the timing path on the input. +// OUT_FIFO_LOG2 : Log base-2 of the output FIFO size. This must be set to +// at least 3. +// INPUT_ORDER : BIT_REVERSE or NATURAL. See fft_reorder_pkg for values. +// MAX_FFT_LEN_LOG2 : Ceiling of log base-2 of the maximum FFT size to be +// supported. +// DATA_W : Data width. Typically 32 for sc16 data type. +// EN_CP_INSERTION : Controls whether or not the CP insertion logic is +// included. +// +// Signals: +// +// i_t* : AXI-Stream data input. Each packet is one FFT to be processed. The +// length of the packet must match the FFT size. i_tuser contains the +// cyclic prefix size to insert for this packet and must be valid +// during the first transfer of the packet. +// o_t* : AXI-Stream data output. Each packet is one FFT with optional cyclic +// prefix. +// + +`default_nettype none + + +module fft_reorder + import fft_reorder_pkg::*; +#( + parameter int IN_FIFO_LOG2 = 1, + parameter int OUT_FIFO_LOG2 = 3, + parameter fft_order_t INPUT_ORDER = BIT_REVERSE, + parameter int MAX_FFT_LEN_LOG2 = 12, + parameter int DATA_W = 32, + parameter bit EN_CP_INSERTION = 1, + + localparam int FFT_LEN_LOG2_W = $clog2(MAX_FFT_LEN_LOG2+1), + localparam int CP_LEN_W = MAX_FFT_LEN_LOG2 +) ( + input wire clk, + input wire rst, + + input wire fft_cfg_wr, + input wire [FFT_LEN_LOG2_W-1:0] fft_len_log2, + input fft_order_t fft_out_order, + + // Data Input + input wire [ DATA_W-1:0] i_tdata, + input wire [CP_LEN_W-1:0] i_tuser, + input wire i_tlast, + input wire i_tvalid, + output wire i_tready, + + // Data Output + output wire [DATA_W-1:0] o_tdata, + output wire o_tlast, + output wire o_tvalid, + input wire o_tready +); + + // These registers track if the current read/write buffers are OK to use + logic ok_to_write = 1'b1; // Current write buffer is free for writes + logic ok_to_read = 1'b0; // Current read buffer has data to read + + + //--------------------------------------------------------------------------- + // Optional Data Input Pipeline + //--------------------------------------------------------------------------- + + logic [ DATA_W-1:0] in_fifo_o_tdata; + logic [CP_LEN_W-1:0] in_fifo_o_tuser; + logic in_fifo_o_tvalid; + logic in_fifo_o_tready; + logic in_fifo_o_tlast; + + if (IN_FIFO_LOG2 >= 0) begin : gen_in_fifo + axi_fifo #( + .WIDTH(1 + CP_LEN_W + DATA_W), + .SIZE (IN_FIFO_LOG2) + ) axi_fifo_in ( + .clk (clk), + .reset (rst), + .clear ('0), + .i_tdata ({i_tlast, i_tuser, i_tdata}), + .i_tvalid(i_tvalid), + .i_tready(i_tready), + .o_tdata ({in_fifo_o_tlast, in_fifo_o_tuser, in_fifo_o_tdata}), + .o_tvalid(in_fifo_o_tvalid), + .o_tready(in_fifo_o_tready), + .space (), + .occupied() + ); + end else begin : gen_no_in_fifo + assign in_fifo_o_tdata = i_tdata; + assign in_fifo_o_tuser = i_tuser; + assign in_fifo_o_tlast = i_tlast; + assign in_fifo_o_tvalid = i_tvalid; + assign i_tready = in_fifo_o_tready; + end + + + //--------------------------------------------------------------------------- + // Optional Data Output Pipeline + //--------------------------------------------------------------------------- + + if (OUT_FIFO_LOG2 < 3) begin + OUT_FIFO_LOG2_must_be_at_least_3(); + end + + logic [DATA_W-1:0] out_fifo_i_tdata; + logic out_fifo_i_tvalid; + logic out_fifo_i_tlast; + + // We use out_fifo_space instead of out_fifo_i_tready to allow extra space + // for the RAM output read delay. + logic [15:0] out_fifo_space; + + axi_fifo #( + .WIDTH(DATA_W+1), + .SIZE (OUT_FIFO_LOG2) + ) axi_fifo_out ( + .clk (clk), + .reset (rst), + .clear ('0), + .i_tdata ({out_fifo_i_tlast, out_fifo_i_tdata}), + .i_tvalid(out_fifo_i_tvalid), + .i_tready(), + .o_tdata ({o_tlast, o_tdata}), + .o_tvalid(o_tvalid), + .o_tready(o_tready), + .space (out_fifo_space), + .occupied() + ); + + + //--------------------------------------------------------------------------- + // Configuration Registers + //--------------------------------------------------------------------------- + // + // Store relevant FFT configuration values in registers for use elsewhere. We + // assume that the configuration is set in advance of any operation and is + // only changed when the FFT is idle, so we ignore the latency here. + // + //--------------------------------------------------------------------------- + + // Number of bits needed to represent the maximum FFT size + localparam FFT_LEN_W = MAX_FFT_LEN_LOG2+1; + + logic fft_cfg_wr_stb = 1'b0; + fft_order_t fft_out_order_reg = NORMAL; + logic [FFT_LEN_LOG2_W-1:0] fft_len_log2_reg = MAX_FFT_LEN_LOG2; + logic [FFT_LEN_W-1:0] fft_len = 1 << MAX_FFT_LEN_LOG2; + logic [FFT_LEN_W-1:0] fft_len_m1 = (1 << MAX_FFT_LEN_LOG2)-1; + + always_ff @(posedge clk) begin + if(rst) begin + fft_cfg_wr_stb <= 1'b0; + fft_out_order_reg <= NORMAL; + fft_len_log2_reg <= MAX_FFT_LEN_LOG2; + fft_len <= 1 << MAX_FFT_LEN_LOG2; + fft_len_m1 <= (1 << MAX_FFT_LEN_LOG2)-1; + end else begin + fft_cfg_wr_stb <= 1'b0; + if (fft_cfg_wr) begin + fft_cfg_wr_stb <= 1'b1; + fft_out_order_reg <= fft_out_order; + fft_len_log2_reg <= fft_len_log2; + fft_len <= (1 << fft_len_log2); + fft_len_m1 <= (1 << fft_len_log2)-1; + end + end + end + + + //--------------------------------------------------------------------------- + // RAM Buffer + //--------------------------------------------------------------------------- + // + // This RAM stores the data that's being input, writing it the order needed + // such that when read out sequentially, it will be in the correct order. + // + // The RAM is divided into two halves, which we'll call buffers. Each buffer + // is used exclusively for read or write, until they switch. + // + //--------------------------------------------------------------------------- + + // Address width for each buffer. Must be big enough to store the maximum + // length FFT. + localparam ADDR_W = MAX_FFT_LEN_LOG2; + + // RAM read latency + localparam READ_LATENCY = 2; + + logic ram_rd_buffer; // Indicates which buffer is currently used for reads + logic ram_wr_buffer; // Indicates which buffer is currently used for writes + logic ram_wr_en; + logic ram_wr_en_0; // One RAM read enable for each buffer + logic ram_wr_en_1; + logic [ADDR_W-1:0] ram_wr_addr; + logic [DATA_W-1:0] ram_wr_data; + logic ram_rd_en; + logic [ADDR_W-1:0] ram_rd_addr; + logic [DATA_W-1:0] ram_rd_data_raw_0; // One RAM read output for each buffer + logic [DATA_W-1:0] ram_rd_data_raw_1; + + ram_2port #( + .DWIDTH (DATA_W), + .AWIDTH (ADDR_W), // Make the RAM two buffers big + .OUT_REG(1) + ) ram_2port_0 ( + .clka (clk), + .ena ('1), + .wea (ram_wr_en_0), + .addra(ram_wr_addr), + .dia (ram_wr_data), + .doa (), + .clkb (clk), + .enb ('1), + .web ('0), + .addrb(ram_rd_addr), + .dib ('0), + .dob (ram_rd_data_raw_0) + ); + + ram_2port #( + .DWIDTH (DATA_W), + .AWIDTH (ADDR_W), // Make the RAM two buffers big + .OUT_REG(1) + ) ram_2port_1 ( + .clka (clk), + .ena ('1), + .wea (ram_wr_en_1), + .addra(ram_wr_addr), + .dia (ram_wr_data), + .doa (), + .clkb (clk), + .enb ('1), + .web ('0), + .addrb(ram_rd_addr), + .dib ('0), + .dob (ram_rd_data_raw_1) + ); + + + //--------------------------------------------------------------------------- + // Write Logic + //--------------------------------------------------------------------------- + // + // Here we write the data into the memory in a carefully controlled order + // such that we can read it out in sequential or bit-reversed order to get + // the order we want. + // + //--------------------------------------------------------------------------- + + logic [FFT_LEN_W-1:0] fft_addr_mask; + logic [FFT_LEN_W-1:0] wr_count; + + logic ram_wr_last; + + assign ram_wr_data = in_fifo_o_tdata; + assign ram_wr_en = in_fifo_o_tvalid && in_fifo_o_tready; + assign ram_wr_en_0 = ram_wr_en && (ram_wr_buffer == 1'b0); + assign ram_wr_en_1 = ram_wr_en && (ram_wr_buffer == 1'b1); + assign in_fifo_o_tready = ok_to_write; + assign ram_wr_last = in_fifo_o_tlast; + + always_ff @(posedge clk) begin + if (fft_cfg_wr_stb || (ram_wr_en && ram_wr_last)) begin + if (fft_out_order_reg == NATURAL) begin + // Natural to natural. No mask needed to affect the order. + fft_addr_mask <= '0; + ram_wr_addr <= '0; + end else if (fft_out_order_reg == REVERSE) begin + // Natural to reverse. Invert all bits except the MSB. Inverting the + // lower bits reverses the order. Leaving the MSB unchanged ensures we + // output positive frequencies first, then negative frequencies. + fft_addr_mask <= fft_len_m1 >> 1; // e.g., 8'b0111_1111 + ram_wr_addr <= fft_len_m1 >> 1; + end else if (fft_out_order_reg == NORMAL) begin + // Natural to normal. Invert the MSB, so that we output negative + // frequencies first, then positive frequencies. + fft_addr_mask <= fft_len >> 1; // e.g., 8'b1000_0000 + ram_wr_addr <= fft_len >> 1; + end else begin // (fft_order_t == BIT_REVERSE) + // Natural to bit-reverse. For this we also use natural order, and we + // enable/disable the bit-reversal on the read side as needed. + fft_addr_mask <= '0; + ram_wr_addr <= '0; + end + end + + if (ram_wr_en) begin + wr_count <= wr_count+1; + + if (ram_wr_last) begin + // Switch to the other buffer + ram_wr_buffer <= ~ram_wr_buffer; + wr_count <= '0; + end else begin + // Calculate the the next write address + if ( + (INPUT_ORDER == BIT_REVERSE && fft_out_order_reg != BIT_REVERSE) || + (INPUT_ORDER == NATURAL && fft_out_order_reg == BIT_REVERSE) + ) begin : bit_reversed + // If the input is bit-reversed and we're not outputting + // bit-reversed, then we bit reverse the RAM address to convert from + // bit-reversed to natural order. Then apply the mask to that to + // convert from natural to the desired output order. + ram_wr_addr <= bit_reverse(wr_count+1, fft_len_log2_reg) ^ fft_addr_mask; + end else begin : natural + // Apply the mask to convert from natural to to the desired output + // order. + ram_wr_addr <= (wr_count+1) ^ fft_addr_mask; + end + end + end + + if (rst) begin + ram_wr_buffer <= '0; + ram_wr_addr <= '0; + wr_count <= '0; + end + end + + + //--------------------------------------------------------------------------- + // CP Insertion Length FIFO + //--------------------------------------------------------------------------- + + // Cyclic prefix logic interface signals + logic cp_valid; // Indicates the CP FIFO has an output + logic cp_non_zero; // Indicates the CP value is > 0 + logic [ADDR_W-1:0] cp_start_addr; // Indicates the CP RAM start address + logic cp_consume; // Control to indicate we've captured the CP length output + + if (EN_CP_INSERTION) begin: gen_cp_ins_fifo + logic [CP_LEN_W-1:0] cp_len_tdata; + logic cp_len_tvalid; + logic cp_len_tready; + logic i_tvalid; + logic in_fifo_o_tfirst = '1; // First transfer of packet + + // Create a register that indicates when the next transfer is the start of + // a new packet. + always_ff @(posedge clk) begin + if (rst) begin + in_fifo_o_tfirst <= '1; + end else begin + if (in_fifo_o_tvalid && in_fifo_o_tready) begin + in_fifo_o_tfirst <= in_fifo_o_tlast; + end + end + end + + // Write the first tuser word of the packet into the CP length FIFO + assign i_tvalid = in_fifo_o_tvalid && in_fifo_o_tready && in_fifo_o_tfirst; + + // The dual RAM buffer can only hold two FFTs at a time, so we can + // guarantee this FIFO has sufficient room and will always be ready by + // setting its size appropriately. + axi_fifo #( + .WIDTH(CP_LEN_W), + .SIZE (1) + ) axi_fifo_cp_length ( + .clk (clk), + .reset (rst), + .clear ('0), + .i_tdata (in_fifo_o_tuser), + .i_tvalid(i_tvalid), + .i_tready(), + .o_tdata (cp_len_tdata), + .o_tvalid(cp_len_tvalid), + .o_tready(cp_consume), + .space (), + .occupied() + ); + + // Add a register to calculate the cyclic prefix start read address and + // figure out if we need to do a cyclic prefix insertion. The latency of + // this register will be much less than the FFT write time. + always_ff @(posedge clk) begin + cp_valid <= cp_len_tvalid; + cp_non_zero <= (cp_len_tdata != 0); + cp_start_addr <= fft_len - cp_len_tdata; + end + end else begin : gen_no_cp_ins_fifo + assign cp_valid = '0; + assign cp_non_zero = '0; + assign cp_start_addr = '0; + end + + + //--------------------------------------------------------------------------- + // Read Logic + //--------------------------------------------------------------------------- + + typedef enum logic [1:0] { READ_CHECK, READ_CP, READ_FFT} read_state_t; + read_state_t read_state = EN_CP_INSERTION ? READ_CHECK : READ_FFT; + read_state_t read_state_nx; + + logic [ADDR_W-1:0] ram_rd_addr_nx; + logic ram_rd_buffer_nx; + logic ram_rd_last; // Indicates when ram_rd_en asserts for the last sample + logic out_fifo_avail; + + // Delayed versions of read signals to align with read output timing + logic [READ_LATENCY-1:0] ram_rd_buffer_del; + logic [READ_LATENCY-1:0] ram_rd_en_del; + logic [READ_LATENCY-1:0] ram_rd_last_del; + + logic [DATA_W-1:0] ram_rd_data; + logic ram_rd_data_valid; // Indicates ram_rd_data has data + logic ram_rd_data_last; // Indicates ram_rd_data is the last of the FFT + + assign out_fifo_i_tdata = ram_rd_data; + assign out_fifo_i_tvalid = ram_rd_data_valid; + assign out_fifo_i_tlast = ram_rd_data_last; + + always_ff @(posedge clk) begin : read_fsm_reg + if (rst) begin + read_state <= EN_CP_INSERTION ? READ_CHECK : READ_FFT; + ram_rd_buffer <= '0; + ram_rd_addr <= '0; + ram_rd_buffer_del <= '0; + ram_rd_en_del <= '0; + ram_rd_last_del <= '0; + ram_rd_data <= 'X; + ram_rd_data_valid <= '0; + ram_rd_data_last <= '0; + out_fifo_avail <= '0; + end else begin + read_state <= read_state_nx; + ram_rd_buffer <= ram_rd_buffer_nx; + ram_rd_addr <= ram_rd_addr_nx; + + // Pipeline the buffer selection, enable, and last to align with RAM output + ram_rd_buffer_del <= (ram_rd_buffer_del << 1) | ram_rd_buffer; + ram_rd_en_del <= (ram_rd_en_del << 1) | ram_rd_en; + ram_rd_last_del <= (ram_rd_last_del << 1) | ram_rd_last; + + // Select the RAM output that was used for the read + ram_rd_data <= ram_rd_buffer_del[READ_LATENCY-1] ? + ram_rd_data_raw_1 : ram_rd_data_raw_0; + ram_rd_data_valid <= ram_rd_en_del[READ_LATENCY-1]; + ram_rd_data_last <= ram_rd_last_del[READ_LATENCY-1]; + + // Ensure there's enough room in the output FIFO to account for the + // latency through the read logic. + out_fifo_avail <= out_fifo_space > 4; + end + end + + always_comb begin : read_fsm_comb + ram_rd_en = '0; + ram_rd_last = '0; + ram_rd_buffer_nx = ram_rd_buffer; + ram_rd_addr_nx = ram_rd_addr; + read_state_nx = read_state; + cp_consume = '0; + + case (read_state) + READ_CHECK : begin + // Wait until the next cyclic prefix is available and update the RAM + // read address appropriately. + if (cp_valid) begin + cp_consume = '1; + if (cp_non_zero) begin + read_state_nx = READ_CP; + ram_rd_addr_nx = cp_start_addr; + end else begin + read_state_nx = READ_FFT; + ram_rd_addr_nx = '0; + end + end + end + READ_CP : begin + // Read out the cyclic prefix + ram_rd_en = (ok_to_read && out_fifo_avail); + if (ram_rd_en) begin + if (ram_rd_addr == fft_len_m1) begin + ram_rd_addr_nx = '0; + read_state_nx = READ_FFT; + end else begin + ram_rd_addr_nx = ram_rd_addr + 1; + end + end + end + default : begin // READ_FFT + // Read out the whole FFT + ram_rd_en = (ok_to_read && out_fifo_avail); + if (ram_rd_en) begin + if (ram_rd_addr == fft_len_m1) begin + ram_rd_last = '1; + ram_rd_addr_nx = '0; + ram_rd_buffer_nx = ~ram_rd_buffer; + read_state_nx = EN_CP_INSERTION ? READ_CHECK : READ_FFT; + end else begin + ram_rd_addr_nx = ram_rd_addr + 1; + end + end + end + endcase + end + + + //--------------------------------------------------------------------------- + // Read/Write Arbitration Logic + //--------------------------------------------------------------------------- + // + // Here we ensure that we only write when the write buffer is free and that + // we only read when the read buffer has an FFT in it. Because we're reading + // and writing simultaneously, we swap between the lower and upper parts of + // the RAM as data gets written and read out. + // + //--------------------------------------------------------------------------- + + always_ff @(posedge clk) begin + if (ram_wr_en && ram_rd_en) begin + if (ram_wr_last && ram_rd_last) begin + // Both buffers are switching on the same cycle + ok_to_write <= 1'b1; + ok_to_read <= 1'b1; + end else if (ram_wr_last) begin + // Switching write buffer to the one being used for reads + ok_to_write <= 1'b0; + end else if (ram_rd_last) begin + // Switching read buffer to the one being used for writes + ok_to_read <= 1'b0; + end + end else if (ram_wr_en && ram_wr_last) begin + // Write buffer is switching + if (ram_wr_buffer == ram_rd_buffer) begin + // Write buffer is switching away from the current read buffer + ok_to_write <= 1'b1; + ok_to_read <= 1'b1; + end else begin + // Write buffer is switching to the current read buffer + ok_to_write <= 1'b0; + end + end else if (ram_rd_en && ram_rd_last) begin + // Read buffer is switching + if (ram_wr_buffer == ram_rd_buffer) begin + // Read buffer is switching away from the current write buffer + ok_to_write <= 1'b1; + ok_to_read <= 1'b1; + end else begin + // Read buffer is switching to the current write buffer + ok_to_read <= 1'b0; + end + end + + //synthesis translate_off + if (ram_wr_en && ram_rd_en && (ram_wr_buffer == ram_rd_buffer)) begin + $error("Attempt to read and write the same buffer!"); + end + //synthesis translate_on + + if (rst) begin + ok_to_write <= 1'b1; // Buffers empty after reset + ok_to_read <= 1'b0; // Can't read until we fill the first buffer + end + end + +endmodule + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_pkg.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_pkg.sv new file mode 100644 index 0000000..24f0cc6 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_pkg.sv @@ -0,0 +1,37 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_reorder_pkg +// +// Description: +// +// Package file for the fft_reorder module. Includes relevant types and +// functions needed by the module. +// + +package fft_reorder_pkg; + + typedef enum bit [1:0] { + NORMAL, + REVERSE, + NATURAL, + BIT_REVERSE + } fft_order_t; + + // Reverse the order of the lower `width` bits on the `index` input. The + // upper bits will be 0. + function automatic int bit_reverse (bit [15:0] index, bit [3:0] width); + bit [15:0] result; + + // Reverse bit order + result = { << { index }}; + + // Right-align + result = result >> (16 - width); + + return result; + endfunction + +endpackage : fft_reorder_pkg diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile new file mode 100644 index 0000000..d0f2a0c --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile @@ -0,0 +1,52 @@ +# +# Copyright 2024 Ettus Research, a National Instruments Brand +# +# SPDX-License-Identifier: LGPL-3.0-or-later +# + +#------------------------------------------------- +# Top-of-Makefile +#------------------------------------------------- +# Define BASE_DIR to point to the "top" dir +BASE_DIR = $(abspath ../../../../../top) +# Include viv_sim_preample after defining BASE_DIR +include $(BASE_DIR)/../tools/make/viv_sim_preamble.mak + +#------------------------------------------------- +# IP Specific +#------------------------------------------------- +# If simulation contains IP, define the IP_DIR and point +# it to the base level IP directory +LIB_IP_DIR = $(BASE_DIR)/../lib/ip + +#------------------------------------------------- +# Design Specific +#------------------------------------------------- +# Include makefiles and sources for the DUT and its +# dependencies. +include $(BASE_DIR)/../lib/rfnoc/core/Makefile.srcs +include $(BASE_DIR)/../lib/rfnoc/utils/Makefile.srcs +include Makefile.srcs + +DESIGN_SRCS += $(abspath \ +$(RFNOC_CORE_SRCS) \ +$(RFNOC_UTIL_SRCS) \ +$(FFT_REORDER_SRCS) \ +) + +#------------------------------------------------- +# Testbench Specific +#------------------------------------------------- +SIM_TOP = fft_reorder_all_tb +SIM_SRCS = $(abspath \ +fft_reorder_tb.sv \ +fft_reorder_all_tb.sv \ +) + +#------------------------------------------------- +# Bottom-of-Makefile +#------------------------------------------------- +# Include all simulator specific makefiles here +# Each should define a unique target to simulate +# e.g. xsim, vsim, etc and a common "clean" target +include $(BASE_DIR)/../tools/make/viv_simulator.mak diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile.srcs b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile.srcs new file mode 100644 index 0000000..e08442b --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/Makefile.srcs @@ -0,0 +1,10 @@ +# +# Copyright 2024 Ettus Research, a National Instruments Brand +# +# SPDX-License-Identifier: LGPL-3.0-or-later +# + +FFT_REORDER_SRCS += $(abspath $(addprefix $(BASE_DIR)/../lib/rfnoc/blocks/rfnoc_block_fft/, \ +fft_reorder_pkg.sv \ +fft_reorder.sv \ +)) diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_all_tb.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_all_tb.sv new file mode 100644 index 0000000..adab2b2 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_all_tb.sv @@ -0,0 +1,27 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_reorder_all_tb +// +// Description: +// +// Top-level testbench for fft_reorder_tb, testing different configurations of +// the module. +// + +module fft_reorder_all_tb; + import fft_reorder_pkg::*; + + // Test different input orders. Do a larger size for bit reverse to test a + // larger memory. + fft_reorder_tb #(.INPUT_ORDER( NATURAL), .MAX_FFT_LEN_LOG2( 6)) tb_01(); + fft_reorder_tb #(.INPUT_ORDER(BIT_REVERSE), .MAX_FFT_LEN_LOG2(12)) tb_02(); + + // Test different input FIFO configs. Use smaller size for quicker test. + fft_reorder_tb #(.IN_FIFO_LOG2(-1), .OUT_FIFO_LOG2(3), .INPUT_ORDER(NATURAL), .MAX_FFT_LEN_LOG2(5)) tb_03(); + fft_reorder_tb #(.IN_FIFO_LOG2( 0), .OUT_FIFO_LOG2(5), .INPUT_ORDER(BIT_REVERSE), .MAX_FFT_LEN_LOG2(5)) tb_04(); + fft_reorder_tb #(.IN_FIFO_LOG2( 1), .OUT_FIFO_LOG2(6), .INPUT_ORDER(NATURAL), .MAX_FFT_LEN_LOG2(5)) tb_05(); + +endmodule : fft_reorder_all_tb diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_tb.sv b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_tb.sv new file mode 100644 index 0000000..4ba9ee5 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/fft_reorder_tb/fft_reorder_tb.sv @@ -0,0 +1,446 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: fft_reorder_tb +// +// Description: +// +// Testbench for fft_reorder. +// + +`default_nettype none + + +module fft_reorder_tb + import fft_reorder_pkg::*; +#( + int IN_FIFO_LOG2 = 1, + int OUT_FIFO_LOG2 = 3, + fft_order_t INPUT_ORDER = BIT_REVERSE, + int MAX_FFT_LEN_LOG2 = 5, + int EN_CP_INSERTION = 1 +) (); + + // Include macros and time declarations for use with PkgTestExec + `include "test_exec.svh" + import PkgTestExec::*; + import PkgAxiStreamBfm::*; + import PkgRandom::*; + + localparam real CLK_PERIOD = 10.0; + localparam int DATA_W = 32; + localparam int CP_LEN_W = MAX_FFT_LEN_LOG2; + localparam int FFT_LEN_LOG2_W = $clog2(MAX_FFT_LEN_LOG2+1); + localparam bit VERBOSE = 0; + localparam int STALL_PROB = 25; + + // Define parameters for packet randomization + localparam int MIN_FFT_LEN_LOG2 = 3; // Same as Xilinx FFT core + localparam int MIN_FFT_LEN = 2**MIN_FFT_LEN_LOG2; + localparam int MAX_FFT_LEN = 2**MAX_FFT_LEN_LOG2; + + + //--------------------------------------------------------------------------- + // Clocks and Resets + //--------------------------------------------------------------------------- + + bit clk; + bit rst; + + sim_clock_gen #(.PERIOD(CLK_PERIOD), .AUTOSTART(0)) + clk_gen (.clk(clk), .rst(rst)); + + + //--------------------------------------------------------------------------- + // Bus Functional Models + //--------------------------------------------------------------------------- + + // Connections to DUT as interfaces: + AxiStreamIf #(DATA_W, CP_LEN_W) i_axis (clk, rst); + AxiStreamIf #(DATA_W, CP_LEN_W) o_axis (clk, rst); + + // AXI-Stream BFM + AxiStreamBfm #(DATA_W, CP_LEN_W) bfm = new(i_axis, o_axis); + + typedef AxiStreamBfm #(DATA_W, CP_LEN_W)::AxisPacket_t AxisPacket_t; + typedef AxiStreamBfm #(DATA_W, CP_LEN_W)::data_t data_t; + typedef AxiStreamBfm #(DATA_W, CP_LEN_W)::user_t user_t; + typedef AxisPacket_t AxisPacketQueue_t [$]; + + + //--------------------------------------------------------------------------- + // Device Under Test (DUT) + //--------------------------------------------------------------------------- + + logic fft_cfg_wr = 1'b0; + logic [FFT_LEN_LOG2_W-1:0] fft_len_log2; + fft_order_t fft_out_order; + + fft_reorder #( + .IN_FIFO_LOG2 (IN_FIFO_LOG2), + .OUT_FIFO_LOG2 (OUT_FIFO_LOG2), + .INPUT_ORDER (INPUT_ORDER), + .MAX_FFT_LEN_LOG2(MAX_FFT_LEN_LOG2), + .DATA_W (DATA_W), + .EN_CP_INSERTION (EN_CP_INSERTION) + ) fft_reorder_i ( + .clk (clk), + .rst (rst), + .fft_cfg_wr (fft_cfg_wr), + .fft_len_log2 (fft_len_log2), + .fft_out_order(fft_out_order), + .i_tdata (i_axis.tdata), + .i_tuser (i_axis.tuser), + .i_tlast (i_axis.tlast), + .i_tvalid (i_axis.tvalid), + .i_tready (i_axis.tready), + .o_tdata (o_axis.tdata), + .o_tlast (o_axis.tlast), + .o_tvalid (o_axis.tvalid), + .o_tready (o_axis.tready) + ); + + + //--------------------------------------------------------------------------- + // Helper Functions + //--------------------------------------------------------------------------- + + // Determine if the expected packet is the same to the actual packet + // received. Return 1 if they are equivalent, 0 if they differ. + function automatic bit check_pkt_equal(AxisPacket_t exp, AxisPacket_t act); + if (exp.data.size() != act.data.size()) begin + $display("Packets differ in size"); + $display("Exp: %0d words, Act: %0d words", exp.data.size(), act.data.size()); + return 0; + end + for (int i = 0; i < exp.data.size(); i++) begin + if (exp.data[i] != act.data[i]) begin + $display("Index %0d, expected 0x%X but received 0x%X", i, exp.data[i], act.data[i]); + return 0; + end + end + + // If it made it to here, all is well + return 1; + endfunction + + + // Generate an FFT packet having the indicated length, order, and start + // value. All data values are sequential. + function automatic AxisPacket_t gen_in_packet( + int len_log2, + fft_order_t order, + int start_value, + int cp_len = 0 + ); + AxisPacket_t pkt; + int data_count = start_value; + data_t data [] = new [2**len_log2]; + user_t user [] = new [1]; + + case (order) + NATURAL : begin + foreach (data[i]) data[i] = data_count++; + end + BIT_REVERSE : begin + foreach (data[i]) data[bit_reverse(i, len_log2)] = data_count++; + end + default : begin + `ASSERT_FATAL(0, "Invalid input FFT order"); + end + endcase + + user[0] = cp_len; + + pkt = new(); + pkt.data = data; + pkt.user = user; + return pkt; + endfunction : gen_in_packet + + + // Generate the expected output packet. + function automatic AxisPacket_t gen_out_packet( + int len_log2, + fft_order_t order, + int start_value, + int cp_len = 0 + ); + AxisPacket_t pkt; + int data_count = start_value; + int length = 2**len_log2; + data_t cp [] = new [cp_len]; + data_t data [] = new [length]; + + case (order) + NORMAL : begin + // 8 9 A B C D E F 0 1 2 3 4 5 6 7 + for (int i = length/2; i < length; i++) data[i] = data_count++; + for (int i = 0; i < length/2; i++) data[i] = data_count++; + end + REVERSE : begin + // 7 6 5 4 3 2 1 0 F E D C B A 9 8 + for (int i = length/2-1; i >= 0; i--) data[i] = data_count++; + for (int i = length-1; i >= length/2; i--) data[i] = data_count++; + end + NATURAL : begin + // 0 1 2 3 4 5 6 7 8 9 A B C D E F + foreach (data[i]) data[i] = data_count++; + end + BIT_REVERSE : begin + // 0 8 4 C 2 A 6 E 1 9 5 D 3 B 7 F + foreach (data[i]) data[bit_reverse(i, len_log2)] = data_count++; + end + default: begin + `ASSERT_FATAL(0, "Invalid input FFT order"); + end + endcase + + foreach (cp[i]) cp[i] = data[length-cp_len+i]; + + pkt = new(); + pkt.data = {cp, data}; + return pkt; + endfunction : gen_out_packet + + + // Run a test of the specific configuration. + // + // num_pkts : Number of test packets to test for this configuration + // len_log2 : FFT size to test + // order : FFT output order to test + // + task automatic test_packets( + int num_pkts, + int len_log2, + fft_order_t order, + int cp_insertions[] = {} + ); + int data_count; + int length = 2**len_log2; + int cp_idx; + AxisPacketQueue_t exp_q; + + if (VERBOSE) begin + $display("test_packets(): num_pkts = %0d, len_log2 = %0d, order = %s", + num_pkts, len_log2, order.name()); + end + + `ASSERT_FATAL(len_log2 >= MIN_FFT_LEN_LOG2 && len_log2 <= MAX_FFT_LEN_LOG2, + $sformatf("FFT length %0d is out of allowed range", 2**len_log2)); + + @(posedge clk); + fft_len_log2 <= len_log2; + fft_out_order <= order; + fft_cfg_wr <= 1'b1; + @(posedge clk); + fft_cfg_wr <= 1'b0; + @(posedge clk); + + repeat (num_pkts) begin + AxisPacket_t pkt, exp; + + // Generate test packet and expected output packet + pkt = gen_in_packet(len_log2, INPUT_ORDER, data_count, cp_insertions[cp_idx]); + exp = gen_out_packet(len_log2, order, data_count, cp_insertions[cp_idx]); + cp_idx++; + data_count += length; + + // Queue up the test packet to be sent + bfm.put(pkt); + + // Save the expected result + exp_q.push_back(exp); + end + + repeat (num_pkts) begin + AxisPacket_t act, exp; + bfm.get(act); + exp = exp_q.pop_front(); + + // Check if the received packet is equivalent to the expected packet + if (!check_pkt_equal(exp, act)) begin + $displayh("Expected: %p", exp.data); + $displayh("Received: %p", act.data); + `ASSERT_FATAL(0, "Received packet does not match expected packet"); + end + end + endtask : test_packets + + + //--------------------------------------------------------------------------- + // Tests + //--------------------------------------------------------------------------- + + task automatic test_orders(); + test.start_test($sformatf("Test %s to NORMAL", INPUT_ORDER.name())); + test_packets(2, 4, NORMAL); + test.end_test(); + + test.start_test($sformatf("Test %s to NATURAL", INPUT_ORDER.name())); + test_packets(2, 4, NATURAL); + test.end_test(); + + test.start_test($sformatf("Test %s to REVERSE", INPUT_ORDER.name())); + test_packets(2, 4, REVERSE); + test.end_test(); + + test.start_test($sformatf("Test %s to BIT_REVERSE", INPUT_ORDER.name())); + test_packets(2, 4, BIT_REVERSE); + test.end_test(); + endtask : test_orders + + + task automatic test_backpressure(); + test.start_test("Test full throttle"); + bfm.set_master_stall_prob(0); + bfm.set_slave_stall_prob(0); + test_packets(16, 4, NORMAL); + test.end_test(); + + test.start_test("Test overflow"); + bfm.set_master_stall_prob(10); + bfm.set_slave_stall_prob(90); + test_packets(16, 4, NORMAL); + test.end_test(); + + test.start_test("Test underflow"); + bfm.set_master_stall_prob(90); + bfm.set_slave_stall_prob(10); + test_packets(16, 4, NORMAL); + test.end_test(); + + // Restore default stall probability + bfm.set_master_stall_prob(STALL_PROB); + bfm.set_slave_stall_prob(STALL_PROB); + endtask : test_backpressure + + + task automatic test_random(int num_tests); + fft_order_t order; + int num_pkts; + int len_log2; + + test.start_test("Test random"); + + repeat (num_tests) begin + int cp_insertions []; + // Choose random parameters for this test + num_pkts = $urandom_range(1, 4); + len_log2 = $urandom_range($clog2(MIN_FFT_LEN), $clog2(MAX_FFT_LEN)); + cp_insertions = new [num_pkts]; + + // Use a CP for half of all packets groups + if ($urandom_range(1)) begin + foreach (cp_insertions[i]) cp_insertions[i] = $urandom_range(0, 2**len_log2-1); + end + + // Choose a random output order + order = fft_order_t'($urandom_range(order.num()-1)); + + test_packets(num_pkts, len_log2, order); + end + + test.end_test(); + endtask : test_random + + + // Test some basic cyclic prefix insertion/removal + task automatic test_cp_insertion(); + int cp_insertions[] = {0, 1, 2, 0}; + if (!EN_CP_INSERTION) return; + test.start_test("Test CP insertion"); + test_packets(cp_insertions.size(), 3, NATURAL, cp_insertions); + test.end_test(); + endtask + + + // Test min/max FFT and CP sizes + task automatic test_min_max(); + int min_insertion [2]; + int max_insertion [2]; + + test.start_test("Test min/max"); + + if (EN_CP_INSERTION) begin + min_insertion = {0, 1}; + max_insertion = {MAX_FFT_LEN-1, 0}; + end else begin + min_insertion = {0, 0}; + max_insertion = {0, 0}; + end + + test_packets(2, $clog2(MIN_FFT_LEN), NATURAL, min_insertion); + test_packets(2, $clog2(MAX_FFT_LEN), NATURAL, max_insertion); + test.end_test(); + endtask + + + //--------------------------------------------------------------------------- + // Main Test Process + //--------------------------------------------------------------------------- + + initial begin : tb_main + //string msg; + string tb_name; + tb_name = $sformatf( { + "fft_reorder_tb\n", + "IN_FIFO_LOG2 = %0d\n", + "OUT_FIFO_LOG2 = %0d\n", + "INPUT_ORDER = %s\n", + "MAX_FFT_LEN_LOG2 = %0d"}, + IN_FIFO_LOG2, OUT_FIFO_LOG2, INPUT_ORDER.name(), MAX_FFT_LEN_LOG2 + ); + test.start_tb(tb_name, 100ms); + + // Don't start the clocks until after start_tb() returns. This ensures that + // the clocks aren't toggling while other instances of this testbench are + // running, which speeds up simulation time. + clk_gen.start(); + + // Start the BFM + bfm.run(); + bfm.set_master_stall_prob(STALL_PROB); + bfm.set_slave_stall_prob(STALL_PROB); + + //-------------------------------- + // Reset + //-------------------------------- + + test.start_test("Reset", 10us); + clk_gen.reset(2); + clk_gen.clk_wait_f(3); + test.end_test(); + + //-------------------------------- + // Test Sequences + //-------------------------------- + + test_orders(); + test_cp_insertion(); + test_backpressure(); + test_min_max(); + + repeat (5) begin + clk_gen.reset(2); + clk_gen.clk_wait_f(3); + test_random(50); + end + + //-------------------------------- + // Finish Up + //-------------------------------- + + // End the TB, but don't $finish, since we don't want to kill other + // instances of this testbench that may be running. + test.end_tb(0); + // Kill the clocks to end this instance of the testbench + clk_gen.kill(); + end : tb_main + +endmodule : fft_reorder_tb + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/noc_shell_fft.v b/lib/rfnoc/blocks/rfnoc_block_fft/noc_shell_fft.v index 96e3dbd..3d22c16 100644 --- a/lib/rfnoc/blocks/rfnoc_block_fft/noc_shell_fft.v +++ b/lib/rfnoc/blocks/rfnoc_block_fft/noc_shell_fft.v @@ -1,5 +1,5 @@ // -// Copyright 2022 Ettus Research, a National Instruments Brand +// Copyright 2024 Ettus Research, a National Instruments Brand // // SPDX-License-Identifier: LGPL-3.0-or-later // @@ -7,7 +7,7 @@ // // Description: // -// This is a tool-generated NoC-shell for the fft block. +// This is a tool-generated NoC-shell for the FFT block. // See the RFNoC specification for more information about NoC shells. // // Parameters: @@ -22,13 +22,12 @@ module noc_shell_fft #( - parameter [9:0] THIS_PORTID = 10'd0, - parameter CHDR_W = 64, - parameter [5:0] MTU = 10, - parameter EN_MAGNITUDE_OUT = 0, - parameter EN_MAGNITUDE_APPROX_OUT = 1, - parameter EN_MAGNITUDE_SQ_OUT = 1, - parameter EN_FFT_SHIFT = 1 + parameter [9:0] THIS_PORTID = 10'd0, + parameter CHDR_W = 64, + parameter [5:0] MTU = 10, + parameter NUM_PORTS = 2, + parameter NIPC = 1, + parameter ITEM_W = 32 ) ( //--------------------- // Framework Interface @@ -49,15 +48,15 @@ module noc_shell_fft #( output wire [511:0] rfnoc_core_status, // AXIS-CHDR Input Ports (from framework) - input wire [(1)*CHDR_W-1:0] s_rfnoc_chdr_tdata, - input wire [(1)-1:0] s_rfnoc_chdr_tlast, - input wire [(1)-1:0] s_rfnoc_chdr_tvalid, - output wire [(1)-1:0] s_rfnoc_chdr_tready, + input wire [NUM_PORTS*CHDR_W-1:0] s_rfnoc_chdr_tdata, + input wire [NUM_PORTS-1:0] s_rfnoc_chdr_tlast, + input wire [NUM_PORTS-1:0] s_rfnoc_chdr_tvalid, + output wire [NUM_PORTS-1:0] s_rfnoc_chdr_tready, // AXIS-CHDR Output Ports (to framework) - output wire [(1)*CHDR_W-1:0] m_rfnoc_chdr_tdata, - output wire [(1)-1:0] m_rfnoc_chdr_tlast, - output wire [(1)-1:0] m_rfnoc_chdr_tvalid, - input wire [(1)-1:0] m_rfnoc_chdr_tready, + output wire [NUM_PORTS*CHDR_W-1:0] m_rfnoc_chdr_tdata, + output wire [NUM_PORTS-1:0] m_rfnoc_chdr_tlast, + output wire [NUM_PORTS-1:0] m_rfnoc_chdr_tvalid, + input wire [NUM_PORTS-1:0] m_rfnoc_chdr_tready, // AXIS-Ctrl Control Input Port (from framework) input wire [31:0] s_rfnoc_ctrl_tdata, @@ -85,33 +84,31 @@ module noc_shell_fft #( input wire m_ctrlport_resp_ack, input wire [31:0] m_ctrlport_resp_data, - // AXI-Stream Payload Context Clock and Reset + // AXI-Stream Data Clock and Reset output wire axis_data_clk, output wire axis_data_rst, - // Payload Stream to User Logic: in_0 - output wire [32*1-1:0] m_in_0_payload_tdata, - output wire [1-1:0] m_in_0_payload_tkeep, - output wire m_in_0_payload_tlast, - output wire m_in_0_payload_tvalid, - input wire m_in_0_payload_tready, - // Context Stream to User Logic: in_0 - output wire [CHDR_W-1:0] m_in_0_context_tdata, - output wire [3:0] m_in_0_context_tuser, - output wire m_in_0_context_tlast, - output wire m_in_0_context_tvalid, - input wire m_in_0_context_tready, - // Payload Stream from User Logic: out_0 - input wire [32*1-1:0] s_out_0_payload_tdata, - input wire [0:0] s_out_0_payload_tkeep, - input wire s_out_0_payload_tlast, - input wire s_out_0_payload_tvalid, - output wire s_out_0_payload_tready, - // Context Stream from User Logic: out_0 - input wire [CHDR_W-1:0] s_out_0_context_tdata, - input wire [3:0] s_out_0_context_tuser, - input wire s_out_0_context_tlast, - input wire s_out_0_context_tvalid, - output wire s_out_0_context_tready + // Data Stream to User Logic: in + output wire [NUM_PORTS*ITEM_W*NIPC-1:0] m_in_axis_tdata, + output wire [NUM_PORTS*NIPC-1:0] m_in_axis_tkeep, + output wire [NUM_PORTS-1:0] m_in_axis_tlast, + output wire [NUM_PORTS-1:0] m_in_axis_tvalid, + input wire [NUM_PORTS-1:0] m_in_axis_tready, + output wire [NUM_PORTS*64-1:0] m_in_axis_ttimestamp, + output wire [NUM_PORTS-1:0] m_in_axis_thas_time, + output wire [NUM_PORTS*16-1:0] m_in_axis_tlength, + output wire [NUM_PORTS-1:0] m_in_axis_teov, + output wire [NUM_PORTS-1:0] m_in_axis_teob, + // Data Stream from User Logic: out + input wire [NUM_PORTS*ITEM_W*NIPC-1:0] s_out_axis_tdata, + input wire [NUM_PORTS*NIPC-1:0] s_out_axis_tkeep, + input wire [NUM_PORTS-1:0] s_out_axis_tlast, + input wire [NUM_PORTS-1:0] s_out_axis_tvalid, + output wire [NUM_PORTS-1:0] s_out_axis_tready, + input wire [NUM_PORTS*64-1:0] s_out_axis_ttimestamp, + input wire [NUM_PORTS-1:0] s_out_axis_thas_time, + input wire [NUM_PORTS*16-1:0] s_out_axis_tlength, + input wire [NUM_PORTS-1:0] s_out_axis_teov, + input wire [NUM_PORTS-1:0] s_out_axis_teob ); //--------------------------------------------------------------------------- @@ -128,9 +125,9 @@ module noc_shell_fft #( wire [63:0] data_o_flush_done; backend_iface #( - .NOC_ID (32'hFF700000), - .NUM_DATA_I (1), - .NUM_DATA_O (1), + .NOC_ID (32'hFF700002), + .NUM_DATA_I (NUM_PORTS), + .NUM_DATA_O (NUM_PORTS), .CTRL_FIFOSIZE ($clog2(32)), .MTU (MTU) ) backend_iface_i ( @@ -230,77 +227,79 @@ module noc_shell_fft #( // Input Data Paths //--------------------- - chdr_to_axis_pyld_ctxt #( - .CHDR_W (CHDR_W), - .ITEM_W (32), - .NIPC (1), - .SYNC_CLKS (0), - .CONTEXT_FIFO_SIZE ($clog2(2)), - .PAYLOAD_FIFO_SIZE ($clog2(32)), - .CONTEXT_PREFETCH_EN (1) - ) chdr_to_axis_pyld_ctxt_in_in_0 ( - .axis_chdr_clk (rfnoc_chdr_clk), - .axis_chdr_rst (rfnoc_chdr_rst), - .axis_data_clk (axis_data_clk), - .axis_data_rst (axis_data_rst), - .s_axis_chdr_tdata (s_rfnoc_chdr_tdata[(0)*CHDR_W+:CHDR_W]), - .s_axis_chdr_tlast (s_rfnoc_chdr_tlast[0]), - .s_axis_chdr_tvalid (s_rfnoc_chdr_tvalid[0]), - .s_axis_chdr_tready (s_rfnoc_chdr_tready[0]), - .m_axis_payload_tdata (m_in_0_payload_tdata), - .m_axis_payload_tkeep (m_in_0_payload_tkeep), - .m_axis_payload_tlast (m_in_0_payload_tlast), - .m_axis_payload_tvalid (m_in_0_payload_tvalid), - .m_axis_payload_tready (m_in_0_payload_tready), - .m_axis_context_tdata (m_in_0_context_tdata), - .m_axis_context_tuser (m_in_0_context_tuser), - .m_axis_context_tlast (m_in_0_context_tlast), - .m_axis_context_tvalid (m_in_0_context_tvalid), - .m_axis_context_tready (m_in_0_context_tready), - .flush_en (data_i_flush_en), - .flush_timeout (data_i_flush_timeout), - .flush_active (data_i_flush_active[0]), - .flush_done (data_i_flush_done[0]) - ); + for (i = 0; i < NUM_PORTS; i = i + 1) begin: gen_input_in + chdr_to_axis_data #( + .CHDR_W (CHDR_W), + .ITEM_W (ITEM_W), + .NIPC (NIPC), + .SYNC_CLKS (0), + .INFO_FIFO_SIZE ($clog2(32)), + .PYLD_FIFO_SIZE ($clog2(32)) + ) chdr_to_axis_data_in_in ( + .axis_chdr_clk (rfnoc_chdr_clk), + .axis_chdr_rst (rfnoc_chdr_rst), + .axis_data_clk (axis_data_clk), + .axis_data_rst (axis_data_rst), + .s_axis_chdr_tdata (s_rfnoc_chdr_tdata[((0+i)*CHDR_W)+:CHDR_W]), + .s_axis_chdr_tlast (s_rfnoc_chdr_tlast[0+i]), + .s_axis_chdr_tvalid (s_rfnoc_chdr_tvalid[0+i]), + .s_axis_chdr_tready (s_rfnoc_chdr_tready[0+i]), + .m_axis_tdata (m_in_axis_tdata[(ITEM_W*NIPC)*i+:(ITEM_W*NIPC)]), + .m_axis_tkeep (m_in_axis_tkeep[NIPC*i+:NIPC]), + .m_axis_tlast (m_in_axis_tlast[i]), + .m_axis_tvalid (m_in_axis_tvalid[i]), + .m_axis_tready (m_in_axis_tready[i]), + .m_axis_ttimestamp (m_in_axis_ttimestamp[64*i+:64]), + .m_axis_thas_time (m_in_axis_thas_time[i]), + .m_axis_tlength (m_in_axis_tlength[16*i+:16]), + .m_axis_teov (m_in_axis_teov[i]), + .m_axis_teob (m_in_axis_teob[i]), + .flush_en (data_i_flush_en), + .flush_timeout (data_i_flush_timeout), + .flush_active (data_i_flush_active[0+i]), + .flush_done (data_i_flush_done[0+i]) + ); + end //--------------------- // Output Data Paths //--------------------- - axis_pyld_ctxt_to_chdr #( - .CHDR_W (CHDR_W), - .ITEM_W (32), - .NIPC (1), - .SYNC_CLKS (0), - .CONTEXT_FIFO_SIZE ($clog2(2)), - .PAYLOAD_FIFO_SIZE ($clog2(32)), - .MTU (MTU), - .CONTEXT_PREFETCH_EN (1) - ) axis_pyld_ctxt_to_chdr_out_out_0 ( - .axis_chdr_clk (rfnoc_chdr_clk), - .axis_chdr_rst (rfnoc_chdr_rst), - .axis_data_clk (axis_data_clk), - .axis_data_rst (axis_data_rst), - .m_axis_chdr_tdata (m_rfnoc_chdr_tdata[(0)*CHDR_W+:CHDR_W]), - .m_axis_chdr_tlast (m_rfnoc_chdr_tlast[0]), - .m_axis_chdr_tvalid (m_rfnoc_chdr_tvalid[0]), - .m_axis_chdr_tready (m_rfnoc_chdr_tready[0]), - .s_axis_payload_tdata (s_out_0_payload_tdata), - .s_axis_payload_tkeep (s_out_0_payload_tkeep), - .s_axis_payload_tlast (s_out_0_payload_tlast), - .s_axis_payload_tvalid (s_out_0_payload_tvalid), - .s_axis_payload_tready (s_out_0_payload_tready), - .s_axis_context_tdata (s_out_0_context_tdata), - .s_axis_context_tuser (s_out_0_context_tuser), - .s_axis_context_tlast (s_out_0_context_tlast), - .s_axis_context_tvalid (s_out_0_context_tvalid), - .s_axis_context_tready (s_out_0_context_tready), - .framer_errors (), - .flush_en (data_o_flush_en), - .flush_timeout (data_o_flush_timeout), - .flush_active (data_o_flush_active[0]), - .flush_done (data_o_flush_done[0]) - ); + for (i = 0; i < NUM_PORTS; i = i + 1) begin: gen_output_out + axis_data_to_chdr #( + .CHDR_W (CHDR_W), + .ITEM_W (ITEM_W), + .NIPC (NIPC), + .SYNC_CLKS (0), + .INFO_FIFO_SIZE ($clog2(32)), + .PYLD_FIFO_SIZE ($clog2(2**MTU)), + .MTU (MTU), + .SIDEBAND_AT_END (1) + ) axis_data_to_chdr_out_out ( + .axis_chdr_clk (rfnoc_chdr_clk), + .axis_chdr_rst (rfnoc_chdr_rst), + .axis_data_clk (axis_data_clk), + .axis_data_rst (axis_data_rst), + .m_axis_chdr_tdata (m_rfnoc_chdr_tdata[(0+i)*CHDR_W+:CHDR_W]), + .m_axis_chdr_tlast (m_rfnoc_chdr_tlast[0+i]), + .m_axis_chdr_tvalid (m_rfnoc_chdr_tvalid[0+i]), + .m_axis_chdr_tready (m_rfnoc_chdr_tready[0+i]), + .s_axis_tdata (s_out_axis_tdata[(ITEM_W*NIPC)*i+:(ITEM_W*NIPC)]), + .s_axis_tkeep (s_out_axis_tkeep[NIPC*i+:NIPC]), + .s_axis_tlast (s_out_axis_tlast[i]), + .s_axis_tvalid (s_out_axis_tvalid[i]), + .s_axis_tready (s_out_axis_tready[i]), + .s_axis_ttimestamp (s_out_axis_ttimestamp[64*i+:64]), + .s_axis_thas_time (s_out_axis_thas_time[i]), + .s_axis_tlength (s_out_axis_tlength[16*i+:16]), + .s_axis_teov (s_out_axis_teov[i]), + .s_axis_teob (s_out_axis_teob[i]), + .flush_en (data_o_flush_en), + .flush_timeout (data_o_flush_timeout), + .flush_active (data_o_flush_active[0+i]), + .flush_done (data_o_flush_done[0+i]) + ); + end endmodule // noc_shell_fft diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.sv b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.sv new file mode 100644 index 0000000..4dac8ba --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.sv @@ -0,0 +1,351 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: rfnoc_block_fft +// +// Description: +// +// RFNoC block for multichannel FFT/IFFT plus cyclic prefix insertion/removal. +// +// User Parameters: +// +// THIS_PORTID : Control crossbar port to which this block is connected +// CHDR_W : AXIS-CHDR data bus width +// MTU : Log2 of maximum transmission unit +// NUM_PORTS : Total number of FFT channels +// NUM_CORES : Number of individual cores to instantiate. +// Setting to 1 means all ports use a shared core +// and therefore all ports share the same control +// logic and all ports must be used simultaneously. +// Setting to NUM_PORTS means that each port will +// use its own core, and therefore each port can +// be configured and used independently. NUM_PORTS +// must be a multiple of NUM_CORES. +// MAX_FFT_SIZE_LOG2 : Log2 of maximum configurable FFT size. That is, +// the FFT size is exactly 2**fft_size_log2. +// MAX_CP_LIST_LEN_INS_LOG2 : Log2 of max length of cyclic prefix insertion +// list. Actual max is 2**MAX_CP_LIST_LEN_INS_LOG2. +// MAX_CP_LIST_LEN_REM_LOG2 : Log2 of max length of cyclic prefix removal +// list. Actual max is 2**MAX_CP_LIST_LEN_REM_LOG2. +// CP_INSERTION_REPEAT : Enable repeating the CP insertion list. When 1, +// the list repeats. When 0, CP insertion will +// stop when the list is finished. +// CP_REMOVAL_REPEAT : Enable repeating the CP removal list. When 1, +// the list repeats. When 0, CP removal will +// stop when the list is finished. +// EN_FFT_BYPASS : Controls whether to include the FFT bypass logic. +// EN_FFT_ORDER : Controls whether to include the FFT reorder logic. +// EN_MAGNITUDE : Controls whether to include the magnitude +// output calculation logic. +// EN_MAGNITUDE_SQ : Controls whether to include the +// magnitude-squared output calculation logic. +// USE_APPROX_MAG : Controls whether to use the low-resource +// approximate calculation (1) or the more exact +// and more resource-intensive calculation (0) for +// the magnitude calculation. +// + +`default_nettype none + + +module rfnoc_block_fft #( + logic [9:0] THIS_PORTID = 10'd0, + int CHDR_W = 64, + logic [5:0] MTU = 6'd10, + int NUM_PORTS = 1, + int NUM_CORES = 1, + int MAX_FFT_SIZE_LOG2 = 12, + int MAX_CP_LIST_LEN_INS_LOG2 = 5, + int MAX_CP_LIST_LEN_REM_LOG2 = 5, + bit CP_INSERTION_REPEAT = 1, + bit CP_REMOVAL_REPEAT = 1, + bit EN_FFT_BYPASS = 1, + bit EN_FFT_ORDER = 1, + bit EN_MAGNITUDE = 1, + bit EN_MAGNITUDE_SQ = 1, + bit USE_APPROX_MAG = 1 +) ( + // RFNoC Framework Clocks and Resets + input wire rfnoc_chdr_clk, + input wire rfnoc_ctrl_clk, + input wire ce_clk, + + // RFNoC Backend Interface + input wire [ 511:0] rfnoc_core_config, + output wire [ 511:0] rfnoc_core_status, + + // AXIS-CHDR Input Ports (from framework) + input wire [CHDR_W*NUM_PORTS-1:0] s_rfnoc_chdr_tdata, + input wire [ NUM_PORTS-1:0] s_rfnoc_chdr_tlast, + input wire [ NUM_PORTS-1:0] s_rfnoc_chdr_tvalid, + output wire [ NUM_PORTS-1:0] s_rfnoc_chdr_tready, + + // AXIS-CHDR Output Ports (to framework) + output wire [CHDR_W*NUM_PORTS-1:0] m_rfnoc_chdr_tdata, + output wire [ NUM_PORTS-1:0] m_rfnoc_chdr_tlast, + output wire [ NUM_PORTS-1:0] m_rfnoc_chdr_tvalid, + input wire [ NUM_PORTS-1:0] m_rfnoc_chdr_tready, + + // AXIS-Ctrl Input Port (from framework) + input wire [ 31:0] s_rfnoc_ctrl_tdata, + input wire s_rfnoc_ctrl_tlast, + input wire s_rfnoc_ctrl_tvalid, + output wire s_rfnoc_ctrl_tready, + + // AXIS-Ctrl Output Port (to framework) + output wire [ 31:0] m_rfnoc_ctrl_tdata, + output wire m_rfnoc_ctrl_tlast, + output wire m_rfnoc_ctrl_tvalid, + input wire m_rfnoc_ctrl_tready +); + + `include "usrp_utils.svh" + + import ctrlport_pkg::*; + import rfnoc_chdr_utils_pkg::*; + import fft_core_regs_pkg::FFT_CORE_ADDR_W; + + localparam ITEM_W = 32; + + + //--------------------------------------------------------------------------- + // Signal Declarations + //--------------------------------------------------------------------------- + + // Clocks and Resets + logic ce_rst; + + logic ctrlport_req_wr; + logic ctrlport_req_rd; + logic [CTRLPORT_ADDR_W-1:0] ctrlport_req_addr; + logic [CTRLPORT_DATA_W-1:0] ctrlport_req_data; + logic ctrlport_resp_ack; + logic [CTRLPORT_DATA_W-1:0] ctrlport_resp_data; + + logic [ ITEM_W*NUM_PORTS-1:0] in_axis_tdata; + logic [ NUM_PORTS-1:0] in_axis_tkeep; + logic [ NUM_PORTS-1:0] in_axis_tlast; + logic [ NUM_PORTS-1:0] in_axis_tvalid; + logic [ NUM_PORTS-1:0] in_axis_tready; + logic [CHDR_TIMESTAMP_W*NUM_PORTS-1:0] in_axis_ttimestamp; + logic [ NUM_PORTS-1:0] in_axis_thas_time; + logic [ CHDR_LENGTH_W*NUM_PORTS-1:0] in_axis_tlength; + logic [ NUM_PORTS-1:0] in_axis_teov; + logic [ NUM_PORTS-1:0] in_axis_teob; + + logic [ ITEM_W*NUM_PORTS-1:0] out_axis_tdata; + logic [ NUM_PORTS-1:0] out_axis_tkeep; + logic [ NUM_PORTS-1:0] out_axis_tlast; + logic [ NUM_PORTS-1:0] out_axis_tvalid; + logic [ NUM_PORTS-1:0] out_axis_tready; + logic [CHDR_TIMESTAMP_W*NUM_PORTS-1:0] out_axis_ttimestamp; + logic [ NUM_PORTS-1:0] out_axis_thas_time; + logic [ CHDR_LENGTH_W*NUM_PORTS-1:0] out_axis_tlength; + logic [ NUM_PORTS-1:0] out_axis_teov; + logic [ NUM_PORTS-1:0] out_axis_teob; + + + //--------------------------------------------------------------------------- + // NoC Shell + //--------------------------------------------------------------------------- + + noc_shell_fft #( + .CHDR_W (CHDR_W), + .THIS_PORTID(THIS_PORTID), + .MTU (MTU), + .NUM_PORTS (NUM_PORTS) + ) noc_shell_fft_i ( + //--------------------- + // Framework Interface + //--------------------- + // Clock Inputs + .rfnoc_chdr_clk (rfnoc_chdr_clk), + .rfnoc_ctrl_clk (rfnoc_ctrl_clk), + .ce_clk (ce_clk), + // Reset Outputs + .rfnoc_chdr_rst (), + .rfnoc_ctrl_rst (), + .ce_rst (ce_rst), + // RFNoC Backend Interface + .rfnoc_core_config (rfnoc_core_config), + .rfnoc_core_status (rfnoc_core_status), + // CHDR Input Ports (from framework) + .s_rfnoc_chdr_tdata (s_rfnoc_chdr_tdata), + .s_rfnoc_chdr_tlast (s_rfnoc_chdr_tlast), + .s_rfnoc_chdr_tvalid (s_rfnoc_chdr_tvalid), + .s_rfnoc_chdr_tready (s_rfnoc_chdr_tready), + // CHDR Output Ports (to framework) + .m_rfnoc_chdr_tdata (m_rfnoc_chdr_tdata), + .m_rfnoc_chdr_tlast (m_rfnoc_chdr_tlast), + .m_rfnoc_chdr_tvalid (m_rfnoc_chdr_tvalid), + .m_rfnoc_chdr_tready (m_rfnoc_chdr_tready), + // AXIS-Ctrl Input Port (from framework) + .s_rfnoc_ctrl_tdata (s_rfnoc_ctrl_tdata), + .s_rfnoc_ctrl_tlast (s_rfnoc_ctrl_tlast), + .s_rfnoc_ctrl_tvalid (s_rfnoc_ctrl_tvalid), + .s_rfnoc_ctrl_tready (s_rfnoc_ctrl_tready), + // AXIS-Ctrl Output Port (to framework) + .m_rfnoc_ctrl_tdata (m_rfnoc_ctrl_tdata), + .m_rfnoc_ctrl_tlast (m_rfnoc_ctrl_tlast), + .m_rfnoc_ctrl_tvalid (m_rfnoc_ctrl_tvalid), + .m_rfnoc_ctrl_tready (m_rfnoc_ctrl_tready), + //--------------------- + // Client Interface + //--------------------- + // CtrlPort Clock and Reset + .ctrlport_clk (), + .ctrlport_rst (), + // CtrlPort Master + .m_ctrlport_req_wr (ctrlport_req_wr), + .m_ctrlport_req_rd (ctrlport_req_rd), + .m_ctrlport_req_addr (ctrlport_req_addr), + .m_ctrlport_req_data (ctrlport_req_data), + .m_ctrlport_resp_ack (ctrlport_resp_ack), + .m_ctrlport_resp_data (ctrlport_resp_data), + // AXI-Stream Clock and Reset + .axis_data_clk (), + .axis_data_rst (), + // Data Stream to User Logic: in + .m_in_axis_tdata (in_axis_tdata), + .m_in_axis_tkeep (in_axis_tkeep), + .m_in_axis_tlast (in_axis_tlast), + .m_in_axis_tvalid (in_axis_tvalid), + .m_in_axis_tready (in_axis_tready), + .m_in_axis_ttimestamp (in_axis_ttimestamp), + .m_in_axis_thas_time (in_axis_thas_time), + .m_in_axis_tlength (in_axis_tlength), + .m_in_axis_teov (in_axis_teov), + .m_in_axis_teob (in_axis_teob), + // Data Stream from User Logic: out + .s_out_axis_tdata (out_axis_tdata), + .s_out_axis_tkeep (out_axis_tkeep), + .s_out_axis_tlast (out_axis_tlast), + .s_out_axis_tvalid (out_axis_tvalid), + .s_out_axis_tready (out_axis_tready), + .s_out_axis_ttimestamp(out_axis_ttimestamp), + .s_out_axis_thas_time (out_axis_thas_time), + .s_out_axis_tlength (out_axis_tlength), + .s_out_axis_teov (out_axis_teov), + .s_out_axis_teob (out_axis_teob) + ); + + + //--------------------------------------------------------------------------- + // CtrlPort Splitter + //--------------------------------------------------------------------------- + + wire [ NUM_CORES-1:0] dec_ctrlport_req_wr; + wire [ NUM_CORES-1:0] dec_ctrlport_req_rd; + wire [CTRLPORT_ADDR_W*NUM_CORES-1:0] dec_ctrlport_req_addr; + wire [CTRLPORT_DATA_W*NUM_CORES-1:0] dec_ctrlport_req_data; + wire [ NUM_CORES-1:0] dec_ctrlport_resp_ack; + wire [CTRLPORT_DATA_W*NUM_CORES-1:0] dec_ctrlport_resp_data; + + generate + if (NUM_CORES > 1) begin : gen_ctrlport_decoder + ctrlport_decoder #( + .NUM_SLAVES (NUM_CORES), + .BASE_ADDR (0), + .SLAVE_ADDR_W (FFT_CORE_ADDR_W) + ) ctrlport_decoder_i ( + .ctrlport_clk (ce_clk), + .ctrlport_rst (ce_rst), + .s_ctrlport_req_wr (ctrlport_req_wr), + .s_ctrlport_req_rd (ctrlport_req_rd), + .s_ctrlport_req_addr (ctrlport_req_addr), + .s_ctrlport_req_data (ctrlport_req_data), + .s_ctrlport_req_byte_en ('1), + .s_ctrlport_req_has_time ('0), + .s_ctrlport_req_time ('0), + .s_ctrlport_resp_ack (ctrlport_resp_ack), + .s_ctrlport_resp_status (), + .s_ctrlport_resp_data (ctrlport_resp_data), + .m_ctrlport_req_wr (dec_ctrlport_req_wr), + .m_ctrlport_req_rd (dec_ctrlport_req_rd), + .m_ctrlport_req_addr (dec_ctrlport_req_addr), + .m_ctrlport_req_data (dec_ctrlport_req_data), + .m_ctrlport_req_byte_en (), + .m_ctrlport_req_has_time (), + .m_ctrlport_req_time (), + .m_ctrlport_resp_ack (dec_ctrlport_resp_ack), + .m_ctrlport_resp_status ('0), + .m_ctrlport_resp_data (dec_ctrlport_resp_data) + ); + end else begin : gen_no_decoder + assign dec_ctrlport_req_wr = ctrlport_req_wr; + assign dec_ctrlport_req_rd = ctrlport_req_rd; + assign dec_ctrlport_req_addr = {{CTRLPORT_DATA_W-FFT_CORE_ADDR_W{1'b0}}, + ctrlport_req_addr[FFT_CORE_ADDR_W-1:0]}; + assign dec_ctrlport_req_data = ctrlport_req_data; + assign ctrlport_resp_ack = dec_ctrlport_resp_ack; + assign ctrlport_resp_data = dec_ctrlport_resp_data; + end + endgenerate + + + //--------------------------------------------------------------------------- + // FFT Core + //--------------------------------------------------------------------------- + + // Calculate the number of ports per core + localparam int NPPC = NUM_PORTS / NUM_CORES; + + if (NUM_CORES * NPPC != NUM_PORTS) begin : check_num_ports_per_core + // We require each FFT core instance to have the same number of channels. + ERROR__NUM_PORTS_must_be_a_multiple_of_NUM_CORES(); + end : check_num_ports_per_core + + genvar core_i; + + for (core_i = 0; core_i < NUM_CORES; core_i = core_i+1) begin : gen_fft_cores + fft_core #( + .NUM_CHAN (NPPC), + .NUM_CORES (NUM_CORES), + .MAX_FFT_SIZE_LOG2 (MAX_FFT_SIZE_LOG2), + .MAX_CP_LIST_LEN_INS_LOG2(MAX_CP_LIST_LEN_INS_LOG2), + .MAX_CP_LIST_LEN_REM_LOG2(MAX_CP_LIST_LEN_REM_LOG2), + .CP_INSERTION_REPEAT (CP_INSERTION_REPEAT), + .CP_REMOVAL_REPEAT (CP_REMOVAL_REPEAT), + .EN_FFT_BYPASS (EN_FFT_BYPASS), + .EN_FFT_ORDER (EN_FFT_ORDER), + .EN_MAGNITUDE (EN_MAGNITUDE), + .EN_MAGNITUDE_SQ (EN_MAGNITUDE_SQ), + .USE_APPROX_MAG (USE_APPROX_MAG) + ) fft_core_i ( + .ce_clk (ce_clk), + .ce_rst (ce_rst), + .s_ctrlport_req_wr (`BUS_I(dec_ctrlport_req_wr, 1, core_i)), + .s_ctrlport_req_rd (`BUS_I(dec_ctrlport_req_rd, 1, core_i)), + .s_ctrlport_req_addr (`BUS_I(dec_ctrlport_req_addr, CTRLPORT_ADDR_W, core_i)), + .s_ctrlport_req_data (`BUS_I(dec_ctrlport_req_data, CTRLPORT_DATA_W, core_i)), + .s_ctrlport_resp_ack (`BUS_I(dec_ctrlport_resp_ack, 1, core_i)), + .s_ctrlport_resp_data (`BUS_I(dec_ctrlport_resp_data, CTRLPORT_DATA_W, core_i)), + .s_in_axis_tdata (`BUS_I(in_axis_tdata, ITEM_W*NPPC, core_i)), + .s_in_axis_tkeep (`BUS_I(in_axis_tkeep, 1*NPPC, core_i)), + .s_in_axis_tlast (`BUS_I(in_axis_tlast, 1*NPPC, core_i)), + .s_in_axis_tvalid (`BUS_I(in_axis_tvalid, 1*NPPC, core_i)), + .s_in_axis_tready (`BUS_I(in_axis_tready, 1*NPPC, core_i)), + .s_in_axis_ttimestamp (`BUS_I(in_axis_ttimestamp, CHDR_TIMESTAMP_W*NPPC, core_i)), + .s_in_axis_thas_time (`BUS_I(in_axis_thas_time, 1*NPPC, core_i)), + .s_in_axis_tlength (`BUS_I(in_axis_tlength, CHDR_LENGTH_W*NPPC, core_i)), + .s_in_axis_teov (`BUS_I(in_axis_teov, 1*NPPC, core_i)), + .s_in_axis_teob (`BUS_I(in_axis_teob, 1*NPPC, core_i)), + .m_out_axis_tdata (`BUS_I(out_axis_tdata, ITEM_W*NPPC, core_i)), + .m_out_axis_tkeep (`BUS_I(out_axis_tkeep, 1*NPPC, core_i)), + .m_out_axis_tlast (`BUS_I(out_axis_tlast, 1*NPPC, core_i)), + .m_out_axis_tvalid (`BUS_I(out_axis_tvalid, 1*NPPC, core_i)), + .m_out_axis_tready (`BUS_I(out_axis_tready, 1*NPPC, core_i)), + .m_out_axis_ttimestamp(`BUS_I(out_axis_ttimestamp, CHDR_TIMESTAMP_W*NPPC, core_i)), + .m_out_axis_thas_time (`BUS_I(out_axis_thas_time, 1*NPPC, core_i)), + .m_out_axis_tlength (`BUS_I(out_axis_tlength, CHDR_LENGTH_W*NPPC, core_i)), + .m_out_axis_teov (`BUS_I(out_axis_teov, 1*NPPC, core_i)), + .m_out_axis_teob (`BUS_I(out_axis_teob, 1*NPPC, core_i)) + ); + end : gen_fft_cores + +endmodule : rfnoc_block_fft + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.v b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.v deleted file mode 100644 index 7096b26..0000000 --- a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft.v +++ /dev/null @@ -1,522 +0,0 @@ -// -// Copyright 2019 Ettus Research, a National Instruments Company -// -// SPDX-License-Identifier: LGPL-3.0-or-later -// -// Module: rfnoc_block_fft -// -// Description: An FFT block for RFNoC. -// -// Parameters: -// -// THIS_PORTID : Control crossbar port to which this block is connected -// CHDR_W : AXIS CHDR interface data width -// MTU : Maximum transmission unit (i.e., maximum packet size) in -// CHDR words is 2**MTU. -// EN_MAGNITUDE_OUT : CORDIC based magnitude calculation -// EN_MAGNITUDE_APPROX_OUT : Multipler-less, lower resource usage -// EN_MAGNITUDE_SQ_OUT : Magnitude squared -// EN_FFT_SHIFT : Center zero frequency bin -// - -module rfnoc_block_fft #( - parameter THIS_PORTID = 0, - parameter CHDR_W = 64, - parameter MTU = 10, - - parameter EN_MAGNITUDE_OUT = 0, - parameter EN_MAGNITUDE_APPROX_OUT = 1, - parameter EN_MAGNITUDE_SQ_OUT = 1, - parameter EN_FFT_SHIFT = 1 - ) -( - //--------------------------------------------------------------------------- - // AXIS CHDR Port - //--------------------------------------------------------------------------- - - input wire rfnoc_chdr_clk, - input wire ce_clk, - - // CHDR inputs from framework - input wire [CHDR_W-1:0] s_rfnoc_chdr_tdata, - input wire s_rfnoc_chdr_tlast, - input wire s_rfnoc_chdr_tvalid, - output wire s_rfnoc_chdr_tready, - - // CHDR outputs to framework - output wire [CHDR_W-1:0] m_rfnoc_chdr_tdata, - output wire m_rfnoc_chdr_tlast, - output wire m_rfnoc_chdr_tvalid, - input wire m_rfnoc_chdr_tready, - - // Backend interface - input wire [511:0] rfnoc_core_config, - output wire [511:0] rfnoc_core_status, - - //--------------------------------------------------------------------------- - // AXIS CTRL Port - //--------------------------------------------------------------------------- - - input wire rfnoc_ctrl_clk, - - // CTRL port requests from framework - input wire [31:0] s_rfnoc_ctrl_tdata, - input wire s_rfnoc_ctrl_tlast, - input wire s_rfnoc_ctrl_tvalid, - output wire s_rfnoc_ctrl_tready, - - // CTRL port requests to framework - output wire [31:0] m_rfnoc_ctrl_tdata, - output wire m_rfnoc_ctrl_tlast, - output wire m_rfnoc_ctrl_tvalid, - input wire m_rfnoc_ctrl_tready -); - - // These are the only supported values for now - localparam ITEM_W = 32; - localparam NIPC = 1; - - `include "../../core/rfnoc_axis_ctrl_utils.vh" - - //--------------------------------------------------------------------------- - // Signal Declarations - //--------------------------------------------------------------------------- - - wire ctrlport_req_wr; - wire ctrlport_req_rd; - wire [19:0] ctrlport_req_addr; - wire [31:0] ctrlport_req_data; - wire ctrlport_resp_ack; - wire [31:0] ctrlport_resp_data; - - wire [ITEM_W-1:0] axis_to_fft_tdata; - wire axis_to_fft_tlast; - wire axis_to_fft_tvalid; - wire axis_to_fft_tready; - - wire [ITEM_W-1:0] axis_from_fft_tdata; - wire axis_from_fft_tlast; - wire axis_from_fft_tvalid; - wire axis_from_fft_tready; - - wire [CHDR_W-1:0] m_axis_context_tdata; - wire [ 3:0] m_axis_context_tuser; - wire [ 0:0] m_axis_context_tlast; - wire [ 0:0] m_axis_context_tvalid; - wire [ 0:0] m_axis_context_tready; - - wire [CHDR_W-1:0] s_axis_context_tdata; - wire [ 3:0] s_axis_context_tuser; - wire [ 0:0] s_axis_context_tlast; - wire [ 0:0] s_axis_context_tvalid; - wire [ 0:0] s_axis_context_tready; - - wire ce_rst; - - - //--------------------------------------------------------------------------- - // NoC Shell - //--------------------------------------------------------------------------- - - noc_shell_fft #( - .THIS_PORTID (THIS_PORTID), - .CHDR_W (CHDR_W ), - .MTU (MTU ) - ) noc_shell_fft_i ( - .rfnoc_chdr_clk (rfnoc_chdr_clk ), - .rfnoc_ctrl_clk (rfnoc_ctrl_clk ), - .ce_clk (ce_clk ), - .rfnoc_chdr_rst ( ), - .rfnoc_ctrl_rst ( ), - .ce_rst (ce_rst ), - .rfnoc_core_config (rfnoc_core_config ), - .rfnoc_core_status (rfnoc_core_status ), - .s_rfnoc_chdr_tdata (s_rfnoc_chdr_tdata ), - .s_rfnoc_chdr_tlast (s_rfnoc_chdr_tlast ), - .s_rfnoc_chdr_tvalid (s_rfnoc_chdr_tvalid ), - .s_rfnoc_chdr_tready (s_rfnoc_chdr_tready ), - .m_rfnoc_chdr_tdata (m_rfnoc_chdr_tdata ), - .m_rfnoc_chdr_tlast (m_rfnoc_chdr_tlast ), - .m_rfnoc_chdr_tvalid (m_rfnoc_chdr_tvalid ), - .m_rfnoc_chdr_tready (m_rfnoc_chdr_tready ), - .s_rfnoc_ctrl_tdata (s_rfnoc_ctrl_tdata ), - .s_rfnoc_ctrl_tlast (s_rfnoc_ctrl_tlast ), - .s_rfnoc_ctrl_tvalid (s_rfnoc_ctrl_tvalid ), - .s_rfnoc_ctrl_tready (s_rfnoc_ctrl_tready ), - .m_rfnoc_ctrl_tdata (m_rfnoc_ctrl_tdata ), - .m_rfnoc_ctrl_tlast (m_rfnoc_ctrl_tlast ), - .m_rfnoc_ctrl_tvalid (m_rfnoc_ctrl_tvalid ), - .m_rfnoc_ctrl_tready (m_rfnoc_ctrl_tready ), - .ctrlport_clk ( ), - .ctrlport_rst ( ), - .m_ctrlport_req_wr (ctrlport_req_wr ), - .m_ctrlport_req_rd (ctrlport_req_rd ), - .m_ctrlport_req_addr (ctrlport_req_addr ), - .m_ctrlport_req_data (ctrlport_req_data ), - .m_ctrlport_resp_ack (ctrlport_resp_ack ), - .m_ctrlport_resp_data (ctrlport_resp_data ), - .axis_data_clk ( ), - .axis_data_rst ( ), - .m_in_0_payload_tdata (axis_to_fft_tdata ), - .m_in_0_payload_tkeep ( ), - .m_in_0_payload_tlast (axis_to_fft_tlast ), - .m_in_0_payload_tvalid (axis_to_fft_tvalid ), - .m_in_0_payload_tready (axis_to_fft_tready ), - .m_in_0_context_tdata (m_axis_context_tdata ), - .m_in_0_context_tuser (m_axis_context_tuser ), - .m_in_0_context_tlast (m_axis_context_tlast ), - .m_in_0_context_tvalid (m_axis_context_tvalid), - .m_in_0_context_tready (m_axis_context_tready), - .s_out_0_payload_tdata (axis_from_fft_tdata ), - .s_out_0_payload_tkeep ({1*NIPC{1'b1}} ), - .s_out_0_payload_tlast (axis_from_fft_tlast ), - .s_out_0_payload_tvalid (axis_from_fft_tvalid ), - .s_out_0_payload_tready (axis_from_fft_tready ), - .s_out_0_context_tdata (s_axis_context_tdata ), - .s_out_0_context_tuser (s_axis_context_tuser ), - .s_out_0_context_tlast (s_axis_context_tlast ), - .s_out_0_context_tvalid (s_axis_context_tvalid), - .s_out_0_context_tready (s_axis_context_tready) - ); - - // The input packets are the same configuration as the output packets, so - // just use the header information for each incoming to create the header for - // each outgoing packet. This is done by connecting m_axis_context to - // directly to s_axis_context. - assign s_axis_context_tdata = m_axis_context_tdata; - assign s_axis_context_tuser = m_axis_context_tuser; - assign s_axis_context_tlast = m_axis_context_tlast; - assign s_axis_context_tvalid = m_axis_context_tvalid; - assign m_axis_context_tready = s_axis_context_tready; - - wire [ 8-1:0] set_addr; - wire [32-1:0] set_data; - wire set_stb; - wire [ 8-1:0] rb_addr; - reg [64-1:0] rb_data; - - ctrlport_to_settings_bus # ( - .NUM_PORTS (1) - ) ctrlport_to_settings_bus_i ( - .ctrlport_clk (ce_clk), - .ctrlport_rst (ce_rst), - .s_ctrlport_req_wr (ctrlport_req_wr), - .s_ctrlport_req_rd (ctrlport_req_rd), - .s_ctrlport_req_addr (ctrlport_req_addr), - .s_ctrlport_req_data (ctrlport_req_data), - .s_ctrlport_req_has_time (1'b0), - .s_ctrlport_req_time (64'b0), - .s_ctrlport_resp_ack (ctrlport_resp_ack), - .s_ctrlport_resp_data (ctrlport_resp_data), - .set_data (set_data), - .set_addr (set_addr), - .set_stb (set_stb), - .set_time (), - .set_has_time (), - .rb_stb (1'b1), - .rb_addr (rb_addr), - .rb_data (rb_data)); - - localparam MAX_FFT_SIZE_LOG2 = 11; - - localparam [31:0] SR_FFT_RESET = 131; - localparam [31:0] SR_FFT_SIZE_LOG2 = 132; - localparam [31:0] SR_MAGNITUDE_OUT = 133; - localparam [31:0] SR_FFT_DIRECTION = 134; - localparam [31:0] SR_FFT_SCALING = 135; - localparam [31:0] SR_FFT_SHIFT_CONFIG = 136; - - localparam RB_FFT_RESET = 0; - localparam RB_MAGNITUDE_OUT = 1; - localparam RB_FFT_SIZE_LOG2 = 2; - localparam RB_FFT_DIRECTION = 3; - localparam RB_FFT_SCALING = 4; - localparam RB_FFT_SHIFT_CONFIG = 5; - - // FFT Output - localparam [1:0] COMPLEX_OUT = 0; - localparam [1:0] MAG_OUT = 1; - localparam [1:0] MAG_SQ_OUT = 2; - - // FFT Direction - localparam [0:0] FFT_REVERSE = 0; - localparam [0:0] FFT_FORWARD = 1; - - wire [1:0] magnitude_out; - wire [31:0] fft_data_o_tdata; - wire fft_data_o_tlast; - wire fft_data_o_tvalid; - wire fft_data_o_tready; - wire [15:0] fft_data_o_tuser; - wire [31:0] fft_shift_o_tdata; - wire fft_shift_o_tlast; - wire fft_shift_o_tvalid; - wire fft_shift_o_tready; - wire [31:0] fft_mag_i_tdata, fft_mag_o_tdata, fft_mag_o_tdata_int; - wire fft_mag_i_tlast, fft_mag_o_tlast; - wire fft_mag_i_tvalid, fft_mag_o_tvalid; - wire fft_mag_i_tready, fft_mag_o_tready; - wire [31:0] fft_mag_sq_i_tdata, fft_mag_sq_o_tdata; - wire fft_mag_sq_i_tlast, fft_mag_sq_o_tlast; - wire fft_mag_sq_i_tvalid, fft_mag_sq_o_tvalid; - wire fft_mag_sq_i_tready, fft_mag_sq_o_tready; - wire [31:0] fft_mag_round_i_tdata, fft_mag_round_o_tdata; - wire fft_mag_round_i_tlast, fft_mag_round_o_tlast; - wire fft_mag_round_i_tvalid, fft_mag_round_o_tvalid; - wire fft_mag_round_i_tready, fft_mag_round_o_tready; - - // Settings Registers - wire fft_reset; - setting_reg #( - .my_addr(SR_FFT_RESET), .awidth(8), .width(1)) - sr_fft_reset ( - .clk(ce_clk), .rst(ce_rst), - .strobe(set_stb), .addr(set_addr), .in(set_data), .out(fft_reset), .changed()); - - // Two instances of FFT size register, one for FFT core and one for FFT shift - localparam DEFAULT_FFT_SIZE = 8; // 256 - wire [7:0] fft_size_log2_tdata ,fft_core_size_log2_tdata; - wire fft_size_log2_tvalid, fft_core_size_log2_tvalid, fft_size_log2_tready, fft_core_size_log2_tready; - axi_setting_reg #( - .ADDR(SR_FFT_SIZE_LOG2), .AWIDTH(8), .WIDTH(8), .DATA_AT_RESET(DEFAULT_FFT_SIZE), .VALID_AT_RESET(1)) - sr_fft_size_log2 ( - .clk(ce_clk), .reset(ce_rst), - .set_stb(set_stb), .set_addr(set_addr), .set_data(set_data), - .o_tdata(fft_size_log2_tdata), .o_tlast(), .o_tvalid(fft_size_log2_tvalid), .o_tready(fft_size_log2_tready)); - - axi_setting_reg #( - .ADDR(SR_FFT_SIZE_LOG2), .AWIDTH(8), .WIDTH(8), .DATA_AT_RESET(DEFAULT_FFT_SIZE), .VALID_AT_RESET(1)) - sr_fft_size_log2_2 ( - .clk(ce_clk), .reset(ce_rst), - .set_stb(set_stb), .set_addr(set_addr), .set_data(set_data), - .o_tdata(fft_core_size_log2_tdata), .o_tlast(), .o_tvalid(fft_core_size_log2_tvalid), .o_tready(fft_core_size_log2_tready)); - - localparam DEFAULT_FFT_DIRECTION = FFT_FORWARD; - wire fft_direction_tdata; - wire fft_direction_tvalid, fft_direction_tready; - axi_setting_reg #( - .ADDR(SR_FFT_DIRECTION), .AWIDTH(8), .WIDTH(1), .DATA_AT_RESET(DEFAULT_FFT_DIRECTION), .VALID_AT_RESET(1)) - sr_fft_direction ( - .clk(ce_clk), .reset(ce_rst), - .set_stb(set_stb), .set_addr(set_addr), .set_data(set_data), - .o_tdata(fft_direction_tdata), .o_tlast(), .o_tvalid(fft_direction_tvalid), .o_tready(fft_direction_tready)); - - localparam [11:0] DEFAULT_FFT_SCALING = 12'b011010101010; // Conservative 1/N scaling - wire [11:0] fft_scaling_tdata; - wire fft_scaling_tvalid, fft_scaling_tready; - axi_setting_reg #( - .ADDR(SR_FFT_SCALING), .AWIDTH(8), .WIDTH(12), .DATA_AT_RESET(DEFAULT_FFT_SCALING), .VALID_AT_RESET(1)) - sr_fft_scaling ( - .clk(ce_clk), .reset(ce_rst), - .set_stb(set_stb), .set_addr(set_addr), .set_data(set_data), - .o_tdata(fft_scaling_tdata), .o_tlast(), .o_tvalid(fft_scaling_tvalid), .o_tready(fft_scaling_tready)); - - wire [1:0] fft_shift_config_tdata; - wire fft_shift_config_tvalid, fft_shift_config_tready; - axi_setting_reg #( - .ADDR(SR_FFT_SHIFT_CONFIG), .AWIDTH(8), .WIDTH(2)) - sr_fft_shift_config ( - .clk(ce_clk), .reset(ce_rst), - .set_stb(set_stb), .set_addr(set_addr), .set_data(set_data), - .o_tdata(fft_shift_config_tdata), .o_tlast(), .o_tvalid(fft_shift_config_tvalid), .o_tready(fft_shift_config_tready)); - - // Synchronize writing configuration to the FFT core - reg fft_config_ready; - wire fft_config_write = fft_config_ready & axis_to_fft_tvalid & axis_to_fft_tready; - always @(posedge ce_clk) begin - if (ce_rst | fft_reset) begin - fft_config_ready <= 1'b1; - end else begin - if (fft_config_write) begin - fft_config_ready <= 1'b0; - end else if (axis_to_fft_tlast) begin - fft_config_ready <= 1'b1; - end - end - end - - wire [23:0] fft_config_tdata = {3'd0, fft_scaling_tdata, fft_direction_tdata, fft_core_size_log2_tdata}; - wire fft_config_tvalid = fft_config_write & (fft_scaling_tvalid | fft_direction_tvalid | fft_core_size_log2_tvalid); - wire fft_config_tready; - assign fft_core_size_log2_tready = fft_config_tready & fft_config_write; - assign fft_direction_tready = fft_config_tready & fft_config_write; - assign fft_scaling_tready = fft_config_tready & fft_config_write; - axi_fft inst_axi_fft ( - .aclk(ce_clk), .aresetn(~(fft_reset)), - .s_axis_data_tvalid(axis_to_fft_tvalid), - .s_axis_data_tready(axis_to_fft_tready), - .s_axis_data_tlast(axis_to_fft_tlast), - .s_axis_data_tdata({axis_to_fft_tdata[15:0],axis_to_fft_tdata[31:16]}), - .m_axis_data_tvalid(fft_data_o_tvalid), - .m_axis_data_tready(fft_data_o_tready), - .m_axis_data_tlast(fft_data_o_tlast), - .m_axis_data_tdata({fft_data_o_tdata[15:0],fft_data_o_tdata[31:16]}), - .m_axis_data_tuser(fft_data_o_tuser), // FFT index - .s_axis_config_tdata(fft_config_tdata), - .s_axis_config_tvalid(fft_config_tvalid), - .s_axis_config_tready(fft_config_tready), - .event_frame_started(), - .event_tlast_unexpected(), - .event_tlast_missing(), - .event_status_channel_halt(), - .event_data_in_channel_halt(), - .event_data_out_channel_halt()); - - // Mux control signals - assign fft_shift_o_tready = (magnitude_out == MAG_OUT) ? fft_mag_i_tready : - (magnitude_out == MAG_SQ_OUT) ? fft_mag_sq_i_tready : axis_from_fft_tready; - assign fft_mag_i_tvalid = (magnitude_out == MAG_OUT) ? fft_shift_o_tvalid : 1'b0; - assign fft_mag_i_tlast = (magnitude_out == MAG_OUT) ? fft_shift_o_tlast : 1'b0; - assign fft_mag_i_tdata = fft_shift_o_tdata; - assign fft_mag_o_tready = (magnitude_out == MAG_OUT) ? fft_mag_round_i_tready : 1'b0; - assign fft_mag_sq_i_tvalid = (magnitude_out == MAG_SQ_OUT) ? fft_shift_o_tvalid : 1'b0; - assign fft_mag_sq_i_tlast = (magnitude_out == MAG_SQ_OUT) ? fft_shift_o_tlast : 1'b0; - assign fft_mag_sq_i_tdata = fft_shift_o_tdata; - assign fft_mag_sq_o_tready = (magnitude_out == MAG_SQ_OUT) ? fft_mag_round_i_tready : 1'b0; - assign fft_mag_round_i_tvalid = (magnitude_out == MAG_OUT) ? fft_mag_o_tvalid : - (magnitude_out == MAG_SQ_OUT) ? fft_mag_sq_o_tvalid : 1'b0; - assign fft_mag_round_i_tlast = (magnitude_out == MAG_OUT) ? fft_mag_o_tlast : - (magnitude_out == MAG_SQ_OUT) ? fft_mag_sq_o_tlast : 1'b0; - assign fft_mag_round_i_tdata = (magnitude_out == MAG_OUT) ? fft_mag_o_tdata : fft_mag_sq_o_tdata; - assign fft_mag_round_o_tready = axis_from_fft_tready; - assign axis_from_fft_tvalid = (magnitude_out == MAG_OUT | magnitude_out == MAG_SQ_OUT) ? fft_mag_round_o_tvalid : fft_shift_o_tvalid; - assign axis_from_fft_tlast = (magnitude_out == MAG_OUT | magnitude_out == MAG_SQ_OUT) ? fft_mag_round_o_tlast : fft_shift_o_tlast; - assign axis_from_fft_tdata = (magnitude_out == MAG_OUT | magnitude_out == MAG_SQ_OUT) ? fft_mag_round_o_tdata : fft_shift_o_tdata; - - // Conditionally synth magnitude / magnitude^2 logic - generate - if (EN_MAGNITUDE_OUT | EN_MAGNITUDE_APPROX_OUT | EN_MAGNITUDE_SQ_OUT) begin : generate_magnitude_out - setting_reg #( - .my_addr(SR_MAGNITUDE_OUT), .awidth(8), .width(2)) - sr_magnitude_out ( - .clk(ce_clk), .rst(ce_rst), - .strobe(set_stb), .addr(set_addr), .in(set_data), .out(magnitude_out), .changed()); - end else begin : generate_magnitude_out_else - // Magnitude calculation logic not included, so always bypass - assign magnitude_out = 2'd0; - end - - if (EN_FFT_SHIFT) begin : generate_fft_shift - fft_shift #( - .MAX_FFT_SIZE_LOG2(MAX_FFT_SIZE_LOG2), - .WIDTH(32)) - inst_fft_shift ( - .clk(ce_clk), .reset(ce_rst | fft_reset), - .config_tdata(fft_shift_config_tdata), - .config_tvalid(fft_shift_config_tvalid), - .config_tready(fft_shift_config_tready), - .fft_size_log2_tdata(fft_size_log2_tdata[$clog2(MAX_FFT_SIZE_LOG2)-1:0]), - .fft_size_log2_tvalid(fft_size_log2_tvalid), - .fft_size_log2_tready(fft_size_log2_tready), - .i_tdata(fft_data_o_tdata), - .i_tlast(fft_data_o_tlast), - .i_tvalid(fft_data_o_tvalid), - .i_tready(fft_data_o_tready), - .i_tuser(fft_data_o_tuser[MAX_FFT_SIZE_LOG2-1:0]), - .o_tdata(fft_shift_o_tdata), - .o_tlast(fft_shift_o_tlast), - .o_tvalid(fft_shift_o_tvalid), - .o_tready(fft_shift_o_tready)); - end - else begin : generate_fft_shift_else - assign fft_shift_o_tdata = fft_data_o_tdata; - assign fft_shift_o_tlast = fft_data_o_tlast; - assign fft_shift_o_tvalid = fft_data_o_tvalid; - assign fft_data_o_tready = fft_shift_o_tready; - end - - // More accurate magnitude calculation takes precedence if enabled - if (EN_MAGNITUDE_OUT) begin : generate_complex_to_magphase - complex_to_magphase - inst_complex_to_magphase ( - .aclk(ce_clk), .aresetn(~(ce_rst | fft_reset)), - .s_axis_cartesian_tvalid(fft_mag_i_tvalid), - .s_axis_cartesian_tlast(fft_mag_i_tlast), - .s_axis_cartesian_tready(fft_mag_i_tready), - .s_axis_cartesian_tdata(fft_mag_i_tdata), - .m_axis_dout_tvalid(fft_mag_o_tvalid), - .m_axis_dout_tlast(fft_mag_o_tlast), - .m_axis_dout_tdata(fft_mag_o_tdata_int), - .m_axis_dout_tready(fft_mag_o_tready)); - assign fft_mag_o_tdata = {1'b0, fft_mag_o_tdata_int[15:0], 15'd0}; - end - else if (EN_MAGNITUDE_APPROX_OUT) begin : generate_complex_to_mag_approx - complex_to_mag_approx - inst_complex_to_mag_approx ( - .clk(ce_clk), .reset(ce_rst | fft_reset), .clear(1'b0), - .i_tvalid(fft_mag_i_tvalid), - .i_tlast(fft_mag_i_tlast), - .i_tready(fft_mag_i_tready), - .i_tdata(fft_mag_i_tdata), - .o_tvalid(fft_mag_o_tvalid), - .o_tlast(fft_mag_o_tlast), - .o_tready(fft_mag_o_tready), - .o_tdata(fft_mag_o_tdata_int[15:0])); - assign fft_mag_o_tdata = {1'b0, fft_mag_o_tdata_int[15:0], 15'd0}; - end - else begin : generate_complex_to_mag_approx_else - assign fft_mag_o_tdata = fft_mag_i_tdata; - assign fft_mag_o_tlast = fft_mag_i_tlast; - assign fft_mag_o_tvalid = fft_mag_i_tvalid; - assign fft_mag_i_tready = fft_mag_o_tready; - end - - if (EN_MAGNITUDE_SQ_OUT) begin : generate_complex_to_magsq - complex_to_magsq - inst_complex_to_magsq ( - .clk(ce_clk), .reset(ce_rst | fft_reset), .clear(1'b0), - .i_tvalid(fft_mag_sq_i_tvalid), - .i_tlast(fft_mag_sq_i_tlast), - .i_tready(fft_mag_sq_i_tready), - .i_tdata(fft_mag_sq_i_tdata), - .o_tvalid(fft_mag_sq_o_tvalid), - .o_tlast(fft_mag_sq_o_tlast), - .o_tready(fft_mag_sq_o_tready), - .o_tdata(fft_mag_sq_o_tdata)); - end - else begin : generate_complex_to_magsq_else - assign fft_mag_sq_o_tdata = fft_mag_sq_i_tdata; - assign fft_mag_sq_o_tlast = fft_mag_sq_i_tlast; - assign fft_mag_sq_o_tvalid = fft_mag_sq_i_tvalid; - assign fft_mag_sq_i_tready = fft_mag_sq_o_tready; - end - - // Convert to SC16 - if (EN_MAGNITUDE_OUT | EN_MAGNITUDE_APPROX_OUT | EN_MAGNITUDE_SQ_OUT) begin : generate_axi_round_and_clip - axi_round_and_clip #( - .WIDTH_IN(32), - .WIDTH_OUT(16), - .CLIP_BITS(1)) - inst_axi_round_and_clip ( - .clk(ce_clk), .reset(ce_rst | fft_reset), - .i_tdata(fft_mag_round_i_tdata), - .i_tlast(fft_mag_round_i_tlast), - .i_tvalid(fft_mag_round_i_tvalid), - .i_tready(fft_mag_round_i_tready), - .o_tdata(fft_mag_round_o_tdata[31:16]), - .o_tlast(fft_mag_round_o_tlast), - .o_tvalid(fft_mag_round_o_tvalid), - .o_tready(fft_mag_round_o_tready)); - assign fft_mag_round_o_tdata[15:0] = {16{16'd0}}; - end - else begin : generate_axi_round_and_clip_else - assign fft_mag_round_o_tdata = fft_mag_round_i_tdata; - assign fft_mag_round_o_tlast = fft_mag_round_i_tlast; - assign fft_mag_round_o_tvalid = fft_mag_round_i_tvalid; - assign fft_mag_round_i_tready = fft_mag_round_o_tready; - end - endgenerate - - // Readback registers - always @* - case(rb_addr) - RB_FFT_RESET : rb_data <= {63'd0, fft_reset}; - RB_MAGNITUDE_OUT : rb_data <= {62'd0, magnitude_out}; - RB_FFT_SIZE_LOG2 : rb_data <= {fft_size_log2_tdata}; - RB_FFT_DIRECTION : rb_data <= {63'd0, fft_direction_tdata}; - RB_FFT_SCALING : rb_data <= {52'd0, fft_scaling_tdata}; - RB_FFT_SHIFT_CONFIG : rb_data <= {62'd0, fft_shift_config_tdata}; - default : rb_data <= 64'h0BADC0DE0BADC0DE; - endcase - -endmodule diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_all_tb.sv b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_all_tb.sv new file mode 100644 index 0000000..d9c7a54 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_all_tb.sv @@ -0,0 +1,52 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: rfnoc_block_fft_all_tb +// +// Description: +// +// This is the testbench for rfnoc_block_fft that instantiates several +// variations of the testbench to test different configurations. +// + + +module rfnoc_block_fft_all_tb; + + //--------------------------------------------------------------------------- + // Test Configurations + //--------------------------------------------------------------------------- + + // Basic tests of multi-ports configurations + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(2), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(10)) tb_0a (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(2), .NUM_CORES(2), .MAX_FFT_SIZE_LOG2(10)) tb_0b (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(4), .NUM_CORES(2), .MAX_FFT_SIZE_LOG2(10)) tb_0c (); + + // Basic tests of other FFT sizes + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(11)) tb_1a (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(12)) tb_1b (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(13)) tb_1c (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(14)) tb_1d (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(15)) tb_1e (); + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(16)) tb_1f (); + + // Test case where USE_APPROX_MAG = 1 + rfnoc_block_fft_tb #(.FULL_TEST(0), .NUM_PORTS(1), .NUM_CORES(1), .MAX_FFT_SIZE_LOG2(10), + .EN_FFT_BYPASS(1), .EN_MAGNITUDE(1), .USE_APPROX_MAG(1)) tb_2a (); + + // Run full suite of tests on 1k FFT configuration + rfnoc_block_fft_tb #( + .FULL_TEST (1 ), + .NUM_PORTS (1 ), + .NUM_CORES (1 ), + .MAX_FFT_SIZE_LOG2 (10), + .MAX_CP_LIST_LEN_INS_LOG2(5 ), + .MAX_CP_LIST_LEN_REM_LOG2(5 ), + .EN_MAGNITUDE_SQ (1 ), + .EN_MAGNITUDE (1 ), + .EN_FFT_BYPASS (1 ), + .USE_APPROX_MAG (0 ) + ) tb_3a (); + +endmodule : rfnoc_block_fft_all_tb diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_tb.sv b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_tb.sv index b243ff2..f738f16 100644 --- a/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_tb.sv +++ b/lib/rfnoc/blocks/rfnoc_block_fft/rfnoc_block_fft_tb.sv @@ -1,424 +1,1306 @@ // -// Copyright 2019 Ettus Research, a National Instruments Company +// Copyright 2024 Ettus Research, a National Instruments Brand // // SPDX-License-Identifier: LGPL-3.0-or-later // // Module: rfnoc_block_fft_tb // -// Description: Testbench for rfnoc_block_fft +// Description: Testbench for the FFT RFNoC block. // -module rfnoc_block_fft_tb(); +`default_nettype none + + +module rfnoc_block_fft_tb #( + bit FULL_TEST = 1, + int NUM_PORTS = 1, + int NUM_CORES = 1, + int MAX_FFT_SIZE_LOG2 = 10, + int MAX_CP_LIST_LEN_INS_LOG2 = 5, + int MAX_CP_LIST_LEN_REM_LOG2 = 5, + bit EN_MAGNITUDE_SQ = 1, + bit EN_MAGNITUDE = 1, + bit EN_FFT_ORDER = 1, + bit EN_FFT_BYPASS = 1, + bit USE_APPROX_MAG = 0, + bit VERBOSE = 0 +); - // Include macros and time declarations for use with PkgTestExec `include "test_exec.svh" + `include "usrp_utils.svh" import PkgTestExec::*; import PkgChdrUtils::*; import PkgRfnocBlockCtrlBfm::*; + import PkgRfnocItemUtils::*; + + import PkgMath::*; + + // Import register descriptions + import fft_core_regs_pkg::*; + + // Import Xilinx FFT helper functions + import xfft_config_pkg::*; + + // FFT reorder constants + import fft_reorder_pkg::*; + //--------------------------------------------------------------------------- - // Local Parameters + // Testbench Configuration //--------------------------------------------------------------------------- - // Simulation parameters - localparam real CHDR_CLK_PER = 5.0; // Clock rate - localparam int SPP = 256; // Samples per packet - localparam int PKT_SIZE_BYTES = SPP*4; // Bytes per packet - localparam int STALL_PROB = 25; // BFM stall probability - // Block configuration - localparam int NOC_ID = 32'hFF70_0000; - localparam int CHDR_W = 64; - localparam int ITEM_W = 32; - localparam int THIS_PORTID = 'h123; - localparam int MTU = 10; - localparam int NUM_PORTS = 1; - localparam int NUM_HB = 3; - localparam int CIC_MAX_DECIM = 255; + localparam int NUM_CHAN_PER_CORE = NUM_PORTS / NUM_CORES; + localparam int MAX_CP_LIST_LEN_INS = 2**MAX_CP_LIST_LEN_INS_LOG2-1; + localparam int MAX_CP_LIST_LEN_REM = 2**MAX_CP_LIST_LEN_REM_LOG2-1; + localparam int CP_INSERTION_REPEAT = 1; + localparam int CP_REMOVAL_REPEAT = 1; + localparam int MAX_FFT_SIZE = 2**MAX_FFT_SIZE_LOG2; + localparam int MAX_CP_LEN_LOG2 = MAX_FFT_SIZE_LOG2; + localparam int MAX_CP_LEN = 2**MAX_CP_LEN_LOG2-1; + localparam int MIN_FFT_SIZE_LOG2 = 3; // Minimum allowed by Xilinx FFT core + localparam int MIN_FFT_SIZE = 2**MIN_FFT_SIZE_LOG2; + localparam int FFT_SCALING = fft_scale_default(MAX_FFT_SIZE_LOG2); + + // RFNoC configuration + localparam [9:0] THIS_PORTID = 10'h123; + localparam int CHDR_W = 64; // CHDR size in bits + localparam int MTU = 10; // Log2 of max transmission unit in CHDR words + localparam int NUM_PORTS_I = NUM_PORTS; + localparam int NUM_PORTS_O = NUM_PORTS; + localparam int ITEM_W = 32; // Sample size in bits + localparam int SPP = 64; // Samples per packet. Must be a power of 2 for FFT. + localparam int PKT_SIZE_BYTES = SPP * (ITEM_W/8); + localparam int STALL_PROB = 50; // Default BFM stall probability + localparam real CHDR_CLK_PER = 5.0; // 200 MHz + localparam real CTRL_CLK_PER = 8.0; // 125 MHz + localparam real CE_CLK_PER = 4.0; // 250 MHz - // FFT specific settings - // FFT settings - localparam [31:0] FFT_SIZE = 256; - localparam [31:0] FFT_SIZE_LOG2 = $clog2(FFT_SIZE); - localparam [31:0] FFT_SCALING = 12'b011010101010; // Conservative scaling of 1/N - localparam [31:0] FFT_SHIFT_CONFIG = 0; // Normal FFT shift - localparam FFT_BIN = FFT_SIZE/8 + FFT_SIZE/2; // 1/8 sample rate freq + FFT shift - localparam NUM_ITERATIONS = 10; //--------------------------------------------------------------------------- - // Clocks + // Clocks and Resets //--------------------------------------------------------------------------- bit rfnoc_chdr_clk; bit rfnoc_ctrl_clk; + bit ce_clk; + + sim_clock_gen #(.PERIOD(CHDR_CLK_PER), .AUTOSTART(0)) + rfnoc_chdr_clk_gen (.clk(rfnoc_chdr_clk), .rst()); + sim_clock_gen #(.PERIOD(CTRL_CLK_PER), .AUTOSTART(0)) + rfnoc_ctrl_clk_gen (.clk(rfnoc_ctrl_clk), .rst()); + sim_clock_gen #(.PERIOD(CE_CLK_PER), .AUTOSTART(0)) + ce_clk_gen (.clk(ce_clk), .rst()); - sim_clock_gen #(CHDR_CLK_PER) rfnoc_chdr_clk_gen (.clk(rfnoc_chdr_clk), .rst()); - sim_clock_gen #(CHDR_CLK_PER) rfnoc_ctrl_clk_gen (.clk(rfnoc_ctrl_clk), .rst()); //--------------------------------------------------------------------------- // Bus Functional Models //--------------------------------------------------------------------------- + // Backend Interface + RfnocBackendIf backend (rfnoc_chdr_clk, rfnoc_ctrl_clk); + + // AXIS-Ctrl Interface + AxiStreamIf #(32) m_ctrl (rfnoc_ctrl_clk, 1'b0); + AxiStreamIf #(32) s_ctrl (rfnoc_ctrl_clk, 1'b0); + + // AXIS-CHDR Interfaces + AxiStreamIf #(CHDR_W) m_chdr [NUM_PORTS_I] (rfnoc_chdr_clk, 1'b0); + AxiStreamIf #(CHDR_W) s_chdr [NUM_PORTS_O] (rfnoc_chdr_clk, 1'b0); + + // Block Controller BFM + RfnocBlockCtrlBfm #(CHDR_W, ITEM_W) blk_ctrl = new(backend, m_ctrl, s_ctrl); + + // CHDR word and item/sample data types typedef ChdrData #(CHDR_W, ITEM_W)::chdr_word_t chdr_word_t; typedef ChdrData #(CHDR_W, ITEM_W)::item_t item_t; - RfnocBackendIf backend (rfnoc_chdr_clk, rfnoc_ctrl_clk); - AxiStreamIf #(32) m_ctrl (rfnoc_ctrl_clk, 1'b0); - AxiStreamIf #(32) s_ctrl (rfnoc_ctrl_clk, 1'b0); - AxiStreamIf #(CHDR_W) m_chdr (rfnoc_chdr_clk, 1'b0); - AxiStreamIf #(CHDR_W) s_chdr (rfnoc_chdr_clk, 1'b0); - - // Bus functional model for a software block controller - RfnocBlockCtrlBfm #(CHDR_W, ITEM_W) blk_ctrl = - new(backend, m_ctrl, s_ctrl); + typedef item_t item_queue_t[$]; // Connect block controller to BFMs - initial begin - blk_ctrl.connect_master_data_port(0, m_chdr, PKT_SIZE_BYTES); - blk_ctrl.connect_slave_data_port(0, s_chdr); - blk_ctrl.set_master_stall_prob(0, STALL_PROB); - blk_ctrl.set_slave_stall_prob(0, STALL_PROB); + for (genvar i = 0; i < NUM_PORTS_I; i++) begin : gen_bfm_input_connections + initial begin + blk_ctrl.connect_master_data_port(i, m_chdr[i], PKT_SIZE_BYTES); + blk_ctrl.set_master_stall_prob(i, STALL_PROB); + end + end + for (genvar i = 0; i < NUM_PORTS_O; i++) begin : gen_bfm_output_connections + initial begin + blk_ctrl.connect_slave_data_port(i, s_chdr[i]); + blk_ctrl.set_slave_stall_prob(i, STALL_PROB); + end end + //--------------------------------------------------------------------------- - // DUT + // Device Under Test (DUT) //--------------------------------------------------------------------------- + // DUT Slave (Input) Port Signals + logic [CHDR_W*NUM_PORTS_I-1:0] s_rfnoc_chdr_tdata; + logic [ NUM_PORTS_I-1:0] s_rfnoc_chdr_tlast; + logic [ NUM_PORTS_I-1:0] s_rfnoc_chdr_tvalid; + logic [ NUM_PORTS_I-1:0] s_rfnoc_chdr_tready; + + // DUT Master (Output) Port Signals + logic [CHDR_W*NUM_PORTS_O-1:0] m_rfnoc_chdr_tdata; + logic [ NUM_PORTS_O-1:0] m_rfnoc_chdr_tlast; + logic [ NUM_PORTS_O-1:0] m_rfnoc_chdr_tvalid; + logic [ NUM_PORTS_O-1:0] m_rfnoc_chdr_tready; + + // Map the array of BFMs to a flat vector for the DUT connections + for (genvar i = 0; i < NUM_PORTS_I; i++) begin : gen_dut_input_connections + // Connect BFM master to DUT slave port + assign s_rfnoc_chdr_tdata[CHDR_W*i+:CHDR_W] = m_chdr[i].tdata; + assign s_rfnoc_chdr_tlast[i] = m_chdr[i].tlast; + assign s_rfnoc_chdr_tvalid[i] = m_chdr[i].tvalid; + assign m_chdr[i].tready = s_rfnoc_chdr_tready[i]; + end + for (genvar i = 0; i < NUM_PORTS_O; i++) begin : gen_dut_output_connections + // Connect BFM slave to DUT master port + assign s_chdr[i].tdata = m_rfnoc_chdr_tdata[CHDR_W*i+:CHDR_W]; + assign s_chdr[i].tlast = m_rfnoc_chdr_tlast[i]; + assign s_chdr[i].tvalid = m_rfnoc_chdr_tvalid[i]; + assign m_rfnoc_chdr_tready[i] = s_chdr[i].tready; + end + rfnoc_block_fft #( - .THIS_PORTID (0 ), - .CHDR_W (64 ), - .MTU (MTU), - - .EN_MAGNITUDE_OUT (0 ), - .EN_MAGNITUDE_APPROX_OUT(1 ), - .EN_MAGNITUDE_SQ_OUT (1 ), - .EN_FFT_SHIFT (1 ) - ) DUT ( - .rfnoc_chdr_clk (backend.chdr_clk), - .ce_clk (backend.chdr_clk), - .s_rfnoc_chdr_tdata (m_chdr.tdata ), - .s_rfnoc_chdr_tlast (m_chdr.tlast ), - .s_rfnoc_chdr_tvalid(m_chdr.tvalid ), - .s_rfnoc_chdr_tready(m_chdr.tready ), - - .m_rfnoc_chdr_tdata (s_chdr.tdata ), - .m_rfnoc_chdr_tlast (s_chdr.tlast ), - .m_rfnoc_chdr_tvalid(s_chdr.tvalid ), - .m_rfnoc_chdr_tready(s_chdr.tready ), - - .rfnoc_core_config (backend.cfg ), - .rfnoc_core_status (backend.sts ), - .rfnoc_ctrl_clk (backend.ctrl_clk), - - .s_rfnoc_ctrl_tdata (m_ctrl.tdata ), - .s_rfnoc_ctrl_tlast (m_ctrl.tlast ), - .s_rfnoc_ctrl_tvalid(m_ctrl.tvalid ), - .s_rfnoc_ctrl_tready(m_ctrl.tready ), - - .m_rfnoc_ctrl_tdata (s_ctrl.tdata ), - .m_rfnoc_ctrl_tlast (s_ctrl.tlast ), - .m_rfnoc_ctrl_tvalid(s_ctrl.tvalid ), - .m_rfnoc_ctrl_tready(s_ctrl.tready ) + .THIS_PORTID (THIS_PORTID), + .CHDR_W (CHDR_W), + .MTU (MTU), + .NUM_PORTS (NUM_PORTS), + .NUM_CORES (NUM_CORES), + .MAX_FFT_SIZE_LOG2 (MAX_FFT_SIZE_LOG2), + .MAX_CP_LIST_LEN_INS_LOG2(MAX_CP_LIST_LEN_INS_LOG2), + .MAX_CP_LIST_LEN_REM_LOG2(MAX_CP_LIST_LEN_REM_LOG2), + .CP_INSERTION_REPEAT (CP_INSERTION_REPEAT), + .CP_REMOVAL_REPEAT (CP_REMOVAL_REPEAT), + .EN_FFT_BYPASS (EN_FFT_BYPASS), + .EN_FFT_ORDER (EN_FFT_ORDER), + .EN_MAGNITUDE (EN_MAGNITUDE), + .EN_MAGNITUDE_SQ (EN_MAGNITUDE_SQ), + .USE_APPROX_MAG (USE_APPROX_MAG) + ) dut ( + .rfnoc_chdr_clk (rfnoc_chdr_clk), + .rfnoc_ctrl_clk (rfnoc_ctrl_clk), + .ce_clk (ce_clk), + .rfnoc_core_config (backend.cfg), + .rfnoc_core_status (backend.sts), + .s_rfnoc_chdr_tdata (s_rfnoc_chdr_tdata), + .s_rfnoc_chdr_tlast (s_rfnoc_chdr_tlast), + .s_rfnoc_chdr_tvalid(s_rfnoc_chdr_tvalid), + .s_rfnoc_chdr_tready(s_rfnoc_chdr_tready), + .m_rfnoc_chdr_tdata (m_rfnoc_chdr_tdata), + .m_rfnoc_chdr_tlast (m_rfnoc_chdr_tlast), + .m_rfnoc_chdr_tvalid(m_rfnoc_chdr_tvalid), + .m_rfnoc_chdr_tready(m_rfnoc_chdr_tready), + .s_rfnoc_ctrl_tdata (m_ctrl.tdata), + .s_rfnoc_ctrl_tlast (m_ctrl.tlast), + .s_rfnoc_ctrl_tvalid(m_ctrl.tvalid), + .s_rfnoc_ctrl_tready(m_ctrl.tready), + .m_rfnoc_ctrl_tdata (s_ctrl.tdata), + .m_rfnoc_ctrl_tlast (s_ctrl.tlast), + .m_rfnoc_ctrl_tvalid(s_ctrl.tvalid), + .m_rfnoc_ctrl_tready(s_ctrl.tready) ); + //--------------------------------------------------------------------------- // Helper Tasks //--------------------------------------------------------------------------- - // Translate the desired register access to a ctrlport write request. - task automatic write_reg(int port, byte addr, bit [31:0] value); - blk_ctrl.reg_write(256*8*port + addr*8, value); - endtask : write_reg - - // Translate the desired register access to a ctrlport read request. - task automatic read_user_reg(int port, byte addr, output logic [63:0] value); - blk_ctrl.reg_read(256*8*port + addr*8 + 0, value[31: 0]); - blk_ctrl.reg_read(256*8*port + addr*8 + 4, value[63:32]); - endtask : read_user_reg - - //--------------------------------------------------------------------------- - // Test Process - //--------------------------------------------------------------------------- - - task automatic test_sine_wave ( - input int unsigned port + task automatic write_reg ( + int unsigned addr, + logic [31:0] write_val, + int core ); - test.start_test("Test sine wave", 20us); + int base_addr = core * (2**FFT_CORE_ADDR_W); + `ASSERT_ERROR(addr < 2**REG_ADDR_W, "Register address is out of bounds"); + `ASSERT_ERROR(core < NUM_CORES, "Specified FFT core does not exist"); + blk_ctrl.reg_write(base_addr + addr, write_val); + endtask - write_reg(port, DUT.SR_FFT_SIZE_LOG2, FFT_SIZE_LOG2); - write_reg(port, DUT.SR_FFT_DIRECTION, DUT.FFT_FORWARD); - write_reg(port, DUT.SR_FFT_SCALING, FFT_SCALING); - write_reg(port, DUT.SR_FFT_SHIFT_CONFIG, FFT_SHIFT_CONFIG); - write_reg(port, DUT.SR_MAGNITUDE_OUT, DUT.COMPLEX_OUT); // Enable real/imag out - // Send a sine wave + task automatic read_reg ( + input int unsigned addr, + output logic [31:0] read_val, + input int core + ); + int base_addr = core * (2**FFT_CORE_ADDR_W); + `ASSERT_ERROR(addr < 2**REG_ADDR_W, "Register address is out of bounds"); + `ASSERT_ERROR(core < NUM_CORES, "Specified FFT core does not exist"); + blk_ctrl.reg_read(base_addr + addr, read_val); + endtask + + + task automatic user_reset (int core); + int dummy_val; + write_reg(REG_RESET_ADDR, 1'b1, core); + // Do a dummy read to ensure reset has time to complete + read_reg(REG_RESET_ADDR, dummy_val, core); + endtask + + + // Configures the FFT core using the given parameters. + // + // fft_size, : Length of FFT (e.g., 4096 for 4k FFT) + // cp_insertions[] : List of cyclic-prefix insertions to load + // cp_removals[] : List of cyclic-prefix removals to load + // fft_scaling : Scaling value to be passed to FFT core + // fft_direction : Direction of FFT (FFT_FORWARD, FFT_INVERSE) + // cp_list_clear : When true, cyclic-prefix list will be reset + // fft_bypass : Enable FFT bypass feature + // fft_order_sel : FFT output order selection + // magnitude_sel : Magnitude output selection + // core : Which FFT core to configure + // + task automatic config_fft ( + int fft_size, + int cp_insertions[] = {}, + int cp_removals[] = {}, + int fft_scaling = fft_scale_default($clog2(fft_size)), + bit fft_direction = FFT_FORWARD, + bit cp_list_clear = 1, + bit fft_bypass = 0, + int fft_order_sel = FFT_ORDER_NATURAL, + int magnitude_sel = 0, + int core = 0 + ); + logic [31:0] fft_size_log2 = $clog2(fft_size); + + if (cp_list_clear) begin + if (VERBOSE) $display("config_fft(): Clearing Cyclic Prefix insertion FIFO"); + write_reg(REG_CP_INS_LIST_CLR_ADDR, 1'b1, core); + if (VERBOSE) $display("config_fft(): Clearing Cyclic Prefix removal FIFO"); + write_reg(REG_CP_REM_LIST_CLR_ADDR, 1'b1, core); + end + if (VERBOSE) $display("config_fft(): Setting FFT Size (Log2) to %0d", fft_size_log2); + write_reg(REG_LENGTH_LOG2_ADDR, fft_size_log2, core); + if (VERBOSE) $display("config_fft(): Setting FFT Scaling to 0x%8h", fft_scaling); + write_reg(REG_SCALING_ADDR, fft_scaling, core); + if (VERBOSE) $display("config_fft(): Setting FFT Direction to %0d", fft_direction); + write_reg(REG_DIRECTION_ADDR, fft_direction, core); + foreach (cp_insertions[i]) begin + `ASSERT_ERROR( + cp_insertions[i] < fft_size, + "Cyclic prefix insertion length must be less than FFT size" + ); + if (VERBOSE) $display("config_fft(): Setting Cyclic Prefix (insertion) %0d", + cp_insertions[i]); + write_reg(REG_CP_INS_LEN_ADDR, cp_insertions[i], core); + write_reg(REG_CP_INS_LIST_LOAD_ADDR, 1'b1, core); + end + foreach (cp_removals[i]) begin + `ASSERT_ERROR( + cp_removals[i] < fft_size, + "Cyclic prefix removal length must be less than FFT size" + ); + if (VERBOSE) $display("config_fft(): Setting Cyclic Prefix (removal) %0d", + cp_removals[i]); + write_reg(REG_CP_REM_LEN_ADDR, cp_removals[i], core); + write_reg(REG_CP_REM_LIST_LOAD_ADDR, 1'b1, core); + end + if (EN_FFT_BYPASS) begin + if (VERBOSE) $display("config_fft(): Setting FFT Bypass to %0d", fft_bypass); + write_reg(REG_BYPASS_ADDR, fft_bypass, core); + end + if (EN_FFT_ORDER) begin + if (VERBOSE) $display("config_fft(): Setting FFT Order to %0d", fft_order_sel); + write_reg(REG_ORDER_ADDR, fft_order_sel, core); + end + if (EN_MAGNITUDE || EN_MAGNITUDE_SQ) begin + if (VERBOSE) $display("config_fft(): Setting Magnitude to %0d", magnitude_sel); + write_reg(REG_MAGNITUDE_ADDR, magnitude_sel, core); + end + endtask + + + // Test a readable/writable register + task automatic test_rw_reg ( + input string reg_name, + input int unsigned addr, + input int unsigned bit_width = 32, + input int unsigned core = 0 + ); + + logic [31:0] init_val, write_val, read_val, check_val; + string s; + + if (VERBOSE) $display("test_rw_reg(): Set and check %s register... ", reg_name); + read_reg(addr, init_val, core); + repeat (2) begin + write_val = $random(); + write_reg(addr, write_val, core); + read_reg(addr, read_val, core); + check_val = write_val & 32'((1 << bit_width)-1); // Mask relevant bits + `ASSERT_FATAL( + read_val == check_val, + $sformatf("%s register incorrect readback! Expected: %0d, Actual %0d", + reg_name, check_val, read_val) + ); + end + write_reg(addr, init_val, core); + endtask + + + // Test read-only register + task automatic test_ro_reg ( + input string reg_name, + input int unsigned addr, + input int unsigned value, + input int unsigned bit_width = 32, + input int unsigned core = 0 + ); + logic [31:0] read_val; + string s; + + if (VERBOSE) $display("test_ro_reg(): Read and check %s register... ", reg_name); + read_reg(addr, read_val, core); + read_val = read_val & 32'((1 << bit_width)-1); // Mask relevant bits + `ASSERT_FATAL( + value == read_val, + $sformatf("%s register incorrect readback! Expected: %0d, Actual %0d", + reg_name, value, read_val) + ); + endtask + + + // Generates a complex sine wave with period of 4 samples. + // + // num_samples : The number of samples to generate + // phase : The phase offset of the first sample. -1 would start one + // sample before the real part crosses y = 9. + // amplitude : Amplitude of the signal + // + function automatic item_queue_t gen_sine_wave( + int num_samples, + int phase = 0, + logic [15:0] amplitude = 16'd16383 + ); + item_t samples [] = new [num_samples]; + + foreach (samples[samp_i]) begin + case (unsigned'(samp_i + phase) % 4) + 0: samples[samp_i] = { amplitude, 16'd0 }; + 1: samples[samp_i] = { 16'd0, amplitude }; + 2: samples[samp_i] = { -amplitude, 16'd0 }; + 3: samples[samp_i] = { 16'd0, -amplitude }; + endcase + end + return samples; + endfunction + + + // Generates a complex sine wave in the frequency domain. + // + // num_samples : The number of samples to generate + // f_norm : Normalized frequency + // amplitude : Amplitude in the frequency domain + // + function automatic item_queue_t gen_tone( + int num_samples, + real f_norm = 0, + logic [15:0] amplitude = 16'd16383 + ); + item_t samples [] = new [num_samples]; + samples = '{default: 0}; + samples[int'(f_norm*num_samples)] = amplitude; + return samples; + endfunction + + + // Test the FFT block by putting a single frequency signal through it. + // + // fft_size : Size of the FFT to test (e.g., 4096 for 4k FFT) + // num_ffts : Number of complete FFTs to test + // cp_insertions[] : Cyclic-prefix insertion list to use + // cp_removals[] : Cyclic-prefix removal list to use + // pkt_size : Packet size to use + // timed : Test timed (1) or untimed (0) packets + // core : Which FFT core to test + // + task automatic test_fft_sine( + int fft_size, + int num_ffts = 1, + int cp_insertions[] = {}, + int cp_removals[] = {}, + int pkt_size = fft_size < SPP ? fft_size : SPP, + bit timed = 1, + int core = 0 + ); + int first_port = core*NUM_CHAN_PER_CORE; + int last_port = (core+1)*NUM_CHAN_PER_CORE - 1; + + if (VERBOSE) begin + $display("test_fft_sine():"); + $display(" fft_size: %0d", fft_size); + $display(" num_ffts: %0d", num_ffts); + $display(" cp_insertions: %p", cp_insertions); + $display(" cp_removals: %p", cp_removals); + $display(" pkt_size: %0d", pkt_size); + $display(" timed: %0d", timed); + $display(" core: %0d", core); + end + + // Having both insertion and removal might work, but that's not a use case + // we're supporting. + assert(!(cp_insertions.size() && cp_removals.size())) else + `ASSERT_ERROR(0, "Cannot specify both CP insertion and CP removal"); + + config_fft(fft_size, cp_insertions, cp_removals, fft_scale_default(fft_size), + FFT_FORWARD, 1, .core(core)); + + // Create a thread for the sender and one (or more) for the receiver(s). fork - begin - chdr_word_t send_payload[$]; + begin : sender + item_queue_t send_data; + packet_info_t send_pkt_info; + int cp_removal_len; + int num_pkts; + int total_fft_size; + int samp_count; - for (int n = 0; n < NUM_ITERATIONS; n++) begin - for (int i = 0; i < (FFT_SIZE/8); i++) begin - send_payload.push_back({ 16'h5A82, 16'h5A82, 16'h7FFF, 16'h0000}); - send_payload.push_back({-16'h5A82, 16'h5A82, 16'h0000, 16'h7FFF}); - send_payload.push_back({-16'h5A82,-16'h5A82,-16'h7FFF, 16'h0000}); - send_payload.push_back({ 16'h5A82,-16'h5A82, 16'h0000,-16'h7FFF}); + send_pkt_info = '0; + + if (VERBOSE) $display("test_fft_sine(): send: Start send thread"); + + // Generate the data we're going to send + for (int fft_count = 0; fft_count < num_ffts; fft_count++) begin + // This emulates the CP removal FIFO's behavior + if (cp_removals.size() == 0) cp_removal_len = 0; + else cp_removal_len = cp_removals[fft_count % cp_removals.size()]; + + total_fft_size += fft_size + cp_removal_len; + + if (VERBOSE) begin + $display("test_fft_sine(): send: Generating FFT %0d for %0d+%0d = %0d samples", + fft_count, cp_removal_len, fft_size, fft_size + cp_removal_len); end - blk_ctrl.send(port, send_payload); - blk_ctrl.wait_complete(port); - send_payload = {}; + // Generate the data we're going to send + send_data = {send_data, gen_sine_wave(fft_size + cp_removal_len, -cp_removal_len)}; end + + num_pkts = $ceil(real'(send_data.size()) / real'(pkt_size)); + + // Send the data one packet at a time + samp_count = 0; + for (int pkt_count = 0; pkt_count < num_pkts; pkt_count++) begin + int start, len; + + start = pkt_count * pkt_size; + len = `MIN(total_fft_size - pkt_count*pkt_size, pkt_size); + + send_pkt_info.eob = (pkt_count == num_pkts-1); + + if (VERBOSE) begin + $display("test_fft_sine(): send: PKT %0d: Sending %0d samples (EOB=%b)", + pkt_count, len, send_pkt_info.eob); + end + + if (timed) begin + send_pkt_info.has_time = 1; + send_pkt_info.timestamp = samp_count + 'hDEADBEEF; + end + + for (int port = first_port; port <= last_port; port++) begin + blk_ctrl.send_items(port, send_data[start : start+len-1], , send_pkt_info); + end + blk_ctrl.wait_complete(first_port); + + samp_count += len; + end + if (VERBOSE) $display("test_fft_sine(): send: End send thread"); + end : sender + + + // Create one thread for each port to be verified + begin : receiver + for (int port_num = first_port; port_num <= last_port; port_num++) begin : recv_port_loop + fork + int port = port_num; // Pass the unique port value to each thread + begin : receive_thread + item_queue_t recv_payload, recv_data; + chdr_word_t recv_metadata[$]; + packet_info_t recv_pkt_info; + longint timestamp = 'hDEADBEEF; + int samp_count, fft_samp_count; + + if (VERBOSE) $display("test_fft_sine(): recv: Start recv thread for port %0d", port); + + // Receive the burst + for (int pkt_count = 0; ; pkt_count++) begin + blk_ctrl.recv_items_adv(port, recv_payload, recv_metadata, recv_pkt_info); + recv_data = { recv_data, recv_payload }; + + if (VERBOSE) begin + $display("test_fft_sine(): recv: PKT %0d: Received %0d samples (EOV=%b, EOB=%b)", + pkt_count, recv_payload.size(), recv_pkt_info.eov, recv_pkt_info.eob); + end + + // Check timestamp + if (timed && pkt_count == 0) begin + `ASSERT_ERROR( + recv_pkt_info.has_time, + $sformatf({"test_fft_sine(): recv: PKT %0d: ", + "Expected timestamp on packet"}, + pkt_count) + ); + `ASSERT_ERROR( + recv_pkt_info.timestamp == timestamp, + $sformatf({"test_fft_sine(): recv: PKT %0d: ", + "Expected timestamp of 0x%X, received 0x%x"}, + pkt_count, timestamp, recv_pkt_info.timestamp) + ); + timestamp += recv_payload.size(); + end else begin + `ASSERT_ERROR( + recv_pkt_info.has_time == 0, + $sformatf({"test_fft_sine(): recv: PKT %0d: ", + "Unexpected timestamp"}, + pkt_count) + ); + end + + if (cp_removals.size() == 0 && cp_insertions.size() == 0 && + recv_data.size() % fft_size == 0) begin + `ASSERT_ERROR(recv_pkt_info.eov, + "test_fft_sine(): recv: EOV is not set on FFT multiple"); + end + + if (recv_pkt_info.eob) break; + end + + + // Verify the received data + samp_count = 0; + for (int fft_count = 0; fft_count < num_ffts; fft_count++) begin + int peak_index; + int cp_peak_index; + int cp_insertion_len; + int total_fft_size; + + // This emulates the CP insertion FIFO's behavior + if (cp_insertions.size() == 0) cp_insertion_len = 0; + else cp_insertion_len = cp_insertions[fft_count % cp_insertions.size()]; + + total_fft_size = fft_size + cp_insertion_len; + + if (VERBOSE) begin + $display("test_fft_sine(): recv: Checking FFT %0d for %0d+%0d = %0d samples", + fft_count, cp_insertion_len, fft_size, total_fft_size); + end + + // We should see a peak in the FFT region but may see one in + // the cyclic prefix as well if it is long enough. + peak_index = cp_insertion_len + fft_size/4; + cp_peak_index = peak_index - fft_size; + + recv_payload = recv_data[samp_count : samp_count+total_fft_size-1]; + + // Verify the sample values + fft_samp_count = 0; + foreach (recv_payload[samp_i]) begin + if (fft_samp_count == peak_index || fft_samp_count == cp_peak_index) begin + bit signed [15:0] real_val, imag_val; + int magnitude; + {real_val, imag_val} = recv_payload[samp_i]; + magnitude = $sqrt(real_val**2 + imag_val**2); + `ASSERT_ERROR( + imag_val == 0, + $sformatf({"test_fft_sine(): recv: FFT %0d: ", + "Expected 0 for im at sample %0d, received 0x%x"}, + fft_count, fft_samp_count, recv_payload[samp_i]) + ); + `ASSERT_ERROR( + real_val >= 16368, + $sformatf({"test_fft_sine(): recv: FFT %0d: ", + "Expected re >= 16368 at sample %0d, received 0x%x"}, + fft_count, fft_samp_count, recv_payload[samp_i]) + ); + end else begin + `ASSERT_ERROR( + recv_payload[samp_i] == 0, + $sformatf({"test_fft_sine(): recv: FFT %0d: ", + "Expected 0 at sample %0d, received 0x%x"}, + fft_count, fft_samp_count, recv_payload[samp_i]) + ); + end + fft_samp_count++; + end + + samp_count += total_fft_size; + end + + // Check the length + `ASSERT_ERROR( + samp_count == recv_data.size(), + $sformatf({"test_fft_sine(): recv: ", + "Expected %0d samples for this burst but received %0d"}, + samp_count, recv_data.size()) + ); + + if (VERBOSE) $display("test_fft_sine(): recv: End recv thread for port %0d", port); + + end : receive_thread + join_none + end : recv_port_loop + + // Wait for all receive threads to finish + wait fork; + end : receiver + join + endtask + + + // Calculates and returns the expected value for the magnitude calculation. + function automatic item_queue_t calc_magnitude(item_queue_t data_in, int select); + item_queue_t data_out; + if (select == MAG_SEL_NONE) begin + return data_in; + end else begin + int re, im, magnitude; + bit signed [15:0] re16, im16; + foreach (data_in[i]) begin + // Resize with sign extension to int. Done in two steps to make Vivado + // XSim happy. Otherwise, we'd do: re = signed'(data_in[i][31:16]); + re16 = data_in[i][31:16]; + im16 = data_in[i][15:0]; + re = re16; + im = im16; + if (select == MAG_SEL_MAG && USE_APPROX_MAG) begin + magnitude = `MAX(`ABS(re), `ABS(im)) + `MIN(`ABS(re), `ABS(im))/4; + end else if (select == MAG_SEL_MAG && !USE_APPROX_MAG) begin + magnitude = int'($sqrt(re**2 + im**2)); + end else if (select == MAG_SEL_MAG_SQ) begin + magnitude = (re**2 + im**2) / 32'sh8000; + end else begin + `ASSERT_FATAL(0, "Invalid magnitude selection") + end + if (magnitude > 16'sh7FFF) magnitude = 16'h7FFF; + data_out[i] = magnitude << 16; + end + return data_out; + end + endfunction + + + // Check if two arrays of samples are approximately equal. Returns 1 if every + // component of every sample is <= thresh. Otherwise returns 0. + function automatic bit approx_equal( + item_queue_t left, + item_queue_t right, + int thresh = 1 + ); + if (left.size() != right.size()) return 0; + foreach (left[i]) begin + bit signed [15:0] left_re, right_re, left_im, right_im; + if (left[i] == right[i]) continue; + left_re = left [i][31:16]; + left_im = left [i][15: 0]; + right_re = right[i][31:16]; + right_im = right[i][15: 0]; + if (`ABS(left_re - right_re) > thresh || `ABS(left_im - right_im) > thresh) begin + $display("At index %0d, left = 0x%X, right = 0x%X", i, + {left_re, left_im}, {right_re, right_im}); + return 0; + end + end + return 1; + endfunction + + + //--------------------------------------------------------------------------- + // Tests + //--------------------------------------------------------------------------- + + // Test a sequence of random configurations + task automatic test_random( + int num_iterations, + int max_fft_size = MAX_FFT_SIZE, + int min_fft_size = MIN_FFT_SIZE + ); + test.start_test( + $sformatf({ + "Test Random\n", + " num_iterations: %0d\n", + " max_fft_size: %0d\n", + " min_fft_size: %0d"}, + num_iterations, max_fft_size, min_fft_size + ), num_iterations*max_fft_size*30ns + ); + + for (int test_iter = 0; test_iter < num_iterations; test_iter++) begin + bit timed = $urandom_range(0, 1); + int fft_size = 2**$urandom_range($clog2(min_fft_size), $clog2(max_fft_size)); + int num_ffts = $urandom_range(1, 3); + int cp_mode = $urandom_range(0, 2); // 0=None, 1=removal, 2=insertion + bit has_cp = $urandom_range(0, 1); + int cp_list_len = $urandom_range(0, + `MIN(num_ffts-1, `MIN(MAX_CP_LIST_LEN_REM, MAX_CP_LIST_LEN_INS))); + int cp_lengths[] = new [cp_list_len]; + int pkt_size; + + // We require the packet size to be a multiple of MIN_FFT_SIZE and up to + // the current FFT size in length. + pkt_size = $urandom_range(MIN_FFT_SIZE, `MIN(fft_size, SPP)); + pkt_size = `DIV_CEIL(pkt_size, MIN_FFT_SIZE)*MIN_FFT_SIZE; + + foreach (cp_lengths[i]) begin + cp_lengths[i] = $urandom_range(0, fft_size-1); end - begin - string msg; - chdr_word_t recv_payload[$], temp_payload[$]; - int data_bytes; - logic [15:0] real_val; - logic [15:0] cplx_val; + if (cp_mode == 2) begin + // Test with CP insertion + test_fft_sine(fft_size, num_ffts, .timed(timed), + .cp_insertions(cp_lengths), .pkt_size(pkt_size)); + end else if(cp_mode == 1) begin + // Test with CP removal + test_fft_sine(fft_size, num_ffts, .timed(timed), + .cp_removals(cp_lengths), .pkt_size(pkt_size)); + end else begin + // Test without cyclic prefix + test_fft_sine(fft_size, num_ffts, .timed(timed), .pkt_size(pkt_size)); + end + end - for (int n = 0; n < NUM_ITERATIONS; n++) begin - blk_ctrl.recv(port, recv_payload, data_bytes); + test.end_test(); + endtask - `ASSERT_ERROR(recv_payload.size * 2 == FFT_SIZE, "received wrong amount of data"); - for (int k = 0; k < FFT_SIZE/2; k++) begin - chdr_word_t payload_word; - payload_word = recv_payload.pop_front(); + task automatic test_fft_config(int fft_size, int pkt_size = SPP); + int cp_lengths[] = new [14]; - for (int i = 0; i < 2; i++) begin - {real_val, cplx_val} = payload_word; - payload_word = payload_word[63:32]; + test.start_test( + $sformatf("Test fft configuration (fft_size=%0d, pkt_size=%0d)", fft_size, pkt_size), + 2ms + ); - if (2*k+i == FFT_BIN) begin - // Assert that for the special case of a 1/8th sample rate sine wave input, - // the real part of the corresponding 1/8th sample rate FFT bin should always be greater than 0 and - // the complex part equal to 0. - $sformat(msg, - "On iteration %0d, sample %0d, FFT real part is 0x%X, expected value > 0", - n, 2*k+i, real_val); - `ASSERT_ERROR(real_val > 32'd0, msg); - $sformat(msg, - "On iteration %0d, sample %0d, FFT complex part is 0x%X, expected 0", - n, 2*k+i, cplx_val); - `ASSERT_ERROR(cplx_val == 32'd0, msg); - end else begin - // Assert all other FFT bins should be 0 for both complex and real parts - $sformat(msg, - "On iteration %0d, sample %0d, FFT real part is 0x%X, expected value 0", - n, 2*k+i, real_val); - `ASSERT_ERROR(real_val == 32'd0, msg); - $sformat(msg, - "On iteration %0d, sample %0d, FFT complex part is 0x%X, expected 0", - n, 2*k+i, cplx_val); - `ASSERT_ERROR(cplx_val == 32'd0, msg); - end - end - end + if (fft_size == 4096) begin + // Assume 122.88 MS/s, 30 kHz subcarrier spacing (mu=1), 28 fft symbols + // per 1 ms subframe. + cp_lengths = '{352, 288, 288, 288, 288, 288, 288, 288, 288, 288, 288, 288, 288, 288}; + end else if (fft_size == 8192) begin + // Assume 245.76 MS/s, 30 kHz subcarrier spacing (mu=1), 28 fft symbols + // per 1 ms subframe. + cp_lengths = '{704, 576, 576, 576, 576, 576, 576, 576, 576, 576, 576, 576, 576, 576}; + end else begin + $fatal(1, "Bad FFT size"); + end + + test_fft_sine(fft_size, cp_lengths.size(), .cp_insertions(cp_lengths)); + test_fft_sine(fft_size, cp_lengths.size(), .cp_removals(cp_lengths)); + + test.end_test(); + endtask + + + // Test loopback, i.e., sending a symbol and getting back the original symbol + // that was sent, like you would do with OFDM. + task automatic test_loopback( + int fft_size, + int num_ffts = 1, + int cp[] = {}, + int pkt_size = fft_size < SPP ? fft_size : SPP, + int port = 0 + ); + logic [31:0] data_in [$]; + logic [31:0] signal [$]; + logic [31:0] data_out [$]; + int fft_scaling = fft_scale_default($clog2(fft_size)); // Get 1/N scaling + int ifft_scaling = 0; + packet_info_t send_pkt_info = '0; + + if (NUM_CHAN_PER_CORE > 1) begin + `ASSERT_ERROR(0, "test_loopback() only works with one channel per core."); + return; + end + + test.start_test( + $sformatf({ + "Test Loopback\n", + " fft_size: %0d\n", + " num_ffts: %0d\n", + " cp: %p\n", + " pkt_size: %0d\n", + " port: %0d"}, + fft_size, num_ffts, cp, pkt_size, port + ), 2ms + ); + + // For this test, we send one FFT per packet, unless the FFT is larger than + // the packet size, which is supported. + assert(fft_size % pkt_size == 0) else + `ASSERT_ERROR(0, "fft_size must be a multiple of pkt_size"); + + // Set the SPP + blk_ctrl.set_max_payload_length(port, pkt_size*4); + + // Do IFFT to convert the frequency domain signal to a time domain signal + // with CP insertion. + config_fft(fft_size, cp, {}, ifft_scaling, FFT_INVERSE, .core(port)); + repeat(num_ffts) data_in = {data_in, gen_tone(fft_size, 0.25)}; + send_pkt_info.eob = '1; + blk_ctrl.send_packets_items(.port(port), .items(data_in), .pkt_info(send_pkt_info)); + blk_ctrl.recv_packets_items(.port(port), .items(signal), .eob(1)); + + // Now do FFT with CP removal to get back the original symbol + config_fft(fft_size, {}, cp, fft_scaling, FFT_FORWARD, .core(port)); + blk_ctrl.send_packets_items(.port(port), .items(signal), .pkt_info(send_pkt_info)); + blk_ctrl.recv_packets_items(.port(port), .items(data_out), .eob(1)); + + `ASSERT_ERROR(data_in == data_out, + "Samples sent does not match samples received"); + + test.end_test(); + endtask : test_loopback + + + // Test max and min FFT size without CP + task automatic test_max_min_fft(); + test.start_test("Test min/max values", MAX_FFT_SIZE*30ns); + test_fft_sine(MAX_FFT_SIZE, 2); + test_fft_sine(MIN_FFT_SIZE, 2); + test.end_test(); + endtask + + + // Test max CP list length (2**MAX_CP_LIST_LEN_LOG2 - 1). + task automatic test_max_cp_list_len(int fft_size = MIN_FFT_SIZE); + test.start_test( + $sformatf("Test maximum CP list size (fft_size = %0d)", fft_size), + 5ms + ); + + for (int insert = 0; insert < 2; insert++) begin + int cp_list_len = insert ? MAX_CP_LIST_LEN_INS : MAX_CP_LIST_LEN_REM; + int cp_lengths [] = new [cp_list_len]; + + foreach (cp_lengths[i]) begin + // Keep the insertion length small to limit test length. We'll test the + // max length in another test. + cp_lengths[i] = $urandom_range(0, 1); + end + + // Test one FFT more than the CP list length to make sure it repeats + if (insert) begin + test_fft_sine( + .fft_size (fft_size), + .num_ffts (cp_list_len+1), + .cp_insertions(cp_lengths) + ); + end else begin + test_fft_sine( + .fft_size (fft_size), + .num_ffts (cp_list_len+1), + .cp_removals (cp_lengths) + ); + end + end + + test.end_test(); + endtask + + + // Test max FFT size with max and min CP insertion and removal lengths + task automatic test_max_min_cp_len(); + int cp_lengths [] = {MAX_CP_LEN, 1, 0}; + test.start_test("Test min/max CP lengths", 5ms); + test_fft_sine( + .fft_size (MAX_FFT_SIZE), + .num_ffts (cp_lengths.size()), + .cp_insertions(cp_lengths) + ); + test_fft_sine( + .fft_size (MAX_FFT_SIZE), + .num_ffts (cp_lengths.size()), + .cp_removals (cp_lengths) + ); + test.end_test(); + endtask + + + // Test all the registers to ensure that they read/write as expected. Except + // the write-only registers which will be exercised during functional tests. + task automatic test_registers(); + localparam bit [31:0] exp_compat = {16'd3, 16'd0}; + localparam bit [31:0] exp_capabilities = + (8'(MAX_CP_LIST_LEN_INS_LOG2) << 24) | + (8'(MAX_CP_LIST_LEN_REM_LOG2) << 16) | + (8'( MAX_CP_LEN_LOG2) << 8) | + (8'( MAX_FFT_SIZE_LOG2) << 0); + localparam bit [31:0] exp_capabilities2 = + (EN_MAGNITUDE_SQ << 3) | + (EN_MAGNITUDE << 2) | + (EN_FFT_ORDER << 1) | + (EN_FFT_BYPASS << 0); + localparam bit [31:0] exp_port_config = + (16'(NUM_CORES ) << 16) | + (16'(NUM_PORTS / NUM_CORES) << 0); + localparam bit [31:0] exp_overflow = '0; + + test.start_test("Test registers", 2ms); + + // Test registers in each core + for (int core = 0; core < NUM_CORES; core++) begin + if (VERBOSE) $display("test_registers(): Testing core %0d", core); + // Test read/write registers + test_rw_reg("FFT Size", REG_LENGTH_LOG2_ADDR, + dut.gen_fft_cores[0].fft_core_i.REG_LENGTH_LOG2_WIDTH, core); + test_rw_reg("FFT Scaling", REG_SCALING_ADDR, + dut.gen_fft_cores[0].fft_core_i.REG_SCALING_WIDTH, core); + test_rw_reg("FFT Direction", REG_DIRECTION_ADDR, + REG_DIRECTION_WIDTH, core); + test_rw_reg("CP Ins Length", REG_CP_INS_LEN_ADDR, + dut.gen_fft_cores[0].fft_core_i.REG_CP_INS_LEN_WIDTH, core); + test_rw_reg("CP Rem Length", REG_CP_REM_LEN_ADDR, + dut.gen_fft_cores[0].fft_core_i.REG_CP_REM_LEN_WIDTH, core); + + // Test optional registers + if (EN_FFT_BYPASS) begin + test_rw_reg("FFT Bypass", REG_BYPASS_ADDR, + REG_BYPASS_WIDTH, core); + end else begin + test_ro_reg("FFT Bypass", REG_BYPASS_ADDR, 0, + REG_BYPASS_WIDTH, core); + end + if (EN_FFT_ORDER) begin + test_rw_reg("FFT Order", REG_ORDER_ADDR, + REG_ORDER_WIDTH, core); + end else begin + test_ro_reg("FFT Order", REG_ORDER_ADDR, 0, + REG_ORDER_WIDTH, core); + end + if (EN_MAGNITUDE || EN_MAGNITUDE_SQ) begin + test_rw_reg("FFT Magnitude", REG_MAGNITUDE_ADDR, + REG_MAGNITUDE_WIDTH, core); + end else begin + test_ro_reg("FFT Magnitude", REG_MAGNITUDE_ADDR, 0, + REG_MAGNITUDE_WIDTH, core); + end + + // Test read-only registers + test_ro_reg("Compatibility", REG_COMPAT_ADDR, exp_compat, + REG_COMPAT_WIDTH, core); + test_ro_reg("Capabilities", REG_CAPABILITIES_ADDR, exp_capabilities, + REG_CAPABILITIES_WIDTH, core); + test_ro_reg("Capabilities 2", REG_CAPABILITIES2_ADDR, exp_capabilities2, + REG_CAPABILITIES2_WIDTH, core); + test_ro_reg("Port Config", REG_PORT_CONFIG_ADDR, exp_port_config, + REG_PORT_CONFIG_WIDTH, core); + test_ro_reg("Overflow", REG_OVERFLOW_ADDR, exp_overflow, 32, core); + end + + test.end_test(); + endtask : test_registers + + + // Run a simple/quick test on all ports to make sure things are connected as + // expected. + task automatic test_basic(); + test.start_test("Test basic", 2ms); + for (int core = 0; core < NUM_CORES; core++) begin + test_fft_sine( + .fft_size(32), .num_ffts(2), + .cp_insertions({}), + .cp_removals({}), + .pkt_size(32), + .timed(1), + .core(core) + ); + end + test.end_test(); + endtask + + + // Test different cyclic-prefix lists with insertion and removal + task automatic test_cyclic_prefix(); + automatic int cp_lengths_single[] = '{ 12 }; + automatic int cp_lengths_multi[] = '{ 320, 288, 111, 0, 320 }; + automatic int cp_lengths_golden[] = '{ 320, 288, 288, 288, 288, 288, 288, + 288, 288, 288, 288, 288, 288, 288 }; + + test.start_test("Test CP insertion and removal", 2ms); + test_fft_sine(128, cp_lengths_single.size(), .cp_insertions(cp_lengths_single)); + test_fft_sine(512, cp_lengths_multi.size(), .cp_insertions(cp_lengths_multi)); + test_fft_sine(128, cp_lengths_single.size(), .cp_removals(cp_lengths_single)); + test_fft_sine(512, cp_lengths_multi.size(), .cp_removals(cp_lengths_multi)); + test_fft_sine(512, cp_lengths_golden.size(), .cp_removals(cp_lengths_golden)); + test.end_test(); + endtask + + + // Test underflow and back pressure + task automatic test_throttle(); + test.start_test("Test throttle", 2ms); + + // Test back-pressure + for (int port = 0; port < NUM_PORTS; port++) begin + blk_ctrl.set_master_stall_prob(port, 10); + blk_ctrl.set_slave_stall_prob(port, 80); + end + test_fft_sine(512, 8, .pkt_size(256)); + + // Test underflow + for (int port = 0; port < NUM_PORTS; port++) begin + blk_ctrl.set_master_stall_prob(port, 80); + blk_ctrl.set_slave_stall_prob(port, 10); + end + test_fft_sine(512, 8, .pkt_size(256)); + + // Test full throttle + for (int port = 0; port < NUM_PORTS; port++) begin + blk_ctrl.set_master_stall_prob(port, 0); + blk_ctrl.set_slave_stall_prob(port, 0); + end + test_fft_sine(512, 8, .pkt_size(256)); + + // Restore the default back-pressure + for (int port = 0; port < NUM_PORTS; port++) begin + blk_ctrl.set_master_stall_prob(port, STALL_PROB); + blk_ctrl.set_slave_stall_prob(port, STALL_PROB); + end + + test.end_test(); + endtask + + + // Measure the throughput to ensure it's adequate for USRP clock + // configurations. For example, with 266.666 MHz CE clock and 250 MHz radio + // clock, we need at least 93.75% efficiency. + task automatic test_throughput( + int fft_size = 512, + int num_ffts = 8, + int pkt_size = fft_size < SPP ? fft_size : SPP, + int port = 0 + ); + localparam real CE_CLK_FREQ_MHZ = 1000.0/CE_CLK_PER; + localparam real MIN_RATE = 0.98 * CE_CLK_FREQ_MHZ; + logic [31:0] data_in [$]; + logic [31:0] signal [$]; + packet_info_t send_pkt_info = '0; + + if (NUM_CHAN_PER_CORE > 1) begin + `ASSERT_ERROR(0, "test_throughput() only works with one channel per core. Test skipped."); + return; + end + + test.start_test( + $sformatf({ + "Test Throughput\n", + " fft_size: %0d\n", + " num_ffts: %0d\n", + " pkt_size: %0d\n", + " port: %0d"}, + fft_size, num_ffts, pkt_size, port + ), 2ms + ); + + // For this test, we send one FFT per packet, unless the FFT is larger than + // the packet size, which is supported. + assert(fft_size % pkt_size == 0) else + `ASSERT_ERROR(0, "fft_size must be a multiple of pkt_size"); + + // Turn off back pressure in the BFM + blk_ctrl.set_master_stall_prob(port, 0); + blk_ctrl.set_slave_stall_prob(port, 0); + + blk_ctrl.set_max_payload_length(port, pkt_size*4); + config_fft(.fft_size(fft_size), .core(port)); + repeat(num_ffts) data_in = {data_in, gen_tone(fft_size, 0.25)}; + send_pkt_info.eob = '1; + fork + begin : sender + realtime start_time, elapsed; + real rate; + // Send the data twice, once to fill the pipes, the second to measure. + repeat (2) begin + blk_ctrl.send_packets_items(.port(port), .items(data_in), .pkt_info(send_pkt_info)); + start_time = $realtime; + blk_ctrl.wait_complete(port); end + elapsed = $realtime - start_time; + rate = 1000.0 * num_ffts * fft_size / elapsed; // In MS/s + if (VERBOSE) $display("Throughput is %f MS/s", rate); + `ASSERT_ERROR( + rate >= MIN_RATE, + $sformatf("Calculated throughput (%f MS/s) is below threshold (%f MS/s)", + rate, MIN_RATE) + ); + end + begin : receiver + repeat (2) blk_ctrl.recv_packets_items(.port(port), .items(signal), .eob(1)); end join + // Set back pressure back to default + blk_ctrl.set_master_stall_prob(port, STALL_PROB); + blk_ctrl.set_slave_stall_prob(port, STALL_PROB); + test.end_test(); - endtask : test_sine_wave + endtask - task automatic test_short ( - input int unsigned port - ); - item_t samples[$], spectrum[$], recv[$]; - string msg; + // Test the magnitude output option + // + // mag_sel : Magnitude output selection + // port : RFNoC port to test + // + task automatic test_magnitude(int mag_sel, int port = 0); + packet_info_t send_pkt_info = '0; + logic [31:0] data_in [$]; + logic [31:0] data_out [$]; + logic [31:0] data_exp [$]; + localparam FFT_SIZE = 128; + localparam NUM_ITEMS = FFT_SIZE*10; + localparam PKT_SIZE = FFT_SIZE; - test.start_test("Test short FFT", 10us); + test.start_test($sformatf("Test Magnitude (mag_sel = %0d)", mag_sel), 2ms); - write_reg(port, DUT.SR_FFT_SIZE_LOG2, 3); - write_reg(port, DUT.SR_FFT_DIRECTION, DUT.FFT_FORWARD); - write_reg(port, DUT.SR_FFT_SCALING, 0); // No scaling - write_reg(port, DUT.SR_FFT_SHIFT_CONFIG, 10); // Bypass shifting - - // Samples to input to FFT (expected output of IFFT) - samples = '{ - 32'h000E_0000, // {16'd14, 16'd0}, // Vivado won't allow concatenation - 32'h000F_0000, // {16'd15, 16'd0}, // in dynamic types. - 32'h0010_0000, // {16'd16, 16'd0}, - 32'h0011_0000, // {16'd17, 16'd0}, - 32'h0012_0000, // {16'd18, 16'd0}, - 32'h0013_0000, // {16'd19, 16'd0}, - 32'h0014_0000, // {16'd20, 16'd0}, - 32'h0015_0000 // {16'd21, 16'd0} - }; - - // Expected spectrum output by FFT (values to input to IFFT) - spectrum = '{ - 32'h008C_0000, // { 16'sd140, 16'd0}, - 32'hFFFC_000A, // {-16'sd4, 16'd10}, - 32'hFFFC_0004, // {-16'sd4, 16'd4}, - 32'hFFFC_0002, // {-16'sd4, 16'd2}, - 32'hFFFC_0000, // {-16'sd4, 16'd0}, - 32'hFFFC_FFFE, // {-16'sd4, -16'd2}, - 32'hFFFC_FFFC, // {-16'sd4, -16'd4}, - 32'hFFFC_FFF6 // {-16'sd4, -16'd10} - }; - - blk_ctrl.send_items(port, samples); - blk_ctrl.recv_items(port, recv); - - foreach (recv[i]) begin - if (recv[i] != spectrum[i]) begin - $sformat(msg, "On sample %d, received (%d,%d), expected (%d,%d)", i, - signed'(recv[i][31:16]), signed'(recv[i][15:0]), - signed'(spectrum[i][31:16]), signed'(spectrum[i][15:0])); - `ASSERT_ERROR(0, msg); - end + if (NUM_CHAN_PER_CORE > 1) begin + `ASSERT_ERROR(0, "test_magnitude() only works with one channel per core."); + return; end - test.end_test(); + `ASSERT_FATAL(EN_MAGNITUDE || EN_MAGNITUDE_SQ, + "Magnitude/Magnitude-squared logic is not enabled in core."); + `ASSERT_FATAL(EN_FFT_BYPASS, + "Magnitude/Magnitude-squared test requires FFT bypass."); - test.start_test("Test short IFFT", 10us); + `ASSERT_FATAL(!(mag_sel == MAG_SEL_MAG && !EN_MAGNITUDE), + "Magnitude logic is not enabled in core."); - write_reg(port, DUT.SR_FFT_DIRECTION, DUT.FFT_REVERSE); - write_reg(port, DUT.SR_FFT_SCALING, 12'b11); - blk_ctrl.send_items(port, spectrum); - blk_ctrl.recv_items(port, recv); + `ASSERT_FATAL(!(mag_sel == MAG_SEL_MAG_SQ && !EN_MAGNITUDE_SQ), + "Magnitude-squared logic is not enabled in core."); - foreach (recv[i]) begin - if (recv[i] != samples[i]) begin - $sformat(msg, "On sample %d, received (%d,%d), expected (%d,%d)", i, - signed'(recv[i][31:16]), signed'(recv[i][15:0]), - signed'(samples[i][31:16]), signed'(samples[i][15:0])); - `ASSERT_ERROR(0, msg); - end - end + // Set the packet size in bytes + blk_ctrl.set_max_payload_length(port, PKT_SIZE*4); + + // Configure the FFT block to something sane, but bypass the FFT processing + // stage so we can measure the effect of the post-processing directly. We + // also set the FFT order to match the default input order so that samples + // are not reordered during this test. + config_fft(.fft_size(FFT_SIZE), .fft_bypass(1), .fft_order_sel(BIT_REVERSE), + .magnitude_sel(mag_sel), .core(port)); + + // Generate the data to send + repeat(NUM_ITEMS) data_in.push_back($urandom()); + + // Send and receive data + send_pkt_info.eob = '1; + blk_ctrl.send_packets_items(.port(port), .items(data_in), .pkt_info(send_pkt_info)); + blk_ctrl.recv_packets_items(.port(port), .items(data_out), .eob(1)); + + data_exp = calc_magnitude(data_in, mag_sel); + `ASSERT_ERROR(approx_equal(data_out, data_exp), + "Magnitude output didn't match expected values"); test.end_test(); - endtask : test_short + endtask - task automatic test_regs ( - input int unsigned port - ); - logic [31:0] val; - - test.start_test("Test registers", 10us); - - write_reg(port, DUT.SR_FFT_RESET, 1); - write_reg(port, DUT.SR_FFT_RESET, 0); - - // SR_FFT_SIZE_LOG2 - read_user_reg(port, DUT.RB_FFT_SIZE_LOG2, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_SIZE, "FFT_SIZE_LOG2 is incorrect"); - write_reg(port, DUT.SR_FFT_SIZE_LOG2, 32'hFFFFFFFF); - read_user_reg(port, DUT.RB_FFT_SIZE_LOG2, val); - `ASSERT_ERROR(val == 32'hFF, "FFT_SIZE_LOG2 is incorrect"); - write_reg(port, DUT.SR_FFT_SIZE_LOG2, 32'h0); - read_user_reg(port, DUT.RB_FFT_SIZE_LOG2, val); - `ASSERT_ERROR(val == 32'h0, "FFT_SIZE_LOG2 is incorrect"); - write_reg(port, DUT.SR_FFT_SIZE_LOG2, DUT.DEFAULT_FFT_SIZE); - read_user_reg(port, DUT.RB_FFT_SIZE_LOG2, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_SIZE, "FFT_SIZE_LOG2 is incorrect"); - - // SR_MAGNITUDE_OUT - read_user_reg(port, DUT.RB_MAGNITUDE_OUT, val); - `ASSERT_ERROR(val == 32'h0, "MAGNITUDE_OUT is incorrect"); - write_reg(port, DUT.SR_MAGNITUDE_OUT, 32'hFFFFFFFF); - read_user_reg(port, DUT.RB_MAGNITUDE_OUT, val); - `ASSERT_ERROR(val == 32'h3, "MAGNITUDE_OUT is incorrect"); - write_reg(port, DUT.SR_MAGNITUDE_OUT, 32'h0); - read_user_reg(port, DUT.RB_MAGNITUDE_OUT, val); - `ASSERT_ERROR(val == 32'h0, "MAGNITUDE_OUT is incorrect"); - - // SR_FFT_DIRECTION - read_user_reg(port, DUT.RB_FFT_DIRECTION, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_DIRECTION, "FFT_DIRECTION is incorrect"); - write_reg(port, DUT.SR_FFT_DIRECTION, 32'hFFFFFFFF); - read_user_reg(port, DUT.RB_FFT_DIRECTION, val); - `ASSERT_ERROR(val == 32'h1, "FFT_DIRECTION is incorrect"); - write_reg(port, DUT.SR_FFT_DIRECTION, 32'h0); - read_user_reg(port, DUT.RB_FFT_DIRECTION, val); - `ASSERT_ERROR(val == 32'h0, "FFT_DIRECTION is incorrect"); - write_reg(port, DUT.SR_FFT_DIRECTION, DUT.DEFAULT_FFT_DIRECTION); - read_user_reg(port, DUT.RB_FFT_DIRECTION, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_DIRECTION, "FFT_DIRECTION is incorrect"); - - // SR_FFT_SCALING - read_user_reg(port, DUT.RB_FFT_SCALING, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_SCALING, "FFT_SCALING is incorrect"); - write_reg(port, DUT.SR_FFT_SCALING, 32'hFFFFFFFF); - read_user_reg(port, DUT.RB_FFT_SCALING, val); - `ASSERT_ERROR(val == 32'hFFF, "FFT_SCALING is incorrect"); - write_reg(port, DUT.SR_FFT_SCALING, 32'h0); - read_user_reg(port, DUT.RB_FFT_SCALING, val); - `ASSERT_ERROR(val == 32'h0, "FFT_SCALING is incorrect"); - write_reg(port, DUT.SR_FFT_SCALING, DUT.DEFAULT_FFT_SCALING); - read_user_reg(port, DUT.RB_FFT_SCALING, val); - `ASSERT_ERROR(val == DUT.DEFAULT_FFT_SCALING, "FFT_SCALING is incorrect"); - - // SR_FFT_SHIFT_CONFIG - read_user_reg(port, DUT.RB_FFT_SHIFT_CONFIG, val); - `ASSERT_ERROR(val == 32'b0, "FFT_SHIFT_CONFIG is incorrect"); - write_reg(port, DUT.SR_FFT_SHIFT_CONFIG, 32'hFFFFFFFF); - read_user_reg(port, DUT.RB_FFT_SHIFT_CONFIG, val); - `ASSERT_ERROR(val == 32'h3, "FFT_SHIFT_CONFIG is incorrect"); - write_reg(port, DUT.SR_FFT_SHIFT_CONFIG, 32'h0); - read_user_reg(port, DUT.RB_FFT_SHIFT_CONFIG, val); - `ASSERT_ERROR(val == 32'h0, "FFT_SHIFT_CONFIG is incorrect"); - write_reg(port, DUT.SR_FFT_SHIFT_CONFIG, 32'b0); - read_user_reg(port, DUT.RB_FFT_SHIFT_CONFIG, val); - `ASSERT_ERROR(val == 32'b0, "FFT_SHIFT_CONFIG is incorrect"); - - test.end_test(); - endtask : test_regs - + //--------------------------------------------------------------------------- + // Main Test Process + //--------------------------------------------------------------------------- initial begin : tb_main - static int port = 0; - test.start_tb("rfnoc_block_fft_tb"); + string tb_name; + + // Generate a string for the name of this instance of the testbench + tb_name = $sformatf({ + "rfnoc_block_fft_tb\n", + "\tFULL_TEST = %0d\n", + "\tNUM_PORTS = %0d\n", + "\tNUM_CORES = %0d\n", + "\tMAX_FFT_SIZE_LOG2 = %0d\n", + "\tMAX_CP_LIST_LEN_INS_LOG2 = %0d\n", + "\tMAX_CP_LIST_LEN_REM_LOG2 = %0d\n", + "\tEN_MAGNITUDE_SQ = %0d\n", + "\tEN_MAGNITUDE = %0d\n", + "\tEN_FFT_ORDER = %0d\n", + "\tEN_FFT_BYPASS = %0d\n", + "\tUSE_APPROX_MAG = %0d\n", + "\tVERBOSE = %0d"}, + FULL_TEST, NUM_PORTS, NUM_CORES, MAX_FFT_SIZE_LOG2, + MAX_CP_LIST_LEN_INS_LOG2, MAX_CP_LIST_LEN_REM_LOG2, EN_MAGNITUDE_SQ, + EN_MAGNITUDE, EN_FFT_ORDER, EN_FFT_BYPASS, USE_APPROX_MAG, VERBOSE + ); + + // Initialize the test exec object for this testbench + test.start_tb(tb_name); + + // Start the clocks + rfnoc_chdr_clk_gen.start(); + rfnoc_ctrl_clk_gen.start(); + ce_clk_gen.start(); // Start the BFMs running blk_ctrl.run(); - //------------------------------------------------------------------------- + //-------------------------------- // Reset - //------------------------------------------------------------------------- - - test.start_test("Wait for Reset", 10us); - fork - blk_ctrl.reset_chdr(); - blk_ctrl.reset_ctrl(); - join; + //-------------------------------- + + test.start_test("Flush block then reset it", 10us); + blk_ctrl.flush_and_reset(); test.end_test(); - - //------------------------------------------------------------------------- - // Check NoC ID and Block Info - //------------------------------------------------------------------------- - + //-------------------------------- + // Verify Block Info + //-------------------------------- + test.start_test("Verify Block Info", 2us); - `ASSERT_ERROR(blk_ctrl.get_noc_id() == NOC_ID, "Incorrect NOC_ID Value"); - `ASSERT_ERROR(blk_ctrl.get_num_data_i() == NUM_PORTS, "Incorrect NUM_DATA_I Value"); - `ASSERT_ERROR(blk_ctrl.get_num_data_o() == NUM_PORTS, "Incorrect NUM_DATA_O Value"); - `ASSERT_ERROR(blk_ctrl.get_mtu() == MTU, "Incorrect MTU Value"); + `ASSERT_ERROR(blk_ctrl.get_noc_id() == 32'hFF700002, "Incorrect NOC_ID Value"); + `ASSERT_ERROR(blk_ctrl.get_num_data_i() == NUM_PORTS_I, "Incorrect NUM_DATA_I Value"); + `ASSERT_ERROR(blk_ctrl.get_num_data_o() == NUM_PORTS_O, "Incorrect NUM_DATA_O Value"); + `ASSERT_ERROR(blk_ctrl.get_mtu() == MTU, "Incorrect MTU Value"); test.end_test(); - //------------------------------------------------------------------------- - // Tests - //------------------------------------------------------------------------- + //-------------------------------- + // Test Sequences + //-------------------------------- - test_regs(port); - test_short(port); - test_sine_wave(port); + test_registers(); + test_basic(); - //------------------------------------------------------------------------- - // Finish - //------------------------------------------------------------------------- + if (FULL_TEST) begin + if (NUM_CHAN_PER_CORE == 1) begin + test_loopback(128, 2, '{80, 72, 111, 0, 80}); + test_loopback(256, 3, '{160, 144, 56, 0, 160}); + end + if (MAX_FFT_SIZE >= 4096) test_fft_config(4096, 1024); + if (MAX_FFT_SIZE >= 8192) test_fft_config(8192, 1024); + test_cyclic_prefix(); + test_throttle(); + test_throughput(); + test_max_min_fft(); + test_max_cp_list_len(); + test_max_min_cp_len(); + test_random(.num_iterations(200), .max_fft_size(128), .min_fft_size(MIN_FFT_SIZE)); + end - // End the TB, but don't $finish, since we don't want to kill other - // instances of this testbench that may be running. + // Always test magnitude as part of the quick tests, so we can quickly test + // different values of USE_APPROX_MAG. + if (NUM_CHAN_PER_CORE == 1) begin + if (EN_MAGNITUDE && EN_FFT_BYPASS) test_magnitude(MAG_SEL_MAG); + if (EN_MAGNITUDE_SQ && EN_FFT_BYPASS) test_magnitude(MAG_SEL_MAG_SQ); + end + + //-------------------------------- + // Finish Up + //-------------------------------- + + // Display final statistics and results test.end_tb(0); // Kill the clocks to end this instance of the testbench rfnoc_chdr_clk_gen.kill(); rfnoc_ctrl_clk_gen.kill(); + ce_clk_gen.kill(); end + endmodule + + +`default_nettype wire diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/xfft_config_pkg.sv b/lib/rfnoc/blocks/rfnoc_block_fft/xfft_config_pkg.sv new file mode 100644 index 0000000..c5c0a72 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/xfft_config_pkg.sv @@ -0,0 +1,165 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: xfft_config_pkg +// +// Description: +// +// This package helps helps build and interpret the s_axis_config_tdata bus +// used to configure the Xilinx FFT core. The bus changes depending on the +// parameters of the IP. See the Xilinx Fast Fourier Transform product guide +// (PG109) for details. You can verify the values for a specific +// configuration by clicking the Implementation tab in the Vivado IP +// customization GUI. +// + + +package xfft_config_pkg; + + // Set to 1 if cyclic prefix is enabled on the Xilinx IP. Set to 0 otherwise. + localparam bit CP_ENABLE = 0; + + localparam int MAX_W = 256; + + // Returns the width of the SCAL_SCH field of the config_tdata bus. + function automatic int fft_scale_w(int max_fft_size_log2); + // max_fft_size_log2 rounded up to the nearest multiple of 2 + return ((max_fft_size_log2+1) / 2) * 2; + endfunction + + // Returns the FFT scale value needed to get the default 1/N scaling for a + // given FFT size. + function automatic int fft_scale_default(int size_log2); + // Should be [ 10 10 ... 10] if N is a power of 4 (size_log2 is even) + // Should be [ 01 10 ... 10] if N is not a power of 4 (size log2 is odd) + int scale; + for (int i = 0; i < (size_log2+1)/2; i++) begin + scale[i*2 +: 2] = i == size_log2/2 ? 2'b01: 2'b10; + end + return scale; + endfunction + + // Returns the width of the FWD_INV field of the config_tdata bus. + function automatic int fft_fwd_inv_w(int max_fft_size_log2); + return 1; + endfunction + + // Returns the width of the CP_LEN field of the config_tdata bus. + function automatic int fft_cp_len_w(int max_fft_size_log2); + if (CP_ENABLE) begin + // Always the same as NFFT if present + return max_fft_size_log2; + end else begin + return 0; + end + endfunction + + // Returns the width of the NFFT field of the config_tdata bus. + function automatic int fft_nfft_w(int max_fft_size_log2); + return 5; + endfunction + + // Returns the LSB position of the SCAL_SCH field of the config_tdata bus. + function automatic int fft_scale_pos(int max_fft_size_log2); + return fft_fwd_inv_pos(max_fft_size_log2) + 1; + endfunction + + // Returns the LSB position of the FWD_INV field of the config_tdata bus. + function automatic int fft_fwd_inv_pos(int max_fft_size_log2); + if (CP_ENABLE) begin + if (fft_cp_len_pos(max_fft_size_log2) + fft_cp_len_w(max_fft_size_log2) > 16) + return 24; + else + return 16; + end else begin + return 8; + end + endfunction + + // Returns the LSB position of the CP_LEN field of the config_tdata bus. + function automatic int fft_cp_len_pos(int max_fft_size_log2); + return 8; + endfunction + + // Returns the LSB position of the NFFT field of the config_tdata bus. + function automatic int fft_nfft_pos(int max_fft_size_log2); + return 0; + endfunction + + // Returns the width of the config_tdata bus. + function automatic int fft_config_w(int max_fft_size_log2); + // It's the length needed to hold SCALE_SCH rounded up to the nearest byte + return ((fft_scale_w(max_fft_size_log2) + fft_scale_pos(max_fft_size_log2) + 7) / 8) * 8; + endfunction + + // Generates a mask of all ones that is num_bits wide. + function automatic logic [MAX_W-1:0] mask(int num_bits); + logic [MAX_W-1:0] bits; + bits = (1 << num_bits) - 1; + return bits; + endfunction + + // Builds the config_tdata value from the provided settings. + function automatic bit [MAX_W-1:0] build_fft_config( + int max_fft_size_log2, + int scale_sch, + bit fwd_inv, + int nfft, + int cp_len = 0 + ); + bit [MAX_W-1:0] cfg; + assert (max_fft_size_log2 >= 4 && max_fft_size_log2 <= 16) else + $fatal(1, "This FFT size is not yet supported"); + assert (!CP_ENABLE || cp_len == 0) else + $fatal(1, "Cyclic prefix must be 0 if not in use"); + cfg = + ((scale_sch & mask(fft_scale_w (max_fft_size_log2))) << fft_scale_pos (max_fft_size_log2)) | + ((fwd_inv & mask(fft_fwd_inv_w(max_fft_size_log2))) << fft_fwd_inv_pos(max_fft_size_log2)) | + ((cp_len & mask(fft_cp_len_w (max_fft_size_log2))) << fft_cp_len_pos (max_fft_size_log2)) | + ((nfft & mask(fft_nfft_w (max_fft_size_log2))) << fft_nfft_pos (max_fft_size_log2)); + return cfg; + endfunction + + + //synthesis translate_off + + // Takes as input the config_tdata bus and prints the settings encoding on it. + function automatic void print_fft_config(int max_fft_size_log2, logic [MAX_W-1:0] cfg); + int scale_sch, fwd_inv, cp_len, nfft; + + scale_sch = (cfg >> fft_scale_pos (max_fft_size_log2)) & mask(fft_scale_w (max_fft_size_log2)); + fwd_inv = (cfg >> fft_fwd_inv_pos(max_fft_size_log2)) & mask(fft_fwd_inv_w(max_fft_size_log2)); + cp_len = (cfg >> fft_cp_len_pos (max_fft_size_log2)) & mask(fft_cp_len_w (max_fft_size_log2)); + nfft = (cfg >> fft_nfft_pos (max_fft_size_log2)) & mask(fft_nfft_w (max_fft_size_log2)); + + $display("SCALE_SCH_0 : 0b%0b", scale_sch); + $display("FWD_INV_0 : %0d", fwd_inv); + if (CP_ENABLE) $display("CP_LEN : %0d", cp_len); + $display("NFFT : %0d", nfft); + endfunction + + // The output of this function should match what's shown in the + // Implementation Details tab of the Xilinx Fast Fourier Transform IP + // generation wizard. + function automatic void print_fft_config_fields(int max_fft_size_log2); + $display("SCALE_SCH_0(%0d:%0d) bit%0d", + fft_scale_w(max_fft_size_log2) + fft_scale_pos(max_fft_size_log2) - 1, + fft_scale_pos(max_fft_size_log2), fft_scale_w(max_fft_size_log2)); + $display("FWD_INV_0(%0d:%0d) bit%0d", + fft_fwd_inv_w(max_fft_size_log2) + fft_fwd_inv_pos(max_fft_size_log2) - 1, + fft_fwd_inv_pos(max_fft_size_log2), fft_fwd_inv_w(max_fft_size_log2)); + if (CP_ENABLE) begin + $display("CP_LEN(%0d:%0d) uint%0d", + fft_cp_len_w(max_fft_size_log2) + fft_cp_len_pos(max_fft_size_log2) - 1, + fft_cp_len_pos(max_fft_size_log2), fft_cp_len_w(max_fft_size_log2)); + end + $display("NFFT(%0d:%0d) uint%0d", + fft_nfft_w(max_fft_size_log2) + fft_nfft_pos(max_fft_size_log2) - 1, + fft_nfft_pos(max_fft_size_log2), fft_nfft_w(max_fft_size_log2)); + endfunction + + //synthesis translate_on + +endpackage : xfft_config_pkg diff --git a/lib/rfnoc/blocks/rfnoc_block_fft/xfft_wrapper.sv b/lib/rfnoc/blocks/rfnoc_block_fft/xfft_wrapper.sv new file mode 100644 index 0000000..2912202 --- /dev/null +++ b/lib/rfnoc/blocks/rfnoc_block_fft/xfft_wrapper.sv @@ -0,0 +1,251 @@ +// +// Copyright 2024 Ettus Research, a National Instruments Brand +// +// SPDX-License-Identifier: LGPL-3.0-or-later +// +// Module: xfft_wrapper +// +// Description: +// +// Wrapper for the Xilinx FFT core, which allows you to configure the maximum +// FFT size using parameters. +// +// Parameters: +// +// MAX_FFT_SIZE_LOG2 : Log2 of maximum configurable FFT size. That is, the +// max supported FFT size will be 2**MAX_FFT_SIZE_LOG2. +// + +`default_nettype none + + +module xfft_wrapper + import xfft_config_pkg::*; +#( + parameter int MAX_FFT_SIZE_LOG2 = 12, + localparam int FFT_CONFIG_W = fft_config_w(MAX_FFT_SIZE_LOG2) +) ( + input wire aclk, + input wire aresetn, + input wire [FFT_CONFIG_W-1:0] s_axis_config_tdata, + input wire s_axis_config_tvalid, + output wire s_axis_config_tready, + input wire [ 31:0] s_axis_data_tdata, + input wire s_axis_data_tvalid, + output wire s_axis_data_tready, + input wire s_axis_data_tlast, + output wire [ 31:0] m_axis_data_tdata, + output wire [ 23:0] m_axis_data_tuser, + output wire m_axis_data_tvalid, + input wire m_axis_data_tready, + output wire m_axis_data_tlast, + output wire [ 7:0] m_axis_status_tdata, + output wire m_axis_status_tvalid, + input wire m_axis_status_tready, + output wire event_frame_started, + output wire event_tlast_unexpected, + output wire event_tlast_missing, + output wire event_fft_overflow, + output wire event_status_channel_halt, + output wire event_data_in_channel_halt, + output wire event_data_out_channel_halt +); + localparam MAX_FFT_SIZE = 2**MAX_FFT_SIZE_LOG2; + + if (MAX_FFT_SIZE == 1024) begin : gen_1k_fft + xfft_1k_16b xfft_1k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 2048) begin : gen_2k_fft + xfft_2k_16b xfft_2k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 4096) begin : gen_4k_fft + xfft_4k_16b xfft_4k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 8192) begin : gen_8k_fft + xfft_8k_16b xfft_8k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 16384) begin : gen_16k_fft + xfft_16k_16b xfft_16k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 32768) begin : gen_32k_fft + xfft_32k_16b xfft_32k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else if (MAX_FFT_SIZE == 65536) begin : gen_64k_fft + xfft_64k_16b xfft_64k_16b_i ( + .aclk (aclk), + .aresetn (aresetn), + .s_axis_config_tdata (s_axis_config_tdata), + .s_axis_config_tvalid (s_axis_config_tvalid), + .s_axis_config_tready (s_axis_config_tready), + .s_axis_data_tdata (s_axis_data_tdata), + .s_axis_data_tlast (s_axis_data_tlast), + .s_axis_data_tvalid (s_axis_data_tvalid), + .s_axis_data_tready (s_axis_data_tready), + .m_axis_data_tdata (m_axis_data_tdata), + .m_axis_data_tuser (m_axis_data_tuser), + .m_axis_data_tlast (m_axis_data_tlast), + .m_axis_data_tvalid (m_axis_data_tvalid), + .m_axis_data_tready (m_axis_data_tready), + .m_axis_status_tdata (m_axis_status_tdata), + .m_axis_status_tvalid (m_axis_status_tvalid), + .m_axis_status_tready (m_axis_status_tready), + .event_frame_started (event_frame_started), + .event_tlast_unexpected (event_tlast_unexpected), + .event_tlast_missing (event_tlast_missing), + .event_fft_overflow (event_fft_overflow), + .event_status_channel_halt (event_status_channel_halt), + .event_data_in_channel_halt (event_data_in_channel_halt), + .event_data_out_channel_halt (event_data_out_channel_halt) + ); + end else begin + ERROR_Invalid_FFT_parameters(); + end + +endmodule : xfft_wrapper + + +`default_nettype wire