diff --git a/rtl/common/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv new file mode 100644 index 0000000..d55d27e --- /dev/null +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -0,0 +1,145 @@ +// Copyright (c) 2026 Ethan Sifferman +// +// Redistribution and use in source and binary forms, with or without modification, are permitted +// provided that the following conditions are met: +// +// 1. Redistributions of source code must retain the above copyright notice, this list of +// conditions and the following disclaimer. +// +// 2. Redistributions in binary form must reproduce the above copyright notice, this list of +// conditions and the following disclaimer in the documentation and/or other materials provided +// with the distribution. +// +// 3. Neither the name of the copyright holder nor the names of its contributors may be used to +// endorse or promote products derived from this software without specific prior written +// permission. +// +// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR +// IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND +// FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR +// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR +// CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR +// OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +// POSSIBILITY OF SUCH DAMAGE. + + +// ternip_fixed_latency_equalizer +// +// Gives a variable-latency ready/valid core a fixed, data-independent latency. + +module ternip_fixed_latency_equalizer #( + parameter int DataWidth = 16, + parameter int NumCycles = 32 +) ( + input logic clk_i, + input logic rst_ni, + + output logic in_ready_o, + input logic in_valid_i, + + input logic wrappedcore_in_ready_i, + output logic wrappedcore_in_valid_o, + + output logic wrappedcore_out_ready_o, + input logic wrappedcore_out_valid_i, + input logic [DataWidth-1:0] wrappedcore_out_data_i, + + input logic out_ready_i, + output logic out_valid_o, + output logic [DataWidth-1:0] out_data_o +); + +localparam int CounterWidth = $clog2(NumCycles); + +enum logic [1:0] { + WAITING_FOR_IN, + WAITING_FOR_RESULT, + HOLDING_RESULT +} state_d, state_q = WAITING_FOR_IN; // for assertions at time=0 + +logic [CounterWidth-1:0] counter_d, counter_q; +logic [DataWidth-1:0] data_d, data_q; + +wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1)); + +assign out_data_o = data_q; + +always_comb begin + state_d = state_q; + counter_d = counter_q; + data_d = data_q; + + in_ready_o = 0; + wrappedcore_in_valid_o = 0; + wrappedcore_out_ready_o = 1; + out_valid_o = 0; + + // Only pass the input handshake through while idle, so a second transaction + // can never restart the window and cost the latency its independence. + if (state_q == WAITING_FOR_IN) begin + + in_ready_o = wrappedcore_in_ready_i; + wrappedcore_in_valid_o = in_valid_i; + + if (in_ready_o && in_valid_i) begin + state_d = WAITING_FOR_RESULT; + counter_d = '0; + end + + end else if (state_q == WAITING_FOR_RESULT) begin + + counter_d++; // assuming deadline_reached==0 +`ifndef SYNTHESIS + if (deadline_reached) + $fatal(0, "core did not produce a result within %0d cycles", NumCycles); +`endif + + if (wrappedcore_out_ready_o && wrappedcore_out_valid_i) begin + data_d = wrappedcore_out_data_i; + state_d = HOLDING_RESULT; + end + + end else if (state_q == HOLDING_RESULT) begin + wrappedcore_out_ready_o = 0; + + if (deadline_reached) + out_valid_o = 1; + else + counter_d++; + + if (out_ready_i && out_valid_o) + state_d = WAITING_FOR_IN; + + end else begin + state_d = WAITING_FOR_IN; + end +end + +always_ff @(posedge clk_i) begin + if (!rst_ni) begin + state_q <= WAITING_FOR_IN; + counter_q <= '0; + data_q <= '0; + end else begin + state_q <= state_d; + counter_q <= counter_d; + data_q <= data_d; + end +end + +`ifndef SYNTHESIS + +// Ensure that correct latency is achieved +property fixed_latency_p; + @(posedge clk_i) disable iff (!rst_ni) + (in_ready_o && in_valid_i) |-> ##NumCycles out_valid_o; +endproperty + +assert property (fixed_latency_p) + else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles); + +`endif + +endmodule diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 0e89ce0..eff6cf6 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -79,14 +79,10 @@ ternip_fixed_point_convert #( .InPrecision(InAPrecision), .InExponent(InAExponent), .OutPrecision(DivInternalPrecision), - .OutExponent(InternalAInternalExponent) + .OutExponent(InternalAInternalExponent), + .NumPipelineStages(0) ) convert_a ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(a_i), .out(internal_a) ); @@ -95,14 +91,10 @@ ternip_fixed_point_convert #( .InPrecision(InBPrecision), .InExponent(InBExponent), .OutPrecision(DivInternalPrecision), - .OutExponent(InternalBInternalExponent) + .OutExponent(InternalBInternalExponent), + .NumPipelineStages(0) ) convert_b ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(b_i), .out(internal_b) ); @@ -111,61 +103,39 @@ ternip_fixed_point_convert #( .InPrecision(DivInternalPrecision), .InExponent(InternalYInternalExponent), .OutPrecision(OutPrecision), - .OutExponent(OutExponent) + .OutExponent(OutExponent), + .NumPipelineStages(0) ) convert_out ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(internal_y), .out(convert_out_y) ); // Fixed-latency equalizer: the iterative divider takes a data-dependent number of -// cycles, which would desync callers that run several dividers in lockstep. The -// divider holds its quotient until yumi, so a single counter just delays the result -// by a fixed, data-independent number of cycles -- long enough that every divide is done. +// cycles, long enough that every divider in all cores is done localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; -localparam int EqualizerCounterWidth = $clog2(RawDivideLatencyUpperBound + 1); - -// counter == 0 is idle; a divide counts to RawDivideLatencyUpperBound, then signals done. -logic [EqualizerCounterWidth-1:0] equalizer_counter_d, equalizer_counter_q; -wire equalizer_idle = (equalizer_counter_q == '0); -wire equalizer_done = (equalizer_counter_q >= EqualizerCounterWidth'(RawDivideLatencyUpperBound)); - -assign raw_in_valid = in_valid_i && equalizer_idle; -assign in_ready_o = raw_in_ready && equalizer_idle; -assign div_out_ready = equalizer_done && out_ready_i; -assign out_valid_o = equalizer_done; -assign y_o = convert_out_y; - -always_comb begin - equalizer_counter_d = equalizer_counter_q; - if (equalizer_idle) begin - if (in_valid_i && in_ready_o) equalizer_counter_d = 1; - end else if (equalizer_done) begin - if (out_ready_i) equalizer_counter_d = '0; - end else begin - equalizer_counter_d = equalizer_counter_q + 1; - end -end -always_ff @(posedge clk_i) begin - if (!rst_ni) equalizer_counter_q <= '0; - else equalizer_counter_q <= equalizer_counter_d; -end +ternip_fixed_latency_equalizer #( + .DataWidth(OutPrecision), + .NumCycles(RawDivideLatencyUpperBound + 1) +) equalizer ( + .clk_i, + .rst_ni, -`ifndef SYNTHESIS -// The fixed window must be long enough that the raw quotient always arrives -// before it elapses; otherwise the latency would become data-dependent again. -always @(posedge clk_i) if (rst_ni) begin - assert (!equalizer_done || div_out_valid) - else $fatal(0, "ternip_div equalizer window (%0d) shorter than raw divide latency", - RawDivideLatencyUpperBound); -end -`endif + .in_ready_o, + .in_valid_i, + + .wrappedcore_in_ready_i(raw_in_ready), + .wrappedcore_in_valid_o(raw_in_valid), + + .wrappedcore_out_ready_o(div_out_ready), + .wrappedcore_out_valid_i(div_out_valid), + .wrappedcore_out_data_i(convert_out_y), + + .out_ready_i, + .out_valid_o, + .out_data_o(y_o) +); if (Implementation == ternip_pkg::DIV_BSG) begin : div_bsg diff --git a/rtl/math/ternip_fixed_point_convert.sv b/rtl/math/ternip_fixed_point_convert.sv index 09b9be2..79101c5 100644 --- a/rtl/math/ternip_fixed_point_convert.sv +++ b/rtl/math/ternip_fixed_point_convert.sv @@ -47,13 +47,13 @@ module ternip_fixed_point_convert #( input logic clk_i, input logic rst_ni, - input logic signed [InPrecision-1:0] in, - input logic in_valid_i, output logic in_ready_o, + input logic in_valid_i, + input logic signed [InPrecision-1:0] in, - output logic signed [OutPrecision-1:0] out, + input logic out_ready_i, output logic out_valid_o, - input logic out_ready_i + output logic signed [OutPrecision-1:0] out ); if (InPrecision < 1) $fatal(0, "InPrecision (%0d) must be positive", InPrecision);