From fc9ab7af65728655a03b75013122d4a8115e3c4e Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Thu, 20 Aug 2026 14:31:44 -0700 Subject: [PATCH 01/10] register the divider equalizer output --- rtl/math/ternip_div.sv | 76 +++++++++++++++++++++++++++++++----------- 1 file changed, 56 insertions(+), 20 deletions(-) diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 0e89ce0..8cc96b8 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -130,38 +130,74 @@ ternip_fixed_point_convert #( localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; localparam int EqualizerCounterWidth = $clog2(RawDivideLatencyUpperBound + 1); -// counter == 0 is idle; a divide counts to RawDivideLatencyUpperBound, then signals done. -logic [EqualizerCounterWidth-1:0] equalizer_counter_d, equalizer_counter_q; -wire equalizer_idle = (equalizer_counter_q == '0); -wire equalizer_done = (equalizer_counter_q >= EqualizerCounterWidth'(RawDivideLatencyUpperBound)); - -assign raw_in_valid = in_valid_i && equalizer_idle; -assign in_ready_o = raw_in_ready && equalizer_idle; -assign div_out_ready = equalizer_done && out_ready_i; -assign out_valid_o = equalizer_done; -assign y_o = convert_out_y; +// The quotient is captured into equalizer_result_q as soon as the raw divider +// presents it, and held until the window elapses. Registering it here (rather +// than driving y_o straight from convert_out_y) also breaks the combinational +// path from the divider's held quotient through the output requantize into the +// consumer's own input conversion, which is the critical path out of rms. +logic equalizer_busy_d, equalizer_busy_q; +logic equalizer_captured_d, equalizer_captured_q; +logic [EqualizerCounterWidth-1:0] equalizer_counter_d, equalizer_counter_q; +logic signed [OutPrecision-1:0] equalizer_result_d, equalizer_result_q; + +wire equalizer_accept = in_valid_i && in_ready_o; +wire equalizer_capture = equalizer_busy_q && div_out_valid && !equalizer_captured_q; +wire equalizer_result_present + = equalizer_busy_q + && (equalizer_counter_q >= EqualizerCounterWidth'(RawDivideLatencyUpperBound)); + +assign raw_in_valid = in_valid_i && !equalizer_busy_q; +assign in_ready_o = raw_in_ready && !equalizer_busy_q; +// Drain the raw divider exactly once, when its quotient first becomes valid. +assign div_out_ready = !equalizer_captured_q; + +assign out_valid_o = equalizer_result_present; +assign y_o = equalizer_result_q; always_comb begin - equalizer_counter_d = equalizer_counter_q; - if (equalizer_idle) begin - if (in_valid_i && in_ready_o) equalizer_counter_d = 1; - end else if (equalizer_done) begin - if (out_ready_i) equalizer_counter_d = '0; - end else begin - equalizer_counter_d = equalizer_counter_q + 1; + equalizer_busy_d = equalizer_busy_q; + equalizer_captured_d = equalizer_captured_q; + equalizer_counter_d = equalizer_counter_q; + equalizer_result_d = equalizer_result_q; + + if (equalizer_accept) begin + equalizer_busy_d = 1; + equalizer_captured_d = 0; + equalizer_counter_d = '0; + end else if (equalizer_busy_q) begin + if (equalizer_counter_q < EqualizerCounterWidth'(RawDivideLatencyUpperBound)) + equalizer_counter_d = equalizer_counter_q + 1; + if (equalizer_result_present && out_ready_i) begin + equalizer_busy_d = 0; + equalizer_captured_d = 0; + end + end + + if (equalizer_capture) begin + equalizer_captured_d = 1; + equalizer_result_d = convert_out_y; end end always_ff @(posedge clk_i) begin - if (!rst_ni) equalizer_counter_q <= '0; - else equalizer_counter_q <= equalizer_counter_d; + if (!rst_ni) begin + equalizer_busy_q <= 0; + equalizer_captured_q <= 0; + equalizer_counter_q <= '0; + equalizer_result_q <= '0; + end else begin + equalizer_busy_q <= equalizer_busy_d; + equalizer_captured_q <= equalizer_captured_d; + equalizer_counter_q <= equalizer_counter_d; + equalizer_result_q <= equalizer_result_d; + end end `ifndef SYNTHESIS // The fixed window must be long enough that the raw quotient always arrives // before it elapses; otherwise the latency would become data-dependent again. always @(posedge clk_i) if (rst_ni) begin - assert (!equalizer_done || div_out_valid) + assert (!equalizer_result_present || equalizer_captured_q) else $fatal(0, "ternip_div equalizer window (%0d) shorter than raw divide latency", RawDivideLatencyUpperBound); end From cb009e0c7606f7b7fc14295664d554c015ab6054 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 09:40:33 -0700 Subject: [PATCH 02/10] extract fixed-latency equalizer with latency assertion --- rtl/math/ternip_div.sv | 123 ++++------------ rtl/math/ternip_fixed_latency_equalizer.sv | 163 +++++++++++++++++++++ 2 files changed, 191 insertions(+), 95 deletions(-) create mode 100644 rtl/math/ternip_fixed_latency_equalizer.sv diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 8cc96b8..67c6aea 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -79,14 +79,10 @@ ternip_fixed_point_convert #( .InPrecision(InAPrecision), .InExponent(InAExponent), .OutPrecision(DivInternalPrecision), - .OutExponent(InternalAInternalExponent) + .OutExponent(InternalAInternalExponent), + .NumPipelineStages(0) ) convert_a ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational .in(a_i), .out(internal_a) ); @@ -95,14 +91,10 @@ ternip_fixed_point_convert #( .InPrecision(InBPrecision), .InExponent(InBExponent), .OutPrecision(DivInternalPrecision), - .OutExponent(InternalBInternalExponent) + .OutExponent(InternalBInternalExponent), + .NumPipelineStages(0) ) convert_b ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational .in(b_i), .out(internal_b) ); @@ -111,97 +103,38 @@ ternip_fixed_point_convert #( .InPrecision(DivInternalPrecision), .InExponent(InternalYInternalExponent), .OutPrecision(OutPrecision), - .OutExponent(OutExponent) + .OutExponent(OutExponent), + .NumPipelineStages(0) ) convert_out ( - .clk_i, - .rst_ni, - .in_valid_i(1'b1), - .in_ready_o(), - .out_valid_o(), - .out_ready_i(1'b1), + .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational .in(internal_y), .out(convert_out_y) ); // Fixed-latency equalizer: the iterative divider takes a data-dependent number of -// cycles, which would desync callers that run several dividers in lockstep. The -// divider holds its quotient until yumi, so a single counter just delays the result -// by a fixed, data-independent number of cycles -- long enough that every divide is done. +// cycles, long enough that every divider in all cores is done localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; -localparam int EqualizerCounterWidth = $clog2(RawDivideLatencyUpperBound + 1); - -// The quotient is captured into equalizer_result_q as soon as the raw divider -// presents it, and held until the window elapses. Registering it here (rather -// than driving y_o straight from convert_out_y) also breaks the combinational -// path from the divider's held quotient through the output requantize into the -// consumer's own input conversion, which is the critical path out of rms. -logic equalizer_busy_d, equalizer_busy_q; -logic equalizer_captured_d, equalizer_captured_q; -logic [EqualizerCounterWidth-1:0] equalizer_counter_d, equalizer_counter_q; -logic signed [OutPrecision-1:0] equalizer_result_d, equalizer_result_q; - -wire equalizer_accept = in_valid_i && in_ready_o; -wire equalizer_capture = equalizer_busy_q && div_out_valid && !equalizer_captured_q; -wire equalizer_result_present - = equalizer_busy_q - && (equalizer_counter_q >= EqualizerCounterWidth'(RawDivideLatencyUpperBound)); - -assign raw_in_valid = in_valid_i && !equalizer_busy_q; -assign in_ready_o = raw_in_ready && !equalizer_busy_q; -// Drain the raw divider exactly once, when its quotient first becomes valid. -assign div_out_ready = !equalizer_captured_q; - -assign out_valid_o = equalizer_result_present; -assign y_o = equalizer_result_q; - -always_comb begin - equalizer_busy_d = equalizer_busy_q; - equalizer_captured_d = equalizer_captured_q; - equalizer_counter_d = equalizer_counter_q; - equalizer_result_d = equalizer_result_q; - - if (equalizer_accept) begin - equalizer_busy_d = 1; - equalizer_captured_d = 0; - equalizer_counter_d = '0; - end else if (equalizer_busy_q) begin - if (equalizer_counter_q < EqualizerCounterWidth'(RawDivideLatencyUpperBound)) - equalizer_counter_d = equalizer_counter_q + 1; - if (equalizer_result_present && out_ready_i) begin - equalizer_busy_d = 0; - equalizer_captured_d = 0; - end - end - if (equalizer_capture) begin - equalizer_captured_d = 1; - equalizer_result_d = convert_out_y; - end -end +ternip_fixed_latency_equalizer #( + .DataWidth(OutPrecision), + .LatencyUpperBound(RawDivideLatencyUpperBound) +) equalizer ( + .clk_i, + .rst_ni, -always_ff @(posedge clk_i) begin - if (!rst_ni) begin - equalizer_busy_q <= 0; - equalizer_captured_q <= 0; - equalizer_counter_q <= '0; - equalizer_result_q <= '0; - end else begin - equalizer_busy_q <= equalizer_busy_d; - equalizer_captured_q <= equalizer_captured_d; - equalizer_counter_q <= equalizer_counter_d; - equalizer_result_q <= equalizer_result_d; - end -end + .in_valid_i, + .in_ready_o, -`ifndef SYNTHESIS -// The fixed window must be long enough that the raw quotient always arrives -// before it elapses; otherwise the latency would become data-dependent again. -always @(posedge clk_i) if (rst_ni) begin - assert (!equalizer_result_present || equalizer_captured_q) - else $fatal(0, "ternip_div equalizer window (%0d) shorter than raw divide latency", - RawDivideLatencyUpperBound); -end -`endif + .core_in_valid_o(raw_in_valid), + .core_in_ready_i(raw_in_ready), + .core_out_valid_i(div_out_valid), + .core_out_ready_o(div_out_ready), + .core_data_i(convert_out_y), + + .data_o(y_o), + .out_valid_o, + .out_ready_i +); if (Implementation == ternip_pkg::DIV_BSG) begin : div_bsg diff --git a/rtl/math/ternip_fixed_latency_equalizer.sv b/rtl/math/ternip_fixed_latency_equalizer.sv new file mode 100644 index 0000000..1095306 --- /dev/null +++ b/rtl/math/ternip_fixed_latency_equalizer.sv @@ -0,0 +1,163 @@ +// Copyright (c) 2026 Ethan Sifferman +// +// Redistribution and use in source and binary forms, with or without modification, are permitted +// provided that the following conditions are met: +// +// 1. Redistributions of source code must retain the above copyright notice, this list of +// conditions and the following disclaimer. +// +// 2. Redistributions in binary form must reproduce the above copyright notice, this list of +// conditions and the following disclaimer in the documentation and/or other materials provided +// with the distribution. +// +// 3. Neither the name of the copyright holder nor the names of its contributors may be used to +// endorse or promote products derived from this software without specific prior written +// permission. +// +// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR +// IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND +// FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR +// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR +// CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR +// OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +// POSSIBILITY OF SUCH DAMAGE. + + +// ternip_fixed_latency_equalizer +// +// Gives a variable-latency ready/valid core a fixed, data-independent latency. +// +// A core whose latency depends on its data (an iterative divider, say) desyncs +// callers that run several of them in lockstep. This holds each result until a +// fixed deadline instead, so every instance retires on the same cycle. +// +// One transaction in flight at a time. out_valid_o rises exactly +// LatencyUpperBound+1 cycles after each accepted input, so LatencyUpperBound +// must be at least the wrapped core's worst-case latency. +// +// data_o is driven straight from a register: the point of holding the result is +// to keep the core's combinational output cone out of the consumer's input path. + +module ternip_fixed_latency_equalizer #( + parameter int DataWidth = 16, + parameter int LatencyUpperBound = 32 +) ( + input logic clk_i, + input logic rst_ni, + + input logic in_valid_i, + output logic in_ready_o, + + output logic core_in_valid_o, + input logic core_in_ready_i, + input logic core_out_valid_i, + output logic core_out_ready_o, + input logic [DataWidth-1:0] core_data_i, + + output logic [DataWidth-1:0] data_o, + output logic out_valid_o, + input logic out_ready_i +); + +localparam int CounterWidth = $clog2(LatencyUpperBound + 1); + +enum logic [1:0] { + WAITING_FOR_IN, + WAITING_FOR_CORE, + HOLDING_RESULT +} state_d, state_q = WAITING_FOR_IN; // for assertions at time=0 + +logic [CounterWidth-1:0] counter_d, counter_q; +logic [DataWidth-1:0] data_d, data_q; + +wire deadline_reached = (counter_q >= CounterWidth'(LatencyUpperBound)); + +assign data_o = data_q; + +always_comb begin + state_d = state_q; + counter_d = counter_q; + data_d = data_q; + + core_in_valid_o = 0; + in_ready_o = 0; + core_out_ready_o = 1; + out_valid_o = 0; + + if (state_q == WAITING_FOR_IN) begin + + core_in_valid_o = in_valid_i; + in_ready_o = core_in_ready_i; + + if (in_valid_i && in_ready_o) begin + state_d = WAITING_FOR_CORE; + counter_d = '0; + end + + end else if (state_q == WAITING_FOR_CORE) begin + + if (!deadline_reached) + counter_d++; + + // Drain the core exactly once, when its result first becomes valid. + if (core_out_valid_i && core_out_ready_o) begin + data_d = core_data_i; + state_d = HOLDING_RESULT; + end + + // The core blew the deadline, so LatencyUpperBound is too small. Retire + // on schedule regardless to keep lockstep peers aligned; data_o is stale. + // Deliberately does not wait for out_ready_i -- this path is fatal. + if (state_d == WAITING_FOR_CORE && deadline_reached) begin + state_d = WAITING_FOR_IN; + out_valid_o = 1; +`ifndef SYNTHESIS + $fatal(0, "equalizer window (%0d) shorter than core latency", LatencyUpperBound); +`endif + end + + end else if (state_q == HOLDING_RESULT) begin + core_out_ready_o = 0; + + if (deadline_reached) + out_valid_o = 1; + else + counter_d++; + + if (out_valid_o && out_ready_i) + state_d = WAITING_FOR_IN; + + end else begin + state_d = WAITING_FOR_IN; + end +end + +always_ff @(posedge clk_i) begin + if (!rst_ni) begin + state_q <= WAITING_FOR_IN; + counter_q <= '0; + data_q <= '0; + end else begin + state_q <= state_d; + counter_q <= counter_d; + data_q <= data_d; + end +end + +`ifndef SYNTHESIS +// The whole point of this module is that latency does not depend on the data, so +// check it end to end rather than trusting the state machine not to add a cycle. +localparam int EqualizerLatency = LatencyUpperBound + 1; + +property fixed_latency_p; + @(posedge clk_i) disable iff (!rst_ni) + (in_valid_i && in_ready_o) |-> ##EqualizerLatency out_valid_o; +endproperty + +assert property (fixed_latency_p) + else $fatal(0, "equalizer latency deviated from the fixed %0d cycles", EqualizerLatency); +`endif + +endmodule From a9711361c5448214cd4a66a31902ef6674977e02 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 10:00:55 -0700 Subject: [PATCH 03/10] name equalizer ports by equalized and unequalized result --- rtl/math/ternip_div.sv | 19 ++--- rtl/math/ternip_fixed_latency_equalizer.sv | 94 ++++++++++------------ 2 files changed, 51 insertions(+), 62 deletions(-) diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 67c6aea..97a477d 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -117,7 +117,7 @@ localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; ternip_fixed_latency_equalizer #( .DataWidth(OutPrecision), - .LatencyUpperBound(RawDivideLatencyUpperBound) + .NumCycles(RawDivideLatencyUpperBound + 1) ) equalizer ( .clk_i, .rst_ni, @@ -125,15 +125,16 @@ ternip_fixed_latency_equalizer #( .in_valid_i, .in_ready_o, - .core_in_valid_o(raw_in_valid), - .core_in_ready_i(raw_in_ready), - .core_out_valid_i(div_out_valid), - .core_out_ready_o(div_out_ready), - .core_data_i(convert_out_y), + .unequalized_in_valid_o(raw_in_valid), + .unequalized_in_ready_i(raw_in_ready), - .data_o(y_o), - .out_valid_o, - .out_ready_i + .unequalized_result_valid_i(div_out_valid), + .unequalized_result_ready_o(div_out_ready), + .unequalized_result_data_i(convert_out_y), + + .equalized_result_valid_o(out_valid_o), + .equalized_result_ready_i(out_ready_i), + .equalized_result_data_o(y_o) ); if (Implementation == ternip_pkg::DIV_BSG) begin : div_bsg diff --git a/rtl/math/ternip_fixed_latency_equalizer.sv b/rtl/math/ternip_fixed_latency_equalizer.sv index 1095306..95294a2 100644 --- a/rtl/math/ternip_fixed_latency_equalizer.sv +++ b/rtl/math/ternip_fixed_latency_equalizer.sv @@ -28,105 +28,95 @@ // ternip_fixed_latency_equalizer // // Gives a variable-latency ready/valid core a fixed, data-independent latency. -// -// A core whose latency depends on its data (an iterative divider, say) desyncs -// callers that run several of them in lockstep. This holds each result until a -// fixed deadline instead, so every instance retires on the same cycle. -// -// One transaction in flight at a time. out_valid_o rises exactly -// LatencyUpperBound+1 cycles after each accepted input, so LatencyUpperBound -// must be at least the wrapped core's worst-case latency. -// -// data_o is driven straight from a register: the point of holding the result is -// to keep the core's combinational output cone out of the consumer's input path. module ternip_fixed_latency_equalizer #( - parameter int DataWidth = 16, - parameter int LatencyUpperBound = 32 + parameter int DataWidth = 16, + parameter int NumCycles = 32 ) ( - input logic clk_i, - input logic rst_ni, + input logic clk_i, + input logic rst_ni, - input logic in_valid_i, - output logic in_ready_o, + input logic in_valid_i, + output logic in_ready_o, - output logic core_in_valid_o, - input logic core_in_ready_i, - input logic core_out_valid_i, - output logic core_out_ready_o, - input logic [DataWidth-1:0] core_data_i, + output logic unequalized_in_valid_o, + input logic unequalized_in_ready_i, - output logic [DataWidth-1:0] data_o, - output logic out_valid_o, - input logic out_ready_i + input logic unequalized_result_valid_i, + output logic unequalized_result_ready_o, + input logic [DataWidth-1:0] unequalized_result_data_i, + + output logic equalized_result_valid_o, + input logic equalized_result_ready_i, + output logic [DataWidth-1:0] equalized_result_data_o ); -localparam int CounterWidth = $clog2(LatencyUpperBound + 1); +localparam int CounterWidth = $clog2(NumCycles); enum logic [1:0] { WAITING_FOR_IN, - WAITING_FOR_CORE, + WAITING_FOR_RESULT, HOLDING_RESULT } state_d, state_q = WAITING_FOR_IN; // for assertions at time=0 logic [CounterWidth-1:0] counter_d, counter_q; logic [DataWidth-1:0] data_d, data_q; -wire deadline_reached = (counter_q >= CounterWidth'(LatencyUpperBound)); +wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1)); -assign data_o = data_q; +assign equalized_result_data_o = data_q; always_comb begin state_d = state_q; counter_d = counter_q; data_d = data_q; - core_in_valid_o = 0; - in_ready_o = 0; - core_out_ready_o = 1; - out_valid_o = 0; + unequalized_in_valid_o = 0; + in_ready_o = 0; + unequalized_result_ready_o = 1; + equalized_result_valid_o = 0; if (state_q == WAITING_FOR_IN) begin - core_in_valid_o = in_valid_i; - in_ready_o = core_in_ready_i; + unequalized_in_valid_o = in_valid_i; + in_ready_o = unequalized_in_ready_i; if (in_valid_i && in_ready_o) begin - state_d = WAITING_FOR_CORE; + state_d = WAITING_FOR_RESULT; counter_d = '0; end - end else if (state_q == WAITING_FOR_CORE) begin + end else if (state_q == WAITING_FOR_RESULT) begin if (!deadline_reached) counter_d++; // Drain the core exactly once, when its result first becomes valid. - if (core_out_valid_i && core_out_ready_o) begin - data_d = core_data_i; + if (unequalized_result_valid_i && unequalized_result_ready_o) begin + data_d = unequalized_result_data_i; state_d = HOLDING_RESULT; end - // The core blew the deadline, so LatencyUpperBound is too small. Retire - // on schedule regardless to keep lockstep peers aligned; data_o is stale. - // Deliberately does not wait for out_ready_i -- this path is fatal. - if (state_d == WAITING_FOR_CORE && deadline_reached) begin - state_d = WAITING_FOR_IN; - out_valid_o = 1; + // The core blew the deadline, so NumCycles is too small. Retire on + // schedule regardless to keep lockstep peers aligned; the data is stale. + // Deliberately does not wait for ready -- this path is fatal. + if (state_d == WAITING_FOR_RESULT && deadline_reached) begin + state_d = WAITING_FOR_IN; + equalized_result_valid_o = 1; `ifndef SYNTHESIS - $fatal(0, "equalizer window (%0d) shorter than core latency", LatencyUpperBound); + $fatal(0, "core did not produce a result within %0d cycles", NumCycles); `endif end end else if (state_q == HOLDING_RESULT) begin - core_out_ready_o = 0; + unequalized_result_ready_o = 0; if (deadline_reached) - out_valid_o = 1; + equalized_result_valid_o = 1; else counter_d++; - if (out_valid_o && out_ready_i) + if (equalized_result_valid_o && equalized_result_ready_i) state_d = WAITING_FOR_IN; end else begin @@ -149,15 +139,13 @@ end `ifndef SYNTHESIS // The whole point of this module is that latency does not depend on the data, so // check it end to end rather than trusting the state machine not to add a cycle. -localparam int EqualizerLatency = LatencyUpperBound + 1; - property fixed_latency_p; @(posedge clk_i) disable iff (!rst_ni) - (in_valid_i && in_ready_o) |-> ##EqualizerLatency out_valid_o; + (in_valid_i && in_ready_o) |-> ##NumCycles equalized_result_valid_o; endproperty assert property (fixed_latency_p) - else $fatal(0, "equalizer latency deviated from the fixed %0d cycles", EqualizerLatency); + else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles); `endif endmodule From c26a476cdcb7e541047a716350023ecb7874d265 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 10:25:10 -0700 Subject: [PATCH 04/10] gate equalizer input with idle instead of passthrough --- rtl/math/ternip_div.sv | 13 ++++++---- rtl/math/ternip_fixed_latency_equalizer.sv | 30 +++++++++++++--------- 2 files changed, 26 insertions(+), 17 deletions(-) diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 97a477d..65e8efd 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -115,6 +115,12 @@ ternip_fixed_point_convert #( // cycles, long enough that every divider in all cores is done localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; +logic equalizer_idle; + +// Only one divide in flight, so the equalizer's window is never restarted early. +assign raw_in_valid = in_valid_i && equalizer_idle; +assign in_ready_o = raw_in_ready && equalizer_idle; + ternip_fixed_latency_equalizer #( .DataWidth(OutPrecision), .NumCycles(RawDivideLatencyUpperBound + 1) @@ -122,11 +128,8 @@ ternip_fixed_latency_equalizer #( .clk_i, .rst_ni, - .in_valid_i, - .in_ready_o, - - .unequalized_in_valid_o(raw_in_valid), - .unequalized_in_ready_i(raw_in_ready), + .idle_o(equalizer_idle), + .in_accepted_i(in_valid_i && in_ready_o), .unequalized_result_valid_i(div_out_valid), .unequalized_result_ready_o(div_out_ready), diff --git a/rtl/math/ternip_fixed_latency_equalizer.sv b/rtl/math/ternip_fixed_latency_equalizer.sv index 95294a2..d8f8a4a 100644 --- a/rtl/math/ternip_fixed_latency_equalizer.sv +++ b/rtl/math/ternip_fixed_latency_equalizer.sv @@ -28,6 +28,10 @@ // ternip_fixed_latency_equalizer // // Gives a variable-latency ready/valid core a fixed, data-independent latency. +// +// Only one transaction may be in flight: the caller must gate its own input +// handshake with idle_o, or the window restarts mid-flight and the latency stops +// being fixed. Asserted below rather than enforced structurally. module ternip_fixed_latency_equalizer #( parameter int DataWidth = 16, @@ -36,11 +40,8 @@ module ternip_fixed_latency_equalizer #( input logic clk_i, input logic rst_ni, - input logic in_valid_i, - output logic in_ready_o, - - output logic unequalized_in_valid_o, - input logic unequalized_in_ready_i, + output logic idle_o, + input logic in_accepted_i, input logic unequalized_result_valid_i, output logic unequalized_result_ready_o, @@ -64,6 +65,7 @@ logic [DataWidth-1:0] data_d, data_q; wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1)); +assign idle_o = (state_q == WAITING_FOR_IN); assign equalized_result_data_o = data_q; always_comb begin @@ -71,17 +73,12 @@ always_comb begin counter_d = counter_q; data_d = data_q; - unequalized_in_valid_o = 0; - in_ready_o = 0; unequalized_result_ready_o = 1; equalized_result_valid_o = 0; if (state_q == WAITING_FOR_IN) begin - unequalized_in_valid_o = in_valid_i; - in_ready_o = unequalized_in_ready_i; - - if (in_valid_i && in_ready_o) begin + if (in_accepted_i) begin state_d = WAITING_FOR_RESULT; counter_d = '0; end @@ -141,11 +138,20 @@ end // check it end to end rather than trusting the state machine not to add a cycle. property fixed_latency_p; @(posedge clk_i) disable iff (!rst_ni) - (in_valid_i && in_ready_o) |-> ##NumCycles equalized_result_valid_o; + in_accepted_i |-> ##NumCycles equalized_result_valid_o; endproperty assert property (fixed_latency_p) else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles); + +// The caller owns the one-in-flight rule, so check that it kept it. +property one_in_flight_p; + @(posedge clk_i) disable iff (!rst_ni) + in_accepted_i |-> idle_o; +endproperty + +assert property (one_in_flight_p) + else $fatal(0, "input accepted while a transaction was still in flight"); `endif endmodule From 6bcb65882509b5ceafb3e9faac0799b32b10a369 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 10:41:57 -0700 Subject: [PATCH 05/10] move fixed-latency equalizer to common and drop dead path --- .../ternip_fixed_latency_equalizer.sv | 19 ++++++++----------- 1 file changed, 8 insertions(+), 11 deletions(-) rename rtl/{math => common}/ternip_fixed_latency_equalizer.sv (90%) diff --git a/rtl/math/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv similarity index 90% rename from rtl/math/ternip_fixed_latency_equalizer.sv rename to rtl/common/ternip_fixed_latency_equalizer.sv index d8f8a4a..d1151ea 100644 --- a/rtl/math/ternip_fixed_latency_equalizer.sv +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -94,17 +94,6 @@ always_comb begin state_d = HOLDING_RESULT; end - // The core blew the deadline, so NumCycles is too small. Retire on - // schedule regardless to keep lockstep peers aligned; the data is stale. - // Deliberately does not wait for ready -- this path is fatal. - if (state_d == WAITING_FOR_RESULT && deadline_reached) begin - state_d = WAITING_FOR_IN; - equalized_result_valid_o = 1; -`ifndef SYNTHESIS - $fatal(0, "core did not produce a result within %0d cycles", NumCycles); -`endif - end - end else if (state_q == HOLDING_RESULT) begin unequalized_result_ready_o = 0; @@ -134,6 +123,14 @@ always_ff @(posedge clk_i) begin end `ifndef SYNTHESIS +// A core that misses the deadline makes the latency data-dependent again, which +// means NumCycles is too small for it. Sampled on the clock rather than checked +// inside the always_comb, so a transient evaluation cannot trip it. +always @(posedge clk_i) if (rst_ni) begin + assert (!(state_q == WAITING_FOR_RESULT && deadline_reached)) + else $fatal(0, "core did not produce a result within %0d cycles", NumCycles); +end + // The whole point of this module is that latency does not depend on the data, so // check it end to end rather than trusting the state machine not to add a cycle. property fixed_latency_p; From 6b01e428bf7b215d54e523128977c878deee04c0 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 10:56:48 -0700 Subject: [PATCH 06/10] restore equalizer input handshake and drop idle --- rtl/common/ternip_fixed_latency_equalizer.sv | 32 +++++++++----------- rtl/math/ternip_div.sv | 13 +++----- 2 files changed, 19 insertions(+), 26 deletions(-) diff --git a/rtl/common/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv index d1151ea..f6e92d6 100644 --- a/rtl/common/ternip_fixed_latency_equalizer.sv +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -28,10 +28,6 @@ // ternip_fixed_latency_equalizer // // Gives a variable-latency ready/valid core a fixed, data-independent latency. -// -// Only one transaction may be in flight: the caller must gate its own input -// handshake with idle_o, or the window restarts mid-flight and the latency stops -// being fixed. Asserted below rather than enforced structurally. module ternip_fixed_latency_equalizer #( parameter int DataWidth = 16, @@ -40,8 +36,11 @@ module ternip_fixed_latency_equalizer #( input logic clk_i, input logic rst_ni, - output logic idle_o, - input logic in_accepted_i, + input logic in_valid_i, + output logic in_ready_o, + + output logic unequalized_in_valid_o, + input logic unequalized_in_ready_i, input logic unequalized_result_valid_i, output logic unequalized_result_ready_o, @@ -65,7 +64,6 @@ logic [DataWidth-1:0] data_d, data_q; wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1)); -assign idle_o = (state_q == WAITING_FOR_IN); assign equalized_result_data_o = data_q; always_comb begin @@ -73,12 +71,19 @@ always_comb begin counter_d = counter_q; data_d = data_q; + unequalized_in_valid_o = 0; + in_ready_o = 0; unequalized_result_ready_o = 1; equalized_result_valid_o = 0; + // Only pass the input handshake through while idle, so a second transaction + // can never restart the window and cost the latency its independence. if (state_q == WAITING_FOR_IN) begin - if (in_accepted_i) begin + unequalized_in_valid_o = in_valid_i; + in_ready_o = unequalized_in_ready_i; + + if (in_valid_i && in_ready_o) begin state_d = WAITING_FOR_RESULT; counter_d = '0; end @@ -135,20 +140,11 @@ end // check it end to end rather than trusting the state machine not to add a cycle. property fixed_latency_p; @(posedge clk_i) disable iff (!rst_ni) - in_accepted_i |-> ##NumCycles equalized_result_valid_o; + (in_valid_i && in_ready_o) |-> ##NumCycles equalized_result_valid_o; endproperty assert property (fixed_latency_p) else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles); - -// The caller owns the one-in-flight rule, so check that it kept it. -property one_in_flight_p; - @(posedge clk_i) disable iff (!rst_ni) - in_accepted_i |-> idle_o; -endproperty - -assert property (one_in_flight_p) - else $fatal(0, "input accepted while a transaction was still in flight"); `endif endmodule diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 65e8efd..97a477d 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -115,12 +115,6 @@ ternip_fixed_point_convert #( // cycles, long enough that every divider in all cores is done localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16; -logic equalizer_idle; - -// Only one divide in flight, so the equalizer's window is never restarted early. -assign raw_in_valid = in_valid_i && equalizer_idle; -assign in_ready_o = raw_in_ready && equalizer_idle; - ternip_fixed_latency_equalizer #( .DataWidth(OutPrecision), .NumCycles(RawDivideLatencyUpperBound + 1) @@ -128,8 +122,11 @@ ternip_fixed_latency_equalizer #( .clk_i, .rst_ni, - .idle_o(equalizer_idle), - .in_accepted_i(in_valid_i && in_ready_o), + .in_valid_i, + .in_ready_o, + + .unequalized_in_valid_o(raw_in_valid), + .unequalized_in_ready_i(raw_in_ready), .unequalized_result_valid_i(div_out_valid), .unequalized_result_ready_o(div_out_ready), From 46ae2615f4b395f6ec3c9b9cece4dbbcdf675041 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 11:10:34 -0700 Subject: [PATCH 07/10] name core-facing equalizer ports wrappedcore in and out --- rtl/common/ternip_fixed_latency_equalizer.sv | 24 ++++++++++---------- rtl/math/ternip_div.sv | 10 ++++---- 2 files changed, 17 insertions(+), 17 deletions(-) diff --git a/rtl/common/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv index f6e92d6..9ca709a 100644 --- a/rtl/common/ternip_fixed_latency_equalizer.sv +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -39,12 +39,12 @@ module ternip_fixed_latency_equalizer #( input logic in_valid_i, output logic in_ready_o, - output logic unequalized_in_valid_o, - input logic unequalized_in_ready_i, + output logic wrappedcore_in_valid_o, + input logic wrappedcore_in_ready_i, - input logic unequalized_result_valid_i, - output logic unequalized_result_ready_o, - input logic [DataWidth-1:0] unequalized_result_data_i, + input logic wrappedcore_out_valid_i, + output logic wrappedcore_out_ready_o, + input logic [DataWidth-1:0] wrappedcore_out_data_i, output logic equalized_result_valid_o, input logic equalized_result_ready_i, @@ -71,17 +71,17 @@ always_comb begin counter_d = counter_q; data_d = data_q; - unequalized_in_valid_o = 0; + wrappedcore_in_valid_o = 0; in_ready_o = 0; - unequalized_result_ready_o = 1; + wrappedcore_out_ready_o = 1; equalized_result_valid_o = 0; // Only pass the input handshake through while idle, so a second transaction // can never restart the window and cost the latency its independence. if (state_q == WAITING_FOR_IN) begin - unequalized_in_valid_o = in_valid_i; - in_ready_o = unequalized_in_ready_i; + wrappedcore_in_valid_o = in_valid_i; + in_ready_o = wrappedcore_in_ready_i; if (in_valid_i && in_ready_o) begin state_d = WAITING_FOR_RESULT; @@ -94,13 +94,13 @@ always_comb begin counter_d++; // Drain the core exactly once, when its result first becomes valid. - if (unequalized_result_valid_i && unequalized_result_ready_o) begin - data_d = unequalized_result_data_i; + if (wrappedcore_out_valid_i && wrappedcore_out_ready_o) begin + data_d = wrappedcore_out_data_i; state_d = HOLDING_RESULT; end end else if (state_q == HOLDING_RESULT) begin - unequalized_result_ready_o = 0; + wrappedcore_out_ready_o = 0; if (deadline_reached) equalized_result_valid_o = 1; diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index 97a477d..ad8331c 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -125,12 +125,12 @@ ternip_fixed_latency_equalizer #( .in_valid_i, .in_ready_o, - .unequalized_in_valid_o(raw_in_valid), - .unequalized_in_ready_i(raw_in_ready), + .wrappedcore_in_valid_o(raw_in_valid), + .wrappedcore_in_ready_i(raw_in_ready), - .unequalized_result_valid_i(div_out_valid), - .unequalized_result_ready_o(div_out_ready), - .unequalized_result_data_i(convert_out_y), + .wrappedcore_out_valid_i(div_out_valid), + .wrappedcore_out_ready_o(div_out_ready), + .wrappedcore_out_data_i(convert_out_y), .equalized_result_valid_o(out_valid_o), .equalized_result_ready_i(out_ready_i), From 6d480a01b0626cd78d9f67e4c3d45eb62534f870 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 11:19:14 -0700 Subject: [PATCH 08/10] name caller-facing equalizer result ports out --- rtl/common/ternip_fixed_latency_equalizer.sv | 16 ++++++++-------- rtl/math/ternip_div.sv | 6 +++--- 2 files changed, 11 insertions(+), 11 deletions(-) diff --git a/rtl/common/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv index 9ca709a..e7f8c48 100644 --- a/rtl/common/ternip_fixed_latency_equalizer.sv +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -46,9 +46,9 @@ module ternip_fixed_latency_equalizer #( output logic wrappedcore_out_ready_o, input logic [DataWidth-1:0] wrappedcore_out_data_i, - output logic equalized_result_valid_o, - input logic equalized_result_ready_i, - output logic [DataWidth-1:0] equalized_result_data_o + output logic out_valid_o, + input logic out_ready_i, + output logic [DataWidth-1:0] out_data_o ); localparam int CounterWidth = $clog2(NumCycles); @@ -64,7 +64,7 @@ logic [DataWidth-1:0] data_d, data_q; wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1)); -assign equalized_result_data_o = data_q; +assign out_data_o = data_q; always_comb begin state_d = state_q; @@ -74,7 +74,7 @@ always_comb begin wrappedcore_in_valid_o = 0; in_ready_o = 0; wrappedcore_out_ready_o = 1; - equalized_result_valid_o = 0; + out_valid_o = 0; // Only pass the input handshake through while idle, so a second transaction // can never restart the window and cost the latency its independence. @@ -103,11 +103,11 @@ always_comb begin wrappedcore_out_ready_o = 0; if (deadline_reached) - equalized_result_valid_o = 1; + out_valid_o = 1; else counter_d++; - if (equalized_result_valid_o && equalized_result_ready_i) + if (out_valid_o && out_ready_i) state_d = WAITING_FOR_IN; end else begin @@ -140,7 +140,7 @@ end // check it end to end rather than trusting the state machine not to add a cycle. property fixed_latency_p; @(posedge clk_i) disable iff (!rst_ni) - (in_valid_i && in_ready_o) |-> ##NumCycles equalized_result_valid_o; + (in_valid_i && in_ready_o) |-> ##NumCycles out_valid_o; endproperty assert property (fixed_latency_p) diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index ad8331c..fd121bd 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -132,9 +132,9 @@ ternip_fixed_latency_equalizer #( .wrappedcore_out_ready_o(div_out_ready), .wrappedcore_out_data_i(convert_out_y), - .equalized_result_valid_o(out_valid_o), - .equalized_result_ready_i(out_ready_i), - .equalized_result_data_o(y_o) + .out_valid_o, + .out_ready_i, + .out_data_o(y_o) ); if (Implementation == ternip_pkg::DIV_BSG) begin : div_bsg From a11c33c8cc689905a769abf77a437eb41335a1c9 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 11:43:47 -0700 Subject: [PATCH 09/10] order equalizer handshake ports ready before valid --- rtl/common/ternip_fixed_latency_equalizer.sv | 43 +++++++++----------- rtl/math/ternip_div.sv | 14 +++---- 2 files changed, 26 insertions(+), 31 deletions(-) diff --git a/rtl/common/ternip_fixed_latency_equalizer.sv b/rtl/common/ternip_fixed_latency_equalizer.sv index e7f8c48..d55d27e 100644 --- a/rtl/common/ternip_fixed_latency_equalizer.sv +++ b/rtl/common/ternip_fixed_latency_equalizer.sv @@ -36,18 +36,18 @@ module ternip_fixed_latency_equalizer #( input logic clk_i, input logic rst_ni, - input logic in_valid_i, output logic in_ready_o, + input logic in_valid_i, - output logic wrappedcore_in_valid_o, input logic wrappedcore_in_ready_i, + output logic wrappedcore_in_valid_o, - input logic wrappedcore_out_valid_i, output logic wrappedcore_out_ready_o, + input logic wrappedcore_out_valid_i, input logic [DataWidth-1:0] wrappedcore_out_data_i, - output logic out_valid_o, input logic out_ready_i, + output logic out_valid_o, output logic [DataWidth-1:0] out_data_o ); @@ -71,30 +71,32 @@ always_comb begin counter_d = counter_q; data_d = data_q; - wrappedcore_in_valid_o = 0; - in_ready_o = 0; + in_ready_o = 0; + wrappedcore_in_valid_o = 0; wrappedcore_out_ready_o = 1; - out_valid_o = 0; + out_valid_o = 0; // Only pass the input handshake through while idle, so a second transaction // can never restart the window and cost the latency its independence. if (state_q == WAITING_FOR_IN) begin - wrappedcore_in_valid_o = in_valid_i; in_ready_o = wrappedcore_in_ready_i; + wrappedcore_in_valid_o = in_valid_i; - if (in_valid_i && in_ready_o) begin + if (in_ready_o && in_valid_i) begin state_d = WAITING_FOR_RESULT; counter_d = '0; end end else if (state_q == WAITING_FOR_RESULT) begin - if (!deadline_reached) - counter_d++; + counter_d++; // assuming deadline_reached==0 +`ifndef SYNTHESIS + if (deadline_reached) + $fatal(0, "core did not produce a result within %0d cycles", NumCycles); +`endif - // Drain the core exactly once, when its result first becomes valid. - if (wrappedcore_out_valid_i && wrappedcore_out_ready_o) begin + if (wrappedcore_out_ready_o && wrappedcore_out_valid_i) begin data_d = wrappedcore_out_data_i; state_d = HOLDING_RESULT; end @@ -107,7 +109,7 @@ always_comb begin else counter_d++; - if (out_valid_o && out_ready_i) + if (out_ready_i && out_valid_o) state_d = WAITING_FOR_IN; end else begin @@ -128,23 +130,16 @@ always_ff @(posedge clk_i) begin end `ifndef SYNTHESIS -// A core that misses the deadline makes the latency data-dependent again, which -// means NumCycles is too small for it. Sampled on the clock rather than checked -// inside the always_comb, so a transient evaluation cannot trip it. -always @(posedge clk_i) if (rst_ni) begin - assert (!(state_q == WAITING_FOR_RESULT && deadline_reached)) - else $fatal(0, "core did not produce a result within %0d cycles", NumCycles); -end -// The whole point of this module is that latency does not depend on the data, so -// check it end to end rather than trusting the state machine not to add a cycle. +// Ensure that correct latency is achieved property fixed_latency_p; @(posedge clk_i) disable iff (!rst_ni) - (in_valid_i && in_ready_o) |-> ##NumCycles out_valid_o; + (in_ready_o && in_valid_i) |-> ##NumCycles out_valid_o; endproperty assert property (fixed_latency_p) else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles); + `endif endmodule diff --git a/rtl/math/ternip_div.sv b/rtl/math/ternip_div.sv index fd121bd..eff6cf6 100644 --- a/rtl/math/ternip_div.sv +++ b/rtl/math/ternip_div.sv @@ -82,7 +82,7 @@ ternip_fixed_point_convert #( .OutExponent(InternalAInternalExponent), .NumPipelineStages(0) ) convert_a ( - .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(a_i), .out(internal_a) ); @@ -94,7 +94,7 @@ ternip_fixed_point_convert #( .OutExponent(InternalBInternalExponent), .NumPipelineStages(0) ) convert_b ( - .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(b_i), .out(internal_b) ); @@ -106,7 +106,7 @@ ternip_fixed_point_convert #( .OutExponent(OutExponent), .NumPipelineStages(0) ) convert_out ( - .clk_i, .rst_ni, .in_valid_i(1'b1), .in_ready_o(), .out_valid_o(), .out_ready_i(1'b1), // unused, module combinational + .clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational .in(internal_y), .out(convert_out_y) ); @@ -122,18 +122,18 @@ ternip_fixed_latency_equalizer #( .clk_i, .rst_ni, - .in_valid_i, .in_ready_o, + .in_valid_i, - .wrappedcore_in_valid_o(raw_in_valid), .wrappedcore_in_ready_i(raw_in_ready), + .wrappedcore_in_valid_o(raw_in_valid), - .wrappedcore_out_valid_i(div_out_valid), .wrappedcore_out_ready_o(div_out_ready), + .wrappedcore_out_valid_i(div_out_valid), .wrappedcore_out_data_i(convert_out_y), - .out_valid_o, .out_ready_i, + .out_valid_o, .out_data_o(y_o) ); From 4c6891eb4c3797160a173767920caf703c3ac840 Mon Sep 17 00:00:00 2001 From: Ethan Sifferman Date: Fri, 21 Aug 2026 11:43:47 -0700 Subject: [PATCH 10/10] order convert handshake ports ready before valid --- rtl/math/ternip_fixed_point_convert.sv | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/rtl/math/ternip_fixed_point_convert.sv b/rtl/math/ternip_fixed_point_convert.sv index 09b9be2..79101c5 100644 --- a/rtl/math/ternip_fixed_point_convert.sv +++ b/rtl/math/ternip_fixed_point_convert.sv @@ -47,13 +47,13 @@ module ternip_fixed_point_convert #( input logic clk_i, input logic rst_ni, - input logic signed [InPrecision-1:0] in, - input logic in_valid_i, output logic in_ready_o, + input logic in_valid_i, + input logic signed [InPrecision-1:0] in, - output logic signed [OutPrecision-1:0] out, + input logic out_ready_i, output logic out_valid_o, - input logic out_ready_i + output logic signed [OutPrecision-1:0] out ); if (InPrecision < 1) $fatal(0, "InPrecision (%0d) must be positive", InPrecision);