Skip to content
Merged
145 changes: 145 additions & 0 deletions rtl/common/ternip_fixed_latency_equalizer.sv
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
// Copyright (c) 2026 Ethan Sifferman
//
// Redistribution and use in source and binary forms, with or without modification, are permitted
// provided that the following conditions are met:
//
// 1. Redistributions of source code must retain the above copyright notice, this list of
// conditions and the following disclaimer.
//
// 2. Redistributions in binary form must reproduce the above copyright notice, this list of
// conditions and the following disclaimer in the documentation and/or other materials provided
// with the distribution.
//
// 3. Neither the name of the copyright holder nor the names of its contributors may be used to
// endorse or promote products derived from this software without specific prior written
// permission.
//
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
// IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
// FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
// CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
// SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
// THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR
// OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
// POSSIBILITY OF SUCH DAMAGE.


// ternip_fixed_latency_equalizer
//
// Gives a variable-latency ready/valid core a fixed, data-independent latency.

module ternip_fixed_latency_equalizer #(
parameter int DataWidth = 16,
parameter int NumCycles = 32
) (
input logic clk_i,
input logic rst_ni,

output logic in_ready_o,
input logic in_valid_i,

input logic wrappedcore_in_ready_i,
output logic wrappedcore_in_valid_o,

output logic wrappedcore_out_ready_o,
input logic wrappedcore_out_valid_i,
input logic [DataWidth-1:0] wrappedcore_out_data_i,

input logic out_ready_i,
output logic out_valid_o,
output logic [DataWidth-1:0] out_data_o
);

localparam int CounterWidth = $clog2(NumCycles);

enum logic [1:0] {
WAITING_FOR_IN,
WAITING_FOR_RESULT,
HOLDING_RESULT
} state_d, state_q = WAITING_FOR_IN; // for assertions at time=0

logic [CounterWidth-1:0] counter_d, counter_q;
logic [DataWidth-1:0] data_d, data_q;

wire deadline_reached = (counter_q >= CounterWidth'(NumCycles-1));

assign out_data_o = data_q;

always_comb begin
state_d = state_q;
counter_d = counter_q;
data_d = data_q;

in_ready_o = 0;
wrappedcore_in_valid_o = 0;
wrappedcore_out_ready_o = 1;
out_valid_o = 0;

// Only pass the input handshake through while idle, so a second transaction
// can never restart the window and cost the latency its independence.
if (state_q == WAITING_FOR_IN) begin

in_ready_o = wrappedcore_in_ready_i;
wrappedcore_in_valid_o = in_valid_i;

if (in_ready_o && in_valid_i) begin
state_d = WAITING_FOR_RESULT;
counter_d = '0;
end

end else if (state_q == WAITING_FOR_RESULT) begin

counter_d++; // assuming deadline_reached==0
`ifndef SYNTHESIS
if (deadline_reached)
$fatal(0, "core did not produce a result within %0d cycles", NumCycles);
`endif

if (wrappedcore_out_ready_o && wrappedcore_out_valid_i) begin
data_d = wrappedcore_out_data_i;
state_d = HOLDING_RESULT;
end

end else if (state_q == HOLDING_RESULT) begin
wrappedcore_out_ready_o = 0;

if (deadline_reached)
out_valid_o = 1;
else
counter_d++;

if (out_ready_i && out_valid_o)
state_d = WAITING_FOR_IN;

end else begin
state_d = WAITING_FOR_IN;
end
end

always_ff @(posedge clk_i) begin
if (!rst_ni) begin
state_q <= WAITING_FOR_IN;
counter_q <= '0;
data_q <= '0;
end else begin
state_q <= state_d;
counter_q <= counter_d;
data_q <= data_d;
end
end

`ifndef SYNTHESIS

// Ensure that correct latency is achieved
property fixed_latency_p;
@(posedge clk_i) disable iff (!rst_ni)
(in_ready_o && in_valid_i) |-> ##NumCycles out_valid_o;
endproperty

assert property (fixed_latency_p)
else $fatal(0, "latency deviated from the fixed %0d cycles", NumCycles);

`endif

endmodule
90 changes: 30 additions & 60 deletions rtl/math/ternip_div.sv
Original file line number Diff line number Diff line change
Expand Up @@ -79,14 +79,10 @@ ternip_fixed_point_convert #(
.InPrecision(InAPrecision),
.InExponent(InAExponent),
.OutPrecision(DivInternalPrecision),
.OutExponent(InternalAInternalExponent)
.OutExponent(InternalAInternalExponent),
.NumPipelineStages(0)
) convert_a (
.clk_i,
.rst_ni,
.in_valid_i(1'b1),
.in_ready_o(),
.out_valid_o(),
.out_ready_i(1'b1),
.clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational
.in(a_i),
.out(internal_a)
);
Expand All @@ -95,14 +91,10 @@ ternip_fixed_point_convert #(
.InPrecision(InBPrecision),
.InExponent(InBExponent),
.OutPrecision(DivInternalPrecision),
.OutExponent(InternalBInternalExponent)
.OutExponent(InternalBInternalExponent),
.NumPipelineStages(0)
) convert_b (
.clk_i,
.rst_ni,
.in_valid_i(1'b1),
.in_ready_o(),
.out_valid_o(),
.out_ready_i(1'b1),
.clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational
.in(b_i),
.out(internal_b)
);
Expand All @@ -111,61 +103,39 @@ ternip_fixed_point_convert #(
.InPrecision(DivInternalPrecision),
.InExponent(InternalYInternalExponent),
.OutPrecision(OutPrecision),
.OutExponent(OutExponent)
.OutExponent(OutExponent),
.NumPipelineStages(0)
) convert_out (
.clk_i,
.rst_ni,
.in_valid_i(1'b1),
.in_ready_o(),
.out_valid_o(),
.out_ready_i(1'b1),
.clk_i, .rst_ni, .in_ready_o(), .in_valid_i(1'b1), .out_ready_i(1'b1), .out_valid_o(), // unused, module combinational
.in(internal_y),
.out(convert_out_y)
);

// Fixed-latency equalizer: the iterative divider takes a data-dependent number of
// cycles, which would desync callers that run several dividers in lockstep. The
// divider holds its quotient until yumi, so a single counter just delays the result
// by a fixed, data-independent number of cycles -- long enough that every divide is done.
// cycles, long enough that every divider in all cores is done
localparam int RawDivideLatencyUpperBound = DivInternalPrecision + 16;
localparam int EqualizerCounterWidth = $clog2(RawDivideLatencyUpperBound + 1);

// counter == 0 is idle; a divide counts to RawDivideLatencyUpperBound, then signals done.
logic [EqualizerCounterWidth-1:0] equalizer_counter_d, equalizer_counter_q;
wire equalizer_idle = (equalizer_counter_q == '0);
wire equalizer_done = (equalizer_counter_q >= EqualizerCounterWidth'(RawDivideLatencyUpperBound));

assign raw_in_valid = in_valid_i && equalizer_idle;
assign in_ready_o = raw_in_ready && equalizer_idle;
assign div_out_ready = equalizer_done && out_ready_i;
assign out_valid_o = equalizer_done;
assign y_o = convert_out_y;

always_comb begin
equalizer_counter_d = equalizer_counter_q;
if (equalizer_idle) begin
if (in_valid_i && in_ready_o) equalizer_counter_d = 1;
end else if (equalizer_done) begin
if (out_ready_i) equalizer_counter_d = '0;
end else begin
equalizer_counter_d = equalizer_counter_q + 1;
end
end

always_ff @(posedge clk_i) begin
if (!rst_ni) equalizer_counter_q <= '0;
else equalizer_counter_q <= equalizer_counter_d;
end
ternip_fixed_latency_equalizer #(
.DataWidth(OutPrecision),
.NumCycles(RawDivideLatencyUpperBound + 1)
) equalizer (
.clk_i,
.rst_ni,

`ifndef SYNTHESIS
// The fixed window must be long enough that the raw quotient always arrives
// before it elapses; otherwise the latency would become data-dependent again.
always @(posedge clk_i) if (rst_ni) begin
assert (!equalizer_done || div_out_valid)
else $fatal(0, "ternip_div equalizer window (%0d) shorter than raw divide latency",
RawDivideLatencyUpperBound);
end
`endif
.in_ready_o,
.in_valid_i,

.wrappedcore_in_ready_i(raw_in_ready),
.wrappedcore_in_valid_o(raw_in_valid),

.wrappedcore_out_ready_o(div_out_ready),
.wrappedcore_out_valid_i(div_out_valid),
.wrappedcore_out_data_i(convert_out_y),

.out_ready_i,
.out_valid_o,
.out_data_o(y_o)
);

if (Implementation == ternip_pkg::DIV_BSG) begin : div_bsg

Expand Down
8 changes: 4 additions & 4 deletions rtl/math/ternip_fixed_point_convert.sv
Original file line number Diff line number Diff line change
Expand Up @@ -47,13 +47,13 @@ module ternip_fixed_point_convert #(
input logic clk_i,
input logic rst_ni,

input logic signed [InPrecision-1:0] in,
input logic in_valid_i,
output logic in_ready_o,
input logic in_valid_i,
input logic signed [InPrecision-1:0] in,

output logic signed [OutPrecision-1:0] out,
input logic out_ready_i,
output logic out_valid_o,
input logic out_ready_i
output logic signed [OutPrecision-1:0] out
);

if (InPrecision < 1) $fatal(0, "InPrecision (%0d) must be positive", InPrecision);
Expand Down
Loading