ba74bbd5aa
Per-vertex GS fog end-to-end (gs_stub emit incl. persp_emit5, gs_prim_list_feeder XYZ2->XYZF2 on PRIM.FGE, gs_make_sh3_scheduler_fixture.py F/FGE packing), new fog TBs, fidelity attribution tooling. Functional baseline before removing the dead bilinear lerp8 clamps (Codex: 161-node comb loop -> -0.042ns setup fail). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
265 lines
14 KiB
Systemverilog
265 lines
14 KiB
Systemverilog
// retroDE_ps2 — gs_async_fifo (Ch318)
|
|
//
|
|
// Generic dual-clock (asynchronous) FIFO with gray-code pointers and 2-FF pointer
|
|
// synchronizers — the standard CDC-safe ring buffer. Used by gs_lpddr_axi_master to
|
|
// cross 256-bit framebuffer-row packets {addr,data,strb} from the GS clock domain to
|
|
// the f2sdram (LPDDR AXI) clock domain. Both domains are treated as GENUINELY async
|
|
// even when nominally the same frequency (GS = PLL design_clk; f2sdram = raw board
|
|
// clock), per the Ch318 directive.
|
|
//
|
|
// DEPTH must be a power of two. `wr`/`rd` are single-cycle handshakes gated by
|
|
// !full / !empty. Standard caveats: do NOT assert wr when full or rd when empty
|
|
// (the wrapper gates both). One-deep gray pointers, single 2-FF synchronizer each
|
|
// way — adequate for the modest packet rate (one 32-byte beat per 16 flushed pixels).
|
|
|
|
module gs_async_fifo #(
|
|
parameter int WIDTH = 320, // {addr[31:0], data[255:0], strb[31:0]}
|
|
parameter int DEPTH = 16, // power of two
|
|
// Infer a synchronous read port when set. This is useful for deep/wide
|
|
// FIFOs whose bank-select mux cannot meet a fast rclk as an FWFT output.
|
|
parameter bit REGISTERED_READ = 1'b0,
|
|
// Ch438 timing cut for very deep/wide registered-read FIFOs. Splitting the
|
|
// payload into two independently inferred RAMs gives each half its own
|
|
// preserved read-address launch register. This removes the single 744-load
|
|
// port-B address net seen on the 93x16K production request FIFO while
|
|
// preserving depth, order, and one-cycle read behavior.
|
|
parameter bit BANKED_READ = 1'b0,
|
|
// Ch439e: split a deep/wide memory in both dimensions. Two depth banks
|
|
// times two width banks leave each physical read-address copy driving
|
|
// roughly one quarter of the original M20K tree. The registered outputs
|
|
// need only a 2:1 depth-bank select; FIFO depth and latency are unchanged.
|
|
parameter bit QUADRANT_READ = 1'b0
|
|
) (
|
|
// write domain
|
|
input logic wclk,
|
|
input logic wrst_n,
|
|
input logic wr,
|
|
input logic [WIDTH-1:0] wdata,
|
|
output logic wfull,
|
|
// read domain
|
|
input logic rclk,
|
|
input logic rrst_n,
|
|
input logic rd,
|
|
output logic [WIDTH-1:0] rdata,
|
|
output logic rempty
|
|
);
|
|
localparam int AW = $clog2(DEPTH);
|
|
|
|
logic [WIDTH-1:0] mem [0:DEPTH-1];
|
|
localparam int BANK_LO_W = WIDTH / 2;
|
|
localparam int BANK_HI_W = WIDTH - BANK_LO_W;
|
|
logic [BANK_LO_W-1:0] mem_lo [0:DEPTH-1];
|
|
logic [BANK_HI_W-1:0] mem_hi [0:DEPTH-1];
|
|
localparam int HALF_DEPTH = DEPTH / 2;
|
|
localparam int HALF_AW = AW - 1;
|
|
logic [BANK_LO_W-1:0] mem_lo0 [0:HALF_DEPTH-1];
|
|
logic [BANK_LO_W-1:0] mem_lo1 [0:HALF_DEPTH-1];
|
|
logic [BANK_HI_W-1:0] mem_hi0 [0:HALF_DEPTH-1];
|
|
logic [BANK_HI_W-1:0] mem_hi1 [0:HALF_DEPTH-1];
|
|
// Dedicated write-port staging lets the fitter duplicate/place the RAM
|
|
// address register beside a wide banked memory. Driving every bank
|
|
// directly from the shared binary pointer created a 310 MHz high-fanout
|
|
// wbin -> RAM-address path in the 321-bit color FIFO. The opposite-domain
|
|
// pointer requires two synchronizer cycles before a reader can observe a
|
|
// write, so committing the RAM one local cycle later is CDC-safe.
|
|
logic [AW-1:0] waddr_q;
|
|
logic [WIDTH-1:0] wdata_q;
|
|
logic wwrite_q;
|
|
|
|
// ---- binary + gray pointers (one extra MSB for full/empty disambiguation) ----
|
|
logic [AW:0] wbin, wgray, wbin_nxt;
|
|
logic [AW:0] wcommit, wcommit_nxt;
|
|
logic wfull_nxt; // Ch352 — combinational next-value for the now-REGISTERED wfull
|
|
logic [AW:0] rbin, rgray, rbin_nxt, rgray_nxt;
|
|
logic [AW:0] rbin_inc, rgray_inc;
|
|
(* keep *) logic rempty_if_hold, rempty_if_pop;
|
|
logic rempty_nxt; // Ch357 — combinational next-value for the now-REGISTERED rempty (read-side twin)
|
|
logic [WIDTH-1:0] rdata_q;
|
|
logic [BANK_LO_W-1:0] rdata_lo_q;
|
|
logic [BANK_HI_W-1:0] rdata_hi_q;
|
|
// Keep the RAM-facing address distinct from the binary/Gray pointer. The
|
|
// production request FIFO is one packed 93-bit x 16K macro; splitting it
|
|
// into explicit width banks wastes M20Ks at each bank boundary. Retain
|
|
// that efficient packing and ask synthesis to duplicate only this launch
|
|
// register so no copy drives the complete physical port-B address tree.
|
|
(* dont_merge, preserve *) logic [AW-1:0] raddr_q /* synthesis maxfan = 64 */;
|
|
(* dont_merge, preserve *) logic [AW-1:0] raddr_lo_q /* synthesis maxfan = 64 */;
|
|
(* dont_merge, preserve *) logic [AW-1:0] raddr_hi_q /* synthesis maxfan = 64 */;
|
|
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo0_q /* synthesis maxfan = 32 */;
|
|
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo1_q /* synthesis maxfan = 32 */;
|
|
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi0_q /* synthesis maxfan = 32 */;
|
|
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi1_q /* synthesis maxfan = 32 */;
|
|
logic [BANK_LO_W-1:0] rdata_lo0_q, rdata_lo1_q;
|
|
logic [BANK_HI_W-1:0] rdata_hi0_q, rdata_hi1_q;
|
|
logic rbank_addr_q, rbank_data_q;
|
|
|
|
// synchronized opposite-domain gray pointers (2-FF)
|
|
logic [AW:0] rgray_s1, rgray_s2; // read gray -> write domain
|
|
logic [AW:0] wgray_s1, wgray_s2; // write gray -> read domain
|
|
|
|
function automatic logic [AW:0] bin2gray(input logic [AW:0] b);
|
|
bin2gray = b ^ (b >> 1);
|
|
endfunction
|
|
|
|
// ---------------- write domain ----------------
|
|
assign wbin_nxt = wbin + (wr && !wfull);
|
|
// full: next write gray == read gray with top two bits inverted. Ch352 — wfull is now a REGISTERED flag
|
|
// (Cummings canonical). The previous `assign wfull = (wgray_nxt == ...)` was combinational, and since
|
|
// wgray_nxt <- wbin_nxt <- wfull, it formed a wbin_nxt->wgray_nxt->wfull->wbin_nxt COMBINATIONAL LOOP that
|
|
// Quartus reports and that made Place churn. Registering it breaks the loop with no overflow-behavior change:
|
|
// wfull still asserts the cycle after the filling write (full is computed from wgray_nxt = the pointer AFTER
|
|
// the current write), so the (DEPTH+1)th write is still blocked. Ch357 — rempty is now the registered read-side twin.
|
|
assign wfull_nxt = (bin2gray(wbin_nxt) == {~rgray_s2[AW:AW-1], rgray_s2[AW-2:0]});
|
|
// `wbin` is the allocation pointer (an input handshake reserves an
|
|
// address). `wcommit` trails it by the one-entry write-port stage and is
|
|
// the ONLY pointer published to the read domain. Publishing allocation
|
|
// early is unsafe when rclk is faster than wclk: the reader can otherwise
|
|
// observe a new pointer before the staged RAM write has occurred.
|
|
assign wcommit_nxt = wcommit + wwrite_q;
|
|
always_ff @(posedge wclk or negedge wrst_n) begin
|
|
if (!wrst_n) begin
|
|
wbin <= '0; wcommit <= '0; wgray <= '0; wfull <= 1'b0;
|
|
rgray_s1 <= '0; rgray_s2 <= '0;
|
|
waddr_q <= '0; wdata_q <= '0; wwrite_q <= 1'b0;
|
|
end else begin
|
|
wbin <= wbin_nxt;
|
|
wcommit <= wcommit_nxt;
|
|
wgray <= bin2gray(wcommit_nxt);
|
|
wfull <= wfull_nxt;
|
|
rgray_s1 <= rgray; // sync read gray into write domain
|
|
rgray_s2 <= rgray_s1;
|
|
waddr_q <= wbin[AW-1:0];
|
|
wdata_q <= wdata;
|
|
wwrite_q <= wr && !wfull;
|
|
end
|
|
end
|
|
// ---------------- read domain ----------------
|
|
// `rd` is an accepted-read handshake by contract: every wrapper gates it
|
|
// with !rempty. Do not gate it again here. The redundant internal gate
|
|
// put rempty in front of the AW+1 pointer adder and, for a deep FIFO, also
|
|
// in front of every RAM read-address bank. That feedback was the complete
|
|
// Ch405 310 MHz setup-failure family.
|
|
// Precompute the increment independent of `rd`, then select between the
|
|
// hold/pop results. Writing this as `rbin + rd` put the registered pop
|
|
// pulse on the carry input of the complete AW+1 adder and then through
|
|
// Gray conversion + empty equality at 310 MHz. The explicit two-result
|
|
// form is behavior-identical but leaves `rd` driving only final muxes.
|
|
assign rbin_inc = rbin + {{AW{1'b0}}, 1'b1};
|
|
assign rgray_inc = bin2gray(rbin_inc);
|
|
assign rbin_nxt = rd ? rbin_inc : rbin;
|
|
assign rgray_nxt = rd ? rgray_inc : rgray;
|
|
assign rempty_if_hold = (rgray == wgray_s2);
|
|
assign rempty_if_pop = (rgray_inc == wgray_s2);
|
|
assign rempty_nxt = rd ? rempty_if_pop : rempty_if_hold;
|
|
always_ff @(posedge rclk or negedge rrst_n) begin
|
|
if (!rrst_n) begin
|
|
rbin <= '0; rgray <= '0; rempty <= 1'b1;
|
|
wgray_s1 <= '0; wgray_s2 <= '0;
|
|
end else begin
|
|
rbin <= rbin_nxt;
|
|
rgray <= rgray_nxt;
|
|
rempty <= rempty_nxt;
|
|
wgray_s1 <= wgray; // sync write gray into read domain
|
|
wgray_s2 <= wgray_s1;
|
|
end
|
|
end
|
|
generate
|
|
if (QUADRANT_READ) begin : g_quadrant_storage
|
|
// Four physical RAM quadrants: low/high payload width crossed with
|
|
// lower/upper address half. Writes remain atomic and use the
|
|
// staged allocation address exactly as the monolithic form does.
|
|
always_ff @(posedge wclk) begin
|
|
if (wwrite_q) begin
|
|
if (waddr_q[AW-1]) begin
|
|
mem_lo1[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W];
|
|
mem_hi1[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W];
|
|
end else begin
|
|
mem_lo0[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W];
|
|
mem_hi0[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W];
|
|
end
|
|
end
|
|
end
|
|
if (REGISTERED_READ) begin : g_registered_read
|
|
always_ff @(posedge rclk) begin
|
|
// Separate launch copies are intentional: each feeds only
|
|
// one depth/width quadrant. rbank_data_q trails the
|
|
// address-bank selector by the same cycle as the four
|
|
// synchronous RAM outputs.
|
|
raddr_lo0_q <= rbin_nxt[HALF_AW-1:0];
|
|
raddr_lo1_q <= rbin_nxt[HALF_AW-1:0];
|
|
raddr_hi0_q <= rbin_nxt[HALF_AW-1:0];
|
|
raddr_hi1_q <= rbin_nxt[HALF_AW-1:0];
|
|
rbank_addr_q <= rbin_nxt[AW-1];
|
|
rbank_data_q <= rbank_addr_q;
|
|
rdata_lo0_q <= mem_lo0[raddr_lo0_q];
|
|
rdata_lo1_q <= mem_lo1[raddr_lo1_q];
|
|
rdata_hi0_q <= mem_hi0[raddr_hi0_q];
|
|
rdata_hi1_q <= mem_hi1[raddr_hi1_q];
|
|
end
|
|
assign rdata = rbank_data_q ? {rdata_hi1_q, rdata_lo1_q}
|
|
: {rdata_hi0_q, rdata_lo0_q};
|
|
end else begin : g_fwft_read
|
|
assign rdata = rbin[AW-1]
|
|
? {mem_hi1[rbin[HALF_AW-1:0]], mem_lo1[rbin[HALF_AW-1:0]]}
|
|
: {mem_hi0[rbin[HALF_AW-1:0]], mem_lo0[rbin[HALF_AW-1:0]]};
|
|
end
|
|
end else if (BANKED_READ) begin : g_banked_storage
|
|
// Two physical payload banks, written atomically from the same
|
|
// staged tuple. Each registered read address drives only its own
|
|
// half of the inferred RAM instead of the entire packed macro.
|
|
always_ff @(posedge wclk) begin
|
|
if (wwrite_q) begin
|
|
mem_lo[waddr_q] <= wdata_q[0 +: BANK_LO_W];
|
|
mem_hi[waddr_q] <= wdata_q[BANK_LO_W +: BANK_HI_W];
|
|
end
|
|
end
|
|
if (REGISTERED_READ) begin : g_registered_read
|
|
always_ff @(posedge rclk) begin
|
|
raddr_lo_q <= rbin_nxt[AW-1:0];
|
|
raddr_hi_q <= rbin_nxt[AW-1:0];
|
|
rdata_lo_q <= mem_lo[raddr_lo_q];
|
|
rdata_hi_q <= mem_hi[raddr_hi_q];
|
|
end
|
|
assign rdata = {rdata_hi_q, rdata_lo_q};
|
|
end else begin : g_fwft_read
|
|
assign rdata = {mem_hi[rbin[AW-1:0]], mem_lo[rbin[AW-1:0]]};
|
|
end
|
|
end else begin : g_monolithic_storage
|
|
always_ff @(posedge wclk)
|
|
if (wwrite_q) mem[waddr_q] <= wdata_q;
|
|
if (REGISTERED_READ) begin : g_registered_read
|
|
// A synchronous read lets Quartus use the memory output register
|
|
// instead of timing a deep bank mux directly into request decode.
|
|
// Read the current head every cycle and qualify rdata only at the
|
|
// interface. The pointer still advances exclusively on `rd`, so
|
|
// this does not consume an entry or change the one-cycle accepted-
|
|
// read latency. Leaving the inferred RAM read enable permanently
|
|
// active is important for a very wide FIFO: using `rd` as the RAM
|
|
// enable made one pop register drive every physical data bank
|
|
// (749 loads in the production request FIFO) at 310 MHz.
|
|
//
|
|
// Ch420: the Ch419 fit proved the enable cut and exposed the same
|
|
// topology on portbaddr: rbin[6] directly drove 713 RAM-address
|
|
// loads. `raddr_q` tracks the pointer's selected next value, so
|
|
// before every edge it equals the current head address. The RAM
|
|
// read therefore returns the same entry on the same edge as the
|
|
// prior `mem[rbin]` form, including consecutive accepted pops,
|
|
// while splitting pointer selection from physical RAM addressing.
|
|
//
|
|
// raddr_q/rdata_q intentionally have neither enables nor resets.
|
|
// The FIFO cannot become nonempty until the synchronized write
|
|
// pointer arrives, giving raddr_q multiple clocks to initialize to
|
|
// zero after reset. Resetting the wide inferred read structure
|
|
// previously created its own high-fanout recovery/setup family.
|
|
always_ff @(posedge rclk) begin
|
|
raddr_q <= rbin_nxt[AW-1:0];
|
|
rdata_q <= mem[raddr_q];
|
|
end
|
|
assign rdata = rdata_q;
|
|
end else begin : g_fwft_read
|
|
assign rdata = mem[rbin[AW-1:0]];
|
|
end
|
|
end
|
|
endgenerate
|
|
endmodule : gs_async_fifo
|