Files
retroDE_ps2/rtl/gif_gs/gs_async_fifo.sv
T
thejayman77 ba74bbd5aa Snapshot: fog implementation + fidelity tooling baseline (pre bilinear-clamp fix)
Per-vertex GS fog end-to-end (gs_stub emit incl. persp_emit5, gs_prim_list_feeder
XYZ2->XYZF2 on PRIM.FGE, gs_make_sh3_scheduler_fixture.py F/FGE packing), new fog
TBs, fidelity attribution tooling. Functional baseline before removing the dead
bilinear lerp8 clamps (Codex: 161-node comb loop -> -0.042ns setup fail).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-20 19:56:46 -04:00

265 lines
14 KiB
Systemverilog

// retroDE_ps2 — gs_async_fifo (Ch318)
//
// Generic dual-clock (asynchronous) FIFO with gray-code pointers and 2-FF pointer
// synchronizers — the standard CDC-safe ring buffer. Used by gs_lpddr_axi_master to
// cross 256-bit framebuffer-row packets {addr,data,strb} from the GS clock domain to
// the f2sdram (LPDDR AXI) clock domain. Both domains are treated as GENUINELY async
// even when nominally the same frequency (GS = PLL design_clk; f2sdram = raw board
// clock), per the Ch318 directive.
//
// DEPTH must be a power of two. `wr`/`rd` are single-cycle handshakes gated by
// !full / !empty. Standard caveats: do NOT assert wr when full or rd when empty
// (the wrapper gates both). One-deep gray pointers, single 2-FF synchronizer each
// way — adequate for the modest packet rate (one 32-byte beat per 16 flushed pixels).
module gs_async_fifo #(
parameter int WIDTH = 320, // {addr[31:0], data[255:0], strb[31:0]}
parameter int DEPTH = 16, // power of two
// Infer a synchronous read port when set. This is useful for deep/wide
// FIFOs whose bank-select mux cannot meet a fast rclk as an FWFT output.
parameter bit REGISTERED_READ = 1'b0,
// Ch438 timing cut for very deep/wide registered-read FIFOs. Splitting the
// payload into two independently inferred RAMs gives each half its own
// preserved read-address launch register. This removes the single 744-load
// port-B address net seen on the 93x16K production request FIFO while
// preserving depth, order, and one-cycle read behavior.
parameter bit BANKED_READ = 1'b0,
// Ch439e: split a deep/wide memory in both dimensions. Two depth banks
// times two width banks leave each physical read-address copy driving
// roughly one quarter of the original M20K tree. The registered outputs
// need only a 2:1 depth-bank select; FIFO depth and latency are unchanged.
parameter bit QUADRANT_READ = 1'b0
) (
// write domain
input logic wclk,
input logic wrst_n,
input logic wr,
input logic [WIDTH-1:0] wdata,
output logic wfull,
// read domain
input logic rclk,
input logic rrst_n,
input logic rd,
output logic [WIDTH-1:0] rdata,
output logic rempty
);
localparam int AW = $clog2(DEPTH);
logic [WIDTH-1:0] mem [0:DEPTH-1];
localparam int BANK_LO_W = WIDTH / 2;
localparam int BANK_HI_W = WIDTH - BANK_LO_W;
logic [BANK_LO_W-1:0] mem_lo [0:DEPTH-1];
logic [BANK_HI_W-1:0] mem_hi [0:DEPTH-1];
localparam int HALF_DEPTH = DEPTH / 2;
localparam int HALF_AW = AW - 1;
logic [BANK_LO_W-1:0] mem_lo0 [0:HALF_DEPTH-1];
logic [BANK_LO_W-1:0] mem_lo1 [0:HALF_DEPTH-1];
logic [BANK_HI_W-1:0] mem_hi0 [0:HALF_DEPTH-1];
logic [BANK_HI_W-1:0] mem_hi1 [0:HALF_DEPTH-1];
// Dedicated write-port staging lets the fitter duplicate/place the RAM
// address register beside a wide banked memory. Driving every bank
// directly from the shared binary pointer created a 310 MHz high-fanout
// wbin -> RAM-address path in the 321-bit color FIFO. The opposite-domain
// pointer requires two synchronizer cycles before a reader can observe a
// write, so committing the RAM one local cycle later is CDC-safe.
logic [AW-1:0] waddr_q;
logic [WIDTH-1:0] wdata_q;
logic wwrite_q;
// ---- binary + gray pointers (one extra MSB for full/empty disambiguation) ----
logic [AW:0] wbin, wgray, wbin_nxt;
logic [AW:0] wcommit, wcommit_nxt;
logic wfull_nxt; // Ch352 — combinational next-value for the now-REGISTERED wfull
logic [AW:0] rbin, rgray, rbin_nxt, rgray_nxt;
logic [AW:0] rbin_inc, rgray_inc;
(* keep *) logic rempty_if_hold, rempty_if_pop;
logic rempty_nxt; // Ch357 — combinational next-value for the now-REGISTERED rempty (read-side twin)
logic [WIDTH-1:0] rdata_q;
logic [BANK_LO_W-1:0] rdata_lo_q;
logic [BANK_HI_W-1:0] rdata_hi_q;
// Keep the RAM-facing address distinct from the binary/Gray pointer. The
// production request FIFO is one packed 93-bit x 16K macro; splitting it
// into explicit width banks wastes M20Ks at each bank boundary. Retain
// that efficient packing and ask synthesis to duplicate only this launch
// register so no copy drives the complete physical port-B address tree.
(* dont_merge, preserve *) logic [AW-1:0] raddr_q /* synthesis maxfan = 64 */;
(* dont_merge, preserve *) logic [AW-1:0] raddr_lo_q /* synthesis maxfan = 64 */;
(* dont_merge, preserve *) logic [AW-1:0] raddr_hi_q /* synthesis maxfan = 64 */;
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo0_q /* synthesis maxfan = 32 */;
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo1_q /* synthesis maxfan = 32 */;
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi0_q /* synthesis maxfan = 32 */;
(* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi1_q /* synthesis maxfan = 32 */;
logic [BANK_LO_W-1:0] rdata_lo0_q, rdata_lo1_q;
logic [BANK_HI_W-1:0] rdata_hi0_q, rdata_hi1_q;
logic rbank_addr_q, rbank_data_q;
// synchronized opposite-domain gray pointers (2-FF)
logic [AW:0] rgray_s1, rgray_s2; // read gray -> write domain
logic [AW:0] wgray_s1, wgray_s2; // write gray -> read domain
function automatic logic [AW:0] bin2gray(input logic [AW:0] b);
bin2gray = b ^ (b >> 1);
endfunction
// ---------------- write domain ----------------
assign wbin_nxt = wbin + (wr && !wfull);
// full: next write gray == read gray with top two bits inverted. Ch352 — wfull is now a REGISTERED flag
// (Cummings canonical). The previous `assign wfull = (wgray_nxt == ...)` was combinational, and since
// wgray_nxt <- wbin_nxt <- wfull, it formed a wbin_nxt->wgray_nxt->wfull->wbin_nxt COMBINATIONAL LOOP that
// Quartus reports and that made Place churn. Registering it breaks the loop with no overflow-behavior change:
// wfull still asserts the cycle after the filling write (full is computed from wgray_nxt = the pointer AFTER
// the current write), so the (DEPTH+1)th write is still blocked. Ch357 — rempty is now the registered read-side twin.
assign wfull_nxt = (bin2gray(wbin_nxt) == {~rgray_s2[AW:AW-1], rgray_s2[AW-2:0]});
// `wbin` is the allocation pointer (an input handshake reserves an
// address). `wcommit` trails it by the one-entry write-port stage and is
// the ONLY pointer published to the read domain. Publishing allocation
// early is unsafe when rclk is faster than wclk: the reader can otherwise
// observe a new pointer before the staged RAM write has occurred.
assign wcommit_nxt = wcommit + wwrite_q;
always_ff @(posedge wclk or negedge wrst_n) begin
if (!wrst_n) begin
wbin <= '0; wcommit <= '0; wgray <= '0; wfull <= 1'b0;
rgray_s1 <= '0; rgray_s2 <= '0;
waddr_q <= '0; wdata_q <= '0; wwrite_q <= 1'b0;
end else begin
wbin <= wbin_nxt;
wcommit <= wcommit_nxt;
wgray <= bin2gray(wcommit_nxt);
wfull <= wfull_nxt;
rgray_s1 <= rgray; // sync read gray into write domain
rgray_s2 <= rgray_s1;
waddr_q <= wbin[AW-1:0];
wdata_q <= wdata;
wwrite_q <= wr && !wfull;
end
end
// ---------------- read domain ----------------
// `rd` is an accepted-read handshake by contract: every wrapper gates it
// with !rempty. Do not gate it again here. The redundant internal gate
// put rempty in front of the AW+1 pointer adder and, for a deep FIFO, also
// in front of every RAM read-address bank. That feedback was the complete
// Ch405 310 MHz setup-failure family.
// Precompute the increment independent of `rd`, then select between the
// hold/pop results. Writing this as `rbin + rd` put the registered pop
// pulse on the carry input of the complete AW+1 adder and then through
// Gray conversion + empty equality at 310 MHz. The explicit two-result
// form is behavior-identical but leaves `rd` driving only final muxes.
assign rbin_inc = rbin + {{AW{1'b0}}, 1'b1};
assign rgray_inc = bin2gray(rbin_inc);
assign rbin_nxt = rd ? rbin_inc : rbin;
assign rgray_nxt = rd ? rgray_inc : rgray;
assign rempty_if_hold = (rgray == wgray_s2);
assign rempty_if_pop = (rgray_inc == wgray_s2);
assign rempty_nxt = rd ? rempty_if_pop : rempty_if_hold;
always_ff @(posedge rclk or negedge rrst_n) begin
if (!rrst_n) begin
rbin <= '0; rgray <= '0; rempty <= 1'b1;
wgray_s1 <= '0; wgray_s2 <= '0;
end else begin
rbin <= rbin_nxt;
rgray <= rgray_nxt;
rempty <= rempty_nxt;
wgray_s1 <= wgray; // sync write gray into read domain
wgray_s2 <= wgray_s1;
end
end
generate
if (QUADRANT_READ) begin : g_quadrant_storage
// Four physical RAM quadrants: low/high payload width crossed with
// lower/upper address half. Writes remain atomic and use the
// staged allocation address exactly as the monolithic form does.
always_ff @(posedge wclk) begin
if (wwrite_q) begin
if (waddr_q[AW-1]) begin
mem_lo1[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W];
mem_hi1[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W];
end else begin
mem_lo0[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W];
mem_hi0[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W];
end
end
end
if (REGISTERED_READ) begin : g_registered_read
always_ff @(posedge rclk) begin
// Separate launch copies are intentional: each feeds only
// one depth/width quadrant. rbank_data_q trails the
// address-bank selector by the same cycle as the four
// synchronous RAM outputs.
raddr_lo0_q <= rbin_nxt[HALF_AW-1:0];
raddr_lo1_q <= rbin_nxt[HALF_AW-1:0];
raddr_hi0_q <= rbin_nxt[HALF_AW-1:0];
raddr_hi1_q <= rbin_nxt[HALF_AW-1:0];
rbank_addr_q <= rbin_nxt[AW-1];
rbank_data_q <= rbank_addr_q;
rdata_lo0_q <= mem_lo0[raddr_lo0_q];
rdata_lo1_q <= mem_lo1[raddr_lo1_q];
rdata_hi0_q <= mem_hi0[raddr_hi0_q];
rdata_hi1_q <= mem_hi1[raddr_hi1_q];
end
assign rdata = rbank_data_q ? {rdata_hi1_q, rdata_lo1_q}
: {rdata_hi0_q, rdata_lo0_q};
end else begin : g_fwft_read
assign rdata = rbin[AW-1]
? {mem_hi1[rbin[HALF_AW-1:0]], mem_lo1[rbin[HALF_AW-1:0]]}
: {mem_hi0[rbin[HALF_AW-1:0]], mem_lo0[rbin[HALF_AW-1:0]]};
end
end else if (BANKED_READ) begin : g_banked_storage
// Two physical payload banks, written atomically from the same
// staged tuple. Each registered read address drives only its own
// half of the inferred RAM instead of the entire packed macro.
always_ff @(posedge wclk) begin
if (wwrite_q) begin
mem_lo[waddr_q] <= wdata_q[0 +: BANK_LO_W];
mem_hi[waddr_q] <= wdata_q[BANK_LO_W +: BANK_HI_W];
end
end
if (REGISTERED_READ) begin : g_registered_read
always_ff @(posedge rclk) begin
raddr_lo_q <= rbin_nxt[AW-1:0];
raddr_hi_q <= rbin_nxt[AW-1:0];
rdata_lo_q <= mem_lo[raddr_lo_q];
rdata_hi_q <= mem_hi[raddr_hi_q];
end
assign rdata = {rdata_hi_q, rdata_lo_q};
end else begin : g_fwft_read
assign rdata = {mem_hi[rbin[AW-1:0]], mem_lo[rbin[AW-1:0]]};
end
end else begin : g_monolithic_storage
always_ff @(posedge wclk)
if (wwrite_q) mem[waddr_q] <= wdata_q;
if (REGISTERED_READ) begin : g_registered_read
// A synchronous read lets Quartus use the memory output register
// instead of timing a deep bank mux directly into request decode.
// Read the current head every cycle and qualify rdata only at the
// interface. The pointer still advances exclusively on `rd`, so
// this does not consume an entry or change the one-cycle accepted-
// read latency. Leaving the inferred RAM read enable permanently
// active is important for a very wide FIFO: using `rd` as the RAM
// enable made one pop register drive every physical data bank
// (749 loads in the production request FIFO) at 310 MHz.
//
// Ch420: the Ch419 fit proved the enable cut and exposed the same
// topology on portbaddr: rbin[6] directly drove 713 RAM-address
// loads. `raddr_q` tracks the pointer's selected next value, so
// before every edge it equals the current head address. The RAM
// read therefore returns the same entry on the same edge as the
// prior `mem[rbin]` form, including consecutive accepted pops,
// while splitting pointer selection from physical RAM addressing.
//
// raddr_q/rdata_q intentionally have neither enables nor resets.
// The FIFO cannot become nonempty until the synchronized write
// pointer arrives, giving raddr_q multiple clocks to initialize to
// zero after reset. Resetting the wide inferred read structure
// previously created its own high-fanout recovery/setup family.
always_ff @(posedge rclk) begin
raddr_q <= rbin_nxt[AW-1:0];
rdata_q <= mem[raddr_q];
end
assign rdata = rdata_q;
end else begin : g_fwft_read
assign rdata = mem[rbin[AW-1:0]];
end
end
endgenerate
endmodule : gs_async_fifo