// retroDE_ps2 — gs_async_fifo (Ch318) // // Generic dual-clock (asynchronous) FIFO with gray-code pointers and 2-FF pointer // synchronizers — the standard CDC-safe ring buffer. Used by gs_lpddr_axi_master to // cross 256-bit framebuffer-row packets {addr,data,strb} from the GS clock domain to // the f2sdram (LPDDR AXI) clock domain. Both domains are treated as GENUINELY async // even when nominally the same frequency (GS = PLL design_clk; f2sdram = raw board // clock), per the Ch318 directive. // // DEPTH must be a power of two. `wr`/`rd` are single-cycle handshakes gated by // !full / !empty. Standard caveats: do NOT assert wr when full or rd when empty // (the wrapper gates both). One-deep gray pointers, single 2-FF synchronizer each // way — adequate for the modest packet rate (one 32-byte beat per 16 flushed pixels). module gs_async_fifo #( parameter int WIDTH = 320, // {addr[31:0], data[255:0], strb[31:0]} parameter int DEPTH = 16, // power of two // Infer a synchronous read port when set. This is useful for deep/wide // FIFOs whose bank-select mux cannot meet a fast rclk as an FWFT output. parameter bit REGISTERED_READ = 1'b0, // Ch438 timing cut for very deep/wide registered-read FIFOs. Splitting the // payload into two independently inferred RAMs gives each half its own // preserved read-address launch register. This removes the single 744-load // port-B address net seen on the 93x16K production request FIFO while // preserving depth, order, and one-cycle read behavior. parameter bit BANKED_READ = 1'b0, // Ch439e: split a deep/wide memory in both dimensions. Two depth banks // times two width banks leave each physical read-address copy driving // roughly one quarter of the original M20K tree. The registered outputs // need only a 2:1 depth-bank select; FIFO depth and latency are unchanged. parameter bit QUADRANT_READ = 1'b0 ) ( // write domain input logic wclk, input logic wrst_n, input logic wr, input logic [WIDTH-1:0] wdata, output logic wfull, // read domain input logic rclk, input logic rrst_n, input logic rd, output logic [WIDTH-1:0] rdata, output logic rempty ); localparam int AW = $clog2(DEPTH); logic [WIDTH-1:0] mem [0:DEPTH-1]; localparam int BANK_LO_W = WIDTH / 2; localparam int BANK_HI_W = WIDTH - BANK_LO_W; logic [BANK_LO_W-1:0] mem_lo [0:DEPTH-1]; logic [BANK_HI_W-1:0] mem_hi [0:DEPTH-1]; localparam int HALF_DEPTH = DEPTH / 2; localparam int HALF_AW = AW - 1; logic [BANK_LO_W-1:0] mem_lo0 [0:HALF_DEPTH-1]; logic [BANK_LO_W-1:0] mem_lo1 [0:HALF_DEPTH-1]; logic [BANK_HI_W-1:0] mem_hi0 [0:HALF_DEPTH-1]; logic [BANK_HI_W-1:0] mem_hi1 [0:HALF_DEPTH-1]; // Dedicated write-port staging lets the fitter duplicate/place the RAM // address register beside a wide banked memory. Driving every bank // directly from the shared binary pointer created a 310 MHz high-fanout // wbin -> RAM-address path in the 321-bit color FIFO. The opposite-domain // pointer requires two synchronizer cycles before a reader can observe a // write, so committing the RAM one local cycle later is CDC-safe. logic [AW-1:0] waddr_q; logic [WIDTH-1:0] wdata_q; logic wwrite_q; // ---- binary + gray pointers (one extra MSB for full/empty disambiguation) ---- logic [AW:0] wbin, wgray, wbin_nxt; logic [AW:0] wcommit, wcommit_nxt; logic wfull_nxt; // Ch352 — combinational next-value for the now-REGISTERED wfull logic [AW:0] rbin, rgray, rbin_nxt, rgray_nxt; logic [AW:0] rbin_inc, rgray_inc; (* keep *) logic rempty_if_hold, rempty_if_pop; logic rempty_nxt; // Ch357 — combinational next-value for the now-REGISTERED rempty (read-side twin) logic [WIDTH-1:0] rdata_q; logic [BANK_LO_W-1:0] rdata_lo_q; logic [BANK_HI_W-1:0] rdata_hi_q; // Keep the RAM-facing address distinct from the binary/Gray pointer. The // production request FIFO is one packed 93-bit x 16K macro; splitting it // into explicit width banks wastes M20Ks at each bank boundary. Retain // that efficient packing and ask synthesis to duplicate only this launch // register so no copy drives the complete physical port-B address tree. (* dont_merge, preserve *) logic [AW-1:0] raddr_q /* synthesis maxfan = 64 */; (* dont_merge, preserve *) logic [AW-1:0] raddr_lo_q /* synthesis maxfan = 64 */; (* dont_merge, preserve *) logic [AW-1:0] raddr_hi_q /* synthesis maxfan = 64 */; (* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo0_q /* synthesis maxfan = 32 */; (* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_lo1_q /* synthesis maxfan = 32 */; (* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi0_q /* synthesis maxfan = 32 */; (* dont_merge, preserve *) logic [HALF_AW-1:0] raddr_hi1_q /* synthesis maxfan = 32 */; logic [BANK_LO_W-1:0] rdata_lo0_q, rdata_lo1_q; logic [BANK_HI_W-1:0] rdata_hi0_q, rdata_hi1_q; logic rbank_addr_q, rbank_data_q; // synchronized opposite-domain gray pointers (2-FF) logic [AW:0] rgray_s1, rgray_s2; // read gray -> write domain logic [AW:0] wgray_s1, wgray_s2; // write gray -> read domain function automatic logic [AW:0] bin2gray(input logic [AW:0] b); bin2gray = b ^ (b >> 1); endfunction // ---------------- write domain ---------------- assign wbin_nxt = wbin + (wr && !wfull); // full: next write gray == read gray with top two bits inverted. Ch352 — wfull is now a REGISTERED flag // (Cummings canonical). The previous `assign wfull = (wgray_nxt == ...)` was combinational, and since // wgray_nxt <- wbin_nxt <- wfull, it formed a wbin_nxt->wgray_nxt->wfull->wbin_nxt COMBINATIONAL LOOP that // Quartus reports and that made Place churn. Registering it breaks the loop with no overflow-behavior change: // wfull still asserts the cycle after the filling write (full is computed from wgray_nxt = the pointer AFTER // the current write), so the (DEPTH+1)th write is still blocked. Ch357 — rempty is now the registered read-side twin. assign wfull_nxt = (bin2gray(wbin_nxt) == {~rgray_s2[AW:AW-1], rgray_s2[AW-2:0]}); // `wbin` is the allocation pointer (an input handshake reserves an // address). `wcommit` trails it by the one-entry write-port stage and is // the ONLY pointer published to the read domain. Publishing allocation // early is unsafe when rclk is faster than wclk: the reader can otherwise // observe a new pointer before the staged RAM write has occurred. assign wcommit_nxt = wcommit + wwrite_q; always_ff @(posedge wclk or negedge wrst_n) begin if (!wrst_n) begin wbin <= '0; wcommit <= '0; wgray <= '0; wfull <= 1'b0; rgray_s1 <= '0; rgray_s2 <= '0; waddr_q <= '0; wdata_q <= '0; wwrite_q <= 1'b0; end else begin wbin <= wbin_nxt; wcommit <= wcommit_nxt; wgray <= bin2gray(wcommit_nxt); wfull <= wfull_nxt; rgray_s1 <= rgray; // sync read gray into write domain rgray_s2 <= rgray_s1; waddr_q <= wbin[AW-1:0]; wdata_q <= wdata; wwrite_q <= wr && !wfull; end end // ---------------- read domain ---------------- // `rd` is an accepted-read handshake by contract: every wrapper gates it // with !rempty. Do not gate it again here. The redundant internal gate // put rempty in front of the AW+1 pointer adder and, for a deep FIFO, also // in front of every RAM read-address bank. That feedback was the complete // Ch405 310 MHz setup-failure family. // Precompute the increment independent of `rd`, then select between the // hold/pop results. Writing this as `rbin + rd` put the registered pop // pulse on the carry input of the complete AW+1 adder and then through // Gray conversion + empty equality at 310 MHz. The explicit two-result // form is behavior-identical but leaves `rd` driving only final muxes. assign rbin_inc = rbin + {{AW{1'b0}}, 1'b1}; assign rgray_inc = bin2gray(rbin_inc); assign rbin_nxt = rd ? rbin_inc : rbin; assign rgray_nxt = rd ? rgray_inc : rgray; assign rempty_if_hold = (rgray == wgray_s2); assign rempty_if_pop = (rgray_inc == wgray_s2); assign rempty_nxt = rd ? rempty_if_pop : rempty_if_hold; always_ff @(posedge rclk or negedge rrst_n) begin if (!rrst_n) begin rbin <= '0; rgray <= '0; rempty <= 1'b1; wgray_s1 <= '0; wgray_s2 <= '0; end else begin rbin <= rbin_nxt; rgray <= rgray_nxt; rempty <= rempty_nxt; wgray_s1 <= wgray; // sync write gray into read domain wgray_s2 <= wgray_s1; end end generate if (QUADRANT_READ) begin : g_quadrant_storage // Four physical RAM quadrants: low/high payload width crossed with // lower/upper address half. Writes remain atomic and use the // staged allocation address exactly as the monolithic form does. always_ff @(posedge wclk) begin if (wwrite_q) begin if (waddr_q[AW-1]) begin mem_lo1[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W]; mem_hi1[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W]; end else begin mem_lo0[waddr_q[HALF_AW-1:0]] <= wdata_q[0 +: BANK_LO_W]; mem_hi0[waddr_q[HALF_AW-1:0]] <= wdata_q[BANK_LO_W +: BANK_HI_W]; end end end if (REGISTERED_READ) begin : g_registered_read always_ff @(posedge rclk) begin // Separate launch copies are intentional: each feeds only // one depth/width quadrant. rbank_data_q trails the // address-bank selector by the same cycle as the four // synchronous RAM outputs. raddr_lo0_q <= rbin_nxt[HALF_AW-1:0]; raddr_lo1_q <= rbin_nxt[HALF_AW-1:0]; raddr_hi0_q <= rbin_nxt[HALF_AW-1:0]; raddr_hi1_q <= rbin_nxt[HALF_AW-1:0]; rbank_addr_q <= rbin_nxt[AW-1]; rbank_data_q <= rbank_addr_q; rdata_lo0_q <= mem_lo0[raddr_lo0_q]; rdata_lo1_q <= mem_lo1[raddr_lo1_q]; rdata_hi0_q <= mem_hi0[raddr_hi0_q]; rdata_hi1_q <= mem_hi1[raddr_hi1_q]; end assign rdata = rbank_data_q ? {rdata_hi1_q, rdata_lo1_q} : {rdata_hi0_q, rdata_lo0_q}; end else begin : g_fwft_read assign rdata = rbin[AW-1] ? {mem_hi1[rbin[HALF_AW-1:0]], mem_lo1[rbin[HALF_AW-1:0]]} : {mem_hi0[rbin[HALF_AW-1:0]], mem_lo0[rbin[HALF_AW-1:0]]}; end end else if (BANKED_READ) begin : g_banked_storage // Two physical payload banks, written atomically from the same // staged tuple. Each registered read address drives only its own // half of the inferred RAM instead of the entire packed macro. always_ff @(posedge wclk) begin if (wwrite_q) begin mem_lo[waddr_q] <= wdata_q[0 +: BANK_LO_W]; mem_hi[waddr_q] <= wdata_q[BANK_LO_W +: BANK_HI_W]; end end if (REGISTERED_READ) begin : g_registered_read always_ff @(posedge rclk) begin raddr_lo_q <= rbin_nxt[AW-1:0]; raddr_hi_q <= rbin_nxt[AW-1:0]; rdata_lo_q <= mem_lo[raddr_lo_q]; rdata_hi_q <= mem_hi[raddr_hi_q]; end assign rdata = {rdata_hi_q, rdata_lo_q}; end else begin : g_fwft_read assign rdata = {mem_hi[rbin[AW-1:0]], mem_lo[rbin[AW-1:0]]}; end end else begin : g_monolithic_storage always_ff @(posedge wclk) if (wwrite_q) mem[waddr_q] <= wdata_q; if (REGISTERED_READ) begin : g_registered_read // A synchronous read lets Quartus use the memory output register // instead of timing a deep bank mux directly into request decode. // Read the current head every cycle and qualify rdata only at the // interface. The pointer still advances exclusively on `rd`, so // this does not consume an entry or change the one-cycle accepted- // read latency. Leaving the inferred RAM read enable permanently // active is important for a very wide FIFO: using `rd` as the RAM // enable made one pop register drive every physical data bank // (749 loads in the production request FIFO) at 310 MHz. // // Ch420: the Ch419 fit proved the enable cut and exposed the same // topology on portbaddr: rbin[6] directly drove 713 RAM-address // loads. `raddr_q` tracks the pointer's selected next value, so // before every edge it equals the current head address. The RAM // read therefore returns the same entry on the same edge as the // prior `mem[rbin]` form, including consecutive accepted pops, // while splitting pointer selection from physical RAM addressing. // // raddr_q/rdata_q intentionally have neither enables nor resets. // The FIFO cannot become nonempty until the synchronized write // pointer arrives, giving raddr_q multiple clocks to initialize to // zero after reset. Resetting the wide inferred read structure // previously created its own high-fanout recovery/setup family. always_ff @(posedge rclk) begin raddr_q <= rbin_nxt[AW-1:0]; rdata_q <= mem[raddr_q]; end assign rdata = rdata_q; end else begin : g_fwft_read assign rdata = mem[rbin[AW-1:0]]; end end endgenerate endmodule : gs_async_fifo