// ============================================================================ // gs_lpddr_scanout_lb.sv (Ch321 Brick 2) // // LINE-BUFFER LPDDR4B scanout — the architectural successor to the whole-frame // cache (gs_lpddr_scanout). Instead of mirroring the entire framebuffer in // on-chip RAM (which defeats the point of putting the FB in LPDDR), this holds // just two scanlines (three with the optional Ch438 low-pass): it displays row // L while prefetching row L+1. On-chip cost is O(width), not O(width*height). // // NARROW SCOPE (Ch321): the 128x128 PSMCT16 demo. The frame is LINEAR (the GS // writer mirrors the rasterizer's linear flush addresses), display window at // origin, 1:1 (MAG off) — so the reader serves pixel (col=pixel_x, line=pixel_y) // directly when inside the window. No general MAG/window handling beyond that. // // Two clock domains: // axi_clk (emif_clk) — AXI4 burst-read one row (ROW_BEATS beats) into a buffer // video_clk (design) — pixel_x/pixel_y index the active line buffer -> r/g/b // // Prefetch handshake: on each new display line (and at frame start) the video // side requests the next FB row via a toggle; the axi side fills the OTHER // buffer. `underflow` flags any pixel read before its row finished loading. // ============================================================================ `timescale 1ns/1ps module gs_lpddr_scanout_lb #( parameter [29:0] FB_BASE = 30'd0, parameter int STRIDE_BYTES = 256, // PSMCT16 128px*2B=256; PSMCT32 128px*4B=512 parameter int ROW_BEATS = 8, // STRIDE_BYTES / 32 (PSMCT16 128px=8; PSMCT32 128px=16) parameter int N_ROWS = 128, // Ch327a — PSMCT32 (ABGR8888, 8 px/256-bit beat) vs the original PSMCT16 (RGBA5551, // 16 px/beat). The Ch326 LPDDR-only spill framebuffer is PSMCT32 @ COLOR_SPILL_BASE, so the // line-buffer must decode it — NOT a config flip of the Ch321 PSMCT16/FB-at-0 path. parameter bit PSMCT32 = 1'b0, // Ch418 — the captured SH3 DISPLAY2 is 512 source pixels wide // (FBW=8) and MAGH=4, i.e. five VCKs per source sample. The board // emits a 640-pixel active line, so reducing that authentic 2560-VCK // display domain to VGA requires the exact nearest-neighbour map // source_x = floor(display_x * 4 / 5). // Keep this opt-in: legacy demos and directed scanout tests remain 1:1. // The implementation below is a five-state phase accumulator, not an // inferred divider, and therefore adds no wide arithmetic timing cone. parameter bit H_STRETCH_5_TO_4 = 1'b0, // Captured SH3 interlace presentation: DISPLAY2.DH=895 represents // 448 source lines, beginning at DISPFB2.DBY=32. Mapping those lines // onto the board's 480-line active raster is source_y = // V_SOURCE_START + floor(display_y*14/15). As with horizontal scale, // this remains opt-in and uses a tiny phase accumulator. parameter int V_SOURCE_START = 0, parameter bit V_STRETCH_15_TO_14 = 1'b0, // Ch436 — optional linear reconstruction between the two resident source // rows. The 15:14 mapper already keeps row L and L+1 in the alternating // line buffers, so this adds no framebuffer traffic or line storage. parameter bit V_LINEAR_FILTER = 1'b0, // Ch437 — horizontal linear reconstruction for the 5:4 presentation map. // A two-beat register cache provides x and x+1 from each resident row while // retaining ONE read port per physical line buffer. H_SOURCE_PIXELS is the // active source width (512 for SH3 DISPLAY2); zero means the physical stride. parameter bit H_LINEAR_FILTER = 1'b0, parameter int H_SOURCE_PIXELS = 0, // Ch438 — separable [1 2 1]/4 low-pass reconstruction in source space. // Three rotating line buffers retain rows y-1/y/y+1; the horizontal taps // reuse the Ch437 two-beat cache, so every physical RAM still has exactly // one registered read port. This is mutually exclusive with the Ch436/437 // linear filters in the SH3 profile and uses only adds plus shifts. parameter bit BINOMIAL_3X3_FILTER = 1'b0 )( // ---- AXI read clock domain (emif_clk) ---- input logic axi_clk, input logic axi_rst_n, input logic enable, // 1 = active (prefetch + serve) // ---- video clock domain (design_clk) ---- input logic video_clk, input logic frame_start, // vsync pulse/level (synced internally) input logic [11:0] pixel_x, // raster column (display) input logic [11:0] pixel_y, // raster line (display) input logic in_window, // PCRTC displayed-frame window gate output logic [7:0] r, output logic [7:0] g, output logic [7:0] b, // ---- status (axi_clk domain; bridge syncs) ---- output logic line_valid, // at least one row has been loaded output logic underflow, // a pixel was read before its row was ready (sticky) output logic [31:0] rd_errs, // non-OKAY read responses (cumulative) // ---- Ch442 A+B diagnostic (read-only; NOTHING here feeds fetch/output/arbitration/underflow) ---- // Splits the single top-level scan-error bit into independently observable causes and captures // the FIRST raw-underflow event of each video-source-enabled session as a bundled-data snapshot. output logic diag_rderr_nz, // LIVE (rd_errs != 0), reduced+registered in the axi_clk domain (single bit; the raw counter never crosses) output logic diag_valid, // first raw-underflow captured; held stable until video source disabled (!enable) output logic [29:0] diag_first, // SNAPSHOT, stable while diag_valid: {nf_s0[9:0], nf_v[9:0], scan_y[9:0]} output logic [6:0] diag_stat, // SNAPSHOT, stable while diag_valid: {vphase[3:0], line_valid, lookahead_cause, base_cause} // ---- AXI4 read channel to the EMIF user port (axi_clk, 256-bit) ---- output logic [29:0] araddr, output logic [1:0] arburst, output logic [6:0] arid, output logic [7:0] arlen, output logic [2:0] arsize, output logic arvalid, input logic arready, input logic [255:0] rdata, input logic [1:0] rresp, input logic rlast, input logic rvalid, output logic rready ); localparam int RB_BITS = $clog2(ROW_BEATS); // 3 for 8 assign arburst = 2'b01; // INCR assign arid = 7'd3; // distinct: writer=0, probe=1, frame-cache=2, line-buf=3 assign arlen = 8'd0; // SINGLE-BEAT per read — the only AXI read pattern proven on this // EMIF (writer/probe/frame-cache all use arlen=0). A multi-beat // burst (arlen=ROW_BEATS-1) was untested and garbled on hardware. assign arsize = 3'b101; // 32 bytes // Two line buffers for legacy/linear scanout; Ch438 enables a third so the // previous, current, and next source rows are resident simultaneously. Ch443e // adds a FOURTH buffer for BINOMIAL only: the 3x3 filter needs r-1/r/r+1 resident // (3 buffers), so to lead the prefetch by 2 (fetch r+2 while displaying r) without // overwriting r-1 (still needed) a 4th rotating buffer is required. Non-binomial // configs leave lb3 unused (pruned) and keep the legacy 2/3-buffer, lead-1 behavior. localparam int NBUF = BINOMIAL_3X3_FILTER ? 4 : 3; // buffers actually rotated logic [255:0] lb0 [0:ROW_BEATS-1]; logic [255:0] lb1 [0:ROW_BEATS-1]; logic [255:0] lb2 [0:ROW_BEATS-1]; logic [255:0] lb3 [0:ROW_BEATS-1]; // ================= video side (video_clk) ================= // No miss-prone request toggle. The video side just exposes the current // in-window display row; the axi side free-runs, fetching rows sequentially // and staying one row ahead (see below). disp_row_v resets on vsync. logic [$clog2(N_ROWS):0] disp_row_v; logic [2:0] fs_sync_v; wire fs_edge_v = fs_sync_v[1] && !fs_sync_v[2]; // RISING edge only: one reset per frame_start pulse // The buffer holding display line L is L&1 (row L is fetched into L&1). Select // it DIRECTLY from pixel_y[0] (tracks the current pixel) — a separately-registered // "disp_buf" lags by one cycle and corrupts col 0 of each line. logic [$clog2(N_ROWS):0] stretch_src_y_q; logic [3:0] stretch_vphase_q; localparam int V_SOURCE_BUF = V_SOURCE_START % NBUF; logic [1:0] stretch_buf_q; logic in_window_v_q; always_ff @(posedge video_clk) begin if (!enable) begin stretch_src_y_q <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); stretch_vphase_q <= 4'd0; stretch_buf_q <= 2'(V_SOURCE_BUF); in_window_v_q <= 1'b0; end else begin in_window_v_q <= in_window; if (fs_edge_v) begin stretch_src_y_q <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); stretch_vphase_q <= 4'd0; stretch_buf_q <= 2'(V_SOURCE_BUF); end else if (V_STRETCH_15_TO_14 && in_window_v_q && !in_window) begin // End of one output line. phase 0 repeats the current // source line once; phases 14..1 advance while counting // down, giving 15 output lines per 14 source lines. if (stretch_vphase_q == 4'd0) stretch_vphase_q <= 4'd14; else begin stretch_src_y_q <= stretch_src_y_q + 1'b1; stretch_vphase_q <= stretch_vphase_q - 1'b1; stretch_buf_q <= (stretch_buf_q == 2'(NBUF-1)) ? 2'd0 : stretch_buf_q + 1'b1; end end end end wire [$clog2(N_ROWS):0] scan_y = V_STRETCH_15_TO_14 ? stretch_src_y_q : ($clog2(N_ROWS)+1)'(pixel_y); wire disp_buf = scan_y[0]; wire [1:0] scan_buf3 = stretch_buf_q; always_ff @(posedge video_clk) begin if (!enable) begin disp_row_v <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); fs_sync_v <= 3'd0; end else begin fs_sync_v <= {fs_sync_v[1:0], frame_start}; if (fs_edge_v) disp_row_v <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); // Publish a stretch-row advance during horizontal blanking, not // only after the first active pixel. The AXI side then has the // full blank interval to replace the retired parity buffer with // row L+1 before linear scanout needs it. else if (scan_y < ($clog2(N_ROWS)+1)'(N_ROWS)) disp_row_v <= scan_y; end end // Ch418 horizontal presentation mapper. At output x=0 the state is // {src=0,phase=0}; successive active clocks produce source columns // 0,0,1,2,3,4,4,5,6,7,...,511 // for output columns 0..639. Blanking resets the state before each line. // phase==0 is the sole repeat; all other phases advance source_x. This is // algebraically identical to floor(x*4/5), using only a 3-bit decrement // and a 12-bit increment. logic [11:0] stretch_src_x_q; logic [2:0] stretch_phase_q; always_ff @(posedge video_clk) begin if (!enable || !in_window) begin stretch_src_x_q <= 12'd0; stretch_phase_q <= 3'd0; end else if (H_STRETCH_5_TO_4) begin if (stretch_phase_q == 3'd0) begin stretch_phase_q <= 3'd4; end else begin stretch_src_x_q <= stretch_src_x_q + 12'd1; stretch_phase_q <= stretch_phase_q - 3'd1; end end end wire [11:0] scan_x = H_STRETCH_5_TO_4 ? stretch_src_x_q : pixel_x; // Registered (sync-read) pixel: pick buffer + beat + within-beat lane from scan_x. // PSMCT32: 8 px/256-bit beat -> beat = scan_x>>3, lane = scan_x[2:0] (32-bit). // PSMCT16: 16 px/beat -> beat = scan_x>>4, lane = scan_x[3:0] (16-bit). localparam int PXSH = PSMCT32 ? 3 : 4; // px-per-beat shift localparam int PX_PER_ROW = PSMCT32 ? (STRIDE_BYTES/4) : (STRIDE_BYTES/2); localparam int H_SOURCE_PX = (H_SOURCE_PIXELS != 0) ? H_SOURCE_PIXELS : PX_PER_ROW; localparam int H_LAST_BEAT = (H_SOURCE_PX-1) >> PXSH; wire [RB_BITS-1:0] col_beat = scan_x[RB_BITS+PXSH-1 -: RB_BITS]; wire [3:0] col_lane = PSMCT32 ? {1'b0, scan_x[2:0]} : scan_x[3:0]; logic [255:0] prv_word_q, cur_word_q, nxt_word_q; // Only lane zero of the lookahead beat can be selected: x+1 crosses a // beat exactly when x is its final lane. Keep 32 bits, not another pair // of 256-bit payload registers. logic [31:0] prv_left_px_q, cur_left_px_q, nxt_left_px_q; logic [31:0] prv_look_px_q, cur_look_px_q, nxt_look_px_q; logic [3:0] lane_q, vphase_q; logic [2:0] hphase_q; logic first_source_x_q, last_source_x_q; logic in_q; generate if (H_LINEAR_FILTER || BINOMIAL_3X3_FILTER) begin : g_hlinear_cache // Slot parity equals beat parity. Horizontal blanking continually // primes beats 0 and 1 after the vertical row selector settles. // On entry to each subsequent beat, the retired slot is refilled // with beat+1. No second line-buffer read port is required. logic [255:0] lb0_cache0_q, lb0_cache1_q; logic [255:0] lb1_cache0_q, lb1_cache1_q; logic [255:0] lb2_cache0_q, lb2_cache1_q; logic [255:0] lb3_cache0_q, lb3_cache1_q; // Ch443e — binomial 4th buffer logic blank_prime_q; logic [RB_BITS-1:0] active_beat_q; logic [RB_BITS-1:0] video_rd_addr_q, video_rd_tag_q; logic video_rd_req_q, video_rd_valid_q; logic [255:0] lb0_video_rd_q, lb1_video_rd_q, lb2_video_rd_q, lb3_video_rd_q; // Keep the inferred line-buffer read ports canonical: exactly one // unconditional registered address and one registered data output // per physical array. The prior conditional multi-address reads // made Quartus expand both arrays into 30,720 flip-flops. always_ff @(posedge video_clk) begin lb0_video_rd_q <= lb0[video_rd_addr_q]; lb1_video_rd_q <= lb1[video_rd_addr_q]; lb2_video_rd_q <= lb2[video_rd_addr_q]; lb3_video_rd_q <= lb3[video_rd_addr_q]; // Ch443e — binomial 4th buffer read port video_rd_tag_q <= video_rd_addr_q; video_rd_valid_q <= video_rd_req_q; end always_ff @(posedge video_clk) begin // The registered RAM response arrives with its beat tag two // control edges after the request. Store physical lb0/lb1 // independently; row parity is selected only at pixel output. if (video_rd_valid_q) begin if (video_rd_tag_q[0]) begin lb0_cache1_q <= lb0_video_rd_q; lb1_cache1_q <= lb1_video_rd_q; lb2_cache1_q <= lb2_video_rd_q; lb3_cache1_q <= lb3_video_rd_q; end else begin lb0_cache0_q <= lb0_video_rd_q; lb1_cache0_q <= lb1_video_rd_q; lb2_cache0_q <= lb2_video_rd_q; lb3_cache0_q <= lb3_video_rd_q; end end if (!enable) begin blank_prime_q <= 1'b0; active_beat_q <= '0; video_rd_addr_q <= '0; video_rd_req_q <= 1'b0; in_q <= 1'b0; end else if (!in_window) begin // Alternate requests for beats 0 and 1 throughout blank. // VGA supplies far more than the four clocks needed for // both registered responses to settle into the cache. video_rd_addr_q <= blank_prime_q ? RB_BITS'(1) : '0; video_rd_req_q <= 1'b1; blank_prime_q <= ~blank_prime_q; active_beat_q <= '0; in_q <= 1'b0; end else begin video_rd_req_q <= 1'b0; // Source x and x+1 come from the two cached beats. The // current beat is selected by parity; the opposite slot // is its already-fetched successor. if (BINOMIAL_3X3_FILTER) begin // Select the three rotating physical rows. The first // displayed source row clamps y-1 to y; the final row // similarly clamps y+1. For a lane-zero sample the // opposite cache slot still contains the preceding // beat; by the time lane seven needs x+1 it contains // the newly fetched successor beat. // Ch443e — modulo-4 rotation across FOUR buffers. Current row is // buffer scan_buf3; prev=(b-1)%4, next=(b+1)%4; the 4th buffer // (b+2)%4 is being prefetched (row r+2, lead-2) and is not read here. // First row clamps prev->cur; last row clamps next->cur. case (scan_buf3) 2'd0: begin // cur=lb0 prev=lb3 next=lb1 prv_word_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb0_cache1_q : lb0_cache0_q) : (col_beat[0] ? lb3_cache1_q : lb3_cache0_q); cur_word_q <= col_beat[0] ? lb0_cache1_q : lb0_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb0_cache1_q : lb0_cache0_q) : (col_beat[0] ? lb1_cache1_q : lb1_cache0_q); prv_left_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb0_cache0_q[255:224] : lb0_cache1_q[255:224]) : (col_beat[0] ? lb3_cache0_q[255:224] : lb3_cache1_q[255:224]); cur_left_px_q <= col_beat[0] ? lb0_cache0_q[255:224] : lb0_cache1_q[255:224]; nxt_left_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb0_cache0_q[255:224] : lb0_cache1_q[255:224]) : (col_beat[0] ? lb1_cache0_q[255:224] : lb1_cache1_q[255:224]); prv_look_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]) : (col_beat[0] ? lb3_cache0_q[31:0] : lb3_cache1_q[31:0]); cur_look_px_q <= col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]) : (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]); end 2'd1: begin // cur=lb1 prev=lb0 next=lb2 prv_word_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb1_cache1_q : lb1_cache0_q) : (col_beat[0] ? lb0_cache1_q : lb0_cache0_q); cur_word_q <= col_beat[0] ? lb1_cache1_q : lb1_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb1_cache1_q : lb1_cache0_q) : (col_beat[0] ? lb2_cache1_q : lb2_cache0_q); prv_left_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb1_cache0_q[255:224] : lb1_cache1_q[255:224]) : (col_beat[0] ? lb0_cache0_q[255:224] : lb0_cache1_q[255:224]); cur_left_px_q <= col_beat[0] ? lb1_cache0_q[255:224] : lb1_cache1_q[255:224]; nxt_left_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb1_cache0_q[255:224] : lb1_cache1_q[255:224]) : (col_beat[0] ? lb2_cache0_q[255:224] : lb2_cache1_q[255:224]); prv_look_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]) : (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]); cur_look_px_q <= col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]) : (col_beat[0] ? lb2_cache0_q[31:0] : lb2_cache1_q[31:0]); end 2'd2: begin // cur=lb2 prev=lb1 next=lb3 prv_word_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb2_cache1_q : lb2_cache0_q) : (col_beat[0] ? lb1_cache1_q : lb1_cache0_q); cur_word_q <= col_beat[0] ? lb2_cache1_q : lb2_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb2_cache1_q : lb2_cache0_q) : (col_beat[0] ? lb3_cache1_q : lb3_cache0_q); prv_left_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb2_cache0_q[255:224] : lb2_cache1_q[255:224]) : (col_beat[0] ? lb1_cache0_q[255:224] : lb1_cache1_q[255:224]); cur_left_px_q <= col_beat[0] ? lb2_cache0_q[255:224] : lb2_cache1_q[255:224]; nxt_left_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb2_cache0_q[255:224] : lb2_cache1_q[255:224]) : (col_beat[0] ? lb3_cache0_q[255:224] : lb3_cache1_q[255:224]); prv_look_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb2_cache0_q[31:0] : lb2_cache1_q[31:0]) : (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]); cur_look_px_q <= col_beat[0] ? lb2_cache0_q[31:0] : lb2_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb2_cache0_q[31:0] : lb2_cache1_q[31:0]) : (col_beat[0] ? lb3_cache0_q[31:0] : lb3_cache1_q[31:0]); end default: begin // cur=lb3 prev=lb2 next=lb0 prv_word_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb3_cache1_q : lb3_cache0_q) : (col_beat[0] ? lb2_cache1_q : lb2_cache0_q); cur_word_q <= col_beat[0] ? lb3_cache1_q : lb3_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb3_cache1_q : lb3_cache0_q) : (col_beat[0] ? lb0_cache1_q : lb0_cache0_q); prv_left_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb3_cache0_q[255:224] : lb3_cache1_q[255:224]) : (col_beat[0] ? lb2_cache0_q[255:224] : lb2_cache1_q[255:224]); cur_left_px_q <= col_beat[0] ? lb3_cache0_q[255:224] : lb3_cache1_q[255:224]; nxt_left_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb3_cache0_q[255:224] : lb3_cache1_q[255:224]) : (col_beat[0] ? lb0_cache0_q[255:224] : lb0_cache1_q[255:224]); prv_look_px_q <= (scan_y <= ($clog2(N_ROWS)+1)'(V_SOURCE_START)) ? (col_beat[0] ? lb3_cache0_q[31:0] : lb3_cache1_q[31:0]) : (col_beat[0] ? lb2_cache0_q[31:0] : lb2_cache1_q[31:0]); cur_look_px_q <= col_beat[0] ? lb3_cache0_q[31:0] : lb3_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb3_cache0_q[31:0] : lb3_cache1_q[31:0]) : (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]); end endcase end else if (disp_buf) begin prv_word_q <= '0; cur_word_q <= col_beat[0] ? lb1_cache1_q : lb1_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb1_cache1_q : lb1_cache0_q) : (col_beat[0] ? lb0_cache1_q : lb0_cache0_q); cur_look_px_q <= col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]) : (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]); end else begin prv_word_q <= '0; cur_word_q <= col_beat[0] ? lb0_cache1_q : lb0_cache0_q; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb0_cache1_q : lb0_cache0_q) : (col_beat[0] ? lb1_cache1_q : lb1_cache0_q); cur_look_px_q <= col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]; nxt_look_px_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? (col_beat[0] ? lb0_cache0_q[31:0] : lb0_cache1_q[31:0]) : (col_beat[0] ? lb1_cache0_q[31:0] : lb1_cache1_q[31:0]); end if (!BINOMIAL_3X3_FILTER) begin prv_left_px_q <= '0; cur_left_px_q <= '0; nxt_left_px_q <= '0; prv_look_px_q <= '0; end lane_q <= col_lane; vphase_q <= stretch_vphase_q; hphase_q <= stretch_phase_q; first_source_x_q <= (scan_x == 12'd0); last_source_x_q <= (scan_x >= 12'(H_SOURCE_PX-1)); in_q <= (pixel_x < PX_PER_ROW) && (pixel_y < N_ROWS); if ((col_beat != active_beat_q) && (col_beat < RB_BITS'(H_LAST_BEAT))) begin active_beat_q <= col_beat; video_rd_addr_q <= col_beat + 1'b1; video_rd_req_q <= 1'b1; end end end end else begin : g_direct_read always_ff @(posedge video_clk) begin // One video read from each physical buffer supplies the current row // and its already-prefetched successor in parallel. Select them in // this same registered stage so parity cannot lag at a line boundary. if (disp_buf) begin prv_word_q <= '0; cur_word_q <= lb1[col_beat]; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? lb1[col_beat] : lb0[col_beat]; end else begin prv_word_q <= '0; cur_word_q <= lb0[col_beat]; nxt_word_q <= (scan_y + 1'b1 >= ($clog2(N_ROWS)+1)'(N_ROWS)) ? lb0[col_beat] : lb1[col_beat]; end cur_look_px_q <= '0; nxt_look_px_q <= '0; prv_left_px_q <= '0; cur_left_px_q <= '0; nxt_left_px_q <= '0; prv_look_px_q <= '0; lane_q <= col_lane; vphase_q <= stretch_vphase_q; hphase_q <= stretch_phase_q; first_source_x_q <= 1'b1; last_source_x_q <= 1'b1; in_q <= in_window && (pixel_x < PX_PER_ROW) && (pixel_y < N_ROWS); end end endgenerate // PSMCT32 ABGR8888 (r=[7:0],g=[15:8],b=[23:16]) — matches gs_lpddr_scanout (frame-cache). wire [31:0] px32_prv = prv_word_q[lane_q[2:0]*32 +: 32]; wire [31:0] px32_cur = cur_word_q[lane_q[2:0]*32 +: 32]; wire [31:0] px32_nxt = nxt_word_q[lane_q[2:0]*32 +: 32]; wire [31:0] px32_prv_left = first_source_x_q ? px32_prv : ((lane_q[2:0] == 3'd0) ? prv_left_px_q : prv_word_q[(lane_q[2:0]-1'b1)*32 +: 32]); wire [31:0] px32_cur_left = first_source_x_q ? px32_cur : ((lane_q[2:0] == 3'd0) ? cur_left_px_q : cur_word_q[(lane_q[2:0]-1'b1)*32 +: 32]); wire [31:0] px32_nxt_left = first_source_x_q ? px32_nxt : ((lane_q[2:0] == 3'd0) ? nxt_left_px_q : nxt_word_q[(lane_q[2:0]-1'b1)*32 +: 32]); wire [31:0] px32_prv_right = last_source_x_q ? px32_prv : ((lane_q[2:0] == 3'd7) ? prv_look_px_q : prv_word_q[(lane_q[2:0]+1'b1)*32 +: 32]); wire [31:0] px32_cur_right = last_source_x_q ? px32_cur : ((lane_q[2:0] == 3'd7) ? cur_look_px_q : cur_word_q[(lane_q[2:0]+1'b1)*32 +: 32]); wire [31:0] px32_nxt_right = last_source_x_q ? px32_nxt : ((lane_q[2:0] == 3'd7) ? nxt_look_px_q : nxt_word_q[(lane_q[2:0]+1'b1)*32 +: 32]); // PSMCT16 RGBA5551 5-bit lanes expanded to 8-bit. wire [15:0] px16_prv = prv_word_q[lane_q*16 +: 16]; wire [15:0] px16_cur = cur_word_q[lane_q*16 +: 16]; wire [15:0] px16_nxt = nxt_word_q[lane_q*16 +: 16]; wire [15:0] px16_prv_left = first_source_x_q ? px16_prv : ((lane_q == 4'd0) ? prv_left_px_q[15:0] : prv_word_q[(lane_q-1'b1)*16 +: 16]); wire [15:0] px16_cur_left = first_source_x_q ? px16_cur : ((lane_q == 4'd0) ? cur_left_px_q[15:0] : cur_word_q[(lane_q-1'b1)*16 +: 16]); wire [15:0] px16_nxt_left = first_source_x_q ? px16_nxt : ((lane_q == 4'd0) ? nxt_left_px_q[15:0] : nxt_word_q[(lane_q-1'b1)*16 +: 16]); wire [15:0] px16_prv_right = last_source_x_q ? px16_prv : ((lane_q == 4'd15) ? prv_look_px_q[15:0] : prv_word_q[(lane_q+1'b1)*16 +: 16]); wire [15:0] px16_cur_right = last_source_x_q ? px16_cur : ((lane_q == 4'd15) ? cur_look_px_q[15:0] : cur_word_q[(lane_q+1'b1)*16 +: 16]); wire [15:0] px16_nxt_right = last_source_x_q ? px16_nxt : ((lane_q == 4'd15) ? nxt_look_px_q[15:0] : nxt_word_q[(lane_q+1'b1)*16 +: 16]); wire [7:0] r16_cur = {px16_cur[4:0], px16_cur[4:2]}; wire [7:0] g16_cur = {px16_cur[9:5], px16_cur[9:7]}; wire [7:0] b16_cur = {px16_cur[14:10], px16_cur[14:12]}; wire [7:0] r16_nxt = {px16_nxt[4:0], px16_nxt[4:2]}; wire [7:0] g16_nxt = {px16_nxt[9:5], px16_nxt[9:7]}; wire [7:0] b16_nxt = {px16_nxt[14:10], px16_nxt[14:12]}; wire [7:0] r16_prv = {px16_prv[4:0], px16_prv[4:2]}; wire [7:0] g16_prv = {px16_prv[9:5], px16_prv[9:7]}; wire [7:0] b16_prv = {px16_prv[14:10], px16_prv[14:12]}; wire [7:0] r16_prv_left = {px16_prv_left[4:0], px16_prv_left[4:2]}; wire [7:0] g16_prv_left = {px16_prv_left[9:5], px16_prv_left[9:7]}; wire [7:0] b16_prv_left = {px16_prv_left[14:10], px16_prv_left[14:12]}; wire [7:0] r16_cur_left = {px16_cur_left[4:0], px16_cur_left[4:2]}; wire [7:0] g16_cur_left = {px16_cur_left[9:5], px16_cur_left[9:7]}; wire [7:0] b16_cur_left = {px16_cur_left[14:10], px16_cur_left[14:12]}; wire [7:0] r16_nxt_left = {px16_nxt_left[4:0], px16_nxt_left[4:2]}; wire [7:0] g16_nxt_left = {px16_nxt_left[9:5], px16_nxt_left[9:7]}; wire [7:0] b16_nxt_left = {px16_nxt_left[14:10], px16_nxt_left[14:12]}; wire [7:0] r16_prv_right = {px16_prv_right[4:0], px16_prv_right[4:2]}; wire [7:0] g16_prv_right = {px16_prv_right[9:5], px16_prv_right[9:7]}; wire [7:0] b16_prv_right = {px16_prv_right[14:10], px16_prv_right[14:12]}; wire [7:0] r16_cur_right = {px16_cur_right[4:0], px16_cur_right[4:2]}; wire [7:0] g16_cur_right = {px16_cur_right[9:5], px16_cur_right[9:7]}; wire [7:0] b16_cur_right = {px16_cur_right[14:10], px16_cur_right[14:12]}; wire [7:0] r16_nxt_right = {px16_nxt_right[4:0], px16_nxt_right[4:2]}; wire [7:0] g16_nxt_right = {px16_nxt_right[9:5], px16_nxt_right[9:7]}; wire [7:0] b16_nxt_right = {px16_nxt_right[14:10], px16_nxt_right[14:12]}; function automatic logic [7:0] blend15( input logic [7:0] cur, input logic [7:0] nxt, input logic [3:0] frac ); logic [12:0] weighted; begin if (frac == 4'd0) blend15 = cur; else begin weighted = ((4'd15-frac) * cur) + (frac * nxt) + 13'd7; blend15 = weighted / 13'd15; end end endfunction function automatic logic [7:0] blend5( input logic [7:0] left, input logic [7:0] right, input logic [2:0] frac ); logic [10:0] weighted; begin if (frac == 3'd0) blend5 = left; else begin weighted = ((3'd5-frac) * left) + (frac * right) + 11'd2; blend5 = weighted / 11'd5; end end endfunction function automatic logic [7:0] binom3( input logic [7:0] left, input logic [7:0] center, input logic [7:0] right ); logic [9:0] weighted; begin weighted = {2'b0,left} + {1'b0,center,1'b0} + {2'b0,right} + 10'd2; binom3 = weighted[9:2]; end endfunction wire [7:0] prv_r = PSMCT32 ? px32_prv[7:0] : r16_prv; wire [7:0] prv_g = PSMCT32 ? px32_prv[15:8] : g16_prv; wire [7:0] prv_b = PSMCT32 ? px32_prv[23:16] : b16_prv; wire [7:0] cur_r = PSMCT32 ? px32_cur[7:0] : r16_cur; wire [7:0] cur_g = PSMCT32 ? px32_cur[15:8] : g16_cur; wire [7:0] cur_b = PSMCT32 ? px32_cur[23:16] : b16_cur; wire [7:0] nxt_r = PSMCT32 ? px32_nxt[7:0] : r16_nxt; wire [7:0] nxt_g = PSMCT32 ? px32_nxt[15:8] : g16_nxt; wire [7:0] nxt_b = PSMCT32 ? px32_nxt[23:16] : b16_nxt; wire [7:0] prv_left_r = PSMCT32 ? px32_prv_left[7:0] : r16_prv_left; wire [7:0] prv_left_g = PSMCT32 ? px32_prv_left[15:8] : g16_prv_left; wire [7:0] prv_left_b = PSMCT32 ? px32_prv_left[23:16] : b16_prv_left; wire [7:0] cur_left_r = PSMCT32 ? px32_cur_left[7:0] : r16_cur_left; wire [7:0] cur_left_g = PSMCT32 ? px32_cur_left[15:8] : g16_cur_left; wire [7:0] cur_left_b = PSMCT32 ? px32_cur_left[23:16] : b16_cur_left; wire [7:0] nxt_left_r = PSMCT32 ? px32_nxt_left[7:0] : r16_nxt_left; wire [7:0] nxt_left_g = PSMCT32 ? px32_nxt_left[15:8] : g16_nxt_left; wire [7:0] nxt_left_b = PSMCT32 ? px32_nxt_left[23:16] : b16_nxt_left; wire [7:0] prv_right_r = PSMCT32 ? px32_prv_right[7:0] : r16_prv_right; wire [7:0] prv_right_g = PSMCT32 ? px32_prv_right[15:8] : g16_prv_right; wire [7:0] prv_right_b = PSMCT32 ? px32_prv_right[23:16] : b16_prv_right; wire [7:0] cur_right_r = PSMCT32 ? px32_cur_right[7:0] : r16_cur_right; wire [7:0] cur_right_g = PSMCT32 ? px32_cur_right[15:8] : g16_cur_right; wire [7:0] cur_right_b = PSMCT32 ? px32_cur_right[23:16] : b16_cur_right; wire [7:0] nxt_right_r = PSMCT32 ? px32_nxt_right[7:0] : r16_nxt_right; wire [7:0] nxt_right_g = PSMCT32 ? px32_nxt_right[15:8] : g16_nxt_right; wire [7:0] nxt_right_b = PSMCT32 ? px32_nxt_right[23:16] : b16_nxt_right; wire [3:0] filter_frac = V_LINEAR_FILTER ? vphase_q : 4'd0; wire [2:0] hfilter_frac = H_LINEAR_FILTER ? hphase_q : 3'd0; wire [7:0] left_r = blend15(cur_r, nxt_r, filter_frac); wire [7:0] left_g = blend15(cur_g, nxt_g, filter_frac); wire [7:0] left_b = blend15(cur_b, nxt_b, filter_frac); wire [7:0] right_r = blend15(cur_right_r, nxt_right_r, filter_frac); wire [7:0] right_g = blend15(cur_right_g, nxt_right_g, filter_frac); wire [7:0] right_b = blend15(cur_right_b, nxt_right_b, filter_frac); wire [7:0] linear_out_r = blend5(left_r, right_r, hfilter_frac); wire [7:0] linear_out_g = blend5(left_g, right_g, hfilter_frac); wire [7:0] linear_out_b = blend5(left_b, right_b, hfilter_frac); wire [7:0] bin_prv_r = binom3(prv_left_r, prv_r, prv_right_r); wire [7:0] bin_prv_g = binom3(prv_left_g, prv_g, prv_right_g); wire [7:0] bin_prv_b = binom3(prv_left_b, prv_b, prv_right_b); wire [7:0] bin_cur_r = binom3(cur_left_r, cur_r, cur_right_r); wire [7:0] bin_cur_g = binom3(cur_left_g, cur_g, cur_right_g); wire [7:0] bin_cur_b = binom3(cur_left_b, cur_b, cur_right_b); wire [7:0] bin_nxt_r = binom3(nxt_left_r, nxt_r, nxt_right_r); wire [7:0] bin_nxt_g = binom3(nxt_left_g, nxt_g, nxt_right_g); wire [7:0] bin_nxt_b = binom3(nxt_left_b, nxt_b, nxt_right_b); wire [7:0] out_r = BINOMIAL_3X3_FILTER ? binom3(bin_prv_r, bin_cur_r, bin_nxt_r) : linear_out_r; wire [7:0] out_g = BINOMIAL_3X3_FILTER ? binom3(bin_prv_g, bin_cur_g, bin_nxt_g) : linear_out_g; wire [7:0] out_b = BINOMIAL_3X3_FILTER ? binom3(bin_prv_b, bin_cur_b, bin_nxt_b) : linear_out_b; assign r = !in_q ? 8'd0 : out_r; assign g = !in_q ? 8'd0 : out_g; assign b = !in_q ? 8'd0 : out_b; // ================= axi side (axi_clk) — row fill FSM ================= // free-running prefetcher: fetch rows sequentially, staying <= disp_row+1 ahead. // disp_row crosses video->axi (slowly-changing; the +1 throttle tolerates a 1-off // transient). frame_start is edge-detected here to reset next_fetch every frame. logic [2:0] fs_sync_e; wire fs_edge_e = fs_sync_e[1] && !fs_sync_e[2]; // RISING edge only: one prefetch restart per frame_start pulse logic [$clog2(N_ROWS):0] disp_row_s0, disp_row_limit_e; logic [$clog2(N_ROWS):0] next_fetch; // next row to load (0..N_ROWS) logic [1:0] next_fetch_buf; typedef enum logic [1:0] { L_IDLE, L_AR, L_R, L_C } lstate_t; lstate_t lst; logic [$clog2(N_ROWS):0] cur_row; logic [1:0] cur_buf; logic [RB_BITS:0] beat; logic fs_pending; // a vsync restart is pending; applied in L_IDLE (never mid-read) // Ch358 (Codex) — one-entry RESPONSE STAGE: the 26.1 STA leader (WNS -0.884 at 640) was the EMIF read-FIFO // RAM output driving lb0/lb1's write port in the SAME rvalid cycle (RAM->bus->RAM in one 310MHz period). // L_R now only CAPTURES {rdata, beat, buf, last} into per-buffer physical write registers. L_C commits the // selected LB RAM; next_fetch/line_valid advance only after that commit. One outstanding read. // Ch439b — physical write-port stage. The Ch438 fit exposed a route-only // r_data_q -> lb1 RAM path at 310 MHz after adding the third line buffer. // Give each inferred RAM its own data/address/enable launch registers so // the fitter can place them beside that RAM instead of routing one shared // 256-bit register bank across all three memories. Capture directly from // AXI in L_R and commit in L_C: the first Ch439 L_R->L_C->L_W form added a // cycle per beat and produced sustained line-buffer underflow on hardware. // Data/address registers intentionally have no reset; the reset write- // enables qualify them. logic [255:0] lb0_wdata_q, lb1_wdata_q, lb2_wdata_q, lb3_wdata_q; logic [RB_BITS-1:0] lb0_waddr_q, lb1_waddr_q, lb2_waddr_q, lb3_waddr_q; logic lb0_we_q, lb1_we_q, lb2_we_q, lb3_we_q; // Ch443e — binomial 4th buffer // Ch442 diag: registered single-bit reduction of the read-error counter (emif domain). // Diagnostic-only; nothing downstream reads it, so it cannot perturb the fetch FSM. logic rd_err_nz_q; assign diag_rderr_nz = rd_err_nz_q; always_ff @(posedge axi_clk) begin if (!axi_rst_n) begin fs_sync_e <= 3'd0; disp_row_s0 <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); disp_row_limit_e <= ($clog2(N_ROWS)+1)'(V_SOURCE_START + (BINOMIAL_3X3_FILTER ? 2 : 1)); next_fetch <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); next_fetch_buf <= BINOMIAL_3X3_FILTER ? 2'(V_SOURCE_BUF) : {1'b0, 1'(V_SOURCE_START)}; lst <= L_IDLE; araddr <= '0; arvalid <= 1'b0; rready <= 1'b0; cur_row <= '0; cur_buf <= 2'd0; beat <= '0; line_valid <= 1'b0; rd_errs <= 32'd0; fs_pending <= 1'b0; lb0_we_q <= 1'b0; lb1_we_q <= 1'b0; lb2_we_q <= 1'b0; lb3_we_q <= 1'b0; rd_err_nz_q <= 1'b0; end else begin fs_sync_e <= {fs_sync_e[1:0], frame_start}; disp_row_s0 <= disp_row_v; // 2-FF sync of the display row // Register the already-incremented throttle limit. This remains the // second CDC stage, but removes disp_row -> (+1) -> compare -> araddr // enable from one 310 MHz cycle (the post-alpha fit's -0.125 ns family). // The extra bit represents N_ROWS exactly on the final display row. // Ch443e — BINOMIAL leads by 2 (fetch disp_row+2 while displaying disp_row) so // the 3x3 filter's lookahead row (disp_row+1) is FULLY resident before it is read; // the 4th rotating buffer holds the in-flight disp_row+2. Legacy stays lead-1. disp_row_limit_e <= disp_row_s0 + (BINOMIAL_3X3_FILTER ? 2'd2 : 2'd1); // Ch439c — RAM-local response pipeline. Commit the response // captured on the preceding cycle while the AXI FSM advances to // (or waits for) the next single-beat read. This keeps the // rdata->local-register->RAM timing cut without paying an L_C // bubble after every beat. The final beat uses L_C only as a // one-cycle row-end flush before next_fetch becomes visible. if (lb0_we_q) lb0[lb0_waddr_q] <= lb0_wdata_q; if (lb1_we_q) lb1[lb1_waddr_q] <= lb1_wdata_q; if (lb2_we_q) lb2[lb2_waddr_q] <= lb2_wdata_q; if (lb3_we_q) lb3[lb3_waddr_q] <= lb3_wdata_q; lb0_we_q <= 1'b0; lb1_we_q <= 1'b0; lb2_we_q <= 1'b0; lb3_we_q <= 1'b0; // Ch442 diag: register the read-error-nonzero flag (emif domain). One cycle of lag // vs rd_errs is irrelevant for a stuck-nonzero indicator; keeps a clean launch flop. rd_err_nz_q <= (rd_errs != 32'd0); // vsync: mark a prefetch restart. DEFER it to L_IDLE so an in-flight AXI // read is never aborted mid-handshake (which would deadlock the slave). if (fs_edge_e) fs_pending <= 1'b1; case (lst) L_IDLE: begin if (fs_pending) begin next_fetch <= ($clog2(N_ROWS)+1)'(V_SOURCE_START); next_fetch_buf <= BINOMIAL_3X3_FILTER ? 2'(V_SOURCE_BUF) : {1'b0, 1'(V_SOURCE_START)}; // restart at the captured display source row fs_pending <= 1'b0; end else if (enable && (next_fetch < N_ROWS) && (next_fetch <= disp_row_limit_e)) begin cur_row <= next_fetch; cur_buf <= next_fetch_buf; araddr <= FB_BASE + (next_fetch * STRIDE_BYTES); beat <= '0; arvalid <= 1'b1; lst <= L_AR; end end L_AR: begin if (arready) begin arvalid <= 1'b0; rready <= 1'b1; lst <= L_R; end end L_R: begin if (rvalid) begin rready <= 1'b0; if (rresp != 2'b00) rd_errs <= rd_errs + 32'd1; // accepted read; a non-OKAY response is real lb0_we_q <= 1'b0; lb1_we_q <= 1'b0; lb2_we_q <= 1'b0; lb3_we_q <= 1'b0; if (fs_pending || fs_edge_e) begin // Ch443c (Codex): a vsync restart is pending. This single-beat AXI // transaction is COMPLETE (rvalid accepted), so it is protocol-safe to // ABANDON the remainder of this now-obsolete row: do NOT commit the beat, // do NOT issue the next old-row AR, and do NOT publish/increment next_fetch. // Return to L_IDLE, where fs_pending restarts the prefetch at V_SOURCE_START. // This removes up to a full row of restart latency during vertical blanking // (the reason the warm-up first-fetch could lag the first displayed row). lst <= L_IDLE; end else begin // Capture directly into the selected RAM-local port stage; L_C commits. case (cur_buf) 2'd1: begin lb1_wdata_q <= rdata; lb1_waddr_q <= beat[RB_BITS-1:0]; lb1_we_q <= 1'b1; end 2'd2: begin lb2_wdata_q <= rdata; lb2_waddr_q <= beat[RB_BITS-1:0]; lb2_we_q <= 1'b1; end 2'd3: begin lb3_wdata_q <= rdata; lb3_waddr_q <= beat[RB_BITS-1:0]; lb3_we_q <= 1'b1; end default: begin lb0_wdata_q <= rdata; lb0_waddr_q <= beat[RB_BITS-1:0]; lb0_we_q <= 1'b1; end endcase if (beat == ROW_BEATS-1) begin // The local register captures this last response // now; L_C flushes it into RAM on the next edge. lst <= L_C; end else begin // RAM-local stage commits independently above, so // immediately issue the next single-beat read. beat <= beat + 1'b1; araddr <= araddr + 30'd32; arvalid <= 1'b1; lst <= L_AR; end end end end L_C: begin // The global pipeline commit above writes the final beat // on this edge. Publish the completed row only now. line_valid <= 1'b1; next_fetch <= next_fetch + 1'b1; // rows 0..next_fetch are now loaded if (BINOMIAL_3X3_FILTER) next_fetch_buf <= (next_fetch_buf == 2'd3) ? 2'd0 // Ch443e — mod-4 rotation : next_fetch_buf + 1'b1; else next_fetch_buf <= {1'b0, ~next_fetch_buf[0]}; lst <= L_IDLE; end default: lst <= L_IDLE; endcase end end // underflow (sticky, video domain): an in-window pixel for line pixel_y is read // before that row was prefetched. The axi side loads rows 0..next_fetch-1, so row // pixel_y is ready iff pixel_y < next_fetch. next_fetch crosses axi->video synced // (slowly-changing; a 1-off transient is harmless). Resets on vsync. logic [$clog2(N_ROWS):0] nf_s0, nf_v; logic underflow_v; always_ff @(posedge video_clk) begin nf_s0 <= next_fetch; nf_v <= nf_s0; if (!enable || fs_edge_v) underflow_v <= 1'b0; else if (in_window && (scan_y < ($clog2(N_ROWS)+1)'(N_ROWS)) && ((scan_y >= nf_v) || ((BINOMIAL_3X3_FILTER || (V_LINEAR_FILTER && (stretch_vphase_q != 4'd0))) && (scan_y + 1'b1 < ($clog2(N_ROWS)+1)'(N_ROWS)) && (scan_y + 1'b1 >= nf_v)))) underflow_v <= 1'b1; end assign underflow = underflow_v; // ================= Ch442 A+B first-failure diagnostic (read-only) ================= // Re-express the EXACT raw-underflow predicate (identical boolean to the sticky latch // above) as combinational wires so the capture can name WHICH sub-cause fired. These // wires and the registers below drive ONLY the diag_* outputs — never the fetch FSM, // the pixel path, arbitration, or underflow_v. The existing latch block is untouched. // NOTE the `!fs_edge_v` term: the sticky latch gives frame-start CLEAR priority // (`if (!enable || fs_edge_v) ... else if (predicate)`), so on an fs_edge_v cycle it // never sets. Mirror that exactly here or the diagnostic could capture a frame-start // coincident miss the real latch suppresses. wire uf_in_range = enable && !fs_edge_v && in_window && (scan_y < ($clog2(N_ROWS)+1)'(N_ROWS)); wire uf_base_cond = uf_in_range && (scan_y >= nf_v); wire uf_look_cond = uf_in_range && (BINOMIAL_3X3_FILTER || (V_LINEAR_FILTER && (stretch_vphase_q != 4'd0))) && (scan_y + 1'b1 < ($clog2(N_ROWS)+1)'(N_ROWS)) && (scan_y + 1'b1 >= nf_v); wire raw_uf_cond = uf_base_cond || uf_look_cond; // line_valid is axi_clk-domain; 2-FF into video_clk for coherent capture. logic [1:0] line_valid_vsync; // First-failure snapshot (video_clk). Ch443c (Codex): PER-FRAME — diag_valid clears on // fs_edge_v as well as !enable, mirroring underflow_v, then records the first raw miss of // the NEW frame. This discards the pre-display WARM-UP capture: the line-buffer reader is // enabled by video_src_emif immediately, but the HDMI mux only switches to it at the next // vsync, so the very first post-enable miss is not an HDMI-visible one. Clearing per frame // makes the snapshot report the first miss AFTER the real frame boundary. The bundled // snapshot ({nf_s0,nf_v,scan_y} + cause/line_valid/vphase) stays stable within a frame, // so the bridge can transfer it coherently with a single synchronized valid. // (* preserve *): keep these capture regs as named keepers so the SDC bundled-data // hold-false/max-skew/net-delay constraint can bind to them (they must not be merged). (* preserve *) logic diag_valid_q, diag_base_q, diag_look_q, diag_lv_q; (* preserve *) logic [3:0] diag_vphase_q; (* preserve *) logic [$clog2(N_ROWS):0] diag_scan_y_q, diag_nf_v_q, diag_nf_s0_q; always_ff @(posedge video_clk) begin line_valid_vsync <= {line_valid_vsync[0], line_valid}; if (!enable || fs_edge_v) begin diag_valid_q <= 1'b0; // Ch443c: per-frame clear (discards pre-display warm-up capture) end else if (!diag_valid_q && raw_uf_cond) begin diag_valid_q <= 1'b1; diag_base_q <= uf_base_cond; diag_look_q <= uf_look_cond; diag_lv_q <= line_valid_vsync[1]; diag_vphase_q <= stretch_vphase_q; diag_scan_y_q <= scan_y; diag_nf_v_q <= nf_v; diag_nf_s0_q <= nf_s0; end end assign diag_valid = diag_valid_q; assign diag_first = {10'(diag_nf_s0_q), 10'(diag_nf_v_q), 10'(diag_scan_y_q)}; assign diag_stat = {diag_vphase_q, diag_lv_q, diag_look_q, diag_base_q}; endmodule