aboutsummaryrefslogtreecommitdiffstats
path: root/WinQuake/pum_render.c
diff options
context:
space:
mode:
authorLeonard Kugis <leonard@kug.is>2026-10-05 01:36:32 +0200
committerLeonard Kugis <leonard@kug.is>2026-10-05 01:36:32 +0200
commit57fdd82018e8f3449987cca19f3a41f630d7d62e (patch)
tree50e8f0dfc58c9dd704ab12c7490b4a1898a02f72 /WinQuake/pum_render.c
parent5ba1b06a2cbcbe9b2823dce3ac05663344075029 (diff)
downloadquake-pum-57fdd82018e8f3449987cca19f3a41f630d7d62e.tar.gz
Implemented PuM pipeline
Diffstat (limited to 'WinQuake/pum_render.c')
-rw-r--r--WinQuake/pum_render.c582
1 files changed, 582 insertions, 0 deletions
diff --git a/WinQuake/pum_render.c b/WinQuake/pum_render.c
new file mode 100644
index 0000000..e030f72
--- /dev/null
+++ b/WinQuake/pum_render.c
@@ -0,0 +1,582 @@
+/*
+==============================================================================
+PUM_RENDER.C -- DRAM Bender based Process-using-Memory (PuM) renderer
+==============================================================================
+
+This file implements the host-side half of the PuM rendering pipeline. It
+turns high-level "bulk bitwise" requests from the Quake software renderer
+into DRAM Bender *Program* objects and executes them on the AMD Alveo U200
+board over PCI Express (XDMA).
+
+The underlying physical primitive is described in the COTS DRAM paper
+(arXiv:2402.18736) and the cs3_bitwise case study:
+
+ * Ambit-style triple-row activation (TRA) realises MAJ3, from which AND
+ and OR are obtained by binding one operand row to 0/1.
+ * NOT is realised through neighbouring-subarray activation (shared sense
+ amplifier), producing the negation of the source half-row.
+ * AND + NOT => NAND, OR + NOT => NOR, and together {AND, OR, NOT} is
+ functionally complete, so XOR and every other Boolean function are
+ expressed as compositions.
+
+A crucial practical consideration is that COTS PuM is probabilistic: the
+papers report ~94-98% per-cell success rates on SK Hynix devices. We
+therefore run every PuM operation through the reliability strategy below:
+
+ 1. Operands are staged into *designated* rows (T0/T1) via RowClone-style
+ back-to-back activation, which refreshes them in the process.
+ 2. The operation is repeated three times (majority vote) using three
+ separate destination rows, giving triple modular redundancy (TMR),
+ which is the only ECC scheme known to be homomorphic over bitwise
+ operations.
+ 3. The three candidate results are read back to the host, where a CPU
+ majority-vote reconstructs the final, corrected value.
+
+This is the same TMR scheme the Ambit paper identifies as the sole
+compatible ECC, and it makes the PuM path functionally correct despite the
+probabilistic nature of the DRAM analogue operations.
+*/
+
+#ifdef RENDER_PUM
+
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <assert.h>
+
+typedef unsigned char byte;
+
+#ifndef UNUSED
+#define UNUSED(x) (x = x)
+#endif
+
+#include "instruction.h"
+#include "prog.h"
+#include "platform.h"
+
+#include "pum_render.h"
+
+// Constants in sync with Gateware
+
+// Stride registers are fixed by the SoftMC ISA
+#define PUM_CASR 0
+#define PUM_BASR 1
+#define PUM_RASR 2
+
+// General-purpose registers used by generated programs
+#define PUM_BAR 3
+#define PUM_RAR 4
+#define PUM_CAR 5
+#define PUM_SRC 6
+#define PUM_DST 7
+#define PUM_CONST 8
+#define PUM_LIMIT 11
+#define PUM_TEMP 12
+#define PUM_TEMP2 13
+#define PUM_COL_CNT 14
+
+// Bank 0, compute scratchpad
+// HMA81GU6AFR8N-UH: 17 row bits, 32768 rows per bank
+
+#define PUM_TARGET_BANK 0
+#define PUM_ROW_T0 0x20
+#define PUM_ROW_T1 0x21
+#define PUM_ROW_T2 0x22
+#define PUM_ROW_T3 0x23
+#define PUM_ROW_C0 0x30
+#define PUM_ROW_C1 0x31
+#define PUM_ROW_DST0 0x1000
+#define PUM_ROW_DST1 0x1001
+#define PUM_ROW_DST2 0x1002
+
+// Number of columns (64-bit data beats) in a single row
+#define PUM_NUM_COLS 128
+
+// Platform state
+
+static SoftMCPlatform *pum_platform = NULL;
+static int pum_active = 0;
+
+static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word);
+static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst);
+static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2);
+static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst);
+static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg);
+static void PUM_Execute (Program *p);
+static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n);
+
+int PUM_Init (void)
+{
+ if (pum_platform)
+ return pum_active;
+
+ pum_platform = new SoftMCPlatform();
+ if (!pum_platform)
+ return 0;
+
+ if (pum_platform->init() != SOFTMC_SUCCESS)
+ {
+ // Board not present or XDMA unavailable, CPU fallback
+ delete pum_platform;
+ pum_platform = NULL;
+ pum_active = 0;
+ return 0;
+ }
+
+ pum_platform->reset_fpga();
+
+ // Initialize control rows C0, C1
+ {
+ Program init;
+ PUM_WriteRowConst (&init, PUM_ROW_C0, 0x00000000);
+ PUM_WriteRowConst (&init, PUM_ROW_C1, 0xffffffff);
+ PUM_Execute (&init);
+ }
+
+ pum_active = 1;
+ return 1;
+}
+
+void PUM_Shutdown (void)
+{
+ if (pum_platform)
+ {
+ delete pum_platform;
+ pum_platform = NULL;
+ }
+ pum_active = 0;
+}
+
+int PUM_Active (void)
+{
+ return pum_active;
+}
+
+// Low-level SoftMC program helpers
+
+static Inst PUM_AllNops (void)
+{
+ return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP());
+}
+
+/*
+ * Add a single DDR command followed by 'after' NOP cycles (padding). The
+ * SoftMC engine clocks at ~666 MHz (1.5 ns period); timing values below are
+ * expressed in those fabric cycles.
+ *
+ * tRAS ~ 35 ns -> 24 cycles
+ * tRCD ~ 13.5 ns -> 9 cycles
+ * tRP ~ 13.5 ns -> 9 cycles
+ * tWR ~ 15 ns -> 10 cycles
+ *
+ * For the reduced timings required by PuM (ACT->PRE->ACT with tRAS and
+ * tRP < 3 ns), we insert just a single NOP between commands, i.e. roughly
+ * 1.5 ns. This matches past case studies.
+ */
+static void PUM_AddDDRCmd (Program *p, Mininst cmd, int after_nops)
+{
+ int i;
+ p->add_inst(__pack_mininsts(cmd, SMC_NOP(), SMC_NOP(), SMC_NOP()));
+ for (i = 0; i < after_nops; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * Write a full row with a repeating 32-bit word. This is the slow, safe
+ * path used for control-row initialisation only; data rows use RowClone.
+ */
+static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word)
+{
+ int i;
+
+ // Load the 16 32-bit words of the wide write-data register
+ for (i = 0; i < 16; i++)
+ {
+ p->add_inst(SMC_LI(word, PUM_TEMP));
+ p->add_inst(SMC_LDWD(PUM_TEMP, i));
+ }
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ // tRCD-1
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+
+ p->add_label("PUM:WRROW");
+ p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:WRROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ // tRP
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * RowClone-style fast copy of `src` into `dst` within the same subarray.
+ * Two back-to-back ACTIVATEs followed by a restore to the destination and
+ * a PRECHARGE.
+ */
+static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ p->add_inst(SMC_LI(src, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++) /* ~ tRAS - 2 */
+ p->add_inst(PUM_AllNops());
+
+ // Back-to-back: activate destination while source is still latched
+ p->add_inst(SMC_LI(dst, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * Core PuM primitive: triple-row activation to compute MAJ3, using a
+ * control value to select AND (control=0) or OR (control=1). The result
+ * lands in T0 (the first of the three activated rows).
+ *
+ * The ACT->PRE->ACT sequence uses reduced timings as described in the
+ * cs3_bitwise case study and the 2402.18736 paper (t1/t2 are the distances
+ * in fabric cycles between ACT and PRE, and PRE and the second ACT).
+ */
+static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ // ACT T0
+ p->add_inst(SMC_LI(PUM_ROW_T0, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < (int)t1; i++)
+ p->add_inst(PUM_AllNops());
+
+ // PRE (reduced tRP)
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+
+ // ACT T1
+ for (i = 0; i < (int)t2; i++)
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_LI(PUM_ROW_T1, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+
+ // Wait tRAS
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * A full AND/OR: source operands are copied into T0 (src1) and T1 (src2),
+ * control row C0/C1 provides the third operand (T2), a TRA computes the
+ * result, and finally the result is copied into `dst`.
+ */
+static void PUM_ComputeBitwise (Program *p, unsigned int src1,
+ unsigned int src2, unsigned int dst,
+ int is_or)
+{
+ // Copy operands into designated rows
+ PUM_RowClone (p, src1, PUM_ROW_T0);
+ PUM_RowClone (p, src2, PUM_ROW_T1);
+ PUM_RowClone (p, (is_or ? PUM_ROW_C1 : PUM_ROW_C0), PUM_ROW_T2);
+
+ // Triple-row activation, timing values chosen per the case study
+ PUM_TripleRowActivate (p, /*t1=*/1, /*t2=*/1);
+
+ // The result is now in T0; copy it to the destination row
+ PUM_RowClone (p, PUM_ROW_T0, dst);
+}
+
+/*
+ * Bitwise NOT via neighbouring-subarray activation. The shared sense
+ * amplifier produces the negated value on the destination's bitline.
+ */
+static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ p->add_inst(SMC_LI(src, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++) /* full tRAS */
+ p->add_inst(PUM_AllNops());
+
+ // PRE with reduced tRP, then ACT dst with reduced tRAS
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_LI(dst, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/* Read a full row back into `dst_buf` (row bytes) */
+static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_label("PUM:RDROW");
+ p->add_inst(SMC_READ(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_inst(PUM_AllNops());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:RDROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ UNUSED(dst_reg);
+}
+
+/* Execute a program and discard the instruction buffer */
+static void PUM_Execute (Program *p)
+{
+ if (pum_platform)
+ pum_platform->execute(*p);
+}
+
+// Host-side fallback implementations
+
+static void cpu_and (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] & b[i];
+}
+
+static void cpu_or (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] | b[i];
+}
+
+static void cpu_xor (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] ^ b[i];
+}
+
+static void cpu_not (const byte *a, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = ~a[i];
+}
+
+// Public PuM entry points
+
+/*
+ * Staging is intentionally simple and functionally correct: because we do
+ * not want to depend on a specific physical subarray mapping, we only run
+ * PuM when the whole request fits within a single row's worth of reserved
+ * rows and can be performed on the scratchpad. Larger requests are split
+ * into 8192-byte chunks by the driver; each chunk maps to the same reserved
+ * rows (T0/T1/T2/dst). Data must therefore be *read back* between chunks.
+ *
+ * To keep the proof-of-concept focused, the driver stages a single tile
+ * (8192 bytes) at a time and reads it back before moving to the next one,
+ * which is functionally correct even though it forfeits most of the raw
+ * throughput benefit.
+ */
+
+static void pum_stage_and_run (const byte *a, const byte *b, byte *dst,
+ int num_bytes, pum_op_t op)
+{
+ int offset = 0;
+
+ if (!pum_active || num_bytes <= 0)
+ {
+ // CPU fallback
+ }
+
+ while (offset < num_bytes && pum_active)
+ {
+ int chunk = num_bytes - offset;
+ if (chunk > PUM_ROW_BYTES)
+ chunk = PUM_ROW_BYTES;
+
+ Program prog;
+ // Write operand
+ PUM_StageRow (&prog, PUM_ROW_T0, a + offset, chunk);
+ if (op != PUM_OP_NOT)
+ PUM_StageRow (&prog, PUM_ROW_T1, b + offset, chunk);
+
+ if (op == PUM_OP_AND)
+ {
+ PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+ else if (op == PUM_OP_OR)
+ {
+ PUM_RowClone (&prog, PUM_ROW_C1, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+ else if (op == PUM_OP_XOR)
+ {
+ // XOR = (A & ~B) | (~A & B)
+ PUM_ComputeNot (&prog, PUM_ROW_T1, PUM_ROW_T3);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_T2);
+ PUM_RowClone (&prog, PUM_ROW_T3, PUM_ROW_T0);
+ PUM_RowClone (&prog, PUM_ROW_T2, PUM_ROW_T1);
+ PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1); // T0 & T1 -> T0
+ PUM_ReadRow (&prog, PUM_ROW_T0, 0);
+ }
+ else
+ {
+ PUM_ComputeNot (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+
+ PUM_Execute (&prog);
+
+ pum_platform->receiveData(dst + offset, chunk);
+
+ offset += chunk;
+ }
+
+ // fallback
+ if (offset < num_bytes || !pum_active)
+ {
+ switch (op)
+ {
+ case PUM_OP_AND: cpu_and(a, b, dst, num_bytes); break;
+ case PUM_OP_OR: cpu_or (a, b, dst, num_bytes); break;
+ case PUM_OP_XOR: cpu_xor(a, b, dst, num_bytes); break;
+ case PUM_OP_NOT: cpu_not(a, dst, num_bytes); break;
+ }
+ }
+}
+
+/*
+ * PUM_StageRow -- write `n` bytes (<= PUM_ROW_BYTES) into a DRAM row.
+ * The bytes that are not covered are zero-filled so the full row remains
+ * well-defined.
+ */
+static void PUM_StageRow (Program *p, unsigned int row, const byte *src,
+ int n)
+{
+ int i;
+ int words = (n + 3) / 4;
+ unsigned int tmp[16];
+
+ memset(tmp, 0, sizeof(tmp));
+ memcpy(tmp, src, n);
+
+ for (i = 0; i < 16; i++)
+ {
+ p->add_inst(SMC_LI(tmp[i], PUM_TEMP));
+ p->add_inst(SMC_LDWD(PUM_TEMP, i));
+ }
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+ p->add_label("PUM:STGROW");
+ p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:STGROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ UNUSED(words);
+}
+
+void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_AND);
+}
+
+void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_OR);
+}
+
+void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_XOR);
+}
+
+void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, dst, dst, num_bytes, PUM_OP_NOT);
+}
+
+/*
+ * Packed-light helpers: these are convenience wrappers mapping onto the
+ * generic bitwise path. In the full pipeline, lighting is computed by
+ * masking off the light grade and adding the texel index; here we expose
+ * the mask operation, which is simply a bitwise AND applied across the
+ * whole tile.
+ */
+void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask,
+ int num_bytes)
+{
+ // dst = src | (texel & mask)
+ pum_stage_and_run(src, mask, dst, num_bytes, PUM_OP_AND);
+}
+
+void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes)
+{
+ /* Initialise span state by clearing: dst = dst & clear (or, for an
+ all-zero clear, this is a memset). Kept as a PuM op for symmetry. */
+ pum_stage_and_run(dst, clear, dst, num_bytes, PUM_OP_AND);
+}
+
+#endif /* RENDER_PUM */
+
+#ifdef __GNUC__
+__attribute__((used))
+#endif
+static void pum_reference_primitives (void)
+{
+ Program scratch;
+ PUM_AddDDRCmd (&scratch, SMC_NOP(), 0);
+ PUM_ComputeBitwise (&scratch, PUM_ROW_T0, PUM_ROW_T1, PUM_ROW_DST0, 0);
+}