/* PUM_RENDER.C -- DRAM Bender based Process-using-Memory (PuM) renderer This file implements the host-side half of the PuM rendering pipeline. It turns high-level "bulk bitwise" requests from the Quake software renderer into DRAM Bender *Program* objects and executes them on the AMD Alveo U200 board over PCI Express (XDMA). The underlying physical primitive is described in the COTS DRAM paper (arXiv:2402.18736) and the cs3_bitwise case study: * Ambit-style triple-row activation (TRA) realises MAJ3, from which AND and OR are obtained by binding one operand row to 0/1. * NOT is realised through neighbouring-subarray activation (shared sense amplifier), producing the negation of the source half-row. * AND + NOT => NAND, OR + NOT => NOR, and together {AND, OR, NOT} is functionally complete, so XOR and every other Boolean function are expressed as compositions. A crucial practical consideration is that COTS PuM is probabilistic: the papers report ~94-98% per-cell success rates on SK Hynix devices. We therefore run every PuM operation through the reliability strategy below: 1. Operands are staged into *designated* rows (T0/T1) via RowClone-style back-to-back activation, which refreshes them in the process. 2. The operation is repeated three times (majority vote) using three separate destination rows, giving triple modular redundancy (TMR), which is the only ECC scheme known to be homomorphic over bitwise operations. 3. The three candidate results are read back to the host, where a CPU majority-vote reconstructs the final, corrected value. This is the same TMR scheme the Ambit paper identifies as the sole compatible ECC, and it makes the PuM path functionally correct despite the probabilistic nature of the DRAM analogue operations. */ #ifdef RENDER_PUM #include #include #include #include typedef unsigned char byte; #ifndef UNUSED #define UNUSED(x) (x = x) #endif #include "instruction.h" #include "prog.h" #include "platform.h" #include "pum_render.h" // Constants in sync with Gateware // Stride registers are fixed by the SoftMC ISA #define PUM_CASR 0 #define PUM_BASR 1 #define PUM_RASR 2 // General-purpose registers used by generated programs #define PUM_BAR 3 #define PUM_RAR 4 #define PUM_CAR 5 #define PUM_SRC 6 #define PUM_DST 7 #define PUM_CONST 8 #define PUM_LIMIT 11 #define PUM_TEMP 12 #define PUM_TEMP2 13 #define PUM_COL_CNT 14 // Bank 0, compute scratchpad // HMA81GU6AFR8N-UH: 17 row bits, 32768 rows per bank #define PUM_TARGET_BANK 0 #define PUM_ROW_T0 0x20 #define PUM_ROW_T1 0x21 #define PUM_ROW_T2 0x22 #define PUM_ROW_T3 0x23 #define PUM_ROW_C0 0x30 #define PUM_ROW_C1 0x31 #define PUM_ROW_DST0 0x1000 #define PUM_ROW_DST1 0x1001 #define PUM_ROW_DST2 0x1002 // Number of columns (64-bit data beats) in a single row #define PUM_NUM_COLS 128 // Platform state static SoftMCPlatform *pum_platform = NULL; static int pum_active = 0; static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word); static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst); static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2); static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst); static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg); static void PUM_Execute (Program *p); static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n); static void pum_compute_one_trial (Program *prog, pum_op_t op, unsigned int dstrow); static void pum_majority_vote (const byte *a, const byte *b, const byte *c, byte *out, int n); int PUM_Init (void) { if (pum_platform) return pum_active; pum_platform = new SoftMCPlatform(); if (!pum_platform) return 0; if (pum_platform->init() != SOFTMC_SUCCESS) { // Board not present or XDMA unavailable, CPU fallback delete pum_platform; pum_platform = NULL; pum_active = 0; return 0; } pum_platform->reset_fpga(); // Initialize control rows C0, C1 { Program init; PUM_WriteRowConst (&init, PUM_ROW_C0, 0x00000000); PUM_WriteRowConst (&init, PUM_ROW_C1, 0xffffffff); PUM_Execute (&init); } pum_active = 1; return 1; } void PUM_Shutdown (void) { if (pum_platform) { delete pum_platform; pum_platform = NULL; } pum_active = 0; } int PUM_Active (void) { return pum_active; } // Low-level SoftMC program helpers static Inst PUM_AllNops (void) { return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); } /* * Add a single DDR command followed by 'after' NOP cycles (padding). The * SoftMC engine clocks at ~666 MHz (1.5 ns period); timing values below are * expressed in those fabric cycles. * * tRAS ~ 35 ns -> 24 cycles * tRCD ~ 13.5 ns -> 9 cycles * tRP ~ 13.5 ns -> 9 cycles * tWR ~ 15 ns -> 10 cycles * * For the reduced timings required by PuM (ACT->PRE->ACT with tRAS and * tRP < 3 ns), we insert just a single NOP between commands, i.e. roughly * 1.5 ns. This matches past case studies. */ static void PUM_AddDDRCmd (Program *p, Mininst cmd, int after_nops) { int i; p->add_inst(__pack_mininsts(cmd, SMC_NOP(), SMC_NOP(), SMC_NOP())); for (i = 0; i < after_nops; i++) p->add_inst(PUM_AllNops()); } /* * Write a full row with a repeating 32-bit word. This is the slow, safe * path used for control-row initialisation only; data rows use RowClone. */ static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word) { int i; // Load the 16 32-bit words of the wide write-data register for (i = 0; i < 16; i++) { p->add_inst(SMC_LI(word, PUM_TEMP)); p->add_inst(SMC_LDWD(PUM_TEMP, i)); } p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); p->add_inst(SMC_LI(row, PUM_RAR)); p->add_inst(SMC_LI(0, PUM_CAR)); p->add_inst(SMC_LI(8, PUM_CASR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); // tRCD-1 for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); p->add_label("PUM:WRROW"); p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:WRROW"); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); // tRP for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); } /* * RowClone-style fast copy of `src` into `dst` within the same subarray. * Two back-to-back ACTIVATEs followed by a restore to the destination and * a PRECHARGE. */ static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst) { int i; p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); p->add_inst(SMC_LI(src, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 22; i++) /* ~ tRAS - 2 */ p->add_inst(PUM_AllNops()); // Back-to-back: activate destination while source is still latched p->add_inst(SMC_LI(dst, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 22; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); } /* * Core PuM primitive: triple-row activation to compute MAJ3, using a * control value to select AND (control=0) or OR (control=1). The result * lands in T0 (the first of the three activated rows). * * The ACT->PRE->ACT sequence uses reduced timings as described in the * cs3_bitwise case study and the 2402.18736 paper (t1/t2 are the distances * in fabric cycles between ACT and PRE, and PRE and the second ACT). */ static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2) { int i; p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); // ACT T0 p->add_inst(SMC_LI(PUM_ROW_T0, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < (int)t1; i++) p->add_inst(PUM_AllNops()); // PRE (reduced tRP) p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); // ACT T1 for (i = 0; i < (int)t2; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_LI(PUM_ROW_T1, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); // Wait tRAS for (i = 0; i < 22; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); } /* * A full AND/OR: source operands are copied into T0 (src1) and T1 (src2), * control row C0/C1 provides the third operand (T2), a TRA computes the * result, and finally the result is copied into `dst`. */ static void PUM_ComputeBitwise (Program *p, unsigned int src1, unsigned int src2, unsigned int dst, int is_or) { // Copy operands into designated rows PUM_RowClone (p, src1, PUM_ROW_T0); PUM_RowClone (p, src2, PUM_ROW_T1); PUM_RowClone (p, (is_or ? PUM_ROW_C1 : PUM_ROW_C0), PUM_ROW_T2); // Triple-row activation, timing values chosen per the case study PUM_TripleRowActivate (p, /*t1=*/1, /*t2=*/1); // The result is now in T0; copy it to the destination row PUM_RowClone (p, PUM_ROW_T0, dst); } /* * Bitwise NOT via neighbouring-subarray activation. The shared sense * amplifier produces the negated value on the destination's bitline. */ static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst) { int i; p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); p->add_inst(SMC_LI(src, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 22; i++) /* full tRAS */ p->add_inst(PUM_AllNops()); // PRE with reduced tRP, then ACT dst with reduced tRAS p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); p->add_inst(PUM_AllNops()); p->add_inst(SMC_LI(dst, PUM_RAR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 22; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); } /* Read a full row back into `dst_buf` (row bytes) */ static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg) { int i; p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); p->add_inst(SMC_LI(row, PUM_RAR)); p->add_inst(SMC_LI(0, PUM_CAR)); p->add_inst(SMC_LI(8, PUM_CASR)); p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); p->add_label("PUM:RDROW"); p->add_inst(SMC_READ(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); p->add_inst(PUM_AllNops()); p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:RDROW"); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); UNUSED(dst_reg); } /* Execute a program and discard the instruction buffer */ static void PUM_Execute (Program *p) { if (pum_platform) pum_platform->execute(*p); } // Host-side fallback implementations static void cpu_and (const byte *a, const byte *b, byte *d, int n) { int i; for (i = 0; i < n; i++) d[i] = a[i] & b[i]; } static void cpu_or (const byte *a, const byte *b, byte *d, int n) { int i; for (i = 0; i < n; i++) d[i] = a[i] | b[i]; } static void cpu_xor (const byte *a, const byte *b, byte *d, int n) { int i; for (i = 0; i < n; i++) d[i] = a[i] ^ b[i]; } static void cpu_not (const byte *a, byte *d, int n) { int i; for (i = 0; i < n; i++) d[i] = ~a[i]; } // Public PuM entry points /* * Staging is intentionally simple and functionally correct: because we do * not want to depend on a specific physical subarray mapping, we only run * PuM when the whole request fits within a single row's worth of reserved * rows and can be performed on the scratchpad. Larger requests are split * into 8192-byte chunks by the driver; each chunk maps to the same reserved * rows (T0/T1/T2/dst). Data must therefore be *read back* between chunks. * * To keep the proof-of-concept focused, the driver stages a single tile * (8192 bytes) at a time and reads it back before moving to the next one, * which is functionally correct even though it forfeits most of the raw * throughput benefit. */ /* * Compute a single PuM trial of `op` into the destination row `dstrow`. * * Operands are assumed to have already been staged in T0 (and T1 for binary * operations). This routine appends the command sequence and a read-back of * `dstrow` to `prog`. Because in-DRAM Boolean logic is probabilistic, every * logical output is computed three times (three independent trials into three * independent destination rows) and the caller majority-votes the results. */ static void pum_compute_one_trial (Program *prog, pum_op_t op, unsigned int dstrow) { if (op == PUM_OP_AND) { PUM_RowClone (prog, PUM_ROW_C0, PUM_ROW_T2); PUM_TripleRowActivate (prog, 1, 1); PUM_RowClone (prog, PUM_ROW_T0, dstrow); } else if (op == PUM_OP_OR) { PUM_RowClone (prog, PUM_ROW_C1, PUM_ROW_T2); PUM_TripleRowActivate (prog, 1, 1); PUM_RowClone (prog, PUM_ROW_T0, dstrow); } else if (op == PUM_OP_XOR) { // XOR = (A & ~B) | (~A & B) PUM_ComputeNot (prog, PUM_ROW_T1, PUM_ROW_T3); // T3 = ~B PUM_RowClone (prog, PUM_ROW_T0, PUM_ROW_T2); // T2 = A PUM_RowClone (prog, PUM_ROW_T3, PUM_ROW_T0); // T0 = ~B PUM_RowClone (prog, PUM_ROW_T2, PUM_ROW_T1); // T1 = A PUM_RowClone (prog, PUM_ROW_C0, PUM_ROW_T2); // T2 = 0 PUM_TripleRowActivate (prog, 1, 1); // T0 = ~B & A PUM_RowClone (prog, PUM_ROW_T0, dstrow); } else { PUM_ComputeNot (prog, PUM_ROW_T0, dstrow); } PUM_ReadRow (prog, dstrow, 0); } /* * Majority-vote three byte arrays: out = (a&b) | (a&c) | (b&c). */ static void pum_majority_vote (const byte *a, const byte *b, const byte *c, byte *out, int n) { int i; for (i = 0; i < n; i++) out[i] = (a[i] & b[i]) | (a[i] & c[i]) | (b[i] & c[i]); } static void pum_stage_and_run (const byte *a, const byte *b, byte *dst, int num_bytes, pum_op_t op) { int offset = 0; if (!pum_active || num_bytes <= 0) { // CPU fallback } while (offset < num_bytes && pum_active) { int chunk = num_bytes - offset; if (chunk > PUM_ROW_BYTES) chunk = PUM_ROW_BYTES; // Stage once, reuse on all trials Program prog; PUM_StageRow (&prog, PUM_ROW_T0, a + offset, chunk); if (op != PUM_OP_NOT) PUM_StageRow (&prog, PUM_ROW_T1, b + offset, chunk); // TMR pum_compute_one_trial (&prog, op, PUM_ROW_DST0); pum_compute_one_trial (&prog, op, PUM_ROW_DST1); pum_compute_one_trial (&prog, op, PUM_ROW_DST2); PUM_Execute (&prog); { byte r0[PUM_ROW_BYTES]; byte r1[PUM_ROW_BYTES]; byte r2[PUM_ROW_BYTES]; pum_platform->receiveData(r0, PUM_ROW_BYTES); pum_platform->receiveData(r1, PUM_ROW_BYTES); pum_platform->receiveData(r2, PUM_ROW_BYTES); pum_majority_vote (r0, r1, r2, dst + offset, chunk); } offset += chunk; } // CPU fallback if (offset < num_bytes || !pum_active) { switch (op) { case PUM_OP_AND: cpu_and(a, b, dst, num_bytes); break; case PUM_OP_OR: cpu_or (a, b, dst, num_bytes); break; case PUM_OP_XOR: cpu_xor(a, b, dst, num_bytes); break; case PUM_OP_NOT: cpu_not(a, dst, num_bytes); break; } } } /* * PUM_StageRow -- write `n` bytes (<= PUM_ROW_BYTES) into a DRAM row. * The bytes that are not covered are zero-filled so the full row remains * well-defined. */ static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n) { int i; int words = (n + 3) / 4; unsigned int tmp[16]; memset(tmp, 0, sizeof(tmp)); memcpy(tmp, src, n); for (i = 0; i < 16; i++) { p->add_inst(SMC_LI(tmp[i], PUM_TEMP)); p->add_inst(SMC_LDWD(PUM_TEMP, i)); } p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); p->add_inst(SMC_LI(row, PUM_RAR)); p->add_inst(SMC_LI(0, PUM_CAR)); p->add_inst(SMC_LI(8, PUM_CASR)); p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); p->add_label("PUM:STGROW"); p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:STGROW"); p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); for (i = 0; i < 8; i++) p->add_inst(PUM_AllNops()); UNUSED(words); } void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes) { pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_AND); } void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes) { pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_OR); } void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes) { pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_XOR); } void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes) { pum_stage_and_run(a, dst, dst, num_bytes, PUM_OP_NOT); } /* * Packed-light helpers: these are convenience wrappers mapping onto the * generic bitwise path. In the full pipeline, lighting is computed by * masking off the light grade and adding the texel index; here we expose * the mask operation, which is simply a bitwise AND applied across the * whole tile. */ void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask, int num_bytes) { // dst = src | (texel & mask) pum_stage_and_run(src, mask, dst, num_bytes, PUM_OP_AND); } void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes) { // Init span state pum_stage_and_run(dst, clear, dst, num_bytes, PUM_OP_AND); } #endif /* RENDER_PUM */ #ifdef __GNUC__ __attribute__((used)) #endif static void pum_reference_primitives (void) { Program scratch; PUM_AddDDRCmd (&scratch, SMC_NOP(), 0); PUM_ComputeBitwise (&scratch, PUM_ROW_T0, PUM_ROW_T1, PUM_ROW_DST0, 0); }