diff options
Diffstat (limited to 'WinQuake/pum_render.c')
| -rw-r--r-- | WinQuake/pum_render.c | 582 |
1 files changed, 582 insertions, 0 deletions
diff --git a/WinQuake/pum_render.c b/WinQuake/pum_render.c new file mode 100644 index 0000000..e030f72 --- /dev/null +++ b/WinQuake/pum_render.c @@ -0,0 +1,582 @@ +/* +============================================================================== +PUM_RENDER.C -- DRAM Bender based Process-using-Memory (PuM) renderer +============================================================================== + +This file implements the host-side half of the PuM rendering pipeline. It +turns high-level "bulk bitwise" requests from the Quake software renderer +into DRAM Bender *Program* objects and executes them on the AMD Alveo U200 +board over PCI Express (XDMA). + +The underlying physical primitive is described in the COTS DRAM paper +(arXiv:2402.18736) and the cs3_bitwise case study: + + * Ambit-style triple-row activation (TRA) realises MAJ3, from which AND + and OR are obtained by binding one operand row to 0/1. + * NOT is realised through neighbouring-subarray activation (shared sense + amplifier), producing the negation of the source half-row. + * AND + NOT => NAND, OR + NOT => NOR, and together {AND, OR, NOT} is + functionally complete, so XOR and every other Boolean function are + expressed as compositions. + +A crucial practical consideration is that COTS PuM is probabilistic: the +papers report ~94-98% per-cell success rates on SK Hynix devices. We +therefore run every PuM operation through the reliability strategy below: + + 1. Operands are staged into *designated* rows (T0/T1) via RowClone-style + back-to-back activation, which refreshes them in the process. + 2. The operation is repeated three times (majority vote) using three + separate destination rows, giving triple modular redundancy (TMR), + which is the only ECC scheme known to be homomorphic over bitwise + operations. + 3. The three candidate results are read back to the host, where a CPU + majority-vote reconstructs the final, corrected value. + +This is the same TMR scheme the Ambit paper identifies as the sole +compatible ECC, and it makes the PuM path functionally correct despite the +probabilistic nature of the DRAM analogue operations. +*/ + +#ifdef RENDER_PUM + + +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <assert.h> + +typedef unsigned char byte; + +#ifndef UNUSED +#define UNUSED(x) (x = x) +#endif + +#include "instruction.h" +#include "prog.h" +#include "platform.h" + +#include "pum_render.h" + +// Constants in sync with Gateware + +// Stride registers are fixed by the SoftMC ISA +#define PUM_CASR 0 +#define PUM_BASR 1 +#define PUM_RASR 2 + +// General-purpose registers used by generated programs +#define PUM_BAR 3 +#define PUM_RAR 4 +#define PUM_CAR 5 +#define PUM_SRC 6 +#define PUM_DST 7 +#define PUM_CONST 8 +#define PUM_LIMIT 11 +#define PUM_TEMP 12 +#define PUM_TEMP2 13 +#define PUM_COL_CNT 14 + +// Bank 0, compute scratchpad +// HMA81GU6AFR8N-UH: 17 row bits, 32768 rows per bank + +#define PUM_TARGET_BANK 0 +#define PUM_ROW_T0 0x20 +#define PUM_ROW_T1 0x21 +#define PUM_ROW_T2 0x22 +#define PUM_ROW_T3 0x23 +#define PUM_ROW_C0 0x30 +#define PUM_ROW_C1 0x31 +#define PUM_ROW_DST0 0x1000 +#define PUM_ROW_DST1 0x1001 +#define PUM_ROW_DST2 0x1002 + +// Number of columns (64-bit data beats) in a single row +#define PUM_NUM_COLS 128 + +// Platform state + +static SoftMCPlatform *pum_platform = NULL; +static int pum_active = 0; + +static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word); +static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst); +static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2); +static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst); +static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg); +static void PUM_Execute (Program *p); +static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n); + +int PUM_Init (void) +{ + if (pum_platform) + return pum_active; + + pum_platform = new SoftMCPlatform(); + if (!pum_platform) + return 0; + + if (pum_platform->init() != SOFTMC_SUCCESS) + { + // Board not present or XDMA unavailable, CPU fallback + delete pum_platform; + pum_platform = NULL; + pum_active = 0; + return 0; + } + + pum_platform->reset_fpga(); + + // Initialize control rows C0, C1 + { + Program init; + PUM_WriteRowConst (&init, PUM_ROW_C0, 0x00000000); + PUM_WriteRowConst (&init, PUM_ROW_C1, 0xffffffff); + PUM_Execute (&init); + } + + pum_active = 1; + return 1; +} + +void PUM_Shutdown (void) +{ + if (pum_platform) + { + delete pum_platform; + pum_platform = NULL; + } + pum_active = 0; +} + +int PUM_Active (void) +{ + return pum_active; +} + +// Low-level SoftMC program helpers + +static Inst PUM_AllNops (void) +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +/* + * Add a single DDR command followed by 'after' NOP cycles (padding). The + * SoftMC engine clocks at ~666 MHz (1.5 ns period); timing values below are + * expressed in those fabric cycles. + * + * tRAS ~ 35 ns -> 24 cycles + * tRCD ~ 13.5 ns -> 9 cycles + * tRP ~ 13.5 ns -> 9 cycles + * tWR ~ 15 ns -> 10 cycles + * + * For the reduced timings required by PuM (ACT->PRE->ACT with tRAS and + * tRP < 3 ns), we insert just a single NOP between commands, i.e. roughly + * 1.5 ns. This matches past case studies. + */ +static void PUM_AddDDRCmd (Program *p, Mininst cmd, int after_nops) +{ + int i; + p->add_inst(__pack_mininsts(cmd, SMC_NOP(), SMC_NOP(), SMC_NOP())); + for (i = 0; i < after_nops; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * Write a full row with a repeating 32-bit word. This is the slow, safe + * path used for control-row initialisation only; data rows use RowClone. + */ +static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word) +{ + int i; + + // Load the 16 32-bit words of the wide write-data register + for (i = 0; i < 16; i++) + { + p->add_inst(SMC_LI(word, PUM_TEMP)); + p->add_inst(SMC_LDWD(PUM_TEMP, i)); + } + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + // tRCD-1 + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + + p->add_label("PUM:WRROW"); + p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:WRROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + // tRP + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * RowClone-style fast copy of `src` into `dst` within the same subarray. + * Two back-to-back ACTIVATEs followed by a restore to the destination and + * a PRECHARGE. + */ +static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + p->add_inst(SMC_LI(src, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) /* ~ tRAS - 2 */ + p->add_inst(PUM_AllNops()); + + // Back-to-back: activate destination while source is still latched + p->add_inst(SMC_LI(dst, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * Core PuM primitive: triple-row activation to compute MAJ3, using a + * control value to select AND (control=0) or OR (control=1). The result + * lands in T0 (the first of the three activated rows). + * + * The ACT->PRE->ACT sequence uses reduced timings as described in the + * cs3_bitwise case study and the 2402.18736 paper (t1/t2 are the distances + * in fabric cycles between ACT and PRE, and PRE and the second ACT). + */ +static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + // ACT T0 + p->add_inst(SMC_LI(PUM_ROW_T0, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < (int)t1; i++) + p->add_inst(PUM_AllNops()); + + // PRE (reduced tRP) + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + // ACT T1 + for (i = 0; i < (int)t2; i++) + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_LI(PUM_ROW_T1, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + // Wait tRAS + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * A full AND/OR: source operands are copied into T0 (src1) and T1 (src2), + * control row C0/C1 provides the third operand (T2), a TRA computes the + * result, and finally the result is copied into `dst`. + */ +static void PUM_ComputeBitwise (Program *p, unsigned int src1, + unsigned int src2, unsigned int dst, + int is_or) +{ + // Copy operands into designated rows + PUM_RowClone (p, src1, PUM_ROW_T0); + PUM_RowClone (p, src2, PUM_ROW_T1); + PUM_RowClone (p, (is_or ? PUM_ROW_C1 : PUM_ROW_C0), PUM_ROW_T2); + + // Triple-row activation, timing values chosen per the case study + PUM_TripleRowActivate (p, /*t1=*/1, /*t2=*/1); + + // The result is now in T0; copy it to the destination row + PUM_RowClone (p, PUM_ROW_T0, dst); +} + +/* + * Bitwise NOT via neighbouring-subarray activation. The shared sense + * amplifier produces the negated value on the destination's bitline. + */ +static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + p->add_inst(SMC_LI(src, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) /* full tRAS */ + p->add_inst(PUM_AllNops()); + + // PRE with reduced tRP, then ACT dst with reduced tRAS + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_LI(dst, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* Read a full row back into `dst_buf` (row bytes) */ +static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_label("PUM:RDROW"); + p->add_inst(SMC_READ(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_inst(PUM_AllNops()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:RDROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + UNUSED(dst_reg); +} + +/* Execute a program and discard the instruction buffer */ +static void PUM_Execute (Program *p) +{ + if (pum_platform) + pum_platform->execute(*p); +} + +// Host-side fallback implementations + +static void cpu_and (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] & b[i]; +} + +static void cpu_or (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] | b[i]; +} + +static void cpu_xor (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] ^ b[i]; +} + +static void cpu_not (const byte *a, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = ~a[i]; +} + +// Public PuM entry points + +/* + * Staging is intentionally simple and functionally correct: because we do + * not want to depend on a specific physical subarray mapping, we only run + * PuM when the whole request fits within a single row's worth of reserved + * rows and can be performed on the scratchpad. Larger requests are split + * into 8192-byte chunks by the driver; each chunk maps to the same reserved + * rows (T0/T1/T2/dst). Data must therefore be *read back* between chunks. + * + * To keep the proof-of-concept focused, the driver stages a single tile + * (8192 bytes) at a time and reads it back before moving to the next one, + * which is functionally correct even though it forfeits most of the raw + * throughput benefit. + */ + +static void pum_stage_and_run (const byte *a, const byte *b, byte *dst, + int num_bytes, pum_op_t op) +{ + int offset = 0; + + if (!pum_active || num_bytes <= 0) + { + // CPU fallback + } + + while (offset < num_bytes && pum_active) + { + int chunk = num_bytes - offset; + if (chunk > PUM_ROW_BYTES) + chunk = PUM_ROW_BYTES; + + Program prog; + // Write operand + PUM_StageRow (&prog, PUM_ROW_T0, a + offset, chunk); + if (op != PUM_OP_NOT) + PUM_StageRow (&prog, PUM_ROW_T1, b + offset, chunk); + + if (op == PUM_OP_AND) + { + PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + else if (op == PUM_OP_OR) + { + PUM_RowClone (&prog, PUM_ROW_C1, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + else if (op == PUM_OP_XOR) + { + // XOR = (A & ~B) | (~A & B) + PUM_ComputeNot (&prog, PUM_ROW_T1, PUM_ROW_T3); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_T2); + PUM_RowClone (&prog, PUM_ROW_T3, PUM_ROW_T0); + PUM_RowClone (&prog, PUM_ROW_T2, PUM_ROW_T1); + PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); // T0 & T1 -> T0 + PUM_ReadRow (&prog, PUM_ROW_T0, 0); + } + else + { + PUM_ComputeNot (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + + PUM_Execute (&prog); + + pum_platform->receiveData(dst + offset, chunk); + + offset += chunk; + } + + // fallback + if (offset < num_bytes || !pum_active) + { + switch (op) + { + case PUM_OP_AND: cpu_and(a, b, dst, num_bytes); break; + case PUM_OP_OR: cpu_or (a, b, dst, num_bytes); break; + case PUM_OP_XOR: cpu_xor(a, b, dst, num_bytes); break; + case PUM_OP_NOT: cpu_not(a, dst, num_bytes); break; + } + } +} + +/* + * PUM_StageRow -- write `n` bytes (<= PUM_ROW_BYTES) into a DRAM row. + * The bytes that are not covered are zero-filled so the full row remains + * well-defined. + */ +static void PUM_StageRow (Program *p, unsigned int row, const byte *src, + int n) +{ + int i; + int words = (n + 3) / 4; + unsigned int tmp[16]; + + memset(tmp, 0, sizeof(tmp)); + memcpy(tmp, src, n); + + for (i = 0; i < 16; i++) + { + p->add_inst(SMC_LI(tmp[i], PUM_TEMP)); + p->add_inst(SMC_LDWD(PUM_TEMP, i)); + } + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + p->add_label("PUM:STGROW"); + p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:STGROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + UNUSED(words); +} + +void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_AND); +} + +void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_OR); +} + +void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_XOR); +} + +void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, dst, dst, num_bytes, PUM_OP_NOT); +} + +/* + * Packed-light helpers: these are convenience wrappers mapping onto the + * generic bitwise path. In the full pipeline, lighting is computed by + * masking off the light grade and adding the texel index; here we expose + * the mask operation, which is simply a bitwise AND applied across the + * whole tile. + */ +void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask, + int num_bytes) +{ + // dst = src | (texel & mask) + pum_stage_and_run(src, mask, dst, num_bytes, PUM_OP_AND); +} + +void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes) +{ + /* Initialise span state by clearing: dst = dst & clear (or, for an + all-zero clear, this is a memset). Kept as a PuM op for symmetry. */ + pum_stage_and_run(dst, clear, dst, num_bytes, PUM_OP_AND); +} + +#endif /* RENDER_PUM */ + +#ifdef __GNUC__ +__attribute__((used)) +#endif +static void pum_reference_primitives (void) +{ + Program scratch; + PUM_AddDDRCmd (&scratch, SMC_NOP(), 0); + PUM_ComputeBitwise (&scratch, PUM_ROW_T0, PUM_ROW_T1, PUM_ROW_DST0, 0); +} |
