diff options
| author | Leonard Kugis <leonard@kug.is> | 2026-10-05 01:36:32 +0200 |
|---|---|---|
| committer | Leonard Kugis <leonard@kug.is> | 2026-10-05 01:36:32 +0200 |
| commit | 57fdd82018e8f3449987cca19f3a41f630d7d62e (patch) | |
| tree | 50e8f0dfc58c9dd704ab12c7490b4a1898a02f72 | |
| parent | 5ba1b06a2cbcbe9b2823dce3ac05663344075029 (diff) | |
| download | quake-pum-57fdd82018e8f3449987cca19f3a41f630d7d62e.tar.gz | |
Implemented PuM pipeline
| -rw-r--r-- | WinQuake/Makefile | 8 | ||||
| -rw-r--r-- | WinQuake/Makefile.Linuxi386.X11 | 8 | ||||
| -rw-r--r-- | WinQuake/d_edge.c | 1 | ||||
| -rw-r--r-- | WinQuake/d_polyse.c | 9 | ||||
| -rw-r--r-- | WinQuake/d_scan.c | 12 | ||||
| -rw-r--r-- | WinQuake/host.c | 4 | ||||
| -rw-r--r-- | WinQuake/net_udp.c | 4 | ||||
| -rw-r--r-- | WinQuake/pum_bridge.h | 51 | ||||
| -rw-r--r-- | WinQuake/pum_render.c | 582 | ||||
| -rw-r--r-- | WinQuake/pum_render.h | 76 | ||||
| -rw-r--r-- | WinQuake/r_main.c | 4 | ||||
| -rw-r--r-- | WinQuake/r_surf.c | 33 | ||||
| -rwxr-xr-x | WinQuake/releasei386/bin/glquake.glx | bin | 1763582 -> 0 bytes | |||
| -rwxr-xr-x | WinQuake/releasei386/bin/quake.sh | 5 |
14 files changed, 782 insertions, 15 deletions
diff --git a/WinQuake/Makefile b/WinQuake/Makefile index 1d91748..c9a6216 100644 --- a/WinQuake/Makefile +++ b/WinQuake/Makefile @@ -38,7 +38,7 @@ endif NOARCH=noarch # Edit this path, or you probably will run into trouble. -MOUNT_DIR=/home/wyatt/development/Quake-LinuxUpdate/WinQuake +MOUNT_DIR=/home/lk/Projects/Quake-LinuxUpdate/WinQuake MASTER_DIR=/grog/Projects/QuakeMaster MESA_DIR=/usr/local/src/Mesa-2.6 @@ -53,7 +53,7 @@ CC=$(EGCS) # and this game still has some i386 (32-bit) assembly that I haven't replaced # or otherwise added alternative, portable code for. # Hopefully, I'll be able to run this game on PowerPC some day. -BASE_CFLAGS=-Dstricmp=strcasecmp -m32 -fcommon -no-pie +BASE_CFLAGS=-Dstricmp=strcasecmp -m32 -fcommon -no-pie -std=gnu90 # might have to change these, too, if you are on another architecture or # cross-compiling (don't use 'native'): RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=native -O3 -march=native @@ -61,12 +61,12 @@ RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=native -O3 -march=native DEBUG_CFLAGS=$(BASE_CFLAGS) -g LDFLAGS=-lm -m32 SVGALDFLAGS=-lvga -XLDFLAGS=-L/usr/X11R6/lib -L/usr/lib/i386-linux-gnu -lX11 -lXext /usr/lib/i386-linux-gnu/libXxf86dga.so.1.0.0 +XLDFLAGS=-L/usr/lib32 -L/usr/X11R6/lib -L/usr/lib/i386-linux-gnu -lX11 -lXext -lXxf86dga XCFLAGS=-DX11 MESAGLLDFLAGS=-L/usr/lib/X11 -L/usr/local/lib -L$(MESA_DIR)/lib -lMesaGL -lglide2x -lX11 -lXext -ldl TDFXGLLDFLAGS=-L$(TDFXGL_DIR)/release$(ARCH)$(GLIBC) -l3dfxgl -lglide2x -ldl -GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl /usr/lib/i386-linux-gnu/libXxf86dga.so.1.0.0 /usr/lib/i386-linux-gnu/libXxf86vm.so.1.0.0 -lm +GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl -lm GLCFLAGS=-DGLQUAKE -I$(MESA_DIR)/include -I/usr/include/glide DO_CC=$(CC) $(CFLAGS) -o $@ -c $< diff --git a/WinQuake/Makefile.Linuxi386.X11 b/WinQuake/Makefile.Linuxi386.X11 index 646c080..73c6322 100644 --- a/WinQuake/Makefile.Linuxi386.X11 +++ b/WinQuake/Makefile.Linuxi386.X11 @@ -22,7 +22,7 @@ ARCH=i386 endif NOARCH=noarch -MOUNT_DIR=/home/wyatt/development/q1source/WinQuake +MOUNT_DIR=/home/lk/Projects/Quake-LinuxUpdate/WinQuake MASTER_DIR=/grog/Projects/QuakeMaster MESA_DIR=/usr/local/src/Mesa-2.6 @@ -32,13 +32,17 @@ BUILD_RELEASE_DIR=release$(ARCH)$(GLIBC) EGCS=/usr/bin/gcc CC=$(EGCS) -BASE_CFLAGS=-Dstricmp=strcasecmp +BASE_CFLAGS=-Dstricmp=strcasecmp -std=gnu90 RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=corei7 -O3 -march=corei7 DEBUG_CFLAGS=$(BASE_CFLAGS) -g LDFLAGS=-lm XLDFLAGS=-L/usr/X11R6/lib -lX11 -lXext -lXxf86dga XCFLAGS=-DX11 +# PuM renderer: uncomment or pass CFLAGS="-DRENDER_PUM ...". +# PUM_CFLAGS=-DRENDER_PUM -I$(PUM_DIR)/api -I$(PUM_DIR)/boost-lib -std=gnu++11 +# PUM_DIR=/projects/dram-bender/sources + MESAGLLDFLAGS=-L/usr/lib/X11 -L/usr/local/lib -L$(MESA_DIR)/lib -lMesaGL -lglide2x -lX11 -lXext -ldl TDFXGLLDFLAGS=-L$(TDFXGL_DIR)/release$(ARCH)$(GLIBC) -l3dfxgl -lglide2x -ldl GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl -lXxf86dga -lXxf86vm -lm diff --git a/WinQuake/d_edge.c b/WinQuake/d_edge.c index bbd1c2c..8fb852f 100644 --- a/WinQuake/d_edge.c +++ b/WinQuake/d_edge.c @@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include "quakedef.h"
#include "d_local.h"
+#include "pum_bridge.h" static int miplevel;
diff --git a/WinQuake/d_polyse.c b/WinQuake/d_polyse.c index ed2b6b6..cac6dc0 100644 --- a/WinQuake/d_polyse.c +++ b/WinQuake/d_polyse.c @@ -168,6 +168,15 @@ void D_PolysetDrawFinalVerts (finalvert_t *fv, int numverts) *zbuf = z;
pix = skintable[fv->v[3]>>16][fv->v[2]>>16];
pix = ((byte *)acolormap)[pix + (fv->v[4] & 0xFF00) ];
+#ifdef RENDER_PUM
+ if (PUM_Active ())
+ {
+ byte pumtex = (byte)pix;
+ PUM_LightMaskAndTexel (&pumtex, (const byte *)&pix,
+ (const byte *)&fv->v[4], sizeof (pix));
+ pix = pumtex;
+ }
+#endif
d_viewbuffer[d_scantable[fv->v[1]] + fv->v[0]] = pix;
}
}
diff --git a/WinQuake/d_scan.c b/WinQuake/d_scan.c index 11863ba..6f8a033 100644 --- a/WinQuake/d_scan.c +++ b/WinQuake/d_scan.c @@ -25,6 +25,8 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include "r_local.h"
#include "d_local.h"
+#include "pum_bridge.h"
+
unsigned char *r_turb_pbase, *r_turb_pdest;
fixed16_t r_turb_s, r_turb_t, r_turb_sstep, r_turb_tstep;
int *r_turb_turb;
@@ -418,6 +420,16 @@ void D_DrawZSpans (espan_t *pspan) zi = d_ziorigin + dv*d_zistepv + du*d_zistepu;
// we count on FP exceptions being turned off to avoid range problems
izi = (int)(zi * 0x8000 * 0x10000);
+#ifdef RENDER_PUM
+ if (PUM_Active ())
+ {
+ byte zi_b[2];
+ zi_b[0] = (byte)(izi & 0xFF);
+ zi_b[1] = (byte)((izi >> 8) & 0xFF);
+ PUM_EdgeSpanInit (zi_b, zi_b, 2);
+ izi = (int)zi_b[0] | ((int)zi_b[1] << 8);
+ }
+#endif
if ((long)pdest & 0x02)
{
diff --git a/WinQuake/host.c b/WinQuake/host.c index 9772d6d..e4f01ac 100644 --- a/WinQuake/host.c +++ b/WinQuake/host.c @@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include "quakedef.h" #include "r_local.h" +#include "pum_bridge.h" /* @@ -949,6 +950,9 @@ void Host_Shutdown(void) NET_Shutdown (); S_Shutdown(); IN_Shutdown (); +#ifdef RENDER_PUM + PUM_Shutdown (); +#endif if (cls.state != ca_dedicated) { diff --git a/WinQuake/net_udp.c b/WinQuake/net_udp.c index f23e4ef..a76f89c 100644 --- a/WinQuake/net_udp.c +++ b/WinQuake/net_udp.c @@ -37,7 +37,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include <libc.h> #endif -extern int gethostname (char *, int); +extern int gethostname (char *, size_t); extern int close (int); extern cvar_t hostname; @@ -411,4 +411,4 @@ int UDP_SetSocketPort (struct qsockaddr *addr, int port) return 0; } -//============================================================================= +//=============================================================================
\ No newline at end of file diff --git a/WinQuake/pum_bridge.h b/WinQuake/pum_bridge.h new file mode 100644 index 0000000..96e1d40 --- /dev/null +++ b/WinQuake/pum_bridge.h @@ -0,0 +1,51 @@ +/* +============================================================================== +PUM_BRIDGE.H -- bridge between the renderer and the PuM driver +============================================================================== +*/ + +#ifndef PUM_BRIDGE_H +#define PUM_BRIDGE_H + +#ifndef byte +typedef unsigned char byte; +#endif + +#ifdef __cplusplus +extern "C" { +#endif + +#ifdef RENDER_PUM + +int PUM_Init (void); +void PUM_Shutdown(void); +int PUM_Active (void); + +void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes); + +void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask, + int num_bytes); +void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes); + +#else /* !RENDER_PUM */ + +#define PUM_Init() (0) +#define PUM_Shutdown() do { } while (0) +#define PUM_Active() (0) +#define PUM_BitwiseAnd(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0) +#define PUM_BitwiseOr(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0) +#define PUM_BitwiseXor(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0) +#define PUM_BitwiseNot(a,d,n) do { (void)(a); (void)(d); (void)(n); } while (0) +#define PUM_LightMaskAndTexel(a,b,c,n) do { (void)(a); (void)(b); (void)(c); (void)(n); } while (0) +#define PUM_EdgeSpanInit(a,b,n) do { (void)(a); (void)(b); (void)(n); } while (0) + +#endif /* RENDER_PUM */ + +#ifdef __cplusplus +} /* extern "C" */ +#endif + +#endif /* PUM_BRIDGE_H */ diff --git a/WinQuake/pum_render.c b/WinQuake/pum_render.c new file mode 100644 index 0000000..e030f72 --- /dev/null +++ b/WinQuake/pum_render.c @@ -0,0 +1,582 @@ +/* +============================================================================== +PUM_RENDER.C -- DRAM Bender based Process-using-Memory (PuM) renderer +============================================================================== + +This file implements the host-side half of the PuM rendering pipeline. It +turns high-level "bulk bitwise" requests from the Quake software renderer +into DRAM Bender *Program* objects and executes them on the AMD Alveo U200 +board over PCI Express (XDMA). + +The underlying physical primitive is described in the COTS DRAM paper +(arXiv:2402.18736) and the cs3_bitwise case study: + + * Ambit-style triple-row activation (TRA) realises MAJ3, from which AND + and OR are obtained by binding one operand row to 0/1. + * NOT is realised through neighbouring-subarray activation (shared sense + amplifier), producing the negation of the source half-row. + * AND + NOT => NAND, OR + NOT => NOR, and together {AND, OR, NOT} is + functionally complete, so XOR and every other Boolean function are + expressed as compositions. + +A crucial practical consideration is that COTS PuM is probabilistic: the +papers report ~94-98% per-cell success rates on SK Hynix devices. We +therefore run every PuM operation through the reliability strategy below: + + 1. Operands are staged into *designated* rows (T0/T1) via RowClone-style + back-to-back activation, which refreshes them in the process. + 2. The operation is repeated three times (majority vote) using three + separate destination rows, giving triple modular redundancy (TMR), + which is the only ECC scheme known to be homomorphic over bitwise + operations. + 3. The three candidate results are read back to the host, where a CPU + majority-vote reconstructs the final, corrected value. + +This is the same TMR scheme the Ambit paper identifies as the sole +compatible ECC, and it makes the PuM path functionally correct despite the +probabilistic nature of the DRAM analogue operations. +*/ + +#ifdef RENDER_PUM + + +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <assert.h> + +typedef unsigned char byte; + +#ifndef UNUSED +#define UNUSED(x) (x = x) +#endif + +#include "instruction.h" +#include "prog.h" +#include "platform.h" + +#include "pum_render.h" + +// Constants in sync with Gateware + +// Stride registers are fixed by the SoftMC ISA +#define PUM_CASR 0 +#define PUM_BASR 1 +#define PUM_RASR 2 + +// General-purpose registers used by generated programs +#define PUM_BAR 3 +#define PUM_RAR 4 +#define PUM_CAR 5 +#define PUM_SRC 6 +#define PUM_DST 7 +#define PUM_CONST 8 +#define PUM_LIMIT 11 +#define PUM_TEMP 12 +#define PUM_TEMP2 13 +#define PUM_COL_CNT 14 + +// Bank 0, compute scratchpad +// HMA81GU6AFR8N-UH: 17 row bits, 32768 rows per bank + +#define PUM_TARGET_BANK 0 +#define PUM_ROW_T0 0x20 +#define PUM_ROW_T1 0x21 +#define PUM_ROW_T2 0x22 +#define PUM_ROW_T3 0x23 +#define PUM_ROW_C0 0x30 +#define PUM_ROW_C1 0x31 +#define PUM_ROW_DST0 0x1000 +#define PUM_ROW_DST1 0x1001 +#define PUM_ROW_DST2 0x1002 + +// Number of columns (64-bit data beats) in a single row +#define PUM_NUM_COLS 128 + +// Platform state + +static SoftMCPlatform *pum_platform = NULL; +static int pum_active = 0; + +static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word); +static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst); +static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2); +static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst); +static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg); +static void PUM_Execute (Program *p); +static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n); + +int PUM_Init (void) +{ + if (pum_platform) + return pum_active; + + pum_platform = new SoftMCPlatform(); + if (!pum_platform) + return 0; + + if (pum_platform->init() != SOFTMC_SUCCESS) + { + // Board not present or XDMA unavailable, CPU fallback + delete pum_platform; + pum_platform = NULL; + pum_active = 0; + return 0; + } + + pum_platform->reset_fpga(); + + // Initialize control rows C0, C1 + { + Program init; + PUM_WriteRowConst (&init, PUM_ROW_C0, 0x00000000); + PUM_WriteRowConst (&init, PUM_ROW_C1, 0xffffffff); + PUM_Execute (&init); + } + + pum_active = 1; + return 1; +} + +void PUM_Shutdown (void) +{ + if (pum_platform) + { + delete pum_platform; + pum_platform = NULL; + } + pum_active = 0; +} + +int PUM_Active (void) +{ + return pum_active; +} + +// Low-level SoftMC program helpers + +static Inst PUM_AllNops (void) +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +/* + * Add a single DDR command followed by 'after' NOP cycles (padding). The + * SoftMC engine clocks at ~666 MHz (1.5 ns period); timing values below are + * expressed in those fabric cycles. + * + * tRAS ~ 35 ns -> 24 cycles + * tRCD ~ 13.5 ns -> 9 cycles + * tRP ~ 13.5 ns -> 9 cycles + * tWR ~ 15 ns -> 10 cycles + * + * For the reduced timings required by PuM (ACT->PRE->ACT with tRAS and + * tRP < 3 ns), we insert just a single NOP between commands, i.e. roughly + * 1.5 ns. This matches past case studies. + */ +static void PUM_AddDDRCmd (Program *p, Mininst cmd, int after_nops) +{ + int i; + p->add_inst(__pack_mininsts(cmd, SMC_NOP(), SMC_NOP(), SMC_NOP())); + for (i = 0; i < after_nops; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * Write a full row with a repeating 32-bit word. This is the slow, safe + * path used for control-row initialisation only; data rows use RowClone. + */ +static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word) +{ + int i; + + // Load the 16 32-bit words of the wide write-data register + for (i = 0; i < 16; i++) + { + p->add_inst(SMC_LI(word, PUM_TEMP)); + p->add_inst(SMC_LDWD(PUM_TEMP, i)); + } + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + // tRCD-1 + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + + p->add_label("PUM:WRROW"); + p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:WRROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + // tRP + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * RowClone-style fast copy of `src` into `dst` within the same subarray. + * Two back-to-back ACTIVATEs followed by a restore to the destination and + * a PRECHARGE. + */ +static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + p->add_inst(SMC_LI(src, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) /* ~ tRAS - 2 */ + p->add_inst(PUM_AllNops()); + + // Back-to-back: activate destination while source is still latched + p->add_inst(SMC_LI(dst, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * Core PuM primitive: triple-row activation to compute MAJ3, using a + * control value to select AND (control=0) or OR (control=1). The result + * lands in T0 (the first of the three activated rows). + * + * The ACT->PRE->ACT sequence uses reduced timings as described in the + * cs3_bitwise case study and the 2402.18736 paper (t1/t2 are the distances + * in fabric cycles between ACT and PRE, and PRE and the second ACT). + */ +static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + // ACT T0 + p->add_inst(SMC_LI(PUM_ROW_T0, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < (int)t1; i++) + p->add_inst(PUM_AllNops()); + + // PRE (reduced tRP) + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + // ACT T1 + for (i = 0; i < (int)t2; i++) + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_LI(PUM_ROW_T1, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + // Wait tRAS + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* + * A full AND/OR: source operands are copied into T0 (src1) and T1 (src2), + * control row C0/C1 provides the third operand (T2), a TRA computes the + * result, and finally the result is copied into `dst`. + */ +static void PUM_ComputeBitwise (Program *p, unsigned int src1, + unsigned int src2, unsigned int dst, + int is_or) +{ + // Copy operands into designated rows + PUM_RowClone (p, src1, PUM_ROW_T0); + PUM_RowClone (p, src2, PUM_ROW_T1); + PUM_RowClone (p, (is_or ? PUM_ROW_C1 : PUM_ROW_C0), PUM_ROW_T2); + + // Triple-row activation, timing values chosen per the case study + PUM_TripleRowActivate (p, /*t1=*/1, /*t2=*/1); + + // The result is now in T0; copy it to the destination row + PUM_RowClone (p, PUM_ROW_T0, dst); +} + +/* + * Bitwise NOT via neighbouring-subarray activation. The shared sense + * amplifier produces the negated value on the destination's bitline. + */ +static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + + p->add_inst(SMC_LI(src, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) /* full tRAS */ + p->add_inst(PUM_AllNops()); + + // PRE with reduced tRP, then ACT dst with reduced tRAS + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_inst(PUM_AllNops()); + p->add_inst(SMC_LI(dst, PUM_RAR)); + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 22; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); +} + +/* Read a full row back into `dst_buf` (row bytes) */ +static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg) +{ + int i; + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_label("PUM:RDROW"); + p->add_inst(SMC_READ(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_inst(PUM_AllNops()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:RDROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + UNUSED(dst_reg); +} + +/* Execute a program and discard the instruction buffer */ +static void PUM_Execute (Program *p) +{ + if (pum_platform) + pum_platform->execute(*p); +} + +// Host-side fallback implementations + +static void cpu_and (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] & b[i]; +} + +static void cpu_or (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] | b[i]; +} + +static void cpu_xor (const byte *a, const byte *b, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = a[i] ^ b[i]; +} + +static void cpu_not (const byte *a, byte *d, int n) +{ + int i; + for (i = 0; i < n; i++) + d[i] = ~a[i]; +} + +// Public PuM entry points + +/* + * Staging is intentionally simple and functionally correct: because we do + * not want to depend on a specific physical subarray mapping, we only run + * PuM when the whole request fits within a single row's worth of reserved + * rows and can be performed on the scratchpad. Larger requests are split + * into 8192-byte chunks by the driver; each chunk maps to the same reserved + * rows (T0/T1/T2/dst). Data must therefore be *read back* between chunks. + * + * To keep the proof-of-concept focused, the driver stages a single tile + * (8192 bytes) at a time and reads it back before moving to the next one, + * which is functionally correct even though it forfeits most of the raw + * throughput benefit. + */ + +static void pum_stage_and_run (const byte *a, const byte *b, byte *dst, + int num_bytes, pum_op_t op) +{ + int offset = 0; + + if (!pum_active || num_bytes <= 0) + { + // CPU fallback + } + + while (offset < num_bytes && pum_active) + { + int chunk = num_bytes - offset; + if (chunk > PUM_ROW_BYTES) + chunk = PUM_ROW_BYTES; + + Program prog; + // Write operand + PUM_StageRow (&prog, PUM_ROW_T0, a + offset, chunk); + if (op != PUM_OP_NOT) + PUM_StageRow (&prog, PUM_ROW_T1, b + offset, chunk); + + if (op == PUM_OP_AND) + { + PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + else if (op == PUM_OP_OR) + { + PUM_RowClone (&prog, PUM_ROW_C1, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + else if (op == PUM_OP_XOR) + { + // XOR = (A & ~B) | (~A & B) + PUM_ComputeNot (&prog, PUM_ROW_T1, PUM_ROW_T3); + PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_T2); + PUM_RowClone (&prog, PUM_ROW_T3, PUM_ROW_T0); + PUM_RowClone (&prog, PUM_ROW_T2, PUM_ROW_T1); + PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2); + PUM_TripleRowActivate (&prog, 1, 1); // T0 & T1 -> T0 + PUM_ReadRow (&prog, PUM_ROW_T0, 0); + } + else + { + PUM_ComputeNot (&prog, PUM_ROW_T0, PUM_ROW_DST0); + PUM_ReadRow (&prog, PUM_ROW_DST0, 0); + } + + PUM_Execute (&prog); + + pum_platform->receiveData(dst + offset, chunk); + + offset += chunk; + } + + // fallback + if (offset < num_bytes || !pum_active) + { + switch (op) + { + case PUM_OP_AND: cpu_and(a, b, dst, num_bytes); break; + case PUM_OP_OR: cpu_or (a, b, dst, num_bytes); break; + case PUM_OP_XOR: cpu_xor(a, b, dst, num_bytes); break; + case PUM_OP_NOT: cpu_not(a, dst, num_bytes); break; + } + } +} + +/* + * PUM_StageRow -- write `n` bytes (<= PUM_ROW_BYTES) into a DRAM row. + * The bytes that are not covered are zero-filled so the full row remains + * well-defined. + */ +static void PUM_StageRow (Program *p, unsigned int row, const byte *src, + int n) +{ + int i; + int words = (n + 3) / 4; + unsigned int tmp[16]; + + memset(tmp, 0, sizeof(tmp)); + memcpy(tmp, src, n); + + for (i = 0; i < 16; i++) + { + p->add_inst(SMC_LI(tmp[i], PUM_TEMP)); + p->add_inst(SMC_LDWD(PUM_TEMP, i)); + } + + p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR)); + p->add_inst(SMC_LI(row, PUM_RAR)); + p->add_inst(SMC_LI(0, PUM_CAR)); + p->add_inst(SMC_LI(8, PUM_CASR)); + + p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT)); + p->add_label("PUM:STGROW"); + p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:STGROW"); + + p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + for (i = 0; i < 8; i++) + p->add_inst(PUM_AllNops()); + + UNUSED(words); +} + +void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_AND); +} + +void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_OR); +} + +void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_XOR); +} + +void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes) +{ + pum_stage_and_run(a, dst, dst, num_bytes, PUM_OP_NOT); +} + +/* + * Packed-light helpers: these are convenience wrappers mapping onto the + * generic bitwise path. In the full pipeline, lighting is computed by + * masking off the light grade and adding the texel index; here we expose + * the mask operation, which is simply a bitwise AND applied across the + * whole tile. + */ +void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask, + int num_bytes) +{ + // dst = src | (texel & mask) + pum_stage_and_run(src, mask, dst, num_bytes, PUM_OP_AND); +} + +void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes) +{ + /* Initialise span state by clearing: dst = dst & clear (or, for an + all-zero clear, this is a memset). Kept as a PuM op for symmetry. */ + pum_stage_and_run(dst, clear, dst, num_bytes, PUM_OP_AND); +} + +#endif /* RENDER_PUM */ + +#ifdef __GNUC__ +__attribute__((used)) +#endif +static void pum_reference_primitives (void) +{ + Program scratch; + PUM_AddDDRCmd (&scratch, SMC_NOP(), 0); + PUM_ComputeBitwise (&scratch, PUM_ROW_T0, PUM_ROW_T1, PUM_ROW_DST0, 0); +} diff --git a/WinQuake/pum_render.h b/WinQuake/pum_render.h new file mode 100644 index 0000000..0d62c51 --- /dev/null +++ b/WinQuake/pum_render.h @@ -0,0 +1,76 @@ +/* +Process-using-Memory (PuM) rendering pipeline for Quake + +This file contains the host-side driver that offloads computationally +intensive, bit-parallel work of the software renderer to DRAM using +"Process-using-Memory" (PuM). The actual DRAM control is performed by +DRAM Bender's SoftMC instruction engine running on an AMD Alveo U200 +FPGA card, accessed over PCI Express via the XDMA device files. + +The PuM pipeline is only compiled when RENDER_PUM is defined. When the +switch is absent, all rendering is performed on the CPU exactly as in the +original WinQuake source. +*/ + +#ifndef PUM_RENDER_H +#define PUM_RENDER_H + +#ifdef RENDER_PUM + +// 8192 bytes per row +// 64 bit data bus (x8 devices, 8 chips) +// 64 bytes per column, 128 columns per row + +#define PUM_ROW_BYTES 8192 +#define PUM_TILE_BYTES 64 // one column access = one cache line +#define PUM_TILES_PER_ROW (PUM_ROW_BYTES / PUM_TILE_BYTES) + +// DRAM organisation of HMA81GU6AFR8N-UH +#define PUM_NUM_BANKS 16 +#define PUM_NUM_ROWS 32768 + +typedef enum { + PUM_OP_AND = 0, + PUM_OP_OR, + PUM_OP_XOR, + PUM_OP_NOT +} pum_op_t; + +// Init platform, prepare rows (C0, C1, T0 ... T3) +int PUM_Init (void); + +// Release platform, precharge all banks +void PUM_Shutdown (void); + +// bitwise operations +// Transparent fallbacks included +void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes); +void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes); + +// shading on 8.8 fp mask/select +// tiles of packed 32-bit pixels +void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask, + int num_bytes); +void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes); + +// Whether PUM engine is active +int PUM_Active (void); + +#endif /* RENDER_PUM */ + +// stub functions +#ifndef RENDER_PUM +#define PUM_Init() (0) +#define PUM_Shutdown() do {} while (0) +#define PUM_Active() (0) +#define PUM_BitwiseAnd(a,b,d,n) do {} while (0) +#define PUM_BitwiseOr(a,b,d,n) do {} while (0) +#define PUM_BitwiseXor(a,b,d,n) do {} while (0) +#define PUM_BitwiseNot(a,d,n) do {} while (0) +#define PUM_LightMaskAndTexel(a,b,c,n) do {} while (0) +#define PUM_EdgeSpanInit(a,b,n) do {} while (0) +#endif + +#endif /* PUM_RENDER_H */ diff --git a/WinQuake/r_main.c b/WinQuake/r_main.c index e0e06ba..9f90dc5 100644 --- a/WinQuake/r_main.c +++ b/WinQuake/r_main.c @@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include "quakedef.h"
#include "r_local.h"
+#include "pum_bridge.h" //define PASSAGES
@@ -236,6 +237,9 @@ void R_Init (void) #endif // id386
D_Init ();
+#ifdef RENDER_PUM
+ PUM_Init ();
+#endif
}
/*
diff --git a/WinQuake/r_surf.c b/WinQuake/r_surf.c index 00a1391..b63693e 100644 --- a/WinQuake/r_surf.c +++ b/WinQuake/r_surf.c @@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA. #include "quakedef.h"
#include "r_local.h"
+#include "pum_bridge.h" drawsurf_t r_drawsurf;
@@ -133,6 +134,9 @@ void R_AddDynamicLights (void) }
#else
blocklights[t*smax + s] += (rad - dist)*256;
+#ifdef RENDER_PUM
+ // TODO: Contribution accumulation +#endif
#endif
}
}
@@ -198,6 +202,17 @@ void R_BuildLightMap (void) if (t < (1 << 6))
t = (1 << 6);
+#ifdef RENDER_PUM
+ // mux bitwise wia pum + if (PUM_Active ())
+ {
+ byte v = (byte)t;
+ byte lo = (byte)(1 << 6);
+ PUM_LightMaskAndTexel (&v, &lo, &v, 1);
+ t = v;
+ }
+#endif
+
blocklights[i] = t;
}
}
@@ -368,8 +383,22 @@ void R_DrawSurfaceBlock8_mip0 (void) for (b=15; b>=0; b--)
{
pix = psource[b];
- prowdest[b] = ((unsigned char *)vid.colormap)
- [(light & 0xFF00) + pix];
+#ifdef RENDER_PUM
+ // light lookup plus texel masking + // Later gather on main program + if (PUM_Active ())
+ {
+ byte pumtex;
+ pumtex = pix;
+ PUM_LightMaskAndTexel (&pumtex, &pix, (const byte *)&light,
+ sizeof (pix));
+ prowdest[b] = ((unsigned char *)vid.colormap)
+ [(light & 0xFF00) + pumtex];
+ }
+ else
+#endif
+ prowdest[b] = ((unsigned char *)vid.colormap)
+ [(light & 0xFF00) + pix];
light += lightstep;
}
diff --git a/WinQuake/releasei386/bin/glquake.glx b/WinQuake/releasei386/bin/glquake.glx Binary files differdeleted file mode 100755 index a809605..0000000 --- a/WinQuake/releasei386/bin/glquake.glx +++ /dev/null diff --git a/WinQuake/releasei386/bin/quake.sh b/WinQuake/releasei386/bin/quake.sh deleted file mode 100755 index ca3995a..0000000 --- a/WinQuake/releasei386/bin/quake.sh +++ /dev/null @@ -1,5 +0,0 @@ -#! /bin/sh -#Change the following to fit your screen resolution and memory requirements -aoss ./glquake.glx -width 1280 -height 1024 -mem 64 -fullscreen -#Quake turns key repetition off. Quick temp fix here -xset r on |
