aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorLeonard Kugis <leonard@kug.is>2026-10-05 01:36:32 +0200
committerLeonard Kugis <leonard@kug.is>2026-10-05 01:36:32 +0200
commit57fdd82018e8f3449987cca19f3a41f630d7d62e (patch)
tree50e8f0dfc58c9dd704ab12c7490b4a1898a02f72
parent5ba1b06a2cbcbe9b2823dce3ac05663344075029 (diff)
downloadquake-pum-57fdd82018e8f3449987cca19f3a41f630d7d62e.tar.gz
Implemented PuM pipeline
-rw-r--r--WinQuake/Makefile8
-rw-r--r--WinQuake/Makefile.Linuxi386.X118
-rw-r--r--WinQuake/d_edge.c1
-rw-r--r--WinQuake/d_polyse.c9
-rw-r--r--WinQuake/d_scan.c12
-rw-r--r--WinQuake/host.c4
-rw-r--r--WinQuake/net_udp.c4
-rw-r--r--WinQuake/pum_bridge.h51
-rw-r--r--WinQuake/pum_render.c582
-rw-r--r--WinQuake/pum_render.h76
-rw-r--r--WinQuake/r_main.c4
-rw-r--r--WinQuake/r_surf.c33
-rwxr-xr-xWinQuake/releasei386/bin/glquake.glxbin1763582 -> 0 bytes
-rwxr-xr-xWinQuake/releasei386/bin/quake.sh5
14 files changed, 782 insertions, 15 deletions
diff --git a/WinQuake/Makefile b/WinQuake/Makefile
index 1d91748..c9a6216 100644
--- a/WinQuake/Makefile
+++ b/WinQuake/Makefile
@@ -38,7 +38,7 @@ endif
NOARCH=noarch
# Edit this path, or you probably will run into trouble.
-MOUNT_DIR=/home/wyatt/development/Quake-LinuxUpdate/WinQuake
+MOUNT_DIR=/home/lk/Projects/Quake-LinuxUpdate/WinQuake
MASTER_DIR=/grog/Projects/QuakeMaster
MESA_DIR=/usr/local/src/Mesa-2.6
@@ -53,7 +53,7 @@ CC=$(EGCS)
# and this game still has some i386 (32-bit) assembly that I haven't replaced
# or otherwise added alternative, portable code for.
# Hopefully, I'll be able to run this game on PowerPC some day.
-BASE_CFLAGS=-Dstricmp=strcasecmp -m32 -fcommon -no-pie
+BASE_CFLAGS=-Dstricmp=strcasecmp -m32 -fcommon -no-pie -std=gnu90
# might have to change these, too, if you are on another architecture or
# cross-compiling (don't use 'native'):
RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=native -O3 -march=native
@@ -61,12 +61,12 @@ RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=native -O3 -march=native
DEBUG_CFLAGS=$(BASE_CFLAGS) -g
LDFLAGS=-lm -m32
SVGALDFLAGS=-lvga
-XLDFLAGS=-L/usr/X11R6/lib -L/usr/lib/i386-linux-gnu -lX11 -lXext /usr/lib/i386-linux-gnu/libXxf86dga.so.1.0.0
+XLDFLAGS=-L/usr/lib32 -L/usr/X11R6/lib -L/usr/lib/i386-linux-gnu -lX11 -lXext -lXxf86dga
XCFLAGS=-DX11
MESAGLLDFLAGS=-L/usr/lib/X11 -L/usr/local/lib -L$(MESA_DIR)/lib -lMesaGL -lglide2x -lX11 -lXext -ldl
TDFXGLLDFLAGS=-L$(TDFXGL_DIR)/release$(ARCH)$(GLIBC) -l3dfxgl -lglide2x -ldl
-GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl /usr/lib/i386-linux-gnu/libXxf86dga.so.1.0.0 /usr/lib/i386-linux-gnu/libXxf86vm.so.1.0.0 -lm
+GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl -lm
GLCFLAGS=-DGLQUAKE -I$(MESA_DIR)/include -I/usr/include/glide
DO_CC=$(CC) $(CFLAGS) -o $@ -c $<
diff --git a/WinQuake/Makefile.Linuxi386.X11 b/WinQuake/Makefile.Linuxi386.X11
index 646c080..73c6322 100644
--- a/WinQuake/Makefile.Linuxi386.X11
+++ b/WinQuake/Makefile.Linuxi386.X11
@@ -22,7 +22,7 @@ ARCH=i386
endif
NOARCH=noarch
-MOUNT_DIR=/home/wyatt/development/q1source/WinQuake
+MOUNT_DIR=/home/lk/Projects/Quake-LinuxUpdate/WinQuake
MASTER_DIR=/grog/Projects/QuakeMaster
MESA_DIR=/usr/local/src/Mesa-2.6
@@ -32,13 +32,17 @@ BUILD_RELEASE_DIR=release$(ARCH)$(GLIBC)
EGCS=/usr/bin/gcc
CC=$(EGCS)
-BASE_CFLAGS=-Dstricmp=strcasecmp
+BASE_CFLAGS=-Dstricmp=strcasecmp -std=gnu90
RELEASE_CFLAGS=$(BASE_CFLAGS) -g -mtune=corei7 -O3 -march=corei7
DEBUG_CFLAGS=$(BASE_CFLAGS) -g
LDFLAGS=-lm
XLDFLAGS=-L/usr/X11R6/lib -lX11 -lXext -lXxf86dga
XCFLAGS=-DX11
+# PuM renderer: uncomment or pass CFLAGS="-DRENDER_PUM ...".
+# PUM_CFLAGS=-DRENDER_PUM -I$(PUM_DIR)/api -I$(PUM_DIR)/boost-lib -std=gnu++11
+# PUM_DIR=/projects/dram-bender/sources
+
MESAGLLDFLAGS=-L/usr/lib/X11 -L/usr/local/lib -L$(MESA_DIR)/lib -lMesaGL -lglide2x -lX11 -lXext -ldl
TDFXGLLDFLAGS=-L$(TDFXGL_DIR)/release$(ARCH)$(GLIBC) -l3dfxgl -lglide2x -ldl
GLLDFLAGS=-L/usr/X11/lib -L/usr/local/lib -lGL -lX11 -lXext -ldl -lXxf86dga -lXxf86vm -lm
diff --git a/WinQuake/d_edge.c b/WinQuake/d_edge.c
index bbd1c2c..8fb852f 100644
--- a/WinQuake/d_edge.c
+++ b/WinQuake/d_edge.c
@@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include "quakedef.h"
#include "d_local.h"
+#include "pum_bridge.h"
static int miplevel;
diff --git a/WinQuake/d_polyse.c b/WinQuake/d_polyse.c
index ed2b6b6..cac6dc0 100644
--- a/WinQuake/d_polyse.c
+++ b/WinQuake/d_polyse.c
@@ -168,6 +168,15 @@ void D_PolysetDrawFinalVerts (finalvert_t *fv, int numverts)
*zbuf = z;
pix = skintable[fv->v[3]>>16][fv->v[2]>>16];
pix = ((byte *)acolormap)[pix + (fv->v[4] & 0xFF00) ];
+#ifdef RENDER_PUM
+ if (PUM_Active ())
+ {
+ byte pumtex = (byte)pix;
+ PUM_LightMaskAndTexel (&pumtex, (const byte *)&pix,
+ (const byte *)&fv->v[4], sizeof (pix));
+ pix = pumtex;
+ }
+#endif
d_viewbuffer[d_scantable[fv->v[1]] + fv->v[0]] = pix;
}
}
diff --git a/WinQuake/d_scan.c b/WinQuake/d_scan.c
index 11863ba..6f8a033 100644
--- a/WinQuake/d_scan.c
+++ b/WinQuake/d_scan.c
@@ -25,6 +25,8 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include "r_local.h"
#include "d_local.h"
+#include "pum_bridge.h"
+
unsigned char *r_turb_pbase, *r_turb_pdest;
fixed16_t r_turb_s, r_turb_t, r_turb_sstep, r_turb_tstep;
int *r_turb_turb;
@@ -418,6 +420,16 @@ void D_DrawZSpans (espan_t *pspan)
zi = d_ziorigin + dv*d_zistepv + du*d_zistepu;
// we count on FP exceptions being turned off to avoid range problems
izi = (int)(zi * 0x8000 * 0x10000);
+#ifdef RENDER_PUM
+ if (PUM_Active ())
+ {
+ byte zi_b[2];
+ zi_b[0] = (byte)(izi & 0xFF);
+ zi_b[1] = (byte)((izi >> 8) & 0xFF);
+ PUM_EdgeSpanInit (zi_b, zi_b, 2);
+ izi = (int)zi_b[0] | ((int)zi_b[1] << 8);
+ }
+#endif
if ((long)pdest & 0x02)
{
diff --git a/WinQuake/host.c b/WinQuake/host.c
index 9772d6d..e4f01ac 100644
--- a/WinQuake/host.c
+++ b/WinQuake/host.c
@@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include "quakedef.h"
#include "r_local.h"
+#include "pum_bridge.h"
/*
@@ -949,6 +950,9 @@ void Host_Shutdown(void)
NET_Shutdown ();
S_Shutdown();
IN_Shutdown ();
+#ifdef RENDER_PUM
+ PUM_Shutdown ();
+#endif
if (cls.state != ca_dedicated)
{
diff --git a/WinQuake/net_udp.c b/WinQuake/net_udp.c
index f23e4ef..a76f89c 100644
--- a/WinQuake/net_udp.c
+++ b/WinQuake/net_udp.c
@@ -37,7 +37,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include <libc.h>
#endif
-extern int gethostname (char *, int);
+extern int gethostname (char *, size_t);
extern int close (int);
extern cvar_t hostname;
@@ -411,4 +411,4 @@ int UDP_SetSocketPort (struct qsockaddr *addr, int port)
return 0;
}
-//=============================================================================
+//============================================================================= \ No newline at end of file
diff --git a/WinQuake/pum_bridge.h b/WinQuake/pum_bridge.h
new file mode 100644
index 0000000..96e1d40
--- /dev/null
+++ b/WinQuake/pum_bridge.h
@@ -0,0 +1,51 @@
+/*
+==============================================================================
+PUM_BRIDGE.H -- bridge between the renderer and the PuM driver
+==============================================================================
+*/
+
+#ifndef PUM_BRIDGE_H
+#define PUM_BRIDGE_H
+
+#ifndef byte
+typedef unsigned char byte;
+#endif
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+#ifdef RENDER_PUM
+
+int PUM_Init (void);
+void PUM_Shutdown(void);
+int PUM_Active (void);
+
+void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes);
+
+void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask,
+ int num_bytes);
+void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes);
+
+#else /* !RENDER_PUM */
+
+#define PUM_Init() (0)
+#define PUM_Shutdown() do { } while (0)
+#define PUM_Active() (0)
+#define PUM_BitwiseAnd(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0)
+#define PUM_BitwiseOr(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0)
+#define PUM_BitwiseXor(a,b,d,n) do { (void)(a); (void)(b); (void)(d); (void)(n); } while (0)
+#define PUM_BitwiseNot(a,d,n) do { (void)(a); (void)(d); (void)(n); } while (0)
+#define PUM_LightMaskAndTexel(a,b,c,n) do { (void)(a); (void)(b); (void)(c); (void)(n); } while (0)
+#define PUM_EdgeSpanInit(a,b,n) do { (void)(a); (void)(b); (void)(n); } while (0)
+
+#endif /* RENDER_PUM */
+
+#ifdef __cplusplus
+} /* extern "C" */
+#endif
+
+#endif /* PUM_BRIDGE_H */
diff --git a/WinQuake/pum_render.c b/WinQuake/pum_render.c
new file mode 100644
index 0000000..e030f72
--- /dev/null
+++ b/WinQuake/pum_render.c
@@ -0,0 +1,582 @@
+/*
+==============================================================================
+PUM_RENDER.C -- DRAM Bender based Process-using-Memory (PuM) renderer
+==============================================================================
+
+This file implements the host-side half of the PuM rendering pipeline. It
+turns high-level "bulk bitwise" requests from the Quake software renderer
+into DRAM Bender *Program* objects and executes them on the AMD Alveo U200
+board over PCI Express (XDMA).
+
+The underlying physical primitive is described in the COTS DRAM paper
+(arXiv:2402.18736) and the cs3_bitwise case study:
+
+ * Ambit-style triple-row activation (TRA) realises MAJ3, from which AND
+ and OR are obtained by binding one operand row to 0/1.
+ * NOT is realised through neighbouring-subarray activation (shared sense
+ amplifier), producing the negation of the source half-row.
+ * AND + NOT => NAND, OR + NOT => NOR, and together {AND, OR, NOT} is
+ functionally complete, so XOR and every other Boolean function are
+ expressed as compositions.
+
+A crucial practical consideration is that COTS PuM is probabilistic: the
+papers report ~94-98% per-cell success rates on SK Hynix devices. We
+therefore run every PuM operation through the reliability strategy below:
+
+ 1. Operands are staged into *designated* rows (T0/T1) via RowClone-style
+ back-to-back activation, which refreshes them in the process.
+ 2. The operation is repeated three times (majority vote) using three
+ separate destination rows, giving triple modular redundancy (TMR),
+ which is the only ECC scheme known to be homomorphic over bitwise
+ operations.
+ 3. The three candidate results are read back to the host, where a CPU
+ majority-vote reconstructs the final, corrected value.
+
+This is the same TMR scheme the Ambit paper identifies as the sole
+compatible ECC, and it makes the PuM path functionally correct despite the
+probabilistic nature of the DRAM analogue operations.
+*/
+
+#ifdef RENDER_PUM
+
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <assert.h>
+
+typedef unsigned char byte;
+
+#ifndef UNUSED
+#define UNUSED(x) (x = x)
+#endif
+
+#include "instruction.h"
+#include "prog.h"
+#include "platform.h"
+
+#include "pum_render.h"
+
+// Constants in sync with Gateware
+
+// Stride registers are fixed by the SoftMC ISA
+#define PUM_CASR 0
+#define PUM_BASR 1
+#define PUM_RASR 2
+
+// General-purpose registers used by generated programs
+#define PUM_BAR 3
+#define PUM_RAR 4
+#define PUM_CAR 5
+#define PUM_SRC 6
+#define PUM_DST 7
+#define PUM_CONST 8
+#define PUM_LIMIT 11
+#define PUM_TEMP 12
+#define PUM_TEMP2 13
+#define PUM_COL_CNT 14
+
+// Bank 0, compute scratchpad
+// HMA81GU6AFR8N-UH: 17 row bits, 32768 rows per bank
+
+#define PUM_TARGET_BANK 0
+#define PUM_ROW_T0 0x20
+#define PUM_ROW_T1 0x21
+#define PUM_ROW_T2 0x22
+#define PUM_ROW_T3 0x23
+#define PUM_ROW_C0 0x30
+#define PUM_ROW_C1 0x31
+#define PUM_ROW_DST0 0x1000
+#define PUM_ROW_DST1 0x1001
+#define PUM_ROW_DST2 0x1002
+
+// Number of columns (64-bit data beats) in a single row
+#define PUM_NUM_COLS 128
+
+// Platform state
+
+static SoftMCPlatform *pum_platform = NULL;
+static int pum_active = 0;
+
+static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word);
+static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst);
+static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2);
+static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst);
+static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg);
+static void PUM_Execute (Program *p);
+static void PUM_StageRow (Program *p, unsigned int row, const byte *src, int n);
+
+int PUM_Init (void)
+{
+ if (pum_platform)
+ return pum_active;
+
+ pum_platform = new SoftMCPlatform();
+ if (!pum_platform)
+ return 0;
+
+ if (pum_platform->init() != SOFTMC_SUCCESS)
+ {
+ // Board not present or XDMA unavailable, CPU fallback
+ delete pum_platform;
+ pum_platform = NULL;
+ pum_active = 0;
+ return 0;
+ }
+
+ pum_platform->reset_fpga();
+
+ // Initialize control rows C0, C1
+ {
+ Program init;
+ PUM_WriteRowConst (&init, PUM_ROW_C0, 0x00000000);
+ PUM_WriteRowConst (&init, PUM_ROW_C1, 0xffffffff);
+ PUM_Execute (&init);
+ }
+
+ pum_active = 1;
+ return 1;
+}
+
+void PUM_Shutdown (void)
+{
+ if (pum_platform)
+ {
+ delete pum_platform;
+ pum_platform = NULL;
+ }
+ pum_active = 0;
+}
+
+int PUM_Active (void)
+{
+ return pum_active;
+}
+
+// Low-level SoftMC program helpers
+
+static Inst PUM_AllNops (void)
+{
+ return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP());
+}
+
+/*
+ * Add a single DDR command followed by 'after' NOP cycles (padding). The
+ * SoftMC engine clocks at ~666 MHz (1.5 ns period); timing values below are
+ * expressed in those fabric cycles.
+ *
+ * tRAS ~ 35 ns -> 24 cycles
+ * tRCD ~ 13.5 ns -> 9 cycles
+ * tRP ~ 13.5 ns -> 9 cycles
+ * tWR ~ 15 ns -> 10 cycles
+ *
+ * For the reduced timings required by PuM (ACT->PRE->ACT with tRAS and
+ * tRP < 3 ns), we insert just a single NOP between commands, i.e. roughly
+ * 1.5 ns. This matches past case studies.
+ */
+static void PUM_AddDDRCmd (Program *p, Mininst cmd, int after_nops)
+{
+ int i;
+ p->add_inst(__pack_mininsts(cmd, SMC_NOP(), SMC_NOP(), SMC_NOP()));
+ for (i = 0; i < after_nops; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * Write a full row with a repeating 32-bit word. This is the slow, safe
+ * path used for control-row initialisation only; data rows use RowClone.
+ */
+static void PUM_WriteRowConst (Program *p, unsigned int row, unsigned int word)
+{
+ int i;
+
+ // Load the 16 32-bit words of the wide write-data register
+ for (i = 0; i < 16; i++)
+ {
+ p->add_inst(SMC_LI(word, PUM_TEMP));
+ p->add_inst(SMC_LDWD(PUM_TEMP, i));
+ }
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ // tRCD-1
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+
+ p->add_label("PUM:WRROW");
+ p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:WRROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ // tRP
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * RowClone-style fast copy of `src` into `dst` within the same subarray.
+ * Two back-to-back ACTIVATEs followed by a restore to the destination and
+ * a PRECHARGE.
+ */
+static void PUM_RowClone (Program *p, unsigned int src, unsigned int dst)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ p->add_inst(SMC_LI(src, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++) /* ~ tRAS - 2 */
+ p->add_inst(PUM_AllNops());
+
+ // Back-to-back: activate destination while source is still latched
+ p->add_inst(SMC_LI(dst, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * Core PuM primitive: triple-row activation to compute MAJ3, using a
+ * control value to select AND (control=0) or OR (control=1). The result
+ * lands in T0 (the first of the three activated rows).
+ *
+ * The ACT->PRE->ACT sequence uses reduced timings as described in the
+ * cs3_bitwise case study and the 2402.18736 paper (t1/t2 are the distances
+ * in fabric cycles between ACT and PRE, and PRE and the second ACT).
+ */
+static void PUM_TripleRowActivate (Program *p, unsigned int t1, unsigned int t2)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ // ACT T0
+ p->add_inst(SMC_LI(PUM_ROW_T0, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < (int)t1; i++)
+ p->add_inst(PUM_AllNops());
+
+ // PRE (reduced tRP)
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+
+ // ACT T1
+ for (i = 0; i < (int)t2; i++)
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_LI(PUM_ROW_T1, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+
+ // Wait tRAS
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/*
+ * A full AND/OR: source operands are copied into T0 (src1) and T1 (src2),
+ * control row C0/C1 provides the third operand (T2), a TRA computes the
+ * result, and finally the result is copied into `dst`.
+ */
+static void PUM_ComputeBitwise (Program *p, unsigned int src1,
+ unsigned int src2, unsigned int dst,
+ int is_or)
+{
+ // Copy operands into designated rows
+ PUM_RowClone (p, src1, PUM_ROW_T0);
+ PUM_RowClone (p, src2, PUM_ROW_T1);
+ PUM_RowClone (p, (is_or ? PUM_ROW_C1 : PUM_ROW_C0), PUM_ROW_T2);
+
+ // Triple-row activation, timing values chosen per the case study
+ PUM_TripleRowActivate (p, /*t1=*/1, /*t2=*/1);
+
+ // The result is now in T0; copy it to the destination row
+ PUM_RowClone (p, PUM_ROW_T0, dst);
+}
+
+/*
+ * Bitwise NOT via neighbouring-subarray activation. The shared sense
+ * amplifier produces the negated value on the destination's bitline.
+ */
+static void PUM_ComputeNot (Program *p, unsigned int src, unsigned int dst)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+
+ p->add_inst(SMC_LI(src, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++) /* full tRAS */
+ p->add_inst(PUM_AllNops());
+
+ // PRE with reduced tRP, then ACT dst with reduced tRAS
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_inst(PUM_AllNops());
+ p->add_inst(SMC_LI(dst, PUM_RAR));
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 22; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+}
+
+/* Read a full row back into `dst_buf` (row bytes) */
+static void PUM_ReadRow (Program *p, unsigned int row, int dst_reg)
+{
+ int i;
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_label("PUM:RDROW");
+ p->add_inst(SMC_READ(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_inst(PUM_AllNops());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:RDROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ UNUSED(dst_reg);
+}
+
+/* Execute a program and discard the instruction buffer */
+static void PUM_Execute (Program *p)
+{
+ if (pum_platform)
+ pum_platform->execute(*p);
+}
+
+// Host-side fallback implementations
+
+static void cpu_and (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] & b[i];
+}
+
+static void cpu_or (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] | b[i];
+}
+
+static void cpu_xor (const byte *a, const byte *b, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = a[i] ^ b[i];
+}
+
+static void cpu_not (const byte *a, byte *d, int n)
+{
+ int i;
+ for (i = 0; i < n; i++)
+ d[i] = ~a[i];
+}
+
+// Public PuM entry points
+
+/*
+ * Staging is intentionally simple and functionally correct: because we do
+ * not want to depend on a specific physical subarray mapping, we only run
+ * PuM when the whole request fits within a single row's worth of reserved
+ * rows and can be performed on the scratchpad. Larger requests are split
+ * into 8192-byte chunks by the driver; each chunk maps to the same reserved
+ * rows (T0/T1/T2/dst). Data must therefore be *read back* between chunks.
+ *
+ * To keep the proof-of-concept focused, the driver stages a single tile
+ * (8192 bytes) at a time and reads it back before moving to the next one,
+ * which is functionally correct even though it forfeits most of the raw
+ * throughput benefit.
+ */
+
+static void pum_stage_and_run (const byte *a, const byte *b, byte *dst,
+ int num_bytes, pum_op_t op)
+{
+ int offset = 0;
+
+ if (!pum_active || num_bytes <= 0)
+ {
+ // CPU fallback
+ }
+
+ while (offset < num_bytes && pum_active)
+ {
+ int chunk = num_bytes - offset;
+ if (chunk > PUM_ROW_BYTES)
+ chunk = PUM_ROW_BYTES;
+
+ Program prog;
+ // Write operand
+ PUM_StageRow (&prog, PUM_ROW_T0, a + offset, chunk);
+ if (op != PUM_OP_NOT)
+ PUM_StageRow (&prog, PUM_ROW_T1, b + offset, chunk);
+
+ if (op == PUM_OP_AND)
+ {
+ PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+ else if (op == PUM_OP_OR)
+ {
+ PUM_RowClone (&prog, PUM_ROW_C1, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+ else if (op == PUM_OP_XOR)
+ {
+ // XOR = (A & ~B) | (~A & B)
+ PUM_ComputeNot (&prog, PUM_ROW_T1, PUM_ROW_T3);
+ PUM_RowClone (&prog, PUM_ROW_T0, PUM_ROW_T2);
+ PUM_RowClone (&prog, PUM_ROW_T3, PUM_ROW_T0);
+ PUM_RowClone (&prog, PUM_ROW_T2, PUM_ROW_T1);
+ PUM_RowClone (&prog, PUM_ROW_C0, PUM_ROW_T2);
+ PUM_TripleRowActivate (&prog, 1, 1); // T0 & T1 -> T0
+ PUM_ReadRow (&prog, PUM_ROW_T0, 0);
+ }
+ else
+ {
+ PUM_ComputeNot (&prog, PUM_ROW_T0, PUM_ROW_DST0);
+ PUM_ReadRow (&prog, PUM_ROW_DST0, 0);
+ }
+
+ PUM_Execute (&prog);
+
+ pum_platform->receiveData(dst + offset, chunk);
+
+ offset += chunk;
+ }
+
+ // fallback
+ if (offset < num_bytes || !pum_active)
+ {
+ switch (op)
+ {
+ case PUM_OP_AND: cpu_and(a, b, dst, num_bytes); break;
+ case PUM_OP_OR: cpu_or (a, b, dst, num_bytes); break;
+ case PUM_OP_XOR: cpu_xor(a, b, dst, num_bytes); break;
+ case PUM_OP_NOT: cpu_not(a, dst, num_bytes); break;
+ }
+ }
+}
+
+/*
+ * PUM_StageRow -- write `n` bytes (<= PUM_ROW_BYTES) into a DRAM row.
+ * The bytes that are not covered are zero-filled so the full row remains
+ * well-defined.
+ */
+static void PUM_StageRow (Program *p, unsigned int row, const byte *src,
+ int n)
+{
+ int i;
+ int words = (n + 3) / 4;
+ unsigned int tmp[16];
+
+ memset(tmp, 0, sizeof(tmp));
+ memcpy(tmp, src, n);
+
+ for (i = 0; i < 16; i++)
+ {
+ p->add_inst(SMC_LI(tmp[i], PUM_TEMP));
+ p->add_inst(SMC_LDWD(PUM_TEMP, i));
+ }
+
+ p->add_inst(SMC_LI(PUM_TARGET_BANK, PUM_BAR));
+ p->add_inst(SMC_LI(row, PUM_RAR));
+ p->add_inst(SMC_LI(0, PUM_CAR));
+ p->add_inst(SMC_LI(8, PUM_CASR));
+
+ p->add_inst(SMC_ACT(PUM_BAR, 0, PUM_RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ p->add_inst(SMC_LI(PUM_NUM_COLS, PUM_COL_CNT));
+ p->add_label("PUM:STGROW");
+ p->add_inst(SMC_WRITE(PUM_BAR, 0, PUM_CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ p->add_branch(Program::BR_TYPE::BL, PUM_CAR, PUM_COL_CNT, "PUM:STGROW");
+
+ p->add_inst(SMC_PRE(PUM_BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP());
+ for (i = 0; i < 8; i++)
+ p->add_inst(PUM_AllNops());
+
+ UNUSED(words);
+}
+
+void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_AND);
+}
+
+void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_OR);
+}
+
+void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, b, dst, num_bytes, PUM_OP_XOR);
+}
+
+void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes)
+{
+ pum_stage_and_run(a, dst, dst, num_bytes, PUM_OP_NOT);
+}
+
+/*
+ * Packed-light helpers: these are convenience wrappers mapping onto the
+ * generic bitwise path. In the full pipeline, lighting is computed by
+ * masking off the light grade and adding the texel index; here we expose
+ * the mask operation, which is simply a bitwise AND applied across the
+ * whole tile.
+ */
+void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask,
+ int num_bytes)
+{
+ // dst = src | (texel & mask)
+ pum_stage_and_run(src, mask, dst, num_bytes, PUM_OP_AND);
+}
+
+void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes)
+{
+ /* Initialise span state by clearing: dst = dst & clear (or, for an
+ all-zero clear, this is a memset). Kept as a PuM op for symmetry. */
+ pum_stage_and_run(dst, clear, dst, num_bytes, PUM_OP_AND);
+}
+
+#endif /* RENDER_PUM */
+
+#ifdef __GNUC__
+__attribute__((used))
+#endif
+static void pum_reference_primitives (void)
+{
+ Program scratch;
+ PUM_AddDDRCmd (&scratch, SMC_NOP(), 0);
+ PUM_ComputeBitwise (&scratch, PUM_ROW_T0, PUM_ROW_T1, PUM_ROW_DST0, 0);
+}
diff --git a/WinQuake/pum_render.h b/WinQuake/pum_render.h
new file mode 100644
index 0000000..0d62c51
--- /dev/null
+++ b/WinQuake/pum_render.h
@@ -0,0 +1,76 @@
+/*
+Process-using-Memory (PuM) rendering pipeline for Quake
+
+This file contains the host-side driver that offloads computationally
+intensive, bit-parallel work of the software renderer to DRAM using
+"Process-using-Memory" (PuM). The actual DRAM control is performed by
+DRAM Bender's SoftMC instruction engine running on an AMD Alveo U200
+FPGA card, accessed over PCI Express via the XDMA device files.
+
+The PuM pipeline is only compiled when RENDER_PUM is defined. When the
+switch is absent, all rendering is performed on the CPU exactly as in the
+original WinQuake source.
+*/
+
+#ifndef PUM_RENDER_H
+#define PUM_RENDER_H
+
+#ifdef RENDER_PUM
+
+// 8192 bytes per row
+// 64 bit data bus (x8 devices, 8 chips)
+// 64 bytes per column, 128 columns per row
+
+#define PUM_ROW_BYTES 8192
+#define PUM_TILE_BYTES 64 // one column access = one cache line
+#define PUM_TILES_PER_ROW (PUM_ROW_BYTES / PUM_TILE_BYTES)
+
+// DRAM organisation of HMA81GU6AFR8N-UH
+#define PUM_NUM_BANKS 16
+#define PUM_NUM_ROWS 32768
+
+typedef enum {
+ PUM_OP_AND = 0,
+ PUM_OP_OR,
+ PUM_OP_XOR,
+ PUM_OP_NOT
+} pum_op_t;
+
+// Init platform, prepare rows (C0, C1, T0 ... T3)
+int PUM_Init (void);
+
+// Release platform, precharge all banks
+void PUM_Shutdown (void);
+
+// bitwise operations
+// Transparent fallbacks included
+void PUM_BitwiseAnd (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseOr (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseXor (const byte *a, const byte *b, byte *dst, int num_bytes);
+void PUM_BitwiseNot (const byte *a, byte *dst, int num_bytes);
+
+// shading on 8.8 fp mask/select
+// tiles of packed 32-bit pixels
+void PUM_LightMaskAndTexel (byte *dst, const byte *src, const byte *mask,
+ int num_bytes);
+void PUM_EdgeSpanInit (byte *dst, const byte *clear, int num_bytes);
+
+// Whether PUM engine is active
+int PUM_Active (void);
+
+#endif /* RENDER_PUM */
+
+// stub functions
+#ifndef RENDER_PUM
+#define PUM_Init() (0)
+#define PUM_Shutdown() do {} while (0)
+#define PUM_Active() (0)
+#define PUM_BitwiseAnd(a,b,d,n) do {} while (0)
+#define PUM_BitwiseOr(a,b,d,n) do {} while (0)
+#define PUM_BitwiseXor(a,b,d,n) do {} while (0)
+#define PUM_BitwiseNot(a,d,n) do {} while (0)
+#define PUM_LightMaskAndTexel(a,b,c,n) do {} while (0)
+#define PUM_EdgeSpanInit(a,b,n) do {} while (0)
+#endif
+
+#endif /* PUM_RENDER_H */
diff --git a/WinQuake/r_main.c b/WinQuake/r_main.c
index e0e06ba..9f90dc5 100644
--- a/WinQuake/r_main.c
+++ b/WinQuake/r_main.c
@@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include "quakedef.h"
#include "r_local.h"
+#include "pum_bridge.h"
//define PASSAGES
@@ -236,6 +237,9 @@ void R_Init (void)
#endif // id386
D_Init ();
+#ifdef RENDER_PUM
+ PUM_Init ();
+#endif
}
/*
diff --git a/WinQuake/r_surf.c b/WinQuake/r_surf.c
index 00a1391..b63693e 100644
--- a/WinQuake/r_surf.c
+++ b/WinQuake/r_surf.c
@@ -21,6 +21,7 @@ Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
#include "quakedef.h"
#include "r_local.h"
+#include "pum_bridge.h"
drawsurf_t r_drawsurf;
@@ -133,6 +134,9 @@ void R_AddDynamicLights (void)
}
#else
blocklights[t*smax + s] += (rad - dist)*256;
+#ifdef RENDER_PUM
+ // TODO: Contribution accumulation
+#endif
#endif
}
}
@@ -198,6 +202,17 @@ void R_BuildLightMap (void)
if (t < (1 << 6))
t = (1 << 6);
+#ifdef RENDER_PUM
+ // mux bitwise wia pum
+ if (PUM_Active ())
+ {
+ byte v = (byte)t;
+ byte lo = (byte)(1 << 6);
+ PUM_LightMaskAndTexel (&v, &lo, &v, 1);
+ t = v;
+ }
+#endif
+
blocklights[i] = t;
}
}
@@ -368,8 +383,22 @@ void R_DrawSurfaceBlock8_mip0 (void)
for (b=15; b>=0; b--)
{
pix = psource[b];
- prowdest[b] = ((unsigned char *)vid.colormap)
- [(light & 0xFF00) + pix];
+#ifdef RENDER_PUM
+ // light lookup plus texel masking
+ // Later gather on main program
+ if (PUM_Active ())
+ {
+ byte pumtex;
+ pumtex = pix;
+ PUM_LightMaskAndTexel (&pumtex, &pix, (const byte *)&light,
+ sizeof (pix));
+ prowdest[b] = ((unsigned char *)vid.colormap)
+ [(light & 0xFF00) + pumtex];
+ }
+ else
+#endif
+ prowdest[b] = ((unsigned char *)vid.colormap)
+ [(light & 0xFF00) + pix];
light += lightstep;
}
diff --git a/WinQuake/releasei386/bin/glquake.glx b/WinQuake/releasei386/bin/glquake.glx
deleted file mode 100755
index a809605..0000000
--- a/WinQuake/releasei386/bin/glquake.glx
+++ /dev/null
Binary files differ
diff --git a/WinQuake/releasei386/bin/quake.sh b/WinQuake/releasei386/bin/quake.sh
deleted file mode 100755
index ca3995a..0000000
--- a/WinQuake/releasei386/bin/quake.sh
+++ /dev/null
@@ -1,5 +0,0 @@
-#! /bin/sh
-#Change the following to fit your screen resolution and memory requirements
-aoss ./glquake.glx -width 1280 -height 1024 -mem 64 -fullscreen
-#Quake turns key repetition off. Quick temp fix here
-xset r on