diff options
| author | Ataberk <olgunataberk@gmail.com> | 2022-09-25 17:22:03 +0200 |
|---|---|---|
| committer | Ataberk <olgunataberk@gmail.com> | 2022-09-25 17:22:03 +0200 |
| commit | dc0b3db1b4f1895a07e5fe280ee3790e87f97b9f (patch) | |
| tree | b47203aa281bdd959def4451c84d310cd9cf2e12 /sources | |
| download | dram-bender-dc0b3db1b4f1895a07e5fe280ee3790e87f97b9f.tar.gz | |
Initial commit
Diffstat (limited to 'sources')
110 files changed, 24515 insertions, 0 deletions
diff --git a/sources/api/.gitignore b/sources/api/.gitignore new file mode 100644 index 0000000..be60ba4 --- /dev/null +++ b/sources/api/.gitignore @@ -0,0 +1,2 @@ +*.o +.vscode/
\ No newline at end of file diff --git a/sources/api/board.cpp b/sources/api/board.cpp new file mode 100644 index 0000000..db990ca --- /dev/null +++ b/sources/api/board.cpp @@ -0,0 +1,132 @@ +#include "board.h" +#include <unistd.h> +#include <fstream> +#include <iostream> +#include <cassert> +#include <fcntl.h> +#include <string.h> + +BoardInterface::BoardInterface(IFACE iface_type) +{ + this -> iface_type = iface_type; +} +BoardInterface::~BoardInterface() +{ + free(send_buf); + free(recv_buf); + close(to_card); + close(from_card); +} + +int BoardInterface::init() +{ + switch(iface_type) + { + case IFACE::XDMA: + { + int fpga_fd = open(TO_FPGA_DEFAULT.c_str(), O_RDWR); + if(fpga_fd<0) + { + std::cerr << "Open to card failed!" << std::endl; + return 1; + } + else + std::cout << "Opened " << TO_FPGA_DEFAULT << " -> " << fpga_fd << std::endl; + to_card = fpga_fd; + fpga_fd = open(FROM_FPGA_DEFAULT.c_str(), O_RDWR); + if(fpga_fd<0) + { + std::cerr << "Open to host failed!" << std::endl; + return 1; + } + else + std::cout << "Opened " << FROM_FPGA_DEFAULT << " -> " << fpga_fd << std::endl; + from_card = fpga_fd; + // allocate page size aligned X page size regions to our buffers + if (posix_memalign((void **)&send_buf, 4096 /*alignment */ , SEND_BUF_SIZE + (4096-(SEND_BUF_SIZE % 4096))) != 0) + { + std::cerr << "Send buffer allocation failed!" << std::endl; + return 1; + } + if (posix_memalign((void **)&recv_buf, 4096 /*alignment */ , RECV_BUF_SIZE + (4096-(RECV_BUF_SIZE % 4096)))) + { + std::cerr << "Receive buffer allocation failed!" << std::endl; + return 1; + } + if( (!send_buf) || (!recv_buf) ) + { + std::cerr << "Buffers cannot be allocated!" << std::endl; + return 1; + } + return 0; + } + default: + std::cerr << "Unknown iface_type!" << std::endl; + return 1; + } +} + +int BoardInterface::sendData(void* data, const uint size) +{ + switch(iface_type) + { + case IFACE::XDMA: + return xdma_send(data,size); + break; + default: + std::cerr << "Unknown iface_type!" << std::endl; + return 1; + } +} + +int BoardInterface::recvData(void* buf, const uint size) +{ + switch(iface_type) + { + case IFACE::XDMA: + return xdma_recv(buf,size); + break; + default: + std::cerr << "Unknown iface_type!" << std::endl; + return 1; + } +} + +int BoardInterface::xdma_send(void* data, const uint size) +{ + memcpy((char*)send_buf, (char*)data, size); + + int fd = to_card; + ssize_t rc; + uint64_t count = 0; + char *buf = (char*) send_buf; + + while (count < size) { + /* write data to file from memory buffer */ + rc = write(fd, buf, size); + assert(rc == size || rc == 0); + count += rc; + } + + // We wrote more than we supposed to + if (count != size) + return 1; + + return 0; +} + +int BoardInterface::xdma_recv(void* buf, const uint size) +{ + assert(size <= RECV_BUF_SIZE && "given read size is too large"); + + int fd = from_card; + uint64_t count = 0; + // Try to read from the card. + count = read(fd, (char*) recv_buf, size); + + // We read more than we were supposed to + assert(count <= size);// || rc == 0); + + memcpy(buf, (char*) recv_buf, count); + return count; +} diff --git a/sources/api/board.h b/sources/api/board.h new file mode 100644 index 0000000..0935392 --- /dev/null +++ b/sources/api/board.h @@ -0,0 +1,35 @@ +#include <string> + +#ifndef BOARD_H +#define BOARD_H +/** This class defines how the host + * interfaces with the board. + */ +class BoardInterface{ + const uint SEND_BUF_SIZE = 32*2048; + // send each instruction as a 256-bit packet + const uint RECV_BUF_SIZE = 1024*32; + const std::string TO_FPGA_DEFAULT = "/dev/xdma0_h2c_0"; + const std::string FROM_FPGA_DEFAULT = "/dev/xdma0_c2h_0"; +public: + enum class IFACE { + XDMA = 0 + }; + BoardInterface(IFACE); + ~BoardInterface(); + int init(); + int sendData(void* data, const uint size); + int recvData(void* buf , const uint size); +private: + IFACE iface_type; + // XDMA related constructs + int to_card; + int from_card; + void* send_buf; + void* recv_buf; + int xdma_send(void* data, const uint size); + int xdma_recv(void* buf, const uint size); + // end XDMA related constructs +}; + +#endif diff --git a/sources/api/instruction.cpp b/sources/api/instruction.cpp new file mode 100644 index 0000000..432188f --- /dev/null +++ b/sources/api/instruction.cpp @@ -0,0 +1,586 @@ +#include "instruction.h" +#include <assert.h> +#include <stdio.h> + +Inst SMC_ADD(int rs1, int rs2, int rt) +{ + Inst fu_code = (uint64_t)__ADD << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_regs | t_reg; + + return inst; +} +Inst SMC_ADDI(int rs1, uint32_t imd, int rt) +{ + Inst fu_code = (uint64_t)__ADDI << __FU_CODE; + Inst s_reg = rs1; + Inst imd1 = imd << __IMD1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_reg | imd1 | t_reg; + + return inst; +} +Inst SMC_SUB(int rs1, int rs2, int rt) +{ + Inst fu_code = (uint64_t)__SUB << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_regs | t_reg; + + return inst; +} +Inst SMC_SUBI(int rs1, uint32_t imd, int rt) +{ + Inst fu_code = (uint64_t)__SUBI << __FU_CODE; + Inst s_reg = rs1; + Inst imd1 = imd << __IMD1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_reg | imd1 | t_reg; + + return inst; +} +Inst SMC_LI(uint32_t imd, int rt) +{ + Inst fu_code = (uint64_t)__LI << __FU_CODE; + Inst imd1 = ((uint32_t)(imd<<16)>>16) << __IMD1; + Inst imd2 = (uint64_t)(((uint32_t)imd)>>16) << __IMD3; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | imd1 | imd2 | t_reg; + + return inst; +} +Inst SMC_MV(int rs1, int rt) +{ + Inst fu_code = (uint64_t)__MV << __FU_CODE; + Inst s_reg = rs1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_reg | t_reg; + + return inst; +} +Inst SMC_SRC(int rs1, int rt) +{ + Inst fu_code = (uint64_t)__SRC << __FU_CODE; + Inst s_reg = rs1; + Inst t_reg = rt << __RT; + + Inst inst = fu_code | s_reg | t_reg; + + return inst; +} +Inst SMC_LDWD(int rs1, int off) +{ + Inst fu_code = (uint64_t)__LDWD << __FU_CODE; + Inst s_reg = rs1; + Inst offset = off << __RT; + + Inst inst = fu_code | s_reg | offset; + + return inst; +} +Inst SMC_LDPC(PC_TYPE pc_type, int rt) +{ + Inst fu_code = (uint64_t)__LDPC << __FU_CODE; + Inst pc_reg = 2; + Inst t_reg = rt << __RT; + switch(pc_type){ + case PC_TYPE::WRITE: + pc_reg = 0; + break; + case PC_TYPE::READ: + pc_reg = 1; + break; + case PC_TYPE::PRE: + pc_reg = 2; + break; + case PC_TYPE::ACT: + pc_reg = 3; + break; + case PC_TYPE::ZQ: + pc_reg = 4; + break; + case PC_TYPE::REF: + pc_reg = 5; + break; + case PC_TYPE::CYC: + pc_reg = 6; + break; + } + + Inst inst = fu_code | pc_reg | t_reg; + + return inst; +} +Inst SMC_BL(int rs1, int rs2, int tgt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BR; + Inst fu_code = (uint64_t)__BL << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst target = (uint64_t)tgt << __BR_TGT; + + Inst inst = op_code | fu_code | s_regs | target; + + return inst; +} +Inst SMC_BEQ(int rs1, int rs2, int tgt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BR; + Inst fu_code = (uint64_t)__BEQ << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst target = (uint64_t)tgt << __BR_TGT; + + Inst inst = op_code | fu_code | s_regs | target; + + return inst; +} +Inst SMC_JUMP(int tgt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BR; + Inst fu_code = (uint64_t)__JUMP << __FU_CODE; + Inst target = tgt; + + Inst inst = op_code | fu_code | target; + + return inst; +} +Inst SMC_SLEEP(uint32_t samt) +{ + assert(samt > 2 && "Cannot sleep for less than 3 cycles."); + Inst op_code = (uint64_t)0x1 << __IS_BR; + Inst fu_code = (uint64_t)__SLEEP << __FU_CODE; + + samt -= 2; + + Inst inst = op_code | fu_code | samt; + + return inst; +} +Inst SMC_LD(int rb, int offset, int rt) +{ + Inst op_code = (uint64_t)0x1 << __IS_MEM; + Inst fu_code = (uint64_t)__LD << __FU_CODE; + Inst s_reg = rb; + Inst imd1 = offset << __IMD1; + Inst t_reg = rt << __RT; + + Inst inst = op_code | fu_code | s_reg | imd1 | t_reg; + + return inst; +} +Inst SMC_ST(int rb, int offset, int rv) +{ + Inst op_code = (uint64_t)0x1 << __IS_MEM; + Inst fu_code = (uint64_t)__ST << __FU_CODE; + Inst b_reg = rb; + Inst imd1 = offset << __IMD1; + Inst v_reg = rv << __RT; // We cannot have imd1 and rs2 present simultaneously + + Inst inst = op_code | fu_code | b_reg | imd1 | v_reg; + + return inst; +} +Inst SMC_AND(int rs1, int rs2, int rt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BW; + Inst fu_code = (uint64_t)__AND << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst t_reg = rt << __RT; + + Inst inst = op_code | fu_code | s_regs | t_reg; + + return inst; +} +Inst SMC_OR(int rs1, int rs2, int rt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BW; + Inst fu_code = (uint64_t)__OR << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst t_reg = rt << __RT; + + Inst inst = op_code | fu_code | s_regs | t_reg; + + return inst; +} +Inst SMC_XOR(int rs1, int rs2, int rt) +{ + Inst op_code = (uint64_t)0x1 << __IS_BW; + Inst fu_code = (uint64_t)__XOR << __FU_CODE; + Inst s_regs = (rs2 << __RS2) | rs1; + Inst t_reg = rt << __RT; + + Inst inst = op_code | fu_code | s_regs | t_reg; + + return inst; +} +Inst SMC_END() +{ + return 0; +} +Inst SMC_INFO(int rdcnt) +{ + Inst op_code = (uint64_t)0x1 << __IS_MISC; + Inst fu_code = (uint64_t)__INFO << __FU_CODE; + return op_code | fu_code | (uint64_t) rdcnt; +} +Mininst SMC_WRITE(int bar, int ibar, int car, int icar, int BL4, int ap) +{ + Mininst fu_code = ((uint64_t)__WRITE) << __DDR_CMD; + Mininst i_bar = bar; + Mininst i_car = car << __DDR_CAR; + Mininst i_ibar = ibar << __DDR_IBAR; + Mininst i_icar = icar << __DDR_ICAR; + Mininst i_BL4 = BL4 << __DDR_BL4; + Mininst i_ap = ap <<__DDR_AP; + + Mininst inst = fu_code | i_bar | i_car | i_ibar | + i_icar | i_BL4 | i_ap; + + return inst; +} +Mininst SMC_READ(int bar, int ibar, int car, int icar, int BL4, int ap) +{ + Mininst fu_code = ((uint64_t)__READ) << __DDR_CMD; + Mininst i_bar = bar; + Mininst i_car = car << __DDR_CAR; + Mininst i_ibar = ibar << __DDR_IBAR; + Mininst i_icar = icar << __DDR_ICAR; + Mininst i_BL4 = BL4 << __DDR_BL4; + Mininst i_ap = ap <<__DDR_AP; + + Mininst inst = fu_code | i_bar | i_car | i_ibar | + i_icar | i_BL4 | i_ap; + + return inst; +} +Mininst SMC_PRE(int bar, int ibar, int pall) +{ + Mininst fu_code = ((uint64_t)__PRE) << __DDR_CMD; + Mininst i_bar = bar; + Mininst i_ibar = ibar << __DDR_IBAR; + Mininst i_pall = pall << __DDR_PALL; + + Mininst inst = fu_code | i_bar | i_ibar | i_pall; + + return inst; +} +Mininst SMC_ACT(int bar, int ibar, int rar, int irar) +{ + Mininst fu_code = ((uint64_t)__ACT) << __DDR_CMD; + Mininst i_bar = bar; + Mininst i_rar = rar << __DDR_RAR; + Mininst i_ibar = ibar << __DDR_IBAR; + Mininst i_irar = irar << __DDR_IRAR; + + Mininst inst = fu_code | i_bar | i_rar | i_ibar | i_irar; + + return inst; +} +Mininst SMC_ZQ() +{ + Mininst fu_code = ((uint64_t)__ZQ) << __DDR_CMD; + + return fu_code; +} +Mininst SMC_REF() +{ + Mininst fu_code = ((uint64_t)__REF) << __DDR_CMD; + + return fu_code; +} +Mininst SMC_NOP() +{ + Mininst fu_code = ((uint64_t)__NOP) << __DDR_CMD; + + return fu_code; +} +Inst SMC_SRE() +{ + Inst op_code = (uint64_t) 0x1 << 56; + Inst fu_code = (uint64_t)__SRE << __FU_CODE; + + return fu_code | op_code; +} +Inst SMC_SRX() +{ + Inst op_code = (uint64_t) 0x1 << 56; + Inst fu_code = (uint64_t)__SRX << __FU_CODE; + + return fu_code | op_code; +} +Inst __pack_mininsts(Mininst i1, Mininst i2, Mininst i3, Mininst i4) +{ + return (uint64_t) i4 << 48 | + (uint64_t) i3 << 32 | + (uint64_t) i2 << 16 | + i1 ; +} + +int is_conditional(Inst i) +{ + uint64_t fcode = (i >> __FU_CODE) & 0x7ff; + switch (fcode) { + case 0: + return 1; + case 1: + return 1; + } + return 0; +} + +int is_branch(Inst i) +{ + uint64_t fcode = (i >> __OP_CODE); + //printf("%ld\n",fcode); + switch (fcode) { + case 8: + return 1; + } + return 0; +} + +int is_ddr(Inst i) +{ + uint64_t fcode = i >> __IS_DDR; + return fcode == 1; +} + +int is_load(Inst i) +{ + uint64_t fcode = i >> __FU_CODE; + return fcode == 4096; +} + +int is_sleep(Inst i) +{ + uint64_t fcode = i >> __FU_CODE; + return fcode == 19; +} + +int is_ddr_read(Inst inst) +{ + int ctr = 0; + for (size_t i = 0 ; i < 4 ; i++) + { + Mininst min = inst >> (i*16); + ctr += (min >> __DDR_CMD) == (Mininst) __READ; + } + return ctr; +} + +void decode_inst(Inst inst) +{ + if(is_ddr(inst)) + { + for(int i = 0 ; i < 4 ; i++) + { + Mininst mini = (inst >> i*16); + decode_ddr(mini); + if (i<3) printf(" : "); + } + } + else + { + int fu_code = inst >> __FU_CODE; + int fc_mask = 0x7ff; + fu_code = fu_code & fc_mask; + int op_code = inst >> __OP_CODE; + int is_br = op_code == 0x8; + int is_mem = op_code == 0x2; + int is_bw = op_code == 0x1; + int is_misc = op_code == 0x4; + int is_arit = op_code == 0x0; + uint64_t rid_mask = 0xf; + uint64_t imd_mask = 0xffff; + uint64_t br_tgt_mask = 0x7ffff; + uint64_t jmp_tgt_mask = 0x7ffffff; + int rs1 = (inst >> __RS1) & rid_mask; + int rs2 = (inst >> __RS2) & rid_mask; + int rt = (inst >> __RT) & rid_mask; + int imd1 = (inst >> __IMD1) & imd_mask; + int imd3 = (inst >> __IMD3) & imd_mask; + int bt = (inst >> __BR_TGT) & br_tgt_mask; + int jt = (inst >> __J_TGT) & jmp_tgt_mask; + int samt = (uint32_t)inst; + uint64_t imd_concat = ((uint64_t)imd3 << 16) + imd1; + if (is_arit) + { + switch (fu_code) + { + case __ADD: + if(inst == 0) + printf("END"); + else + printf("ADD r%d r%d r%d", rt, rs1, rs2); + break; + case __ADDI: + printf("ADDI r%d r%d %d", rt, rs1, imd1); + break; + case __SUB: + printf("SUB r%d r%d r%d", rt, rs1, rs2); + break; + case __SUBI: + printf("SUBI r%d r%d %d", rt, rs1, imd1); + break; + case __MV: + printf("MOV r%d r%d", rt, rs1); + break; + case __LI: + printf("LI r%d %ld", rt, imd_concat); + break; + case __SRC: + printf("SRC r%d r%d", rt, rs1); + break; + case __LDWD: + printf("LDWD %d r%d", rt, rs1); + break; + case __LDPC: + switch(rs1){ + case 0: + printf("LDPC r%d WRITE_COUNTER", rt); + break; + case 1: + printf("LDPC r%d READ_COUNTER", rt); + break; + case 2: + printf("LDPC r%d PRE_COUNTER", rt); + break; + case 3: + printf("LDPC r%d ACT_COUNTER", rt); + break; + case 4: + printf("LDPC r%d ZQ_COUNTER", rt); + break; + case 5: + printf("LDPC r%d REF_COUNTER", rt); + break; + case 6: + printf("LDPC r%d TOTAL_CYCLE", rt); + break; + } + break; + case __SRE: + printf("SRE"); + break; + case __SRX: + printf("SRX"); + break; + } + } else if (is_br) + { + switch (fu_code) + { + case __BL: + printf("BL PC:%d r%d r%d", bt, rs1, rs2); + break; + case __BEQ: + printf("BEQ PC:0x%x r%d r%d", bt, rs1, rs2); + break; + case __JUMP: + printf("JUMP PC:%d", jt); + break; + case __SLEEP: + printf("SLEEP %d cycles", samt); + break; + } + } else if (is_misc) + { + switch (fu_code) + { + case __INFO: + printf("Auto-generated instruction."); + break; + } + } else if (is_mem) + { + switch (fu_code) + { + case __LD: + printf("LD r%d [r%d]%d", rt, rs1, imd1); + break; + case __ST: + printf("ST [r%d]%d r%d", rs1, imd1, rt); + break; + } + } else if (is_bw) + { + switch (fu_code) + { + case __AND: + printf("AND r%d r%d r%d", rt, rs1, rs2); + break; + case __OR: + printf("OR r%d r%d r%d", rt, rs1, rs2); + break; + case __XOR: + printf("XOR r%d r%d r%d", rt, rs1, rs2); + break; + } + } + } +} + +void decode_ddr(Mininst i) +{ + int ddr_code = i >> __DDR_CMD; + uint64_t rid_mask = 0xf; + int car = (i >> __DDR_CAR) & rid_mask; + int bar = (i >> __DDR_BAR) & rid_mask; + int rar = (i >> __DDR_RAR) & rid_mask; + int icar = (i >> __DDR_ICAR) & 0x1; + int ibar = (i >> __DDR_IBAR) & 0x1; + int irar = (i >> __DDR_IRAR) & 0x1; + int pall = (i >> __DDR_PALL) & 0x1; + int ap = (i >> __DDR_AP) & 0x1; + int bc = (i >> __DDR_BL4) & 0x1; + + switch(ddr_code) + { + case __WRITE: + printf("WR r%d%s r%d%s%s%s", bar, ibar?"++":"", car, icar?"++":"", + ap?" AP":"",bc?" BC":""); + break; + case __READ: + printf("RD r%d%s r%d%s%s%s", bar, ibar?"++":"", car, icar?"++":"", + ap?" AP":"",bc?" BC":""); + break; + case __PRE: + printf("PRE r%d%s%s", bar, ibar?"++":"", pall?" PALL":""); + break; + case __ACT: + printf("ACT r%d%s r%d%s", bar, ibar?"++":"", rar, irar?"++":""); + break; + case __ZQ: + printf("ZQ"); + break; + case __REF: + printf("REF"); + break; + case __NOP: + printf("NOP"); + break; + } +} + +void print_bits(size_t const size, void const * const ptr) +{ + unsigned char *b = (unsigned char*) ptr; + unsigned char byte; + int i, j; + + for (i=size-1;i>=0;i--) + { + for (j=7;j>=0;j--) + { + byte = (b[i] >> j) & 1; + printf("%u", byte); + } + } + printf("\n"); +} diff --git a/sources/api/instruction.h b/sources/api/instruction.h new file mode 100644 index 0000000..1be8691 --- /dev/null +++ b/sources/api/instruction.h @@ -0,0 +1,200 @@ +#ifndef INSTRUCTION_H +#define INSTRUCTION_H + +#include <stdint.h> +#include <stdlib.h> +// SoftMC decode +#define __OP_CODE 59 +#define __FU_CODE 48 +#define __IS_BR 62 +#define __IS_DDR 63 +#define __IS_MISC 61 +#define __IS_MEM 60 +#define __IS_BW 59 +// EXE decode +#define __RS1 0 +#define __RS2 4 +#define __RT 20 +#define __IMD1 4 +#define __IMD2 0 +#define __IMD3 24 +#define __BR_TGT 8 +#define __J_TGT 0 +#define __SLP_AMT 0 +// DDR decode +#define __DDR_CMD 12 +#define __DDR_CAR 4 +#define __DDR_BAR 0 +#define __DDR_RAR 4 +#define __DDR_IBAR 10 +#define __DDR_ICAR 11 +#define __DDR_IRAR 11 +#define __DDR_PALL 11 +#define __DDR_AP 9 +#define __DDR_BL4 8 +// EXE function codes +#define __ADD 0 +#define __ADDI 1 +#define __SUB 2 +#define __SUBI 3 +#define __MV 4 +#define __SRC 5 +#define __LI 6 +#define __LDWD 7 +#define __LDPC 8 +#define __SRE 0x100 +#define __SRX 0x101 +#define __BL 0 +#define __BEQ 1 +#define __JUMP 2 +#define __SLEEP 3 +#define __INFO 0 // Various information about upcoming block +#define __AND 0 +#define __OR 1 +#define __XOR 2 +#define __LD 0 +#define __ST 1 +// DDR function codes +#define __WRITE 8 +#define __READ 9 +#define __PRE 10 +#define __ACT 11 +#define __ZQ 12 +#define __REF 13 +#define __NOP 15 + +// 64 bit instructions +typedef uint64_t Inst; +// 16 bit mini ddr-instructions +typedef uint16_t Mininst; + +//Counter types for LDPC Inst +enum class PC_TYPE { WRITE, READ, PRE, ACT, ZQ, REF, CYC }; + +/** + * Load a word from memory into rt + * rt = [rb] + offset + * To handle structural hazards, the API adds a + * "buffer" NOP instruction following loads. + * @param rb register holding the base address + * @param offset memory address offset + * @param rt register to load the value with + */ +Inst SMC_LD(int rb, int offset, int rt); +/** + * Store a word to memory + * [rb] + offset = rt + * @param rb register holding the base address + * @param offset memory address offset + * @param rv register holding the value to store + */ +Inst SMC_ST(int rb, int offset, int rv); + +Inst SMC_AND(int rs1, int rs2, int rt); +Inst SMC_OR(int rs1, int rs2, int rt); +Inst SMC_XOR(int rs1, int rs2, int rt); + +Inst SMC_ADD(int rs1, int rs2, int rt); +Inst SMC_ADDI(int rs1, uint32_t imd, int rt); +Inst SMC_SUB(int rs1, int rs2, int rt); +Inst SMC_SUBI(int rs1, uint32_t imd, int rt); +Inst SMC_LI(uint32_t imd, int rt); +Inst SMC_MV(int rs1, int rt); +/** + * Shift right circular, shift one bit to right and copy + * the rightmost bit to the leftmost bit of the result. + * @param rs1 register to shift + * @param rt register to load the shifted value into + */ +Inst SMC_SRC(int rs1, int rt); +/** + * Move 32 bit data to wide register's specified offset + * @param rs1 source register where 32-bit data resides + * @param off 32-bit offset (e.g. 0 = bytes(0,4) - 5 = bytes(20,24)) + */ +Inst SMC_LDWD(int rs1, int off); +Inst SMC_LDPC(PC_TYPE pc_type, int rt); +Inst SMC_BL(int rs1, int rs2, int tgt); +Inst SMC_BEQ(int rs1, int rs2, int tgt); +Inst SMC_JUMP(int tgt); +Inst SMC_END(); +Inst SMC_INFO(int rdcnt); + +/** + * Wait for a specified amount of fabric cycles (6ns by default) before + * executing the next instruction + * @param samt how many cycles to wait for, must be greater than 2 + */ +Inst SMC_SLEEP(uint32_t samt); + +/** + * Generate a DDR-WR command + * @param bar bank address register ID + * @param ibar increment BAR (BAR=BAR+BASR) after issuing the write + * @param car column address register ID + * @param icar increment CAR (CAR=CAR+CASR) after issuing the write + * @param BL4 burst-chop write + * @param ap auto-precharge after issuing write + */ +Mininst SMC_WRITE(int bar, int ibar, int car, int icar, int BL4, int ap); +/** + * Generate a DDR-RD command + * @param bar bank address register ID + * @param ibar increment BAR (BAR=BAR+BASR) after issuing the read + * @param car column address register ID + * @param icar increment CAR (CAR=CAR+CASR) after issuing the read + * @param BL4 burst-chop read + * @param ap auto-precharge after issuing read + */ +Mininst SMC_READ(int bar, int ibar, int car, int icar, int BL4, int ap); +/** + * Generate a DDR-PRE command + * @param bar bank address register ID + * @param ibar increment BAR (BAR=BAR+BASR) after issuing the write + * @param pall precharge all banks + */ +Mininst SMC_PRE(int bar, int ibar, int pall); +/** + * Generate a DDR-RD command + * @param bar bank address register ID + * @param ibar increment BAR (BAR=BAR+BASR) after issuing the activate + * @param rar row address register ID + * @param irar increment RAR (RAR=RAR+RASR) after issuing the activate + */ +Mininst SMC_ACT(int bar, int ibar, int rar, int irar); +/** + * Generate a zq-calibration command + */ +Mininst SMC_ZQ(); +/** + * Generate a refresh command + */ +Mininst SMC_REF(); +/** + * Generate a no-operation command + */ +Mininst SMC_NOP(); +/** + * Enters Self-Refresh Mode + */ +Inst SMC_SRE(); +/** + * Exits Self-Refresh Mode + */ +Inst SMC_SRX(); +/** + * Packs four DDR commands together + */ +Inst __pack_mininsts(Mininst i1, Mininst i2, Mininst i3, Mininst i4); + +int is_branch(Inst i); +int is_conditional(Inst i); +int is_load(Inst i); +int is_ddr(Inst i); +int is_sleep(Inst i); +int is_ddr_read(Inst i); +void decode_inst(Inst inst); +void decode_ddr(Mininst i); +void print_bits(size_t const size, void const * const ptr); + +#endif diff --git a/sources/api/lexyacc/Makefile b/sources/api/lexyacc/Makefile new file mode 100644 index 0000000..67a05fa --- /dev/null +++ b/sources/api/lexyacc/Makefile @@ -0,0 +1,29 @@ +program_NAME := smc_parser +program_CXX_SRCS := lex.yy.c y.tab.c $(wildcard ../*.c*) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../ ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) lex.yy.c + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +lex.yy.c: y.tab.c smc.l + flex smc.l + +y.tab.c: smc.y + yacc -d smc.y + +distclean: clean diff --git a/sources/api/lexyacc/example.smc b/sources/api/lexyacc/example.smc new file mode 100644 index 0000000..f9cc4f6 --- /dev/null +++ b/sources/api/lexyacc/example.smc @@ -0,0 +1,102 @@ +// Initialize register ids +// Similar to #define +casr = 0 +basr = 1 +rasr = 2 +bar = 2 +rar = 3 +car = 4 +pattern_reg = 5 +no_row_reg = 6 +no_bank_reg = 7 + +// Begin writing the program +LI no_row_reg 32768 // Write 2^15 to no_row_Reg +LI no_bank_reg 16 // write 16 to no_bank_reg + +// Set stride registers (not required if you +// are not going to increment address registers +// with ddr commands) +LI casr 8 // set column address stride to 8 +LI basr 1 // set bas to 1 +LI rasr 1 // set ras to 1 + +LI bar 0 // Initialize bank address register + +// Init pattern register, we are going to fill +// dram with 0xffs. +LI pattern_reg 0xffffffff + +// Load the data register with the pattern +// second argument here is the word offset +// we are writing to in the data register +// e.g. LDWD 0 pattern reg fills the first four bytes + +LDWD 0 pattern_reg +LDWD 1 pattern_reg +LDWD 2 pattern_reg +LDWD 3 pattern_reg +LDWD 4 pattern_reg +LDWD 5 pattern_reg +LDWD 6 pattern_reg +LDWD 7 pattern_reg +LDWD 8 pattern_reg +LDWD 9 pattern_reg +LDWD 10 pattern_reg +LDWD 11 pattern_reg +LDWD 12 pattern_reg +LDWD 13 pattern_reg +LDWD 14 pattern_reg +LDWD 15 pattern_reg + +BANK_BEGIN: // Add label to loop over banks +LI rar 0 // Reload rar per bank loop + +ROW_BEGIN: // Add label to loop over rows +LI car 0 // Reload car per row loop + +PRE bar // PREcharge bar, don't increment, no precharge-all +WAIT 7 // Insert seven cycles of NOPs + +ACT bar rar // ACTivate bar, rar +WAIT 7 + +// This is a basic primitive that can be used +// to replicate instructions-commands +// this block replicates the WR command 128 times +// and adds it to the program +times 128 { + WR bar car++ + WAIT 3 +} + +// Reload car, we are now going to read from DRAM +LI car 0 + +// Wait tWR, might not be precise. +WAIT 8 +PRE bar +WAIT 7 + +ACT bar rar +WAIT 7 + +times 128 { + RD bar car++ + WAIT 3 +} + +WAIT 8 +PRE bar + +// Increment rar per row loop +ADDI rar rar 1 +// if rar < no_row_reg, execute the "row loop" once more +BL ROW_BEGIN rar no_row_reg + +ADDI bar bar 1 +BL BANK_BEGIN bar no_bank_reg + +// Tell softmc that the program has ended +// This is important! +END
\ No newline at end of file diff --git a/sources/api/lexyacc/smc.l b/sources/api/lexyacc/smc.l new file mode 100644 index 0000000..4d9e478 --- /dev/null +++ b/sources/api/lexyacc/smc.l @@ -0,0 +1,77 @@ +%{ + #include <stdlib.h> + #include "instruction.h" + #include "y.tab.h" +%} + +%% + +"//"[^\n]* ; /* Skip comments */ +[%\t ]+ ; /* Skip whitespace */ + +0x[0-9a-f]+ { + yylval.lit = yytext+2; + return HEX_NUMBER; + } + +[0-9]+ { + yylval.num = atoi(yytext); + return DEC_NUMBER; + } + +r[0-9][0-9]? { + yylval.num = atoi(++yytext); /* Skip the first character? */ + return REG_ID; + } + +"wait"|"WAIT" return WAIT; + +"add"|"ADD" return ADD; +"addi"|"ADDI" return ADDI; +"sub"|"SUB" return SUB; +"subi"|"SUBI" return SUBI; +"mov"|"MOV" return MOV; +"src"|"SRC" return SRC; +"li"|"LI" return LI; +"ldwd"|"LDWD" return LDWD; + +"bl"|"BL" return BRL; +"beq"|"beq" return BREQ; +"jmp"|"JMP" return JMP; + +"wr"|"WR" return WR; +"rd"|"RD" return RD; +"act"|"ACT" return ACT; +"pre"|"PRE" return PRE; +"ref"|"REF" return REF; +"zq"|"ZQ" return ZQ; +"end"|"END" return END; + + +"++" return INCREMENT; +"AP"|"ap" return AUTO_PRECHARGE; +"BC"|"bc" return BURST_CHOP; +"PA"|"pa" return PRECHARGE_ALL; + +"times" return GENFOR; +"ENDPROGRAM" return ENDPROGRAM; + + [a-z_]+ { + char* var_name = (char*) malloc((yyleng+1)*sizeof(char)); + strcpy(var_name, yytext); + yylval.lit = var_name; + return VARIABLE; + } + + [A-Z_]+ { + char* lbl_name = (char*) malloc((yyleng+1)*sizeof(char)); + strcpy(lbl_name, yytext); + yylval.lit = lbl_name; + return LABEL; + } + + . return *yytext; + +%% + +int yywrap(void) { return 1; } diff --git a/sources/api/lexyacc/smc.y b/sources/api/lexyacc/smc.y new file mode 100644 index 0000000..f9ff06d --- /dev/null +++ b/sources/api/lexyacc/smc.y @@ -0,0 +1,427 @@ +%{ + #include "prog.h" + #include "instruction.h" + #include <iostream> + #include <stdlib.h> + #include <stdint.h> + #include <string> + #include <string.h> + #include <queue> + #include <map> + #include <stack> + + extern char* yytext; + extern int yyleng; + extern int yylex(void); + extern "C" void yyerror(char const*); + + int block_levels = 0; + + Program* prog; + std::stack <Program*> prog_addr_stack; + void process_ddr_command_queue(); + std::stack <uint32_t> num_generate_stack; + std::queue <Mininst> ddr_command_queue; + std::map<std::string, uint32_t> variable_map; +%} + +%start program + +%union { void* prg; + Mininst mini_i_type; + Inst i_type; + uint32_t num; + char* lit; + } + +%token <lit> HEX_NUMBER +%token <num> DEC_NUMBER +%token <num> REG_ID +%token <lit> LABEL +%token <lit> VARIABLE + +/* Instruction OP-Codes */ +%token WAIT + +%token ADD +%token ADDI +%token SUB +%token SUBI +%token MOV +%token SRC +%token LI +%token LDWD + +%token BRL +%token BREQ +%token JMP + +%token WR +%token RD +%token ACT +%token PRE +%token REF +%token ZQ +%token END + +/* Misc. operators */ +%token INCREMENT +%token AUTO_PRECHARGE +%token BURST_CHOP +%token PRECHARGE_ALL + +/* Other Expressions */ +%token GENFOR +%token ENDPROGRAM + +%type <num> gen_begin +%type <num> immediate +%type <i_type> instruction +%type <mini_i_type> ddr_command +%type <num> address_register +%type <num> reg_id +%type <num> rw_args +%type <num> pre_args + +%% + +program: + expression_list + | program ENDPROGRAM + { + } + ; + +expression_list: + expression {} + | expression_list expression {} + +expression: + instruction + { + // See if DDR command queue is empty + if (ddr_command_queue.size() > 0) + { + printf("WARNING: I just appended some NOPs to your code\n"); + int nops_to_append = 4 - ddr_command_queue.size(); + Mininst ddr_cmd_batch[4]; + for (int i = 0 ; i < ddr_command_queue.size() ; i++) + { + ddr_cmd_batch[i] = ddr_command_queue.front(); + ddr_command_queue.pop(); + } + for (int i = ddr_command_queue.size() ; i < 4 ; i++) + ddr_cmd_batch[i] = SMC_NOP(); + prog->add_inst(__pack_mininsts(ddr_cmd_batch[0],ddr_cmd_batch[1], + ddr_cmd_batch[2],ddr_cmd_batch[3])); + } + prog->add_inst($1); + } + | branch + { + + } + | pseudo_instruction + { + + } + | LABEL ':' + { + prog->add_label(std::string($1)); + } + | ddr_command + { + ddr_command_queue.push($1); + process_ddr_command_queue(); + } + | definition + { + + } + | gen_begin + { + num_generate_stack.push($1); + } + | '}' // end of a generate block + { + if(block_levels == 0) // pop an error: no generate blocks present, cannot end a block definition + printf("Expected \'}\' somewhere.\n"); // TODO change this with a correct error displaying function + else + { + Program* previous_in_stack = prog_addr_stack.top(); + prog_addr_stack.pop(); + int num_gen = num_generate_stack.top(); + num_generate_stack.pop(); + for (int i = 0 ; i < num_gen ; i++) + previous_in_stack->add_below(*prog); + free(prog); + prog = previous_in_stack; + } + } + ; + +instruction: + ADD reg_id reg_id reg_id + { + $$ = SMC_ADD($3, $4, $2); + } + | ADDI reg_id reg_id immediate + { + $$ = SMC_ADDI($3, $4, $2); + } + | SUB reg_id reg_id reg_id + { + $$ = SMC_SUB($3, $4, $2); + } + | SUBI reg_id reg_id immediate + { + $$ = SMC_SUBI($3, $4, $2); + } + | MOV reg_id reg_id + { + $$ = SMC_MV($3, $2); + } + | SRC reg_id reg_id + { + $$ = SMC_SRC($3, $2); + } + | LI reg_id immediate + { + $$ = SMC_LI($3, $2); + } + | LDWD immediate reg_id + { + $$ = SMC_LDWD($3, $2); + } + | END + { + $$ = SMC_END(); + } + ; + +branch: + BRL LABEL reg_id reg_id + { + prog->add_branch(prog->BR_TYPE::BL, $3, $4, $2); + } + | BREQ LABEL reg_id reg_id + { + prog->add_branch(prog->BR_TYPE::BEQ, $3, $4, $2); + } + | JMP LABEL + { + prog->add_branch(prog->BR_TYPE::JUMP, 0, 0, $2); + } + ; + +pseudo_instruction: + WAIT immediate + { + // currently functions only by adding NOPs + // TODO depending on the # of cycles specified + // we can instead generate a "loop" which enables + // us to wait for more with less instructions. + for(int i = 0 ; i < $2 ; i++) + ddr_command_queue.push(SMC_NOP()); + process_ddr_command_queue(); + } + ; + +immediate: + HEX_NUMBER + { + $$ = (uint32_t) strtol($1, NULL, 16); + } + | DEC_NUMBER + { + $$ = $1; + } + ; +ddr_command: + WR address_register address_register rw_args + { + int reg1_id = $2; + int reg2_id = $3; + int do_increment1 = reg1_id >= 100 ? 1 : 0; + int do_increment2 = reg2_id >= 100 ? 1 : 0; + if(do_increment1) + reg1_id -= 100; + if(do_increment2) + reg2_id -= 100; + int do_ap = $4 == 3 ? 1 : $4 == 1 ? 1 : 0; + int do_bc = $4 == 3 ? 1 : $4 == 2 ? 1 : 0; + $$ = SMC_WRITE(reg1_id, do_increment1, reg2_id, do_increment2, do_bc, do_ap); + } + | RD address_register address_register rw_args + { + int reg1_id = $2; + int reg2_id = $3; + int do_increment1 = reg1_id >= 100 ? 1 : 0; + int do_increment2 = reg2_id >= 100 ? 1 : 0; + if(do_increment1) + reg1_id -= 100; + if(do_increment2) + reg2_id -= 100; + int do_ap = $4 == 3 ? 1 : $4 == 1 ? 1 : 0; + int do_bc = $4 == 3 ? 1 : $4 == 2 ? 1 : 0; + $$ = SMC_READ(reg1_id, do_increment1, reg2_id, do_increment2, do_bc, do_ap); + } + | ACT address_register address_register + { + int reg1_id = $2; + int reg2_id = $3; + int do_increment1 = reg1_id >= 100 ? 1 : 0; + int do_increment2 = reg2_id >= 100 ? 1 : 0; + if(do_increment1) + reg1_id -= 100; + if(do_increment2) + reg2_id -= 100; + $$ = SMC_ACT(reg1_id, do_increment1, reg2_id, do_increment2); + } + | PRE address_register pre_args + { + int reg_id = $2; + int do_increment = reg_id>=100 ? 1 : 0; + if(do_increment) + reg_id -= 100; + $$ = SMC_PRE(reg_id, do_increment, $3); + } + | REF + { + $$ = SMC_REF(); + } + | ZQ + { + $$ = SMC_ZQ(); + } + ; + +address_register: + reg_id + { + $$ = $1; + } + | reg_id INCREMENT + { + $$ = 100 + $1; + } + ; + +definition: + VARIABLE '=' immediate + { + std::string key = std::string($1); + variable_map[key] = $3; + } + ; + +reg_id: + REG_ID + { + $$ = $1; + } + | VARIABLE + { + std::string key = std::string($1); + if (variable_map.find(key) == variable_map.end()) + { + // TODO proper error handling + std::cout << "Variable " << key << " is not defined." << std::endl; + $$ = 0; + } else + $$ = variable_map[key]; + } + ; + +rw_args: + %empty + { + $$ = 0; + } + | AUTO_PRECHARGE + { + $$ = 1; + } + | BURST_CHOP + { + $$ = 2; + } + | AUTO_PRECHARGE BURST_CHOP + { + $$ = 3; + } + | BURST_CHOP AUTO_PRECHARGE + { + $$ = 3; + } + ; + +pre_args: + %empty + { + $$ = 0; + } + | PRECHARGE_ALL + { + $$ = 1; + } + ; + +gen_begin: + GENFOR immediate '{' + { + // See if DDR command queue is empty + if (ddr_command_queue.size() > 0) + { + printf("WARNING: I just appended some NOPs to your code\n"); + int nops_to_append = 4 - ddr_command_queue.size(); + Mininst ddr_cmd_batch[4]; + for (int i = 0 ; i < ddr_command_queue.size() ; i++) + { + ddr_cmd_batch[i] = ddr_command_queue.front(); + ddr_command_queue.pop(); + } + for (int i = ddr_command_queue.size() ; i < 4 ; i++) + ddr_cmd_batch[i] = SMC_NOP(); + prog->add_inst(__pack_mininsts(ddr_cmd_batch[0],ddr_cmd_batch[1], + ddr_cmd_batch[2],ddr_cmd_batch[3])); + } + prog_addr_stack.push(prog); + prog = new Program(); + block_levels++; + $$ = $2; // Number of times this block is generated + } + ; + +%% + +void process_ddr_command_queue() +{ + while (ddr_command_queue.size() >= 4) + { + Mininst ddr_cmd_batch[4]; + for (int i = 0 ; i < 4 ; i++) + { + ddr_cmd_batch[i] = ddr_command_queue.front(); + ddr_command_queue.pop(); + } + prog->add_inst(__pack_mininsts(ddr_cmd_batch[0],ddr_cmd_batch[1], + ddr_cmd_batch[2],ddr_cmd_batch[3])); + } +} + +void yyerror (char const *s) { + fprintf (stderr, "%s\n", s); +} + +int main(int argc, char* argv[]) +{ + if(argc != 2){ + printf("I need you to supply me with the filename!\n"); + exit(1); + } + prog = new Program(); + yyparse(); + prog->save_bin(std::string(argv[1])); +} + diff --git a/sources/api/platform.cpp b/sources/api/platform.cpp new file mode 100644 index 0000000..2dae891 --- /dev/null +++ b/sources/api/platform.cpp @@ -0,0 +1,296 @@ +#include <stdlib.h> +#include <string.h> +#include <sys/time.h> +#include <stdlib.h> +#include <cassert> +#include <stdint.h> +#include <stdio.h> +#include <thread> +#include <iostream> +#include <unistd.h> +#include <boost/lockfree/spsc_queue.hpp> + +#include "platform.h" +#include "board.h" +#include "prog.h" + +[[maybe_unused]] static int consume_total = 0; +[[maybe_unused]] static int receive_total = 0; + +SoftMCPlatform::SoftMCPlatform() +{ + // 32 KB large read buffer + is_dummy = false; + xdma_recv_buf = malloc(32*1024); + + #ifdef PYSMC + py_data_buffer = (uint8_t*)malloc(32*1024*sizeof(uint8_t)); + #endif +} + +SoftMCPlatform::SoftMCPlatform(bool sandbox) +{ + // 32 KB large read buffer + is_dummy = sandbox; + xdma_recv_buf = malloc(32*1024); +} + +SoftMCPlatform::~SoftMCPlatform(){ + if (receiver.joinable()) + receiver.join(); + + if (xdma_recv_buf) + free(xdma_recv_buf); + + if (instr_buf) + free(instr_buf); + + if (iface) + delete iface; + + #ifdef PYSMC + if (py_data_buffer) + free(py_data_buffer); + #endif +} + +int SoftMCPlatform::init(){ + if(is_dummy) + { + instr_buf = malloc(INSTR_BUF_SIZE); + memset(instr_buf, 0, INSTR_BUF_SIZE); + return SOFTMC_SUCCESS; + } + else + { + instr_buf = malloc(INSTR_BUF_SIZE); + memset(instr_buf, 0, INSTR_BUF_SIZE); + + iface = new BoardInterface(BoardInterface::IFACE::XDMA); + if(!iface -> init()) + return SOFTMC_SUCCESS; + else + return SOFTMC_ERR; + } +} + +/** + * This sends a 256 bit data which has it's + * 33rd bit set to '1'. + */ +void SoftMCPlatform::reset_fpga() +{ + if(is_dummy) + { + return; + } + else + { + ((uint8_t*) instr_buf)[8] = (uint8_t) 1; + int sent = iface -> sendData(instr_buf, 32 /*in bytes*/); + // We do not need to zero out the whole buffer + memset(instr_buf, 0, 32); + if(sent) + std::cerr << "Could not reset the FPGA!" << std::endl; + else + std::cout << "Successfully reset the FPGA!" << std::endl; + } +} + +void SoftMCPlatform::execute(Program &prog) +{ + if(is_dummy) + { + [[maybe_unused]] uint64_t* iseq = (uint64_t*) prog.get_inst_array(); + int bytes = prog.size(); + assert (bytes <= INSTR_BUF_SIZE/4 && " too many instructions in the buffer, the limit is 2048."); + return; + } + else + { + uint64_t* iseq = (uint64_t*) prog.get_inst_array(); + uint64_t* temp_ptr = (uint64_t*) instr_buf; + int bytes = prog.size(); + assert (bytes <= INSTR_BUF_SIZE/4 && " too many instructions in the buffer, the limit is 2048."); + + for(int i = 0 ; i < bytes/8 ; i++) + temp_ptr[i*4] = iseq[i]; + + if(receiver.joinable()) + receiver.join(); + + receiver = std::thread(&SoftMCPlatform::consumeData, this); + int sent = iface -> sendData(instr_buf, bytes*4 /*in bytes*/); + memset(instr_buf, 0, bytes*4); + free(iseq); + assert(!sent && "could not send instructions"); + } +} + +void SoftMCPlatform::consumeData() +{ + while(true) + { + int xdma_read_intent = 32*1024; + int xdma_read_size = xdma_read_intent; + int recvd = iface -> recvData((void*)xdma_recv_buf, xdma_read_size); + if(recvd == 0) + break; + // xdma read size in words + // If we did not read 32KBs then the program probably ended + // and the last transfer is trash so we discard it + if(recvd != xdma_read_intent) + xdma_read_size = recvd-32; + xdma_read_size /= 4; + int total_size = xdma_read_size; + int pushsz = api_recv_buf.push(((int*) xdma_recv_buf), xdma_read_size); + while(pushsz < total_size) + pushsz += api_recv_buf.push(((int*) xdma_recv_buf) + pushsz, xdma_read_size = (total_size - pushsz)); + +// printf("SoftMCPlatform::consumedata(): spsc filled with %d beats of data\n", pushsz); +// printf("Consume total: %d\n",++consume_total); + + assert(pushsz == total_size && + "Unexpected amount of data pushed into spsc\n"); + + if(recvd != xdma_read_intent) + break; + + } +} + +/** +* Try to read param(size) bytes from FPGA, function will block until +* all data is read +* @param recv_buf where to copy read data +* @param size number of bytes to read +* returns the number of bytes read on success +*/ +int SoftMCPlatform::receiveData(void* recv_buf, int size){ + if(is_dummy) + { + assert(size>0 && size%4 == 0 && "size is expected to be a multiple of four\n"); + size /= 4; + int * my_buf = (int *) recv_buf; + for(int i = 0; i < size ; ++i) { + my_buf[i] = 0x0; + } + return size * 4; + } + else + { + assert(size>0 && size%4 == 0 && "size is expected to be a multiple of four\n"); + + size /= 4; + int total_size = size; + int rdsz = api_recv_buf.pop((int*) recv_buf, size); + while (rdsz < total_size) + rdsz += api_recv_buf.pop(((int*) recv_buf) + rdsz, size = (total_size-rdsz)); + + assert(rdsz == total_size && "Unexpected amount of data popped from spsc\n"); + return total_size*4; + } +} + +#ifdef PYSMC +int SoftMCPlatform::py_receiveData(int size){ + if (size > 32 * 1024) + { + std::cerr << "Python version only supports read buffer size of up to 32KB!" << std::endl; + return 0; + } + + if(is_dummy) + { + assert(size>0 && size%4 == 0 && "size is expected to be a multiple of four\n"); + size /= 4; + int * my_buf = (int *) py_data_buffer; + for(int i = 0; i < size ; ++i) { + my_buf[i] = 0x0; + } + return size * 4; + } + else + { + assert(size>0 && size%4 == 0 && "size is expected to be a multiple of four\n"); + + size /= 4; + int total_size = size; + int rdsz = api_recv_buf.pop((int*) py_data_buffer, size); + while (rdsz < total_size) + rdsz += api_recv_buf.pop(((int*) py_data_buffer) + rdsz, size = (total_size-rdsz)); + + assert(rdsz == total_size && "Unexpected amount of data popped from spsc\n"); + return total_size*4; + } +} + +py::memoryview SoftMCPlatform::get_buffer_memoryview() +{ + return py::memoryview::from_memory((uint64_t*) py_data_buffer, sizeof(uint8_t) * 32 * 1024, true); +} +#endif + +int SoftMCPlatform::count_bitflips_in_row(unsigned char comp_pattern){ + int num_bitflips = 0; + unsigned char buf[8192]; + receiveData(buf, 8192); // read one row each iteration + + for(int j = 0 ; j < 8192 ; j++){ + if(comp_pattern != buf[j]) + { + for(int i = 0 ; i < 8 ; i++) + { + if(((comp_pattern >> i) & 1) != ((buf[j] >> i) & 1)) + num_bitflips ++; + } + } + } + + return num_bitflips; +} + +void SoftMCPlatform::set_aref(const bool on) +{ + if(is_dummy) + { + std::cout << (on ? "Enabled" : "Disabled") << " autorefresh!" << std::endl; + return; + } + else + { + ((uint8_t*) instr_buf)[8] = (uint8_t) 0x8; + ((uint8_t*) instr_buf)[0] = on; + int sent = iface -> sendData(instr_buf, 32 /*in bytes*/); + // We do not need to zero out the whole buffer + memset(instr_buf, 0, 32); + + if(sent) + std::cerr << "Could not set auto refresh!" << std::endl; + // else + // std::cout << (on ? "Enabled" : "Disabled") << " autorefresh!" << std::endl; + } +} + +void SoftMCPlatform::readRegisterDump() +{ + if(is_dummy) + return; + + /** The author decided to use printfs explicitly within this function + * because printing unsigned hex bytes is uglier with stdio + */ + uint8_t readData[64]; + this -> receiveData((void*)readData, 64); // first read wdata content + printf("WDATA: 0x"); + for(int i = 63 ; i >= 0 ; i--) + printf("%x", readData[i]); + printf("\n"); + this -> receiveData((void*)readData, 64); // read register content + for(int r = 0 ; r < 16 ; r++) + { + printf("R%d: 0x", r); + printf("%x", ((uint32_t*)readData)[r]); + printf("\n"); + } +}
\ No newline at end of file diff --git a/sources/api/platform.h b/sources/api/platform.h new file mode 100644 index 0000000..277dcdf --- /dev/null +++ b/sources/api/platform.h @@ -0,0 +1,83 @@ +#include "board.h" +#include "prog.h" +#include <thread> +#include <boost/lockfree/spsc_queue.hpp> + +#ifdef PYSMC +#include "ext/pybind11/include/pybind11/pybind11.h" +namespace py = pybind11; +#endif + +//ERROR CODES +#define SOFTMC_SUCCESS 0 +#define SOFTMC_ERR -1 +#define SOFTMC_NO_PLATFORM -2 +#define SOFTMC_ERR_OPEN_FPGA -3 +#define SOFTMC_NO_SUCH_FPGA -4 + +class SoftMCPlatform{ + #define INSTR_BUF_SIZE 32*2048 + #define API_BUF_SIZE 1024*1024*2 + public: + SoftMCPlatform(); + SoftMCPlatform(bool); + ~SoftMCPlatform(); + /** + * Initializes the whole platform + * @return SOFTMC_SUCCESS on sucessful initialization + */ + int init(); + /** + * Resets SoftMC logic, won't reset PCI-E endpoint or the PHY interface + */ + void reset_fpga(); + /** + * Sends SoftMC program to the FPGA board over PCI-E + * @param prog reference to the program object to send + */ + void execute(Program & prog); + /** + * Receive data from the FPGA board over PCI-E + * @param dst_buf pointer to the buffer that will receive the data + * @param num_words number of bytes to read + */ + int receiveData(void* dst_buf, int num_words); + + #ifdef PYSMC + int py_receiveData(int num_words); + #endif + + /** + * Compare data with a given data pattern (repeating bytes) + * and return number of bitflips in 8KB of data + * @param comp_pattern one byte data pattern to compare the read data + */ + int count_bitflips_in_row(unsigned char comp_pattern); + + /** + * Turn auto-refresh on-off + * @param on true to turn aref on false otherwise + */ + void set_aref(bool on); + + /** + * Used along with Program::dumpRegisters to read register content + */ + void readRegisterDump(); + + private: + bool is_dummy; + + BoardInterface *iface; + void* instr_buf; + std::thread receiver; + boost::lockfree::spsc_queue<int, boost::lockfree::capacity<API_BUF_SIZE/4>> api_recv_buf; + void* xdma_recv_buf; + void consumeData(); + + #ifdef PYSMC + public: + uint8_t* py_data_buffer = nullptr; + py::memoryview get_buffer_memoryview(); + #endif +}; diff --git a/sources/api/prog.cpp b/sources/api/prog.cpp new file mode 100644 index 0000000..781c079 --- /dev/null +++ b/sources/api/prog.cpp @@ -0,0 +1,675 @@ +#include "instruction.h" +#include "prog.h" +#include <string> +#include <vector> +#include <cassert> +#include <iostream> +#include <fstream> +#include <bitset> +#include <algorithm> +#include <sys/wait.h> +#include <sys/stat.h> +#include <unistd.h> +#include <fcntl.h> +#include <string.h> + +Program::Program() +{ + dumpRegsCalled = false; +} + +Program::Program(std::string fname) +{ + std::cout << "WARNING: make sure that the executable \"smc_parser\" exists in the working directory" << std::endl; + + std::string correctFileName = "./smc_parser " + fname + " < " + fname + " >/dev/null"; + if (system(correctFileName.c_str()) != 0) + { + std::cerr << "Error parsing SMC program, abort." << std::endl; + exit(-1); + } + + // Read instruction binary data from fname.inst + std::ifstream binFile (fname + ".bin", std::ios::in | std::ios::binary); + int no_insts = 0; + // Read no of insts + binFile.read((char*) &no_insts, 4); + Inst insts[no_insts]; + // Read all insts + binFile.read((char*) insts, sizeof(Inst)*no_insts); + for(int i = 0 ; i < no_insts ; i++) + { + program.push_back(insts[i]); + spc++; + } + binFile.close(); + // Read branch label positions from fname.meta + std::ifstream metaFile (fname + ".meta", std::ios::in); + std::string line; + bool parseLabels = true; + while(getline(metaFile,line)) + { + if(parseLabels) + { + if (line == "-") + { + parseLabels = false; + continue; + } + std::string label = line.substr(0,line.find(" ")); + std::string target = line.substr(line.find(" ")+1,line.length()); + (labels)[label] = stoi(target); + } + else + { + std::string target = line.substr(0,line.find(" ")); + std::string label = line.substr(line.find(" ")+1,line.length()); + (branches)[stoi(target)] = label; + } + } + metaFile.close(); +} + + +void Program::add_label(std::string name) +{ + if(labels.count(name)) + std::cerr << "Trying to add label " << name << " multiple times!" << std::endl; + (labels)[name] = spc; +} +void Program::add_wait(int wait_cycles) +{ + for(; wait_cycles > 0 ; wait_cycles--) + { + minprogram.push_back(SMC_NOP()); + } +} + +void Program::add_mininst(Mininst mi, int wait_after) +{ + minprogram.push_back(mi); + for ( ; wait_after>0 ; wait_after--) + minprogram.push_back(SMC_NOP()); + // if (verbose) { + // std::cout << "Added new mininst with wait." << std::endl; + // for(auto& mininst : minprogram){ + // std::cout << std::hex << mininst << " "; + // } + // std::cout << std::endl; + // } +} + +void Program::pack_minprogram() +{ + // if (verbose) std::cout << "Packing minprogram" << std::endl; + while(minprogram.size() >= 4) + { + this->add_inst(__pack_mininsts(minprogram[0], minprogram[1], minprogram[2], minprogram[3])); + minprogram.erase(minprogram.begin(), minprogram.begin()+4); + } + + switch (minprogram.size()) + { + case 0: + return; + case 1: + this->add_inst(__pack_mininsts(minprogram[0], SMC_NOP(), SMC_NOP(), SMC_NOP())); + break; + case 2: + this->add_inst(__pack_mininsts(minprogram[0], minprogram[1], SMC_NOP(), SMC_NOP())); + break; + case 3: + this->add_inst(__pack_mininsts(minprogram[0], minprogram[1], minprogram[2], SMC_NOP())); + break; + } + // std::cerr << "WARNING: Number of mininsts is not multiple of four. " + // << "Appending "<< 4-minprogram.size() << " extra NOPs for alignment." + // << std::endl; + minprogram.erase(minprogram.begin(), minprogram.end()); +} + +void Program::add_inst(Inst i) +{ + program.push_back(i); + if (is_load(i)) + this->add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); + spc += 1; +} + +void Program::add_inst(Mininst m1, Mininst m2, Mininst m3, Mininst m4) +{ + Inst i = (uint64_t) m4 << 48 | + (uint64_t) m3 << 32 | + (uint64_t) m2 << 16 | + m1; + program.push_back(i); + spc += 1; +} + +void Program::add_branch(BR_TYPE bt, int rs1, int rs2, std::string tgt) +{ + Inst place_holder = 2; + switch(bt) + { + case BR_TYPE::BL: + place_holder = SMC_BL(rs1, rs2, 0); + break; + case BR_TYPE::BEQ: + place_holder = SMC_BEQ(rs1, rs2, 0); + break; + case BR_TYPE::JUMP: + place_holder = SMC_JUMP(0); + break; + } + (branches)[spc] = tgt; + add_inst(place_holder); +} + +void Program::add_below(const Program &p) +{ + // move instructions from p to the end of + // this program + program.insert(program.end(), p.program.begin(), p.program.end()); + + // update branch targets and branch instruction pcs + for ( const auto &branch : (p.branches) ) + ((this->branches))[branch.first + this -> spc] = branch.second; + + for ( const auto &label : (p.labels) ) + ((this->labels))[label.first] = label.second + this -> spc; + + // update pc + this -> spc += p.spc; +} + +Inst* Program::get_inst_array() +{ + linear_analysis(); + insert_generated(); +// printf("Inserted auto-generated info instructions\n"); + preprocess_branches(); +// printf("Assigned branch targets to labels\n"); + //pretty_print(); + int bytes = size(); + Inst* arr = (Inst*) malloc(bytes); + int ind = 0; + for (auto & element : (program)) + arr[ind++] = element; + return arr; +} + +Inst* Program::get_insts() +{ + int bytes = size(); + Inst* arr = (Inst*) malloc(bytes); + int ind = 0; + for (auto & element : (program)) + arr[ind++] = element; + return arr; +} + +int Program::size() +{ + return program.size() * 8; +} + +void Program::preprocess_branches() +{ + for ( const auto &branch : (branches) ) { + Inst br = (program)[branch.first]; + int lbl = (labels)[branch.second]; +// printf("branch addr: %d branch label: %s lbl_adr: %d is_cond: %d\n", branch.first, branch.second.c_str(), +// lbl,is_conditional(br)); + assert(is_branch(br) && + "preprocess_branches(): expected a branch but got something else"); + uint64_t target_b = ((uint64_t)lbl) << 8; + uint64_t target_j = (uint64_t) lbl; + br |= is_conditional(br) ? target_b : target_j; + (program)[branch.first] = br; + } +} + +/** + * Currently only goes through the whole program + * to see if any constraint is violated + */ +void Program::linear_analysis() +{ + bool inSequence = false; + uint seqPc = 0; + int readCtr = 0; + for (uint pc = 0 ; pc < program.size() ; pc++) + { + Inst cur_inst = (program)[pc]; + if(!inSequence) + { + if(is_ddr(cur_inst)) + { + //printf("Program::linear_analysis(): enter segment at pc: %d\n", pc); + inSequence = true; + seqPc = pc; + readCtr = is_ddr_read(cur_inst); + } + }else + { + // we see a non-ddr inst, + // can commit whatever info we want + // related to the previous segment + if(!is_ddr(cur_inst) && !is_sleep(cur_inst)) + { + //printf("Program::linear_analysis(): exit segment at pc: %d with %d reads \n", pc, readCtr); + inSequence = false; + // TODO either directly insert into the vector + // or keep these records elsewhere + if(readCtr > 1024) + { + // TODO Add code that will warn user violently here. + // maybe even an assertion? + std::cerr << "WARNING: SoftMC won't be able to run your program timing-critically!" << std::endl; + }else if(readCtr > 0) + { + // Insert info-encapsulating packet into program + Inst warn_pipeline = SMC_INFO(readCtr); + (warnings)[seqPc] = warn_pipeline; + } + }else + { + readCtr += is_ddr_read(cur_inst); + } + } + } +} + + +void Program::insert_generated() +{ + // WARNING this assumes that map keys are iterated + // in ascending order + + std::vector<uint> warn_keys; + std::vector<Inst> warn_values; + + for ( const auto &warn : (this->warnings)) + { + warn_keys.push_back(warn.first); + warn_values.push_back(warn.second); + } + + while (warn_keys.size()>0) + { + // new branch and label mappings + uint warn_key = warn_keys.front(); + Inst warn_value = warn_values.front(); + warn_keys.erase(warn_keys.begin()); + warn_values.erase(warn_values.begin()); + + std::map<std::string, uint> n_labels; + std::map<uint, std::string> n_branches; + program.insert(program.begin() + warn_key, warn_value); + for ( const auto &branch : (this->branches) ) + { + // branch target > inserted instruction pc + if((labels)[branch.second] > warn_key) + { + (n_labels)[branch.second] = (labels)[branch.second] + 1; + }else + { + (n_labels)[branch.second] = (labels)[branch.second]; + } + // branching instruction pc > inserted instruction pc + if(branch.first >= warn_key) + { + (n_branches)[branch.first + 1] = branch.second; + }else + { + (n_branches)[branch.first] = branch.second; + } + } + for (uint i = 0 ; i < warn_keys.size() ; i++) + warn_keys[i] += 1; + + labels = n_labels; + branches = n_branches; + } +} + +void Program::pretty_print() +{ + for (uint pc = 0 ; pc < program.size() ; pc++) + { + for ( const auto &label : (labels) ) + { + if(label.second == pc) + std::cout << label.first << std::endl; + } + printf("%04d: ", pc); + decode_inst((program)[pc]); + for ( const auto &branch : (branches) ) + { + if(branch.first == pc) + std::cout << " (TARGET LABEL: " << branch.second << ")"; + } + printf("\n"); + } + +} + +void Program::bin_dump() +{ + for (uint pc = 0 ; pc < program.size() ; pc++) + { + for ( const auto &label : (labels) ) + { + if(label.second == pc) + std::cout << label.first << std::endl; + } + std::cout << "PC: " << pc << " Inst: " << std::hex << (program)[pc] << std::dec << std::endl; + + for ( const auto &branch : (branches) ) + { + if(branch.first == pc) + std::cout << branch.second; + } + printf("\n"); + } +} + +void Program::save_bin(const std::string &fname) +{ + // Write instruction binary data to fname.inst + std::ofstream binFile (fname + ".bin", std::ios::out | std::ios::binary); + Inst* insts = this -> get_insts(); + int no_insts = spc; + binFile.write ((char*)&no_insts, sizeof(no_insts)); + binFile.write ((char*) insts, sizeof(Inst)*no_insts); + binFile.close(); + // Write branch label positions to fname.meta + std::ofstream metaFile (fname + ".meta", std::ios::out); + for ( const auto &label : (labels) ) + metaFile << label.first << " " << label.second << "\n"; + metaFile << "-" << "\n"; + for ( const auto &branch : (branches) ) + metaFile << branch.first << " " << branch.second << "\n"; + metaFile.close(); +} + +void Program::save_coe_here(const std::string &prj_dir) +{ + std::ofstream coeFile ("program.coe", std::ios::out | std::ios::binary); + uint64_t* insts = (uint64_t*) this -> get_inst_array(); + coeFile << "memory_initialization_radix=2;\n"; + coeFile << "memory_initialization_vector=\n"; + for(uint i = 0 ; i < program.size() ; i++){ + coeFile << std::bitset<64>(insts[i]); + if( i == (program.size()) -1){ + coeFile << ";"; + }else{ + coeFile << ",\n"; + } + } + coeFile.close(); +} + + +void Program::save_coe(const std::string &prj_dir) +{ + std::ofstream coeFile (prj_dir + "/VU095.sim/sim_1/behav/xsim/simmem.coe", std::ios::out | std::ios::binary); + std::ofstream coeFile2 (prj_dir + "/coe/simmem.coe", std::ios::out | std::ios::binary); + std::ofstream mifFile (prj_dir + "/VU095.sim/sim_1/behav/xsim/instr_blk_mem_sim.mif", std::ios::out | std::ios::binary); + uint64_t* insts = (uint64_t*) this -> get_inst_array(); + coeFile << "memory_initialization_radix=2;\n"; + coeFile2 << "memory_initialization_radix=2;\n"; + coeFile << "memory_initialization_vector=\n"; + coeFile2 << "memory_initialization_vector=\n"; + for(uint i = 0 ; i < program.size() ; i++){ + coeFile << std::bitset<64>(insts[i]); + coeFile2 << std::bitset<64>(insts[i]); + mifFile << std::bitset<64>(insts[i])<<"\n"; + if( i == (program.size()) -1){ + coeFile << ";"; + coeFile2 << ";"; + }else{ + coeFile << ",\n"; + coeFile2 << ",\n"; + } + } + coeFile.close(); + coeFile2.close(); + mifFile.close(); +} + + +void Program::dump_registers() +{ + if(dumpRegsCalled) + { + std::cerr << "You can use your trump card (dump_register) only once per program!" << std::endl; + exit(1); + } + dumpRegsCalled = true; + + this -> add_inst(SMC_LI(15,13)); // BAR + this -> add_inst(SMC_LI(0,14)); // RAR + this -> add_inst(SMC_LI(0,15)); // CAR + + this -> add_inst(__pack_mininsts(SMC_PRE(13,0,1), // PRECHARGE ALL + SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_ACT(13,0,14,0), // ACT B15 R0 + SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_WRITE(13,0,15,0,0,0), // WRITE WDATA CONTENT + SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_READ(13,0,15,0,0,0), // READ WDATA CONTENT + SMC_NOP(),SMC_NOP(),SMC_NOP())); + + for (int i = 0 ; i < 16 ; i++) // SWAP WDATA WITH REG VALUES + this -> add_inst(SMC_LDWD(i,i)); + + this -> add_inst(__pack_mininsts(SMC_WRITE(13,0,15,0,0,0), // WRITE WDATA CONTENT + SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_READ(13,0,15,0,0,0), // READ WDATA CONTENT + SMC_NOP(),SMC_NOP(),SMC_NOP())); + + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP())); + this -> add_inst(__pack_mininsts(SMC_PRE(13,0,1), // PRECHARGE ALL + SMC_NOP(),SMC_NOP(),SMC_NOP())); +} + +bool Program::isDumpRegsCalled() +{ + return dumpRegsCalled; +} + +void Program::debug(const std::string &prj_dir, bool first) +{ + int pid = fork(); + if(pid == 0){ + + std::string source = prj_dir + "/source.tcl"; + std::string project = prj_dir + "/VU095.xpr"; + std::string arg; + int fd = open("/dev/null", O_WRONLY); + dup2(fd, 1); + if(first) + arg = "first"; + else + arg = "debug"; + + execl("/opt/Xilinx/Vivado/2018.2/bin/vivado", + "vivado","-mode","tcl","-nojournal","-applog","-log","/tmp/asmc.log", + "-source", source.c_str(), project.c_str(),"-tclargs", arg.c_str(), NULL); + + }else if(pid > 0){ + std::cout << "Debugger starting...\n"; + + mkfifo("/tmp/fifo_1", 0666); + int f1 = open("/tmp/fifo_1", O_WRONLY); + mkfifo("/tmp/fifo_2", 0666); + int f2 = open("/tmp/fifo_2", O_RDONLY); + + char buf[100]; + [[maybe_unused]] int rv; + + std::vector<std::string> vios; + std::string line, start, end, vio; + + start = "sim_tb_top.mem_model_x8.memModels_Ri1[0].memModel1[0].ddr4_model.always_diff_ck.if_diff_ck:VIOLATION:"; + end = "sim_tb_top.mem_model_x8.memModels_Ri1[0].memModel1[1].ddr4_model.always_diff_ck.if_diff_ck:VIOLATION:"; + + std::string msg, com, param,param2; + std::vector<std::string> commands {"reg", "mem", "time", "step", "run", + "until", "exit", "btwn","stat"}; + + int state = 0; + + while(state != 5){ + switch (state){ + case 0: + std::cout << "Enter command: "; + std::cin >> com; + if(std::find(commands.begin(), commands.end(), com) != commands.end()){ + if(com == "step" || com == "exit" || com == "time" || com == "stat"){ + msg = com + "_\n"; + }else if(com == "btwn"){ + std::cin >> param; + std::cin >> param2; + msg = com + "_" + param + "_" + param2 + "\n"; + }else{ + std::cin >> param; + msg = com + "_" + param + "\n"; + } + state = 1; + }else{ + std::cout << "Please enter a valid command...\n"; + } + break; + + case 1: + rv = write( f1, (char*)msg.c_str(), msg.size()); + memset(buf, 0, sizeof(buf)); + while(read(f2, buf, sizeof(buf)) == 0); + state = 2; + break; + + case 2: + if(com == "reg" || com == "mem") + std::cout << com << " " << param << ": " << buf; + else if(com == "btwn") + std::cout << "Time between "<< param << "-" << param2 << ": " << buf; + else if(com == "time") + std::cout << "Current time: " << buf; + else if(com == "stat") + std::cout << "Stats:\n" << buf; + else if(com == "exit"){ + state = 5; + break; + } + state = 3; + break; + + case 3: + std::ifstream logfile ("/tmp/asmc.log"); + while (std::getline(logfile, line)) + { + if(line.find(start) != std::string::npos) + { + vio = line; + if(std::find(vios.begin(), vios.end(), vio) != vios.end()) + continue; + std::cout << line.substr(91) << "\n"; + while (std::getline(logfile, line)) + { + if (line.find(end) != std::string::npos) + break; + std::cout << line << "\n"; + } + vios.insert(vios.begin(), vio); + } + } + logfile.close(); + state = 0; + break; + + } + } + close(f1); + close(f2); + wait(NULL); + remove("/tmp/fifo_1"); + remove("/tmp/fifo_2"); + remove("/tmp/asmc.log"); + std::cout << "Exiting debugger.\n"; + } +} + +void Program::debug(const std::string &prj_dir, const std::string &filename) +{ + int pid = fork(); + if(pid == 0){ + + std::string source = prj_dir + "/source.tcl"; + std::string project = prj_dir + "/VU095.xpr"; + int fd = open("/dev/null", O_WRONLY); + dup2(fd, 1); + execl("/opt/Xilinx/Vivado/2018.2/bin/vivado", + "vivado","-mode","tcl","-nojournal","-applog","-log","/tmp/asmc.log", + "-source", source.c_str(), project.c_str(),"-tclargs", "update", NULL); + + }else if(pid > 0){ + + std::string path, line, param, value, time_path, temp_path; + + path = prj_dir.substr(0, prj_dir.find("VU095")) + "phy_ddr4_ex/imports/"; + time_path = path + "timing_tasks.sv"; + temp_path = path + "temp.sv"; + + std::ifstream timing_tasks (time_path); + std::ifstream param_file (filename); + std::ofstream temp_file (temp_path); + + std::map <std::string, std::string> params; + + while (std::getline(param_file, line)){ + param = line.substr(0, line.find("=")); + param = param.substr(param.find_first_not_of(" \n\r\t\f\v"),param.find_last_not_of(" \n\r\t\f\v") + 1); + param = " SetTSArray (i" + param; + value = line.substr(line.find("=") + 1); + value = value.substr(value.find_first_not_of(" \n\r\t\f\v"),value.find_last_not_of(" \n\r\t\f\v") + 1); + value = std::string((7- value.size()), ' ') + value; + params[param] = value; + } + param_file.close(); + + while (std::getline(timing_tasks, line)){ + if(line.find(" SetTSArray (i") == std::string::npos){ + temp_file << line << "\n"; + continue; + } + for(const auto & param : params){ + if(line.find(param.first) != std::string::npos){ + line = line.replace(35, 7, param.second); + break; + } + } + temp_file << line << "\n"; + } + + timing_tasks.close(); + temp_file.close(); + remove(time_path.c_str()); + rename(temp_path.c_str(), time_path.c_str()); + wait(NULL); + std::cout << "Timing parameters are updated...\n"; + } +} diff --git a/sources/api/prog.h b/sources/api/prog.h new file mode 100644 index 0000000..d9febd7 --- /dev/null +++ b/sources/api/prog.h @@ -0,0 +1,120 @@ +#ifndef PROG_H +#define PROG_H + +#include <vector> +#include <map> +#include <string> +#include "instruction.h" + +class Program{ +private: + std::map<std::string, uint> labels; + std::map<uint, std::string> branches; + std::map<uint, Inst> warnings; + std::vector<Inst> program; + std::vector<Mininst> minprogram; + int spc = 0; + + bool dumpRegsCalled; + + void preprocess_branches(); + void linear_analysis(); + void insert_generated(); + Inst* get_insts(); + +public: + enum class SEQ_TYPE { WRITE, READ }; + enum BR_TYPE { BEQ, BL, JUMP }; + + + Program(); + + /** + * Load an instruction stream from file + * @param fname name of the file + */ + Program(std::string fname); + + void add_mininst(Mininst mi, int wait_after); + void add_wait(int wait_cycles); + void pack_minprogram(); + /** + * Add an instruction to the program + * @param i instruction to add, usually generated by instruction.h functions (e.g. SMC_ADD()) + */ + void add_inst(Inst i); + /** + * Add four mininsts to the program + */ + void add_inst(Mininst,Mininst,Mininst,Mininst); + void add_label(std::string name); + /** + * Add a conditional branch to the program + * @bt the type of branch instruction to be executed: + * either one of: BR_TYPE.BEQ, BR_TYPE.BL or BR_TYPE.JUMP + * @rs1 branch instr. operand 1 + * @rs2 branch instr. operand 2 + * @tgt branch target to jump to if condition is satisfied + */ + void add_branch(BR_TYPE bt, int rs1, int rs2, std::string tgt); + + /** + * Append another program to this program. + * @param p program to append to the end of this program. + */ + void add_below(const Program &p); + + /** + * Parse SMC programs + * @param fname file to read from + */ + void parse_from_file(std::string fname); + + Inst* get_inst_array(); + + /** + * @return size of the program in bytes + */ + int size(); + + /** + * Print instructions in a human readable form + */ + void pretty_print(); + + /** + * Print instruction binaries + */ + void bin_dump(); + /** + * Save this instruction stream to file + * @param fname name of the file + */ + void save_bin(const std::string &fname); + void save_coe_here(const std::string &prj_dir); + void save_coe(const std::string &prj_dir); + void debug(const std::string &prj_dir, bool first); + void debug(const std::string &prj_dir, const std::string &filename); + + /** + * Insert routine required to dump register file content. + * The generated routine requires registers 13-15 to operate correctly. + * It won't do any context-switch and registers 13-15 will have garbage + * values when this routine is executed. It will also use DRAM bank + * 15 row 0-1 to store and readback data, it will also precharge all banks. + * This routine will also mess with WDATA reg content *after* dumping it. + * + * Aside from all the above, you need to intercept reads yourself. Make sure + * that there is no other data left in the buffers before this routine executes + * and call platform::readRegisterDump(). + * You have been warned. + * You have been warned twice. + * You did not hear that wrong, this will dump all register content. + */ + void dump_registers(); + + bool isDumpRegsCalled(); + +}; + +#endif diff --git a/sources/api/pySoftMC/.gitignore b/sources/api/pySoftMC/.gitignore new file mode 100644 index 0000000..f1fe8d1 --- /dev/null +++ b/sources/api/pySoftMC/.gitignore @@ -0,0 +1 @@ +*.so
\ No newline at end of file diff --git a/sources/api/pySoftMC/Makefile b/sources/api/pySoftMC/Makefile new file mode 100644 index 0000000..771df53 --- /dev/null +++ b/sources/api/pySoftMC/Makefile @@ -0,0 +1,36 @@ +# https://stackoverflow.com/questions/4933285/how-to-determine-python-version-in-makefile +python_version_full := $(wordlist 2,4,$(subst ., ,$(shell python3 --version 2>&1))) +python_version := $(word 1,${python_version_full}).$(word 2,${python_version_full}) + +program_NAME := pySoftMC +program_CXX_SRCS := $(wildcard ../*.c) $(wildcard ../*.cpp) ./pySoftMC.cpp +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) + +program_INCLUDE_DIRS := . .. /usr/include/python${python_version} ../ext/pybind11/include/pybind11 +program_LIBRARY_DIRS := +program_LIBRARIES := python${python_version} +CPPFLAGS += -g -std=c++14 -pthread -O3 -fPIC -Wall + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) -DPYSMC +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +SHARED_LIB_FLAGS := -shared -fPIC + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + @- $(RM) $(program_OBJS) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) ${SHARED_LIB_FLAGS} $(program_OBJS) -o $(program_NAME)$(shell python3-config --extension-suffix) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME)$(shell python3-config --extension-suffix) + @- $(RM) $(program_OBJS) + +distclean: clean diff --git a/sources/api/pySoftMC/example.py b/sources/api/pySoftMC/example.py new file mode 100644 index 0000000..2dd2615 --- /dev/null +++ b/sources/api/pySoftMC/example.py @@ -0,0 +1,153 @@ +from pySoftMC import * +import numpy as np + +# Initialize SoftMC platform +platform = SoftMCPlatform() +err = platform.init() +if (err != 0): + print("SoftMC platform initialization failed!") + exit(-1) +platform.reset_fpga() + +# Define registers +CASR = 0 +BASR = 1 +RASR = 2 +BAR = 3 +RAR = 4 +CAR = 5 +PATTERN_REG = 6 + + +def initialize_row(platform: SoftMCPlatform, data32: int, bank: int, row: int): + p = Program() + p.add_inst(SMC_LI(bank, BAR)) + p.add_inst(SMC_LI(row, RAR)) + p.add_inst(SMC_LI(8, CASR)) + + p.add_inst(SMC_LI(data32, PATTERN_REG)) + for i in range(16): + p.add_inst(SMC_LDWD(PATTERN_REG, i)) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_LI(0, CAR)) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + p.add_inst(SMC_ACT(BAR, 0, RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + for i in range(128): + p.add_inst(SMC_WRITE(BAR, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_SLEEP(3)) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + p.add_inst(SMC_END()) + platform.execute(p) + +def read_row(platform: SoftMCPlatform, bank: int, row: int): + p = Program() + p.add_inst(SMC_LI(bank, BAR)) + p.add_inst(SMC_LI(row, RAR)) + p.add_inst(SMC_LI(8, CASR)) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_LI(0, CAR)) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + p.add_inst(SMC_ACT(BAR, 0, RAR, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + for i in range(128): + p.add_inst(SMC_READ(BAR, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_SLEEP(3)) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + + p.add_inst(SMC_END()) + platform.execute(p) + +def hammer_row_double(bank, aggressor_row_1, aggressor_row_2, hammer_count, RAS_scale, RP_scale): + HMR_COUNTER_REG = 7 + NUM_HMR_REG = 8 + + p = Program() + p.add_inst(SMC_LI(bank, BAR)) + p.add_inst(SMC_LI(aggressor_row_1, RAR)) + + p.add_inst(SMC_LI(0, HMR_COUNTER_REG)) + p.add_inst(SMC_LI(hammer_count, NUM_HMR_REG)) + + p.add_label("HMR_BEGIN") + if (RAS_scale > 1): + p.add_inst(SMC_SLEEP(5 * (RAS_scale - 1))) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_LI(aggressor_row_1, RAR)) + if (RP_scale > 1): + p.add_inst(SMC_SLEEP(3 * (RP_scale - 1))) + + + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_ACT(BAR, 0, RAR, 0)) + p.add_inst(SMC_LI(aggressor_row_2, RAR)) + p.add_inst(SMC_SLEEP(5)) + if (RAS_scale > 1): + p.add_inst(SMC_SLEEP(5 * (RAS_scale - 1))) + + p.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_ADDI(HMR_COUNTER_REG, 1, HMR_COUNTER_REG)) + if (RP_scale > 1): + p.add_inst(SMC_SLEEP(3 * (RP_scale - 1))) + + + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_ACT(BAR, 0, RAR, 0)) + p.add_branch(Program.BR_TYPE.BL, HMR_COUNTER_REG, NUM_HMR_REG, "HMR_BEGIN") + + + p.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()) + p.add_inst(SMC_END()) + + platform.execute(p) + +def check_flips(buffer, pattern): + d = [] + out = np.bitwise_xor(buffer, pattern) + for cl in range(128): + for w in range(8): + if (out[cl * 8 + w]) != 0: + _w = out[cl * 8 + w] + for byte in range(8): + for bit in range(8): + bit_id = np.uint64(byte * 8 + bit) + if ((np.right_shift(_w, bit_id)) & np.uint64(0x1)): + d.append((i, cl, w, byte, bit)) + print(f"ITR {i}\tCL {cl}\t Word {w}\t Byte {byte}\t Bit {bit}") + + +# Get the buffer from the underlying C++ API +buffer = np.frombuffer(platform.get_buffer_memoryview()[:8192], dtype=np.uint64) + +ITR_COUNT=10 +ACT_COUNT = 100000 +pattern = 0xFFFFFFFF + +for i in range(ITR_COUNT): + initialize_row(platform, 0xFFFFFFFF, 0, 233) + initialize_row(platform, 0x0, 0, 234) + initialize_row(platform, 0xFFFFFFFF, 0, 235) + + hammer_row_double(0, 233, 235, ACT_COUNT, 1, 1) + + read_row(platform, 0, 234) + platform.receiveData(8*1024) + + check_flips(buffer, 0x0) + diff --git a/sources/api/pySoftMC/pySoftMC.cpp b/sources/api/pySoftMC/pySoftMC.cpp new file mode 100644 index 0000000..992d9a5 --- /dev/null +++ b/sources/api/pySoftMC/pySoftMC.cpp @@ -0,0 +1,95 @@ +#include "platform.h" +#include "instruction.h" +#include "prog.h" + +#include "ext/pybind11/include/pybind11/pybind11.h" +namespace py = pybind11; + +PYBIND11_MODULE(pySoftMC, m) +{ + py::enum_<PC_TYPE>(m, "PC_TYPE") + .value("WRITE", PC_TYPE::WRITE) + .value("READ", PC_TYPE::READ) + .value("PRE", PC_TYPE::PRE) + .value("ACT", PC_TYPE::ACT) + .value("ZQ", PC_TYPE::ZQ) + .value("REF", PC_TYPE::REF) + .value("CYC", PC_TYPE::CYC) + .export_values(); + + m.def("SMC_LD", &SMC_LD, py::arg("rb"), py::arg("offset"), py::arg("rt")); + m.def("SMC_ST", &SMC_ST, py::arg("rb"), py::arg("offset"), py::arg("rt")); + + m.def("SMC_AND", &SMC_AND, py::arg("rs1"), py::arg("rs2"), py::arg("rt")); + m.def("SMC_OR", &SMC_OR, py::arg("rs1"), py::arg("rs2"), py::arg("rt")); + m.def("SMC_XOR", &SMC_XOR, py::arg("rs1"), py::arg("rs2"), py::arg("rt")); + m.def("SMC_ADD", &SMC_ADD, py::arg("rs1"), py::arg("rs2"), py::arg("rt")); + m.def("SMC_ADDI", &SMC_ADDI, py::arg("rs1"), py::arg("imd"), py::arg("rt")); + m.def("SMC_SUB", &SMC_SUB, py::arg("rs1"), py::arg("rs2"), py::arg("rt")); + m.def("SMC_SUBI", &SMC_SUBI, py::arg("rs1"), py::arg("imd"), py::arg("rt")); + m.def("SMC_SRC", &SMC_SRC, py::arg("rs1"), py::arg("rt")); + + m.def("SMC_LI", &SMC_LI, py::arg("imd"), py::arg("rt")); + m.def("SMC_MV", &SMC_MV, py::arg("rs1"), py::arg("rt")); + + m.def("SMC_LDWD", &SMC_LDWD, py::arg("rs1"), py::arg("offset")); + m.def("SMC_LDPC", &SMC_LDPC, py::arg("type"), py::arg("rt")); + + m.def("SMC_BL", &SMC_BL, py::arg("rs1"), py::arg("rs2"), py::arg("tgt")); + m.def("SMC_BEQ", &SMC_BEQ, py::arg("rs1"), py::arg("rs2"), py::arg("tgt")); + m.def("SMC_JUMP", &SMC_JUMP, py::arg("tgt")); + + m.def("SMC_END", &SMC_END); + m.def("SMC_INFO", &SMC_INFO, py::arg("rdcnt")); + m.def("SMC_SLEEP", &SMC_SLEEP, py::arg("samt")); + + m.def("SMC_WRITE", &SMC_WRITE, py::arg("bar"), py::arg("ibar"), py::arg("car"), py::arg("icar"), py::arg("BL4"), py::arg("ap")); + m.def("SMC_READ", &SMC_READ, py::arg("bar"), py::arg("ibar"), py::arg("car"), py::arg("icar"), py::arg("BL4"), py::arg("ap")); + m.def("SMC_PRE", &SMC_PRE, py::arg("bar"), py::arg("ibar"), py::arg("pall")); + m.def("SMC_ACT", &SMC_ACT, py::arg("bar"), py::arg("ibar"), py::arg("rar"), py::arg("irar")); + + m.def("SMC_ZQ", &SMC_ZQ); + m.def("SMC_REF", &SMC_REF); + m.def("SMC_NOP", &SMC_NOP); + m.def("SMC_SRE", &SMC_SRE); + m.def("SMC_SRX", &SMC_SRX); + + m.def("__pack_mininsts", &__pack_mininsts, py::arg("i1"), py::arg("i2"), py::arg("i3"), py::arg("i4")); + + py::class_<SoftMCPlatform>(m, "SoftMCPlatform") + .def(py::init()) + .def("init", &SoftMCPlatform::init) + .def("reset_fpga", &SoftMCPlatform::reset_fpga) + .def("execute", &SoftMCPlatform::execute, py::arg("program")) + .def("receiveData", &SoftMCPlatform::py_receiveData, py::arg("num_words")) + .def("get_buffer_memoryview", &SoftMCPlatform::get_buffer_memoryview) + .def("set_aref", &SoftMCPlatform::set_aref, py::arg("on")) + .def("readRegisterDump", &SoftMCPlatform::readRegisterDump) + .def("count_bitflips_in_row", &SoftMCPlatform::count_bitflips_in_row, py::arg("char_data")); + + py::class_<Program> program(m, "Program"); + + py::enum_<Program::SEQ_TYPE>(program, "SEQ_TYPE") + .value("WRITE", Program::SEQ_TYPE::WRITE) + .value("READ", Program::SEQ_TYPE::READ) + .export_values(); + + py::enum_<Program::BR_TYPE>(program, "BR_TYPE") + .value("BEQ", Program::BR_TYPE::BEQ) + .value("BL", Program::BR_TYPE::BL) + .value("JUMP", Program::BR_TYPE::JUMP) + .export_values(); + + program + .def(py::init()) + .def("add_mininst", &Program::add_mininst, py::arg("mi"), py::arg("wait_after")) + .def("add_wait", &Program::add_wait, py::arg("wait_cycles")) + .def("pack_minprogram", &Program::pack_minprogram) + .def("add_inst", py::overload_cast<Inst>(&Program::add_inst), py::arg("inst")) + .def("add_inst", py::overload_cast<Mininst, Mininst, Mininst, Mininst>(&Program::add_inst)) + .def("add_label", &Program::add_label, py::arg("name")) + .def("add_branch", &Program::add_branch, py::arg("type"), py::arg("rs1"), py::arg("rs2"), py::arg("tgt")) + .def("add_below", &Program::add_below) + .def("pretty_print", &Program::pretty_print) + .def("dump_registers", &Program::dump_registers); +} diff --git a/sources/apps/.gitignore b/sources/apps/.gitignore new file mode 100644 index 0000000..5761abc --- /dev/null +++ b/sources/apps/.gitignore @@ -0,0 +1 @@ +*.o diff --git a/sources/apps/DebugExample/Makefile b/sources/apps/DebugExample/Makefile new file mode 100755 index 0000000..13ecbb0 --- /dev/null +++ b/sources/apps/DebugExample/Makefile @@ -0,0 +1,32 @@ +program_NAME := Debug_example +program_CXX_SRCS := debug_example.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +parser: + $(MAKE) -C ../../api/lexyacc + cp ../../api/lexyacc/smc_parser . + +distclean: clean diff --git a/sources/apps/DebugExample/debug_example.cpp b/sources/apps/DebugExample/debug_example.cpp new file mode 100644 index 0000000..6581354 --- /dev/null +++ b/sources/apps/DebugExample/debug_example.cpp @@ -0,0 +1,58 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> + +using namespace std; + +#define PROJECT_DIRECTORY "<YOUR PROJECT DIRECTORY>" + +#define ADD_REG 7 +#define LD_REG 8 +#define STR_REG 9 + +/** + * @return an instruction formed by NOPs + */ +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +int main() +{ + + Program program; + + program.add_inst(SMC_LI(5, 3)); //R3=5 + program.add_inst(SMC_LI(15, 4)); //R4=15 + program.add_label("LOOP"); + program.add_inst(SMC_ADDI(3, 1, 3)); //R3++ + //If R4>R3, go to LOOP + program.add_branch(program.BR_TYPE::BL,3,4, "LOOP"); + program.add_inst(all_nops()); //R3=R4=15 + program.add_inst(SMC_LI(121, LD_REG)); + program.add_inst(SMC_LI(62, STR_REG)); + program.add_inst(SMC_LI(100, ADD_REG)); + program.add_inst(SMC_ST(ADD_REG, 0, STR_REG)); + program.add_inst(all_nops()); + program.add_inst(SMC_LD(ADD_REG, 0, LD_REG)); + program.add_inst(all_nops()); + program.add_inst(SMC_END()); + + //DRAM timing parameters are updated with timings.txt + program.debug(PROJECT_DIRECTORY, "timings.txt"); + + //Load the program to instruction memory + program.save_coe(PROJECT_DIRECTORY); + program.pretty_print(); + + //Start the debugger + program.debug(PROJECT_DIRECTORY, false); +} diff --git a/sources/apps/DebugExample/timings.txt b/sources/apps/DebugExample/timings.txt new file mode 100644 index 0000000..8de3bed --- /dev/null +++ b/sources/apps/DebugExample/timings.txt @@ -0,0 +1,14 @@ +tRTP = 7500 +tWTRc_S = 2 +tWTRc_L = 5 +tWR = 15000 +tMOD = 36000 +tRCD = 16000 +tRC = 55000 +tRP = 15000 +tCCDc_S = 4 +tCCDc_L = 5 +tRAS = 39000 +tRRDc_S = 4 +tRRDc_L = 4 +tFAW = 20000 diff --git a/sources/apps/QUAC-TRNG/.gitignore b/sources/apps/QUAC-TRNG/.gitignore new file mode 100644 index 0000000..8fce603 --- /dev/null +++ b/sources/apps/QUAC-TRNG/.gitignore @@ -0,0 +1 @@ +data/ diff --git a/sources/apps/QUAC-TRNG/Makefile.ent b/sources/apps/QUAC-TRNG/Makefile.ent new file mode 100644 index 0000000..468a10b --- /dev/null +++ b/sources/apps/QUAC-TRNG/Makefile.ent @@ -0,0 +1,28 @@ +program_NAME := bin/ENT +program_CXX_SRCS := src/slow_entropy.cpp src/util.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +distclean: clean diff --git a/sources/apps/QUAC-TRNG/Makefile.observe b/sources/apps/QUAC-TRNG/Makefile.observe new file mode 100644 index 0000000..d5b1adb --- /dev/null +++ b/sources/apps/QUAC-TRNG/Makefile.observe @@ -0,0 +1,32 @@ +program_NAME := bin/OBSERVE +program_CXX_SRCS := src/observe.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +parser: + $(MAKE) -C ../../api/lexyacc + cp ../../api/lexyacc/smc_parser . + +distclean: clean diff --git a/sources/apps/QUAC-TRNG/Makefile.sha b/sources/apps/QUAC-TRNG/Makefile.sha new file mode 100644 index 0000000..bd0c463 --- /dev/null +++ b/sources/apps/QUAC-TRNG/Makefile.sha @@ -0,0 +1,28 @@ +program_NAME := bin/SHA +program_CXX_SRCS := src/slow_sha.cpp src/util.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := ssl crypto +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +distclean: clean diff --git a/sources/apps/QUAC-TRNG/README.md b/sources/apps/QUAC-TRNG/README.md new file mode 100644 index 0000000..437d0dd --- /dev/null +++ b/sources/apps/QUAC-TRNG/README.md @@ -0,0 +1,58 @@ +## Scripts to Reproduce QUAC-TRNG + +This directory provides all necessary files and instructions to reproduce the results of [QUAC-TRNG](https://people.inf.ethz.ch/omutlu/pub/QUAC-TRNG-DRAM_isca21.pdf). + +``` +@inproceedings{olgun2022quactrng, + title={QUAC-TRNG: High-Throughput True Random Number Generation Using Quadruple Row Activation in Commodity DRAM Chips}, + author={Ataberk Olgun, Minesh Patel, A. Giray Yaglikci, Haocong Luo, Jeremie S. Kim, F. Nisa Bostanci, Nandita Vijaykumar, Oguz Ergin, and Onur Mutlu}, + year={2021}, + booktitle={ISCA} +} +``` + +### Obtain NIST STS + +Download the NIST standard test suite (NIST STS) for random number generators using [this link](https://github.com/arcetri/sts). Install it under `tools` using the instructions provided in the linked repository's README. + +After installing NIST STS, `QUAC-TRNG/tools/sts` should point to the NIST STS executable. + +### Copy SoftMC_reset to bin/ + +Run the following command to copy the reset program under bin/ directory. + +`cp ../ResetBoard/SoftMC_reset bin/` + +### Entropy Tests + +To run entropy tests, simply execute the `run_ent.py` script (you have to execute this in QUAC-TRNG directory): + +``` +make -f Makefile.ent +python3 scripts/run_ent.py <DIMM_NAME> <TEMPERATURE_IN_CELCIUS> <QUAC_ITERATIONS> <ROW_STRIDE> +``` + +Example: `python3 scripts/run_ent.py hytt04 50 1000 4` + +This script will output one text file for every tested 4-bit data pattern (variable named placement in run_ent.py). Each line in the text file starts with the starting `row id` of a DRAM segment, followed by the observed entropy numbers for each bit in the segment. + +### von Neumann Corrected Bitstreams + +This will collect 1 Mb bitstreams from frequently switching bitlines and input them to NIST STS. The script will stop the moment it finds a cell that passes all NIST tests. + +``` +make -f Makefile.observe +python3 scripts/run_vn_nist.py <DIMM_NAME> <TEMPERATURE_IN_CELCIUS> +``` + +Example: `python3 scripts/run_vn_nist.py hytt04 50` + +### SHA-256 Bitstreams + +This section describes how to get the NIST STS results for SHA-256 processed bitstreams. Doing so consists of two steps: 1) finding the DRAM segments with top 10 entropy, 2) generating bitstreams from the top-10-entropy segments and processing them using SHA-256 and performing NIST STS tests. + +First, run `make -f Makefile.sha`. + +1- Obtain the entropy distribution to find the desired 4-bit data pattern (refer to "Entropy Tests"). +2- Get the segments with max. entropy: `python3 scripts/top_ents.py <DIMM_NAME> <PATTERN> <TEMPERATURE>` +3- Run the SHA-256 bitstream generator script: `python3 scripts/sha_bitstreams.py <DIMM_NAME> <PATTERN> <TEMPERATURE>`
\ No newline at end of file diff --git a/sources/apps/QUAC-TRNG/bin/.gitkeep b/sources/apps/QUAC-TRNG/bin/.gitkeep new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/sources/apps/QUAC-TRNG/bin/.gitkeep diff --git a/sources/apps/QUAC-TRNG/scripts/convert_ent.py b/sources/apps/QUAC-TRNG/scripts/convert_ent.py new file mode 100644 index 0000000..b2c66ea --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/convert_ent.py @@ -0,0 +1,32 @@ +import os +import argparse +from numpy import array + +parser = argparse.ArgumentParser(description = "Run this script to preprocess entropy files") +parser.add_argument('indir', help="Path to entropy files") +args = parser.parse_args() + +files = [] +for fname in os.listdir(args.indir): + files.append(fname) + +for file in files: + print("Processing:", file) + # Add this placement + fpath = os.path.normpath(args.indir + '/' + file) + infile = open(fpath,'r') + wfile = file.split('.')[0] + '_floats.bin' + wfpath = os.path.normpath(args.indir + '/' + wfile) + output_file = open(wfpath, 'wb') + + lines = infile.readlines() + + # First row reserved for metadata + for lidx in range(1,len(lines)): + line = lines[lidx].strip() + split = line.split() + elements = [float(i) for i in split[1:]] + a = array(elements,'float32') + a.tofile(output_file) + + output_file.close()
\ No newline at end of file diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/README.md b/sources/apps/QUAC-TRNG/scripts/ft200_rc/README.md new file mode 100644 index 0000000..de91b27 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/README.md @@ -0,0 +1,21 @@ +# FT200 remote read and write + +Make sure that the FT200 controller match the parameters you are using in your script. +The relevant parameters are: + +PASS-0101 menu (see documentation in doc folder): +* ldN0 (Device address) +* bAUd (rate) +* UCR (parity) + +## Requirements +* pip3 install -U minimalmodbus + + +## Add your user to the dialout group + +sudo gpasswd --add ${USER} dialout +(You need to log out and log back in again for it to be effective) + +## Execute test +* python3 ft200.py diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/__init__.py b/sources/apps/QUAC-TRNG/scripts/ft200_rc/__init__.py new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/__init__.py diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/1189902393_ft200_lcd.pdf b/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/1189902393_ft200_lcd.pdf Binary files differnew file mode 100644 index 0000000..8c18eaa --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/1189902393_ft200_lcd.pdf diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/210.pdf b/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/210.pdf Binary files differnew file mode 100644 index 0000000..e4d0d39 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/docs/210.pdf diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/ft200.py b/sources/apps/QUAC-TRNG/scripts/ft200_rc/ft200.py new file mode 100644 index 0000000..aa50a1f --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/ft200.py @@ -0,0 +1,182 @@ +""""" +Code for monitoring and controlling temperature in the Maxwell FT200 controller using the Modbus RS485 protocol +@author: Lois Orosa +""""" +import minimalmodbus as mm, serial, time +import time +import os +import sys +import pathlib + +class FT200: + """ + Controls and monitors the Maxwell FT200 tempoerature controller + """ + READ_ERROR = -999 + WRITE_ERROR = -998 + def __init__(self, _baud=9600, _ft200_addr=1, _temp_tolerance=2): + self.baud = _baud + self.write_reg = 5 # Write register + self.read_reg = 0 # Read register + self.ft200_addr = _ft200_addr # Address of the device (by default is 1) + self.temp_tolerance = _temp_tolerance # Temperature tolerance. By default, we consider that we reach the target value if the measured value is within +-0.2C + + try: + self.device = self.findDevice() # Get the USB-RS485 device + print("USB-RS485 Device: "+str(self.device)) + self.instrument = mm.Instrument(self.device , 3 , debug = False) + self.instrument.serial.baudrate = self.baud + self.instrument.address = self.ft200_addr + self.instrument.serial.timeout = 1 + self.instrument.serial.parity = serial.PARITY_NONE + self.instrument.mode = mm.MODE_RTU + print("Device "+self.device+": conected") + except: + raise + + # Automatically find the USB-RS485 Device + def findDevice(self): + ''' + Find the USB-RS485 device + This function might raise two exceptions: + -> NameError: There is more than 1 USB-RS485 device, or other USB device has a similar name as the USB-RS485 device + -> IOError: Device not found. + ''' + DEVICENAME = "1a86_USB" + name = '' + while True: + path = pathlib.Path(__file__).parent.absolute() + f = os.popen(str(path)+"/usb_dev_path.sh") + for device in f: + if DEVICENAME in device: + if name == '': + name = device.split()[0] + else: + raise NameError("There are at least two USB devices whose name contains "+str(DEVICENAME)) + if name == '': + raise IOError("Device not found.") + return name + + # Read Temperature from FT200 + # The output value has 1 decimal, and it is multiplied by 10 + # E.g., data= 102 represents a temperature of 10.2C + def readTemp(self): + try: + data = self.instrument.read_register(self.read_reg, functioncode=3) + return data + except IOError: + print("Failed to read from instrument") + return self.READ_ERROR + + # Check if the temperature reached a value + # The input value has 1 decimal, and it is multiplied by 10 + # E.g., data= 102 represents a temperature of 10.2C + def isTemp(self, temperature): + try: + data = self.instrument.read_register(self.read_reg, functioncode=3) + if data >= (temperature-self.temp_tolerance) and data <= (temperature+self.temp_tolerance): + return True + else: + return False + except IOError: + print("Failed to read from instrument") + return False + + # Set Temperature FT200 + # The value should be a integer, and it should be the temperature value with 1 decimal multiplied by 10 + # E.g., value= 102 represents a temperature of 10.2C + def setTemp(self, value): + try: + self.instrument.write_register(self.write_reg, value, functioncode=6) + except IOError: + print("Failed to write to instrument") + return self.WRITE_ERROR + return 0 + + # Set the temperature and wait for the temperature to be stable + # Optimized version that reach the target temperature earlier + # The value should be a integer, and it should be the temperature value with 1 decimal multiplied by 10 + # E.g., value= 102 represents a temperature of 10.2C + def autoSetAndWait(self, temperature, debug = False): + acceleration_factor = 5 + stability = 10 # Iterations that the temperature needs to be stable + time_sleep = 1 # 1 second wait per loop iteration + stable_count = stability # Counter for taking into account the stability of the temperature + stable_margin = 20 # We do not change the set temperature if we are within 2C from the target + optTemp = temperature + self.setTemp(optTemp) # Set the temperature + curTemp = self.readTemp() # Read the temperature + if debug: + print("[TEMPERATURE] current: "+str(curTemp/10.0)+", goal: "+str(temperature/10.0)+", programmed: "+str(optTemp/10.0)) + while (((curTemp > (temperature + self.temp_tolerance)) or (curTemp < (temperature - self.temp_tolerance))) or (stable_count!=0)): + if debug: + print("[TEMPERATURE] current: "+str(curTemp/10.0)+", goal: "+str(temperature/10.0)+", programmed: "+str(optTemp/10.0)+", stable_count: "+str(stable_count)) + prev_optTemp = optTemp + if (temperature - curTemp) > 0: #(stable_margin): # If we are far away more than 2C + optTemp = temperature + (temperature - curTemp)*acceleration_factor # Optimization to accelerate convergence + else: + optTemp = temperature + if optTemp != prev_optTemp: + self.setTemp(optTemp) # Set temperature only if different from the previous iteration + + time.sleep(time_sleep) # WAit for a while + + curTemp = self.readTemp() # Read the temperature + while (curTemp == self.READ_ERROR): # If there is an error, read again + curTemp = self.readTemp() + # Five iterations in the range + if (curTemp > (temperature + self.temp_tolerance)) or (curTemp < (temperature - self.temp_tolerance)): + stable_count = stability + else: + stable_count -= 1 # The temperature has to remain stable for a while + + def autoSetAndWait_2(self, temperature, debug = False): + acceleration_factor = 150 + stability = 10 # Iterations that the temperature needs to be stable + time_sleep = 1 # 1 second wait per loop iteration + stable_count = stability # Counter for taking into account the stability of the temperature + stable_margin = 5 # We do not change the set temperature if we are within 2C from the target + optTemp = temperature + self.setTemp(optTemp) # Set the temperature + curTemp = self.readTemp() # Read the temperature + if debug: + print("[TEMPERATURE] current: "+str(curTemp/10.0)+", goal: "+str(temperature/10.0)+", programmed: "+str(optTemp/10.0)) + while (((curTemp > (temperature + self.temp_tolerance)) or (curTemp < (temperature - self.temp_tolerance))) or (stable_count!=0)): + if debug: + print("[TEMPERATURE] current: "+str(curTemp/10.0)+", goal: "+str(temperature/10.0)+", programmed: "+str(optTemp/10.0)+", stable_count: "+str(stable_count)) + prev_optTemp = optTemp + if abs(temperature - curTemp) > (stable_margin): # If we are far away more than 2C + if temperature > curTemp: + optTemp = temperature + acceleration_factor # Optimization to accelerate convergence + else: + optTemp = temperature - acceleration_factor # Optimization to accelerate convergence + else: + optTemp = temperature + + time.sleep(time_sleep) # WAit for a while + + curTemp = self.readTemp() # Read the temperature + while (curTemp == self.READ_ERROR): # If there is an error, read again + curTemp = self.readTemp() + # Five iterations in the range + if (curTemp > (temperature + self.temp_tolerance)) or (curTemp < (temperature - self.temp_tolerance)): + stable_count = stability + else: + stable_count -= 1 # The temperature has to remain stable for a while + +if __name__ == '__main__': + ## Example of how to use the FT200 class + try: + tc = FT200() + except Exception as error: + print("[ERROR] "+ repr(error)) + sys.exit(0) + + # Temperature + value = int(sys.argv[1]) * 10 + print(value) + print(tc.readTemp()) + tc.autoSetAndWait(value,True) + + + diff --git a/sources/apps/QUAC-TRNG/scripts/ft200_rc/usb_dev_path.sh b/sources/apps/QUAC-TRNG/scripts/ft200_rc/usb_dev_path.sh new file mode 100755 index 0000000..241aef1 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/ft200_rc/usb_dev_path.sh @@ -0,0 +1,13 @@ +#!/bin/bash +# https://unix.stackexchange.com/questions/144029/command-to-determine-ports-of-a-device-like-dev-ttyusb0 +echo "USB devices:" +for sysdevpath in $(find /sys/bus/usb/devices/usb*/ -name dev); do + ( + syspath="${sysdevpath%/dev}" + devname="$(udevadm info -q name -p $syspath)" +[[ "$devname" == "bus/"* ]] && exit + eval "$(udevadm info -q property --export -p $syspath)" +[[ -z "$ID_SERIAL" ]] && exit + echo "/dev/$devname - $ID_SERIAL" + ) +done diff --git a/sources/apps/QUAC-TRNG/scripts/run_ent.py b/sources/apps/QUAC-TRNG/scripts/run_ent.py new file mode 100644 index 0000000..5163c5f --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/run_ent.py @@ -0,0 +1,43 @@ +import os +import argparse +import subprocess +import time + +parser = argparse.ArgumentParser(description = "Collects entropy for different data patterns") + +parser.add_argument('dimm', help="DIMM label under test") +parser.add_argument('temperature', help="DIMM temperature in centigrades") +parser.add_argument('iters', help="How many times do we QUAC to find entropy") +parser.add_argument('stride', help="Stride (in rows) used to skim through segments, default: 4") + +args = parser.parse_args() + +# ---------------- SETUP DIRECTORIES ------------------- + +if not os.path.isdir("data"): + os.mkdir("data") + +if not os.path.isdir("data/" + args.dimm): + os.mkdir("data/" + args.dimm) + +if not os.path.isdir("data/" + args.dimm + "/ent"): + os.mkdir("data/" + args.dimm + "/ent") + +if not os.path.isdir("data/" + args.dimm + "/ent/" + args.temperature + "C"): + os.mkdir("data/" + args.dimm + "/ent/" + args.temperature + "C") + +outdir = os.path.normpath("data/" + args.dimm + "/ent/" + args.temperature + "C") + +# -------------------- RUN TESTS ----------------------- + +placements = ["1000", "1001", "1010", "1011", "1100" , "1101", "1110", "1111", \ + "0111", "0110", "0101", "0100", "0011" , "0010", "0001", "0000"] + +ent_path = os.path.normpath("bin/ENT") +results_path = outdir +stride = args.stride +iters = args.iters + +for placement in placements: + print("sudo " + ent_path + " " + results_path + " " + placement + " " + stride + " " + iters) + os.system("sudo " + ent_path + " " + results_path + " " + placement + " " + stride + " " + iters) diff --git a/sources/apps/QUAC-TRNG/scripts/run_vn_nist.py b/sources/apps/QUAC-TRNG/scripts/run_vn_nist.py new file mode 100644 index 0000000..c439da8 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/run_vn_nist.py @@ -0,0 +1,125 @@ +import os +import argparse +import subprocess +import time + +parser = argparse.ArgumentParser(description = "Run this script to evaluate bitstreams") + +parser.add_argument('dimm', help="DIMM label under test") +parser.add_argument('temperature', help="DIMM temperature in centigrades") + +args = parser.parse_args() + +# ---------------- SETUP DIRECTORIES ------------------- + +if not os.path.isdir("data"): + os.mkdir("data") + +if not os.path.isdir("data/" + args.dimm): + os.mkdir("data/" + args.dimm) + +if not os.path.isdir("data/" + args.dimm + "/vnc"): + os.mkdir("data/" + args.dimm + "/vnc") + +if not os.path.isdir("data/" + args.dimm + "/vnc/" + args.temperature + "C"): + os.mkdir("data/" + args.dimm + "/vnc/" + args.temperature + "C") + +if not os.path.isdir("data/" + args.dimm + "/vnc/" + + args.temperature + "C/nist_results"): + os.mkdir("data/" + args.dimm + "/vnc/" + args.temperature + "C/nist_results") + +if not os.path.isdir("data/" + args.dimm + "/vnc/" + + args.temperature + "C/binaries"): + os.mkdir("data/" + args.dimm + "/vnc/" + args.temperature + "C/binaries") + +outdir = os.path.normpath("data/" + args.dimm + "/vnc/" + args.temperature + "C/nist_results") +indir = os.path.normpath("data/" + args.dimm + "/vnc/" + args.temperature + "C/binaries") + +if not os.path.isfile(indir + "/lastrow.txt"): + input("I am going to overwrite 'lastrow.txt', proceed with caution") + init_file = open(indir + "/lastrow.txt", "w") + init_file.write(str(0) + "\n") + +# Hard-reset SoftMC before other experiments +reset_path = os.path.normpath("bin/SoftMC_reset") +reset = subprocess.Popen("exec sudo " + reset_path, shell=True, preexec_fn=os.setpgrp) +time.sleep(5) +os.system("sudo pkill -9 -P" + str(os.getpgid(reset.pid))) +reset.wait() + +# First run OBSERVE to collect von Neumann corrected bitstreams +observe_path = os.path.normpath("bin/OBSERVE " + indir) +observe_args = " > " + os.path.normpath(outdir + "/output.log") +observe = subprocess.Popen("exec sudo " + observe_path + observe_args, shell=True, preexec_fn=os.setpgrp) + +# Watch the bitstream directory for new files +before = ["lastrow.txt"] + +while True: + time.sleep(1) + after = os.listdir(indir) + s = set(before) + diff = [x for x in after if x not in s] + before = after + if len(diff) == 0: + continue + + bitstream = [[] for i in range(8)] # at most 8 STS runs + + # Partition + i = 0 + for fname in diff: + bitstream[i].append(fname) + i = (i+1) % 8 + + found = False + for i in range(len(bitstream[0])): + print("Finished " + str(i) + " out of " + str(len(bitstream[7]))) + processes = [] + for thr in range(0,8): + if len(bitstream[thr]) <= i: + break + command_output = os.path.normpath(outdir + "/thr" + str(thr)) + command_input = os.path.normpath(indir + '/' + bitstream[thr][i]) + command = 'tools/sts -O -i 1 -I 1 -S '+ str(1024*1024) +' -w ' + command_output +' -F r ' + command_input + " 2>/dev/null" + process = subprocess.Popen(command, shell=True) + processes.append(process) + + # Collect statuses + output = [p.wait() for p in processes] + + for thr in range(0,8): + if len(bitstream[thr]) <= i: + break + res_txt = open(os.path.normpath(outdir + '/thr' + str(thr) + '/finalAnalysisReport.txt'), "r") + lines = res_txt.readlines() + n_success = 0 + for j in range(7, len(lines)): + line = lines[j].strip() + if not ("1" in line or "0" in line): + break + split = line.split() + if "/" in split[11]: + n_success += int(split[11].split("/")[0]) + res_txt.close() + print(bitstream[thr][i] + " succeeded in " + str(n_success) + "tests + bits: " + str(1024*1024)) + if n_success < 187: + os.system("rm -f " + indir + "/" + bitstream[thr][i]) + if n_success > 186: + os.system("mkdir " + outdir + "/" + bitstream[thr][i]) + os.system("cp " + outdir + '/thr' + str(thr) + "/finalAnalysisReport.txt " + outdir + "/" + bitstream[thr][i]) + if n_success == 188: + os.system("sudo pkill -9 -P" + str(os.getpgid(observe.pid))) + observe.wait() # wait until the process actually terminates + found = True + if found: + break + +print("Finally found one cell that passes all NIST tests!") + +# Reset SoftMC because last process was killed prematurely +reset_path = os.path.normpath("bin/SoftMC_reset") +reset = subprocess.Popen("exec sudo " + observe_path, shell=True, preexec_fn=os.setpgrp) +time.sleep(5) +os.system("sudo pkill -9 -P" + str(os.getpgid(reset.pid))) +reset.wait() diff --git a/sources/apps/QUAC-TRNG/scripts/sha_bitstreams.py b/sources/apps/QUAC-TRNG/scripts/sha_bitstreams.py new file mode 100644 index 0000000..fd4dbfb --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/sha_bitstreams.py @@ -0,0 +1,87 @@ +import os +import argparse +import subprocess +import time + +parser = argparse.ArgumentParser(description = "Collects sha bitstreams for a data pattern") + +parser.add_argument('dimm', help="DIMM label under test") +parser.add_argument('pattern', help="data pattern to use") +parser.add_argument('temperature', help="DIMM temperature in centigrades") + +args = parser.parse_args() + +bitstreams_available = False + +if not os.path.isdir("data/" + args.dimm + "/sha"): + os.mkdir("data/" + args.dimm + "/sha") + +if os.path.isdir("data/" + args.dimm + "/sha/" + args.temperature + "C"): + bitstreams_available = True + +if not os.path.isdir("data/" + args.dimm + "/sha/" + args.temperature + "C"): + os.mkdir("data/" + args.dimm + "/sha/" + args.temperature + "C") + +infile = os.path.normpath("data/" + args.dimm + "/ent/" + args.temperature + "C/top.txt") +outdir = os.path.normpath("data/" + args.dimm + "/sha/" + args.temperature + "C") + +if not os.path.isdir("data/" + args.dimm + "/sha"): + os.mkdir("data/" + args.dimm + "/sha_nist_results") + +if not os.path.isdir("data/" + args.dimm + "/sha/" + args.temperature + "C"): + os.mkdir("data/" + args.dimm + "/sha_nist_results/" + args.temperature + "C") + +sha_output = os.path.normpath("data/" + args.dimm + "/sha_nist_results/" + args.temperature + "C") + +if not bitstreams_available: + sha_path = "bin/SHA" + print("sudo " + sha_path + " " + outdir + " " + infile) + os.system("sudo " + sha_path + " " + outdir + " " + infile) +else: + print("I think we have the bitstreams available, I won't run softmc code again") + +i=0 +bitstream = [] + +for j in range(0,10): + bitstream.append([]) + +for fname in os.listdir(outdir): + bitstream[i].append(fname) + i+=1 + i = i % 10 + +print(bitstream) + +for i in range(len(bitstream[9])): + print("Finished " + str(i) + " out of " + str(len(bitstream[9]))) + processes = [] + for thr in range(0,10): + outpath = os.path.normpath(sha_output+'/thr'+str(thr)) + if not os.path.isdir(outpath): + os.mkdir(outpath) + command = 'tools/sts -O -i 1024 -I 1 -P 11=0.001 -S '+ str(1024*1024) +' -w '+ outpath +' -F r ' + outdir+'/'+bitstream[thr][i] + " 2>/dev/null" + process = subprocess.Popen(command, shell=True) + processes.append(process) + print(command) + + # Collect statuses + output = [p.wait() for p in processes] + + for thr in range(0,10): + outpath = os.path.normpath(sha_output+'/thr'+str(thr)) + res_txt = open(outpath + '/finalAnalysisReport.txt', "r") + lines = res_txt.readlines() + n_success = 0 + for j in range(7, len(lines)): + line = lines[j].strip() + if not ("1" in line or "0" in line): + break + split = line.split() + if "/" in split[11]: + n_success += int(split[11].split("/")[0]) + res_txt.close() + print(bitstream[thr][i] + " succeeded in " + str(n_success) + "tests + bits: " + str(1024*1024)) + if n_success > 186: + os.system("mkdir "+sha_output+"/" + bitstream[thr][i]) + os.system("cp "+sha_output+'/thr'+str(thr)+"/finalAnalysisReport.txt "+sha_output+"/" + bitstream[thr][i]) diff --git a/sources/apps/QUAC-TRNG/scripts/top_ents.py b/sources/apps/QUAC-TRNG/scripts/top_ents.py new file mode 100644 index 0000000..2a74569 --- /dev/null +++ b/sources/apps/QUAC-TRNG/scripts/top_ents.py @@ -0,0 +1,35 @@ +import os +import argparse +import pandas as pd +import numpy as np + +parser = argparse.ArgumentParser(description = "Postprocess entropy files and obtain top-10 segments with the highest entropy") +parser.add_argument('dimm', help="DIMM label") +parser.add_argument('pattern', help="4-bit data pattern [1000, 1001, 0111, etc]") +parser.add_argument('temp', help="temperature in centigrades") +args = parser.parse_args() + +dimm = args.dimm +pattern = args.pattern +temp = args.temp + +no_segments = 8192 +no_bitlines = 512 * 128 + +fn = "data/" + dimm + "/ent/" + temp + "C/1_1000_128_32768_" + pattern + "_floats.bin" +dirname = "data/" + dimm + "/ent/" + temp + "C/" + +arr = np.fromfile(fn, dtype=np.float32) +arr = arr.reshape(no_segments, no_bitlines) +arr = arr.sum(axis = 1) +indices = arr.argsort()[-10:][::-1] +print(indices) +print(arr[indices]) + +log = open(dirname + "top.txt", "w") +log.write("10\n") +for index in indices: + # bank is always 1 + # multiply segment # by 4 to get starting row addr + log.write("1 " + str(int(index) * 4) + " " + str(arr[index]) + " " + pattern + "\n") +log.close() diff --git a/sources/apps/QUAC-TRNG/src/observe.cpp b/sources/apps/QUAC-TRNG/src/observe.cpp new file mode 100644 index 0000000..275c516 --- /dev/null +++ b/sources/apps/QUAC-TRNG/src/observe.cpp @@ -0,0 +1,726 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> +#include <algorithm> +#include <iterator> +#include <iomanip> + +#define NEWFILTER + +using namespace std; + +#define NO_BANKS 2 +#define NO_ROWS 32*1024 + +// Stride register ids are fixed and should not be changed +// CASR should always be reg 0 +#define CASR 0 +// BASR should always be reg 1 +#define BASR 1 +// RASR should always be reg 2 +#define RASR 2 + +#define PATTERN_REG 12 +#define INV_PATTERN_REG 13 +#define TEMP_PATTERN_REG 14 + +#define BAR 3 +#define CAR 4 +#define RARBASE 5 +#define RAR1 6 +#define RAR2 9 + +#define ONE_REG 10 +#define ZERO_REG 11 + +#define ITER_REG 7 +#define CTR_REG 8 + +#define INIT_LOOP 15 + +Program genRowCopy(int t1, int t2, int row1_reg, int row2_reg, int bank_reg) +{ + Program ret; + // t1 -> ACT to PRE, t2 -> PRE to 2nd ACT + int sz = 3 + t1 + t2; + sz = (4-(sz%4)) + sz; + Mininst buff[sz]; + for(int i = 0 ; i < sz ; i++) + buff[i] = SMC_NOP(); + + // Overwrite with the actual sequence + buff[0] = SMC_ACT(bank_reg, 0, row1_reg, 0); + buff[1+t1] = SMC_PRE(bank_reg, 0, 0); + buff[2+t1+t2] = SMC_ACT(bank_reg, 0, row2_reg, 0); + + for(int i = 0 ; i < sz ; i+=4) + ret.add_inst(buff[i], buff[i+1], buff[i+2], buff[i+3]); + + return ret; +} + +/** + * To send an ACT -> SLEEP t1 -> PRE -> SLEEP t2 -> ACT sequence + * This seq. will use BAR RAR1 and RAR2 registers + */ +Program genActPreActSequence(int t1, int t2, int row1_reg, int row2_reg, int bank_reg) +{ + Program ret; + // t1 -> ACT to PRE, t2 -> PRE to 2nd ACT + int sz = 3 + t1 + t2; + sz = (4-(sz%4)) + sz; + Mininst buff[sz]; + for(int i = 0 ; i < sz ; i++) + buff[i] = SMC_NOP(); + + // Overwrite with the actual sequence + buff[0] = SMC_ACT(bank_reg, 0, row1_reg, 0); + buff[1+t1] = SMC_PRE(bank_reg, 0, 0); + buff[2+t1+t2] = SMC_ACT(bank_reg, 0, row2_reg, 0); + + for(int i = 0 ; i < sz ; i+=4) + ret.add_inst(buff[i], buff[i+1], buff[i+2], buff[i+3]); + + return ret; +} + + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ + +Program genCopyRange(int row_reg, int no_rows, int bank_reg, bool stripes) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_inst(SMC_LI(0, ONE_REG)); + ret.add_inst(SMC_LI(1, ZERO_REG)); + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_below(genRowCopy(4,4,ONE_REG,row_reg,bank_reg)); + ret.add_inst(SMC_SLEEP(6)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_ADDI(row_reg, 1, row_reg)); + + ret.add_inst(SMC_SLEEP(6)); + ret.add_below(genRowCopy(4,4,ZERO_REG,row_reg,bank_reg)); + ret.add_inst(SMC_SLEEP(6)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_ADDI(row_reg, 1, row_reg)); + + ret.add_inst(SMC_SLEEP(6)); + + ret.add_below(genRowCopy(4,4,ONE_REG,row_reg,bank_reg)); + ret.add_inst(SMC_SLEEP(6)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_ADDI(row_reg, 1, row_reg)); + ret.add_inst(SMC_SLEEP(6)); + + ret.add_below(genRowCopy(4,4,ZERO_REG,row_reg,bank_reg)); + ret.add_inst(SMC_SLEEP(6)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_ADDI(row_reg, 1, row_reg)); + ret.add_inst(SMC_SLEEP(6)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1000(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ + +Program genWriteRange(int row_reg, int no_rows, int bank_reg, bool stripes) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + for(int i = 0 ; i < 64 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + if(stripes) + { + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + } + ret.add_branch(ret.BR_TYPE::BL, row_reg, INIT_LOOP, "genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +Program genReadRange(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // read until this row + ret.add_label("genReadRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); // TODO: fix this + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + for(int i = 0 ; i < 64 ; i++) + { + ret.add_inst(SMC_READ(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_READ(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_branch(ret.BR_TYPE::BL, row_reg, INIT_LOOP, "genReadRange:LOOP_BEGIN"); + /* + ret.add_inst(SMC_SLEEP(5)); + ret.dump_registers(); + ret.add_inst(SMC_END()); + */ + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} +/** + * @return an instruction formed by NOPs + */ +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +Program gen_test_prog(int bank, int r1, int r2,int iters) +{ + Program program; + + program.add_inst(SMC_LI(8, CASR)); // Load 8 into CASR since each READ reads 8 columns + program.add_inst(SMC_LI(1, BASR)); // Load 1 into BASR + program.add_inst(SMC_LI(1, RASR)); // Load 1 into RASR + + program.add_inst(SMC_LI(0xffffffff, PATTERN_REG)); + program.add_inst(SMC_LI(0x00000000, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + program.add_inst(SMC_LI(bank, BAR)); + program.add_inst(SMC_LI(r1-(r1%4), RARBASE)); + + program.add_inst(SMC_LI(iters, ITER_REG)); + program.add_inst(SMC_LI(0, CTR_REG)); + + program.add_label("WRITE_BEGIN"); + program.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + program.add_inst(SMC_SLEEP(3)); + program.add_below(genWriteRange(RARBASE, 4, BAR, true)); + //program.add_below(genWriteRange1000(RARBASE, 4, BAR)); + //program.add_below(genCopyRange(RARBASE, 4, BAR, true)); + program.add_inst(SMC_LI(r1, RAR1)); + program.add_inst(SMC_LI(r2, RAR2)); + + program.add_below(genActPreActSequence(1, 1, RAR1, RAR2, BAR)); + + program.add_inst(SMC_SLEEP(6)); + program.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + program.add_below(genReadRange(RARBASE, 1, BAR)); + program.add_inst(SMC_ADDI(CTR_REG, 1, CTR_REG)); + + program.add_branch(program.BR_TYPE::BL, CTR_REG, ITER_REG, "WRITE_BEGIN"); + program.add_inst(SMC_END()); + + return program; +} + +bool nonoverlapping_pattern(std::vector<uint8_t> bv) +{ + unsigned wsize = bv.size()/8; + uint8_t buf[wsize]; + for(int j = 0 ; j < wsize ; j++) + { + uint8_t intermediate = 0; + for(int k = 0 ; k < 8 ; k++) + intermediate = (intermediate << 1) | bv[j*8 + k]; + buf[j] = intermediate; + } + + int bin[8] = {0}; + for (int i = 0 ; i < (1024*128*8 - 3) ; i += 3) + { + if (i % 8 < 6) + bin[(buf[i/8] >> (i%8)) & 0x7]++; + else + { + int next_bits = 3 - (8-(i%8)); + int next_mask = next_bits == 2 ? 0x3 : 0x1; + int this_mask = next_bits == 2 ? 0x1 : 0x3; + uint8_t next_pattern = (buf[(i/8)+1]) & next_mask; + uint8_t this_pattern = (buf[i/8] >> (i%8)) & this_mask; + bin[(next_pattern << (3-next_bits)) | this_pattern]++; + } + } + + bool pass = true; + for (int i = 0 ; i < 8 ; i++) + { + cout << i << ": " << bin[i] << endl; + if (bin[i] < 30000 || bin[i] > 55000) + pass = false; + } + return pass; +} + +vector<int> pre_filter(SoftMCPlatform& platform, int bank, int row) +{ + vector<int> cells_to_track; + + const int NO_CELLS = 64 * 1024; + const int pre_pre_filter_iterations = 128*1024; + const int threshold_size = 4096; + + Program prog = gen_test_prog(bank, row, row+3, pre_pre_filter_iterations); + platform.execute(prog); + uint8_t buf1[8*1024]; + uint8_t buf2[8*1024]; + + vector<vector<uint8_t>> vnc_bitstreams; + + for (int i = 0 ; i < NO_CELLS ; i++) + vnc_bitstreams.push_back(vector<uint8_t>()); + + // Von neumann corrector first + for (int i = 0 ; i < pre_pre_filter_iterations/2 ; i++) + { + platform.receiveData((char*)buf1, 8*1024); + platform.receiveData((char*)buf2, 8*1024); + + for (int b = 0 ; b < NO_CELLS ; b++) + { + if(vnc_bitstreams[b].size() == threshold_size) + continue; + + int b1 = (buf1[b/8] >> b%8) & 0x1; + int b2 = (buf2[b/8] >> b%8) & 0x1; + + if (~b1 & b2) + vnc_bitstreams[b].push_back(1); + else if (b1 & ~b2) + vnc_bitstreams[b].push_back(0); + } + } + + int cnt = 0; + for (int b = 0 ; b < NO_CELLS ; b++) + { + if (vnc_bitstreams[b].size() == threshold_size) + cnt++; + } + + cout << cnt << " cells have passed initial filtering." << endl; + + // counter bins for each bitline + int bin[NO_CELLS][8]; + for (int i = 0 ; i < NO_CELLS ; i++) + { + for (int j = 0 ; j < 8 ; j++) + bin[i][j] = 0; + } + + int freq[NO_CELLS] = {0}; + + for (int b = 0 ; b < NO_CELLS ; b++) + { + if (vnc_bitstreams[b].size() < threshold_size - 1) + continue; + for (int i = 0 ; i < (threshold_size - 1)/3 ; i++) + { + int b1 = vnc_bitstreams[b][i] & 0x1; + int b2 = vnc_bitstreams[b][i+1] & 0x1; + int b3 = vnc_bitstreams[b][i+2] & 0x1; + + if (b1) + freq[b]++; + if (b2) + freq[b]++; + if (b3) + freq[b]++; + + bin[b][(b3 << 2) | (b2 << 1) | b1]++; + } + } + + bool passed_frequency[NO_CELLS] = {false}; + + int upper_freq = threshold_size/2 + threshold_size/2/100; + int lower_freq = threshold_size/2 - threshold_size/2/100; + for (int i = 0 ; i < NO_CELLS ; i++) + { + if (vnc_bitstreams[i].size() < threshold_size - 1) + continue; + if (freq[i] < upper_freq && freq[i] > lower_freq) + passed_frequency[i] = true; + } + + cnt = 0; + for (int b = 0 ; b < NO_CELLS ; b++) + { + if (passed_frequency[b]) + cnt++; + } + + cout << cnt << " cells have passed frequency filtering." << endl; + + //cout << "Cell: " << i << endl; + int lower_limit = ((threshold_size - 1) / 3 / 8) - (((threshold_size - 1) / 3 / 8) / 10); + int upper_limit = ((threshold_size - 1) / 3 / 8) + (((threshold_size - 1) / 3 / 8) / 10); + cout << "UL: " << upper_limit << " LL: " << lower_limit << endl; + bool passed[NO_CELLS]; + + for (int b = 0 ; b < NO_CELLS ; b++) + passed[b] = true; + + for (int i = 0 ; i < NO_CELLS ; i++) + { + if (!passed_frequency[i]) + { + passed[i] = false; + continue; + } + for (int j = 0 ; j < 8 ; j++) + { + //cout << "Bin" << j << ": " << bin[i][j] << endl; + if (bin[i][j] < lower_limit || bin[i][j] > upper_limit) + { + //cout << "Cell " << i << " Bin" << j << ": " << bin[i][j] << endl; + passed[i] = false; + break; + } + } + if (passed[i]) + cout << i << " " << "PASSed" << endl; + } + + for (int i = 0 ; i < NO_CELLS ; i++) + { + if (passed[i]) + cells_to_track.push_back(i); + } + + return cells_to_track; +} + +int main(int argc, char *argv[]) +{ + SoftMCPlatform platform; + int err; + + // buffer allocated for reading data from the board + uint8_t buf[8*1024]; + uint8_t prev_read[8*1024]; + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + platform.set_aref(true); + + ifstream lfFile; + string trackFn = string(argv[1]) + "/" + "lastrow.txt"; + lfFile.open(trackFn); + + int x=0; + lfFile >> x; + + lfFile.close(); + + for(int bank = 1 ; bank < NO_BANKS ; bank++) + { + for(int r1 = x ; r1 < NO_ROWS ; r1 += 4) + { + bool satisfied = false; + #ifdef NEWFILTER + vector<int> cell_pos = pre_filter(platform, bank, r1); + #else + // Enrollment phase: + // Look at switching activity of each cell in a row + Program prog = gen_test_prog(bank, r1, r1+3, 32*1024); + platform.execute(prog); + int n_switches[64*1024]; // how many times each cell have switched + fill(n_switches, n_switches + 64*1024, 0); + platform.receiveData((char*)prev_read, 8*1024); + for(int k = 0; k < 32*1024-1 ; k++) + { + platform.receiveData((char*)buf, 8*1024); // read one segment each iteration + for(int i = 0 ; i < 8 * 1024 ; i++) + { + uint8_t diff = prev_read[i] ^ buf[i]; + for(int j = 0 ; j < 8 ; j++) + n_switches[i*8 + j] += (diff >> j) & 0x1; + } + copy(begin(buf), end(buf), begin(prev_read)); + } + vector<int> cell_pos; // Switching cell positions + float total_bps = 0; + for(int i = 0 ; i < 8*8*1024 ; i++) + { + // In total it takes 132ms to execute all the DRAM commands + // We compute bps based on switching + total_bps += (1000/132.f) * n_switches[i]; + if(n_switches[i] > 1000) + cell_pos.push_back(i); + } + + cout << "Finished enrollment for bank " << bank << + " row " << r1 << "... " << cell_pos.size() << " cells passed... Predicted bps " + << fixed << setw(11) << setprecision(6) << total_bps << endl; + #endif + + cout << "Finished enrollment for bank " << bank << + " row " << r1 << "... " << cell_pos.size() << " cells passed..." << endl; + + if(cell_pos.size() == 0) + continue; + + for (int i = 0 ; i < cell_pos.size() ; i++) + cout << cell_pos[i] << " "; + cout << endl; + + // Begin filtered acquisition phase + vector<vector<uint8_t>> stream_per_cell; + for(int i = 0 ; i < cell_pos.size() ; i++) + { + stream_per_cell.push_back(vector<uint8_t>()); + } + + int prev_bpc_i = 0; + int prev_cell_flips[cell_pos.size()]{0}; + + int loop_count = 0; + int loop_limit = 15; + bool bcf = false; // has the best cell exceeded the threshold + + while(!satisfied) + { + if(loop_count == loop_limit) + { + cout << "Exiting because we have iterated enough times" << endl; + break; + } + int iterations = 1024*1024; + Program prog = gen_test_prog(bank, r1, r1+3, iterations); + platform.execute(prog); + + for(int k = 0; k < iterations/2 ; k++) + { + platform.receiveData((char*)prev_read, 8*1024); // read one segment each iteration + platform.receiveData((char*)buf, 8*1024); // read one segment each iteration + //VON NEUMANN CORRECTOR BELOW + for(int i = 0 ; i < cell_pos.size() ; i++) + { + uint8_t b1 = (prev_read[cell_pos[i]/8] >> (cell_pos[i]%8)) & 0x1; + uint8_t b2 = (buf[cell_pos[i]/8] >> (cell_pos[i]%8)) & 0x1; + uint8_t b3 = b1 ^ b2; + if(!b3) continue; + if(b2) stream_per_cell[i].push_back(1); + else stream_per_cell[i].push_back(0); + } + } + + int bpc_flips = 0; // best performing cell flip count + int bpc_i = 0; // best performing cell index + for(int i = 0 ; i < cell_pos.size() ; i++) + { + if(stream_per_cell[i].size() > bpc_flips) + { + bpc_flips = stream_per_cell[i].size(); + bpc_i = i; + } + } + + /* + cout << "Dumping # of switches per cell..." << endl; + for(int i = 0 ; i < cell_pos.size() ; i++) + { + cout << "Cell#" << cell_pos[i] << " Got:" << stream_per_cell[i].size() - prev_cell_flips[i] + << "Exp:" << 32*n_switches[cell_pos[i]] << endl; + } + */ + + if(!bcf) + { + if(prev_bpc_i == bpc_i) + { + int remaining_bits = 1024*1024 - bpc_flips; + int count_diff = bpc_flips - prev_cell_flips[bpc_i]; + if(remaining_bits/count_diff + 1 > 16) + { + cout << "Exiting because it will take too long" << endl; + break; + } + if(remaining_bits < 0){ + cout << "Finished one cell, will keep iterating for 5 iterations" << endl; + bcf = true; + loop_limit = loop_count + 5; + } + else{ + cout << "It will take " << remaining_bits/count_diff + 1 << " more iterations to finish" << endl; + cout << "Based on the best cell's performance which has flipped " << bpc_flips << " times so far" << endl; + } + }else + { + cout << "Cannot approximate test time because the best performing cell changed" << endl; + cout << "Was " << cell_pos[prev_bpc_i] << " Now " << cell_pos[bpc_i] << endl; + } + } + cout << "Cells about to finish: " << endl; + for(int i = 0 ; i < cell_pos.size() ; i++){ + if(stream_per_cell[i].size() > 768*1024) + cout << cell_pos[i] << ":" << stream_per_cell[i].size() << " "; + + prev_cell_flips[i] = stream_per_cell[i].size(); + } + cout << endl; + prev_bpc_i = bpc_i; + loop_count ++; + } + + cout << "Writing bitstreams..." << endl; + int filtered_streams = 0; + for(int i = 0 ; i < cell_pos.size() ; i++){ + if(stream_per_cell[i].size() >= 1024*1024){ //&& nonoverlapping_pattern(stream_per_cell[i])){ + ofstream dump_file; + filtered_streams++; + string fn = string(argv[1]) + "/" + to_string(bank) + "_" + to_string(r1) + + "_" + to_string(r1+3) + "_" + to_string(cell_pos[i]); + dump_file.open(fn, ofstream::binary); + unsigned wsize = stream_per_cell[i].size()/8; + uint8_t wbuf[wsize]; + for(int j = 0 ; j < wsize ; j++) + { + uint8_t intermediate = 0; + for(int k = 0 ; k < 8 ; k++) + intermediate = (intermediate << 1) | stream_per_cell[i][j*8 + k]; + wbuf[j] = intermediate; + } + dump_file.write((char*)wbuf, wsize); + dump_file.close(); + } + } + //cout << "Filtered streams: " << filtered_streams << endl; + cout << "Finished processing b:" << bank << " r1:" << r1 << " r2:" << r1+3 << endl; + ofstream lfFile; + string trackFn = string(argv[1]) + "/" + "lastrow.txt"; + lfFile.open(trackFn); + lfFile << r1; + lfFile.close(); + } + } +} diff --git a/sources/apps/QUAC-TRNG/src/slow_entropy.cpp b/sources/apps/QUAC-TRNG/src/slow_entropy.cpp new file mode 100644 index 0000000..a18cf6e --- /dev/null +++ b/sources/apps/QUAC-TRNG/src/slow_entropy.cpp @@ -0,0 +1,204 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include "util.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> +#include <math.h> +#include <algorithm> +#include <iterator> +#include <iomanip> + +using namespace std; + +Program gen_test_prog(int bank, int r1, int r2, int iters, int read_cols, int placement) +{ + Program program; + + program.add_inst(SMC_LI(8, CASR)); // Load 8 into CASR since each READ reads 8 columns + program.add_inst(SMC_LI(1, RASR)); // Load 1 into RASR + + program.add_inst(SMC_LI(bank, BAR)); + program.add_inst(SMC_LI(r1, RARBASE)); + + program.add_inst(SMC_LI(iters, ITER_REG)); + program.add_inst(SMC_LI(0, CTR_REG)); + + program.add_inst(SMC_PRE(BAR,0,1),SMC_NOP(),SMC_NOP(),SMC_NOP()); + program.add_inst(SMC_SLEEP(5)); + + program.add_inst(SMC_LI(r1, RAR1)); + program.add_inst(SMC_LI(r2, RAR2)); + + program.add_label("WRITE_BEGIN"); + //program.add_below(genWriteRange(RARBASE, 4, BAR, true)); + switch(placement) + { + // Invert 0-1 start pattern before here when we set the + // WIDEREG + case 1000: + program.add_below(genWriteRange1000(RARBASE, 4, BAR)); + break; + case 1001: + program.add_below(genWriteRange1001(RARBASE, 4, BAR)); + break; + case 1010: + program.add_below(genWriteRange1010(RARBASE, 4, BAR)); + break; + case 1011: + program.add_below(genWriteRange1011(RARBASE, 4, BAR)); + break; + case 1100: + program.add_below(genWriteRange1100(RARBASE, 4, BAR)); + break; + case 1101: + program.add_below(genWriteRange1101(RARBASE, 4, BAR)); + break; + case 1110: + program.add_below(genWriteRange1110(RARBASE, 4, BAR)); + break; + case 1111: + program.add_below(genWriteRange1111(RARBASE, 4, BAR)); + break; + default: + printf("Unexpected placement encoding\n"); + exit(0); + } + + program.add_below(genActPreActSequence(1, 1, RAR1, RAR2, BAR)); + program.add_inst(SMC_SLEEP(4)); + program.add_inst(SMC_PRE(BAR, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + program.add_inst(SMC_SLEEP(4)); + program.add_below(genReadRange(RARBASE, 1, BAR, read_cols)); + + program.add_inst(SMC_ADDI(CTR_REG, 1, CTR_REG)); + program.add_branch(program.BR_TYPE::BL, CTR_REG, ITER_REG, "WRITE_BEGIN"); + program.add_inst(SMC_END()); + + return program; +} + +int main(int argc, char *argv[]) +{ + bool DUMP_RAW = false; + + SoftMCPlatform platform; + int err; + + // buffer allocated for reading data from the board + uint8_t buf[4*READ_CL*64]; + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + platform.set_aref(true); + + int bank = 1; + + int row_stride = atoi(argv[3]); + + string placement = argv[2]; + bool invert = false; + + int iters = atoi(argv[4]); + + ofstream ent_file; + string fn = string(argv[1]) + "/" + to_string(bank) + "_" + to_string(iters) + "_" + to_string(READ_CL) + "_" + to_string(NO_ROWS) + "_" + placement + ".txt"; + cout << "Saving to file " << fn << endl; + ent_file.open(fn); + + char new_placement[10]; + if (placement[0] == '0') + { + for (std::string::size_type i = 0 ; i < placement.size() ; i++) + if(placement[i] == '0') + new_placement[i] = '1'; + else if(placement[i] == '1') + new_placement[i] = '0'; + else + new_placement[i] = placement[i]; + placement = string(new_placement); + invert = true; + } + + Program program; + program.add_inst(SMC_LI(invert ? 0x0 : 0xffffffff, PATTERN_REG)); + program.add_inst(SMC_LI(invert ? 0xffffffff : 0x0, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + program.add_inst(SMC_END()); + platform.execute(program); + + unsigned char *byte_buf[READ_CL*64]; + for(int i = 0 ; i < READ_CL*64 ; i++) + byte_buf[i] = new unsigned char[iters]; + + // Frequency of 1s and 0s generated by each bitline + int alphabet[READ_CL*64*8][2]; + + ent_file << NO_ROWS/row_stride << " " << READ_CL << endl; + + for(int r1 = 0 ; r1 < NO_ROWS ; r1+=row_stride) + { + ent_file << r1 << " "; + Program prog = gen_test_prog(bank, r1, r1+3, iters, READ_CL, std::stoi(placement)); + platform.execute(prog); + + for(int i = 0 ; i < iters ; i++) + { + platform.receiveData((char*)buf, READ_CL*64); // read one segment each iteration + for(int j = 0 ; j < READ_CL*64 ; j++) + byte_buf[j][i] = buf[j]; + } + + for(int i = 0 ; i < READ_CL*64*8 ; i++) + for(int j = 0 ; j < 2 ; j++) + alphabet[i][j] = 0; + + for(int j = 0 ; j < READ_CL*64*8 ; j++) + { + for(int i = 0 ; i < iters ; i++) + alphabet[j][(byte_buf[j/8][i]>>(j%8))&0x1] += 1; + + float approxent = 0; + for(int i = 0 ; i < 2 ; i++) + { + int freq = alphabet[j][i]; + float p = ((float) freq) / ((float) iters); + approxent -= freq == 0 ? 0 : p * log2(p); + } + + ent_file << approxent << " "; + + /* + if(approxent > 0.995) + { + unsigned char bits[iters/8]{0}; + for(int i = 0 ; i < iters ; i++) + bits[i/8] |= ((byte_buf[j/8][i] >> (j%8)) & 0x1) << (7-(i%8)); + ofstream dump_file; + string fn = string(argv[2]) + "/" + to_string(r1) + "_" + to_string(j) + ".bin"; + dump_file.open(fn, ofstream::binary); + dump_file.write((char*)(bits), iters/8); + dump_file.close(); + } + */ + } + + ent_file << endl; + + printf("\rrow %d", r1); + fflush(stdout); + } + for(int i = 0 ; i < READ_CL*64 ; i++) + delete byte_buf[i]; + ent_file.close(); +} diff --git a/sources/apps/QUAC-TRNG/src/slow_sha.cpp b/sources/apps/QUAC-TRNG/src/slow_sha.cpp new file mode 100644 index 0000000..69bf72f --- /dev/null +++ b/sources/apps/QUAC-TRNG/src/slow_sha.cpp @@ -0,0 +1,233 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> +#include <algorithm> +#include <iterator> +#include <iomanip> +#include <openssl/sha.h> +#include "util.h" + +using namespace std; + +// each iter generates 320 bytes of RN +// #define ITERS_SHA 450000 +#define READ_CL 128 + +// Stride register ids are fixed and should not be changed +// CASR should always be reg 0 +#define CASR 0 +// BASR should always be reg 1 +#define BASR 1 +// RASR should always be reg 2 +#define RASR 2 + +#define PATTERN_REG 12 +#define INV_PATTERN_REG 13 +#define TEMP_PATTERN_REG 14 + +#define BAR 3 +#define CAR 4 +#define RARBASE 5 +#define RAR1 6 +#define RAR2 9 + +#define COPY_REG1 10 +#define COPY_REG2 11 + +#define ITER_REG 7 +#define CTR_REG 8 + +#define INIT_LOOP 15 + +Program gen_test_prog(int bank, int r1, int r2, int iters, int placement) +{ + Program program; + + program.add_inst(SMC_LI(8, CASR)); // Load 8 into CASR since each READ reads 8 columns + program.add_inst(SMC_LI(1, BASR)); // Load 1 into BASR + program.add_inst(SMC_LI(1, RASR)); // Load 1 into RASR + + program.add_inst(SMC_LI(bank, BAR)); + program.add_inst(SMC_LI(r1-(r1%4), RARBASE)); + + program.add_inst(SMC_LI(iters, ITER_REG)); + program.add_inst(SMC_LI(0, CTR_REG)); + + program.add_label("WRITE_BEGIN"); + program.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + program.add_inst(SMC_SLEEP(3)); + switch(placement) + { + // Invert 0-1 start pattern before here when we set the + // WIDEREG + case 1000: + program.add_below(genWriteRange1000(RARBASE, 4, BAR)); + break; + case 1001: + program.add_below(genWriteRange1001(RARBASE, 4, BAR)); + break; + case 1010: + program.add_below(genWriteRange1010(RARBASE, 4, BAR)); + break; + case 1011: + program.add_below(genWriteRange1011(RARBASE, 4, BAR)); + break; + case 1100: + program.add_below(genWriteRange1100(RARBASE, 4, BAR)); + break; + case 1101: + program.add_below(genWriteRange1101(RARBASE, 4, BAR)); + break; + case 1110: + program.add_below(genWriteRange1110(RARBASE, 4, BAR)); + break; + case 1111: + program.add_below(genWriteRange1111(RARBASE, 4, BAR)); + break; + default: + printf("Unexpected placement encoding\n"); + exit(0); + } + + //program.add_below(genCopyRange(RARBASE, 4, BAR, true)); + program.add_inst(SMC_LI(r1, RAR1)); + program.add_inst(SMC_LI(r2, RAR2)); + + program.add_below(genActPreActSequence(1, 1, RAR1, RAR2, BAR)); + + program.add_inst(SMC_SLEEP(6)); + program.add_inst(SMC_PRE(BAR, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + program.add_below(genReadRange(RARBASE, 1, BAR, 128)); + program.add_inst(SMC_ADDI(CTR_REG, 1, CTR_REG)); + + program.add_branch(program.BR_TYPE::BL, CTR_REG, ITER_REG, "WRITE_BEGIN"); + program.add_inst(SMC_END()); + + return program; +} + +int main(int argc, char *argv[]) +{ + SoftMCPlatform platform; + int err; + + // buffer allocated for reading data from the board + unsigned char buf[4*READ_CL*64]; + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + platform.set_aref(true); + // Read selected segments from file + // first line: <number of segments> + // each line: <bank> <segment_start> <placement> + ifstream sf; + string sf_name = string(argv[2]); + sf.open(sf_name); + + int n_segments; sf >> n_segments; + std::vector<std::string> placements; + std::vector<int> banks, segments; + std::vector<float> ents; + for(int i = 0 ; i < n_segments ; i++) + { + int read; + sf >> read; banks.push_back(read); + sf >> read; segments.push_back(read); + float ent; + sf >> ent; ents.push_back(ent); + std::string rd; + sf >> rd; placements.push_back(rd); + } + + for(int i = 0 ; i < n_segments ; i++) + { + int bank = banks[i]; + int r1 = segments[i]; + float ent = ents[i]; + + if (ent < 1) + continue; + + std::string placement = placements[i]; + + bool invert = false; + // placement_magic + char new_placement[10]; + if (placement[0] == '0') + { + for (std::string::size_type i = 0 ; i < placement.size() ; i++) + if(placement[i] == '0') + new_placement[i] = '1'; + else if(placement[i] == '1') + new_placement[i] = '0'; + else + new_placement[i] = placement[i]; + placement = string(new_placement); + invert = true; + } + + Program program; + program.add_inst(SMC_LI(invert ? 0x0 : 0xffffffff, PATTERN_REG)); + program.add_inst(SMC_LI(invert ? 0xffffffff : 0x0, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + program.add_inst(SMC_END()); + platform.execute(program); + + unsigned char *bitstream; + + unsigned int sha_input_ent = 256; + unsigned int sha_input_blocks = ent/sha_input_ent + 1; + unsigned int row_partition_size = 8192/sha_input_blocks; + // UNTODO //TODO: from now on this is going to be 16Mbits + unsigned int test_iters = (1024*1024*1024)/(8*sha_input_blocks*32); + + + bitstream = new unsigned char[test_iters*32*sha_input_blocks]; + + cout << "Running for " << test_iters << " iterations." << endl; + cout << "Total bits of entropy: " << ent << " -> " << sha_input_blocks << " sha input blocks for the row" << endl; + + Program prog = gen_test_prog(bank, r1, r1+3, test_iters, std::stoi(placement)); + platform.execute(prog); + + for(int i = 0 ; i < test_iters ; i++) + { + platform.receiveData((char*)buf, 8192); // read one segment each iteration + // https://stackoverflow.com/q/13784434 + for(int j = 0 ; j < sha_input_blocks ; j++) + { + unsigned char hash[SHA256_DIGEST_LENGTH]; + SHA256_CTX sha256; + SHA256_Init(&sha256); + SHA256_Update(&sha256, (char*)(buf+row_partition_size*j), row_partition_size); + SHA256_Final(hash, &sha256); + for(int k = 0 ; k < SHA256_DIGEST_LENGTH ; k++) + { + bitstream[i*sha_input_blocks*SHA256_DIGEST_LENGTH + j*SHA256_DIGEST_LENGTH + k] = hash[k]; + } + } + } + ofstream dump_file; + string fn = string(argv[1]) + "/" + placements[i] + "_" + to_string(r1) + ".bin"; + dump_file.open(fn, ofstream::binary); + dump_file.write((char*)(bitstream), test_iters*32*sha_input_blocks); + dump_file.close(); + printf("\rrow %d", r1); + fflush(stdout); + delete bitstream; + } + +} diff --git a/sources/apps/QUAC-TRNG/src/util.cpp b/sources/apps/QUAC-TRNG/src/util.cpp new file mode 100644 index 0000000..41e4d50 --- /dev/null +++ b/sources/apps/QUAC-TRNG/src/util.cpp @@ -0,0 +1,670 @@ +#include "util.h" + +/** + * To send an ACT -> SLEEP t1 -> PRE -> SLEEP t2 -> ACT sequence + * This seq. will use BAR RAR1 and RAR2 registers + */ +Program genActPreActSequence(int t1, int t2, int row1_reg, int row2_reg, int bank_reg) +{ + Program ret; + // t1 -> ACT to PRE, t2 -> PRE to 2nd ACT + int sz = 3 + t1 + t2; + sz = (4-(sz%4)) + sz; + Mininst buff[sz]; + for(int i = 0 ; i < sz ; i++) + buff[i] = SMC_NOP(); + + // Overwrite with the actual sequence + buff[0] = SMC_ACT(bank_reg, 0, row1_reg, 0); + buff[1+t1] = SMC_PRE(bank_reg, 0, 0); + buff[2+t1+t2] = SMC_ACT(bank_reg, 0, row2_reg, 0); + + for(int i = 0 ; i < sz ; i+=4) + ret.add_inst(buff[i], buff[i+1], buff[i+2], buff[i+3]); + + return ret; +} + +Program genReadRange(int row_reg, int no_rows, int bank_reg, int read_cols) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // read until this row + ret.add_label("genReadRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < read_cols/4 ; i++) + { + ret.add_inst(SMC_READ(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_READ(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_READ(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_READ(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_branch(ret.BR_TYPE::BL, row_reg, INIT_LOOP, "genReadRange:LOOP_BEGIN"); + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + + + + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1000(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1001(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1010(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_branch(ret.BR_TYPE::BL, row_reg, INIT_LOOP, "genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1011(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1100(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_SLEEP(4)); + + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1101(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_SLEEP(4)); + + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1110(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + ret.add_inst(SMC_MV(PATTERN_REG, TEMP_PATTERN_REG)); + ret.add_inst(SMC_MV(INV_PATTERN_REG, PATTERN_REG)); + ret.add_inst(SMC_MV(TEMP_PATTERN_REG, INV_PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + ret.add_inst(SMC_LDWD(PATTERN_REG,i)); + + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * This function will generate a block of instructions that will + * write to a range of rows. It assumes the bank will be precharged. + */ +Program genWriteRange1111(int row_reg, int no_rows, int bank_reg) +{ + Program ret; + ret.add_inst(SMC_LI(no_rows, INIT_LOOP)); + ret.add_inst(SMC_ADD(INIT_LOOP, row_reg, INIT_LOOP)); // write until this row + + ret.add_label("genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_LI(0, CAR)); + + ret.add_inst(SMC_ACT(bank_reg, 0, row_reg, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_SLEEP(4)); + + for(int i = 0 ; i < 32 ; i++) + { + ret.add_inst(SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP(), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0), SMC_NOP()); + ret.add_inst(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_WRITE(bank_reg, 0, CAR, 1, 0, 0)); + ret.add_inst(SMC_NOP(),SMC_NOP(),SMC_NOP(),SMC_NOP()); + } + ret.add_inst(SMC_SLEEP(4)); + ret.add_inst(SMC_PRE(bank_reg, 0, 1), SMC_NOP(), SMC_NOP(), SMC_NOP()); + + ret.add_branch(ret.BR_TYPE::BL, row_reg, INIT_LOOP, "genWriteRange:LOOP_BEGIN"); + ret.add_inst(SMC_SUBI(row_reg, no_rows, row_reg)); + return ret; +} + +/** + * @return an instruction formed by NOPs + */ +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} diff --git a/sources/apps/QUAC-TRNG/src/util.h b/sources/apps/QUAC-TRNG/src/util.h new file mode 100644 index 0000000..1f07e0e --- /dev/null +++ b/sources/apps/QUAC-TRNG/src/util.h @@ -0,0 +1,55 @@ +#ifndef UTIL_HH +#define UTIL_HH + +#include "prog.h" + +#define ITERS_DUMP 1024*1024 +#define READ_CL_DUMP 8 + +#define ITERS 10000 //1024*1024 // 256*8 +#define READ_CL 128 + +#define NO_BANKS 1 +#define NO_ROWS 1024*32 + +// Stride register ids are fixed and should not be changed +// CASR should always be reg 0 +#define CASR 0 +// BASR should always be reg 1 +#define BASR 1 +// RASR should always be reg 2 +#define RASR 2 + +#define PATTERN_REG 12 +#define INV_PATTERN_REG 13 +#define TEMP_PATTERN_REG 14 + +#define BAR 3 +#define CAR 4 +#define RARBASE 5 +#define RAR1 6 +#define RAR2 9 + +#define COPY_REG1 10 +#define COPY_REG2 11 + +#define ITER_REG 7 +#define CTR_REG 8 + +#define INIT_LOOP 15 + +Program genActPreActSequence(int t1, int t2, int row1_reg, int row2_reg, int bank_reg); +Program genReadRange(int row_reg, int no_rows, int bank_reg, int read_cols); +Program genWriteRange1000(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1111(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1001(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1010(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1011(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1100(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1101(int row_reg, int no_rows, int bank_reg); +Program genWriteRange1110(int row_reg, int no_rows, int bank_reg); +Inst all_nops(); + + + +#endif diff --git a/sources/apps/ResetBoard/SoftMC_reset b/sources/apps/ResetBoard/SoftMC_reset Binary files differnew file mode 100755 index 0000000..709b93c --- /dev/null +++ b/sources/apps/ResetBoard/SoftMC_reset diff --git a/sources/apps/ResetBoard/full_reset.sh b/sources/apps/ResetBoard/full_reset.sh new file mode 100755 index 0000000..577a9a9 --- /dev/null +++ b/sources/apps/ResetBoard/full_reset.sh @@ -0,0 +1,9 @@ +#!/bin/bash + + +($( dirname "${BASH_SOURCE[0]}" )/SoftMC_reset) & +pid=$! + +sleep 2 + +kill -9 $pid diff --git a/sources/apps/RetentionTest/Makefile b/sources/apps/RetentionTest/Makefile new file mode 100755 index 0000000..c3828af --- /dev/null +++ b/sources/apps/RetentionTest/Makefile @@ -0,0 +1,28 @@ +program_NAME := SoftMC_Retention +program_CXX_SRCS := retention.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := pthread +CPPFLAGS += -g -std=c++11 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +distclean: clean diff --git a/sources/apps/RetentionTest/retention.cpp b/sources/apps/RetentionTest/retention.cpp new file mode 100644 index 0000000..86b8d98 --- /dev/null +++ b/sources/apps/RetentionTest/retention.cpp @@ -0,0 +1,205 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> +#include <cassert> +#include <sys/time.h> + +using namespace std; + +#define RETENTION_IN_MS 1000 +#define NUM_BANKS 1 +#define NUM_ROWS 32*1024 + +#define GET_TIME_INIT(num) struct timeval _timers[num] +#define GET_TIME_VAL(num) gettimeofday(&_timers[num], NULL) +#define TIME_VAL_TO_MS(num) (((double)_timers[num].tv_sec*1000.0) + ((double)_timers[num].tv_usec/1000.0)) + +// Need to hand-calculate below value? +// indicates how many rows we can write to +// before testing for retention +// TODO SET THIS TO A REASONABLE VALUE +#define MAX_ROW_PER_ITER 512 + +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +int main(int argc, char*argv[]) +{ + Program program; + SoftMCPlatform platform; + int err; + + if(argc != 3) + { + printf("Usage: \n ./SoftMC_RETENTION <retention time in ms> <output file name>\n"); + exit(0); + } + + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + + char* buf = (char*) malloc(sizeof(char)*32*1024); + platform.reset_fpga(); + // sleep(1); + + uint64_t sleep_counter = (uint64_t) ((double) RETENTION_IN_MS * 1000 * 1000 // retention in ns + / 6) ; // 6ns clock period + + assert(sleep_counter < (uint64_t)4*1024*1024*1024 && "Retention time too large to fit into a 32 bit register"); + + printf("Retention time given as %d ms\n", RETENTION_IN_MS); + printf("Will sleep for %ld cycles\n", sleep_counter*6); + + program.add_inst(SMC_LI(8, 0)); // CASR + program.add_inst(SMC_LI(1, 1)); // BASR + program.add_inst(SMC_LI(1, 2)); // RASR + + program.add_inst(SMC_LI(NUM_ROWS, 8)); // NUM_ROWS + program.add_inst(SMC_LI(NUM_BANKS, 11)); + + // To fill wide-reg with data + program.add_inst(SMC_LI(0xffffffff, 7)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(7,i)); + + + program.add_inst(SMC_LI(0, 5)); // BAR + program.add_label("BANK_BEGIN"); + program.add_inst(SMC_LI(MAX_ROW_PER_ITER,7)); + program.add_inst(SMC_LI(0, 4)); // CAR + program.add_inst(SMC_LI(0, 6)); // RAR + + program.add_label("WRITE"); + + // PRE & wait for tRP + program.add_inst(__pack_mininsts( + SMC_PRE(5, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst(__pack_mininsts( + SMC_ACT(5, 0, 6, 1), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // Write to a whole row + for(int i = 0 ; i < 128 ; i++){ + program.add_inst(__pack_mininsts( + SMC_WRITE(5, 0, 4, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + } + + // Wait for t(write-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(__pack_mininsts( + SMC_PRE(5, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + + program.add_branch(program.BR_TYPE::BL, 6, 7, "WRITE"); + + // implement "sleep" as a counter + //program.add_label("SLEEP"); + //program.add_inst(SMC_LI(0, 9)); + //program.add_inst(SMC_LI(sleep_counter/7, 10)); + //program.add_label("SLEEP_INNER"); + //program.add_inst(SMC_ADDI(9,1,9)); + //program.add_branch(program.BR_TYPE::BL, 9, 10, "SLEEP_INNER"); + program.add_inst(SMC_SLEEP(sleep_counter)); + + // decrement RAR to read from newly written rows + program.add_inst(SMC_SUBI(6, MAX_ROW_PER_ITER, 6)); + + program.add_label("READ"); + // PRE & wait for tRP + program.add_inst(__pack_mininsts( + SMC_PRE(5, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst(__pack_mininsts( + SMC_ACT(5, 0, 6, 1), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // Read from a whole row + for(int i = 0 ; i < 128 ; i++){ + program.add_inst(__pack_mininsts( + SMC_READ(5, 0, 4, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + program.add_inst(all_nops()); + } + + // Wait for t(read-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(__pack_mininsts( + SMC_PRE(5, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + )); + + program.add_branch(program.BR_TYPE::BEQ, 6, 8, "INC_BANK"); + program.add_branch(program.BR_TYPE::BL, 6, 7, "READ"); + program.add_inst(SMC_ADDI(7, MAX_ROW_PER_ITER, 7)); + program.add_branch(program.BR_TYPE::JUMP, 0, 0, "WRITE"); + + program.add_label("INC_BANK"); + program.add_inst(SMC_ADDI(5,1,5)); + program.add_branch(program.BR_TYPE::BEQ, 5, 11, "END"); + program.add_branch(program.BR_TYPE::JUMP,0,0,"BANK_BEGIN"); + + program.add_label("END"); + program.add_inst(SMC_END()); + + platform.execute(program); + + uint8_t pattern = 0xff; + uint64_t r_count = 0; + + ofstream failures_file; + failures_file.open(std::string(argv[2]), ofstream::binary); + while(1) + { + platform.receiveData(buf,8*1024); + for(int j = 0 ; j < 8*1024 ; j++) + failures_file << (buf[j] ^ pattern); + r_count++; + if(r_count % (NUM_ROWS) == 0) + printf("Bank %ld complete\n", r_count/(NUM_ROWS)); + if(r_count == NUM_BANKS*NUM_ROWS) + break; + printf("\rRead: %ld",r_count); + fflush(stdout); + } + printf("Test complete!\n"); +} diff --git a/sources/apps/SanityCheck/.gitignore b/sources/apps/SanityCheck/.gitignore new file mode 100644 index 0000000..5f2829a --- /dev/null +++ b/sources/apps/SanityCheck/.gitignore @@ -0,0 +1 @@ +SoftMC_Sanity
\ No newline at end of file diff --git a/sources/apps/SanityCheck/Makefile b/sources/apps/SanityCheck/Makefile new file mode 100755 index 0000000..19e8a5c --- /dev/null +++ b/sources/apps/SanityCheck/Makefile @@ -0,0 +1,33 @@ +program_NAME := SoftMC_Sanity +program_CXX_SRCS := sanity.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) +LDFLAGS += -lboost_program_options + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +parser: + $(MAKE) -C ../../api/lexyacc + cp ../../api/lexyacc/smc_parser . + +distclean: clean diff --git a/sources/apps/SanityCheck/sanity.cpp b/sources/apps/SanityCheck/sanity.cpp new file mode 100644 index 0000000..0c7b0f2 --- /dev/null +++ b/sources/apps/SanityCheck/sanity.cpp @@ -0,0 +1,294 @@ +#include <iostream> + +#include <boost/program_options.hpp> + +#include "instruction.h" +#include "prog.h" +#include "platform.h" + +using namespace std; +namespace po = boost::program_options; + +#define USE_SMC_FILE 0 + +#define NUM_ROWS_REG 8 +#define NUM_BANKS_REG 11 +#define CASR 0 +#define BASR 1 +#define RASR 2 + +#define PATTERN_REG 12 + +#define BAR 7 +#define RAR 6 +#define CAR 4 + +#define NUM_COLS_REG 14 +#define LOOP_COLS 13 + +#define TEMP_PATTERN_REG 15 + +/** + * @return an instruction formed by NOPs + */ +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +int main(int argc, char* argv[]) +{ + uint num_banks, num_rows, num_cols; + + po::options_description help("General options"); + help.add_options()("help,h", "Display the help message\n"); + + po::options_description dram_org("DRAM organization options"); + dram_org.add_options() + ("num_banks", po::value<uint>(&num_banks)->required(), "Specify the number of banks in the chip") + ("num_rows", po::value<uint>(&num_rows )->required(), "Specify the number of rows per bank") + ("num_cols", po::value<uint>(&num_cols )->required(), "Specify the number of columns (cachelines) per row\n") + ; + + po::options_description all; + all.add(help).add(dram_org); + + po::variables_map vm; + try + { + po::store(po::parse_command_line(argc, argv, all), vm); + po::notify(vm); + } + catch(const po::error& e) + { + if (vm.count("help") || argc == 1) + { + std::cout << all << "\n"; + std::exit(1); + } + else + { + std::cerr << "CRITICAL: " << e.what() << std::endl; + std::cerr << " Use \"--help\" to view all options" << std::endl; + std::exit(1); + } + } + + SoftMCPlatform platform; + int err; + uint32_t wr_pattern; + // buffer allocated for reading data from the board + char* buf = (char*) malloc(sizeof(char)*32*1024); + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS) + { + cerr << "Could not initialize SoftMC Platform: " << err << endl; + exit(-1); + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + + Program program; + /* SoftMC programs are formed sequentially by adding + * instructions to the program one by one. Instructions + * can be added to the program using add_inst(). + */ + + program.add_inst(SMC_LI(num_rows, NUM_ROWS_REG)); // load NUM_ROWS into NUM_ROWS_REG + program.add_inst(SMC_LI(num_banks, NUM_BANKS_REG)); // load NUM_BANKS into NUM_BANKS_REG + + // Stride registers are used to increment the row,bank,column registers without + // using regular (non-DDR) instructions. This way the DRAM array can be traversed much faster + + program.add_inst(SMC_LI(8, CASR)); // Load 8 into CASR since each READ reads 8 columns + program.add_inst(SMC_LI(1, BASR)); // Load 1 into BASR + program.add_inst(SMC_LI(1, RASR)); // Load 1 into RASR + + // Initialize PRNG, this test will randomly generate + // the first data to write into DRAM, it will modify + // the data by cyclic shifting and multiplying it with itself + // during runtime (this will be further explained down below). + srand(time(NULL)+getpid()); + uint16_t rand1 = (uint16_t) rand(); + wr_pattern = rand1 << 16; + wr_pattern |= ((uint16_t)rand()); + printf("wr_pattern: %x\n", wr_pattern); + + // Load WR data into pattern register whose content + // will be replicated among all words in the wide register + // Wide register is a 512-bit register, its content will + // be written to DRAM with SMC_WR commands. + program.add_inst(SMC_LI(wr_pattern, PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + program.add_inst(SMC_LI(0, BAR)); // Initialize BAR (bank address register) with 0 + + // add_label is used to add a label that can be used as a branch target to the program + program.add_label("BANK_BEGIN"); + + program.add_inst(SMC_LI(0, RAR)); // Initialize RAR with 0 + + program.add_label("ROW_BEGIN"); + + program.add_inst(SMC_LI(0, CAR)); // Initialize CAR with 0 + + // Four DRAM commands are issued at the same time + // to DRAM by SoftMC. + + // PREcharge bank BAR and wait for tRP (11 NOPs, 11 SoftMC cycles) + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst( + SMC_ACT(BAR, 0, RAR, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // We are now going to loop over all 512-bits in the row + // Since each row is 2^13 bytes we need to WR 128 times + program.add_inst(SMC_LI(128,NUM_COLS_REG)); // Load COL_SIZE register + program.add_inst(SMC_LI(0,LOOP_COLS)); // Load loop variable + + /* + * Try to write a different pattern to each cache line + * in DRAM. + */ + program.add_label("WR_BEGIN"); + + // Write to a whole row, increment CAR per WR + program.add_inst( + SMC_WRITE(BAR, 0, CAR, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + // Reload wide reg with the new data pattern + program.add_inst(SMC_SRC(PATTERN_REG,PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + program.add_inst(SMC_ADDI(LOOP_COLS,1,LOOP_COLS)); + // add a branch lower instruction which will jump to WR_BEGIN + // as long as LOOP_COLS<NUM_COLS, to iterate over the whole row + program.add_branch(program.BR_TYPE::BL,LOOP_COLS,NUM_COLS_REG, "WR_BEGIN"); + // END "WR_BEGIN" LOOP + + // Add more randomness to the PATTERN_REG by multiplying it by 3 + // This happens when a row is written to + program.add_inst(SMC_MV(PATTERN_REG,TEMP_PATTERN_REG)); + program.add_inst(SMC_ADD(PATTERN_REG,TEMP_PATTERN_REG,PATTERN_REG)); + program.add_inst(SMC_ADD(PATTERN_REG,TEMP_PATTERN_REG,PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + // Wait for t(write-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + program.add_inst(SMC_LI(0, CAR)); // reload CAR we are now going to read the whole row + + // PRE & wait for tRP + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst( + SMC_ACT(BAR, 0, RAR, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // Read from a whole row + // Essentially this can be imagined as an unrolled read whole row loop + for(int i = 0 ; i < num_cols ; i++){ + program.add_inst( + SMC_READ(BAR, 0, CAR, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + } + + // Wait for t(read-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + // Increment row address by one since we are finished with this row + program.add_inst(SMC_ADDI(RAR,1,RAR)); + program.add_branch(program.BR_TYPE::BL,RAR,NUM_ROWS_REG, "ROW_BEGIN"); + program.add_inst(SMC_ADDI(BAR,1,BAR)); + program.add_branch(program.BR_TYPE::BL,BAR,NUM_BANKS_REG,"BANK_BEGIN"); + + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // Tells SoftMC that the user program has ended + program.add_inst(SMC_END()); + + // Transfer the program to the FPGA board + platform.execute(program); + + // Read count + int rc = 0; + + uint32_t pattern = wr_pattern; + uint64_t err_count = 0; + // Below loop reads continuously from the FPGA board + // since we know in what order we are issuing our reads + // we know what physical address they are reading + while(1) + { + platform.receiveData(buf, 64 * num_cols); // read one row each iteration + for(int j = 0 ; j < num_cols ; j++){ + // printf("Whole pattern for cacheline %d: %x\n", j, pattern); + for (int k = 0 ; k < 64 ; k++){ // each byte in cache line + uint8_t mini_patt = pattern >> ((uint32_t)(k%4)*8); + if(mini_patt != (uint8_t)buf[j*64 + k]){ + err_count++; + fprintf(stderr,"t1: Pattern mismatch! Bank: %d Row: %d CacheLine: %d Byte: %d Expect: %x Read: %x\n", + rc/(num_rows), rc%(num_rows), j, k, mini_patt, (uint8_t)buf[j*64 + k]); + } + } + + //shift pattern at each row iteration + uint32_t imd = pattern >> (uint32_t)1; + uint32_t imd2 = pattern << (uint32_t)31; + pattern = imd | imd2; + } + pattern = pattern*(uint32_t)3; + rc++; + if(rc % (num_rows) == 0) + printf("Bank %d finished!\n", rc/num_rows); + if(rc == num_rows*num_banks) + break; + } + + if (err_count != 0) + exit(-1); + else + return 0; +} diff --git a/sources/apps/Smalltest/.gitignore b/sources/apps/Smalltest/.gitignore new file mode 100644 index 0000000..57855ab --- /dev/null +++ b/sources/apps/Smalltest/.gitignore @@ -0,0 +1 @@ +SoftMC_rdwr diff --git a/sources/apps/Smalltest/Makefile b/sources/apps/Smalltest/Makefile new file mode 100755 index 0000000..c7498df --- /dev/null +++ b/sources/apps/Smalltest/Makefile @@ -0,0 +1,32 @@ +program_NAME := SoftMC_rdwr +program_CXX_SRCS := read_write.cpp $(wildcard ../../api/*.c) $(wildcard ../../api/*.cpp) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_CXX_OBJS := ${program_CXX_OBJS:.c=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api ../../../boost-lib +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 -pthread -O3 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +parser: + $(MAKE) -C ../../api/lexyacc + cp ../../api/lexyacc/smc_parser . + +distclean: clean diff --git a/sources/apps/Smalltest/read_write.cpp b/sources/apps/Smalltest/read_write.cpp new file mode 100644 index 0000000..608b17a --- /dev/null +++ b/sources/apps/Smalltest/read_write.cpp @@ -0,0 +1,293 @@ +#include "instruction.h" +#include "prog.h" +#include "platform.h" +#include <fstream> +#include <iostream> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> +#include <cstring> +#include <list> + +using namespace std; + +#define USE_SMC_FILE 0 + +#define NUM_BANKS 16 +#define NUM_ROWS 1024*32 + +#define NUM_ROWS_REG 8 +#define NUM_BANKS_REG 11 +// Stride register ids are fixed and should not be changed +// CASR should always be reg 0 +// BASR should always be reg 1 +// RASR should always be reg 2 +#define CASR 0 +#define BASR 1 +#define RASR 2 + +#define PATTERN_REG 12 + +#define BAR 7 +#define RAR 6 +#define CAR 4 + +#define NUM_COLS_REG 14 +#define LOOP_COLS 13 + +#define TEMP_PATTERN_REG 15 + +/** + * @return an instruction formed by NOPs + */ +Inst all_nops() +{ + return __pack_mininsts(SMC_NOP(), SMC_NOP(), SMC_NOP(), SMC_NOP()); +} + +int main() +{ + SoftMCPlatform platform; + int err; + uint32_t wr_pattern; + // buffer allocated for reading data from the board + char* buf = (char*) malloc(sizeof(char)*32*1024); + + // Demo using the smc file as program input + if(USE_SMC_FILE) + { + // Load demo smc program from file using + // program's constructor + Program program("read_write.smc"); + wr_pattern = 0x1a2b3c4d; + printf("Program Ready\n"); + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + // Transfer the program to the FPGA board + platform.execute(program); + printf("Sent instructions\n"); + } + else + { + + // Initialize the platform, opens file descriptors for the board PCI-E interface. + if((err = platform.init()) != SOFTMC_SUCCESS){ + cerr << "Could not initialize SoftMC Platform: " << err << endl; + } + // reset the board to hopefully restore the board's state + platform.reset_fpga(); + + Program program; + /* SoftMC programs are formed sequentially by adding + * instructions to the program one by one. Instructions + * can be added to the program using add_inst(). + */ + + program.add_inst(SMC_LI(NUM_ROWS, NUM_ROWS_REG)); // load NUM_ROWS into NUM_ROWS_REG + program.add_inst(SMC_LI(NUM_BANKS, NUM_BANKS_REG)); // load NUM_BANKS into NUM_BANKS_REG + + // Stride registers are used to increment the row,bank,column registers without + // using regular (non-DDR) instructions. This way the DRAM array can be traversed much faster + + program.add_inst(SMC_LI(8, CASR)); // Load 8 into CASR since each READ reads 8 columns + program.add_inst(SMC_LI(1, BASR)); // Load 1 into BASR + program.add_inst(SMC_LI(1, RASR)); // Load 1 into RASR + + // Initialize PRNG, this test will randomly generate + // the first data to write into DRAM, it will modify + // the data by cyclic shifting and multiplying it with itself + // during runtime (this will be further explained down below). + srand(time(NULL)+getpid()); + uint16_t rand1 = (uint16_t) rand(); + wr_pattern = rand1 << 16; + wr_pattern |= ((uint16_t)rand()); + printf("wr_pattern: %x\n", wr_pattern); + + // Load WR data into pattern register whose content + // will be replicated among all words in the wide register + // Wide register is a 512-bit register, its content will + // be written to DRAM with SMC_WR commands. + program.add_inst(SMC_LI(wr_pattern, PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + program.add_inst(SMC_LI(0, BAR)); // Initialize BAR (bank address register) with 0 + + // add_label is used to add a label that can be used as a branch target to the program + program.add_label("BANK_BEGIN"); + + program.add_inst(SMC_LI(0, RAR)); // Initialize RAR with 0 + + program.add_label("ROW_BEGIN"); + + program.add_inst(SMC_LI(0, CAR)); // Initialize CAR with 0 + + // Four DRAM commands are issued at the same time + // to DRAM by SoftMC. + + // PREcharge bank BAR and wait for tRP (11 NOPs, 11 SoftMC cycles) + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst( + SMC_ACT(BAR, 0, RAR, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // We are now going to loop over all 512-bits in the row + // Since each row is 2^13 bytes we need to WR 128 times + program.add_inst(SMC_LI(128,NUM_COLS_REG)); // Load COL_SIZE register + program.add_inst(SMC_LI(0,LOOP_COLS)); // Load loop variable + + /* + * Try to write a different pattern to each cache line + * in DRAM. + */ + program.add_label("WR_BEGIN"); + + // Write to a whole row, increment CAR per WR + program.add_inst( + SMC_WRITE(BAR, 0, CAR, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + // Reload wide reg with the new data pattern + program.add_inst(SMC_SRC(PATTERN_REG,PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + program.add_inst(SMC_ADDI(LOOP_COLS,1,LOOP_COLS)); + // add a branch lower instruction which will jump to WR_BEGIN + // as long as LOOP_COLS<NUM_COLS, to iterate over the whole row + program.add_branch(program.BR_TYPE::BL,LOOP_COLS,NUM_COLS_REG, "WR_BEGIN"); + // END "WR_BEGIN" LOOP + + // Add more randomness to the PATTERN_REG by multiplying it by 3 + // This happens when a row is written to + program.add_inst(SMC_MV(PATTERN_REG,TEMP_PATTERN_REG)); + program.add_inst(SMC_ADD(PATTERN_REG,TEMP_PATTERN_REG,PATTERN_REG)); + program.add_inst(SMC_ADD(PATTERN_REG,TEMP_PATTERN_REG,PATTERN_REG)); + for(int i = 0 ; i < 16 ; i++) + program.add_inst(SMC_LDWD(PATTERN_REG,i)); + + // Wait for t(write-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + program.add_inst(SMC_LI(0, CAR)); // reload CAR we are now going to read the whole row + + // PRE & wait for tRP + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // ACT & wait for tRCD + program.add_inst( + SMC_ACT(BAR, 0, RAR, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + // Read from a whole row + // Essentially this can be imagined as an unrolled read whole row loop + for(int i = 0 ; i < 128 ; i++){ + program.add_inst( + SMC_READ(BAR, 0, CAR, 1, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + program.add_inst(all_nops()); + } + + // Wait for t(read-precharge) + // & precharge the open bank + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst( + SMC_PRE(BAR, 0, 0), + SMC_NOP(), SMC_NOP(), SMC_NOP() + ); + + // Increment row address by one since we are finished with this row + program.add_inst(SMC_ADDI(RAR,1,RAR)); + program.add_branch(program.BR_TYPE::BL,RAR,NUM_ROWS_REG, "ROW_BEGIN"); + program.add_inst(SMC_ADDI(BAR,1,BAR)); + program.add_branch(program.BR_TYPE::BL,BAR,NUM_BANKS_REG,"BANK_BEGIN"); + + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + program.add_inst(all_nops()); + + program.dump_registers(); + + // Tells SoftMC that the user program has ended + program.add_inst(SMC_END()); + + // Transfer the program to the FPGA board + platform.execute(program); + printf("Program Ready\n"); + program.pretty_print(); + printf("Sent instructions\n"); + } + + // Read count + int rc = 0; + + uint32_t pattern = wr_pattern; + uint64_t err_count = 0; + // Below loop reads continuously from the FPGA board + // since we know in what order we are issuing our reads + // we know what physical address they are reading + while(1) + { + platform.receiveData(buf, 8*1024); // read one row each iteration + for(int j = 0 ; j < 128 ; j++){ + // printf("Whole pattern for cacheline %d: %x\n", j, pattern); + for (int k = 0 ; k < 64 ; k++){ // each byte in cache line + uint8_t mini_patt = pattern >> ((uint32_t)(k%4)*8); + if(mini_patt != (uint8_t)buf[j*64 + k]){ + err_count++; + fprintf(stderr,"t1: Pattern mismatch! Bank: %d Row: %d CacheLine: %d Byte: %d Expect: %x Read: %x\n", + rc/(1024*32), rc%(1024*32), j, k, mini_patt, (uint8_t)buf[j*64 + k]); + //fprintf(stderr,"%d %d %d %x\n", + // rc/(1024*32), rc%(1024*32), j*64 + k, mini_patt^(uint8_t)buf[j*64 + k]); + } + } + + //shift pattern at each row iteration + uint32_t imd = pattern >> (uint32_t)1; + uint32_t imd2 = pattern << (uint32_t)31; + pattern = imd | imd2; + } + pattern = pattern*(uint32_t)3; + rc++; + if(rc % (1024*32) == 0) + printf("Bank %d finished!\n", rc/1024/32); + if(rc == 1024*32*16) + break; + } + printf("%ld out of %ld bytes have errors, last pattern: 0x%x\n", + err_count, (uint64_t)NUM_BANKS*NUM_ROWS*8192, pattern); + platform.readRegisterDump(); +} diff --git a/sources/apps/Smalltest/read_write.smc b/sources/apps/Smalltest/read_write.smc new file mode 100644 index 0000000..c4bfd5f --- /dev/null +++ b/sources/apps/Smalltest/read_write.smc @@ -0,0 +1,170 @@ +// Initialize register ids +// Similar to #define +casr = 0 +basr = 1 +rasr = 2 +bar = 7 +rar = 6 +car = 4 +pattern_reg = 12 +no_row_reg = 8 +no_bank_reg = 11 +no_cols_reg = 14 +loop_cols_reg = 13 +temp_pattern_reg = 15 + +// Begin writing the program +LI no_row_reg 32768 // Write 2^15 to no_row_Reg +LI no_bank_reg 16 // write 16 to no_bank_reg + +// Set stride registers (not required if you +// are not going to increment address registers +// with ddr commands) +LI casr 8 // set column address stride to 8 +LI basr 1 // set bas to 1 +LI rasr 1 // set ras to 1 + +// Init pattern register, we are going to fill +// dram with some random pattern. +LI pattern_reg 0x1a2b3c4d + +// Load the data register with the pattern +// second argument here is the word offset +// we are writing to in the data register +// e.g. LDWD 0 pattern reg fills the first four bytes + +LDWD 0 pattern_reg +LDWD 1 pattern_reg +LDWD 2 pattern_reg +LDWD 3 pattern_reg +LDWD 4 pattern_reg +LDWD 5 pattern_reg +LDWD 6 pattern_reg +LDWD 7 pattern_reg +LDWD 8 pattern_reg +LDWD 9 pattern_reg +LDWD 10 pattern_reg +LDWD 11 pattern_reg +LDWD 12 pattern_reg +LDWD 13 pattern_reg +LDWD 14 pattern_reg +LDWD 15 pattern_reg + +LI bar 0 // Initialize bank address register + +BANK_BEGIN: // Add label to loop over banks +LI rar 0 // Reload rar per bank loop + +ROW_BEGIN: // Add label to loop over rows +LI car 0 // Reload car per row loop + +PRE bar // PREcharge bar, don't increment, no precharge-all +WAIT 11 // Insert seven cycles of NOPs + +ACT bar rar // ACTivate bar, rar +WAIT 11 + +// Each write will write a different pattern +// "car++" indicates that we will increment +// car by casr value. + +LI no_cols_reg 128 +LI loop_cols_reg 0 + +WR_BEGIN: + +WR bar car++ +WAIT 3 + +// Cyclic shift pattern register to add "randomness" +SRC pattern_reg pattern_reg + +// Reload the wide data register +LDWD 0 pattern_reg +LDWD 1 pattern_reg +LDWD 2 pattern_reg +LDWD 3 pattern_reg +LDWD 4 pattern_reg +LDWD 5 pattern_reg +LDWD 6 pattern_reg +LDWD 7 pattern_reg +LDWD 8 pattern_reg +LDWD 9 pattern_reg +LDWD 10 pattern_reg +LDWD 11 pattern_reg +LDWD 12 pattern_reg +LDWD 13 pattern_reg +LDWD 14 pattern_reg +LDWD 15 pattern_reg + +// increment loop variable and iterate +ADDI loop_cols_reg loop_cols_reg 1 +BL WR_BEGIN loop_cols_reg no_cols_reg + +// multiply pattern_reg by 3 +MOV temp_pattern_reg pattern_reg +ADD pattern_reg pattern_reg temp_pattern_reg +ADD pattern_reg pattern_reg temp_pattern_reg + +// Reload the wide data register +LDWD 0 pattern_reg +LDWD 1 pattern_reg +LDWD 2 pattern_reg +LDWD 3 pattern_reg +LDWD 4 pattern_reg +LDWD 5 pattern_reg +LDWD 6 pattern_reg +LDWD 7 pattern_reg +LDWD 8 pattern_reg +LDWD 9 pattern_reg +LDWD 10 pattern_reg +LDWD 11 pattern_reg +LDWD 12 pattern_reg +LDWD 13 pattern_reg +LDWD 14 pattern_reg +LDWD 15 pattern_reg + +WAIT 8 +PRE bar +WAIT 3 + +// Reload car, we are now going to read from DRAM +LI car 0 + +// Precharge the open bank +PRE bar +WAIT 11 + +ACT bar rar +WAIT 11 + +// This is a basic primitive that can be used +// to replicate instructions-commands +// this block replicates the RD command 128 times +// and adds it to the program +times 128 { + RD bar car++ + WAIT 7 +} + +WAIT 8 +PRE bar +WAIT 3 + +// Increment rar per row loop +ADDI rar rar 1 +// if rar < no_row_reg, execute the "row loop" once more +BL ROW_BEGIN rar no_row_reg + +ADDI bar bar 1 +BL BANK_BEGIN bar no_bank_reg + +WAIT 16 + +// Tell softmc that the program has ended +// This is important! +END + +// Tell the parser the program definition has ended +// This is even more important!! +ENDPROGRAM diff --git a/sources/apps/parsertest/Makefile b/sources/apps/parsertest/Makefile new file mode 100755 index 0000000..6ca92f2 --- /dev/null +++ b/sources/apps/parsertest/Makefile @@ -0,0 +1,27 @@ +program_NAME := SoftMC_Example +program_CXX_SRCS := ptest.cpp $(wildcard ../../api/*.c*) +program_CXX_OBJS := ${program_CXX_SRCS:.cpp=.o} +program_OBJS := $(program_CXX_OBJS) +program_INCLUDE_DIRS := ../../api/ +program_LIBRARY_DIRS := +program_LIBRARIES := +CPPFLAGS += -g -std=c++11 + +CPPFLAGS += $(foreach includedir,$(program_INCLUDE_DIRS),-I$(includedir)) +LDFLAGS += $(foreach librarydir,$(program_LIBRARY_DIRS),-L$(librarydir)) +LDFLAGS += $(foreach library,$(program_LIBRARIES),-l$(library)) + +CC=g++ + +.PHONY: all clean distclean + +all: $(program_NAME) + +$(program_NAME): $(program_OBJS) + $(CC) $(CPPFLAGS) $(program_OBJS) -o $(program_NAME) $(LDFLAGS) + +clean: + @- $(RM) $(program_NAME) + @- $(RM) $(program_OBJS) + +distclean: clean diff --git a/sources/apps/parsertest/SoftMC_Example b/sources/apps/parsertest/SoftMC_Example Binary files differnew file mode 100755 index 0000000..c5146bf --- /dev/null +++ b/sources/apps/parsertest/SoftMC_Example diff --git a/sources/apps/parsertest/example.smc b/sources/apps/parsertest/example.smc new file mode 100644 index 0000000..a8cac7a --- /dev/null +++ b/sources/apps/parsertest/example.smc @@ -0,0 +1,9 @@ +LI 0 8; +LI 1 1; +LI 2 1; +LI 3 0; +LI 4 0; +LI 5 0; +BEGIN:; +WR 5 0 4 0 0 0, NOP, NOP, NOP; +END; diff --git a/sources/apps/parsertest/ptest.cpp b/sources/apps/parsertest/ptest.cpp new file mode 100644 index 0000000..28aabac --- /dev/null +++ b/sources/apps/parsertest/ptest.cpp @@ -0,0 +1,18 @@ +#include <string> +#include <fstream> +#include <iostream> +#include <sstream> +#include <string.h> +#include "prog.h" +#include "parser.h" + +int main() +{ + // https://stackoverflow.com/questions/2602013/read-whole-ascii-file-into-c-stdstring + std::ifstream t("example.smc"); + std::stringstream buffer; + buffer << t.rdbuf(); + string program = buffer.str(); + Parser p; + p.parse_program(program); +} diff --git a/sources/hdl/header_verilog/encoding.vh b/sources/hdl/header_verilog/encoding.vh new file mode 100644 index 0000000..f5d5697 --- /dev/null +++ b/sources/hdl/header_verilog/encoding.vh @@ -0,0 +1,103 @@ +// SoftMC instructions +`define FU_CODE_OFFSET 48 +`define BRANCH_OFFSET 62 +`define DDR_OFFSET 63 +`define INFO_OFFSET 61 +`define MEM_OFFSET 60 +`define BW_OFFSET 59 +`define DEC_RS1 0 +`define DEC_RS2 4 +`define DEC_IMD1 4 +`define DEC_IMD2 0 +`define DEC_IMD3 24 +`define DEC_RT 20 +`define DEC_WO 20 +`define DEC_JUMP_OFFSET 0 +`define DEC_SLEEP_OFFSET 0 +`define DEC_BR_TGT_OFFSET 8 +`define SR_OFFSET 56 +// DDR related +`define DDR_CODE_OFFSET 12 +`define DEC_CAR 4 +`define DEC_BAR 0 +`define DEC_RAR 4 +`define DEC_INC_BAR 10 +`define DEC_PRE_ALL 11 +`define DEC_INC_RAR 11 +`define DEC_INC_CAR 11 +`define DEC_AP 9 +`define DEC_BL4 8 +// function codes - exe +`define ADD 0 +`define ADDI 1 +`define SUB 2 +`define SUBI 3 +`define MV 4 +`define SRC 5 +`define LI 6 +`define LDWD 7 +`define LDPC 8 +`define SRE 0 //DDR command but no space left in ISA DDR instructions. +`define SRX 1 //DDR command but no space left in ISA DDR instructions. +`define BL 0 +`define BEQ 1 +`define JUMP 2 +`define SLEEP 3 +`define LD 0 +`define ST 1 +`define AND 0 +`define OR 1 +`define XOR 2 +// function codes - ddr +`define WRITE 0 +`define READ 1 +`define PRE 2 +`define ACT 3 +`define ZQ 4 +`define REF 5 +`define NOP 7 + +// SoftMC ddr uops +`define IS_WRITE 0 +`define IS_READ 1 +`define IS_PRE 2 +`define IS_ACT 3 +`define IS_ZQ 4 +`define IS_REF 5 +`define CAR 6 // column address register identifier +`define RAR 10 // row address register identifier +`define BAR 14 // bank address register identifier +`define PRE_ALL 18 +`define INC_CAR 19 // increment CAR after executing this +`define INC_RAR 20 +`define INC_BAR 21 +`define IS_NOP 22 +`define IS_BL4 23 +`define DO_AP 24 +`define IS_SRE 25 +`define IS_SRX 26 +// SoftMC exe uops +`define IS_ADD 0 +`define IS_SUB 1 +`define IS_MOV 2 +`define IS_LI 3 +`define IS_LDWD 4 +`define HAS_IMD 5 +`define IS_BL 6 +`define IS_BEQ 7 +`define IS_JUMP 8 +`define IS_SLEEP 9 +`define RS1 10 +`define RS2 14 +`define RT 18 +`define IMD 22 +`define IMD2 38 +`define IS_SRC 54 +`define IS_MEM 55 +`define IS_LD 56 +`define IS_ST 57 +`define IS_AND 58 +`define IS_OR 59 +`define IS_XOR 60 +`define IS_LDPC 61 + diff --git a/sources/hdl/header_verilog/parameters.vh b/sources/hdl/header_verilog/parameters.vh new file mode 100644 index 0000000..916c973 --- /dev/null +++ b/sources/hdl/header_verilog/parameters.vh @@ -0,0 +1,21 @@ +// Fetch +`define INSTR_WIDTH 64 +`define IMEM_ADDR_WIDTH 11 + +// Decode - Execute +`define DDR_UOP_WIDTH 27 +`define EXE_UOP_WIDTH 62 +`define BG_WIDTH 2 +`define BANK_WIDTH 2 +`define COL_WIDTH 10 +`define ROW_WIDTH 17 + +//Frontend +`define XDMA_AXI_DATA_WIDTH 256 +`define IMEM_RD_LATENCY 1 +//`define IMEM_SR // please only define when IMEM_RD_LATENCY > 1 + +//Common +`define HIGH 1'b1 +`define LOW 1'b0 + diff --git a/sources/hdl/sim_verilog/example_top.sv b/sources/hdl/sim_verilog/example_top.sv new file mode 100644 index 0000000..dbf4fea --- /dev/null +++ b/sources/hdl/sim_verilog/example_top.sv @@ -0,0 +1,557 @@ + +`ifdef MODEL_TECH + `ifndef CALIB_SIM + `define SIMULATION + `endif +`elsif INCA + `ifndef CALIB_SIM + `define SIMULATION + `endif +`elsif VCS + `ifndef CALIB_SIM + `define SIMULATION + `endif +`elsif XILINX_SIMULATOR + `ifndef CALIB_SIM + `define SIMULATION + `endif +`endif + +`timescale 1ps/1ps + +// Fetch +`define INSTR_WIDTH 64 +`define IMEM_ADDR_WIDTH 11 + +// Decode - Execute +`define DDR_UOP_WIDTH 27 +`define EXE_UOP_WIDTH 62 +`define BG_WIDTH 2 +`define BANK_WIDTH 2 +`define COL_WIDTH 10 +`define ROW_WIDTH 17 + +//Frontend +`define XDMA_AXI_DATA_WIDTH 256 +`define IMEM_RD_LATENCY 1 +//`define IMEM_SR // please only define when IMEM_RD_LATENCY > 1 + +//Common +`define HIGH 1'b1 +`define LOW 1'b0 + + + +`define UDIMM_x8 + +`define DQ_WIDTH 64 +`define ODT_WIDTH 2 +`define CS_WIDTH 2 +`define CKE_WIDTH 2 +`define CK_WIDTH 1 +`define ROW_ADDR_WIDTH 17 + + +module example_top # + ( + parameter SIMULATION = "FALSE" + ) + ( + + // common signals + input c0_sys_clk_p, + input c0_sys_clk_n, + input sys_rst, + + // iob <> ddr4 sdram ip signals + output c0_ddr4_act_n, + output [`ROW_ADDR_WIDTH-1:0] c0_ddr4_adr, + output [1:0] c0_ddr4_ba, + output [1:0] c0_ddr4_bg, + output [`CKE_WIDTH-1:0] c0_ddr4_cke, + output [`ODT_WIDTH-1:0] c0_ddr4_odt, + output [`CS_WIDTH-1:0] c0_ddr4_cs_n, + output [`CK_WIDTH-1:0] c0_ddr4_ck_t, + output [`CK_WIDTH-1:0] c0_ddr4_ck_c, + output c0_ddr4_reset_n, + + `ifdef RDIMM_x4 + inout [17:0] c0_ddr4_dqs_c, + inout [17:0] c0_ddr4_dqs_t, + inout [71:0] c0_ddr4_dq, + output c0_ddr4_parity, + `elsif UDIMM_x8 + inout [7:0] c0_ddr4_dqs_c, + inout [7:0] c0_ddr4_dqs_t, + inout [63:0] c0_ddr4_dq, + inout [7:0] c0_ddr4_dm_dbi_n, + output c0_ddr4_parity, + `elsif RDIMM_x8 + inout [8:0] c0_ddr4_dqs_c, + inout [8:0] c0_ddr4_dqs_t, + inout [71:0] c0_ddr4_dq, + inout [8:0] c0_ddr4_dm_dbi_n, + output c0_ddr4_parity, + `endif + + output c0_init_calib_complete, + output c0_data_compare_error +); + + `ifdef RDIMM_x4 + assign c0_ddr4_odt[1] = 1'b0; + assign c0_ddr4_cs_n[1] = 1'b1; + assign c0_ddr4_cke[1] = 1'b0; + `elsif RDIMM_x8 + assign c0_ddr4_odt[1] = 1'b0; + assign c0_ddr4_cs_n[1] = 1'b1; + assign c0_ddr4_cke[1] = 1'b0; + //assign c0_ddr4_parity = 1'b0; + `elsif UDIMM_x8 + assign c0_ddr4_odt[1] = 1'b0; + assign c0_ddr4_cs_n[1] = 1'b1; + assign c0_ddr4_cke[1] = 1'b0; + assign c0_ddr4_parity = 1'b0; + `endif + + // Frontend control signals + wire softmc_fin; + wire user_rst; + + // Frontend <-> Fetch signals + wire [`IMEM_ADDR_WIDTH-1:0] fr_addr_in; + wire fr_valid_in; + wire [`INSTR_WIDTH-1:0] fr_data_out; + wire fr_valid_out; + wire [`IMEM_ADDR_WIDTH-1:0] fr_addr_out; + wire fr_ready_out; + + // Frontend <-> misc. control signals + wire per_rd_init; + wire per_zq_init; + wire rbe_switch_mode; + + // AXI streaming ports + wire [`XDMA_AXI_DATA_WIDTH-1:0] m_axis_h2c_tdata_0,xdma_h2c_tdata_0; + wire m_axis_h2c_tlast_0, xdma_h2c_tlast_0; + wire m_axis_h2c_tvalid_0, xdma_h2c_tvalid_0; + wire m_axis_h2c_tready_0, xdma_h2c_tready_0; + wire [`XDMA_AXI_DATA_WIDTH/8-1:0] m_axis_h2c_tkeep_0, xdma_h2c_tkeep_0; + wire [`XDMA_AXI_DATA_WIDTH-1:0] s_axis_c2h_tdata_0, xdma_c2h_tdata_0; + wire s_axis_c2h_tlast_0, xdma_c2h_tlast_0; + wire s_axis_c2h_tvalid_0, xdma_c2h_tvalid_0; + wire s_axis_c2h_tready_0, xdma_c2h_tready_0; + wire [`XDMA_AXI_DATA_WIDTH/8-1:0] s_axis_c2h_tkeep_0, xdma_c2h_tkeep_0; + + // ddr_pipeline <-> outer module if + wire [3:0] ddr_write; + wire [3:0] ddr_read; + wire [3:0] ddr_pre; + wire [3:0] ddr_act; + wire [3:0] ddr_ref; + wire [3:0] ddr_zq; + wire [3:0] ddr_nop; + wire [3:0] ddr_sre; + wire [3:0] ddr_srx; + wire [3:0] ddr_ap; + wire [3:0] ddr_pall; + wire [3:0] ddr_half_bl; + wire [4*`BG_WIDTH-1:0] ddr_bg; + wire [4*`BANK_WIDTH-1:0] ddr_bank; + wire [4*`COL_WIDTH-1:0] ddr_col; + wire [4*`ROW_WIDTH-1:0] ddr_row; + wire [511:0] ddr_wdata; + + // periodic maintenance signals + wire ddr_maint_read; + + // phy <-> ddr adapter and dlltoggler signals + // dlltoggler + wire clk_sel = 0; + wire [7:0] dllt_mc_ACT_n; + wire [135:0] dllt_mc_ADR; + wire [15:0] dllt_mc_BA; + wire [15:0] dllt_mc_BG; + wire [7:0] dllt_mc_CKE; + wire [7:0] dllt_mc_CS_n; + wire dllt_done; + // adapter + wire [4:0] dBufAdr; + wire [`DQ_WIDTH*8-1:0] wrData; + wire [`DQ_WIDTH-1:0] wrDataMask; + wire [511:0] rdData; + wire [4:0] rdDataAddr; + wire [0:0] rdDataEn; + wire [0:0] rdDataEnd; + wire [0:0] per_rd_done; + wire [0:0] rmw_rd_done; + wire [4:0] wrDataAddr; + wire [0:0] wrDataEn; + wire [7:0] mc_ACT_n; + wire [135:0] mc_ADR; + wire [15:0] mc_BA; + wire [15:0] mc_BG; + wire [`CKE_WIDTH*8-1:0] mc_CKE; + wire [`CS_WIDTH*8-1:0] mc_CS_n; + wire [`ODT_WIDTH*8-1:0] mc_ODT; + wire [0:0] mcRdCAS; + wire [0:0] mcWrCAS; + wire [0:0] winInjTxn; + wire [0:0] winRmw; + wire [4:0] winBuf; + wire [1:0] winRank; + wire [5:0] tCWL; + wire dbg_clk; + wire c0_wr_rd_complete; + wire c0_ddr4_clk; + wire c0_ddr4_dll_off_clk; + wire ddr4_ui_clk; + wire c0_ddr4_rst; + wire [511:0] dbg_bus; + wire [1:0] mcCasSlot; + wire mcCasSlot2; + wire gt_data_ready; + + wire read_seq_incoming; // next few instructions will read from DRAM + wire [11:0] incoming_reads; // how many reads next few instructions will issue + wire [11:0] buffer_space; // remaining buffer size + + // There is a possibility that these signals are on + // the critical path as observed in + // the previous iteration of SoftMC + reg c0_init_calib_complete_r, sys_rst_r; + wire iq_full, processing_iseq, rdback_fifo_empty; + reg dllt_active = 1'b0; + + reg toggle_dll = 1'b0; + reg dont = 1'b1; + + always @(posedge c0_ddr4_clk) begin + c0_init_calib_complete_r <= c0_init_calib_complete; + sys_rst_r <= sys_rst; + `ifdef ENABLE_DLL_TOGGLER + if(c0_init_calib_complete_r && dont) begin + toggle_dll <= 1'b1; + dont <= 1'b0; + end + if(toggle_dll) begin + dllt_active <= ~dllt_active; + toggle_dll <= 1'b0; + end + if(dllt_done) begin + dllt_active <= ~dllt_active; + end + `endif + end + + + `ifdef UDIMM_x8 + phy_ddr4_udimm phy_ddr4_i( + .sys_rst (sys_rst), + .c0_sys_clk_p (c0_sys_clk_p), + .c0_sys_clk_n (c0_sys_clk_n), + + `ifdef ENABLE_DLL_TOGGLER + .c0_ddr4_ui_clk (ddr4_ui_clk), + .addn_ui_clkout1 (c0_ddr4_dll_off_clk), + `else + .c0_ddr4_ui_clk (c0_ddr4_clk), + `endif + .c0_ddr4_ui_clk_sync_rst (c0_ddr4_rst), + .c0_init_calib_complete (c0_init_calib_complete), + .dbg_clk (dbg_clk), + .c0_ddr4_act_n (c0_ddr4_act_n), + .c0_ddr4_adr (c0_ddr4_adr), + .c0_ddr4_ba (c0_ddr4_ba), + .c0_ddr4_bg (c0_ddr4_bg), + .c0_ddr4_cke (c0_ddr4_cke), + .c0_ddr4_odt (c0_ddr4_odt), + .c0_ddr4_cs_n (c0_ddr4_cs_n), + .c0_ddr4_ck_t (c0_ddr4_ck_t), + .c0_ddr4_ck_c (c0_ddr4_ck_c), + .c0_ddr4_reset_n (c0_ddr4_reset_n), + //.ddr4_par (c0_ddr4_parity), + .c0_ddr4_dq (c0_ddr4_dq), + .c0_ddr4_dqs_c (c0_ddr4_dqs_c), + .c0_ddr4_dqs_t (c0_ddr4_dqs_t), + .c0_ddr4_dm_dbi_n (c0_ddr4_dm_dbi_n), + + .dBufAdr (dBufAdr), + .wrData (wrData), + .rdData (rdData), + .rdDataAddr (rdDataAddr), + .rdDataEn (rdDataEn), + .rdDataEnd (rdDataEnd), + .per_rd_done (per_rd_done), + .rmw_rd_done (rmw_rd_done), + .wrDataAddr (wrDataAddr), + .wrDataEn (wrDataEn), + .wrDataMask (wrDataMask), + + .mc_ACT_n (dllt_active ? dllt_mc_ACT_n : mc_ACT_n), + .mc_ADR (dllt_active ? dllt_mc_ADR : mc_ADR), + .mc_BA (dllt_active ? dllt_mc_BA : mc_BA), + .mc_BG (dllt_active ? dllt_mc_BG : mc_BG), + // DRAM CKE. 8 bits for each DRAM pin. The mc_CKE signal is always set to '1'. + .mc_CKE (dllt_active ? dllt_mc_CKE : {8{1'b1}}), + .mc_CS_n (dllt_active ? dllt_mc_CS_n : mc_CS_n), + .mc_ODT (mc_ODT), + // CAS command slot select. Slot0 is enabled for example design. + .mcCasSlot (dllt_active ? 0 : mcCasSlot), + // CAS slot 2 select. mcCasSlot2 serves a similar purpose as the mcCasSlot[1:0] signal, but mcCasSlot2 is used in timing + // critical logic in the Phy. Slot0 is enabled for example design. + .mcCasSlot2 (dllt_active ? 0 : mcCasSlot2), + .mcRdCAS (dllt_active ? 0 : mcRdCAS), + .mcWrCAS (dllt_active ? 0 : mcWrCAS), + // Optional read command type indication. The winInjTxn signal is set to '0' for example design. + .winInjTxn ({1{1'b0}}), + // Optional read command type indication. The winRmw signal is set to '0' for example design. + .winRmw ({1{1'b0}}), + // Update VT Tracking. The gt_data_ready signal is set to '0' in this example design. + // This signal must be asserted periodically to keep the DQS Gate aligned as voltage and temperature drift. + // For more information, Refer to PG150 document. + .gt_data_ready (gt_data_ready), + .winBuf (winBuf), + .winRank (winRank), + .tCWL (tCWL), + // Debug Port + .dbg_bus (dbg_bus) + ); + `endif + + `ifdef ENABLE_DLL_TOGGLER + //BUFGMUX:GeneralClockMuxBuffer + //UltraScale + //XilinxHDLLibrariesGuide, version2014.4 + BUFGMUX#(.CLK_SEL_TYPE("SYNC") //ASYNC,SYNC + )BUFGMUX_inst( + .O(c0_ddr4_clk), //1-bitoutput:Clockoutput + .I0(ddr4_ui_clk), //1-bitinput:Clockinput(S=0) + .I1(c0_ddr4_dll_off_clk), //1-bitinput:Clockinput(S=1) + .S(clk_sel) //1-bitinput:Clockselect + ); + //End of BUFGMUX_inst instantiation + `endif + + softmc_pipeline pipeline( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst || ~c0_init_calib_complete_r), + + .softmc_end(softmc_fin), + .read_size(incoming_reads), + .read_seq_incoming(read_seq_incoming), + .buffer_space(buffer_space), + + .addr_out(fr_addr_in), + .valid_out(fr_valid_in), + .data_in(fr_data_out), + .valid_in(fr_valid_out), + .addr_in(fr_addr_out), + .ready_out(fr_ready_out), + + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + .ddr_sre(ddr_sre), + .ddr_srx(ddr_srx), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata) + ); + + wire frontend_ready; + + reg keep_frontend_reset = 1'b1; + + `ifdef ENABLE_DLL_TOGGLER + always @(posedge c0_ddr4_clk) begin + keep_frontend_reset <= c0_init_calib_complete_r; + end + `endif + + frontend #(.SIM_MEM("true"))frontend( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || ~keep_frontend_reset || ~c0_init_calib_complete_r || dllt_active), + + .init_calib_complete(c0_init_calib_complete_r), + .softmc_fin(softmc_fin), + .user_rst(user_rst), + //.dllt_begin(toggle_dll), + + // indicates read_back unit is ready for the next iseq + .frontend_ready(frontend_ready), + + // frontend <-> fetch stage if + .addr_in(fr_addr_in), + .valid_in(fr_valid_in), + .data_out(fr_data_out), + .valid_out(fr_valid_out), + .addr_out(fr_addr_out), + .ready_in(fr_ready_out), + + // frontend <-> xdma interface + .h2c_tdata_0(m_axis_h2c_tdata_0), + .h2c_tlast_0(m_axis_h2c_tlast_0), + .h2c_tvalid_0(m_axis_h2c_tvalid_0), + .h2c_tready_0(m_axis_h2c_tready_0), + .h2c_tkeep_0(m_axis_h2c_tkeep_0), + + .per_rd_init(per_rd_init), + .per_zq_init(per_zq_init), + .rbe_switch_mode(rbe_switch_mode) + ); + + ddr4_adapter #( + .DQ_WIDTH(`DQ_WIDTH) + ) ddr4_adapter + ( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst || ~c0_init_calib_complete_r), + .init_calib_complete(c0_init_calib_complete_r), + //.io_config_strobe, + //.io_config, + .dBufAdr(dBufAdr), // Reserved. Should be tied low. + .wrData(wrData), // DRAM write data. There are 8 bits for each DQ lane on the DRAM bus. + .wrDataMask(wrDataMask),// DRAM write DM/DBI port.There is one bit for each byte of the wrData port. + .wrDataEn(wrDataEn), // Write data Enable. The Phy will assert this port for one cycle for each write CAS command. + .mc_ACT_n(mc_ACT_n), // DRAM ACT_n command signal for four DRAM clock cycles. + .mc_ADR(mc_ADR), // DRAM address. There are 8 bits in the fabric interface for each address bit on the DRAM bus. + .mc_BA(mc_BA), // DRAM bank address. 8 bits for each DRAM bank address. + .mc_BG(mc_BG), // DRAM bank group address. + .mc_CS_n(mc_CS_n), // DRAM CS_n + //.mc_CKE(mc_CKE), // DRAM CKE + //.mc_ODT(mc_ODT), // DRAM ODT + .mcRdCAS(mcRdCAS), // Read CAS command issued. + .mcWrCAS(mcWrCAS), // Write CAS command issued. + .winRank(winRank), // Target rank for CAS commands. This value indicates which rank a CAS command is issued to. + .winBuf(winBuf), // Optional control signal. When either mcRdCAS or mcWrCAS is asserted, the Phy will store the value on the winBuf signal. + //.rdData(rdData), // DRAM read data. + .rdDataEn(rdDataEn), // Read data valid. This signal asserts for one fabric cycle for each completed read operation. + .rdDataEnd(rdDataEnd), // Unused. Tied high. + .mcCasSlot(mcCasSlot), + .mcCasSlot2(mcCasSlot2), + .gt_data_ready(gt_data_ready), + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + //.ddr_sre(ddr_sre), + //.ddr_srx(ddr_srx), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata), + + .ddr_maint_read(per_rd_init) + ); + + localparam ODTWRDEL = 5'd9; + localparam ODTWRDUR = 4'd6; + localparam ODTWRODEL = 5'd9; + localparam ODTWRODUR = 4'd6; + localparam ODTRDDEL = 5'd10; + localparam ODTRDDUR = 4'd6; + localparam ODTRDODEL = 5'd9; + localparam ODTRDODUR = 4'd6; + localparam ODTNOP = 16'h0000; + localparam ODTWR = 16'h0001; + localparam ODTRD = 16'h0000; + + + wire tranSentC; + assign tranSentC = mcRdCAS | mcWrCAS; + + //synthesis translate_on + //******************************************************************************* + ddr4_mc_odt # ( + .ODTWR (ODTWR) + ,.ODTWRDEL (ODTWRDEL) + ,.ODTWRDUR (ODTWRDUR) + ,.ODTWRODEL (ODTWRODEL) + ,.ODTWRODUR (ODTWRODUR) + + ,.ODTRD (ODTRD) + ,.ODTRDDEL (ODTRDDEL) + ,.ODTRDDUR (ODTRDDUR) + ,.ODTRDODEL (ODTRDODEL) + ,.ODTRDODUR (ODTRDODUR) + + ,.ODTNOP (ODTNOP) + ,.ODTBITS (`ODT_WIDTH) + ,.TCQ (0.1) + )u_ddr_tb_odt( + .clk (c0_ddr4_clk) + ,.rst (c0_ddr4_rst) + ,.mc_ODT (mc_ODT) + ,.casSlot (mcCasSlot) + ,.casSlot2 (mcCasSlot2) + ,.rank (winRank) + ,.winRead (mcRdCAS) + ,.winWrite (mcWrCAS) + ,.tranSentC (tranSentC) + ); + + readback_engine rbe( + + // common signals + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst || ~c0_init_calib_complete_r), + + // other ctrl signals + .flush(frontend_ready), + .switch_mode(rbe_switch_mode), + .read_seq_incoming(read_seq_incoming), // next few instructions will read from DRAM + .incoming_reads(incoming_reads), // how many reads next few instructions will issue + .buffer_space(buffer_space), // remaining buffer size + // DRAM <-> engine if + .rd_data(rdData), + .rd_valid(rdDataEn), + + // rbe <-> rf interface + .ddr_wdata(ddr_wdata), + + .per_rd_init(per_rd_init), + .per_zq_init(per_zq_init), + + // rbe <-> xdma if + .c2h_tdata_0(s_axis_c2h_tdata_0), + .c2h_tlast_0(s_axis_c2h_tlast_0), + .c2h_tvalid_0(s_axis_c2h_tvalid_0), + .c2h_tready_0(1'b1), + .c2h_tkeep_0(s_axis_c2h_tkeep_0) + + ); + + `ifdef ENABLE_DLL_TOGGLER + dll_toggler dllt + ( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst || ~c0_init_calib_complete_r), + .toggle_valid(toggle_dll), + .dllt_done(dllt_done), + .mc_ACT_n(dllt_mc_ACT_n), // DRAM ACT_n command signal for four DRAM clock cycles. + .mc_ADR(dllt_mc_ADR), // DRAM address. There are 8 bits in the fabric interface for each address bit on the DRAM bus. + .mc_BA(dllt_mc_BA), // DRAM bank address. 8 bits for each DRAM bank address. + .mc_BG(dllt_mc_BG), // DRAM bank group address. + .mc_CS_n(dllt_mc_CS_n), // DRAM CS_n + .mc_CKE(dllt_mc_CKE), + .clk_sel(clk_sel) + ); + `endif +endmodule diff --git a/sources/hdl/sim_verilog/tb_decode_stage.v b/sources/hdl/sim_verilog/tb_decode_stage.v new file mode 100644 index 0000000..1842cf7 --- /dev/null +++ b/sources/hdl/sim_verilog/tb_decode_stage.v @@ -0,0 +1,26 @@ +module tb_decode_stage( + + ); + + reg [63:0] instr; + reg [11:0] instr_pc; + reg instr_valid; + + decode_stage ds( + .instr(instr), + .instr_pc(instr_pc), + .instr_valid(instr_valid) + ); + + reg clk = 0, rst = 1; + + initial begin + #100; + rst = 0; + end + + always begin + #5; + clk = ~clk; + end +endmodule diff --git a/sources/hdl/sim_verilog/tb_frontend.v b/sources/hdl/sim_verilog/tb_frontend.v new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/sources/hdl/sim_verilog/tb_frontend.v diff --git a/sources/hdl/sim_verilog/tb_softmc_top.v b/sources/hdl/sim_verilog/tb_softmc_top.v new file mode 100644 index 0000000..6a61f44 --- /dev/null +++ b/sources/hdl/sim_verilog/tb_softmc_top.v @@ -0,0 +1,42 @@ +`timescale 1ns / 1ps +////////////////////////////////////////////////////////////////////////////////// +// Company: +// Engineer: +// +// Create Date: 12/06/2018 05:43:09 PM +// Design Name: +// Module Name: tb_softmc_top +// Project Name: +// Target Devices: +// Tool Versions: +// Description: +// +// Dependencies: +// +// Revision: +// Revision 0.01 - File Created +// Additional Comments: +// +////////////////////////////////////////////////////////////////////////////////// + + +module tb_only_softmc( + + ); + + reg clk = 0, rst = 1; + softmc_top smct( + .clk(clk), + .rst(rst) + ); + + initial begin + #50; + rst = 0; + end + + always begin + #5; + clk = ~clk; + end +endmodule diff --git a/sources/hdl/verilog/ddr_pipeline.v b/sources/hdl/verilog/ddr_pipeline.v new file mode 100644 index 0000000..f8ae5aa --- /dev/null +++ b/sources/hdl/verilog/ddr_pipeline.v @@ -0,0 +1,206 @@ +`include "parameters.vh" +`include "encoding.vh" + +module ddr_pipeline( + + // common signals + input clk, + input rst, + + // execute_stage <-> ddr_pipeline if + input ddr_valid, + input [`DDR_UOP_WIDTH*4-1:0] ddr_uop, + + // ddr_pipeline <-> outer DDRX IP interface + output [3:0] ddr_write, + output [3:0] ddr_read, + output [3:0] ddr_pre, + output [3:0] ddr_act, + output [3:0] ddr_ref, + output [3:0] ddr_zq, + output [3:0] ddr_nop, + output [3:0] ddr_sre, + output [3:0] ddr_srx, + output [3:0] ddr_ap, + output [3:0] ddr_pall, + output [3:0] ddr_half_bl, + output [4*`BG_WIDTH-1:0] ddr_bg, + output [4*`BANK_WIDTH-1:0] ddr_bank, + output [4*`COL_WIDTH-1:0] ddr_col, + output [4*`ROW_WIDTH-1:0] ddr_row, + output [511:0] ddr_wdata, + + // ddr_pipeline <-> regfile interface + input [511:0] wide_reg, + output [7:0] update_en, + output [4*8-1:0] update_ids, + output [32*8-1:0] update_vals, + input [`COL_WIDTH-1:0] casr, + input [`BANK_WIDTH+`BG_WIDTH-1:0] basr, + input [`ROW_WIDTH-1:0] rasr, + output [4*8-1:0] reg_ids, // registers we need to read + input [32*8-1:0] reg_vals // register values + ); + + //Split input uop into four ways + wire [`DDR_UOP_WIDTH-1:0] uop [3:0]; + genvar uops; + generate + for(uops = 0 ; uops < 4 ; uops = uops + 1) begin: split_uops + assign uop[uops] = ddr_uop[`DDR_UOP_WIDTH*uops +: `DDR_UOP_WIDTH]; + end + endgenerate + + // Figure out reg ids in the bubble cycle + reg [4*8-1:0] reg_ids_ns, reg_ids_r; + + reg s2_valid; + reg [`DDR_UOP_WIDTH-1:0] s2_uop [3:0]; + reg [7:0] s2_update_en; // which registers will we update + reg [8*32-1:0] s2_update_val; + + reg [3:0] ddr_write_ns, ddr_write_r; + reg [3:0] ddr_read_ns, ddr_read_r; + reg [3:0] ddr_pre_ns, ddr_pre_r; + reg [3:0] ddr_act_ns, ddr_act_r; + reg [3:0] ddr_ref_ns, ddr_ref_r; + reg [3:0] ddr_sre_ns, ddr_sre_r; + reg [3:0] ddr_srx_ns, ddr_srx_r; + reg [3:0] ddr_zq_ns, ddr_zq_r; + reg [3:0] ddr_nop_ns, ddr_nop_r; + reg [3:0] ddr_ap_ns, ddr_ap_r; + reg [3:0] ddr_pall_ns, ddr_pall_r; + reg [3:0] ddr_half_bl_ns, ddr_half_bl_r; + reg [4*`BG_WIDTH-1:0] ddr_bg_ns, ddr_bg_r; + reg [4*`BANK_WIDTH-1:0] ddr_bank_ns, ddr_bank_r; + reg [4*`COL_WIDTH-1:0] ddr_col_ns, ddr_col_r; + reg [4*`ROW_WIDTH-1:0] ddr_row_ns, ddr_row_r; + reg [511:0] ddr_data_r; + + assign update_vals = s2_update_val; + assign update_en = s2_update_en; + assign update_ids = reg_ids_r; + assign reg_ids = reg_ids_r; + // these values are registered before + // being offloaded to the outer module + assign ddr_write = ddr_write_r; + assign ddr_read = ddr_read_r; + assign ddr_pre = ddr_pre_r; + assign ddr_act = ddr_act_r; + assign ddr_ref = ddr_ref_r; + assign ddr_sre = ddr_sre_r; + assign ddr_srx = ddr_srx_r; + assign ddr_zq = ddr_zq_r; + assign ddr_nop = ddr_nop_r; + assign ddr_ap = ddr_ap_r; + assign ddr_pall = ddr_pall_r; + assign ddr_half_bl = ddr_half_bl_r; + assign ddr_bg = ddr_bg_r; + assign ddr_bank = ddr_bank_r; + assign ddr_col = ddr_col_r; + assign ddr_row = ddr_row_r; + assign ddr_wdata = ddr_data_r; + + integer i; + always @* begin + ddr_nop_ns = {4{`HIGH}}; + s2_update_en = {4{`LOW}}; + reg_ids_ns = reg_ids_r; + s2_update_val = 8*32'bX; + for(i = 0 ; i < 4 ; i = i + 1) begin: gen_ddrx_sigs + // Decide which registers to read in stage 1 + if(uop[i][`IS_WRITE] || uop[i][`IS_READ]) begin + reg_ids_ns[i*8 +: 4] = uop[i][`BAR +: 4]; + reg_ids_ns[i*8+4 +: 4] = uop[i][`CAR +: 4]; + end + if(uop[i][`IS_PRE]) begin + reg_ids_ns[i*8 +: 4] = uop[i][`BAR +: 4]; + end + if(uop[i][`IS_ACT]) begin + reg_ids_ns[i*8 +: 4] = uop[i][`BAR +: 4]; + reg_ids_ns[i*8+4 +: 4] = uop[i][`RAR +: 4]; + end + // stage 2 control sigs + ddr_pall_ns[i] = s2_uop[i][`PRE_ALL] & s2_uop[i][`IS_PRE]; + ddr_write_ns[i] = s2_uop[i][`IS_WRITE]; + ddr_read_ns[i] = s2_uop[i][`IS_READ]; + ddr_pre_ns[i] = s2_uop[i][`IS_PRE]; + ddr_act_ns[i] = s2_uop[i][`IS_ACT]; + ddr_zq_ns[i] = s2_uop[i][`IS_ZQ]; + ddr_ref_ns[i] = s2_uop[i][`IS_REF]; + ddr_sre_ns[i] = s2_uop[i][`IS_SRE]; + ddr_srx_ns[i] = s2_uop[i][`IS_SRX]; + ddr_ap_ns[i] = s2_uop[i][`DO_AP] & (s2_uop[i][`IS_WRITE] | s2_uop[i][`IS_READ]); + ddr_half_bl_ns[i] = s2_uop[i][`IS_BL4] & (s2_uop[i][`IS_WRITE] | s2_uop[i][`IS_READ]); + ddr_nop_ns[i] = s2_valid ? s2_uop[i][`IS_NOP] : {4{`HIGH}}; + // stage 2 address calculation + ddr_row_ns[i*`ROW_WIDTH +: `ROW_WIDTH] + = reg_vals[i*64+32 +: `ROW_WIDTH]; + ddr_bank_ns[i*`BANK_WIDTH +: `BANK_WIDTH] + = reg_vals[i*64 +: `BANK_WIDTH]; + ddr_bg_ns[i*`BG_WIDTH +: `BG_WIDTH] + = reg_vals[i*64+`BANK_WIDTH +: `BG_WIDTH]; + ddr_col_ns[i*`COL_WIDTH +: `COL_WIDTH] + = reg_vals[i*64+32 +: `COL_WIDTH]; + // stage 2 reg update + if(s2_uop[i][`INC_CAR] & (s2_uop[i][`IS_WRITE] | s2_uop[i][`IS_READ])) begin + s2_update_en[i*2+1] = `HIGH; + s2_update_val[i*64+32 +: 32] = reg_vals[i*64+32 +: `COL_WIDTH] + + casr; + end + else if(s2_uop[i][`INC_RAR] & (s2_uop[i][`IS_ACT])) begin + s2_update_en[i*2+1] = `HIGH; + s2_update_val[i*64+32 +: 32] = reg_vals[i*64+32 +: `ROW_WIDTH] + + rasr; + end + if(s2_uop[i][`INC_BAR]) begin + s2_update_en[i*2] = `HIGH; + s2_update_val[i*64 +: 32] = reg_vals[i*64 +: `BANK_WIDTH+`BG_WIDTH] + + basr; + end + end // for + end + + always @(posedge clk) begin + if(rst) begin + s2_valid <= `LOW; + for(i = 0 ; i < 4 ; i = i + 1) + s2_uop[i] <= {`DDR_UOP_WIDTH{`LOW}}; + ddr_write_r <= {4{`LOW}}; + ddr_read_r <= {4{`LOW}}; + ddr_pre_r <= {4{`LOW}}; + ddr_act_r <= {4{`LOW}}; + ddr_ref_r <= {4{`LOW}}; + ddr_sre_r <= {4{`LOW}}; + ddr_srx_r <= {4{`LOW}}; + ddr_zq_r <= {4{`LOW}}; + ddr_nop_r <= {4{`HIGH}}; + ddr_ap_r <= {4{`LOW}}; + ddr_half_bl_r <= {4{`LOW}}; + end + else begin + reg_ids_r <= reg_ids_ns; + s2_valid <= ddr_valid; + for(i = 0 ; i < 4 ; i = i + 1) + s2_uop[i] <= uop[i]; + ddr_write_r <= ddr_write_ns; + ddr_read_r <= ddr_read_ns; + ddr_pre_r <= ddr_pre_ns; + ddr_act_r <= ddr_act_ns; + ddr_ref_r <= ddr_ref_ns; + ddr_sre_r <= ddr_sre_ns; + ddr_srx_r <= ddr_srx_ns; + ddr_zq_r <= ddr_zq_ns; + ddr_nop_r <= ddr_nop_ns; + ddr_ap_r <= ddr_ap_ns; + ddr_pall_r <= ddr_pall_ns; + ddr_half_bl_r <= ddr_half_bl_ns; + ddr_bg_r <= ddr_bg_ns; + ddr_bank_r <= ddr_bank_ns; + ddr_col_r <= ddr_col_ns; + ddr_row_r <= ddr_row_ns; + ddr_data_r <= wide_reg; + end + end + +endmodule diff --git a/sources/hdl/verilog/decode_stage.v b/sources/hdl/verilog/decode_stage.v new file mode 100644 index 0000000..7f40cba --- /dev/null +++ b/sources/hdl/verilog/decode_stage.v @@ -0,0 +1,318 @@ +`include "parameters.vh" +`include "encoding.vh" + +module decode_stage( + + input clk, + input rst, + + // fetch stage <-> decode stage interface + input [`INSTR_WIDTH-1:0] instr, + input [`IMEM_ADDR_WIDTH-1:0] instr_pc, + input instr_valid, + + // decode stage <-> execute stage interface + output ddr_valid, + output exe_valid, + output [`DDR_UOP_WIDTH*4-1:0] ddr_uop, + output [`EXE_UOP_WIDTH-1:0] exe_uop, + output [`IMEM_ADDR_WIDTH-1:0] exe_pc, + output [32*7-1:0] ddr_stat + ); + + reg ddr_valid_r, ddr_valid_ns; + reg exe_valid_r, exe_valid_ns; + reg [`IMEM_ADDR_WIDTH-1:0] exe_pc_r; + + reg [`DDR_UOP_WIDTH-1:0] ddr_uop_r [3:0], ddr_uop_ns [3:0]; + reg [`EXE_UOP_WIDTH-1:0] exe_uop_r, exe_uop_ns; + reg[31:0] ddr_stat_r[0:6]; //WRITE, READ, PRE, ACT, ZQ, REF, CYC counts + + // split the instr into four individual insts + localparam div = 16; + wire [`INSTR_WIDTH/4-1:0] ddr_insts [3:0]; + assign ddr_insts[0] = instr [0 +: div]; + assign ddr_insts[1] = instr [div +: div]; + assign ddr_insts[2] = instr [div*2 +: div]; + assign ddr_insts[3] = instr [div*3 +: div]; + // gather all ddr uops into one output (¯\_(ツ)_/¯) + // verilog does not let us define output vectors + genvar uops; + generate + for(uops = 0 ; uops < 4 ; uops = uops + 1) begin: gather_uops + assign ddr_uop[uops*`DDR_UOP_WIDTH +: `DDR_UOP_WIDTH] = + ddr_uop_r[uops]; + end + endgenerate + assign exe_uop = exe_uop_r; + assign exe_pc = exe_pc_r; + assign ddr_valid = ddr_valid_r; + assign exe_valid = exe_valid_r; + assign ddr_stat = {ddr_stat_r[6], ddr_stat_r[5], + ddr_stat_r[4], ddr_stat_r[3], + ddr_stat_r[2], ddr_stat_r[1], + ddr_stat_r[0]}; + + reg is_started = 0; + reg[26:0] imd; + integer i; + + always @* begin + ddr_valid_ns = `LOW; + exe_valid_ns = `LOW; + exe_uop_ns = {`EXE_UOP_WIDTH{`LOW}}; + for(i = 0 ; i < 4 ; i = i + 1) + ddr_uop_ns[i] = {`DDR_UOP_WIDTH{`LOW}}; + if(instr_valid) begin + // Decoding a DDR packet + if(instr[`DDR_OFFSET]) begin + ddr_valid_ns = `HIGH; + for(i = 0 ; i < 4 ; i = i + 1) begin: gen_ddr_uops + case(ddr_insts[i][`DDR_CODE_OFFSET +: 3]) + `WRITE: begin + ddr_uop_ns[i][`IS_WRITE] = `HIGH; + ddr_uop_ns[i][`INC_CAR] = ddr_insts[i][`DEC_INC_CAR]; + ddr_uop_ns[i][`INC_BAR] = ddr_insts[i][`DEC_INC_BAR]; + ddr_uop_ns[i][`CAR+:4] = ddr_insts[i][`DEC_CAR+:4]; + ddr_uop_ns[i][`BAR+:4] = ddr_insts[i][`DEC_BAR+:4]; + ddr_uop_ns[i][`DO_AP] = ddr_insts[i][`DEC_AP]; + ddr_uop_ns[i][`IS_BL4] = ddr_insts[i][`DEC_BL4]; + end + `READ: begin + ddr_uop_ns[i][`IS_READ] = `HIGH; + ddr_uop_ns[i][`INC_CAR] = ddr_insts[i][`DEC_INC_CAR]; + ddr_uop_ns[i][`INC_BAR] = ddr_insts[i][`DEC_INC_BAR]; + ddr_uop_ns[i][`CAR+:4] = ddr_insts[i][`DEC_CAR+:4]; + ddr_uop_ns[i][`BAR+:4] = ddr_insts[i][`DEC_BAR+:4]; + ddr_uop_ns[i][`DO_AP] = ddr_insts[i][`DEC_AP]; + ddr_uop_ns[i][`IS_BL4] = ddr_insts[i][`DEC_BL4]; + end + `PRE: begin + ddr_uop_ns[i][`IS_PRE] = `HIGH; + ddr_uop_ns[i][`PRE_ALL] = ddr_insts[i][`DEC_PRE_ALL]; + ddr_uop_ns[i][`INC_BAR] = ddr_insts[i][`DEC_INC_BAR]; + ddr_uop_ns[i][`BAR+:4] = ddr_insts[i][`DEC_BAR+:4]; + end + `ACT: begin + ddr_uop_ns[i][`IS_ACT] = `HIGH; + ddr_uop_ns[i][`INC_RAR] = ddr_insts[i][`DEC_INC_RAR]; + ddr_uop_ns[i][`INC_BAR] = ddr_insts[i][`DEC_INC_BAR]; + ddr_uop_ns[i][`RAR+:4] = ddr_insts[i][`DEC_RAR+:4]; + ddr_uop_ns[i][`BAR+:4] = ddr_insts[i][`DEC_BAR+:4]; + end + `ZQ: begin + ddr_uop_ns[i][`IS_ZQ] = `HIGH; + end + `REF: begin + ddr_uop_ns[i][`IS_REF] = `HIGH; + end + `NOP: begin + ddr_uop_ns[i][`IS_NOP] = `HIGH; + end + endcase + end + end + else if(instr[`SR_OFFSET]) begin + ddr_valid_ns = `HIGH; + case(instr[`FU_CODE_OFFSET]) + `SRE: begin + ddr_uop_ns[0][`IS_SRE] = `HIGH; + ddr_uop_ns[1][`IS_NOP] = `HIGH; + ddr_uop_ns[2][`IS_NOP] = `HIGH; + ddr_uop_ns[3][`IS_NOP] = `HIGH; + end + `SRX: begin + ddr_uop_ns[0][`IS_SRX] = `HIGH; + ddr_uop_ns[1][`IS_NOP] = `HIGH; + ddr_uop_ns[2][`IS_NOP] = `HIGH; + ddr_uop_ns[3][`IS_NOP] = `HIGH; + end + endcase + end + else begin + exe_valid_ns = `HIGH; + // Decoding a branch instruction + if(instr[`BRANCH_OFFSET]) begin + case(instr[`FU_CODE_OFFSET +: 8]) + `BL: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`IMD +: 19] = instr[`DEC_BR_TGT_OFFSET +: 19]; + exe_uop_ns[`IS_BL] = `HIGH; + end + `BEQ: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`IMD +: 19] = instr[`DEC_BR_TGT_OFFSET +: 19]; + exe_uop_ns[`IS_BEQ] = `HIGH; + end + `JUMP: begin + exe_uop_ns[`IMD +: 27] = instr[`DEC_JUMP_OFFSET +: 27]; + exe_uop_ns[`IS_JUMP] = `HIGH; + end + `SLEEP: begin + exe_uop_ns[`IS_SLEEP] = `HIGH; + exe_uop_ns[`IMD +: 27] = instr[`DEC_SLEEP_OFFSET +: 27]; + end + endcase + end + else if(instr[`MEM_OFFSET]) begin + case(instr[`FU_CODE_OFFSET +: 8]) + `LD: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IMD +: 16] = instr[`DEC_IMD1 +: 16]; + exe_uop_ns[`IS_MEM] = `HIGH; + exe_uop_ns[`IS_LD] = `HIGH; + exe_uop_ns[`HAS_IMD] = `HIGH; + end + `ST: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; // The address base to write to + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RT +: 4]; // The value to write + exe_uop_ns[`IMD +: 16] = instr[`DEC_IMD1 +: 16]; + exe_uop_ns[`IS_MEM] = `HIGH; + exe_uop_ns[`IS_ST] = `HIGH; + exe_uop_ns[`HAS_IMD] = `HIGH; + end + endcase + end + else if(instr[`BW_OFFSET]) begin + case(instr[`FU_CODE_OFFSET +: 8]) + `AND: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_AND] = `HIGH; + end + `OR: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_OR] = `HIGH; + end + `XOR: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_XOR] = `HIGH; + end + endcase + end + // Decoding a normal instruction + else begin + case(instr[`FU_CODE_OFFSET +: 8]) + `ADD: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_ADD] = `HIGH; + end + `ADDI: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`IMD +: 16] = instr[`DEC_IMD1 +: 16]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_ADD] = `HIGH; + exe_uop_ns[`HAS_IMD] = `HIGH; + end + `SUB: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RS2 +: 4] = instr[`DEC_RS2 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_SUB] = `HIGH; + end + `SUBI: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`IMD +: 16] = instr[`DEC_IMD1 +: 16]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_SUB] = `HIGH; + exe_uop_ns[`HAS_IMD] = `HIGH; + end + `MV: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_MOV] = `HIGH; + end + `SRC: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_SRC] = `HIGH; + end + `LI: begin + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_LI] = `HIGH; + exe_uop_ns[`IMD +: 16] = instr[`DEC_IMD1 +: 16]; + exe_uop_ns[`IMD2 +: 16]= instr[`DEC_IMD3 +: 16]; + exe_uop_ns[`HAS_IMD] = `HIGH; + end + `LDWD: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + // used to select word offset in wide write data register + exe_uop_ns[`RT +: 4] = instr[`DEC_WO +: 4]; + exe_uop_ns[`IS_LDWD] = `HIGH; + end + `LDPC: begin + exe_uop_ns[`RS1 +: 4] = instr[`DEC_RS1 +: 4]; + exe_uop_ns[`RT +: 4] = instr[`DEC_RT +: 4]; + exe_uop_ns[`IS_LDPC] = `HIGH; + end + endcase + end + end + end + end + + always @(posedge clk) begin + + if(rst) begin + exe_pc_r <= {`IMEM_ADDR_WIDTH{`LOW}}; + exe_uop_r <= {`EXE_UOP_WIDTH{`LOW}}; + exe_valid_r <= `LOW; + ddr_valid_r <= `LOW; + for(i = 0 ; i < 4 ; i = i + 1) + ddr_uop_r[i] <= {`DDR_UOP_WIDTH{`LOW}}; + is_started <= `LOW; + end + else begin + exe_uop_r <= exe_uop_ns; + exe_valid_r <= exe_valid_ns; + ddr_valid_r <= ddr_valid_ns; + for(i = 0 ; i < 4 ; i = i + 1) + ddr_uop_r[i] <= ddr_uop_ns[i]; + exe_pc_r <= instr_pc; + end + + if(~is_started) begin + for(i = 0 ; i < 7 ; i = i + 1) begin + ddr_stat_r[i] <= 0; + end + if(instr_pc == 1) + is_started <= 1; + end + else begin + ddr_stat_r[6] <= ddr_stat_r[6] + 1; + if(instr_valid) begin + if(instr[`DDR_OFFSET]) begin + for(i = 0 ; i < 4 ; i = i + 1) begin + case(ddr_insts[i][`DDR_CODE_OFFSET +: 3]) + `WRITE: + ddr_stat_r[0] <= ddr_stat_r[0] + 1; + `READ: + ddr_stat_r[1] <= ddr_stat_r[1] + 1; + `PRE: + ddr_stat_r[2] <= ddr_stat_r[2] + 1; + `ACT: + ddr_stat_r[3] <= ddr_stat_r[3] + 1; + `ZQ: + ddr_stat_r[4] <= ddr_stat_r[4] + 1; + `REF: + ddr_stat_r[5] <= ddr_stat_r[5] + 1; + endcase + end + end + end + end + end + + + +endmodule diff --git a/sources/hdl/verilog/diff_shift_reg.v b/sources/hdl/verilog/diff_shift_reg.v new file mode 100644 index 0000000..ae39c3a --- /dev/null +++ b/sources/hdl/verilog/diff_shift_reg.v @@ -0,0 +1,69 @@ +`timescale 1ns / 1ps +////////////////////////////////////////////////////////////////////////////////// +// Company: +// Engineer: +// +// Create Date: 12/19/2018 01:07:06 PM +// Design Name: +// Module Name: diff_shift_reg +// Project Name: +// Target Devices: +// Tool Versions: +// Description: +// +// Dependencies: +// +// Revision: +// Revision 0.01 - File Created +// Additional Comments: +// +////////////////////////////////////////////////////////////////////////////////// + + +module diff_shift_reg( + input clk, + input rst, + + input flush, + + input [15:0] in, + input in_valid, + + output[511:0] out, + output out_valid + ); + + reg [4:0] ctr_r; + reg [511:0] shift_r; + reg valid_r; + + always @(posedge clk) begin + if(rst) begin + shift_r <= 512'bX; + ctr_r <= 5'b0; + valid_r <= 1'b0; + end + else begin + if(flush) begin + valid_r <= `HIGH; + ctr_r <= 5'b0; + end + else begin + if(in_valid) begin + shift_r[16 +: 16*31] <= shift_r[0 +: 16*31]; + shift_r[0 +: 16] <= in; + ctr_r <= ctr_r + 1; + end + else begin + ctr_r <= ctr_r; + shift_r <= shift_r; + end + valid_r <= (|ctr_r) && in_valid; + end + end + end + + assign out_valid = valid_r; + assign out = shift_r; + +endmodule diff --git a/sources/hdl/verilog/dll_toggler.v b/sources/hdl/verilog/dll_toggler.v new file mode 100644 index 0000000..a6cb5bf --- /dev/null +++ b/sources/hdl/verilog/dll_toggler.v @@ -0,0 +1,206 @@ +`include "parameters.vh" + +module dll_toggler #(parameter CKE_WIDTH = 1, RANK_WIDTH = 1, DQ_WIDTH = 64, DRAM_CMD_SLOTS = 4, + DATA_BUF_ADDR_WIDTH = 5, DBUF_WIDTH = 4, DQ_BURST = 8) + +( + + input clk, + input rst, + input toggle_valid, + output reg dllt_done, + + // ---------- DDR4 Signals ---------- + output [7:0] mc_ACT_n, // DRAM ACT_n command signal for four DRAM clock cycles. + output [`ADDR_WIDTH*8-1:0] mc_ADR, // DRAM address. There are 8 bits in the fabric interface for each address bit on the DRAM bus. + output [`BANK_WIDTH*8-1:0] mc_BA, // DRAM bank address. 8 bits for each DRAM bank address. + output [`BG_WIDTH*8-1:0] mc_BG, // DRAM bank group address. + output [`CS_WIDTH*8-1:0] mc_CS_n, // DRAM CS_n + + // NOTE: CKE is transmitted within another clock domain that is 4x faster than fabric clock. + output [`CKE_WIDTH*8-1:0] mc_CKE, + output clk_sel + + ); + + wire [13:0] MR1_CONF = 14'b00001100000000; + + reg [`ADDR_WIDTH*8-1:0] ADR_ns, ADR_r; + reg [`BANK_WIDTH*8-1:0] BA_ns, BA_r; + reg [`BG_WIDTH*8-1:0] BG_ns, BG_r; + reg [`CS_WIDTH*8-1:0] CS_n_ns, CS_n_r; + reg [`ODT_WIDTH*8-1:0] ODT_ns, ODT_r; + reg [`CKE_WIDTH*8-1:0] CKE_ns, CKE_r; + + reg clk_sel_ns, clk_sel_r; + + assign mc_ACT_n = {8{`HIGH}}; + assign mc_ADR = ADR_r; + assign mc_BA = BA_r; + assign mc_BG = BG_r; + assign mc_CS_n = CS_n_r; + assign mc_CKE = CKE_r; + assign clk_sel = clk_sel_r; + + + // TODO we need to implement the following routine (TO TURN DLL OFF ONLY) + // 1 - Precharge all banks (IDLE STATE) + // 2 - Set MR1 A0 to 1 (DISABLE DLL), wait tMOD + // 3 - Enter self-refresh mode, wait until tCKSRE/tCKSRE_PAR + // 4 - Change clock frequency + // 5 - Wait at least tCKSRX (until clock signal stabilizes) + // 6 - Exit self-refresh mode, keep CKE high from now on + // if any ODT feature was enabled in self-ref. mode + // ODT signal must be LOW. + // 7 - Wait tXS and set mode registers to appropriate values + // (UG says that CL, CWL and WR may need to be updated), + // wait for another tMOD + + localparam IDLE_S = 0; + localparam IDLE_WAIT_S = 1; + localparam SET_MR_1_S = 2; + localparam WAIT_MR_1_S = 3; + localparam ENTER_SELF_REF_S = 4; + localparam WAIT_ENTER_SELF_REF_S = 5; + localparam CHANGE_CLK_FREQ_S = 6; + localparam WAIT_CHANGE_CLK_FREQ_S = 7; + localparam EXIT_SELF_REF_S = 8; + localparam WAIT_EXIT_SELF_REF_S = 9; + + localparam T_PRECHARGE = 5; // in terms of MC cycles (which is 4x less frequent than the DDR4) + localparam T_MOD = 24; // Max(24CK,15ns) + localparam T_CKSRE = 300; // Max(5CK,10ns) + localparam T_CKSRX = 300; // Max(5CK, 10ns) + additional room for clock to stabilize; + localparam T_XS = 1000; + + + reg[3:0] state_r, state_ns; + + reg[9:0] wait_r, wait_ns; + + integer adr_bit_i; + + always @* begin + // Set bank and bank group signals + dllt_done = `LOW; + ADR_ns = {`ADDR_WIDTH*8{`HIGH}}; + BG_ns = {`BG_WIDTH*8{`LOW}}; + BA_ns = {`BANK_WIDTH*8{`LOW}}; + CS_n_ns = {`CS_WIDTH*8{`HIGH}}; // by default we don't issue any commands + CKE_ns = CKE_r; // register this signal because it needs to be LOW during self-ref. + wait_ns = wait_r; + state_ns = state_r; + clk_sel_ns = clk_sel_r; + case (state_r) + IDLE_S: begin + if(toggle_valid) begin + CS_n_ns[1:0] = {2*`CS_WIDTH{`LOW}}; + ADR_ns[`ADDR_WIDTH*8-3*8 +: 2] = {2{`LOW}}; // WE + ADR_ns[`ADDR_WIDTH*8-2*8 +: 2] = {2{`HIGH}}; // ~CAS + ADR_ns[`ADDR_WIDTH*8-8 +: 2] = {2{`LOW}}; // RAS + ADR_ns[10*8 +: 2] = {2{`HIGH}}; // Pre ALL + wait_ns = T_PRECHARGE; + state_ns = IDLE_WAIT_S; + end + end + IDLE_WAIT_S: begin + if(wait_r > 0) + wait_ns = wait_r - 1'b1; + else + state_ns = SET_MR_1_S; + end + SET_MR_1_S: begin + CS_n_ns[1:0] = {2*`CS_WIDTH{`LOW}}; + ADR_ns[`ADDR_WIDTH*8-3*8 +: 2] = {2{`LOW}}; // WE + ADR_ns[`ADDR_WIDTH*8-2*8 +: 2] = {2{`LOW}}; // CAS + ADR_ns[`ADDR_WIDTH*8-8 +: 2] = {2{`LOW}}; // RAS + for(adr_bit_i = 0 ; adr_bit_i < 14 ; adr_bit_i = adr_bit_i + 1) begin + ADR_ns[adr_bit_i*8 +: 2] = + {2{MR1_CONF[adr_bit_i]}}; + end + // Bank + Bank group bits indicate which register this MRS is writing to. + BA_ns[0 +: 2] = {2{`HIGH}}; // Select MR1 + ADR_ns[0 +: 2] = {2{`LOW}}; // Set A0 to 0 + state_ns = WAIT_MR_1_S; + wait_ns = T_MOD; + end + WAIT_MR_1_S: begin + if(wait_r > 0) + wait_ns = wait_r - 1'b1; + else + state_ns = ENTER_SELF_REF_S; + end + ENTER_SELF_REF_S: begin + CKE_ns = `LOW; + CS_n_ns[1:0] = {2*`CS_WIDTH{`LOW}}; + ADR_ns[`ADDR_WIDTH*8-3*8 +: 2] = {2{`HIGH}}; // ~WE + ADR_ns[`ADDR_WIDTH*8-2*8 +: 2] = {2{`LOW}}; // CAS + ADR_ns[`ADDR_WIDTH*8-8 +: 2] = {2{`LOW}}; // RAS + state_ns = WAIT_ENTER_SELF_REF_S; + wait_ns = T_CKSRE; + end + WAIT_ENTER_SELF_REF_S: begin + if(wait_r > 0) + wait_ns = wait_r - 1'b1; + else begin + state_ns = CHANGE_CLK_FREQ_S; + end + end + CHANGE_CLK_FREQ_S: begin + clk_sel_ns = ~clk_sel_r; + wait_ns = T_CKSRX; + state_ns = WAIT_CHANGE_CLK_FREQ_S; + end + WAIT_CHANGE_CLK_FREQ_S: begin + if(wait_r > 0) + wait_ns = wait_r - 1'b1; + else begin + state_ns = EXIT_SELF_REF_S; + end + end + EXIT_SELF_REF_S: begin + CKE_ns = {`CKE_WIDTH*8{`HIGH}}; + CS_n_ns[7:0] = {8*`CS_WIDTH{`HIGH}}; + ADR_ns[`ADDR_WIDTH*8-3*8 +: 2] = {2{`HIGH}}; // ~WE + ADR_ns[`ADDR_WIDTH*8-2*8 +: 2] = {2{`HIGH}}; // CAS + ADR_ns[`ADDR_WIDTH*8-8 +: 2] = {2{`HIGH}}; // RAS + state_ns = WAIT_EXIT_SELF_REF_S; + wait_ns = T_XS; + end + WAIT_EXIT_SELF_REF_S: begin + CS_n_ns[7:0] = {8*`CS_WIDTH{`HIGH}}; + if(wait_r > 0) + wait_ns = wait_r - 1'b1; + else begin + state_ns = IDLE_S; + dllt_done = `HIGH; + end + end + endcase + + end + + always @(posedge clk) begin + if(rst) begin + state_r <= IDLE_S; + wait_r <= `LOW; + clk_sel_r <= `LOW; + CKE_r <= {`CKE_WIDTH*8{`HIGH}}; + ADR_r = {`ADDR_WIDTH*8{`LOW}}; + BG_r = {`BG_WIDTH*8{`LOW}}; + BA_r = {`BANK_WIDTH*8{`LOW}}; + CS_n_r = {`CS_WIDTH*8{`HIGH}}; + end + else begin + state_r <= state_ns; + wait_r <= wait_ns; + clk_sel_r <= clk_sel_ns; + CKE_r <= CKE_ns; + ADR_r <= ADR_ns; + BA_r <= BA_ns; + BG_r <= BG_ns; + CS_n_r <= CS_n_ns; + end + end + +endmodule diff --git a/sources/hdl/verilog/exe_pipeline.v b/sources/hdl/verilog/exe_pipeline.v new file mode 100644 index 0000000..e2111a3 --- /dev/null +++ b/sources/hdl/verilog/exe_pipeline.v @@ -0,0 +1,191 @@ +`include "parameters.vh" +`include "encoding.vh" + +module exe_pipeline( + // common signals + input clk, + input rst, + + // exe_pipeline <-> execution stage if + input exe_valid, + input [`EXE_UOP_WIDTH-1:0] exe_uop, + input [`IMEM_ADDR_WIDTH-1:0] exe_pc, + + // branch unit if + output br_resolve, + output [`IMEM_ADDR_WIDTH-1:0] br_target, + + // exe_pipeline <-> register file if + output wide_wen, + output [31:0] rf_wdata, + output rf_wen, + output [7:0] rf_raddr, + output [3:0] rf_waddr, + input [2*32-1:0] rf_rdata, + input [32*7-1:0] ddr_stat, + + // exe_pipeline <-> scratchpad + output mem_wen, + output mem_ren, + output [9:0] mem_addr, + output [31:0] mem_wdata, + input [31:0] mem_rdata + ); + + // calculated at stage one, registers + reg s2_valid; + reg [`EXE_UOP_WIDTH-1:0] s2_uop; + reg [`IMEM_ADDR_WIDTH-1:0] s2_pc; + reg [3:0] s2_rs1; + reg [3:0] s2_rs2; + reg [3:0] s2_rt; + reg s2_wen; + reg s2_mem_ren; + reg s2_mem_wen; + reg s3_wen; // for loads + reg [3:0] s3_rt; // for loads + reg s2_wide_wen; + reg [31:0] s2_imd_r, s2_imd_ns; + + // delayed branch resolution signals + // calculated at stage two, registers + reg s3_br_resolve; + reg [`IMEM_ADDR_WIDTH-1:0] s3_br_target; + + // combinational elements + reg [31:0] s2_wdata; + wire [31:0] s2_rs1_data, s2_rs2_data; + reg [31:0] s2_mem_wdata; + reg [`IMEM_ADDR_WIDTH-1:0] fetch_pc; + + assign wide_wen = s2_wide_wen; + assign rf_wdata = s3_wen ? mem_rdata : s2_wdata; + assign rf_wen = s2_wen || s3_wen; + assign rf_raddr[0+:4] = s2_rs1; + assign rf_raddr[4+:4] = s2_rs2; + assign rf_waddr = s3_wen ? s3_rt : s2_rt; + + assign s2_rs1_data = rf_rdata[0+:32]; + assign s2_rs2_data = rf_rdata[32+:32]; + assign br_resolve = s3_br_resolve; + assign br_target = s3_br_target; + + assign mem_addr = s2_rs1_data + s2_imd_r; + assign mem_wen = s2_mem_wen; + assign mem_ren = s2_mem_ren; + assign mem_wdata = s2_mem_wdata; + + always @* begin + // stage one, decode immediate value + s2_mem_wen = `LOW; + s2_imd_ns = s2_imd_r; + fetch_pc = {`IMEM_ADDR_WIDTH{1'bX}}; + s2_wdata = 32'bX; + s2_mem_wen = `LOW; + s2_mem_ren = `LOW; + if(exe_uop[`HAS_IMD]) begin + s2_imd_ns[15:0] = exe_uop[`IMD +: 16]; + s2_imd_ns[31:16] = {16{`LOW}}; + end + if(exe_uop[`IS_LI]) begin + s2_imd_ns[31:16] = exe_uop[`IMD2 +: 16]; + end + if(exe_uop[`IS_BL] | exe_uop[`IS_BEQ]) begin + s2_imd_ns[31:0] = exe_uop[`IMD +: 19]; + end + if(exe_uop[`IS_JUMP]) begin + s2_imd_ns[31:0] = exe_uop[`IMD +: 27]; + end + // stage two, actual computation + if(s2_uop[`IS_ADD]) begin + if(s2_uop[`HAS_IMD]) + s2_wdata = s2_rs1_data + s2_imd_r; + else + s2_wdata = s2_rs1_data + s2_rs2_data; + end + if(s2_uop[`IS_SUB]) begin + if(s2_uop[`HAS_IMD]) + s2_wdata = s2_rs1_data - s2_imd_r; + else + s2_wdata = s2_rs1_data - s2_rs2_data; + end + if(s2_uop[`IS_MOV] || s2_uop[`IS_LDWD]) begin + s2_wdata = s2_rs1_data; + end + if(s2_uop[`IS_LI]) begin + s2_wdata = s2_imd_r; + end + if(s2_uop[`IS_LDPC]) begin + s2_wdata = ddr_stat[s2_rs1*32 +: 32]; + end + if(s2_uop[`IS_SRC]) begin + s2_wdata[30:0] = s2_rs1_data[31:1]; + s2_wdata[31] = s2_rs1_data[0]; + end + // Bitwise ops + if(s2_uop[`IS_AND]) begin + s2_wdata = s2_rs1_data & s2_rs2_data; + end + if(s2_uop[`IS_OR]) begin + s2_wdata = s2_rs1_data | s2_rs2_data; + end + if(s2_uop[`IS_XOR]) begin + s2_wdata = s2_rs1_data ^ s2_rs2_data; + end + // mem unit + if(s2_uop[`IS_LD]) begin + s2_mem_ren = `HIGH; + end + if(s2_uop[`IS_ST]) begin + s2_mem_wdata = s2_rs2_data; + s2_mem_wen = `HIGH; + end + + // branch unit + if(s2_uop[`IS_BL]) begin + if(s2_rs1_data < s2_rs2_data) + fetch_pc = s2_imd_r; + else // not taken + fetch_pc = s2_pc + 1; + end + if(s2_uop[`IS_BEQ]) begin + if(s2_rs1_data == s2_rs2_data) + fetch_pc = s2_imd_r; + else // not taken + fetch_pc = s2_pc + 1; + end + if(s2_uop[`IS_JUMP]) begin + fetch_pc = s2_imd_r; + end + if(s2_uop[`IS_SLEEP]) begin + // not implemented + // frontent/fetch handles api sleeps + end + + end + + always @(posedge clk) begin + // stage one, further decode + s2_valid <= exe_valid; + s2_uop <= exe_uop; + s2_pc <= exe_pc; + s2_rs1 <= exe_uop[`RS1 +: 4]; + s2_rs2 <= exe_uop[`RS2 +: 4]; + s2_rt <= exe_uop[`RT +: 4]; + s2_imd_r <= s2_imd_ns; + s2_wen <= exe_uop[`IS_ADD] || exe_uop[`IS_SUB] || + exe_uop[`IS_MOV] || exe_uop[`IS_LI] || + exe_uop[`IS_LDWD] || exe_uop[`IS_SRC] || + exe_uop[`IS_AND] || exe_uop[`IS_OR] || + exe_uop[`IS_XOR] || exe_uop[`IS_LDPC]; + s2_wide_wen <= exe_uop[`IS_LDWD]; + // stage two, delayed branch signals + s3_br_resolve <= s2_valid && (s2_uop[`IS_BL] || + s2_uop[`IS_BEQ] || + s2_uop[`IS_JUMP]); + s3_br_target <= fetch_pc; + s3_wen <= s2_uop[`IS_LD]; // write path for loads + s3_rt <= s2_uop[`RT +: 4]; + end + +endmodule diff --git a/sources/hdl/verilog/execute_stage.v b/sources/hdl/verilog/execute_stage.v new file mode 100644 index 0000000..0b690f8 --- /dev/null +++ b/sources/hdl/verilog/execute_stage.v @@ -0,0 +1,194 @@ +`include "parameters.vh" + +module execute_stage( + + // common signals + input clk, + input rst, + + // decode stage <-> execute stage if + input ddr_valid, + input exe_valid, + input [`DDR_UOP_WIDTH*4-1:0] ddr_uop, + input [`EXE_UOP_WIDTH-1:0] exe_uop, + input [`IMEM_ADDR_WIDTH-1:0] exe_pc, + + // branch unit <-> fetch stage if + output br_resolve, + output [`IMEM_ADDR_WIDTH-1:0] br_target, + + // execute <-> outer DDRX IP interface + output [3:0] ddr_write, + output [3:0] ddr_read, + output [3:0] ddr_pre, + output [3:0] ddr_act, + output [3:0] ddr_ref, + output [3:0] ddr_sre, + output [3:0] ddr_srx, + output [3:0] ddr_zq, + output [3:0] ddr_nop, + output [3:0] ddr_ap, + output [3:0] ddr_pall, + output [3:0] ddr_half_bl, + output [4*`BG_WIDTH-1:0] ddr_bg, + output [4*`BANK_WIDTH-1:0] ddr_bank, + output [4*`COL_WIDTH-1:0] ddr_col, + output [4*`ROW_WIDTH-1:0] ddr_row, + output [511:0] ddr_wdata, + + // execute <-> reg file if + output rf_wide_wen, + output [3:0] rf_wide_offset, + input [32*8-1:0] rf_rdata, + output [32*8-1:0] rf_wdata, + output [7:0] rf_wen, + output [4*8-1:0] rf_raddr, + output [4*8-1:0] rf_waddr, + input [`COL_WIDTH-1:0] casr, + input [`BANK_WIDTH+`BG_WIDTH-1:0] basr, + input [`ROW_WIDTH-1:0] rasr, + input [511:0] wide_reg, + + // update stride registers, wen is OH (3'b001 = update rasr) + output [31:0] srf_value, + output [2:0] srf_wen, + input [32*7-1:0] ddr_stat + ); + + reg s2_exe; // exe_pipeline has regfile ports + + // ddr operations can optionally update + // registers. TODO when do we read the + // stride values? + wire [7:0] rf_wen_ddr; + wire [4*8-1:0] rf_waddr_ddr; + wire [32*8-1:0] rf_wdata_ddr; + wire [4*8-1:0] rf_raddr_ddr; + + ddr_pipeline dp( + .clk(clk), + .rst(rst), + + // execute <-> ddr_pipe if + .ddr_valid(ddr_valid), + .ddr_uop(ddr_uop), + + // ddr_pipeline <-> outer DDRX IP interface + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_sre(ddr_sre), + .ddr_srx(ddr_srx), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata), + // ddr_pipeline <-> regfile interface + .update_en(rf_wen_ddr), + .update_ids(rf_waddr_ddr), + .update_vals(rf_wdata_ddr), + .casr(casr), + .basr(basr), + .rasr(rasr), + .reg_ids(rf_raddr_ddr), // registers we need to read + .reg_vals(rf_rdata), // register values + .wide_reg(wide_reg) // write data register + ); + + + wire[31:0] rf_wdata_exe; + wire[7:0] rf_raddr_exe; + wire[3:0] rf_waddr_exe; + wire rf_wen_exe; + wire rf_wide_wen_exe; + + wire mem_wen; + wire mem_ren; + wire[9:0] mem_addr; + wire[31:0] mem_wdata; + wire[31:0] mem_rdata; + + exe_pipeline ep( + .clk(clk), + .rst(rst), + + // exe_pipeline <-> execute stage if + .exe_valid(exe_valid), + .exe_uop(exe_uop), + .exe_pc(exe_pc), + + // branch unit <-> fetch stage if + .br_resolve(br_resolve), + .br_target(br_target), + + // exe_pipeline <-> regfile if + .wide_wen(rf_wide_wen_exe), + .rf_wdata(rf_wdata_exe), + .rf_wen(rf_wen_exe), + .rf_raddr(rf_raddr_exe), + .rf_rdata(rf_rdata[0 +: 32*2]), + .rf_waddr(rf_waddr_exe), + + // exe_pipeline <-> data_mem + .mem_wen(mem_wen), + .mem_ren(mem_ren), + .mem_addr(mem_addr), + .mem_wdata(mem_wdata), + .mem_rdata(mem_rdata), + .ddr_stat(ddr_stat) + + ); +`ifdef XILINX_SIMULATOR + reg_mem data_mem( + .addr(mem_addr), + .clk(clk), + .din(mem_wdata), + .dout(mem_rdata), + .en(mem_wen || mem_ren), + .we(mem_wen) + ); +`else + scratchpad data_mem( + .addra(mem_addr), + .clka(clk), + .dina(mem_wdata), + .douta(mem_rdata), + .ena(mem_wen || mem_ren), + .wea(mem_wen) + ); +`endif + + wire exe_target_srf = rf_waddr_exe == 4'b0000 || + rf_waddr_exe == 4'b0001 || + rf_waddr_exe == 4'b0010; + + assign rf_wide_wen = rf_wide_wen_exe; + assign rf_wide_offset = rf_waddr_exe; + assign rf_wdata[32 +: 7*32] = rf_wdata_ddr[32 +: 7*32]; + assign rf_wdata[0 +: 32] = rf_wen_exe ? rf_wdata_exe : rf_wdata_ddr[0 +: 32]; + assign rf_wen = (rf_wen_exe & ~rf_wide_wen_exe & ~exe_target_srf) | rf_wen_ddr; + assign rf_raddr[8 +: 6*4] = rf_raddr_ddr[8 +: 6*4]; + assign rf_raddr[0 +: 2*4] = s2_exe ? rf_raddr_exe : rf_raddr_ddr[0 +: 2*8]; + assign rf_waddr[4 +: 7*4] = rf_waddr_ddr[4 +: 7*4]; + assign rf_waddr[0 +: 4] = rf_wen_exe ? rf_waddr_exe : rf_waddr_ddr[0 +: 4]; + + // First 3 register ids implicitly target stride registers + assign srf_wen[0] = rf_wen_exe & ~rf_wide_wen_exe & (rf_waddr_exe == 4'b0000); + assign srf_wen[1] = rf_wen_exe & ~rf_wide_wen_exe & (rf_waddr_exe == 4'b0001); + assign srf_wen[2] = rf_wen_exe & ~rf_wide_wen_exe & (rf_waddr_exe == 4'b0010); + + assign srf_value = rf_wdata_exe; + + always @(posedge clk) begin + s2_exe <= exe_valid; + end + +endmodule diff --git a/sources/hdl/verilog/fetch_stage.v b/sources/hdl/verilog/fetch_stage.v new file mode 100644 index 0000000..bd72fe5 --- /dev/null +++ b/sources/hdl/verilog/fetch_stage.v @@ -0,0 +1,147 @@ +`include "parameters.vh" + +module fetch_stage( + // common signals + input clk, + input rst, + + // other control signals + output softmc_end, + output [11:0] read_size, + output reg read_seq_incoming, + input [11:0] buffer_space, + + // branch unit <-> fetch stage interface + input br_resolve, + input [`IMEM_ADDR_WIDTH-1:0] br_target, + + // fetch stage <-> frontend interface + output [`IMEM_ADDR_WIDTH-1:0] addr_out, + output valid_out, + input [`INSTR_WIDTH-1:0] data_in, + input valid_in, + input [`IMEM_ADDR_WIDTH-1:0] addr_in, + input ready_out, // frontend is ready for a valid request + + // fetch stage <-> decode stage interface + output [`INSTR_WIDTH-1:0] instr, + output [`IMEM_ADDR_WIDTH-1:0] instr_pc, + output instr_valid + ); + + wire inst_is_br, is_end, is_ddr_start, need_flush, is_sleep; + + reg [31:0] sleep_ctr_r, sleep_ctr_ns; + + localparam WAIT_RESOLVE_S = 0; + localparam FETCH_NEXT_LINE_S = 1; + localparam WAIT_BUFFER_SPACE_S = 2; + localparam WAIT_SLEEP_S = 3; + + reg [2:0] state_r, state_ns; + + // kind of confusing but these PCs map to + // an instruction instead of to a byte. + // i.e. each PC addresses an instruction. + reg [`IMEM_ADDR_WIDTH-1:0] pc_r, pc_ns; + + // register outputs, decode will receive + // stuff we've received with one cycle latency + // i.e. marks the end of fetch_stage cycle + reg [`IMEM_ADDR_WIDTH-1:0] instr_pc_r, instr_pc_ns; + reg [`INSTR_WIDTH-1:0] instr_r, instr_ns; + reg instr_valid_r, instr_valid_ns; + + pre_decode pdec( + .buffer_space(buffer_space), + .read_size(read_size), + .instruction(state_r == WAIT_BUFFER_SPACE_S ? instr_r : data_in), + .is_branch(inst_is_br), + .is_end(is_end), + .is_ddr_start(is_ddr_start), + .need_flush(need_flush), + .is_sleep(is_sleep) + ); + + assign instr = instr_r; + assign instr_valid = instr_valid_r; + assign instr_pc = instr_pc_r; + + // request instr @ pc from frontend + assign valid_out = ready_out && (state_r == FETCH_NEXT_LINE_S) && ~need_flush && ~(valid_in && is_sleep); + assign addr_out = pc_r; + + assign softmc_end = is_end && valid_in && (state_r == FETCH_NEXT_LINE_S); + + always @* begin + sleep_ctr_ns = sleep_ctr_r; + state_ns = state_r; + pc_ns = pc_r; + instr_ns = instr_r; + instr_pc_ns = instr_pc_r; + instr_valid_ns = ~is_end && valid_in && (state_r == FETCH_NEXT_LINE_S) + && ~is_ddr_start; + read_seq_incoming = `LOW; + case(state_r) + WAIT_RESOLVE_S: begin + if(br_resolve) begin + state_ns = FETCH_NEXT_LINE_S; + pc_ns = br_target; + end + end + FETCH_NEXT_LINE_S: begin + if(ready_out && valid_out) + pc_ns = pc_r + 1; + if(valid_in) begin + instr_ns = data_in; + instr_pc_ns = addr_in; + if(is_sleep) begin + sleep_ctr_ns = data_in[31:0]; + state_ns = WAIT_SLEEP_S; + end + if(is_ddr_start && ~need_flush && (|data_in[9:0])) + read_seq_incoming = `HIGH; + if(inst_is_br && ~is_sleep) // we don't have the ability to perform well + state_ns = WAIT_RESOLVE_S; + else if(is_end) + pc_ns = {`IMEM_ADDR_WIDTH{`LOW}}; + else if(need_flush) begin + state_ns = WAIT_BUFFER_SPACE_S; + pc_ns = addr_in; // register the info packet + // as the next instruction to fetch + end + end + end + WAIT_BUFFER_SPACE_S: begin + if(~need_flush) begin + state_ns = FETCH_NEXT_LINE_S; + end + end + WAIT_SLEEP_S: begin + sleep_ctr_ns = sleep_ctr_r - 1; + if(sleep_ctr_r == 32'b1) + state_ns = FETCH_NEXT_LINE_S; + end + endcase + end + + always @(posedge clk) begin + if (rst) begin + pc_r <= {`IMEM_ADDR_WIDTH{`LOW}}; + state_r <= FETCH_NEXT_LINE_S; + instr_valid_r <= `LOW; + instr_r <= {`INSTR_WIDTH{`LOW}}; + instr_pc_r <= {`IMEM_ADDR_WIDTH{`LOW}}; + sleep_ctr_r <= {32{`LOW}}; + end + else begin + state_r <= state_ns; + pc_r <= pc_ns; + instr_r <= instr_ns; + instr_pc_r <= instr_pc_ns; + instr_valid_r <= instr_valid_ns; + sleep_ctr_r <= sleep_ctr_ns; + end + end + +endmodule diff --git a/sources/hdl/verilog/frontend.v b/sources/hdl/verilog/frontend.v new file mode 100644 index 0000000..9225bd9 --- /dev/null +++ b/sources/hdl/verilog/frontend.v @@ -0,0 +1,237 @@ +`include "parameters.vh" +/* + * This module is responsible for the interface + * between XDMA IP and the fetch stage. + * This module encapsulates a X KiB BRAM which + * is used as an instruction memory. + */ +module frontend#(parameter SIM_MEM = "false")( + // common signals + input clk, + input rst, + + // other control signals + input softmc_fin, + output user_rst, + input init_calib_complete, + output reg rbe_switch_mode, + output reg dllt_begin, + output frontend_ready, + + // frontend <-> fetch stage interface + input [`IMEM_ADDR_WIDTH-1:0] addr_in, + input valid_in, + output [`INSTR_WIDTH-1:0] data_out, + output valid_out, + output [`IMEM_ADDR_WIDTH-1:0] addr_out, + output ready_in, + + // frontend <-> xdma interface + input [`XDMA_AXI_DATA_WIDTH-1:0] h2c_tdata_0, + input h2c_tlast_0, + input h2c_tvalid_0, + output h2c_tready_0, + input [`XDMA_AXI_DATA_WIDTH/8-1:0] h2c_tkeep_0, + + // maintenance signals + output per_rd_init, + output per_zq_init, + output per_ref_init + ); + + reg[31:0] delay_fin; + + always @(posedge clk) begin + if(rst || user_rst) + delay_fin <= 32'b0; + else + delay_fin[1+:31] <= delay_fin[0+:31]; + delay_fin[0] <= softmc_fin; + end + + assign frontend_ready = delay_fin[31]; + + wire imem_wr_en, imem_rd_en; + wire [`IMEM_ADDR_WIDTH-1:0] imem_addr; + wire [`INSTR_WIDTH-1:0] imem_wr_data, imem_rd_data; + + generate + if(SIM_MEM == "true") begin + instr_blk_mem_sim imem( + .addra(imem_addr), + .clka(clk), + .dina(imem_wr_data), + .douta(imem_rd_data), + .ena(imem_rd_en || imem_wr_en), + .wea(imem_wr_en) + ); + end + else begin + instr_blk_mem imem( + .addra(imem_addr), + .clka(clk), + .dina(imem_wr_data), + .douta(imem_rd_data), + .ena(imem_rd_en || imem_wr_en), + .wea(imem_wr_en) + ); + end + endgenerate + + wire [`INSTR_WIDTH-1:0] maint_inst; + wire maint_valid; + wire [`IMEM_ADDR_WIDTH-1:0] maint_addr; + wire maint_req; + reg maint_ack; + wire maint_process; + wire program_process; + reg aref_en; + reg aref_en_valid; + + maintenance_controller maint_ctrl + ( + .clk(clk), + .rst(rst | user_rst), + + .init_calib_complete(init_calib_complete), + .softmc_fin(softmc_fin), + + .aref_en(aref_en), + .aref_en_valid(aref_en_valid), + .maint_req(maint_req), + .maint_ack(maint_ack), + .per_rd_init(per_rd_init), + .per_zq_init(per_zq_init), + .per_ref_init(per_ref_init), + .maint_process(maint_process), + .program_process(program_process), + + .in_addr(addr_in), + .in_valid(valid_in), + + .out_data(maint_inst), + .out_valid(maint_valid), + .out_addr(maint_addr) + ); + + localparam IDLE_S = 2'd0; + localparam INIT_MEM_S = 2'd1; + localparam EXECUTE_S = 2'd2; + + reg [1:0] state_r, state_ns; + + reg [4:0] rst_ctr_ns, rst_ctr_r; + reg [`IMEM_ADDR_WIDTH-1:0] xfer_ctr_r, xfer_ctr_ns; + reg [`IMEM_RD_LATENCY-1:0] valid_out_sr; + reg [(`IMEM_RD_LATENCY * `IMEM_ADDR_WIDTH)-1:0] addr_out_sr; + + assign user_rst = (|rst_ctr_r); + + // imem <-> xdma interface + // TODO do we need tkeep? + assign h2c_tready_0 = state_r == INIT_MEM_S; + assign imem_wr_en = h2c_tvalid_0 && (state_r == INIT_MEM_S); + assign imem_wr_data = h2c_tdata_0[`INSTR_WIDTH-1:0]; + assign imem_addr = state_r == INIT_MEM_S ? xfer_ctr_r : addr_in; + // imem <-> pipeline interface + assign imem_rd_en = valid_in && (program_process); + assign data_out = program_process ? imem_rd_data : maint_inst; + assign valid_out = program_process ? valid_out_sr[0] : maint_valid; + assign addr_out = program_process ? addr_out_sr[`IMEM_ADDR_WIDTH-1:0] : maint_addr; + + generate + if(SIM_MEM=="false") + assign ready_in = state_r == EXECUTE_S; + else + assign ready_in = state_r == EXECUTE_S && ~rst; + endgenerate + assign program_process = (state_r == EXECUTE_S) && ~maint_process; + + always @* begin + aref_en_valid = `LOW; + aref_en = `LOW; + state_ns = state_r; + xfer_ctr_ns = xfer_ctr_r; + rst_ctr_ns = {5{`LOW}}; + maint_ack = `LOW; + rbe_switch_mode = `LOW; + dllt_begin = `LOW; + case (state_r) + IDLE_S: begin + if(~((|delay_fin) || softmc_fin)) begin + if(h2c_tvalid_0) + state_ns = INIT_MEM_S; + else begin + if(maint_req) begin + maint_ack = `HIGH; + state_ns = EXECUTE_S; + end + end + end + end + INIT_MEM_S: begin + if(h2c_tvalid_0) begin + if(h2c_tdata_0[`INSTR_WIDTH]) //indicates a reset + rst_ctr_ns = {5{1'b1}}; + else if(h2c_tdata_0[`INSTR_WIDTH+1]) // indicate switch between readback modes + rbe_switch_mode = `HIGH; + else if(h2c_tdata_0[`INSTR_WIDTH+2]) // indicate dll toggle off WIP + dllt_begin = `HIGH; + else if(h2c_tdata_0[`INSTR_WIDTH+3]) begin // enable-disable autoref + aref_en_valid = `HIGH; + aref_en = h2c_tdata_0[0]; + state_ns = IDLE_S; + end + else begin + xfer_ctr_ns = xfer_ctr_r + 1; + if(h2c_tlast_0) begin + state_ns = EXECUTE_S; + xfer_ctr_ns = {`IMEM_ADDR_WIDTH{`LOW}}; + end + end + end + end + EXECUTE_S: begin + if(h2c_tvalid_0) begin + if(h2c_tdata_0[`INSTR_WIDTH]) //indicates a reset + rst_ctr_ns = {5{1'b1}}; + end + if(softmc_fin) + state_ns = IDLE_S; + end + endcase + + end + + always @(posedge clk) begin + if(rst || (|rst_ctr_r)) begin + if(SIM_MEM == "false") + state_r <= IDLE_S; + else + state_r <= EXECUTE_S; + xfer_ctr_r <= {`IMEM_ADDR_WIDTH{`LOW}}; + valid_out_sr <= {`IMEM_RD_LATENCY{`LOW}}; + addr_out_sr <= 0; + if(rst_ctr_r > 0) + rst_ctr_r <= rst_ctr_r - 1; + else + rst_ctr_r <= 0; + end + else begin + state_r <= state_ns; + xfer_ctr_r <= xfer_ctr_ns; + rst_ctr_r <= rst_ctr_ns; + // compute when we should assert valid data to + // fetch stage. + valid_out_sr[`IMEM_RD_LATENCY-1] <= valid_in && (state_r == EXECUTE_S); + addr_out_sr[`IMEM_RD_LATENCY*`IMEM_ADDR_WIDTH-1 : + (`IMEM_RD_LATENCY-1)*`IMEM_ADDR_WIDTH] <= addr_in; + `ifdef IMEM_SR + valid_out_sr[`IMEM_RD_LATENCY-1:0] <= valid_out_sr >> 1; + addr_out_sr[(`IMEM_RD_LATENCY-1)*`IMEM_ADDR_WIDTH-1:0] + <= addr_out_sr >> `IMEM_ADDR_WIDTH; + `endif + end + end +endmodule + diff --git a/sources/hdl/verilog/maintenance_controller.v b/sources/hdl/verilog/maintenance_controller.v new file mode 100644 index 0000000..9354d17 --- /dev/null +++ b/sources/hdl/verilog/maintenance_controller.v @@ -0,0 +1,265 @@ +`include "parameters.vh" + +module maintenance_controller#(parameter tCK = 1500)( + + input clk, + input rst, + + input init_calib_complete, + input softmc_fin, + + input aref_en, + input aref_en_valid, + + input maint_ack, + output maint_req, + output per_rd_init, + output per_zq_init, + output per_ref_init, + output maint_process, + input program_process, + + input [`IMEM_ADDR_WIDTH-1:0] in_addr, + input in_valid, + + output [`INSTR_WIDTH-1:0] out_data, + output reg out_valid, + output [`IMEM_ADDR_WIDTH-1:0] out_addr + ); + + wire zq_ack, per_rd_ack, ref_ack; + reg zq_request, zq_process; + reg per_rd_request, per_rd_process; + reg pr_ref_request, pr_ref_process; + + wire [`INSTR_WIDTH-1 : 0] pr_read_out, pr_zq_out, pr_ref_out; + + assign maint_req = per_rd_request | zq_request | pr_ref_request; + assign out_addr = {`IMEM_ADDR_WIDTH{`HIGH}}; + assign out_data = zq_process ? pr_zq_out : per_rd_process ? pr_read_out : pr_ref_out; + assign per_ref_init = maint_ack && pr_ref_request && ~per_rd_request && ~zq_request; + assign per_rd_init = maint_ack && per_rd_request && ~zq_request; + assign per_zq_init = maint_ack && zq_request; + assign maint_process = zq_process || per_rd_process || pr_ref_process; + + assign zq_ack = softmc_fin && zq_process; + assign per_rd_ack = softmc_fin && per_rd_process; + assign ref_ack = softmc_fin && pr_ref_process; + + pr_read_mem prm + ( + .addra(in_addr[3:0]), + .clka(clk), + .douta(pr_read_out), + .dina(64'bX), + .wea(`LOW), + .ena(in_valid && per_rd_process) + ); + + zq_calib_mem pzm + ( + .addra(in_addr[5:0]), + .clka(clk), + .douta(pr_zq_out), + .dina(64'bX), + .wea(`LOW), + .ena(in_valid && zq_process) + ); + + pr_ref_mem prefm + ( + .addra(in_addr[6:0]), + .clka(clk), + .douta(pr_ref_out), + .dina(64'bX), + .wea(`LOW), + .ena(in_valid && pr_ref_process) + ); + + always @(posedge clk) begin + if(rst) begin + zq_process <= `LOW; + per_rd_process <= `LOW; + out_valid <= `LOW; + pr_ref_process <= `LOW; + end + else begin + if(softmc_fin) begin + if(zq_process) + zq_process <= `LOW; + if(per_rd_process) + per_rd_process <= `LOW; + if(pr_ref_process) + pr_ref_process <= `LOW; + out_valid <= `LOW; + end + else if(maint_ack) begin + if(zq_request) + zq_process <= `HIGH; + else if(per_rd_request) + per_rd_process <= `HIGH; + else if(pr_ref_request) + pr_ref_process <= `HIGH; + out_valid <= `LOW; + end + else begin + zq_process <= zq_process; + per_rd_process <= per_rd_process; + pr_ref_process <= pr_ref_process; + if(maint_process) + out_valid <= in_valid; + else + out_valid <= `LOW; + end + end + end + + // Maintenance control logic + // Set request bits according to timers + + localparam MAINT_PRESCALER_PERIOD = 500_000; // .5 us + + function integer clogb2 (input integer size); // ceiling logb2 + begin + size = size - 1; + for (clogb2=1; size>1; clogb2=clogb2+1) + size = size >> 1; + end + endfunction // clogb2 + + localparam MAINT_PRESCALER_DIV = MAINT_PRESCALER_PERIOD/(tCK * 4); // softmc is clocked 4 times slower than the memory interface + localparam MAINT_PRESCALER_WIDTH = clogb2(MAINT_PRESCALER_DIV + 1); + localparam ONE = 1; + + reg maint_prescaler_tick_r_lcl; + reg [MAINT_PRESCALER_WIDTH-1:0] maint_prescaler_r; + reg [MAINT_PRESCALER_WIDTH-1:0] maint_prescaler_ns; + + wire maint_prescaler_tick_ns = (maint_prescaler_r == ONE[MAINT_PRESCALER_WIDTH-1:0]); + + always @(/*AS*/init_calib_complete or maint_prescaler_r + or maint_prescaler_tick_ns) begin + maint_prescaler_ns = maint_prescaler_r; + if (~init_calib_complete || maint_prescaler_tick_ns) + maint_prescaler_ns = MAINT_PRESCALER_DIV[MAINT_PRESCALER_WIDTH-1:0]; + else if (|maint_prescaler_r) + maint_prescaler_ns = maint_prescaler_r - ONE[MAINT_PRESCALER_WIDTH-1:0]; + end + + always @(posedge clk) maint_prescaler_r <= maint_prescaler_ns; + + always @(posedge clk) maint_prescaler_tick_r_lcl <= maint_prescaler_tick_ns; + + localparam tPRDI = 1_000_000; + localparam PERIODIC_RD_TIMER_DIV = tPRDI/MAINT_PRESCALER_PERIOD; + localparam PERIODIC_RD_TIMER_WIDTH = clogb2(PERIODIC_RD_TIMER_DIV + /*idle state*/ 1); + + reg [PERIODIC_RD_TIMER_WIDTH-1:0] periodic_rd_timer_r, periodic_rd_timer; + + always @* begin + periodic_rd_timer = periodic_rd_timer_r; + + if(~init_calib_complete) begin + periodic_rd_timer = {PERIODIC_RD_TIMER_WIDTH{1'b0}}; + end + else if (per_rd_ack || program_process) begin + periodic_rd_timer = PERIODIC_RD_TIMER_DIV[0+:PERIODIC_RD_TIMER_WIDTH]; + end + else if (|periodic_rd_timer_r && maint_prescaler_tick_r_lcl) begin + periodic_rd_timer = periodic_rd_timer_r - ONE[0+:PERIODIC_RD_TIMER_WIDTH]; + end + end //always + + wire periodic_rd_timer_one = maint_prescaler_tick_r_lcl && (periodic_rd_timer_r == ONE[0+:PERIODIC_RD_TIMER_WIDTH]); + + wire periodic_rd_request = ~rst && (/*((PERIODIC_RD_TIMER_DIV != 0) && ~dfi_init_complete) ||*/ + (~per_rd_ack && (per_rd_request || periodic_rd_timer_one))); + + always @(posedge clk) begin + if(~init_calib_complete) + periodic_rd_timer_r <= PERIODIC_RD_TIMER_DIV[0+:PERIODIC_RD_TIMER_WIDTH]; + else + periodic_rd_timer_r <= periodic_rd_timer; + per_rd_request <= periodic_rd_request; + end //always + + // ZQ timebase. Nominally 128 mS + localparam MAINT_PRESCALER_PERIOD_NS = MAINT_PRESCALER_PERIOD / 1000; + localparam tZQI = 128_000_000; + localparam ZQ_TIMER_DIV = tZQI/MAINT_PRESCALER_PERIOD_NS; + localparam ZQ_TIMER_WIDTH = clogb2(ZQ_TIMER_DIV + 1); + + generate + begin : zq_cntrl + reg zq_tick = 1'b0; + + if (ZQ_TIMER_DIV !=0) begin : zq_timer + reg [ZQ_TIMER_WIDTH-1:0] zq_timer_r; + reg [ZQ_TIMER_WIDTH-1:0] zq_timer_ns; + + always @(/*AS*/init_calib_complete or maint_prescaler_tick_r_lcl or zq_tick or zq_timer_r or program_process) begin + zq_timer_ns = zq_timer_r; + if (~init_calib_complete || zq_tick || program_process) + zq_timer_ns = ZQ_TIMER_DIV[ZQ_TIMER_WIDTH-1:0]; + else if (|zq_timer_r && maint_prescaler_tick_r_lcl) + zq_timer_ns = zq_timer_r - ONE[ZQ_TIMER_WIDTH-1:0]; + end + + always @(posedge clk) zq_timer_r <= zq_timer_ns; + + always @(/*AS*/maint_prescaler_tick_r_lcl or zq_timer_r) + zq_tick = (zq_timer_r == ONE[ZQ_TIMER_WIDTH-1:0] && maint_prescaler_tick_r_lcl); + end // zq_timer + + // ZQ request. Set request with timer tick, and when exiting PHY init. Never + // request if ZQ_TIMER_DIV == 0. + begin : zq_request_logic + wire zq_clear = zq_ack; + reg zq_request_r; + wire zq_request_ns = ~rst && ((~init_calib_complete && (ZQ_TIMER_DIV != 0)) || + (zq_request_r && ~zq_clear) || zq_tick); + + always @(posedge clk) zq_request_r <= zq_request_ns; + + always @(/*AS*/init_calib_complete or zq_request_r) + zq_request = init_calib_complete && zq_request_r; + end // zq_request_logic + end + endgenerate + + localparam tAREF = 7_800; + localparam PERIODIC_REF_TIMER_DIV = tAREF/MAINT_PRESCALER_PERIOD_NS; + localparam PERIODIC_REF_TIMER_WIDTH = clogb2(PERIODIC_REF_TIMER_DIV + /*idle state*/ 1); + + reg [PERIODIC_REF_TIMER_WIDTH-1:0] autoref_timer_r, autoref_timer; + reg aref_switch_ns, aref_switch_r; + + always @* begin + aref_switch_ns = aref_en_valid ? aref_en : aref_switch_r; + pr_ref_request = `LOW; + autoref_timer = autoref_timer_r; + if(~aref_switch_r || ~init_calib_complete) + autoref_timer = PERIODIC_REF_TIMER_DIV[PERIODIC_REF_TIMER_WIDTH-1:0]; + else begin + if(autoref_timer_r > 1 && maint_prescaler_tick_r_lcl) // you were here + autoref_timer = autoref_timer_r - ONE[PERIODIC_REF_TIMER_WIDTH-1:0]; + else if(autoref_timer_r == 1) begin + pr_ref_request = `HIGH; + if(ref_ack) + autoref_timer = PERIODIC_REF_TIMER_DIV[PERIODIC_REF_TIMER_WIDTH-1:0]; + end + end + end + + always @(posedge clk) begin + if(rst) begin + aref_switch_r <= `LOW; + autoref_timer_r = PERIODIC_REF_TIMER_DIV[PERIODIC_REF_TIMER_WIDTH-1:0]; + end + else begin + aref_switch_r <= aref_switch_ns; + autoref_timer_r <= autoref_timer; + end + end + +endmodule diff --git a/sources/hdl/verilog/pop_count4.v b/sources/hdl/verilog/pop_count4.v new file mode 100644 index 0000000..80bea46 --- /dev/null +++ b/sources/hdl/verilog/pop_count4.v @@ -0,0 +1,68 @@ +`timescale 1ns / 1ps +////////////////////////////////////////////////////////////////////////////////// +// Company: +// Engineer: +// +// Create Date: 12/19/2018 10:51:52 AM +// Design Name: +// Module Name: pop_count4 +// Project Name: +// Target Devices: +// Tool Versions: +// Description: +// +// Dependencies: +// +// Revision: +// Revision 0.01 - File Created +// Additional Comments: +// +////////////////////////////////////////////////////////////////////////////////// + + +module pop_count4( + input [3:0] in, + output [2:0] out + ); + + reg [2:0] out_r; + + always @* begin + out_r = 3'd0; + case(in) + 4'b0000: + out_r = 3'd0; + 4'b0001: + out_r = 3'd1; + 4'b0010: + out_r = 3'd1; + 4'b0011: + out_r = 3'd2; + 4'b0100: + out_r = 3'd1; + 4'b0101: + out_r = 3'd2; + 4'b0110: + out_r = 3'd2; + 4'b0111: + out_r = 3'd3; + 4'b1000: + out_r = 3'd1; + 4'b1001: + out_r = 3'd2; + 4'b1010: + out_r = 3'd2; + 4'b1011: + out_r = 3'd3; + 4'b1100: + out_r = 3'd2; + 4'b1101: + out_r = 3'd3; + 4'b1111: + out_r = 3'd4; + endcase + end + + assign out = out_r; + +endmodule diff --git a/sources/hdl/verilog/pre_decode.v b/sources/hdl/verilog/pre_decode.v new file mode 100644 index 0000000..7621b24 --- /dev/null +++ b/sources/hdl/verilog/pre_decode.v @@ -0,0 +1,39 @@ +`include "parameters.vh" +`include "encoding.vh" + +/* + * This combinational logic will output + * whether or not an instruction is a branch + * or not, for now. + */ +module pre_decode( + input [`INSTR_WIDTH-1:0] instruction, + // available readback_fifo entries in terms of read count + input [11:0] buffer_space, + output [11:0] read_size, + output is_branch, + output is_end, + output is_ddr_start, + output need_flush, + output is_sleep + ); + + // When an instruction is non_ddr and has + // is_branch flag set + assign is_branch = instruction[`BRANCH_OFFSET] + && ~instruction[`DDR_OFFSET]; + + assign is_end = &(~instruction); + + // encountered a ddr command segments + assign is_ddr_start = instruction[`INFO_OFFSET] + && ~instruction[`DDR_OFFSET]; + + assign read_size = instruction[9:0]; + // if there are more reads than we can buffer + assign need_flush = is_ddr_start && (read_size > buffer_space); + + assign is_sleep = instruction[`BRANCH_OFFSET] && + instruction[`FU_CODE_OFFSET +: 8] == `SLEEP; + +endmodule diff --git a/sources/hdl/verilog/readback_engine.v b/sources/hdl/verilog/readback_engine.v new file mode 100644 index 0000000..496d83e --- /dev/null +++ b/sources/hdl/verilog/readback_engine.v @@ -0,0 +1,244 @@ +`include "parameters.vh" + +// Process data coming from DRAM before sending it to the host. +module readback_engine( + + // common signals + input clk, + input rst, + + // other control signals + input flush, + input read_seq_incoming, // next few instructions will read from DRAM + input [11:0] incoming_reads, // how many reads next few instructions will issue + output[11:0] buffer_space, // remaining buffer size + input switch_mode, + + // DRAM <-> engine if + input [511:0] rd_data, + input rd_valid, + + input per_rd_init, + input per_zq_init, + input per_ref_init, + + // engine <-> regfile if + input [511:0] ddr_wdata, // to compare read data against + + // readback <-> XDMA if + output [`XDMA_AXI_DATA_WIDTH-1:0] c2h_tdata_0, + output c2h_tlast_0, + output c2h_tvalid_0, + input c2h_tready_0, + output [`XDMA_AXI_DATA_WIDTH/8-1:0] c2h_tkeep_0 + + ); + + + localparam READ_MODE = 0; + localparam DIFF_MODE = 1; + reg mode_r, mode_ns; // Switch between diff count and read modes + + reg rd_valid_r; + reg ignore_read_r, ignore_read_ns; + reg ignore_flush_r, ignore_flush_ns; + + // Popcount computation part + reg[511:0] read_diff; + reg diff_valid; + always @(posedge clk) begin + if(rst) begin + read_diff <= 512'bX; + diff_valid <= `LOW; + end + read_diff <= rd_valid ? rd_data ^ ddr_wdata : read_diff; + diff_valid <= rd_valid && ~ignore_read_r && mode_r == DIFF_MODE ? `HIGH : `LOW; + end + + genvar pcs; // popcount modules + + wire[2:0] pc_out [127:0]; + reg[3:0] pc_out_l2 [63:0]; + reg[4:0] pc_out_l3 [31:0]; + reg[5:0] pc_out_l4 [15:0]; + reg[6:0] pc_out_l5 [7:0]; + reg[7:0] pc_out_l6 [3:0]; + reg[8:0] pc_out_l7 [1:0]; + reg[15:0] pop_count_value; + reg pop_count_valid; + + generate + for(pcs = 0 ; pcs < 128 ; pcs = pcs + 1) begin: gen_pcs + pop_count4 pci + ( + .in(read_diff[pcs*4 +: 4]), + .out(pc_out[pcs]) + ); + end + endgenerate + + integer l1, l2, l3, l4, l5, l6; + always @* begin + for(l1 = 0 ; l1 < 64 ; l1 = l1+1) + pc_out_l2[l1] = pc_out[2*l1] + pc_out[2*l1+1]; + for(l2 = 0 ; l2 < 32 ; l2 = l2+1) + pc_out_l3[l2] = pc_out_l2[2*l2] + pc_out_l2[2*l2+1]; + for(l3 = 0 ; l3 < 16 ; l3 = l3+1) + pc_out_l4[l3] = pc_out_l3[2*l3] + pc_out_l3[2*l3+1]; + for(l4 = 0 ; l4 < 8 ; l4 = l4+1) + pc_out_l5[l4] = pc_out_l4[2*l4] + pc_out_l4[2*l4+1]; + for(l5 = 0 ; l5 < 4 ; l5 = l5+1) + pc_out_l6[l5] = pc_out_l5[2*l5] + pc_out_l5[2*l5+1]; + for(l6 = 0 ; l6 < 2 ; l6 = l6+1) + pc_out_l7[l6] = pc_out_l6[2*l6] + pc_out_l6[2*l6+1]; + end + + always @(posedge clk) begin + if(rst) begin + pop_count_value <= 16'bX; + pop_count_valid <= `LOW; + end + else begin + pop_count_value <= diff_valid ? pc_out_l7[0] + pc_out_l7[1] : pop_count_value; + pop_count_valid <= diff_valid ? `HIGH : `LOW; + end + end + + wire[511:0] dsr_out; + wire dsr_valid; + // We put popcounted data into a shift register + // to fill up 512 bit I/O fifo. + diff_shift_reg dsr( + .clk(clk), + .rst(rst), + + .in(pop_count_value), + .in_valid(pop_count_valid), + + .flush(flush&~ignore_flush_r), + + .out(dsr_out), + .out_valid(dsr_valid) + ); + // End popcount computation part + + // Count up to 1024 32-byte transfers + reg[9:0] xctr_r; + + reg tlast; // indicating c2h's last transfer + + // We read DQ_WIDTH*DQ_BURST (512 as of now) bits + // from DRAM, and have to pipe 256 bit partitions of + // it to the PCI. We may read data each cycle from + // DRAM and have to buffer some of those. + wire rbf_empty, rbf_rd_valid, fifo_almost_full, fifo_valid; + (*KEEP = "TRUE"*) wire rbf_full; + (*KEEP = "TRUE"*) reg [19:0] dbg_rd_ctr; + rdback_fifo rbf( + .full(rbf_full), + .prog_full(fifo_almost_full), + .empty(rbf_empty), + .wr_en(mode_r == READ_MODE ? rd_valid && ~ignore_read_r: dsr_valid), + // shuffle data because fifo outputs them on wrong order + .din(mode_r == READ_MODE ? {rd_data[255:0],rd_data[511:256]} : {dsr_out[255:0],dsr_out[511:256]}), + .rd_en(c2h_tready_0), + .dout(c2h_tdata_0), + .valid(fifo_valid), + .clk(clk), + .srst(rst) + ); + + reg proc_flush_ns, proc_flush_r; + // we count the remaining space in terms of + // AXI transactions + // e.g. 1024 reads will take up 2048 + reg [11:0] buffer_space_ns, buffer_space_r; + + always @* begin + tlast = `LOW; + ignore_read_ns = ignore_read_r; + ignore_flush_ns = ignore_flush_r; + buffer_space_ns = buffer_space_r; + if(per_rd_init || per_zq_init || per_ref_init) begin + ignore_read_ns = per_rd_init; + ignore_flush_ns = `HIGH; + end + if(rd_valid_r) + ignore_read_ns = `LOW; + proc_flush_ns = proc_flush_r; + if(flush) begin + if(ignore_flush_r) + ignore_flush_ns = `LOW; + else + proc_flush_ns = `HIGH; + end + mode_ns = mode_r; + if(switch_mode) + mode_ns = ~mode_r; + if(&xctr_r && (c2h_tready_0 && c2h_tvalid_0)) begin + tlast = `HIGH; + end + // Send what's remaining in the fifo + // to host with a random length transfer + // (tlast is not based on the counter value) + if(proc_flush_r) begin + if(c2h_tready_0 && rbf_empty && ~dsr_valid) begin + tlast = `HIGH; + proc_flush_ns = `LOW; + end + else + proc_flush_ns = `HIGH; + end + if(read_seq_incoming) begin + if(c2h_tvalid_0 && c2h_tready_0) begin + buffer_space_ns = (buffer_space_r - (incoming_reads << 1)) + 1; + end + else begin + buffer_space_ns = (buffer_space_r - (incoming_reads << 1)); + end + end + else begin + if(c2h_tvalid_0 && c2h_tready_0) begin + if(~(proc_flush_r && rbf_empty && ~dsr_valid)) + buffer_space_ns = buffer_space_r + 1; + end + end + end + + always @(posedge clk) begin + if(rst) begin + dbg_rd_ctr <= 20'b0; + xctr_r <= 15'b0; + proc_flush_r <= `LOW; + mode_r <= READ_MODE; + ignore_read_r <= 1'b0; + ignore_flush_r <= 1'b0; + rd_valid_r <= 1'b0; + buffer_space_r <= 12'd2048; + end + else begin + if(rd_valid && ~ignore_read_r && ~rbf_full) + dbg_rd_ctr <= dbg_rd_ctr + 1'b1; + else + dbg_rd_ctr <= dbg_rd_ctr; + buffer_space_r <= buffer_space_ns; + mode_r <= mode_ns; + rd_valid_r <= rd_valid; + ignore_read_r <= ignore_read_ns; + ignore_flush_r <= ignore_flush_ns; + if(proc_flush_r && tlast) + xctr_r <= 15'b0; + else if(c2h_tready_0 && c2h_tvalid_0) begin + xctr_r <= xctr_r + 1; + end + proc_flush_r <= proc_flush_ns; + end + end + + assign c2h_tkeep_0 = {(`XDMA_AXI_DATA_WIDTH/8){1'b1}}; + assign c2h_tlast_0 = tlast; + assign c2h_tvalid_0 = proc_flush_r && rbf_empty && ~dsr_valid ? `HIGH : fifo_valid; + + assign buffer_space = buffer_space_r >> 1; + +endmodule diff --git a/sources/hdl/verilog/reg_mem.v b/sources/hdl/verilog/reg_mem.v new file mode 100644 index 0000000..a8d37cd --- /dev/null +++ b/sources/hdl/verilog/reg_mem.v @@ -0,0 +1,33 @@ +`timescale 1ns / 1ps + +module reg_mem( + input [9:0] addr, + input clk, + input [31:0] din, + output [31:0] dout, + input en, + input we + ); + + reg [31:0] mem [1023:0]; + + integer i; + initial + begin + for(i=0; i<1024; i=i+1) + mem[i]=0; + end + + reg [31:0] out; + + assign dout=out; + + always @(posedge clk) + begin + if(we) + mem[addr]<=din; + else if(en) + out<=mem[addr]; + end + +endmodule diff --git a/sources/hdl/verilog/register_file.v b/sources/hdl/verilog/register_file.v new file mode 100644 index 0000000..2d3b693 --- /dev/null +++ b/sources/hdl/verilog/register_file.v @@ -0,0 +1,92 @@ +`include "parameters.vh" + +/* + * An 8 read 8 write port reg file + * satisfying ddr pipeline's needs + */ +module register_file( + + // common signals + input clk, + input rst, + + // execute stage <-> reg file if + input rf_wide_wen, + input [3:0] rf_wide_offset, + output [511:0] rf_wide_data, + output [32*8-1:0] rf_rdata, + input [32*8-1:0] rf_wdata, + input [7:0] rf_wen, + input [4*8-1:0] rf_raddr, + input [4*8-1:0] rf_waddr, + output [`COL_WIDTH-1:0] casr, + output [`BANK_WIDTH+`BG_WIDTH-1:0] basr, + output [`ROW_WIDTH-1:0] rasr, + // update stride registers, wen is OH (3'b001 = update rasr) + input [31:0] srf_value, + input [2:0] srf_wen + ); + + // Stride registers + reg [31:0] casr_r, basr_r, rasr_r; + + assign casr = casr_r; + assign basr = basr_r; + assign rasr = rasr_r; + + // Update stride registers + always @(posedge clk) begin + if(srf_wen[0]) begin + casr_r <= srf_value; + end + if(srf_wen[1]) begin + basr_r <= srf_value; + end + if(srf_wen[2]) begin + rasr_r <= srf_value; + end + end + + // General purpose registers + reg [31:0] reg_file [15:0]; + + // split r&w data signals into human readable form + // also drive rdata with appropriate register values + wire [31:0] wdata [7:0]; + wire wen [7:0]; + wire [3:0] waddr [7:0]; + wire [3:0] raddr [7:0]; + genvar i; + generate + for(i = 0 ; i < 8 ; i = i + 1) begin: gen_ports + assign wdata[i] = rf_wdata[32*i +: 32]; + assign wen[i] = rf_wen[i]; + assign waddr[i] = rf_waddr[4*i +: 4]; + assign raddr[i] = rf_raddr[4*i +: 4]; + // TODO will this compile? + assign rf_rdata[32*i +: 32] = raddr[i] == 0 ? casr_r : + raddr[i] == 1 ? basr_r : + raddr[i] == 2 ? rasr_r : reg_file[raddr[i]]; + end + endgenerate + + integer j; + // Write to the register file + always @(posedge clk) begin + for(j = 0 ; j < 8 ; j = j+1) begin: regfile_write + if(wen[j]) + reg_file[waddr[j]] <= wdata[j]; + end + end + + // DDR4 8-burst write data register + reg [511:0] wide_reg; + always @(posedge clk) begin + // TODO this probably won't compile + if(rf_wide_wen) + wide_reg[rf_wide_offset*32 +: 32] <= wdata[0]; + end + + assign rf_wide_data = wide_reg; + +endmodule diff --git a/sources/hdl/verilog/softmc_pipeline.v b/sources/hdl/verilog/softmc_pipeline.v new file mode 100644 index 0000000..07c106e --- /dev/null +++ b/sources/hdl/verilog/softmc_pipeline.v @@ -0,0 +1,178 @@ +`include "parameters.vh" + +module softmc_pipeline( + + // common signals + input clk, + input rst, + + // readback <-> fetch_stage backpressure + + output softmc_end, + output [11:0] read_size, + output read_seq_incoming, + input [11:0] buffer_space, + + // frontend <-> fetch stage interface + output [`IMEM_ADDR_WIDTH-1:0] addr_out, + output valid_out, + input [`INSTR_WIDTH-1:0] data_in, + input valid_in, + input [`IMEM_ADDR_WIDTH-1:0] addr_in, + input ready_out, + + // ddr_pipeline <-> outer DDR module + output [3:0] ddr_write, + output [3:0] ddr_read, + output [3:0] ddr_pre, + output [3:0] ddr_act, + output [3:0] ddr_ref, + output [3:0] ddr_sre, + output [3:0] ddr_srx, + output [3:0] ddr_zq, + output [3:0] ddr_nop, + output [3:0] ddr_ap, + output [3:0] ddr_pall, + output [3:0] ddr_half_bl, + output [4*`BG_WIDTH-1:0] ddr_bg, + output [4*`BANK_WIDTH-1:0] ddr_bank, + output [4*`COL_WIDTH-1:0] ddr_col, + output [4*`ROW_WIDTH-1:0] ddr_row, + output [511:0] ddr_wdata + ); + + wire br_resolve; + wire [`IMEM_ADDR_WIDTH-1:0] br_target; + + wire [`INSTR_WIDTH-1:0] dec_instr; + wire dec_instr_valid; + wire [`IMEM_ADDR_WIDTH-1:0] dec_instr_pc; + + fetch_stage fs( + .clk(clk), + .rst(rst), + + .softmc_end(softmc_end), + .read_size(read_size), + .read_seq_incoming(read_seq_incoming), + .buffer_space(buffer_space), + + + .addr_out(addr_out), + .valid_out(valid_out), + .data_in(data_in), + .valid_in(valid_in), + .addr_in(addr_in), + .ready_out(ready_out), + + .br_resolve(br_resolve), + .br_target(br_target), + + .instr(dec_instr), + .instr_pc(dec_instr_pc), + .instr_valid(dec_instr_valid) + ); + + wire ddr_valid, exe_valid; + wire [`DDR_UOP_WIDTH*4-1:0] ddr_uop; + wire [`EXE_UOP_WIDTH-1:0] exe_uop; + wire [`IMEM_ADDR_WIDTH-1:0] exe_pc; + wire [32*7-1:0] ddr_stat; + + decode_stage ds( + .clk(clk), + .rst(rst), + + .instr(dec_instr), + .instr_pc(dec_instr_pc), + .instr_valid(dec_instr_valid), + + .ddr_valid(ddr_valid), + .exe_valid(exe_valid), + .ddr_uop(ddr_uop), + .exe_uop(exe_uop), + .exe_pc(exe_pc), + .ddr_stat(ddr_stat) + ); + + wire rf_wide_wen; + wire [3:0] rf_wide_offset; + wire [511:0] rf_wide_data; + wire [32*8-1:0] rf_rdata; + wire [32*8-1:0] rf_wdata; + wire [7:0] rf_wen; + wire [4*8-1:0] rf_raddr; + wire [4*8-1:0] rf_waddr; + wire [`COL_WIDTH-1:0] casr; + wire [`BANK_WIDTH+`BG_WIDTH-1:0] basr; + wire [`ROW_WIDTH-1:0] rasr; + wire [31:0] srf_value; + wire [2:0] srf_wen; + + execute_stage es( + .clk(clk), + .rst(rst), + + .br_resolve(br_resolve), + .br_target(br_target), + + .ddr_valid(ddr_valid), + .exe_valid(exe_valid), + .ddr_uop(ddr_uop), + .exe_uop(exe_uop), + .exe_pc(exe_pc), + + .rf_wide_wen(rf_wide_wen), + .rf_wide_offset(rf_wide_offset), + .wide_reg(rf_wide_data), + .rf_rdata(rf_rdata), + .rf_wdata(rf_wdata), + .rf_wen(rf_wen), + .rf_raddr(rf_raddr), + .rf_waddr(rf_waddr), + .casr(casr), + .basr(basr), + .rasr(rasr), + .srf_value(srf_value), + .srf_wen(srf_wen), + .ddr_stat(ddr_stat), + + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_sre(ddr_sre), + .ddr_srx(ddr_srx), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata) + ); + + register_file rf( + .clk(clk), + .rst(rst), + + .rf_wide_wen(rf_wide_wen), + .rf_wide_offset(rf_wide_offset), + .rf_wide_data(rf_wide_data), + .rf_rdata(rf_rdata), + .rf_wdata(rf_wdata), + .rf_wen(rf_wen), + .rf_raddr(rf_raddr), + .rf_waddr(rf_waddr), + .casr(casr), + .basr(basr), + .rasr(rasr), + .srf_value(srf_value), + .srf_wen(srf_wen) + ); + +endmodule diff --git a/sources/hdl/verilog/softmc_top.v b/sources/hdl/verilog/softmc_top.v new file mode 100644 index 0000000..c5b1c76 --- /dev/null +++ b/sources/hdl/verilog/softmc_top.v @@ -0,0 +1,838 @@ +`include "parameters.vh" +`include "project.vh" + +`ifdef XUPP3R_x4 + `define XUPP3R +`elsif XUPP3R_x8 + `define XUPP3R +`elsif XUPP3R_x8_1R_UDIMM + `define XUPP3R +`endif + + +module softmc_top #(parameter tCK = 1500, SIM = "false") + ( + // common signals + input c0_sys_clk_p, + input c0_sys_clk_n, + input sys_rst_l, + + // iob <> ddr4 sdram ip signals + output c0_ddr4_act_n, + output [16:0] c0_ddr4_adr, + output [1:0] c0_ddr4_ba, + output [1:0] c0_ddr4_bg, + output [`CKE_WIDTH-1:0] c0_ddr4_cke, + output [`ODT_WIDTH-1:0] c0_ddr4_odt, + output [`CS_WIDTH-1:0] c0_ddr4_cs_n, + output [`CK_WIDTH-1:0] c0_ddr4_ck_t, + output [`CK_WIDTH-1:0] c0_ddr4_ck_c, + output c0_ddr4_reset_n, + `ifdef XUPP3R_x4 + inout [17:0] c0_ddr4_dqs_c, + inout [17:0] c0_ddr4_dqs_t, + inout [71:0] c0_ddr4_dq, + output c0_ddr4_parity, + `elsif XUPP3R_x8 + inout [8:0] c0_ddr4_dm_dbi_n, + inout [71:0] c0_ddr4_dq, + inout [8:0] c0_ddr4_dqs_c, + inout [8:0] c0_ddr4_dqs_t, + output c0_ddr4_parity, + `else + inout [7:0] c0_ddr4_dm_dbi_n, + inout [63:0] c0_ddr4_dq, + inout [7:0] c0_ddr4_dqs_c, + inout [7:0] c0_ddr4_dqs_t, + `endif + // xdma signals + input clk_ref_p, + input clk_ref_n, + input pcie_rst, + output [7:0] pci_exp_txp, + output [7:0] pci_exp_txn, + input [7:0] pci_exp_rxp, + input [7:0] pci_exp_rxn + + ); + + // Frontend control signals + wire softmc_fin; + wire user_rst; + + // Frontend <-> Fetch signals + wire [`IMEM_ADDR_WIDTH-1:0] fr_addr_in; + wire fr_valid_in; + wire [`INSTR_WIDTH-1:0] fr_data_out; + wire fr_valid_out; + wire [`IMEM_ADDR_WIDTH-1:0] fr_addr_out; + wire fr_ready_out; + + // Frontend <-> misc. control signals + wire per_rd_init; + wire per_zq_init; + wire per_ref_init; + wire rbe_switch_mode; + wire toggle_dll; + + // AXI streaming ports + wire [`XDMA_AXI_DATA_WIDTH-1:0] m_axis_h2c_tdata_0,xdma_h2c_tdata_0; + wire m_axis_h2c_tlast_0, xdma_h2c_tlast_0; + wire m_axis_h2c_tvalid_0, xdma_h2c_tvalid_0; + wire m_axis_h2c_tready_0, xdma_h2c_tready_0; + wire [`XDMA_AXI_DATA_WIDTH/8-1:0] m_axis_h2c_tkeep_0, xdma_h2c_tkeep_0; + wire [`XDMA_AXI_DATA_WIDTH-1:0] s_axis_c2h_tdata_0, xdma_c2h_tdata_0; + wire s_axis_c2h_tlast_0, xdma_c2h_tlast_0; + wire s_axis_c2h_tvalid_0, xdma_c2h_tvalid_0; + wire s_axis_c2h_tready_0, xdma_c2h_tready_0; + wire [`XDMA_AXI_DATA_WIDTH/8-1:0] s_axis_c2h_tkeep_0, xdma_c2h_tkeep_0; + + // ddr_pipeline <-> outer module if + wire [3:0] ddr_write; + wire [3:0] ddr_read; + wire [3:0] ddr_pre; + wire [3:0] ddr_act; + wire [3:0] ddr_ref; + wire [3:0] ddr_sre; + wire [3:0] ddr_srx; + wire [3:0] ddr_zq; + wire [3:0] ddr_nop; + wire [3:0] ddr_ap; + wire [3:0] ddr_pall; + wire [3:0] ddr_half_bl; + wire [4*`BG_WIDTH-1:0] ddr_bg; + wire [4*`BANK_WIDTH-1:0] ddr_bank; + wire [4*`COL_WIDTH-1:0] ddr_col; + wire [4*`ROW_WIDTH-1:0] ddr_row; + wire [511:0] ddr_wdata; + + // periodic maintenance signals + wire ddr_maint_read; + + // phy <-> ddr adapter and xdma app signals + // dlltoggler + wire clk_sel = 0; + wire [7:0] dllt_mc_ACT_n; + wire [135:0] dllt_mc_ADR; + wire [15:0] dllt_mc_BA; + wire [15:0] dllt_mc_BG; + wire [7:0] dllt_mc_CKE; + wire [7:0] dllt_mc_CS_n; + wire dllt_done; + // ddr adapter + wire [4:0] dBufAdr; + wire [`DQ_WIDTH*8-1:0] wrData; + wire [`DQ_WIDTH-1:0] wrDataMask; + wire [511:0] rdData; + wire [4:0] rdDataAddr; + wire [0:0] rdDataEn; + wire [0:0] rdDataEnd; + wire [0:0] per_rd_done; + wire [0:0] rmw_rd_done; + wire [4:0] wrDataAddr; + wire [0:0] wrDataEn; + wire [7:0] mc_ACT_n; + wire [135:0] mc_ADR; + wire [15:0] mc_BA; + wire [15:0] mc_BG; + wire [`CKE_WIDTH*8-1:0] mc_CKE; + wire [`CS_WIDTH*8-1:0] mc_CS_n; + wire [`ODT_WIDTH*8-1:0] mc_ODT; + wire [0:0] mcRdCAS; + wire [0:0] mcWrCAS; + wire [0:0] winInjTxn; + wire [0:0] winRmw; + wire [4:0] winBuf; + wire [1:0] winRank; + wire [5:0] tCWL; + wire dbg_clk; + wire c0_wr_rd_complete; + wire c0_ddr4_clk; + wire c0_ddr4_dll_off_clk; + wire ddr4_ui_clk; + wire c0_ddr4_rst; + wire [511:0] dbg_bus; + wire [1:0] mcCasSlot; + wire mcCasSlot2; + wire gt_data_ready; + + wire read_seq_incoming; // next few instructions will read from DRAM + wire [11:0] incoming_reads; // how many reads next few instructions will issue + wire [11:0] buffer_space; // remaining buffer size + + wire sys_rst = ~sys_rst_l; // low active signal + wire c0_init_calib_complete; + + // There is a possibility that these signals are on + // the critical path as observed in + // the previous iteration of SoftMC + reg c0_init_calib_complete_r, sys_rst_r; + wire iq_full, processing_iseq, rdback_fifo_empty; + + always @(posedge c0_ddr4_clk) begin + c0_init_calib_complete_r <= c0_init_calib_complete; + sys_rst_r <= sys_rst; + end + + reg dllt_active = 1'b0; + + `ifdef ENABLE_DLL_TOGGLER + always @(posedge c0_ddr4_clk) begin + if(toggle_dll) begin + dllt_active <= ~dllt_active; + end + if(dllt_done) begin + dllt_active <= ~dllt_active; + end + end + `endif + + `ifdef XUPP3R_x8 + phy_ddr4_x8 phy_ddr4_i( + .sys_rst (sys_rst), + .c0_sys_clk_p (c0_sys_clk_p), + .c0_sys_clk_n (c0_sys_clk_n), + `ifdef ENABLE_DLL_TOGGLER + .c0_ddr4_ui_clk (ddr4_ui_clk), + .addn_ui_clkout1 (c0_ddr4_dll_off_clk), + `else + .c0_ddr4_ui_clk (c0_ddr4_clk), + `endif + .c0_ddr4_ui_clk_sync_rst (c0_ddr4_rst), + .c0_init_calib_complete (c0_init_calib_complete), + .dbg_clk (dbg_clk), + .c0_ddr4_act_n (c0_ddr4_act_n), + .c0_ddr4_adr (c0_ddr4_adr), + .c0_ddr4_ba (c0_ddr4_ba), + .c0_ddr4_bg (c0_ddr4_bg), + .c0_ddr4_cke (c0_ddr4_cke), + .c0_ddr4_odt (c0_ddr4_odt), + .c0_ddr4_cs_n (c0_ddr4_cs_n), + .c0_ddr4_ck_t (c0_ddr4_ck_t), + .c0_ddr4_ck_c (c0_ddr4_ck_c), + .c0_ddr4_reset_n (c0_ddr4_reset_n), + .c0_ddr4_parity (c0_ddr4_parity), + .wrDataMask (wrDataMask), + .c0_ddr4_dm_dbi_n (c0_ddr4_dm_dbi_n), + .c0_ddr4_dq (c0_ddr4_dq), + .c0_ddr4_dqs_c (c0_ddr4_dqs_c), + .c0_ddr4_dqs_t (c0_ddr4_dqs_t), + + .dBufAdr (dBufAdr), + .wrData (wrData), + .rdData (rdData), + .rdDataAddr (rdDataAddr), + .rdDataEn (rdDataEn), + .rdDataEnd (rdDataEnd), + .per_rd_done (per_rd_done), + .rmw_rd_done (rmw_rd_done), + .wrDataAddr (wrDataAddr), + .wrDataEn (wrDataEn), + + .mc_ACT_n (dllt_active ? dllt_mc_ACT_n : mc_ACT_n), + .mc_ADR (dllt_active ? dllt_mc_ADR : mc_ADR), + .mc_BA (dllt_active ? dllt_mc_BA : mc_BA), + .mc_BG (dllt_active ? dllt_mc_BG : mc_BG), + .mc_CKE (dllt_active ? dllt_mc_CKE : mc_CKE), + .mc_CS_n (dllt_active ? dllt_mc_CS_n : mc_CS_n), + .mc_ODT (mc_ODT), + // CAS command slot select. Slot0 is enabled for example design. + .mcCasSlot (dllt_active ? 2'b0 : mcCasSlot), + // CAS slot 2 select. mcCasSlot2 serves a similar purpose as the mcCasSlot[1:0] signal, but mcCasSlot2 is used in timing + // critical logic in the Phy. Slot0 is enabled for example design. + .mcCasSlot2 (dllt_active ? 1'b0 : mcCasSlot2), + .mcRdCAS (dllt_active ? 1'b0 : mcRdCAS), + .mcWrCAS (dllt_active ? 1'b0 : mcWrCAS), + // Optional read command type indication. The winInjTxn signal is set to '0' for example design. + .winInjTxn ({1{1'b0}}), + // Optional read command type indication. The winRmw signal is set to '0' for example design. + .winRmw ({1{1'b0}}), + // Update VT Tracking. The gt_data_ready signal is set to '0' in this example design. + // This signal must be asserted periodically to keep the DQS Gate aligned as voltage and temperature drift. + // For more information, Refer to PG150 document. + .gt_data_ready (gt_data_ready), + .winBuf (winBuf), + .winRank (winRank), + .tCWL (tCWL), + // Debug Port + .dbg_bus (dbg_bus) + ); + `elsif XUPP3R_x4 + phy_ddr4 phy_ddr4_i( + .sys_rst (sys_rst), + .c0_sys_clk_p (c0_sys_clk_p), + .c0_sys_clk_n (c0_sys_clk_n), + + `ifdef ENABLE_DLL_TOGGLER + .c0_ddr4_ui_clk (ddr4_ui_clk), + .addn_ui_clkout1 (c0_ddr4_dll_off_clk), + `else + .c0_ddr4_ui_clk (c0_ddr4_clk), + `endif + .c0_ddr4_ui_clk_sync_rst (c0_ddr4_rst), + .c0_init_calib_complete (c0_init_calib_complete), + .dbg_clk (dbg_clk), + .c0_ddr4_act_n (c0_ddr4_act_n), + .c0_ddr4_adr (c0_ddr4_adr), + .c0_ddr4_ba (c0_ddr4_ba), + .c0_ddr4_bg (c0_ddr4_bg), + .c0_ddr4_cke (c0_ddr4_cke), + .c0_ddr4_odt (c0_ddr4_odt), + .c0_ddr4_cs_n (c0_ddr4_cs_n), + .c0_ddr4_ck_t (c0_ddr4_ck_t), + .c0_ddr4_ck_c (c0_ddr4_ck_c), + .c0_ddr4_reset_n (c0_ddr4_reset_n), + .c0_ddr4_parity (c0_ddr4_parity), + .c0_ddr4_dq (c0_ddr4_dq), + .c0_ddr4_dqs_c (c0_ddr4_dqs_c), + .c0_ddr4_dqs_t (c0_ddr4_dqs_t), + + .dBufAdr (dBufAdr), + .wrData (wrData), + .rdData (rdData), + .rdDataAddr (rdDataAddr), + .rdDataEn (rdDataEn), + .rdDataEnd (rdDataEnd), + .per_rd_done (per_rd_done), + .rmw_rd_done (rmw_rd_done), + .wrDataAddr (wrDataAddr), + .wrDataEn (wrDataEn), + + .mc_ACT_n (dllt_active ? dllt_mc_ACT_n : mc_ACT_n), + .mc_ADR (dllt_active ? dllt_mc_ADR : mc_ADR), + .mc_BA (dllt_active ? dllt_mc_BA : mc_BA), + .mc_BG (dllt_active ? dllt_mc_BG : mc_BG), + .mc_CKE (dllt_active ? dllt_mc_CKE : mc_CKE), + .mc_CS_n (dllt_active ? dllt_mc_CS_n : mc_CS_n), + .mc_ODT (mc_ODT), + // CAS command slot select. Slot0 is enabled for example design. + .mcCasSlot (dllt_active ? 0 : mcCasSlot), + // CAS slot 2 select. mcCasSlot2 serves a similar purpose as the mcCasSlot[1:0] signal, but mcCasSlot2 is used in timing + // critical logic in the Phy. Slot0 is enabled for example design. + .mcCasSlot2 (dllt_active ? 0 : mcCasSlot2), + .mcRdCAS (dllt_active ? 0 : mcRdCAS), + .mcWrCAS (dllt_active ? 0 : mcWrCAS), + // Optional read command type indication. The winInjTxn signal is set to '0' for example design. + .winInjTxn ({1{1'b0}}), + // Optional read command type indication. The winRmw signal is set to '0' for example design. + .winRmw ({1{1'b0}}), + // Update VT Tracking. The gt_data_ready signal is set to '0' in this example design. + // This signal must be asserted periodically to keep the DQS Gate aligned as voltage and temperature drift. + // For more information, Refer to PG150 document. + .gt_data_ready (gt_data_ready), + .winBuf (winBuf), + .winRank (winRank), + .tCWL (tCWL), + // Debug Port + .dbg_bus (dbg_bus) + ); + `elsif XUPP3R_x8_1R_UDIMM + phy_ddr4_udimm phy_ddr4_i( + .sys_rst (sys_rst), + .c0_sys_clk_p (c0_sys_clk_p), + .c0_sys_clk_n (c0_sys_clk_n), + `ifdef ENABLE_DLL_TOGGLER + .c0_ddr4_ui_clk (ddr4_ui_clk), + .addn_ui_clkout1 (c0_ddr4_dll_off_clk), + `else + .c0_ddr4_ui_clk (c0_ddr4_clk), + `endif + .c0_ddr4_ui_clk_sync_rst (c0_ddr4_rst), + .c0_init_calib_complete (c0_init_calib_complete), + .dbg_clk (dbg_clk), + .c0_ddr4_act_n (c0_ddr4_act_n), + .c0_ddr4_adr (c0_ddr4_adr), + .c0_ddr4_ba (c0_ddr4_ba), + .c0_ddr4_bg (c0_ddr4_bg), + .c0_ddr4_cke (c0_ddr4_cke), + .c0_ddr4_odt (c0_ddr4_odt), + .c0_ddr4_cs_n (c0_ddr4_cs_n), + .c0_ddr4_ck_t (c0_ddr4_ck_t), + .c0_ddr4_ck_c (c0_ddr4_ck_c), + .c0_ddr4_reset_n (c0_ddr4_reset_n), + .wrDataMask (wrDataMask), + .c0_ddr4_dm_dbi_n (c0_ddr4_dm_dbi_n), + .c0_ddr4_dq (c0_ddr4_dq), + .c0_ddr4_dqs_c (c0_ddr4_dqs_c), + .c0_ddr4_dqs_t (c0_ddr4_dqs_t), + + .dBufAdr (dBufAdr), + .wrData (wrData), + .rdData (rdData), + .rdDataAddr (rdDataAddr), + .rdDataEn (rdDataEn), + .rdDataEnd (rdDataEnd), + .per_rd_done (per_rd_done), + .rmw_rd_done (rmw_rd_done), + .wrDataAddr (wrDataAddr), + .wrDataEn (wrDataEn), + + .mc_ACT_n (dllt_active ? dllt_mc_ACT_n : mc_ACT_n), + .mc_ADR (dllt_active ? dllt_mc_ADR : mc_ADR), + .mc_BA (dllt_active ? dllt_mc_BA : mc_BA), + .mc_BG (dllt_active ? dllt_mc_BG : mc_BG), + .mc_CKE (dllt_active ? dllt_mc_CKE : mc_CKE), + .mc_CS_n (dllt_active ? dllt_mc_CS_n : mc_CS_n), + .mc_ODT (mc_ODT), + // CAS command slot select. Slot0 is enabled for example design. + .mcCasSlot (dllt_active ? 2'b0 : mcCasSlot), + // CAS slot 2 select. mcCasSlot2 serves a similar purpose as the mcCasSlot[1:0] signal, but mcCasSlot2 is used in timing + // critical logic in the Phy. Slot0 is enabled for example design. + .mcCasSlot2 (dllt_active ? 1'b0 : mcCasSlot2), + .mcRdCAS (dllt_active ? 1'b0 : mcRdCAS), + .mcWrCAS (dllt_active ? 1'b0 : mcWrCAS), + // Optional read command type indication. The winInjTxn signal is set to '0' for example design. + .winInjTxn ({1{1'b0}}), + // Optional read command type indication. The winRmw signal is set to '0' for example design. + .winRmw ({1{1'b0}}), + // Update VT Tracking. The gt_data_ready signal is set to '0' in this example design. + // This signal must be asserted periodically to keep the DQS Gate aligned as voltage and temperature drift. + // For more information, Refer to PG150 document. + .gt_data_ready (gt_data_ready), + .winBuf (winBuf), + .winRank (winRank), + .tCWL (tCWL), + // Debug Port + .dbg_bus (dbg_bus) + ); + `else + phy_ddr4 phy_ddr4_i( + .sys_rst (sys_rst), + .c0_sys_clk_p (c0_sys_clk_p), + .c0_sys_clk_n (c0_sys_clk_n), + + `ifdef ENABLE_DLL_TOGGLER + .c0_ddr4_ui_clk (ddr4_ui_clk), + .addn_ui_clkout1 (c0_ddr4_dll_off_clk), + `else + .c0_ddr4_ui_clk (c0_ddr4_clk), + `endif + .c0_ddr4_ui_clk_sync_rst (c0_ddr4_rst), + .c0_init_calib_complete (c0_init_calib_complete), + .dbg_clk (dbg_clk), + .c0_ddr4_act_n (c0_ddr4_act_n), + .c0_ddr4_adr (c0_ddr4_adr), + .c0_ddr4_ba (c0_ddr4_ba), + .c0_ddr4_bg (c0_ddr4_bg), + .c0_ddr4_cke (c0_ddr4_cke), + .c0_ddr4_odt (c0_ddr4_odt), + .c0_ddr4_cs_n (c0_ddr4_cs_n), + .c0_ddr4_ck_t (c0_ddr4_ck_t), + .c0_ddr4_ck_c (c0_ddr4_ck_c), + .c0_ddr4_reset_n (c0_ddr4_reset_n), + .wrDataMask (wrDataMask), + .c0_ddr4_dm_dbi_n (c0_ddr4_dm_dbi_n), + .c0_ddr4_dq (c0_ddr4_dq), + .c0_ddr4_dqs_c (c0_ddr4_dqs_c), + .c0_ddr4_dqs_t (c0_ddr4_dqs_t), + + .dBufAdr (dBufAdr), + .wrData (wrData), + .rdData (rdData), + .rdDataAddr (rdDataAddr), + .rdDataEn (rdDataEn), + .rdDataEnd (rdDataEnd), + .per_rd_done (per_rd_done), + .rmw_rd_done (rmw_rd_done), + .wrDataAddr (wrDataAddr), + .wrDataEn (wrDataEn), + + .mc_ACT_n (dllt_active ? dllt_mc_ACT_n : mc_ACT_n), + .mc_ADR (dllt_active ? dllt_mc_ADR : mc_ADR), + .mc_BA (dllt_active ? dllt_mc_BA : mc_BA), + .mc_BG (dllt_active ? dllt_mc_BG : mc_BG), + .mc_CKE (dllt_active ? dllt_mc_CKE : mc_CKE), + .mc_CS_n (dllt_active ? dllt_mc_CS_n : mc_CS_n), + .mc_ODT (mc_ODT), + // CAS command slot select. Slot0 is enabled for example design. + .mcCasSlot (dllt_active ? 0 : mcCasSlot), + // CAS slot 2 select. mcCasSlot2 serves a similar purpose as the mcCasSlot[1:0] signal, but mcCasSlot2 is used in timing + // critical logic in the Phy. Slot0 is enabled for example design. + .mcCasSlot2 (dllt_active ? 0 : mcCasSlot2), + .mcRdCAS (dllt_active ? 0 : mcRdCAS), + .mcWrCAS (dllt_active ? 0 : mcWrCAS), + // Optional read command type indication. The winInjTxn signal is set to '0' for example design. + .winInjTxn ({1{1'b0}}), + // Optional read command type indication. The winRmw signal is set to '0' for example design. + .winRmw ({1{1'b0}}), + // Update VT Tracking. The gt_data_ready signal is set to '0' in this example design. + // This signal must be asserted periodically to keep the DQS Gate aligned as voltage and temperature drift. + // For more information, Refer to PG150 document. + .gt_data_ready (gt_data_ready), + .winBuf (winBuf), + .winRank (winRank), + .tCWL (tCWL), + // Debug Port + .dbg_bus (dbg_bus) + ); + `endif + + softmc_pipeline pipeline( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst), + + .softmc_end(softmc_fin), + .read_size(incoming_reads), + .read_seq_incoming(read_seq_incoming), + .buffer_space(buffer_space), + + .addr_out(fr_addr_in), + .valid_out(fr_valid_in), + .data_in(fr_data_out), + .valid_in(fr_valid_out), + .addr_in(fr_addr_out), + .ready_out(fr_ready_out), + + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_sre(ddr_sre), + .ddr_srx(ddr_srx), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata) + ); + + `ifdef ENABLE_DLL_TOGGLER + //BUFGMUX:GeneralClockMuxBuffer + //UltraScale + //XilinxHDLLibrariesGuide, version2014.4 + BUFGMUX#(.CLK_SEL_TYPE("SYNC") //ASYNC,SYNC + )BUFGMUX_inst( + .O(c0_ddr4_clk), //1-bitoutput:Clockoutput + .I0(ddr4_ui_clk), //1-bitinput:Clockinput(S=0) + .I1(c0_ddr4_dll_off_clk), //1-bitinput:Clockinput(S=1) + .S(clk_sel) //1-bitinput:Clockselect + ); + //End of BUFGMUX_inst instantiation + `endif + + + wire frontend_ready; + + frontend #(.SIM_MEM(SIM)) frontend( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst), + + .init_calib_complete(c0_init_calib_complete_r), + .softmc_fin(softmc_fin), + .user_rst(user_rst), + + .dllt_begin(toggle_dll), + + // indicates read_back unit is ready for the next iseq + .frontend_ready(frontend_ready), + + // frontend <-> fetch stage if + .addr_in(fr_addr_in), + .valid_in(fr_valid_in), + .data_out(fr_data_out), + .valid_out(fr_valid_out), + .addr_out(fr_addr_out), + .ready_in(fr_ready_out), + + // frontend <-> xdma interface + .h2c_tdata_0(m_axis_h2c_tdata_0), + .h2c_tlast_0(m_axis_h2c_tlast_0), + .h2c_tvalid_0(m_axis_h2c_tvalid_0), + .h2c_tready_0(m_axis_h2c_tready_0), + .h2c_tkeep_0(m_axis_h2c_tkeep_0), + + .per_rd_init(per_rd_init), + .per_zq_init(per_zq_init), + .per_ref_init(per_ref_init), + .rbe_switch_mode(rbe_switch_mode) + ); + + ddr4_adapter#( + `ifdef XUPP3R + `ifdef XUPP3R_x8_1R_UDIMM + .DQ_WIDTH(64) + `else + .DQ_WIDTH(72) + `endif + `endif + ) ddr4_adapter + ( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst), + .init_calib_complete(c0_init_calib_complete_r), + //.io_config_strobe, + //.io_config, + .dBufAdr(dBufAdr), // Reserved. Should be tied low. + .wrData(wrData), // DRAM write data. There are 8 bits for each DQ lane on the DRAM bus. + .wrDataMask(wrDataMask),// DRAM write DM/DBI port.There is one bit for each byte of the wrData port. + .wrDataEn(wrDataEn), // Write data Enable. The Phy will assert this port for one cycle for each write CAS command. + .mc_ACT_n(mc_ACT_n), // DRAM ACT_n command signal for four DRAM clock cycles. + .mc_ADR(mc_ADR), // DRAM address. There are 8 bits in the fabric interface for each address bit on the DRAM bus. + .mc_BA(mc_BA), // DRAM bank address. 8 bits for each DRAM bank address. + .mc_BG(mc_BG), // DRAM bank group address. + .mc_CS_n(mc_CS_n), // DRAM CS_n + .mc_CKE(mc_CKE), // DRAM CKE + //.mc_ODT(mc_ODT), // DRAM ODT + .mcRdCAS(mcRdCAS), // Read CAS command issued. + .mcWrCAS(mcWrCAS), // Write CAS command issued. + .winRank(winRank), // Target rank for CAS commands. This value indicates which rank a CAS command is issued to. + .winBuf(winBuf), // Optional control signal. When either mcRdCAS or mcWrCAS is asserted, the Phy will store the value on the winBuf signal. + //.rdData(rdData), // DRAM read data. + .rdDataEn(rdDataEn), // Read data valid. This signal asserts for one fabric cycle for each completed read operation. + .rdDataEnd(rdDataEnd), // Unused. Tied high. + .mcCasSlot(mcCasSlot), + .mcCasSlot2(mcCasSlot2), + .gt_data_ready(gt_data_ready), + .ddr_write(ddr_write), + .ddr_read(ddr_read), + .ddr_pre(ddr_pre), + .ddr_act(ddr_act), + .ddr_ref(ddr_ref), + .ddr_sre(ddr_sre), + .ddr_srx(ddr_srx), + .ddr_zq(ddr_zq), + .ddr_nop(ddr_nop), + .ddr_ap(ddr_ap), + .ddr_pall(ddr_pall), + .ddr_half_bl(ddr_half_bl), + .ddr_bg(ddr_bg), + .ddr_bank(ddr_bank), + .ddr_col(ddr_col), + .ddr_row(ddr_row), + .ddr_wdata(ddr_wdata), + + .ddr_maint_read(per_rd_init) + ); + `ifdef XUPP3R_x8 + localparam ODTWRDEL = 5'd9; + localparam ODTWRDUR = 4'd6; + localparam ODTWRODEL = 5'd9; + localparam ODTWRODUR = 4'd6; + localparam ODTRDDEL = 5'd10; + localparam ODTRDDUR = 4'd6; + localparam ODTRDODEL = 5'd9; + localparam ODTRDODUR = 4'd6; + localparam ODTNOP = 16'h0000; + localparam ODTWR = 16'h0021; + localparam ODTRD = 16'h0012; + `elsif XUPP3R_x8_1R_UDIMM + localparam ODTWRDEL = 5'd9; + localparam ODTWRDUR = 4'd6; + localparam ODTWRODEL = 5'd9; + localparam ODTWRODUR = 4'd6; + localparam ODTRDDEL = 5'd9; + localparam ODTRDDUR = 4'd6; + localparam ODTRDODEL = 5'd9; + localparam ODTRDODUR = 4'd6; + localparam ODTNOP = 16'h0000; + localparam ODTWR = 16'h0001; + localparam ODTRD = 16'h0000; + `else + localparam ODTWRDEL = 5'd11; + localparam ODTWRDUR = 4'd6; + localparam ODTWRODEL = 5'd9; + localparam ODTWRODUR = 4'd6; + localparam ODTRDDEL = 5'd11; + localparam ODTRDDUR = 4'd6; + localparam ODTRDODEL = 5'd9; + localparam ODTRDODUR = 4'd6; + localparam ODTNOP = 16'h0000; + localparam ODTWR = 16'h0001; + localparam ODTRD = 16'h0000; + `endif + + wire tranSentC; + assign tranSentC = mcRdCAS | mcWrCAS; + + //synthesis translate_on + //******************************************************************************* + ddr4_mc_odt # ( + .ODTWR (ODTWR) + ,.ODTWRDEL (ODTWRDEL) + ,.ODTWRDUR (ODTWRDUR) + ,.ODTWRODEL (ODTWRODEL) + ,.ODTWRODUR (ODTWRODUR) + + ,.ODTRD (ODTRD) + ,.ODTRDDEL (ODTRDDEL) + ,.ODTRDDUR (ODTRDDUR) + ,.ODTRDODEL (ODTRDODEL) + ,.ODTRDODUR (ODTRDODUR) + + ,.ODTNOP (ODTNOP) + ,.ODTBITS (`ODT_WIDTH) + ,.TCQ (0.1) + )u_ddr_tb_odt( + .clk (c0_ddr4_clk) + ,.rst (c0_ddr4_rst) + ,.mc_ODT (mc_ODT) + ,.casSlot (mcCasSlot) + ,.casSlot2 (mcCasSlot2) + ,.rank (winRank) + ,.winRead (mcRdCAS) + ,.winWrite (mcWrCAS) + ,.tranSentC (tranSentC) + ); + + wire sys_clk, sys_clk_gt; + wire [2:0] msi_vector_width; + wire msi_enable; + wire user_lnk_up, usr_irq_req, usr_irq_ack; + `ifdef XUPP3R + IBUFDS_GTE4 refclk_ibuf (.O(sys_clk_gt), .ODIV2(sys_clk), .I(clk_ref_p), .CEB(1'b0), .IB(clk_ref_n)); + `else + IBUFDS_GTE3 # (.REFCLK_HROW_CK_SEL(2'b01)) refclk_ibuf (.O(sys_clk_gt), .ODIV2(sys_clk), .I(clk_ref_p), .CEB(1'b0), .IB(clk_ref_n)); + `endif + wire axi_clk, axi_rst; + + xdma xdma_i + ( + //---------------------------------------------------------------------------------------// + // PCI Express (pci_exp) Interface // + //---------------------------------------------------------------------------------------// + .sys_rst_n ( pcie_rst ), + .sys_clk ( sys_clk ), + .sys_clk_gt ( sys_clk_gt), + + // Tx + .pci_exp_txn ( pci_exp_txn ), + .pci_exp_txp ( pci_exp_txp ), + + // Rx + .pci_exp_rxn ( pci_exp_rxn ), + .pci_exp_rxp ( pci_exp_rxp ), + + // AXI streaming ports + .s_axis_c2h_tdata_0(xdma_c2h_tdata_0), + .s_axis_c2h_tlast_0(xdma_c2h_tlast_0), + .s_axis_c2h_tvalid_0(xdma_c2h_tvalid_0), + .s_axis_c2h_tready_0(xdma_c2h_tready_0), + .s_axis_c2h_tkeep_0(xdma_c2h_tkeep_0), + .m_axis_h2c_tdata_0(xdma_h2c_tdata_0), + .m_axis_h2c_tlast_0(xdma_h2c_tlast_0), + .m_axis_h2c_tvalid_0(xdma_h2c_tvalid_0), + .m_axis_h2c_tready_0(xdma_h2c_tready_0), + .m_axis_h2c_tkeep_0(xdma_h2c_tkeep_0), + + .usr_irq_req (1'b0), + .usr_irq_ack (usr_irq_ack), + .msi_enable (msi_enable), + .msi_vector_width (msi_vector_width), + + + // Config managemnet interface + .cfg_mgmt_addr ( 19'b0 ), + .cfg_mgmt_write ( 1'b0 ), + .cfg_mgmt_write_data ( 32'b0 ), + .cfg_mgmt_byte_enable ( 4'b0 ), + .cfg_mgmt_read ( 1'b0 ), + .cfg_mgmt_read_data (), + .cfg_mgmt_read_write_done (), + `ifndef XUPP3R + .cfg_mgmt_type1_cfg_reg_access ( 1'b0 ), + //---------- Shared Logic Internal ------------------------- + .int_qpll1lock_out ( ), + .int_qpll1outrefclk_out ( ), + .int_qpll1outclk_out ( ), + `endif + + //-- AXI Global + .axi_aclk (axi_clk), // AXI i-face clock driven from pcie clk + .axi_aresetn (axi_rst), // reset synchronous to axi_clk + + .user_lnk_up ( user_lnk_up ) + ); + + // Clock converter for the c2h interface + axis_clock_converter axis_clk_conv_i0 + ( + .s_axis_tvalid(s_axis_c2h_tvalid_0), + .s_axis_tlast(s_axis_c2h_tlast_0), + .s_axis_tdata(s_axis_c2h_tdata_0), + .s_axis_tkeep(s_axis_c2h_tkeep_0), + .s_axis_tready(s_axis_c2h_tready_0), + .m_axis_tvalid(xdma_c2h_tvalid_0), + .m_axis_tlast(xdma_c2h_tlast_0), + .m_axis_tdata(xdma_c2h_tdata_0), + .m_axis_tkeep(xdma_c2h_tkeep_0), + .m_axis_tready(xdma_c2h_tready_0), + .s_axis_aresetn(~c0_ddr4_rst), + .s_axis_aclk(c0_ddr4_clk), + .m_axis_aresetn(axi_rst), + .m_axis_aclk(axi_clk) + ); + + // Clock converter for the h2c interface + axis_clock_converter axis_clk_conv_i1 + ( + .m_axis_tvalid(m_axis_h2c_tvalid_0), + .m_axis_tlast(m_axis_h2c_tlast_0), + .m_axis_tdata(m_axis_h2c_tdata_0), + .m_axis_tkeep(m_axis_h2c_tkeep_0), + .m_axis_tready(m_axis_h2c_tready_0), + .s_axis_tvalid(xdma_h2c_tvalid_0), + .s_axis_tlast(xdma_h2c_tlast_0), + .s_axis_tdata(xdma_h2c_tdata_0), + .s_axis_tkeep(xdma_h2c_tkeep_0), + .s_axis_tready(xdma_h2c_tready_0), + .m_axis_aresetn(~c0_ddr4_rst), + .m_axis_aclk(c0_ddr4_clk), + .s_axis_aresetn(axi_rst), + .s_axis_aclk(axi_clk) + ); + + readback_engine rbe( + + // common signals + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst), + + // other ctrl signals + .flush(frontend_ready), + .switch_mode(rbe_switch_mode), + .read_seq_incoming(read_seq_incoming), // next few instructions will read from DRAM + .incoming_reads(incoming_reads), // how many reads next few instructions will issue + .buffer_space(buffer_space), // remaining buffer size + // DRAM <-> engine if + .rd_data(rdData), + .rd_valid(rdDataEn), + + // rbe <-> rf interface + .ddr_wdata(ddr_wdata), + + .per_rd_init(per_rd_init), + .per_zq_init(per_zq_init), + .per_ref_init(per_ref_init), + + // rbe <-> xdma if + .c2h_tdata_0(s_axis_c2h_tdata_0), + .c2h_tlast_0(s_axis_c2h_tlast_0), + .c2h_tvalid_0(s_axis_c2h_tvalid_0), + .c2h_tready_0(s_axis_c2h_tready_0), + .c2h_tkeep_0(s_axis_c2h_tkeep_0) + + ); + + `ifdef ENABLE_DLL_TOGGLER + dll_toggler dllt + ( + .clk(c0_ddr4_clk), + .rst(c0_ddr4_rst || user_rst || ~c0_init_calib_complete_r), + .toggle_valid(toggle_dll), + .mc_ACT_n(dllt_mc_ACT_n), // DRAM ACT_n command signal for four DRAM clock cycles. + .mc_ADR(dllt_mc_ADR), // DRAM address. There are 8 bits in the fabric interface for each address bit on the DRAM bus. + .mc_BA(dllt_mc_BA), // DRAM bank address. 8 bits for each DRAM bank address. + .mc_BG(dllt_mc_BG), // DRAM bank group address. + .mc_CS_n(dllt_mc_CS_n), // DRAM CS_n + .mc_CKE(dllt_mc_CKE), + .clk_sel(clk_sel), + .dllt_done(dllt_done) + ); + `endif +endmodule diff --git a/sources/scripts/generate.tcl b/sources/scripts/generate.tcl new file mode 100644 index 0000000..375890b --- /dev/null +++ b/sources/scripts/generate.tcl @@ -0,0 +1,111 @@ +# Returns all directory names in a given directory (non-recursive) +proc get_dir_names {path} { + set dirs [glob -directory "$path" -type d *] + set dir_names {} + foreach project $dirs { + lappend dir_names [string map "$path/ \"\"" $project] + } + return $dir_names +} + +# Version safe wait procedure for a single run. wait_on_run changed to wait_on_runs after Vivado 2021.2 +proc safe_wait { run } { + if { [catch wait_on_runs $run] } { + wait_on_run $run + } +} + +# Version safe wait procedure for multiple runs. wait_on_run changed to wait_on_runs after Vivado 2021.2 +proc safe_wait_multiple { run_list } { + if { [catch wait_on_runs $run_list] } { + foreach run $run_list { wait_on_run $run } + } +} + +# Check required args +if { $argc != 1 } { + puts "Please provide <number_of_threads> to script." + puts "For example, vivado -mode batch generate.tcl -tclargs 12" + exit 0 +} + +# Set path variables +set PROJECT_PATH [pwd] +set PROJECT_NAME [string map ".xpr \"\" $PROJECT_PATH/ \"\"" [glob -directory $PROJECT_PATH *.xpr]] +set PROJECT_IP_PATH "$PROJECT_PATH/$PROJECT_NAME.srcs/sources_1/ip" +set NUM_JOBS [lindex $argv 0] + +# Check if project exists +if { ![file exist $PROJECT_PATH] } { + puts "Project not found on $PROJECT_PATH." + exit 0 +} + +# Open the project +open_project "$PROJECT_PATH/$PROJECT_NAME.xpr" + +update_compile_order -fileset sources_1 + +# Reset previous runs +reset_project + +# IP run variables +set ip_list [get_ips] +set ip_runs {} +set phy_list {} +foreach ip $ip_list { + if {[string first "phy_" $ip] != -1} { + lappend phy_list "$ip" + } + lappend ip_runs "${ip}_synth_1" +} + + +# Reset ourput products of all IPs +reset_target all $ip_list + +# Set fileset to main sources +update_compile_order -fileset sources_1 + +# Re-generate all output products +generate_target all $ip_list + +foreach ip $ip_list { + create_ip_run [get_ips $ip] +} + +launch_runs $ip_runs -verbose -jobs $NUM_JOBS +safe_wait_multiple $ip_runs + +# Lock PHY and Apply patches +foreach phy_ip $phy_list { + set_property IS_LOCKED true [get_files "${phy_ip}.xci"] +} + +if { [catch {exec "${PROJECT_PATH}/apply_patches.sh"} result] != 0 } { + puts "Script 'apply_patches.sh' has no return value, assumed exit value 0 (TCL_OK). Check first incase any errors." +} + +# Reset and relaunch PHY runs +foreach phy_ip $phy_list { + reset_run [get_runs "${phy_ip}_synth_1"] +} + +foreach phy_ip $phy_list { + launch_run [get_runs "${phy_ip}_synth_1"] +} + +foreach phy_ip $phy_list { + safe_wait "${phy_ip}_synth_1" +} + +# Synthesis +launch_runs synth_1 -verbose -jobs $NUM_JOBS +safe_wait "synth_1" + +# Implementation and Bitstream generation +launch_runs impl_1 -to_step write_bitstream -verbose -jobs $NUM_JOBS +safe_wait "impl_1" + +# Done +puts "Bitstream generated successfully. Exitting." diff --git a/sources/xdma_driver/.gitignore b/sources/xdma_driver/.gitignore new file mode 100644 index 0000000..951f319 --- /dev/null +++ b/sources/xdma_driver/.gitignore @@ -0,0 +1,9 @@ +xdma/*.cmd +xdma/*.d +xdma/*.o +xdma/*.ko +xdma/*.o.cmd +xdma/*.order +xdma/*.mod +xdma/*.mod.c +xdma/*.symvers
\ No newline at end of file diff --git a/sources/xdma_driver/COPYING b/sources/xdma_driver/COPYING new file mode 100644 index 0000000..3912109 --- /dev/null +++ b/sources/xdma_driver/COPYING @@ -0,0 +1,340 @@ + GNU GENERAL PUBLIC LICENSE + Version 2, June 1991 + + Copyright (C) 1989, 1991 Free Software Foundation, Inc. + 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA + Everyone is permitted to copy and distribute verbatim copies + of this license document, but changing it is not allowed. + + Preamble + + The licenses for most software are designed to take away your +freedom to share and change it. By contrast, the GNU General Public +License is intended to guarantee your freedom to share and change free +software--to make sure the software is free for all its users. This +General Public License applies to most of the Free Software +Foundation's software and to any other program whose authors commit to +using it. (Some other Free Software Foundation software is covered by +the GNU Library General Public License instead.) You can apply it to +your programs, too. + + When we speak of free software, we are referring to freedom, not +price. Our General Public Licenses are designed to make sure that you +have the freedom to distribute copies of free software (and charge for +this service if you wish), that you receive source code or can get it +if you want it, that you can change the software or use pieces of it +in new free programs; and that you know you can do these things. + + To protect your rights, we need to make restrictions that forbid +anyone to deny you these rights or to ask you to surrender the rights. +These restrictions translate to certain responsibilities for you if you +distribute copies of the software, or if you modify it. + + For example, if you distribute copies of such a program, whether +gratis or for a fee, you must give the recipients all the rights that +you have. You must make sure that they, too, receive or can get the +source code. And you must show them these terms so they know their +rights. + + We protect your rights with two steps: (1) copyright the software, and +(2) offer you this license which gives you legal permission to copy, +distribute and/or modify the software. + + Also, for each author's protection and ours, we want to make certain +that everyone understands that there is no warranty for this free +software. If the software is modified by someone else and passed on, we +want its recipients to know that what they have is not the original, so +that any problems introduced by others will not reflect on the original +authors' reputations. + + Finally, any free program is threatened constantly by software +patents. We wish to avoid the danger that redistributors of a free +program will individually obtain patent licenses, in effect making the +program proprietary. To prevent this, we have made it clear that any +patent must be licensed for everyone's free use or not licensed at all. + + The precise terms and conditions for copying, distribution and +modification follow. + + GNU GENERAL PUBLIC LICENSE + TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION + + 0. This License applies to any program or other work which contains +a notice placed by the copyright holder saying it may be distributed +under the terms of this General Public License. The "Program", below, +refers to any such program or work, and a "work based on the Program" +means either the Program or any derivative work under copyright law: +that is to say, a work containing the Program or a portion of it, +either verbatim or with modifications and/or translated into another +language. (Hereinafter, translation is included without limitation in +the term "modification".) Each licensee is addressed as "you". + +Activities other than copying, distribution and modification are not +covered by this License; they are outside its scope. The act of +running the Program is not restricted, and the output from the Program +is covered only if its contents constitute a work based on the +Program (independent of having been made by running the Program). +Whether that is true depends on what the Program does. + + 1. You may copy and distribute verbatim copies of the Program's +source code as you receive it, in any medium, provided that you +conspicuously and appropriately publish on each copy an appropriate +copyright notice and disclaimer of warranty; keep intact all the +notices that refer to this License and to the absence of any warranty; +and give any other recipients of the Program a copy of this License +along with the Program. + +You may charge a fee for the physical act of transferring a copy, and +you may at your option offer warranty protection in exchange for a fee. + + 2. You may modify your copy or copies of the Program or any portion +of it, thus forming a work based on the Program, and copy and +distribute such modifications or work under the terms of Section 1 +above, provided that you also meet all of these conditions: + + a) You must cause the modified files to carry prominent notices + stating that you changed the files and the date of any change. + + b) You must cause any work that you distribute or publish, that in + whole or in part contains or is derived from the Program or any + part thereof, to be licensed as a whole at no charge to all third + parties under the terms of this License. + + c) If the modified program normally reads commands interactively + when run, you must cause it, when started running for such + interactive use in the most ordinary way, to print or display an + announcement including an appropriate copyright notice and a + notice that there is no warranty (or else, saying that you provide + a warranty) and that users may redistribute the program under + these conditions, and telling the user how to view a copy of this + License. (Exception: if the Program itself is interactive but + does not normally print such an announcement, your work based on + the Program is not required to print an announcement.) + +These requirements apply to the modified work as a whole. If +identifiable sections of that work are not derived from the Program, +and can be reasonably considered independent and separate works in +themselves, then this License, and its terms, do not apply to those +sections when you distribute them as separate works. But when you +distribute the same sections as part of a whole which is a work based +on the Program, the distribution of the whole must be on the terms of +this License, whose permissions for other licensees extend to the +entire whole, and thus to each and every part regardless of who wrote it. + +Thus, it is not the intent of this section to claim rights or contest +your rights to work written entirely by you; rather, the intent is to +exercise the right to control the distribution of derivative or +collective works based on the Program. + +In addition, mere aggregation of another work not based on the Program +with the Program (or with a work based on the Program) on a volume of +a storage or distribution medium does not bring the other work under +the scope of this License. + + 3. You may copy and distribute the Program (or a work based on it, +under Section 2) in object code or executable form under the terms of +Sections 1 and 2 above provided that you also do one of the following: + + a) Accompany it with the complete corresponding machine-readable + source code, which must be distributed under the terms of Sections + 1 and 2 above on a medium customarily used for software interchange; or, + + b) Accompany it with a written offer, valid for at least three + years, to give any third party, for a charge no more than your + cost of physically performing source distribution, a complete + machine-readable copy of the corresponding source code, to be + distributed under the terms of Sections 1 and 2 above on a medium + customarily used for software interchange; or, + + c) Accompany it with the information you received as to the offer + to distribute corresponding source code. (This alternative is + allowed only for noncommercial distribution and only if you + received the program in object code or executable form with such + an offer, in accord with Subsection b above.) + +The source code for a work means the preferred form of the work for +making modifications to it. For an executable work, complete source +code means all the source code for all modules it contains, plus any +associated interface definition files, plus the scripts used to +control compilation and installation of the executable. However, as a +special exception, the source code distributed need not include +anything that is normally distributed (in either source or binary +form) with the major components (compiler, kernel, and so on) of the +operating system on which the executable runs, unless that component +itself accompanies the executable. + +If distribution of executable or object code is made by offering +access to copy from a designated place, then offering equivalent +access to copy the source code from the same place counts as +distribution of the source code, even though third parties are not +compelled to copy the source along with the object code. + + 4. You may not copy, modify, sublicense, or distribute the Program +except as expressly provided under this License. Any attempt +otherwise to copy, modify, sublicense or distribute the Program is +void, and will automatically terminate your rights under this License. +However, parties who have received copies, or rights, from you under +this License will not have their licenses terminated so long as such +parties remain in full compliance. + + 5. You are not required to accept this License, since you have not +signed it. However, nothing else grants you permission to modify or +distribute the Program or its derivative works. These actions are +prohibited by law if you do not accept this License. Therefore, by +modifying or distributing the Program (or any work based on the +Program), you indicate your acceptance of this License to do so, and +all its terms and conditions for copying, distributing or modifying +the Program or works based on it. + + 6. Each time you redistribute the Program (or any work based on the +Program), the recipient automatically receives a license from the +original licensor to copy, distribute or modify the Program subject to +these terms and conditions. You may not impose any further +restrictions on the recipients' exercise of the rights granted herein. +You are not responsible for enforcing compliance by third parties to +this License. + + 7. If, as a consequence of a court judgment or allegation of patent +infringement or for any other reason (not limited to patent issues), +conditions are imposed on you (whether by court order, agreement or +otherwise) that contradict the conditions of this License, they do not +excuse you from the conditions of this License. If you cannot +distribute so as to satisfy simultaneously your obligations under this +License and any other pertinent obligations, then as a consequence you +may not distribute the Program at all. For example, if a patent +license would not permit royalty-free redistribution of the Program by +all those who receive copies directly or indirectly through you, then +the only way you could satisfy both it and this License would be to +refrain entirely from distribution of the Program. + +If any portion of this section is held invalid or unenforceable under +any particular circumstance, the balance of the section is intended to +apply and the section as a whole is intended to apply in other +circumstances. + +It is not the purpose of this section to induce you to infringe any +patents or other property right claims or to contest validity of any +such claims; this section has the sole purpose of protecting the +integrity of the free software distribution system, which is +implemented by public license practices. Many people have made +generous contributions to the wide range of software distributed +through that system in reliance on consistent application of that +system; it is up to the author/donor to decide if he or she is willing +to distribute software through any other system and a licensee cannot +impose that choice. + +This section is intended to make thoroughly clear what is believed to +be a consequence of the rest of this License. + + 8. If the distribution and/or use of the Program is restricted in +certain countries either by patents or by copyrighted interfaces, the +original copyright holder who places the Program under this License +may add an explicit geographical distribution limitation excluding +those countries, so that distribution is permitted only in or among +countries not thus excluded. In such case, this License incorporates +the limitation as if written in the body of this License. + + 9. The Free Software Foundation may publish revised and/or new versions +of the General Public License from time to time. Such new versions will +be similar in spirit to the present version, but may differ in detail to +address new problems or concerns. + +Each version is given a distinguishing version number. If the Program +specifies a version number of this License which applies to it and "any +later version", you have the option of following the terms and conditions +either of that version or of any later version published by the Free +Software Foundation. If the Program does not specify a version number of +this License, you may choose any version ever published by the Free Software +Foundation. + + 10. If you wish to incorporate parts of the Program into other free +programs whose distribution conditions are different, write to the author +to ask for permission. For software which is copyrighted by the Free +Software Foundation, write to the Free Software Foundation; we sometimes +make exceptions for this. Our decision will be guided by the two goals +of preserving the free status of all derivatives of our free software and +of promoting the sharing and reuse of software generally. + + NO WARRANTY + + 11. BECAUSE THE PROGRAM IS LICENSED FREE OF CHARGE, THERE IS NO WARRANTY +FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN +OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES +PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED +OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF +MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS +TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE +PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING, +REPAIR OR CORRECTION. + + 12. IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING +WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MAY MODIFY AND/OR +REDISTRIBUTE THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, +INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING +OUT OF THE USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED +TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY +YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER +PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE +POSSIBILITY OF SUCH DAMAGES. + + END OF TERMS AND CONDITIONS + + How to Apply These Terms to Your New Programs + + If you develop a new program, and you want it to be of the greatest +possible use to the public, the best way to achieve this is to make it +free software which everyone can redistribute and change under these terms. + + To do so, attach the following notices to the program. It is safest +to attach them to the start of each source file to most effectively +convey the exclusion of warranty; and each file should have at least +the "copyright" line and a pointer to where the full notice is found. + + <one line to give the program's name and a brief idea of what it does.> + Copyright (C) <year> <name of author> + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation; either version 2 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA + + +Also add information on how to contact you by electronic and paper mail. + +If the program is interactive, make it output a short notice like this +when it starts in an interactive mode: + + Gnomovision version 69, Copyright (C) year name of author + Gnomovision comes with ABSOLUTELY NO WARRANTY; for details type `show w'. + This is free software, and you are welcome to redistribute it + under certain conditions; type `show c' for details. + +The hypothetical commands `show w' and `show c' should show the appropriate +parts of the General Public License. Of course, the commands you use may +be called something other than `show w' and `show c'; they could even be +mouse-clicks or menu items--whatever suits your program. + +You should also get your employer (if you work as a programmer) or your +school, if any, to sign a "copyright disclaimer" for the program, if +necessary. Here is a sample; alter the names: + + Yoyodyne, Inc., hereby disclaims all copyright interest in the program + `Gnomovision' (which makes passes at compilers) written by James Hacker. + + <signature of Ty Coon>, 1 April 1989 + Ty Coon, President of Vice + +This General Public License does not permit incorporating your program into +proprietary programs. If your program is a subroutine library, you may +consider it more useful to permit linking proprietary applications with the +library. If this is what you want to do, use the GNU Library General +Public License instead of this License. diff --git a/sources/xdma_driver/LICENSE b/sources/xdma_driver/LICENSE new file mode 100644 index 0000000..703e647 --- /dev/null +++ b/sources/xdma_driver/LICENSE @@ -0,0 +1,30 @@ +BSD License + +For Xilinx DMA IP software + +Copyright (c) 2016-present, Xilinx, Inc. All rights reserved. + +Redistribution and use in source and binary forms, with or without modification, +are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, this + list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + + * Neither the name Xilinx nor the names of its contributors may be used to + endorse or promote products derived from this software without specific + prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND +ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES +(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON +ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/sources/xdma_driver/include/libxdma_api.h b/sources/xdma_driver/include/libxdma_api.h new file mode 100644 index 0000000..00d4355 --- /dev/null +++ b/sources/xdma_driver/include/libxdma_api.h @@ -0,0 +1,130 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XDMA_BASE_API_H__ +#define __XDMA_BASE_API_H__ + +#include <linux/types.h> +#include <linux/scatterlist.h> +#include <linux/interrupt.h> + +/* + * functions exported by the xdma driver + */ + +typedef struct { + u64 write_submitted; + u64 write_completed; + u64 read_requested; + u64 read_completed; + u64 restart; + u64 open; + u64 close; + u64 msix_trigger; +} xdma_statistics; + +/* + * This struct should be constantly updated by XMDA using u64_stats_* APIs + * The front end will read the structure without locking (That's why updating atomically is a must) + * every time it prints the statistics. + */ +//static XDMA_Statistics stats; + +/* + * xdma_device_open - read the pci bars and configure the fpga + * should be called from probe() + * NOTE: + * user interrupt will not enabled until xdma_user_isr_enable() + * is called + * @pdev: ptr to pci_dev + * @mod_name: the module name to be used for request_irq + * @user_max: max # of user/event (interrupts) to be configured + * @channel_max: max # of c2h and h2c channels to be configured + * NOTE: if the user/channel provisioned is less than the max specified, + * libxdma will update the user_max/channel_max + * returns + * a opaque handle (for libxdma to identify the device) + * NULL, in case of error + */ +void *xdma_device_open(const char *mod_name, struct pci_dev *pdev, + int *user_max, int *h2c_channel_max, int *c2h_channel_max); + +/* + * xdma_device_close - prepare fpga for removal: disable all interrupts (users + * and xdma) and release all resources + * should called from remove() + * @pdev: ptr to struct pci_dev + * @tuples: from xdma_device_open() + */ +void xdma_device_close(struct pci_dev *pdev, void *dev_handle); + +/* + * xdma_device_restart - restart the fpga + * @pdev: ptr to struct pci_dev + * TODO: + * may need more refining on the parameter list + * return < 0 in case of error + * TODO: exact error code will be defined later + */ +int xdma_device_restart(struct pci_dev *pdev, void *dev_handle); + +/* + * xdma_user_isr_register - register a user ISR handler + * It is expected that the xdma will register the ISR, and for the user + * interrupt, it will call the corresponding handle if it is registered and + * enabled. + * + * @pdev: ptr to the the pci_dev struct + * @mask: bitmask of user interrupts (0 ~ 15)to be registered + * bit 0: user interrupt 0 + * ... + * bit 15: user interrupt 15 + * any bit above bit 15 will be ignored. + * @handler: the correspoinding handler + * a NULL handler will be treated as de-registeration + * @name: to be passed to the handler, ignored if handler is NULL` + * @dev: to be passed to the handler, ignored if handler is NULL` + * return < 0 in case of error + * TODO: exact error code will be defined later + */ +int xdma_user_isr_register(void *dev_hndl, unsigned int mask, + irq_handler_t handler, void *dev); + +/* + * xdma_user_isr_enable/disable - enable or disable user interrupt + * @pdev: ptr to the the pci_dev struct + * @mask: bitmask of user interrupts (0 ~ 15)to be registered + * return < 0 in case of error + * TODO: exact error code will be defined later + */ +int xdma_user_isr_enable(void *dev_hndl, unsigned int mask); +int xdma_user_isr_disable(void *dev_hndl, unsigned int mask); + +/* + * xdma_xfer_submit - submit data for dma operation (for both read and write) + * This is a blocking call + * @channel: channle number (< channel_max) + * == channel_max means libxdma can pick any channel available:q + + * @dir: DMA_FROM/TO_DEVICE + * @offset: offset into the DDR/BRAM memory to read from or write to + * @sg_tbl: the scatter-gather list of data buffers + * @timeout: timeout in mili-seconds, *currently ignored + * return # of bytes transfered or + * < 0 in case of error + * TODO: exact error code will be defined later + */ +ssize_t xdma_xfer_submit(void *dev_hndl, int channel, bool write, u64 ep_addr, + struct sg_table *sgt, bool dma_mapped, int timeout_ms); + + +#endif diff --git a/sources/xdma_driver/libxdma/Makefile b/sources/xdma_driver/libxdma/Makefile new file mode 100644 index 0000000..1003f0f --- /dev/null +++ b/sources/xdma_driver/libxdma/Makefile @@ -0,0 +1,25 @@ +SHELL = /bin/bash + +topdir := $(shell cd $(src)/.. && pwd) + +TARGET_MODULE:=libxdma + +EXTRA_CFLAGS := -I$(topdir)/include +EXTRA_CFLAGS += -D__LIBXDMA_MOD__ + +ifneq ($(KERNELRELEASE),) + obj-m := $(TARGET_MODULE).o +# $(TARGET_MODULE)-objs := libxdma.o +else + BUILDSYSTEM_DIR:=/lib/modules/$(shell uname -r)/build + PWD:=$(shell pwd) +all : + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) modules + +clean: + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) clean + +install: all + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) modules_install + +endif diff --git a/sources/xdma_driver/libxdma/libxdma.c b/sources/xdma_driver/libxdma/libxdma.c new file mode 100644 index 0000000..b9884d7 --- /dev/null +++ b/sources/xdma_driver/libxdma/libxdma.c @@ -0,0 +1,4442 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include <linux/module.h> +#include <linux/kernel.h> +#include <linux/string.h> +#include <linux/mm.h> +#include <linux/errno.h> +#include <linux/sched.h> +#include <linux/vmalloc.h> + +#include "libxdma.h" +#include "libxdma_api.h" +#include "cdev_sgdma.h" + +/* SECTION: Module licensing */ + +#ifdef __LIBXDMA_MOD__ +#include "version.h" +#define DRV_MODULE_NAME "libxdma" +#define DRV_MODULE_DESC "Xilinx XDMA Base Driver" +#define DRV_MODULE_RELDATE "Feb. 2018" + +static char version[] = + DRV_MODULE_DESC " " DRV_MODULE_NAME " v" DRV_MODULE_VERSION "\n"; + +MODULE_AUTHOR("Xilinx, Inc."); +MODULE_DESCRIPTION(DRV_MODULE_DESC); +MODULE_VERSION(DRV_MODULE_VERSION); +MODULE_LICENSE("Dual BSD/GPL"); +#endif + +/* Module Parameters */ +static unsigned int poll_mode; +module_param(poll_mode, uint, 0644); +MODULE_PARM_DESC(poll_mode, "Set 1 for hw polling, default is 0 (interrupts)"); + +static unsigned int interrupt_mode; +module_param(interrupt_mode, uint, 0644); +MODULE_PARM_DESC(interrupt_mode, "0 - MSI-x , 1 - MSI, 2 - Legacy"); + +static unsigned int enable_credit_mp; +module_param(enable_credit_mp, uint, 0644); +MODULE_PARM_DESC(enable_credit_mp, "Set 1 to enable creidt feature, default is 0 (no credit control)"); + +unsigned int desc_blen_max = XDMA_DESC_BLEN_MAX; +module_param(desc_blen_max, uint, 0644); +MODULE_PARM_DESC(desc_blen_max, "per descriptor max. buffer length, default is (1 << 28) - 1"); + +/* + * xdma device management + * maintains a list of the xdma devices + */ +static LIST_HEAD(xdev_list); +static DEFINE_MUTEX(xdev_mutex); + +static LIST_HEAD(xdev_rcu_list); +static DEFINE_SPINLOCK(xdev_rcu_lock); + +#ifndef list_last_entry +#define list_last_entry(ptr, type, member) \ + list_entry((ptr)->prev, type, member) +#endif + +static inline void xdev_list_add(struct xdma_dev *xdev) +{ + mutex_lock(&xdev_mutex); + if (list_empty(&xdev_list)) + xdev->idx = 0; + else { + struct xdma_dev *last; + + last = list_last_entry(&xdev_list, struct xdma_dev, list_head); + xdev->idx = last->idx + 1; + } + list_add_tail(&xdev->list_head, &xdev_list); + mutex_unlock(&xdev_mutex); + + dbg_init("dev %s, xdev 0x%p, xdma idx %d.\n", + dev_name(&xdev->pdev->dev), xdev, xdev->idx); + + spin_lock(&xdev_rcu_lock); + list_add_tail_rcu(&xdev->rcu_node, &xdev_rcu_list); + spin_unlock(&xdev_rcu_lock); +} + +#undef list_last_entry + +static inline void xdev_list_remove(struct xdma_dev *xdev) +{ + mutex_lock(&xdev_mutex); + list_del(&xdev->list_head); + mutex_unlock(&xdev_mutex); + + spin_lock(&xdev_rcu_lock); + list_del_rcu(&xdev->rcu_node); + spin_unlock(&xdev_rcu_lock); + synchronize_rcu(); +} + +struct xdma_dev *xdev_find_by_pdev(struct pci_dev *pdev) +{ + struct xdma_dev *xdev, *tmp; + + mutex_lock(&xdev_mutex); + list_for_each_entry_safe(xdev, tmp, &xdev_list, list_head) { + if (xdev->pdev == pdev) { + mutex_unlock(&xdev_mutex); + return xdev; + } + } + mutex_unlock(&xdev_mutex); + return NULL; +} +EXPORT_SYMBOL_GPL(xdev_find_by_pdev); + +static inline int debug_check_dev_hndl(const char *fname, struct pci_dev *pdev, + void *hndl) +{ + struct xdma_dev *xdev; + + if (!pdev) + return -EINVAL; + + xdev = xdev_find_by_pdev(pdev); + if (!xdev) { + pr_info("%s pdev 0x%p, hndl 0x%p, NO match found!\n", + fname, pdev, hndl); + return -EINVAL; + } + if (xdev != hndl) { + pr_err("%s pdev 0x%p, hndl 0x%p != 0x%p!\n", + fname, pdev, hndl, xdev); + return -EINVAL; + } + + return 0; +} + +#ifdef __LIBXDMA_DEBUG__ +/* SECTION: Function definitions */ +inline void __write_register(const char *fn, u32 value, void *iomem, unsigned long off) +{ + pr_err("%s: w reg 0x%lx(0x%p), 0x%x.\n", fn, off, iomem, value); + iowrite32(value, iomem); +} +#define write_register(v,mem,off) __write_register(__func__, v, mem, off) +#else +#define write_register(v,mem,off) iowrite32(v, mem) +#endif + +inline u32 read_register(void *iomem) +{ + return ioread32(iomem); +} + +static inline u32 build_u32(u32 hi, u32 lo) +{ + return ((hi & 0xFFFFUL) << 16) | (lo & 0xFFFFUL); +} + +static inline u64 build_u64(u64 hi, u64 lo) +{ + return ((hi & 0xFFFFFFFULL) << 32) | (lo & 0xFFFFFFFFULL); +} + +static void check_nonzero_interrupt_status(struct xdma_dev *xdev) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + u32 w; + + w = read_register(®->user_int_enable); + if (w) + pr_info("%s xdma%d user_int_enable = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->channel_int_enable); + if (w) + pr_info("%s xdma%d channel_int_enable = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->user_int_request); + if (w) + pr_info("%s xdma%d user_int_request = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + w = read_register(®->channel_int_request); + if (w) + pr_info("%s xdma%d channel_int_request = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->user_int_pending); + if (w) + pr_info("%s xdma%d user_int_pending = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + w = read_register(®->channel_int_pending); + if (w) + pr_info("%s xdma%d channel_int_pending = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); +} + +/* channel_interrupts_enable -- Enable interrupts we are interested in */ +static void channel_interrupts_enable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->channel_int_enable_w1s, XDMA_OFS_INT_CTRL); +} + +/* channel_interrupts_disable -- Disable interrupts we not interested in */ +static void channel_interrupts_disable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->channel_int_enable_w1c, XDMA_OFS_INT_CTRL); +} + +/* user_interrupts_enable -- Enable interrupts we are interested in */ +static void user_interrupts_enable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->user_int_enable_w1s, XDMA_OFS_INT_CTRL); +} + +/* user_interrupts_disable -- Disable interrupts we not interested in */ +static void user_interrupts_disable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->user_int_enable_w1c, XDMA_OFS_INT_CTRL); +} + +/* read_interrupts -- Print the interrupt controller status */ +static u32 read_interrupts(struct xdma_dev *xdev) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + u32 lo; + u32 hi; + + /* extra debugging; inspect complete engine set of registers */ + hi = read_register(®->user_int_request); + dbg_io("ioread32(0x%p) returned 0x%08x (user_int_request).\n", + ®->user_int_request, hi); + lo = read_register(®->channel_int_request); + dbg_io("ioread32(0x%p) returned 0x%08x (channel_int_request)\n", + ®->channel_int_request, lo); + + /* return interrupts: user in upper 16-bits, channel in lower 16-bits */ + return build_u32(hi, lo); +} + +void enable_perf(struct xdma_engine *engine) +{ + u32 w; + + w = XDMA_PERF_CLEAR; + write_register(w, &engine->regs->perf_ctrl, + (unsigned long)(&engine->regs->perf_ctrl) - + (unsigned long)(&engine->regs)); + read_register(&engine->regs->identifier); + w = XDMA_PERF_AUTO | XDMA_PERF_RUN; + write_register(w, &engine->regs->perf_ctrl, + (unsigned long)(&engine->regs->perf_ctrl) - + (unsigned long)(&engine->regs)); + read_register(&engine->regs->identifier); + + dbg_perf("IOCTL_XDMA_PERF_START\n"); + +} +EXPORT_SYMBOL_GPL(enable_perf); + +void get_perf_stats(struct xdma_engine *engine) +{ + u32 hi; + u32 lo; + + BUG_ON(!engine); + + if (!engine->xdma_perf) { + pr_info("%s perf struct not set up.\n", engine->name); + return; + } + + hi = 0; + lo = read_register(&engine->regs->completed_desc_count); + engine->xdma_perf->iterations = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_cyc_hi); + lo = read_register(&engine->regs->perf_cyc_lo); + + engine->xdma_perf->clock_cycle_count = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_dat_hi); + lo = read_register(&engine->regs->perf_dat_lo); + engine->xdma_perf->data_cycle_count = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_pnd_hi); + lo = read_register(&engine->regs->perf_pnd_lo); + engine->xdma_perf->pending_count = build_u64(hi, lo); +} +EXPORT_SYMBOL_GPL(get_perf_stats); + +static void engine_reg_dump(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + + w = read_register(&engine->regs->identifier); + pr_info("%s: ioread32(0x%p) = 0x%08x (id).\n", + engine->name, &engine->regs->identifier, w); + w &= BLOCK_ID_MASK; + if (w != BLOCK_ID_HEAD) { + pr_info("%s: engine id missing, 0x%08x exp. & 0x%x = 0x%x\n", + engine->name, w, BLOCK_ID_MASK, BLOCK_ID_HEAD); + return; + } + /* extra debugging; inspect complete engine set of registers */ + w = read_register(&engine->regs->status); + pr_info("%s: ioread32(0x%p) = 0x%08x (status).\n", + engine->name, &engine->regs->status, w); + w = read_register(&engine->regs->control); + pr_info("%s: ioread32(0x%p) = 0x%08x (control)\n", + engine->name, &engine->regs->control, w); + w = read_register(&engine->sgdma_regs->first_desc_lo); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_lo)\n", + engine->name, &engine->sgdma_regs->first_desc_lo, w); + w = read_register(&engine->sgdma_regs->first_desc_hi); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_hi)\n", + engine->name, &engine->sgdma_regs->first_desc_hi, w); + w = read_register(&engine->sgdma_regs->first_desc_adjacent); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_adjacent).\n", + engine->name, &engine->sgdma_regs->first_desc_adjacent, w); + w = read_register(&engine->regs->completed_desc_count); + pr_info("%s: ioread32(0x%p) = 0x%08x (completed_desc_count).\n", + engine->name, &engine->regs->completed_desc_count, w); + w = read_register(&engine->regs->interrupt_enable_mask); + pr_info("%s: ioread32(0x%p) = 0x%08x (interrupt_enable_mask)\n", + engine->name, &engine->regs->interrupt_enable_mask, w); +} + +/** + * engine_status_read() - read status of SG DMA engine (optionally reset) + * + * Stores status in engine->status. + * + * @return -1 on failure, status register otherwise + */ +static void engine_status_dump(struct xdma_engine *engine) +{ + u32 v = engine->status; + char buffer[256]; + char *buf = buffer; + int len = 0; + + len = sprintf(buf, "SG engine %s status: 0x%08x: ", engine->name, v); + + if ((v & XDMA_STAT_BUSY)) + len += sprintf(buf + len, "BUSY,"); + if ((v & XDMA_STAT_DESC_STOPPED)) + len += sprintf(buf + len, "DESC_STOPPED,"); + if ((v & XDMA_STAT_DESC_COMPLETED)) + len += sprintf(buf + len, "DESC_COMPL,"); + + /* common H2C & C2H */ + if ((v & XDMA_STAT_COMMON_ERR_MASK)) { + if ((v & XDMA_STAT_ALIGN_MISMATCH)) + len += sprintf(buf + len, "ALIGN_MISMATCH "); + if ((v & XDMA_STAT_MAGIC_STOPPED)) + len += sprintf(buf + len, "MAGIC_STOPPED "); + if ((v & XDMA_STAT_INVALID_LEN)) + len += sprintf(buf + len, "INVLIAD_LEN "); + if ((v & XDMA_STAT_IDLE_STOPPED)) + len += sprintf(buf + len, "IDLE_STOPPED "); + buf[len - 1] = ','; + } + + if ((engine->dir == DMA_TO_DEVICE)) { + /* H2C only */ + if ((v & XDMA_STAT_H2C_R_ERR_MASK)) { + len += sprintf(buf + len, "R:"); + if ((v & XDMA_STAT_H2C_R_UNSUPP_REQ)) + len += sprintf(buf + len, "UNSUPP_REQ "); + if ((v & XDMA_STAT_H2C_R_COMPL_ABORT)) + len += sprintf(buf + len, "COMPL_ABORT "); + if ((v & XDMA_STAT_H2C_R_PARITY_ERR)) + len += sprintf(buf + len, "PARITY "); + if ((v & XDMA_STAT_H2C_R_HEADER_EP)) + len += sprintf(buf + len, "HEADER_EP "); + if ((v & XDMA_STAT_H2C_R_UNEXP_COMPL)) + len += sprintf(buf + len, "UNEXP_COMPL "); + buf[len - 1] = ','; + } + + if ((v & XDMA_STAT_H2C_W_ERR_MASK)) { + len += sprintf(buf + len, "W:"); + if ((v & XDMA_STAT_H2C_W_DECODE_ERR)) + len += sprintf(buf + len, "DECODE_ERR "); + if ((v & XDMA_STAT_H2C_W_SLAVE_ERR)) + len += sprintf(buf + len, "SLAVE_ERR "); + buf[len - 1] = ','; + } + + } else { + /* C2H only */ + if ((v & XDMA_STAT_C2H_R_ERR_MASK)) { + len += sprintf(buf + len, "R:"); + if ((v & XDMA_STAT_C2H_R_DECODE_ERR)) + len += sprintf(buf + len, "DECODE_ERR "); + if ((v & XDMA_STAT_C2H_R_SLAVE_ERR)) + len += sprintf(buf + len, "SLAVE_ERR "); + buf[len - 1] = ','; + } + } + + /* common H2C & C2H */ + if ((v & XDMA_STAT_DESC_ERR_MASK)) { + len += sprintf(buf + len, "DESC_ERR:"); + if ((v & XDMA_STAT_DESC_UNSUPP_REQ)) + len += sprintf(buf + len, "UNSUPP_REQ "); + if ((v & XDMA_STAT_DESC_COMPL_ABORT)) + len += sprintf(buf + len, "COMPL_ABORT "); + if ((v & XDMA_STAT_DESC_PARITY_ERR)) + len += sprintf(buf + len, "PARITY "); + if ((v & XDMA_STAT_DESC_HEADER_EP)) + len += sprintf(buf + len, "HEADER_EP "); + if ((v & XDMA_STAT_DESC_UNEXP_COMPL)) + len += sprintf(buf + len, "UNEXP_COMPL "); + buf[len - 1] = ','; + } + + buf[len - 1] = '\0'; + pr_info("%s\n", buffer); +} + +static u32 engine_status_read(struct xdma_engine *engine, bool clear, bool dump) +{ + u32 value; + + BUG_ON(!engine); + + if (dump) + engine_reg_dump(engine); + + /* read status register */ + if (clear) + value = engine->status = + read_register(&engine->regs->status_rc); + else + value = engine->status = read_register(&engine->regs->status); + + if (dump) + engine_status_dump(engine); + + return value; +} + +/** + * xdma_engine_stop() - stop an SG DMA engine + * + */ +static void xdma_engine_stop(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + dbg_tfr("xdma_engine_stop(engine=%p)\n", engine); + + w = 0; + w |= (u32)XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= (u32)XDMA_CTRL_IE_MAGIC_STOPPED; + w |= (u32)XDMA_CTRL_IE_READ_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ERROR; + + if (poll_mode) { + w |= (u32) XDMA_CTRL_POLL_MODE_WB; + } else { + w |= (u32)XDMA_CTRL_IE_DESC_STOPPED; + w |= (u32)XDMA_CTRL_IE_DESC_COMPLETED; + + /* Disable IDLE STOPPED for MM */ + if ((engine->streaming && (engine->dir == DMA_FROM_DEVICE)) || + (engine->xdma_perf)) + w |= (u32)XDMA_CTRL_IE_IDLE_STOPPED; + } + + dbg_tfr("Stopping SG DMA %s engine; writing 0x%08x to 0x%p.\n", + engine->name, w, (u32 *)&engine->regs->control); + write_register(w, &engine->regs->control, + (unsigned long)(&engine->regs->control) - + (unsigned long)(&engine->regs)); + /* dummy read of status register to flush all previous writes */ + dbg_tfr("xdma_engine_stop(%s) done\n", engine->name); +} + +static void engine_start_mode_config(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + + /* If a perf test is running, enable the engine interrupts */ + if (engine->xdma_perf) { + w = XDMA_CTRL_IE_DESC_STOPPED; + w |= XDMA_CTRL_IE_DESC_COMPLETED; + w |= XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= XDMA_CTRL_IE_MAGIC_STOPPED; + w |= XDMA_CTRL_IE_IDLE_STOPPED; + w |= XDMA_CTRL_IE_READ_ERROR; + w |= XDMA_CTRL_IE_DESC_ERROR; + + write_register(w, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + } + + /* write control register of SG DMA engine */ + w = (u32)XDMA_CTRL_RUN_STOP; + w |= (u32)XDMA_CTRL_IE_READ_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= (u32)XDMA_CTRL_IE_MAGIC_STOPPED; + + if (poll_mode) { + w |= (u32)XDMA_CTRL_POLL_MODE_WB; + } else { + w |= (u32)XDMA_CTRL_IE_DESC_STOPPED; + w |= (u32)XDMA_CTRL_IE_DESC_COMPLETED; + + if ((engine->streaming && (engine->dir == DMA_FROM_DEVICE)) || + (engine->xdma_perf)) + w |= (u32)XDMA_CTRL_IE_IDLE_STOPPED; + + /* set non-incremental addressing mode */ + if (engine->non_incr_addr) + w |= (u32)XDMA_CTRL_NON_INCR_ADDR; + } + + dbg_tfr("iowrite32(0x%08x to 0x%p) (control)\n", w, + (void *)&engine->regs->control); + /* start the engine */ + write_register(w, &engine->regs->control, + (unsigned long)(&engine->regs->control) - + (unsigned long)(&engine->regs)); + + /* dummy read of status register to flush all previous writes */ + w = read_register(&engine->regs->status); + dbg_tfr("ioread32(0x%p) = 0x%08x (dummy read flushes writes).\n", + &engine->regs->status, w); +} + +/** + * engine_start() - start an idle engine with its first transfer on queue + * + * The engine will run and process all transfers that are queued using + * transfer_queue() and thus have their descriptor lists chained. + * + * During the run, new transfers will be processed if transfer_queue() has + * chained the descriptors before the hardware fetches the last descriptor. + * A transfer that was chained too late will invoke a new run of the engine + * initiated from the engine_service() routine. + * + * The engine must be idle and at least one transfer must be queued. + * This function does not take locks; the engine spinlock must already be + * taken. + * + */ +static struct xdma_transfer *engine_start(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer; + u32 w; + int extra_adj = 0; + + /* engine must be idle */ + BUG_ON(engine->running); + /* engine transfer queue must not be empty */ + BUG_ON(list_empty(&engine->transfer_list)); + /* inspect first transfer queued on the engine */ + transfer = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + BUG_ON(!transfer); + + /* engine is no longer shutdown */ + engine->shutdown = ENGINE_SHUTDOWN_NONE; + + dbg_tfr("engine_start(%s): transfer=0x%p.\n", engine->name, transfer); + + /* initialize number of descriptors of dequeued transfers */ + engine->desc_dequeued = 0; + + /* write lower 32-bit of bus address of transfer first descriptor */ + w = cpu_to_le32(PCI_DMA_L(transfer->desc_bus)); + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_lo)\n", w, + (void *)&engine->sgdma_regs->first_desc_lo); + write_register(w, &engine->sgdma_regs->first_desc_lo, + (unsigned long)(&engine->sgdma_regs->first_desc_lo) - + (unsigned long)(&engine->sgdma_regs)); + /* write upper 32-bit of bus address of transfer first descriptor */ + w = cpu_to_le32(PCI_DMA_H(transfer->desc_bus)); + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_hi)\n", w, + (void *)&engine->sgdma_regs->first_desc_hi); + write_register(w, &engine->sgdma_regs->first_desc_hi, + (unsigned long)(&engine->sgdma_regs->first_desc_hi) - + (unsigned long)(&engine->sgdma_regs)); + + if (transfer->desc_adjacent > 0) { + extra_adj = transfer->desc_adjacent - 1; + if (extra_adj > MAX_EXTRA_ADJ) + extra_adj = MAX_EXTRA_ADJ; + } + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_adjacent)\n", + extra_adj, (void *)&engine->sgdma_regs->first_desc_adjacent); + write_register(extra_adj, &engine->sgdma_regs->first_desc_adjacent, + (unsigned long)(&engine->sgdma_regs->first_desc_adjacent) - + (unsigned long)(&engine->sgdma_regs)); + + dbg_tfr("ioread32(0x%p) (dummy read flushes writes).\n", + &engine->regs->status); + mmiowb(); + + engine_start_mode_config(engine); + + engine_status_read(engine, 0, 0); + + dbg_tfr("%s engine 0x%p now running\n", engine->name, engine); + /* remember the engine is running */ + engine->running = 1; + return transfer; +} + +/** + * engine_service() - service an SG DMA engine + * + * must be called with engine->lock already acquired + * + * @engine pointer to struct xdma_engine + * + */ +static void engine_service_shutdown(struct xdma_engine *engine) +{ + /* if the engine stopped with RUN still asserted, de-assert RUN now */ + dbg_tfr("engine just went idle, resetting RUN_STOP.\n"); + xdma_engine_stop(engine); + engine->running = 0; + + /* awake task on engine's shutdown wait queue */ + wake_up_interruptible(&engine->shutdown_wq); +} + +struct xdma_transfer *engine_transfer_completion(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + BUG_ON(!engine); + + if (unlikely(!transfer)) { + pr_info("%s: xfer empty.\n", engine->name); + return NULL; + } + + /* synchronous I/O? */ + /* awake task on transfer's wait queue */ + wake_up_interruptible(&transfer->wq); + + return transfer; +} + +struct xdma_transfer *engine_service_transfer_list(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 *pdesc_completed) +{ + BUG_ON(!engine); + BUG_ON(!pdesc_completed); + + if (unlikely(!transfer)) { + pr_info("%s xfer empty, pdesc completed %u.\n", + engine->name, *pdesc_completed); + return NULL; + } + + /* + * iterate over all the transfers completed by the engine, + * except for the last (i.e. use > instead of >=). + */ + while (transfer && (!transfer->cyclic) && + (*pdesc_completed > transfer->desc_num)) { + /* remove this transfer from pdesc_completed */ + *pdesc_completed -= transfer->desc_num; + dbg_tfr("%s engine completed non-cyclic xfer 0x%p (%d desc)\n", + engine->name, transfer, transfer->desc_num); + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); + /* add to dequeued number of descriptors during this run */ + engine->desc_dequeued += transfer->desc_num; + /* mark transfer as succesfully completed */ + transfer->state = TRANSFER_STATE_COMPLETED; + + /* Complete transfer - sets transfer to NULL if an async + * transfer has completed */ + transfer = engine_transfer_completion(engine, transfer); + + /* if exists, get the next transfer on the list */ + if (!list_empty(&engine->transfer_list)) { + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + dbg_tfr("Non-completed transfer %p\n", transfer); + } else { + /* no further transfers? */ + transfer = NULL; + } + } + + return transfer; +} + +static void engine_err_handle(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 desc_completed) +{ + u32 value; + + /* + * The BUSY bit is expected to be clear now but older HW has a race + * condition which could cause it to be still set. If it's set, re-read + * and check again. If it's still set, log the issue. + */ + if (engine->status & XDMA_STAT_BUSY) { + value = read_register(&engine->regs->status); + if ((value & XDMA_STAT_BUSY) && printk_ratelimit()) + pr_info("%s has errors but is still BUSY\n", + engine->name); + } + + if (printk_ratelimit()) { + pr_info("%s, s 0x%x, aborted xfer 0x%p, cmpl %d/%d\n", + engine->name, engine->status, transfer, desc_completed, + transfer->desc_num); + } + + /* mark transfer as failed */ + transfer->state = TRANSFER_STATE_FAILED; + xdma_engine_stop(engine); +} + +struct xdma_transfer *engine_service_final_transfer(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 *pdesc_completed) +{ + BUG_ON(!engine); + BUG_ON(!pdesc_completed); + + /* inspect the current transfer */ + if (unlikely(!transfer)) { + pr_info("%s xfer empty, pdesc completed %u.\n", + engine->name, *pdesc_completed); + return NULL; + } else { + if (((engine->dir == DMA_FROM_DEVICE) && + (engine->status & XDMA_STAT_C2H_ERR_MASK)) || + ((engine->dir == DMA_TO_DEVICE) && + (engine->status & XDMA_STAT_H2C_ERR_MASK))) { + pr_info("engine %s, status error 0x%x.\n", + engine->name, engine->status); + engine_status_dump(engine); + engine_err_handle(engine, transfer, *pdesc_completed); + goto transfer_del; + } + + if (engine->status & XDMA_STAT_BUSY) + pr_debug("engine %s is unexpectedly busy - ignoring\n", + engine->name); + + /* the engine stopped on current transfer? */ + if (*pdesc_completed < transfer->desc_num) { + transfer->state = TRANSFER_STATE_FAILED; + pr_info("%s, xfer 0x%p, stopped half-way, %d/%d.\n", + engine->name, transfer, *pdesc_completed, + transfer->desc_num); + } else { + dbg_tfr("engine %s completed transfer\n", engine->name); + dbg_tfr("Completed transfer ID = 0x%p\n", transfer); + dbg_tfr("*pdesc_completed=%d, transfer->desc_num=%d", + *pdesc_completed, transfer->desc_num); + + if (!transfer->cyclic) { + /* + * if the engine stopped on this transfer, + * it should be the last + */ + WARN_ON(*pdesc_completed > transfer->desc_num); + } + /* mark transfer as succesfully completed */ + transfer->state = TRANSFER_STATE_COMPLETED; + } + +transfer_del: + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); + /* add to dequeued number of descriptors during this run */ + engine->desc_dequeued += transfer->desc_num; + + /* + * Complete transfer - sets transfer to NULL if an asynchronous + * transfer has completed + */ + transfer = engine_transfer_completion(engine, transfer); + } + + return transfer; +} + +static void engine_service_perf(struct xdma_engine *engine, u32 desc_completed) +{ + BUG_ON(!engine); + + /* performance measurement is running? */ + if (engine->xdma_perf) { + /* a descriptor was completed? */ + if (engine->status & XDMA_STAT_DESC_COMPLETED) { + engine->xdma_perf->iterations = desc_completed; + dbg_perf("transfer->xdma_perf->iterations=%d\n", + engine->xdma_perf->iterations); + } + + /* a descriptor stopped the engine? */ + if (engine->status & XDMA_STAT_DESC_STOPPED) { + engine->xdma_perf->stopped = 1; + /* + * wake any XDMA_PERF_IOCTL_STOP waiting for + * the performance run to finish + */ + wake_up_interruptible(&engine->xdma_perf_wq); + dbg_perf("transfer->xdma_perf stopped\n"); + } + } +} + +static void engine_transfer_dequeue(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer; + + BUG_ON(!engine); + + /* pick first transfer on the queue (was submitted to the engine) */ + transfer = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + if (!transfer || transfer != &engine->cyclic_req->xfer) { + pr_info("%s, xfer 0x%p != 0x%p.\n", + engine->name, transfer, &engine->cyclic_req->xfer); + return; + } + dbg_tfr("%s engine completed cyclic transfer 0x%p (%d desc).\n", + engine->name, transfer, transfer->desc_num); + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); +} + +static int engine_ring_process(struct xdma_engine *engine) +{ + struct xdma_result *result; + int start; + int eop_count = 0; + + BUG_ON(!engine); + result = engine->cyclic_result; + BUG_ON(!result); + + /* where we start receiving in the ring buffer */ + start = engine->rx_tail; + + /* iterate through all newly received RX result descriptors */ + dbg_tfr("%s, result %d, 0x%x, len 0x%x.\n", + engine->name, engine->rx_tail, result[engine->rx_tail].status, + result[engine->rx_tail].length); + while (result[engine->rx_tail].status && !engine->rx_overrun) { + /* EOP bit set in result? */ + if (result[engine->rx_tail].status & RX_STATUS_EOP){ + eop_count++; + } + + /* increment tail pointer */ + engine->rx_tail = (engine->rx_tail + 1) % CYCLIC_RX_PAGES_MAX; + + dbg_tfr("%s, head %d, tail %d, 0x%x, len 0x%x.\n", + engine->name, engine->rx_head, engine->rx_tail, + result[engine->rx_tail].status, + result[engine->rx_tail].length); + + /* overrun? */ + if (engine->rx_tail == engine->rx_head) { + dbg_tfr("%s: overrun\n", engine->name); + /* flag to user space that overrun has occurred */ + engine->rx_overrun = 1; + } + } + + return eop_count; +} + +static int engine_service_cyclic_polled(struct xdma_engine *engine) +{ + int eop_count = 0; + int rc = 0; + struct xdma_poll_wb *writeback_data; + u32 sched_limit = 0; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + writeback_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + while (eop_count == 0) { + if (sched_limit != 0) { + if ((sched_limit % NUM_POLLS_PER_SCHED) == 0) + schedule(); + } + sched_limit++; + + /* Monitor descriptor writeback address for errors */ + if ((writeback_data->completed_desc_count) & WB_ERR_MASK) { + rc = -1; + break; + } + + eop_count = engine_ring_process(engine); + } + + if (eop_count == 0) { + engine_status_read(engine, 1, 0); + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + } + + return rc; +} + +static int engine_service_cyclic_interrupt(struct xdma_engine *engine) +{ + int eop_count = 0; + struct xdma_transfer *xfer; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + engine_status_read(engine, 1, 0); + + eop_count = engine_ring_process(engine); + /* + * wake any reader on EOP, as one or more packets are now in + * the RX buffer + */ + xfer = &engine->cyclic_req->xfer; + if(enable_credit_mp){ + if (eop_count > 0) { + //engine->eop_found = 1; + } + wake_up_interruptible(&xfer->wq); + }else{ + if (eop_count > 0) { + /* awake task on transfer's wait queue */ + dbg_tfr("wake_up_interruptible() due to %d EOP's\n", eop_count); + engine->eop_found = 1; + wake_up_interruptible(&xfer->wq); + } + } + + /* engine was running but is no longer busy? */ + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + + return 0; +} + +/* must be called with engine->lock already acquired */ +static int engine_service_cyclic(struct xdma_engine *engine) +{ + int rc = 0; + + dbg_tfr("engine_service_cyclic()"); + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + if (poll_mode) + rc = engine_service_cyclic_polled(engine); + else + rc = engine_service_cyclic_interrupt(engine); + + return rc; +} + + +static void engine_service_resume(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer_started; + + BUG_ON(!engine); + + /* engine stopped? */ + if (!engine->running) { + /* in the case of shutdown, let it finish what's in the Q */ + if (!list_empty(&engine->transfer_list)) { + /* (re)start engine */ + transfer_started = engine_start(engine); + pr_info("re-started %s engine with pending xfer 0x%p\n", + engine->name, transfer_started); + /* engine was requested to be shutdown? */ + } else if (engine->shutdown & ENGINE_SHUTDOWN_REQUEST) { + engine->shutdown |= ENGINE_SHUTDOWN_IDLE; + /* awake task on engine's shutdown wait queue */ + wake_up_interruptible(&engine->shutdown_wq); + } else { + dbg_tfr("no pending transfers, %s engine stays idle.\n", + engine->name); + } + } else { + /* engine is still running? */ + if (list_empty(&engine->transfer_list)) { + pr_warn("no queued transfers but %s engine running!\n", + engine->name); + WARN_ON(1); + } + } +} + +/** + * engine_service() - service an SG DMA engine + * + * must be called with engine->lock already acquired + * + * @engine pointer to struct xdma_engine + * + */ +static int engine_service(struct xdma_engine *engine, int desc_writeback) +{ + struct xdma_transfer *transfer = NULL; + u32 desc_count = desc_writeback & WB_COUNT_MASK; + u32 err_flag = desc_writeback & WB_ERR_MASK; + int rv = 0; + struct xdma_poll_wb *wb_data; + + BUG_ON(!engine); + + /* If polling detected an error, signal to the caller */ + if (err_flag) + rv = -1; + + /* Service the engine */ + if (!engine->running) { + dbg_tfr("Engine was not running!!! Clearing status\n"); + engine_status_read(engine, 1, 0); + return 0; + } + + /* + * If called by the ISR or polling detected an error, read and clear + * engine status. For polled mode descriptor completion, this read is + * unnecessary and is skipped to reduce latency + */ + if ((desc_count == 0) || (err_flag != 0)) + engine_status_read(engine, 1, 0); + + /* + * engine was running but is no longer busy, or writeback occurred, + * shut down + */ + if ((engine->running && !(engine->status & XDMA_STAT_BUSY)) || + (desc_count != 0)) + engine_service_shutdown(engine); + + /* + * If called from the ISR, or if an error occurred, the descriptor + * count will be zero. In this scenario, read the descriptor count + * from HW. In polled mode descriptor completion, this read is + * unnecessary and is skipped to reduce latency + */ + if (!desc_count) + desc_count = read_register(&engine->regs->completed_desc_count); + dbg_tfr("desc_count = %d\n", desc_count); + + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) { + /* pick first transfer on queue (was submitted to the engine) */ + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + + dbg_tfr("head of queue transfer 0x%p has %d descriptors\n", + transfer, (int)transfer->desc_num); + + dbg_tfr("Engine completed %d desc, %d not yet dequeued\n", + (int)desc_count, + (int)desc_count - engine->desc_dequeued); + + engine_service_perf(engine, desc_count); + } + + /* account for already dequeued transfers during this engine run */ + desc_count -= engine->desc_dequeued; + + /* Process all but the last transfer */ + transfer = engine_service_transfer_list(engine, transfer, &desc_count); + + /* + * Process final transfer - includes checks of number of descriptors to + * detect faulty completion + */ + transfer = engine_service_final_transfer(engine, transfer, &desc_count); + + /* Before starting engine again, clear the writeback data */ + if (poll_mode) { + wb_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + wb_data->completed_desc_count = 0; + } + + /* Restart the engine following the servicing */ + engine_service_resume(engine); + + return 0; +} + +/* engine_service_work */ +static void engine_service_work(struct work_struct *work) +{ + struct xdma_engine *engine; + unsigned long flags; + + engine = container_of(work, struct xdma_engine, work); + BUG_ON(engine->magic != MAGIC_ENGINE); + + /* lock the engine */ + spin_lock_irqsave(&engine->lock, flags); + + dbg_tfr("engine_service() for %s engine %p\n", + engine->name, engine); + if (engine->cyclic_req) + engine_service_cyclic(engine); + else + engine_service(engine, 0); + + /* re-enable interrupts for this engine */ + if (engine->xdev->msix_enabled){ + write_register(engine->interrupt_enable_mask_value, + &engine->regs->interrupt_enable_mask_w1s, + (unsigned long)(&engine->regs->interrupt_enable_mask_w1s) - + (unsigned long)(&engine->regs)); + } else + channel_interrupts_enable(engine->xdev, engine->irq_bitmask); + + /* unlock the engine */ + spin_unlock_irqrestore(&engine->lock, flags); +} + +static u32 engine_service_wb_monitor(struct xdma_engine *engine, + u32 expected_wb) +{ + struct xdma_poll_wb *wb_data; + u32 desc_wb = 0; + u32 sched_limit = 0; + unsigned long timeout; + + BUG_ON(!engine); + wb_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + /* + * Poll the writeback location for the expected number of + * descriptors / error events This loop is skipped for cyclic mode, + * where the expected_desc_count passed in is zero, since it cannot be + * determined before the function is called + */ + + timeout = jiffies + (POLL_TIMEOUT_SECONDS * HZ); + while (expected_wb != 0) { + desc_wb = wb_data->completed_desc_count; + + if (desc_wb & WB_ERR_MASK) + break; + else if (desc_wb == expected_wb) + break; + + /* RTO - prevent system from hanging in polled mode */ + if (time_after(jiffies, timeout)) { + dbg_tfr("Polling timeout occurred"); + dbg_tfr("desc_wb = 0x%08x, expected 0x%08x\n", desc_wb, + expected_wb); + if ((desc_wb & WB_COUNT_MASK) > expected_wb) + desc_wb = expected_wb | WB_ERR_MASK; + + break; + } + + /* + * Define NUM_POLLS_PER_SCHED to limit how much time is spent + * in the scheduler + */ + + if (sched_limit != 0) { + if ((sched_limit % NUM_POLLS_PER_SCHED) == 0) + schedule(); + } + sched_limit++; + } + + return desc_wb; +} + +static int engine_service_poll(struct xdma_engine *engine, + u32 expected_desc_count) +{ + struct xdma_poll_wb *writeback_data; + u32 desc_wb = 0; + unsigned long flags; + int rv = 0; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + writeback_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + if ((expected_desc_count & WB_COUNT_MASK) != expected_desc_count) { + dbg_tfr("Queued descriptor count is larger than supported\n"); + return -1; + } + + /* + * Poll the writeback location for the expected number of + * descriptors / error events This loop is skipped for cyclic mode, + * where the expected_desc_count passed in is zero, since it cannot be + * determined before the function is called + */ + + desc_wb = engine_service_wb_monitor(engine, expected_desc_count); + + spin_lock_irqsave(&engine->lock, flags); + dbg_tfr("%s service.\n", engine->name); + if (engine->cyclic_req) { + rv = engine_service_cyclic(engine); + } else { + rv = engine_service(engine, desc_wb); + } + spin_unlock_irqrestore(&engine->lock, flags); + + return rv; +} + +static irqreturn_t user_irq_service(int irq, struct xdma_user_irq *user_irq) +{ + unsigned long flags; + + BUG_ON(!user_irq); + + if (user_irq->handler) + return user_irq->handler(user_irq->user_idx, user_irq->dev); + + spin_lock_irqsave(&(user_irq->events_lock), flags); + if (!user_irq->events_irq) { + user_irq->events_irq = 1; + wake_up_interruptible(&(user_irq->events_wq)); + } + spin_unlock_irqrestore(&(user_irq->events_lock), flags); + + return IRQ_HANDLED; +} + +/* + * xdma_isr() - Interrupt handler + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_isr(int irq, void *dev_id) +{ + u32 ch_irq; + u32 user_irq; + u32 mask; + struct xdma_dev *xdev; + struct interrupt_regs *irq_regs; + + dbg_irq("(irq=%d, dev 0x%p) <<<< ISR.\n", irq, dev_id); + BUG_ON(!dev_id); + xdev = (struct xdma_dev *)dev_id; + + if (!xdev) { + WARN_ON(!xdev); + dbg_irq("xdma_isr(irq=%d) xdev=%p ??\n", irq, xdev); + return IRQ_NONE; + } + + irq_regs = (struct interrupt_regs *)(xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + + /* read channel interrupt requests */ + ch_irq = read_register(&irq_regs->channel_int_request); + dbg_irq("ch_irq = 0x%08x\n", ch_irq); + + /* + * disable all interrupts that fired; these are re-enabled individually + * after the causing module has been fully serviced. + */ + if (ch_irq) + channel_interrupts_disable(xdev, ch_irq); + + /* read user interrupts - this read also flushes the above write */ + user_irq = read_register(&irq_regs->user_int_request); + dbg_irq("user_irq = 0x%08x\n", user_irq); + + if (user_irq) { + int user = 0; + u32 mask = 1; + int max = xdev->h2c_channel_max; + + for (; user < max && user_irq; user++, mask <<= 1) { + if (user_irq & mask) { + user_irq &= ~mask; + user_irq_service(irq, &xdev->user_irq[user]); + } + } + } + + mask = ch_irq & xdev->mask_irq_h2c; + if (mask) { + int channel = 0; + int max = xdev->h2c_channel_max; + + /* iterate over H2C (PCIe read) */ + for (channel = 0; channel < max && mask; channel++) { + struct xdma_engine *engine = &xdev->engine_h2c[channel]; + + /* engine present and its interrupt fired? */ + if((engine->irq_bitmask & mask) && + (engine->magic == MAGIC_ENGINE)) { + mask &= ~engine->irq_bitmask; + dbg_tfr("schedule_work, %s.\n", engine->name); + schedule_work(&engine->work); + } + } + } + + mask = ch_irq & xdev->mask_irq_c2h; + if (mask) { + int channel = 0; + int max = xdev->c2h_channel_max; + + /* iterate over C2H (PCIe write) */ + for (channel = 0; channel < max && mask; channel++) { + struct xdma_engine *engine = &xdev->engine_c2h[channel]; + + /* engine present and its interrupt fired? */ + if((engine->irq_bitmask & mask) && + (engine->magic == MAGIC_ENGINE)) { + mask &= ~engine->irq_bitmask; + dbg_tfr("schedule_work, %s.\n", engine->name); + schedule_work(&engine->work); + } + } + } + + xdev->irq_count++; + return IRQ_HANDLED; +} + +/* + * xdma_user_irq() - Interrupt handler for user interrupts in MSI-X mode + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_user_irq(int irq, void *dev_id) +{ + struct xdma_user_irq *user_irq; + + dbg_irq("(irq=%d) <<<< INTERRUPT SERVICE ROUTINE\n", irq); + + BUG_ON(!dev_id); + user_irq = (struct xdma_user_irq *)dev_id; + + return user_irq_service(irq, user_irq); +} + +/* + * xdma_channel_irq() - Interrupt handler for channel interrupts in MSI-X mode + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_channel_irq(int irq, void *dev_id) +{ + struct xdma_dev *xdev; + struct xdma_engine *engine; + struct interrupt_regs *irq_regs; + + dbg_irq("(irq=%d) <<<< INTERRUPT service ROUTINE\n", irq); + BUG_ON(!dev_id); + + engine = (struct xdma_engine *)dev_id; + xdev = engine->xdev; + + if (!xdev) { + WARN_ON(!xdev); + dbg_irq("xdma_channel_irq(irq=%d) xdev=%p ??\n", irq, xdev); + return IRQ_NONE; + } + + irq_regs = (struct interrupt_regs *)(xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + + /* Disable the interrupt for this engine */ + write_register(engine->interrupt_enable_mask_value, + &engine->regs->interrupt_enable_mask_w1c, + (unsigned long) + (&engine->regs->interrupt_enable_mask_w1c) - + (unsigned long)(&engine->regs)); + /* Dummy read to flush the above write */ + read_register(&irq_regs->channel_int_pending); + /* Schedule the bottom half */ + schedule_work(&engine->work); + + /* + * RTO - need to protect access here if multiple MSI-X are used for + * user interrupts + */ + xdev->irq_count++; + return IRQ_HANDLED; +} + +/* + * Unmap the BAR regions that had been mapped earlier using map_bars() + */ +static void unmap_bars(struct xdma_dev *xdev, struct pci_dev *dev) +{ + int i; + + for (i = 0; i < XDMA_BAR_NUM; i++) { + /* is this BAR mapped? */ + if (xdev->bar[i]) { + /* unmap BAR */ + pci_iounmap(dev, xdev->bar[i]); + /* mark as unmapped */ + xdev->bar[i] = NULL; + } + } +} + +static int map_single_bar(struct xdma_dev *xdev, struct pci_dev *dev, int idx) +{ + resource_size_t bar_start; + resource_size_t bar_len; + resource_size_t map_len; + + bar_start = pci_resource_start(dev, idx); + bar_len = pci_resource_len(dev, idx); + map_len = bar_len; + + xdev->bar[idx] = NULL; + + /* do not map BARs with length 0. Note that start MAY be 0! */ + if (!bar_len) { + //pr_info("BAR #%d is not present - skipping\n", idx); + return 0; + } + + /* BAR size exceeds maximum desired mapping? */ + if (bar_len > INT_MAX) { + pr_info("Limit BAR %d mapping from %llu to %d bytes\n", idx, + (u64)bar_len, INT_MAX); + map_len = (resource_size_t)INT_MAX; + } + /* + * map the full device memory or IO region into kernel virtual + * address space + */ + dbg_init("BAR%d: %llu bytes to be mapped.\n", idx, (u64)map_len); + xdev->bar[idx] = pci_iomap(dev, idx, map_len); + + if (!xdev->bar[idx]) { + pr_info("Could not map BAR %d.\n", idx); + return -1; + } + + pr_info("BAR%d at 0x%llx mapped at 0x%p, length=%llu(/%llu)\n", idx, + (u64)bar_start, xdev->bar[idx], (u64)map_len, (u64)bar_len); + + return (int)map_len; +} + +static int is_config_bar(struct xdma_dev *xdev, int idx) +{ + u32 irq_id = 0; + u32 cfg_id = 0; + int flag = 0; + u32 mask = 0xffff0000; /* Compare only XDMA ID's not Version number */ + struct interrupt_regs *irq_regs = + (struct interrupt_regs *) (xdev->bar[idx] + XDMA_OFS_INT_CTRL); + struct config_regs *cfg_regs = + (struct config_regs *)(xdev->bar[idx] + XDMA_OFS_CONFIG); + + irq_id = read_register(&irq_regs->identifier); + cfg_id = read_register(&cfg_regs->identifier); + + if (((irq_id & mask)== IRQ_BLOCK_ID) && + ((cfg_id & mask)== CONFIG_BLOCK_ID)) { + dbg_init("BAR %d is the XDMA config BAR\n", idx); + flag = 1; + } else { + dbg_init("BAR %d is NOT the XDMA config BAR: 0x%x, 0x%x.\n", + idx, irq_id, cfg_id); + flag = 0; + } + + return flag; +} + +static void identify_bars(struct xdma_dev *xdev, int *bar_id_list, int num_bars, + int config_bar_pos) +{ + /* + * The following logic identifies which BARs contain what functionality + * based on the position of the XDMA config BAR and the number of BARs + * detected. The rules are that the user logic and bypass logic BARs + * are optional. When both are present, the XDMA config BAR will be the + * 2nd BAR detected (config_bar_pos = 1), with the user logic being + * detected first and the bypass being detected last. When one is + * omitted, the type of BAR present can be identified by whether the + * XDMA config BAR is detected first or last. When both are omitted, + * only the XDMA config BAR is present. This somewhat convoluted + * approach is used instead of relying on BAR numbers in order to work + * correctly with both 32-bit and 64-bit BARs. + */ + + BUG_ON(!xdev); + BUG_ON(!bar_id_list); + + dbg_init("xdev 0x%p, bars %d, config at %d.\n", + xdev, num_bars, config_bar_pos); + + switch (num_bars) { + case 1: + /* Only one BAR present - no extra work necessary */ + break; + + case 2: + if (config_bar_pos == 0) { + xdev->bypass_bar_idx = bar_id_list[1]; + } else if (config_bar_pos == 1) { + xdev->user_bar_idx = bar_id_list[0]; + } else { + pr_info("2, XDMA config BAR unexpected %d.\n", + config_bar_pos); + } + break; + + case 3: + case 4: + if ((config_bar_pos == 1) || (config_bar_pos == 2)) { + /* user bar at bar #0 */ + xdev->user_bar_idx = bar_id_list[0]; + /* bypass bar at the last bar */ + xdev->bypass_bar_idx = bar_id_list[num_bars - 1]; + } else { + pr_info("3/4, XDMA config BAR unexpected %d.\n", + config_bar_pos); + } + break; + + default: + /* Should not occur - warn user but safe to continue */ + pr_info("Unexpected # BARs (%d), XDMA config BAR only.\n", + num_bars); + break; + + } + pr_info("%d BARs: config %d, user %d, bypass %d.\n", + num_bars, config_bar_pos, xdev->user_bar_idx, + xdev->bypass_bar_idx); +} + +/* map_bars() -- map device regions into kernel virtual address space + * + * Map the device memory regions into kernel virtual address space after + * verifying their sizes respect the minimum sizes needed + */ +static int map_bars(struct xdma_dev *xdev, struct pci_dev *dev) +{ + int rv; + int i; + int bar_id_list[XDMA_BAR_NUM]; + int bar_id_idx = 0; + int config_bar_pos = 0; + + /* iterate through all the BARs */ + for (i = 0; i < XDMA_BAR_NUM; i++) { + int bar_len; + + bar_len = map_single_bar(xdev, dev, i); + if (bar_len == 0) { + continue; + } else if (bar_len < 0) { + rv = -EINVAL; + goto fail; + } + + /* Try to identify BAR as XDMA control BAR */ + if ((bar_len >= XDMA_BAR_SIZE) && (xdev->config_bar_idx < 0)) { + + if (is_config_bar(xdev, i)) { + xdev->config_bar_idx = i; + config_bar_pos = bar_id_idx; + pr_info("config bar %d, pos %d.\n", + xdev->config_bar_idx, config_bar_pos); + } + } + + bar_id_list[bar_id_idx] = i; + bar_id_idx++; + } + + /* The XDMA config BAR must always be present */ + if (xdev->config_bar_idx < 0) { + pr_info("Failed to detect XDMA config BAR\n"); + rv = -EINVAL; + goto fail; + } + +#ifdef __LIBXDMA_CONFIG_BAR_ONLY__ + /* unmapped all other bars, except XDMA config. bar */ + for (i = 0; i < XDMA_BAR_NUM; i++) { + if (i == xdev->config_bar_idx) + continue; + + /* is this BAR mapped? */ + if (xdev->bar[i]) { + /* unmap BAR */ + pci_iounmap(dev, xdev->bar[i]); + /* mark as unmapped */ + xdev->bar[i] = NULL; + pr_info("unmap non-config bar %d.\n", i); + } + } +#else + identify_bars(xdev, bar_id_list, bar_id_idx, config_bar_pos); +#endif + + /* successfully mapped all required BAR regions */ + return 0; + +fail: + /* unwind; unmap any BARs that we did map */ + unmap_bars(xdev, dev); + return rv; +} + +/* + * MSI-X interrupt: + * <h2c+c2h channel_max> vectors, followed by <user_max> vectors + */ + +/* + * RTO - code to detect if MSI/MSI-X capability exists is derived + * from linux/pci/msi.c - pci_msi_check_device + */ + +#ifndef arch_msi_check_device +int arch_msi_check_device(struct pci_dev *dev, int nvec, int type) +{ + return 0; +} +#endif + +/* type = PCI_CAP_ID_MSI or PCI_CAP_ID_MSIX */ +static int msi_msix_capable(struct pci_dev *dev, int type) +{ + struct pci_bus *bus; + int ret; + + if (!dev || dev->no_msi) + return 0; + + for (bus = dev->bus; bus; bus = bus->parent) + if (bus->bus_flags & PCI_BUS_FLAGS_NO_MSI) + return 0; + + ret = arch_msi_check_device(dev, 1, type); + if (ret) + return 0; + + if (!pci_find_capability(dev, type)) + return 0; + + return 1; +} + +static void disable_msi_msix(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + if (xdev->msix_enabled) { + pci_disable_msix(pdev); + xdev->msix_enabled = 0; + } else if (xdev->msi_enabled) { + pci_disable_msi(pdev); + xdev->msi_enabled = 0; + } +} + +static int enable_msi_msix(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv = 0; + + BUG_ON(!xdev); + BUG_ON(!pdev); + + if (!interrupt_mode && msi_msix_capable(pdev, PCI_CAP_ID_MSIX)) { + int req_nvec = xdev->c2h_channel_max + xdev->h2c_channel_max + + xdev->user_max; + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + dbg_init("Enabling MSI-X\n"); + rv = pci_alloc_irq_vectors(pdev, req_nvec, req_nvec, + PCI_IRQ_MSIX); +#else + int i; + + dbg_init("Enabling MSI-X\n"); + for (i = 0; i < req_nvec; i++) + xdev->entry[i].entry = i; + + rv = pci_enable_msix(pdev, xdev->entry, req_nvec); +#endif + if (rv < 0) + dbg_init("Couldn't enable MSI-X mode: %d\n", rv); + + xdev->msix_enabled = 1; + + } else if (interrupt_mode == 1 && + msi_msix_capable(pdev, PCI_CAP_ID_MSI)) { + /* enable message signalled interrupts */ + dbg_init("pci_enable_msi()\n"); + rv = pci_enable_msi(pdev); + if (rv < 0) + dbg_init("Couldn't enable MSI mode: %d\n", rv); + xdev->msi_enabled = 1; + + } else { + dbg_init("MSI/MSI-X not detected - using legacy interrupts\n"); + } + + return rv; +} + +static void pci_check_intr_pend(struct pci_dev *pdev) +{ + u16 v; + + pci_read_config_word(pdev, PCI_STATUS, &v); + if (v & PCI_STATUS_INTERRUPT) { + pr_info("%s PCI STATUS Interrupt pending 0x%x.\n", + dev_name(&pdev->dev), v); + pci_write_config_word(pdev, PCI_STATUS, PCI_STATUS_INTERRUPT); + } +} + +static void pci_keep_intx_enabled(struct pci_dev *pdev) +{ + /* workaround to a h/w bug: + * when msix/msi become unavaile, default to legacy. + * However the legacy enable was not checked. + * If the legacy was disabled, no ack then everything stuck + */ + u16 pcmd, pcmd_new; + + pci_read_config_word(pdev, PCI_COMMAND, &pcmd); + pcmd_new = pcmd & ~PCI_COMMAND_INTX_DISABLE; + if (pcmd_new != pcmd) { + pr_info("%s: clear INTX_DISABLE, 0x%x -> 0x%x.\n", + dev_name(&pdev->dev), pcmd, pcmd_new); + pci_write_config_word(pdev, PCI_COMMAND, pcmd_new); + } +} + +static void prog_irq_msix_user(struct xdma_dev *xdev, bool clear) +{ + /* user */ + struct interrupt_regs *int_regs = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + u32 i = xdev->c2h_channel_max + xdev->h2c_channel_max; + u32 max = i + xdev->user_max; + int j; + + for (j = 0; i < max; j++) { + u32 val = 0; + int k; + int shift = 0; + + if (clear) + i += 4; + else + for (k = 0; k < 4 && i < max; i++, k++, shift += 8) + val |= (i & 0x1f) << shift; + + write_register(val, &int_regs->user_msi_vector[j], + XDMA_OFS_INT_CTRL + + ((unsigned long)&int_regs->user_msi_vector[j] - + (unsigned long)int_regs)); + + dbg_init("vector %d, 0x%x.\n", j, val); + } +} + +static void prog_irq_msix_channel(struct xdma_dev *xdev, bool clear) +{ + struct interrupt_regs *int_regs = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + u32 max = xdev->c2h_channel_max + xdev->h2c_channel_max; + u32 i; + int j; + + /* engine */ + for (i = 0, j = 0; i < max; j++) { + u32 val = 0; + int k; + int shift = 0; + + if (clear) + i += 4; + else + for (k = 0; k < 4 && i < max; i++, k++, shift += 8) + val |= (i & 0x1f) << shift; + + write_register(val, &int_regs->channel_msi_vector[j], + XDMA_OFS_INT_CTRL + + ((unsigned long)&int_regs->channel_msi_vector[j] - + (unsigned long)int_regs)); + dbg_init("vector %d, 0x%x.\n", j, val); + } +} + +static void irq_msix_channel_teardown(struct xdma_dev *xdev) +{ + struct xdma_engine *engine; + int j = 0; + int i = 0; + + if (!xdev->msix_enabled) + return; + + prog_irq_msix_channel(xdev, 1); + + engine = xdev->engine_h2c; + for (i = 0; i < xdev->h2c_channel_max; i++, j++, engine++) { + if (!engine->msix_irq_line) + break; + dbg_sg("Release IRQ#%d for engine %p\n", engine->msix_irq_line, + engine); + free_irq(engine->msix_irq_line, engine); + } + + engine = xdev->engine_c2h; + for (i = 0; i < xdev->c2h_channel_max; i++, j++, engine++) { + if (!engine->msix_irq_line) + break; + dbg_sg("Release IRQ#%d for engine %p\n", engine->msix_irq_line, + engine); + free_irq(engine->msix_irq_line, engine); + } +} + +static int irq_msix_channel_setup(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max; + int rv = 0; + u32 vector; + struct xdma_engine *engine; + + BUG_ON(!xdev); + if (!xdev->msix_enabled) + return 0; + + engine = xdev->engine_h2c; + for (i = 0; i < xdev->h2c_channel_max; i++, engine++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + vector = pci_irq_vector(xdev->pdev, i); +#else + vector = xdev->entry[i].vector; +#endif + rv = request_irq(vector, xdma_channel_irq, 0, xdev->mod_name, + engine); + if (rv) { + pr_info("requesti irq#%d failed %d, engine %s.\n", + vector, rv, engine->name); + return rv; + } + pr_info("engine %s, irq#%d.\n", engine->name, vector); + engine->msix_irq_line = vector; + } + + engine = xdev->engine_c2h; + for (i = 0; i < xdev->c2h_channel_max; i++, j++, engine++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + vector = pci_irq_vector(xdev->pdev, j); +#else + vector = xdev->entry[j].vector; +#endif + rv = request_irq(vector, xdma_channel_irq, 0, xdev->mod_name, + engine); + if (rv) { + pr_info("requesti irq#%d failed %d, engine %s.\n", + vector, rv, engine->name); + return rv; + } + pr_info("engine %s, irq#%d.\n", engine->name, vector); + engine->msix_irq_line = vector; + } + + return 0; +} + +static void irq_msix_user_teardown(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max + xdev->c2h_channel_max; + + BUG_ON(!xdev); + + if (!xdev->msix_enabled) + return; + + prog_irq_msix_user(xdev, 1); + + for (i = 0; i < xdev->user_max; i++, j++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + dbg_init("user %d, releasing IRQ#%d\n", i, vector); + free_irq(vector, &xdev->user_irq[i]); + } +} + +static int irq_msix_user_setup(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max + xdev->c2h_channel_max; + int rv = 0; + + /* vectors set in probe_scan_for_msi() */ + for (i = 0; i < xdev->user_max; i++, j++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + rv = request_irq(vector, xdma_user_irq, 0, xdev->mod_name, + &xdev->user_irq[i]); + if (rv) { + pr_info("user %d couldn't use IRQ#%d, %d\n", + i, vector, rv); + break; + } + pr_info("%d-USR-%d, IRQ#%d with 0x%p\n", xdev->idx, i, vector, + &xdev->user_irq[i]); + } + + /* If any errors occur, free IRQs that were successfully requested */ + if (rv) { + for (i--, j--; i >= 0; i--, j--) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + free_irq(vector, &xdev->user_irq[i]); + } + } + + return rv; +} + +static int irq_msi_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv; + + xdev->irq_line = (int)pdev->irq; + rv = request_irq(pdev->irq, xdma_isr, 0, xdev->mod_name, xdev); + if (rv) + dbg_init("Couldn't use IRQ#%d, %d\n", pdev->irq, rv); + else + dbg_init("Using IRQ#%d with 0x%p\n", pdev->irq, xdev); + + return rv; +} + +static int irq_legacy_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + u32 w; + u8 val; + void *reg; + int rv; + + pci_read_config_byte(pdev, PCI_INTERRUPT_PIN, &val); + dbg_init("Legacy Interrupt register value = %d\n", val); + if (val > 1) { + val--; + w = (val<<24) | (val<<16) | (val<<8)| val; + /* Program IRQ Block Channel vactor and IRQ Block User vector + * with Legacy interrupt value */ + reg = xdev->bar[xdev->config_bar_idx] + 0x2080; // IRQ user + write_register(w, reg, 0x2080); + write_register(w, reg+0x4, 0x2084); + write_register(w, reg+0x8, 0x2088); + write_register(w, reg+0xC, 0x208C); + reg = xdev->bar[xdev->config_bar_idx] + 0x20A0; // IRQ Block + write_register(w, reg, 0x20A0); + write_register(w, reg+0x4, 0x20A4); + } + + xdev->irq_line = (int)pdev->irq; + rv = request_irq(pdev->irq, xdma_isr, IRQF_SHARED, xdev->mod_name, + xdev); + if (rv) + dbg_init("Couldn't use IRQ#%d, %d\n", pdev->irq, rv); + else + dbg_init("Using IRQ#%d with 0x%p\n", pdev->irq, xdev); + + return rv; +} + +static void irq_teardown(struct xdma_dev *xdev) +{ + if (xdev->msix_enabled) { + irq_msix_channel_teardown(xdev); + irq_msix_user_teardown(xdev); + } else if (xdev->irq_line != -1) { + dbg_init("Releasing IRQ#%d\n", xdev->irq_line); + free_irq(xdev->irq_line, xdev); + } +} + +static int irq_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + pci_keep_intx_enabled(pdev); + + if (xdev->msix_enabled) { + int rv = irq_msix_channel_setup(xdev); + if (rv) + return rv; + rv = irq_msix_user_setup(xdev); + if (rv) + return rv; + prog_irq_msix_channel(xdev, 0); + prog_irq_msix_user(xdev, 0); + + return 0; + } else if (xdev->msi_enabled) + return irq_msi_setup(xdev, pdev); + + return irq_legacy_setup(xdev, pdev); +} + +#ifdef __LIBXDMA_DEBUG__ +static void dump_desc(struct xdma_desc *desc_virt) +{ + int j; + u32 *p = (u32 *)desc_virt; + static char * const field_name[] = { + "magic|extra_adjacent|control", "bytes", "src_addr_lo", + "src_addr_hi", "dst_addr_lo", "dst_addr_hi", "next_addr", + "next_addr_pad"}; + char *dummy; + + /* remove warning about unused variable when debug printing is off */ + dummy = field_name[0]; + + for (j = 0; j < 8; j += 1) { + pr_info("0x%08lx/0x%02lx: 0x%08x 0x%08x %s\n", + (uintptr_t)p, (uintptr_t)p & 15, (int)*p, + le32_to_cpu(*p), field_name[j]); + p++; + } + pr_info("\n"); +} + +static void transfer_dump(struct xdma_transfer *transfer) +{ + int i; + struct xdma_desc *desc_virt = transfer->desc_virt; + + pr_info("xfer 0x%p, state 0x%x, f 0x%x, dir %d, len %u, last %d.\n", + transfer, transfer->state, transfer->flags, transfer->dir, + transfer->len, transfer->last_in_request); + + pr_info("transfer 0x%p, desc %d, bus 0x%llx, adj %d.\n", + transfer, transfer->desc_num, (u64)transfer->desc_bus, + transfer->desc_adjacent); + for (i = 0; i < transfer->desc_num; i += 1) + dump_desc(desc_virt + i); +} +#endif /* __LIBXDMA_DEBUG__ */ + +/* xdma_desc_alloc() - Allocate cache-coherent array of N descriptors. + * + * Allocates an array of 'number' descriptors in contiguous PCI bus addressable + * memory. Chains the descriptors as a singly-linked list; the descriptor's + * next * pointer specifies the bus address of the next descriptor. + * + * + * @dev Pointer to pci_dev + * @number Number of descriptors to be allocated + * @desc_bus_p Pointer where to store the first descriptor bus address + * + * @return Virtual address of the first descriptor + * + */ +static void transfer_desc_init(struct xdma_transfer *transfer, int count) +{ + struct xdma_desc *desc_virt = transfer->desc_virt; + dma_addr_t desc_bus = transfer->desc_bus; + int i; + int adj = count - 1; + int extra_adj; + u32 temp_control; + + BUG_ON(count > XDMA_TRANSFER_MAX_DESC); + + /* create singly-linked list for SG DMA controller */ + for (i = 0; i < count - 1; i++) { + /* increment bus address to next in array */ + desc_bus += sizeof(struct xdma_desc); + + /* singly-linked list uses bus addresses */ + desc_virt[i].next_lo = cpu_to_le32(PCI_DMA_L(desc_bus)); + desc_virt[i].next_hi = cpu_to_le32(PCI_DMA_H(desc_bus)); + desc_virt[i].bytes = cpu_to_le32(0); + + /* any adjacent descriptors? */ + if (adj > 0) { + extra_adj = adj - 1; + if (extra_adj > MAX_EXTRA_ADJ) + extra_adj = MAX_EXTRA_ADJ; + + adj--; + } else { + extra_adj = 0; + } + + temp_control = DESC_MAGIC | (extra_adj << 8); + + desc_virt[i].control = cpu_to_le32(temp_control); + } + /* { i = number - 1 } */ + /* zero the last descriptor next pointer */ + desc_virt[i].next_lo = cpu_to_le32(0); + desc_virt[i].next_hi = cpu_to_le32(0); + desc_virt[i].bytes = cpu_to_le32(0); + + temp_control = DESC_MAGIC; + + desc_virt[i].control = cpu_to_le32(temp_control); +} + +/* xdma_desc_link() - Link two descriptors + * + * Link the first descriptor to a second descriptor, or terminate the first. + * + * @first first descriptor + * @second second descriptor, or NULL if first descriptor must be set as last. + * @second_bus bus address of second descriptor + */ +static void xdma_desc_link(struct xdma_desc *first, struct xdma_desc *second, + dma_addr_t second_bus) +{ + /* + * remember reserved control in first descriptor, but zero + * extra_adjacent! + */ + /* RTO - what's this about? Shouldn't it be 0x0000c0ffUL? */ + u32 control = le32_to_cpu(first->control) & 0x0000f0ffUL; + /* second descriptor given? */ + if (second) { + /* + * link last descriptor of 1st array to first descriptor of + * 2nd array + */ + first->next_lo = cpu_to_le32(PCI_DMA_L(second_bus)); + first->next_hi = cpu_to_le32(PCI_DMA_H(second_bus)); + WARN_ON(first->next_hi); + /* no second descriptor given */ + } else { + /* first descriptor is the last */ + first->next_lo = 0; + first->next_hi = 0; + } + /* merge magic, extra_adjacent and control field */ + control |= DESC_MAGIC; + + /* write bytes and next_num */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_adjacent -- Set how many descriptors are adjacent to this one */ +static void xdma_desc_adjacent(struct xdma_desc *desc, int next_adjacent) +{ + int extra_adj = 0; + /* remember reserved and control bits */ + u32 control = le32_to_cpu(desc->control) & 0x0000f0ffUL; + u32 max_adj_4k = 0; + + if (next_adjacent > 0) { + extra_adj = next_adjacent - 1; + if (extra_adj > MAX_EXTRA_ADJ){ + extra_adj = MAX_EXTRA_ADJ; + } + max_adj_4k = (0x1000 - ((le32_to_cpu(desc->next_lo))&0xFFF))/32 - 1; + if (extra_adj>max_adj_4k) { + extra_adj = max_adj_4k; + } + if(extra_adj<0){ + printk("Warning: extra_adj<0, converting it to 0\n"); + extra_adj = 0; + } + } + /* merge adjacent and control field */ + control |= 0xAD4B0000UL | (extra_adj << 8); + /* write control and next_adjacent */ + desc->control = cpu_to_le32(control); +} + +/* xdma_desc_control -- Set complete control field of a descriptor. */ +static void xdma_desc_control_set(struct xdma_desc *first, u32 control_field) +{ + /* remember magic and adjacent number */ + u32 control = le32_to_cpu(first->control) & ~(LS_BYTE_MASK); + + BUG_ON(control_field & ~(LS_BYTE_MASK)); + /* merge adjacent and control field */ + control |= control_field; + /* write control and next_adjacent */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_clear -- Clear bits in control field of a descriptor. */ +static void xdma_desc_control_clear(struct xdma_desc *first, u32 clear_mask) +{ + /* remember magic and adjacent number */ + u32 control = le32_to_cpu(first->control); + + BUG_ON(clear_mask & ~(LS_BYTE_MASK)); + + /* merge adjacent and control field */ + control &= (~clear_mask); + /* write control and next_adjacent */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_done - recycle cache-coherent linked list of descriptors. + * + * @dev Pointer to pci_dev + * @number Number of descriptors to be allocated + * @desc_virt Pointer to (i.e. virtual address of) first descriptor in list + * @desc_bus Bus address of first descriptor in list + */ +static inline void xdma_desc_done(struct xdma_desc *desc_virt) +{ + memset(desc_virt, 0, XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc)); +} + +/* xdma_desc() - Fill a descriptor with the transfer details + * + * @desc pointer to descriptor to be filled + * @addr root complex address + * @ep_addr end point address + * @len number of bytes, must be a (non-negative) multiple of 4. + * @dir, dma direction + * is the end point address. If zero, vice versa. + * + * Does not modify the next pointer + */ +static void xdma_desc_set(struct xdma_desc *desc, dma_addr_t rc_bus_addr, + u64 ep_addr, int len, int dir) +{ + /* transfer length */ + desc->bytes = cpu_to_le32(len); + if (dir == DMA_TO_DEVICE) { + /* read from root complex memory (source address) */ + desc->src_addr_lo = cpu_to_le32(PCI_DMA_L(rc_bus_addr)); + desc->src_addr_hi = cpu_to_le32(PCI_DMA_H(rc_bus_addr)); + /* write to end point address (destination address) */ + desc->dst_addr_lo = cpu_to_le32(PCI_DMA_L(ep_addr)); + desc->dst_addr_hi = cpu_to_le32(PCI_DMA_H(ep_addr)); + } else { + /* read from end point address (source address) */ + desc->src_addr_lo = cpu_to_le32(PCI_DMA_L(ep_addr)); + desc->src_addr_hi = cpu_to_le32(PCI_DMA_H(ep_addr)); + /* write to root complex memory (destination address) */ + desc->dst_addr_lo = cpu_to_le32(PCI_DMA_L(rc_bus_addr)); + desc->dst_addr_hi = cpu_to_le32(PCI_DMA_H(rc_bus_addr)); + } +} + +/* + * should hold the engine->lock; + */ +static void transfer_abort(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + struct xdma_transfer *head; + + BUG_ON(!engine); + BUG_ON(!transfer); + BUG_ON(transfer->desc_num == 0); + + pr_info("abort transfer 0x%p, desc %d, engine desc queued %d.\n", + transfer, transfer->desc_num, engine->desc_dequeued); + + head = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + if (head == transfer) + list_del(engine->transfer_list.next); + else + pr_info("engine %s, transfer 0x%p NOT found, 0x%p.\n", + engine->name, transfer, head); + + if (transfer->state == TRANSFER_STATE_SUBMITTED) + transfer->state = TRANSFER_STATE_ABORTED; +} + +/* transfer_queue() - Queue a DMA transfer on the engine + * + * @engine DMA engine doing the transfer + * @transfer DMA transfer submitted to the engine + * + * Takes and releases the engine spinlock + */ +static int transfer_queue(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + int rv = 0; + struct xdma_transfer *transfer_started; + struct xdma_dev *xdev; + unsigned long flags; + + BUG_ON(!engine); + BUG_ON(!engine->xdev); + BUG_ON(!transfer); + BUG_ON(transfer->desc_num == 0); + dbg_tfr("transfer_queue(transfer=0x%p).\n", transfer); + + xdev = engine->xdev; + if (xdma_device_flag_check(xdev, XDEV_FLAG_OFFLINE)) { + pr_info("dev 0x%p offline, transfer 0x%p not queued.\n", + xdev, transfer); + return -EBUSY; + } + + /* lock the engine state */ + spin_lock_irqsave(&engine->lock, flags); + + engine->prev_cpu = get_cpu(); + put_cpu(); + + /* engine is being shutdown; do not accept new transfers */ + if (engine->shutdown & ENGINE_SHUTDOWN_REQUEST) { + pr_info("engine %s offline, transfer 0x%p not queued.\n", + engine->name, transfer); + rv = -EBUSY; + goto shutdown; + } + + /* mark the transfer as submitted */ + transfer->state = TRANSFER_STATE_SUBMITTED; + /* add transfer to the tail of the engine transfer queue */ + list_add_tail(&transfer->entry, &engine->transfer_list); + + /* engine is idle? */ + if (!engine->running) { + /* start engine */ + dbg_tfr("transfer_queue(): starting %s engine.\n", + engine->name); + transfer_started = engine_start(engine); + dbg_tfr("transfer=0x%p started %s engine with transfer 0x%p.\n", + transfer, engine->name, transfer_started); + } else { + dbg_tfr("transfer=0x%p queued, with %s engine running.\n", + transfer, engine->name); + } + +shutdown: + /* unlock the engine state */ + dbg_tfr("engine->running = %d\n", engine->running); + spin_unlock_irqrestore(&engine->lock, flags); + return rv; +} + +static void engine_alignments(struct xdma_engine *engine) +{ + u32 w; + u32 align_bytes; + u32 granularity_bytes; + u32 address_bits; + + w = read_register(&engine->regs->alignments); + dbg_init("engine %p name %s alignments=0x%08x\n", engine, + engine->name, (int)w); + + /* RTO - add some macros to extract these fields */ + align_bytes = (w & 0x00ff0000U) >> 16; + granularity_bytes = (w & 0x0000ff00U) >> 8; + address_bits = (w & 0x000000ffU); + + dbg_init("align_bytes = %d\n", align_bytes); + dbg_init("granularity_bytes = %d\n", granularity_bytes); + dbg_init("address_bits = %d\n", address_bits); + + if (w) { + engine->addr_align = align_bytes; + engine->len_granularity = granularity_bytes; + engine->addr_bits = address_bits; + } else { + /* Some default values if alignments are unspecified */ + engine->addr_align = 1; + engine->len_granularity = 1; + engine->addr_bits = 64; + } +} + +static void engine_free_resource(struct xdma_engine *engine) +{ + struct xdma_dev *xdev = engine->xdev; + + /* Release memory use for descriptor writebacks */ + if (engine->poll_mode_addr_virt) { + dbg_sg("Releasing memory for descriptor writeback\n"); + dma_free_coherent(&xdev->pdev->dev, + sizeof(struct xdma_poll_wb), + engine->poll_mode_addr_virt, + engine->poll_mode_bus); + dbg_sg("Released memory for descriptor writeback\n"); + engine->poll_mode_addr_virt = NULL; + } + + if (engine->desc) { + dbg_init("device %s, engine %s pre-alloc desc 0x%p,0x%llx.\n", + dev_name(&xdev->pdev->dev), engine->name, + engine->desc, engine->desc_bus); + dma_free_coherent(&xdev->pdev->dev, + XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc), + engine->desc, engine->desc_bus); + engine->desc = NULL; + } + + if (engine->cyclic_result) { + dma_free_coherent(&xdev->pdev->dev, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result), + engine->cyclic_result, engine->cyclic_result_bus); + engine->cyclic_result = NULL; + } +} + +static void engine_destroy(struct xdma_dev *xdev, struct xdma_engine *engine) +{ + BUG_ON(!xdev); + BUG_ON(!engine); + + dbg_sg("Shutting down engine %s%d", engine->name, engine->channel); + + /* Disable interrupts to stop processing new events during shutdown */ + write_register(0x0, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + + if (enable_credit_mp && engine->streaming && + engine->dir == DMA_FROM_DEVICE) { + u32 reg_value = (0x1 << engine->channel) << 16; + struct sgdma_common_regs *reg = (struct sgdma_common_regs *) + (xdev->bar[xdev->config_bar_idx] + + (0x6*TARGET_SPACING)); + write_register(reg_value, ®->credit_mode_enable_w1c, 0); + } + + /* Release memory use for descriptor writebacks */ + engine_free_resource(engine); + + memset(engine, 0, sizeof(struct xdma_engine)); + /* Decrement the number of engines available */ + xdev->engines_num--; +} + +/** + *engine_cyclic_stop() - stop a cyclic transfer running on an SG DMA engine + * + *engine->lock must be taken + */ +struct xdma_transfer *engine_cyclic_stop(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer = 0; + + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) { + /* pick first transfer on the queue (was submitted to engine) */ + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + BUG_ON(!transfer); + + xdma_engine_stop(engine); + + if (transfer->cyclic) { + if (engine->xdma_perf) + dbg_perf("Stopping perf transfer on %s\n", + engine->name); + else + dbg_perf("Stopping cyclic transfer on %s\n", + engine->name); + /* make sure the handler sees correct transfer state */ + transfer->cyclic = 1; + /* + * set STOP flag and interrupt on completion, on the + * last descriptor + */ + xdma_desc_control_set( + transfer->desc_virt + transfer->desc_num - 1, + XDMA_DESC_COMPLETED | XDMA_DESC_STOPPED); + } else { + dbg_sg("(engine=%p) running transfer is not cyclic\n", + engine); + } + } else { + dbg_sg("(engine=%p) found not running transfer.\n", engine); + } + return transfer; +} +EXPORT_SYMBOL_GPL(engine_cyclic_stop); + +static int engine_writeback_setup(struct xdma_engine *engine) +{ + u32 w; + struct xdma_dev *xdev; + struct xdma_poll_wb *writeback; + + BUG_ON(!engine); + xdev = engine->xdev; + BUG_ON(!xdev); + + /* + * RTO - doing the allocation per engine is wasteful since a full page + * is allocated each time - better to allocate one page for the whole + * device during probe() and set per-engine offsets here + */ + writeback = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + writeback->completed_desc_count = 0; + + dbg_init("Setting writeback location to 0x%llx for engine %p", + engine->poll_mode_bus, engine); + w = cpu_to_le32(PCI_DMA_L(engine->poll_mode_bus)); + write_register(w, &engine->regs->poll_mode_wb_lo, + (unsigned long)(&engine->regs->poll_mode_wb_lo) - + (unsigned long)(&engine->regs)); + w = cpu_to_le32(PCI_DMA_H(engine->poll_mode_bus)); + write_register(w, &engine->regs->poll_mode_wb_hi, + (unsigned long)(&engine->regs->poll_mode_wb_hi) - + (unsigned long)(&engine->regs)); + + return 0; +} + + +/* engine_create() - Create an SG DMA engine bookkeeping data structure + * + * An SG DMA engine consists of the resources for a single-direction transfer + * queue; the SG DMA hardware, the software queue and interrupt handling. + * + * @dev Pointer to pci_dev + * @offset byte address offset in BAR[xdev->config_bar_idx] resource for the + * SG DMA * controller registers. + * @dir: DMA_TO/FROM_DEVICE + * @streaming Whether the engine is attached to AXI ST (rather than MM) + */ +static int engine_init_regs(struct xdma_engine *engine) +{ + u32 reg_value; + int rv = 0; + + write_register(XDMA_CTRL_NON_INCR_ADDR, &engine->regs->control_w1c, + (unsigned long)(&engine->regs->control_w1c) - + (unsigned long)(&engine->regs)); + + engine_alignments(engine); + + /* Configure error interrupts by default */ + reg_value = XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + reg_value |= XDMA_CTRL_IE_MAGIC_STOPPED; + reg_value |= XDMA_CTRL_IE_MAGIC_STOPPED; + reg_value |= XDMA_CTRL_IE_READ_ERROR; + reg_value |= XDMA_CTRL_IE_DESC_ERROR; + + /* if using polled mode, configure writeback address */ + if (poll_mode) { + rv = engine_writeback_setup(engine); + if (rv) { + dbg_init("%s descr writeback setup failed.\n", + engine->name); + goto fail_wb; + } + } else { + /* enable the relevant completion interrupts */ + reg_value |= XDMA_CTRL_IE_DESC_STOPPED; + reg_value |= XDMA_CTRL_IE_DESC_COMPLETED; + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) + reg_value |= XDMA_CTRL_IE_IDLE_STOPPED; + } + + /* Apply engine configurations */ + write_register(reg_value, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + + engine->interrupt_enable_mask_value = reg_value; + + /* only enable credit mode for AXI-ST C2H */ + if (enable_credit_mp && engine->streaming && + engine->dir == DMA_FROM_DEVICE) { + + struct xdma_dev *xdev = engine->xdev; + u32 reg_value = (0x1 << engine->channel) << 16; + struct sgdma_common_regs *reg = (struct sgdma_common_regs *) + (xdev->bar[xdev->config_bar_idx] + + (0x6*TARGET_SPACING)); + + write_register(reg_value, ®->credit_mode_enable_w1s, 0); + } + + return 0; + +fail_wb: + return rv; +} + +static int engine_alloc_resource(struct xdma_engine *engine) +{ + struct xdma_dev *xdev = engine->xdev; + + engine->desc = dma_alloc_coherent(&xdev->pdev->dev, + XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc), + &engine->desc_bus, GFP_KERNEL); + if (!engine->desc) { + pr_warn("dev %s, %s pre-alloc desc OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + + if (poll_mode) { + engine->poll_mode_addr_virt = dma_alloc_coherent( + &xdev->pdev->dev, + sizeof(struct xdma_poll_wb), + &engine->poll_mode_bus, GFP_KERNEL); + if (!engine->poll_mode_addr_virt) { + pr_warn("%s, %s poll pre-alloc writeback OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + } + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) { + engine->cyclic_result = dma_alloc_coherent(&xdev->pdev->dev, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result), + &engine->cyclic_result_bus, GFP_KERNEL); + + if (!engine->cyclic_result) { + pr_warn("%s, %s pre-alloc result OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + } + + return 0; + +err_out: + engine_free_resource(engine); + return -ENOMEM; +} + +static int engine_init(struct xdma_engine *engine, struct xdma_dev *xdev, + int offset, enum dma_data_direction dir, int channel) +{ + int rv; + u32 val; + + dbg_init("channel %d, offset 0x%x, dir %d.\n", channel, offset, dir); + + /* set magic */ + engine->magic = MAGIC_ENGINE; + + engine->channel = channel; + + /* engine interrupt request bit */ + engine->irq_bitmask = (1 << XDMA_ENG_IRQ_NUM) - 1; + engine->irq_bitmask <<= (xdev->engines_num * XDMA_ENG_IRQ_NUM); + engine->bypass_offset = xdev->engines_num * BYPASS_MODE_SPACING; + + /* parent */ + engine->xdev = xdev; + /* register address */ + engine->regs = (xdev->bar[xdev->config_bar_idx] + offset); + engine->sgdma_regs = xdev->bar[xdev->config_bar_idx] + offset + + SGDMA_OFFSET_FROM_CHANNEL; + val = read_register(&engine->regs->identifier); + if (val & 0x8000U) + engine->streaming = 1; + + /* remember SG DMA direction */ + engine->dir = dir; + sprintf(engine->name, "%d-%s%d-%s", xdev->idx, + (dir == DMA_TO_DEVICE) ? "H2C" : "C2H", channel, + engine->streaming ? "ST" : "MM"); + + dbg_init("engine %p name %s irq_bitmask=0x%08x\n", engine, engine->name, + (int)engine->irq_bitmask); + + /* initialize the deferred work for transfer completion */ + INIT_WORK(&engine->work, engine_service_work); + + if (dir == DMA_TO_DEVICE) + xdev->mask_irq_h2c |= engine->irq_bitmask; + else + xdev->mask_irq_c2h |= engine->irq_bitmask; + xdev->engines_num++; + + rv = engine_alloc_resource(engine); + if (rv) + return rv; + + rv = engine_init_regs(engine); + if (rv) + return rv; + + return 0; +} + +/* transfer_destroy() - free transfer */ +static void transfer_destroy(struct xdma_dev *xdev, struct xdma_transfer *xfer) +{ + /* free descriptors */ + xdma_desc_done(xfer->desc_virt); + + if (xfer->last_in_request && (xfer->flags & XFER_FLAG_NEED_UNMAP)) { + struct sg_table *sgt = xfer->sgt; + + if (sgt->nents) { + pci_unmap_sg(xdev->pdev, sgt->sgl, sgt->nents, + xfer->dir); + sgt->nents = 0; + } + } +} + +static int transfer_build(struct xdma_engine *engine, + struct xdma_request_cb *req, unsigned int desc_max) +{ + struct xdma_transfer *xfer = &req->xfer; + struct sw_desc *sdesc = &(req->sdesc[req->sw_desc_idx]); + int i = 0; + int j = 0; + + for (; i < desc_max; i++, j++, sdesc++) { + dbg_desc("sw desc %d/%u: 0x%llx, 0x%x, ep 0x%llx.\n", + i + req->sw_desc_idx, req->sw_desc_cnt, + sdesc->addr, sdesc->len, req->ep_addr); + + /* fill in descriptor entry j with transfer details */ + xdma_desc_set(xfer->desc_virt + j, sdesc->addr, req->ep_addr, + sdesc->len, xfer->dir); + xfer->len += sdesc->len; + + /* for non-inc-add mode don't increment ep_addr */ + if (!engine->non_incr_addr) + req->ep_addr += sdesc->len; + } + req->sw_desc_idx += desc_max; + return 0; +} + +static int transfer_init(struct xdma_engine *engine, struct xdma_request_cb *req) +{ + struct xdma_transfer *xfer = &req->xfer; + unsigned int desc_max = min_t(unsigned int, + req->sw_desc_cnt - req->sw_desc_idx, + XDMA_TRANSFER_MAX_DESC); + int i = 0; + int last = 0; + u32 control; + + memset(xfer, 0, sizeof(*xfer)); + + /* initialize wait queue */ + init_waitqueue_head(&xfer->wq); + + /* remember direction of transfer */ + xfer->dir = engine->dir; + + xfer->desc_virt = engine->desc; + xfer->desc_bus = engine->desc_bus; + + transfer_desc_init(xfer, desc_max); + + dbg_sg("transfer->desc_bus = 0x%llx.\n", (u64)xfer->desc_bus); + + transfer_build(engine, req, desc_max); + + /* terminate last descriptor */ + last = desc_max - 1; + xdma_desc_link(xfer->desc_virt + last, 0, 0); + /* stop engine, EOP for AXI ST, req IRQ on last descriptor */ + control = XDMA_DESC_STOPPED; + control |= XDMA_DESC_EOP; + control |= XDMA_DESC_COMPLETED; + xdma_desc_control_set(xfer->desc_virt + last, control); + + xfer->desc_num = xfer->desc_adjacent = desc_max; + + dbg_sg("transfer 0x%p has %d descriptors\n", xfer, xfer->desc_num); + /* fill in adjacent numbers */ + for (i = 0; i < xfer->desc_num; i++) + xdma_desc_adjacent(xfer->desc_virt + i, xfer->desc_num - i - 1); + + return 0; +} + +#ifdef __LIBXDMA_DEBUG__ +static void sgt_dump(struct sg_table *sgt) +{ + int i; + struct scatterlist *sg = sgt->sgl; + + pr_info("sgt 0x%p, sgl 0x%p, nents %u/%u.\n", + sgt, sgt->sgl, sgt->nents, sgt->orig_nents); + + for (i = 0; i < sgt->orig_nents; i++, sg = sg_next(sg)) + pr_info("%d, 0x%p, pg 0x%p,%u+%u, dma 0x%llx,%u.\n", + i, sg, sg_page(sg), sg->offset, sg->length, + sg_dma_address(sg), sg_dma_len(sg)); +} + +static void xdma_request_cb_dump(struct xdma_request_cb *req) +{ + int i; + + pr_info("request 0x%p, total %u, ep 0x%llx, sw_desc %u, sgt 0x%p.\n", + req, req->total_len, req->ep_addr, req->sw_desc_cnt, req->sgt); + sgt_dump(req->sgt); + for (i = 0; i < req->sw_desc_cnt; i++) + pr_info("%d/%u, 0x%llx, %u.\n", + i, req->sw_desc_cnt, req->sdesc[i].addr, + req->sdesc[i].len); +} +#endif + +static void xdma_request_free(struct xdma_request_cb *req) +{ + if (((unsigned long)req) >= VMALLOC_START && + ((unsigned long)req) < VMALLOC_END) + vfree(req); + else + kfree(req); +} + +static struct xdma_request_cb * xdma_request_alloc(unsigned int sdesc_nr) +{ + struct xdma_request_cb *req; + unsigned int size = sizeof(struct xdma_request_cb) + + sdesc_nr * sizeof(struct sw_desc); + + req = kzalloc(size, GFP_KERNEL); + if (!req) { + req = vmalloc(size); + if (req) + memset(req, 0, size); + } + if (!req) { + pr_info("OOM, %u sw_desc, %u.\n", sdesc_nr, size); + return NULL; + } + + return req; +} + +static struct xdma_request_cb * xdma_init_request(struct sg_table *sgt, + u64 ep_addr) +{ + struct xdma_request_cb *req; + struct scatterlist *sg = sgt->sgl; + int max = sgt->nents; + int extra = 0; + int i, j = 0; + + for (i = 0; i < max; i++, sg = sg_next(sg)) { + unsigned int len = sg_dma_len(sg); + + if (unlikely(len > desc_blen_max)) + extra += (len + desc_blen_max - 1) / desc_blen_max; + } + +//pr_info("ep 0x%llx, desc %u+%u.\n", ep_addr, max, extra); + + max += extra; + req = xdma_request_alloc(max); + if (!req) + return NULL; + + req->sgt = sgt; + req->ep_addr = ep_addr; + + for (i = 0, sg = sgt->sgl; i < sgt->nents; i++, sg = sg_next(sg)) { + unsigned int tlen = sg_dma_len(sg); + dma_addr_t addr = sg_dma_address(sg); + + req->total_len += tlen; + while (tlen) { + req->sdesc[j].addr = addr; + if (tlen > desc_blen_max) { + req->sdesc[j].len = desc_blen_max; + addr += desc_blen_max; + tlen -= desc_blen_max; + } else { + req->sdesc[j].len = tlen; + tlen = 0; + } + j++; + } + } + BUG_ON(j > max); + + req->sw_desc_cnt = j; +#ifdef __LIBXDMA_DEBUG__ + xdma_request_cb_dump(req); +#endif + return req; +} + +ssize_t xdma_xfer_submit(void *dev_hndl, int channel, bool write, u64 ep_addr, + struct sg_table *sgt, bool dma_mapped, int timeout_ms) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + int rv = 0; + ssize_t done = 0; + struct scatterlist *sg = sgt->sgl; + int nents; + enum dma_data_direction dir = write ? DMA_TO_DEVICE : DMA_FROM_DEVICE; + struct xdma_request_cb *req = NULL; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + if (write == 1) { + if (channel >= xdev->h2c_channel_max) { + pr_warn("H2C channel %d >= %d.\n", + channel, xdev->h2c_channel_max); + return -EINVAL; + } + engine = &xdev->engine_h2c[channel]; + } else if (write == 0) { + if (channel >= xdev->c2h_channel_max) { + pr_warn("C2H channel %d >= %d.\n", + channel, xdev->c2h_channel_max); + return -EINVAL; + } + engine = &xdev->engine_c2h[channel]; + } else { + pr_warn("write %d, exp. 0|1.\n", write); + return -EINVAL; + } + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + xdev = engine->xdev; + if (xdma_device_flag_check(xdev, XDEV_FLAG_OFFLINE)) { + pr_info("xdev 0x%p, offline.\n", xdev); + return -EBUSY; + } + + /* check the direction */ + if (engine->dir != dir) { + pr_info("0x%p, %s, %d, W %d, 0x%x/0x%x mismatch.\n", + engine, engine->name, channel, write, engine->dir, dir); + return -EINVAL; + } + + if (!dma_mapped) { + nents = pci_map_sg(xdev->pdev, sg, sgt->orig_nents, dir); + if (!nents) { + pr_info("map sgl failed, sgt 0x%p.\n", sgt); + return -EIO; + } + sgt->nents = nents; + } else { + BUG_ON(!sgt->nents); + } + + req = xdma_init_request(sgt, ep_addr); + if (!req) { + rv = -ENOMEM; + goto unmap_sgl; + } + + dbg_tfr("%s, len %u sg cnt %u.\n", + engine->name, req->total_len, req->sw_desc_cnt); + + sg = sgt->sgl; + nents = req->sw_desc_cnt; + while (nents) { + unsigned long flags; + struct xdma_transfer *xfer; + + /* one transfer at a time */ + spin_lock(&engine->desc_lock); + + /* build transfer */ + rv = transfer_init(engine, req); + if (rv < 0) { + spin_unlock(&engine->desc_lock); + goto unmap_sgl; + } + xfer = &req->xfer; + + if (!dma_mapped) + xfer->flags = XFER_FLAG_NEED_UNMAP; + + /* last transfer for the given request? */ + nents -= xfer->desc_num; + if (!nents) { + xfer->last_in_request = 1; + xfer->sgt = sgt; + } + + dbg_tfr("xfer, %u, ep 0x%llx, done %lu, sg %u/%u.\n", + xfer->len, req->ep_addr, done, req->sw_desc_idx, + req->sw_desc_cnt); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); +#endif + + rv = transfer_queue(engine, xfer); + if (rv < 0) { + spin_unlock(&engine->desc_lock); + pr_info("unable to submit %s, %d.\n", engine->name, rv); + goto unmap_sgl; + } + + /* + * When polling, determine how many descriptors have been queued * on the engine to determine the writeback value expected + */ + if (poll_mode) { + unsigned int desc_count; + + spin_lock_irqsave(&engine->lock, flags); + desc_count = xfer->desc_num; + spin_unlock_irqrestore(&engine->lock, flags); + + dbg_tfr("%s poll desc_count=%d\n", + engine->name, desc_count); + rv = engine_service_poll(engine, desc_count); + + } else { + rv = wait_event_interruptible_timeout(xfer->wq, + (xfer->state != TRANSFER_STATE_SUBMITTED), + msecs_to_jiffies(timeout_ms)); + } + + spin_lock_irqsave(&engine->lock, flags); + + switch(xfer->state) { + case TRANSFER_STATE_COMPLETED: + spin_unlock_irqrestore(&engine->lock, flags); + + dbg_tfr("transfer %p, %u, ep 0x%llx compl, +%lu.\n", + xfer, xfer->len, req->ep_addr - xfer->len, done); + done += xfer->len; + rv = 0; + break; + case TRANSFER_STATE_FAILED: + pr_info("xfer 0x%p,%u, failed, ep 0x%llx.\n", + xfer, xfer->len, req->ep_addr - xfer->len); + spin_unlock_irqrestore(&engine->lock, flags); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); + sgt_dump(sgt); +#endif + rv = -EIO; + break; + default: + /* transfer can still be in-flight */ + pr_info("xfer 0x%p,%u, s 0x%x timed out, ep 0x%llx.\n", + xfer, xfer->len, xfer->state, req->ep_addr); + engine_status_read(engine, 0, 1); + //engine_status_dump(engine); + transfer_abort(engine, xfer); + + xdma_engine_stop(engine); + spin_unlock_irqrestore(&engine->lock, flags); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); + sgt_dump(sgt); +#endif + rv = -ERESTARTSYS; + break; + } + + transfer_destroy(xdev, xfer); + spin_unlock(&engine->desc_lock); + + if (rv < 0) + goto unmap_sgl; + } /* while (sg) */ + +unmap_sgl: + if (!dma_mapped && sgt->nents) { + pci_unmap_sg(xdev->pdev, sgt->sgl, sgt->orig_nents, dir); + sgt->nents = 0; + } + + if (req) + xdma_request_free(req); + + if (rv < 0) + return rv; + + return done; +} +EXPORT_SYMBOL_GPL(xdma_xfer_submit); + +int xdma_performance_submit(struct xdma_dev *xdev, struct xdma_engine *engine) +{ + u8 *buffer_virt; + u32 max_consistent_size = 128 * 32 * 1024; /* 1024 pages, 4MB */ + dma_addr_t buffer_bus; /* bus address */ + struct xdma_transfer *transfer; + u64 ep_addr = 0; + int num_desc_in_a_loop = 128; + int size_in_desc = engine->xdma_perf->transfer_size; + int size = size_in_desc * num_desc_in_a_loop; + int i; + + BUG_ON(size_in_desc > max_consistent_size); + + if (size > max_consistent_size) { + size = max_consistent_size; + num_desc_in_a_loop = size / size_in_desc; + } + + buffer_virt = dma_alloc_coherent(&xdev->pdev->dev, size, + &buffer_bus, GFP_KERNEL); + + /* allocate transfer data structure */ + transfer = kzalloc(sizeof(struct xdma_transfer), GFP_KERNEL); + BUG_ON(!transfer); + + /* 0 = write engine (to_dev=0) , 1 = read engine (to_dev=1) */ + transfer->dir = engine->dir; + /* set number of descriptors */ + transfer->desc_num = num_desc_in_a_loop; + + /* allocate descriptor list */ + if (!engine->desc) { + engine->desc = dma_alloc_coherent(&xdev->pdev->dev, + num_desc_in_a_loop * sizeof(struct xdma_desc), + &engine->desc_bus, GFP_KERNEL); + BUG_ON(!engine->desc); + dbg_init("device %s, engine %s pre-alloc desc 0x%p,0x%llx.\n", + dev_name(&xdev->pdev->dev), engine->name, + engine->desc, engine->desc_bus); + } + transfer->desc_virt = engine->desc; + transfer->desc_bus = engine->desc_bus; + + transfer_desc_init(transfer, transfer->desc_num); + + dbg_sg("transfer->desc_bus = 0x%llx.\n", (u64)transfer->desc_bus); + + for (i = 0; i < transfer->desc_num; i++) { + struct xdma_desc *desc = transfer->desc_virt + i; + dma_addr_t rc_bus_addr = buffer_bus + size_in_desc * i; + + /* fill in descriptor entry with transfer details */ + xdma_desc_set(desc, rc_bus_addr, ep_addr, size_in_desc, + engine->dir); + } + + /* stop engine and request interrupt on last descriptor */ + xdma_desc_control_set(transfer->desc_virt, 0); + /* create a linked loop */ + xdma_desc_link(transfer->desc_virt + transfer->desc_num - 1, + transfer->desc_virt, transfer->desc_bus); + + transfer->cyclic = 1; + + /* initialize wait queue */ + init_waitqueue_head(&transfer->wq); + + //printk("=== Descriptor print for PERF \n"); + //transfer_dump(transfer); + + dbg_perf("Queueing XDMA I/O %s request for performance measurement.\n", + engine->dir ? "write (to dev)" : "read (from dev)"); + transfer_queue(engine, transfer); + return 0; + +} +EXPORT_SYMBOL_GPL(xdma_performance_submit); + +static struct xdma_dev *alloc_dev_instance(struct pci_dev *pdev) +{ + int i; + struct xdma_dev *xdev; + struct xdma_engine *engine; + + BUG_ON(!pdev); + + /* allocate zeroed device book keeping structure */ + xdev = kzalloc(sizeof(struct xdma_dev), GFP_KERNEL); + if (!xdev) { + pr_info("OOM, xdma_dev.\n"); + return NULL; + } + spin_lock_init(&xdev->lock); + + xdev->magic = MAGIC_DEVICE; + xdev->config_bar_idx = -1; + xdev->user_bar_idx = -1; + xdev->bypass_bar_idx = -1; + xdev->irq_line = -1; + + /* create a driver to device reference */ + xdev->pdev = pdev; + dbg_init("xdev = 0x%p\n", xdev); + + /* Set up data user IRQ data structures */ + for (i = 0; i < 16; i++) { + xdev->user_irq[i].xdev = xdev; + spin_lock_init(&xdev->user_irq[i].events_lock); + init_waitqueue_head(&xdev->user_irq[i].events_wq); + xdev->user_irq[i].handler = NULL; + xdev->user_irq[i].user_idx = i; /* 0 based */ + } + + engine = xdev->engine_h2c; + for (i = 0; i < XDMA_CHANNEL_NUM_MAX; i++, engine++) { + spin_lock_init(&engine->lock); + spin_lock_init(&engine->desc_lock); + INIT_LIST_HEAD(&engine->transfer_list); + init_waitqueue_head(&engine->shutdown_wq); + init_waitqueue_head(&engine->xdma_perf_wq); + } + + engine = xdev->engine_c2h; + for (i = 0; i < XDMA_CHANNEL_NUM_MAX; i++, engine++) { + spin_lock_init(&engine->lock); + spin_lock_init(&engine->desc_lock); + INIT_LIST_HEAD(&engine->transfer_list); + init_waitqueue_head(&engine->shutdown_wq); + init_waitqueue_head(&engine->xdma_perf_wq); + } + + return xdev; +} + +static int request_regions(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv; + + BUG_ON(!xdev); + BUG_ON(!pdev); + + dbg_init("pci_request_regions()\n"); + rv = pci_request_regions(pdev, xdev->mod_name); + /* could not request all regions? */ + if (rv) { + dbg_init("pci_request_regions() = %d, device in use?\n", rv); + /* assume device is in use so do not disable it later */ + xdev->regions_in_use = 1; + } else { + xdev->got_regions = 1; + } + + return rv; +} + +static int set_dma_mask(struct pci_dev *pdev) +{ + BUG_ON(!pdev); + + dbg_init("sizeof(dma_addr_t) == %ld\n", sizeof(dma_addr_t)); + /* 64-bit addressing capability for XDMA? */ + if (!pci_set_dma_mask(pdev, DMA_BIT_MASK(64))) { + /* query for DMA transfer */ + /* @see Documentation/DMA-mapping.txt */ + dbg_init("pci_set_dma_mask()\n"); + /* use 64-bit DMA */ + dbg_init("Using a 64-bit DMA mask.\n"); + /* use 32-bit DMA for descriptors */ + pci_set_consistent_dma_mask(pdev, DMA_BIT_MASK(32)); + /* use 64-bit DMA, 32-bit for consistent */ + } else if (!pci_set_dma_mask(pdev, DMA_BIT_MASK(32))) { + dbg_init("Could not set 64-bit DMA mask.\n"); + pci_set_consistent_dma_mask(pdev, DMA_BIT_MASK(32)); + /* use 32-bit DMA */ + dbg_init("Using a 32-bit DMA mask.\n"); + } else { + dbg_init("No suitable DMA possible.\n"); + return -EINVAL; + } + + return 0; +} + +static u32 get_engine_channel_id(struct engine_regs *regs) +{ + u32 value; + + BUG_ON(!regs); + + value = read_register(®s->identifier); + + return (value & 0x00000f00U) >> 8; +} + +static u32 get_engine_id(struct engine_regs *regs) +{ + u32 value; + + BUG_ON(!regs); + + value = read_register(®s->identifier); + return (value & 0xffff0000U) >> 16; +} + +static void remove_engines(struct xdma_dev *xdev) +{ + struct xdma_engine *engine; + int i; + + BUG_ON(!xdev); + + /* iterate over channels */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + if (engine->magic == MAGIC_ENGINE) { + dbg_sg("Remove %s, %d", engine->name, i); + engine_destroy(xdev, engine); + dbg_sg("%s, %d removed", engine->name, i); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + dbg_sg("Remove %s, %d", engine->name, i); + engine_destroy(xdev, engine); + dbg_sg("%s, %d removed", engine->name, i); + } + } +} + +static int probe_for_engine(struct xdma_dev *xdev, enum dma_data_direction dir, + int channel) +{ + struct engine_regs *regs; + int offset = channel * CHANNEL_SPACING; + u32 engine_id; + u32 engine_id_expected; + u32 channel_id; + struct xdma_engine *engine; + int rv; + + /* register offset for the engine */ + /* read channels at 0x0000, write channels at 0x1000, + * channels at 0x100 interval */ + if (dir == DMA_TO_DEVICE) { + engine_id_expected = XDMA_ID_H2C; + engine = &xdev->engine_h2c[channel]; + } else { + offset += H2C_CHANNEL_OFFSET; + engine_id_expected = XDMA_ID_C2H; + engine = &xdev->engine_c2h[channel]; + } + + regs = xdev->bar[xdev->config_bar_idx] + offset; + engine_id = get_engine_id(regs); + channel_id = get_engine_channel_id(regs); + + if ((engine_id != engine_id_expected) || (channel_id != channel)) { + dbg_init("%s %d engine, reg off 0x%x, id mismatch 0x%x,0x%x," + "exp 0x%x,0x%x, SKIP.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", + channel, offset, engine_id, channel_id, + engine_id_expected, channel_id != channel); + return -EINVAL; + } + + dbg_init("found AXI %s %d engine, reg. off 0x%x, id 0x%x,0x%x.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", channel, + offset, engine_id, channel_id); + + /* allocate and initialize engine */ + rv = engine_init(engine, xdev, offset, dir, channel); + if (rv != 0) { + pr_info("failed to create AXI %s %d engine.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", + channel); + return rv; + } + + return 0; +} + +static int probe_engines(struct xdma_dev *xdev) +{ + int i; + int rv = 0; + + BUG_ON(!xdev); + + /* iterate over channels */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + rv = probe_for_engine(xdev, DMA_TO_DEVICE, i); + if (rv) + break; + } + xdev->h2c_channel_max = i; + + for (i = 0; i < xdev->c2h_channel_max; i++) { + rv = probe_for_engine(xdev, DMA_FROM_DEVICE, i); + if (rv) + break; + } + xdev->c2h_channel_max = i; + + return 0; +} + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(3,5,0) +static void pci_enable_relaxed_ordering(struct pci_dev *pdev) +{ + pcie_capability_set_word(pdev, PCI_EXP_DEVCTL, PCI_EXP_DEVCTL_RELAX_EN); +} +#else +static void pci_enable_relaxed_ordering(struct pci_dev *pdev) +{ + u16 v; + int pos; + + pos = pci_pcie_cap(pdev); + if (pos > 0) { + pci_read_config_word(pdev, pos + PCI_EXP_DEVCTL, &v); + v |= PCI_EXP_DEVCTL_RELAX_EN; + pci_write_config_word(pdev, pos + PCI_EXP_DEVCTL, v); + } +} +#endif + +static void pci_check_extended_tag(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + u16 cap; + u32 v; + void *__iomem reg; + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(3,5,0) + pcie_capability_read_word(pdev, PCI_EXP_DEVCTL, &cap); +#else + int pos; + + pos = pci_pcie_cap(pdev); + if (pos > 0) + pci_read_config_word(pdev, pos + PCI_EXP_DEVCTL, &cap); + else { + pr_info("pdev 0x%p, unable to access pcie cap.\n", pdev); + return; + } +#endif + + if ((cap & PCI_EXP_DEVCTL_EXT_TAG)) + return; + + /* extended tag not enabled */ + pr_info("0x%p EXT_TAG disabled.\n", pdev); + + if (xdev->config_bar_idx < 0) { + pr_info("pdev 0x%p, xdev 0x%p, config bar UNKNOWN.\n", + pdev, xdev); + return; + } + + reg = xdev->bar[xdev->config_bar_idx] + XDMA_OFS_CONFIG + 0x4C; + v = read_register(reg); + v = (v & 0xFF) | (((u32)32) << 8); + write_register(v, reg, XDMA_OFS_CONFIG + 0x4C); +} + +void *xdma_device_open(const char *mname, struct pci_dev *pdev, int *user_max, + int *h2c_channel_max, int *c2h_channel_max) +{ + struct xdma_dev *xdev = NULL; + int rv = 0; + + pr_info("%s device %s, 0x%p.\n", mname, dev_name(&pdev->dev), pdev); + + /* allocate zeroed device book keeping structure */ + xdev = alloc_dev_instance(pdev); + if (!xdev) + return NULL; + xdev->mod_name = mname; + xdev->user_max = *user_max; + xdev->h2c_channel_max = *h2c_channel_max; + xdev->c2h_channel_max = *c2h_channel_max; + + xdma_device_flag_set(xdev, XDEV_FLAG_OFFLINE); + xdev_list_add(xdev); + + if (xdev->user_max == 0 || xdev->user_max > MAX_USER_IRQ) + xdev->user_max = MAX_USER_IRQ; + if (xdev->h2c_channel_max == 0 || + xdev->h2c_channel_max > XDMA_CHANNEL_NUM_MAX) + xdev->h2c_channel_max = XDMA_CHANNEL_NUM_MAX; + if (xdev->c2h_channel_max == 0 || + xdev->c2h_channel_max > XDMA_CHANNEL_NUM_MAX) + xdev->c2h_channel_max = XDMA_CHANNEL_NUM_MAX; + + rv = pci_enable_device(pdev); + if (rv) { + dbg_init("pci_enable_device() failed, %d.\n", rv); + goto err_enable; + } + + /* keep INTx enabled */ + pci_check_intr_pend(pdev); + + /* enable relaxed ordering */ + pci_enable_relaxed_ordering(pdev); + + pci_check_extended_tag(xdev, pdev); + + /* force MRRS to be 512 */ + rv = pcie_set_readrq(pdev, 512); + if (rv) + pr_info("device %s, error set PCI_EXP_DEVCTL_READRQ: %d.\n", + dev_name(&pdev->dev), rv); + + /* enable bus master capability */ + pci_set_master(pdev); + + rv = request_regions(xdev, pdev); + if (rv) + goto err_regions; + + rv = map_bars(xdev, pdev); + if (rv) + goto err_map; + + rv = set_dma_mask(pdev); + if (rv) + goto err_mask; + + check_nonzero_interrupt_status(xdev); + /* explicitely zero all interrupt enable masks */ + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + + rv = probe_engines(xdev); + if (rv) + goto err_engines; + + rv = enable_msi_msix(xdev, pdev); + if (rv < 0) + goto err_enable_msix; + + rv = irq_setup(xdev, pdev); + if (rv < 0) + goto err_interrupts; + + if (!poll_mode) + channel_interrupts_enable(xdev, ~0); + + /* Flush writes */ + read_interrupts(xdev); + + *user_max = xdev->user_max; + *h2c_channel_max = xdev->h2c_channel_max; + *c2h_channel_max = xdev->c2h_channel_max; + + xdma_device_flag_clear(xdev, XDEV_FLAG_OFFLINE); + return (void *)xdev; + +err_interrupts: + irq_teardown(xdev); +err_enable_msix: + disable_msi_msix(xdev, pdev); +err_engines: + remove_engines(xdev); +err_mask: + unmap_bars(xdev, pdev); +err_map: + if (xdev->got_regions) + pci_release_regions(pdev); +err_regions: + if (!xdev->regions_in_use) + pci_disable_device(pdev); +err_enable: + xdev_list_remove(xdev); + kfree(xdev); + return NULL; +} +EXPORT_SYMBOL_GPL(xdma_device_open); + +void xdma_device_close(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + dbg_init("pdev 0x%p, xdev 0x%p.\n", pdev, dev_hndl); + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + + dbg_sg("remove(dev = 0x%p) where pdev->dev.driver_data = 0x%p\n", + pdev, xdev); + if (xdev->pdev != pdev) { + dbg_sg("pci_dev(0x%lx) != pdev(0x%lx)\n", + (unsigned long)xdev->pdev, (unsigned long)pdev); + } + + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + + irq_teardown(xdev); + disable_msi_msix(xdev, pdev); + + remove_engines(xdev); + unmap_bars(xdev, pdev); + + if (xdev->got_regions) { + dbg_init("pci_release_regions 0x%p.\n", pdev); + pci_release_regions(pdev); + } + + if (!xdev->regions_in_use) { + dbg_init("pci_disable_device 0x%p.\n", pdev); + pci_disable_device(pdev); + } + + xdev_list_remove(xdev); + + kfree(xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_close); + +void xdma_device_offline(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + int i; + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + +pr_info("pdev 0x%p, xdev 0x%p.\n", pdev, xdev); + xdma_device_flag_set(xdev, XDEV_FLAG_OFFLINE); + + /* wait for all engines to be idle */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + unsigned long flags; + + engine = &xdev->engine_h2c[i]; + + if (engine->magic == MAGIC_ENGINE) { + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown |= ENGINE_SHUTDOWN_REQUEST; + + xdma_engine_stop(engine); + engine->running = 0; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + unsigned long flags; + + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown |= ENGINE_SHUTDOWN_REQUEST; + + xdma_engine_stop(engine); + engine->running = 0; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + /* turn off interrupts */ + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + irq_teardown(xdev); + + pr_info("xdev 0x%p, done.\n", xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_offline); + +void xdma_device_online(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + unsigned long flags; + int i; + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + +pr_info("pdev 0x%p, xdev 0x%p.\n", pdev, xdev); + + for (i = 0; i < xdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + if (engine->magic == MAGIC_ENGINE) { + engine_init_regs(engine); + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown &= ~ENGINE_SHUTDOWN_REQUEST; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + engine_init_regs(engine); + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown &= ~ENGINE_SHUTDOWN_REQUEST; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + /* re-write the interrupt table */ + if (!poll_mode) { + irq_setup(xdev, pdev); + + channel_interrupts_enable(xdev, ~0); + user_interrupts_enable(xdev, xdev->mask_irq_user); + read_interrupts(xdev); + } + + xdma_device_flag_clear(xdev, XDEV_FLAG_OFFLINE); +pr_info("xdev 0x%p, done.\n", xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_online); + +int xdma_device_restart(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return -EINVAL; + + pr_info("NOT implemented, 0x%p.\n", xdev); + return -EINVAL; +} +EXPORT_SYMBOL_GPL(xdma_device_restart); + +int xdma_user_isr_register(void *dev_hndl, unsigned int mask, + irq_handler_t handler, void *dev) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + int i; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + for (i = 0; i < xdev->user_max && mask; i++) { + unsigned int bit = (1 << i); + + if ((bit & mask) == 0) + continue; + + mask &= ~bit; + xdev->user_irq[i].handler = handler; + xdev->user_irq[i].dev = dev; + } + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_register); + +int xdma_user_isr_enable(void *dev_hndl, unsigned int mask) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + xdev->mask_irq_user |= mask; + /* enable user interrupts */ + user_interrupts_enable(xdev, mask); + read_interrupts(xdev); + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_enable); + +int xdma_user_isr_disable(void *dev_hndl, unsigned int mask) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + xdev->mask_irq_user &= ~mask; + user_interrupts_disable(xdev, mask); + read_interrupts(xdev); + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_disable); + +#ifdef __LIBXDMA_MOD__ +static int __init xdma_base_init(void) +{ + printk(KERN_INFO "%s", version); + return 0; +} + +static void __exit xdma_base_exit(void) +{ + return; +} + +module_init(xdma_base_init); +module_exit(xdma_base_exit); +#endif +/* makes an existing transfer cyclic */ +static void xdma_transfer_cyclic(struct xdma_transfer *transfer) +{ + /* link last descriptor to first descriptor */ + xdma_desc_link(transfer->desc_virt + transfer->desc_num - 1, + transfer->desc_virt, transfer->desc_bus); + /* remember transfer is cyclic */ + transfer->cyclic = 1; +} + +static int transfer_monitor_cyclic(struct xdma_engine *engine, + struct xdma_transfer *transfer, int timeout_ms) +{ + struct xdma_result *result; + int rc = 0; + + BUG_ON(!engine); + BUG_ON(!transfer); + + result = engine->cyclic_result; + BUG_ON(!result); + + if (poll_mode) { + int i ; + for (i = 0; i < 5; i++) { + rc = engine_service_poll(engine, 0); + if (rc) { + pr_info("%s service_poll failed %d.\n", + engine->name, rc); + rc = -ERESTARTSYS; + } + if (result[engine->rx_head].status) + return 0; + } + } else { + if (enable_credit_mp){ + dbg_tfr("%s: rx_head=%d,rx_tail=%d, wait ...\n", + engine->name, engine->rx_head, engine->rx_tail); + rc = wait_event_interruptible_timeout( transfer->wq, + (engine->rx_head!=engine->rx_tail || + engine->rx_overrun), + msecs_to_jiffies(timeout_ms)); + dbg_tfr("%s: wait returns %d, rx %d/%d, overrun %d.\n", + engine->name, rc, engine->rx_head, + engine->rx_tail, engine->rx_overrun); + } else { + rc = wait_event_interruptible_timeout( transfer->wq, + engine->eop_found, + msecs_to_jiffies(timeout_ms)); + dbg_tfr("%s: wait returns %d, eop_found %d.\n", + engine->name, rc, engine->eop_found); + } + } + + return 0; +} + +struct scatterlist *sglist_index(struct sg_table *sgt, unsigned int idx) +{ + struct scatterlist *sg = sgt->sgl; + int i; + + if (idx >= sgt->orig_nents) + return NULL; + + if (!idx) + return sg; + + for (i = 0; i < idx; i++, sg = sg_next(sg)) + ; + + return sg; +} + +static int copy_cyclic_to_user(struct xdma_engine *engine, int pkt_length, + int head, char __user *buf, size_t count) +{ + struct scatterlist *sg; + int more = pkt_length; + + BUG_ON(!engine); + BUG_ON(!buf); + + dbg_tfr("%s, pkt_len %d, head %d, user buf idx %u.\n", + engine->name, pkt_length, head, engine->user_buffer_index); + + sg = sglist_index(&engine->cyclic_sgt, head); + if (!sg) { + pr_info("%s, head %d OOR, sgl %u.\n", + engine->name, head, engine->cyclic_sgt.orig_nents); + return -EIO; + } + + /* EOP found? Transfer anything from head to EOP */ + while (more) { + unsigned int copy = more > PAGE_SIZE ? PAGE_SIZE : more; + unsigned int blen = count - engine->user_buffer_index; + int rv; + + if (copy > blen) + copy = blen; + + dbg_tfr("%s sg %d, 0x%p, copy %u to user %u.\n", + engine->name, head, sg, copy, + engine->user_buffer_index); + + rv = copy_to_user(&buf[engine->user_buffer_index], + page_address(sg_page(sg)), copy); + if (rv) { + pr_info("%s copy_to_user %u failed %d\n", + engine->name, copy, rv); + return -EIO; + } + + more -= copy; + engine->user_buffer_index += copy; + + if (engine->user_buffer_index == count) { + /* user buffer used up */ + break; + } + + head++; + if (head >= CYCLIC_RX_PAGES_MAX) { + head = 0; + sg = engine->cyclic_sgt.sgl; + } else + sg = sg_next(sg); + } + + return pkt_length; +} + +static int complete_cyclic(struct xdma_engine *engine, char __user *buf, + size_t count) +{ + struct xdma_result *result; + int pkt_length = 0; + int fault = 0; + int eop = 0; + int head; + int rc = 0; + int num_credit = 0; + unsigned long flags; + + BUG_ON(!engine); + result = engine->cyclic_result; + BUG_ON(!result); + + spin_lock_irqsave(&engine->lock, flags); + + /* where the host currently is in the ring buffer */ + head = engine->rx_head; + + /* iterate over newly received results */ + while (engine->rx_head != engine->rx_tail||engine->rx_overrun) { + + WARN_ON(result[engine->rx_head].status==0); + + dbg_tfr("%s, result[%d].status = 0x%x length = 0x%x.\n", + engine->name, engine->rx_head, + result[engine->rx_head].status, + result[engine->rx_head].length); + + if ((result[engine->rx_head].status >> 16) != C2H_WB) { + pr_info("%s, result[%d].status 0x%x, no magic.\n", + engine->name, engine->rx_head, + result[engine->rx_head].status); + fault = 1; + } else if (result[engine->rx_head].length > PAGE_SIZE) { + pr_info("%s, result[%d].len 0x%x, > PAGE_SIZE 0x%lx.\n", + engine->name, engine->rx_head, + result[engine->rx_head].length, PAGE_SIZE); + fault = 1; + } else if (result[engine->rx_head].length == 0) { + pr_info("%s, result[%d].length 0x%x.\n", + engine->name, engine->rx_head, + result[engine->rx_head].length); + fault = 1; + /* valid result */ + } else { + pkt_length += result[engine->rx_head].length; + num_credit++; + /* seen eop? */ + //if (result[engine->rx_head].status & RX_STATUS_EOP) + if (result[engine->rx_head].status & RX_STATUS_EOP){ + eop = 1; + engine->eop_found = 1; + } + + dbg_tfr("%s, pkt_length=%d (%s)\n", + engine->name, pkt_length, + eop ? "with EOP" : "no EOP yet"); + } + /* clear result */ + result[engine->rx_head].status = 0; + result[engine->rx_head].length = 0; + /* proceed head pointer so we make progress, even when fault */ + engine->rx_head = (engine->rx_head + 1) % CYCLIC_RX_PAGES_MAX; + + /* stop processing if a fault/eop was detected */ + if (fault || eop){ + break; + } + } + + spin_unlock_irqrestore(&engine->lock, flags); + + if (fault) + return -EIO; + + rc = copy_cyclic_to_user(engine, pkt_length, head, buf, count); + engine->rx_overrun = 0; + /* if copy is successful, release credits */ + if(rc > 0) + write_register(num_credit,&engine->sgdma_regs->credits, 0); + + return rc; +} + +ssize_t xdma_engine_read_cyclic(struct xdma_engine *engine, char __user *buf, + size_t count, int timeout_ms) +{ + int i = 0; + int rc = 0; + int rc_len = 0; + struct xdma_transfer *transfer; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + transfer = &engine->cyclic_req->xfer; + BUG_ON(!transfer); + + engine->user_buffer_index = 0; + + do { + rc = transfer_monitor_cyclic(engine, transfer, timeout_ms); + if (rc < 0) + return rc; + rc = complete_cyclic(engine, buf, count); + if (rc < 0) + return rc; + rc_len += rc; + + i++; + if (i > 10) + break; + } while (!engine->eop_found); + + if(enable_credit_mp) + engine->eop_found = 0; + + return rc_len; +} + +static void sgt_free_with_pages(struct sg_table *sgt, int dir, + struct pci_dev *pdev) +{ + struct scatterlist *sg = sgt->sgl; + int npages = sgt->orig_nents; + int i; + + for (i = 0; i < npages; i++, sg = sg_next(sg)) { + struct page *pg = sg_page(sg); + dma_addr_t bus = sg_dma_address(sg); + + if (pg) { + if (pdev) + pci_unmap_page(pdev, bus, PAGE_SIZE, dir); + __free_page(pg); + } else + break; + } + sg_free_table(sgt); + memset(sgt, 0, sizeof(struct sg_table)); +} + +static int sgt_alloc_with_pages(struct sg_table *sgt, unsigned int npages, + int dir, struct pci_dev *pdev) +{ + struct scatterlist *sg; + int i; + + if (sg_alloc_table(sgt, npages, GFP_KERNEL)) { + pr_info("sgt OOM.\n"); + return -ENOMEM; + } + + sg = sgt->sgl; + for (i = 0; i < npages; i++, sg = sg_next(sg)) { + struct page *pg = alloc_page(GFP_KERNEL); + + if (!pg) { + pr_info("%d/%u, page OOM.\n", i, npages); + goto err_out; + } + + if (pdev) { + dma_addr_t bus = pci_map_page(pdev, pg, 0, PAGE_SIZE, + dir); + if (unlikely(pci_dma_mapping_error(pdev, bus))) { + pr_info("%d/%u, page 0x%p map err.\n", + i, npages, pg); + __free_page(pg); + goto err_out; + } + sg_dma_address(sg) = bus; + sg_dma_len(sg) = PAGE_SIZE; + } + sg_set_page(sg, pg, PAGE_SIZE, 0); + } + + sgt->orig_nents = sgt->nents = npages; + + return 0; + +err_out: + sgt_free_with_pages(sgt, dir, pdev); + return -ENOMEM; +} + +/* + * !NOTE! reference/demo purpose only + * xdma_cyclic_transfer_setup is used for streaming C2H transfers: + * - A list of buffers are pre-allocated for incoming streaming data + * - the ring of the buffers is allowed to wrap around + */ +int xdma_cyclic_transfer_setup(struct xdma_engine *engine) +{ + struct xdma_dev *xdev; + struct xdma_transfer *xfer; + dma_addr_t bus; + unsigned long flags; + int i; + int rc; + + BUG_ON(!engine); + xdev = engine->xdev; + BUG_ON(!xdev); + + if (engine->cyclic_req) { + pr_info("%s: exclusive access already taken.\n", + engine->name); + return -EBUSY; + } + + spin_lock_irqsave(&engine->lock, flags); + + engine->rx_tail = 0; + engine->rx_head = 0; + engine->rx_overrun = 0; + engine->eop_found = 0; + + rc = sgt_alloc_with_pages(&engine->cyclic_sgt, CYCLIC_RX_PAGES_MAX, + engine->dir, xdev->pdev); + if (rc < 0) { + pr_info("%s cyclic pages %u OOM.\n", + engine->name, CYCLIC_RX_PAGES_MAX); + goto err_out; + } + + engine->cyclic_req = xdma_init_request(&engine->cyclic_sgt, 0); + if (!engine->cyclic_req) { + pr_info("%s cyclic request OOM.\n", engine->name); + rc = -ENOMEM; + goto err_out; + } + +#ifdef __LIBXDMA_DEBUG__ + xdma_request_cb_dump(engine->cyclic_req); +#endif + + rc = transfer_init(engine, engine->cyclic_req); + if (rc < 0) + goto err_out; + + xfer = &engine->cyclic_req->xfer; + + /* replace source addresses with result write-back addresses */ + memset(engine->cyclic_result, 0, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result)); + bus = engine->cyclic_result_bus; + for (i = 0; i < xfer->desc_num; i++) { + xfer->desc_virt[i].src_addr_lo = cpu_to_le32(PCI_DMA_L(bus)); + xfer->desc_virt[i].src_addr_hi = cpu_to_le32(PCI_DMA_H(bus)); + bus += sizeof(struct xdma_result); + } + /* set control of all descriptors */ + for (i = 0; i < xfer->desc_num; i++) { + xdma_desc_control_clear(xfer->desc_virt + i, LS_BYTE_MASK); + xdma_desc_control_set(xfer->desc_virt + i, + XDMA_DESC_EOP | XDMA_DESC_COMPLETED); + } + + /* make this a cyclic transfer */ + xdma_transfer_cyclic(xfer); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); +#endif + + if(enable_credit_mp){ + //write_register(RX_BUF_PAGES,&engine->sgdma_regs->credits); + write_register(128, &engine->sgdma_regs->credits, 0); + } + + spin_unlock_irqrestore(&engine->lock, flags); + + /* start cyclic transfer */ + transfer_queue(engine, xfer); + + return 0; + + /* unwind on errors */ +err_out: + if (engine->cyclic_req) { + xdma_request_free(engine->cyclic_req); + engine->cyclic_req = NULL; + } + + if (engine->cyclic_sgt.orig_nents) { + sgt_free_with_pages(&engine->cyclic_sgt, engine->dir, + xdev->pdev); + engine->cyclic_sgt.orig_nents = 0; + engine->cyclic_sgt.nents = 0; + engine->cyclic_sgt.sgl = NULL; + } + + spin_unlock_irqrestore(&engine->lock, flags); + + return rc; +} + + +static int cyclic_shutdown_polled(struct xdma_engine *engine) +{ + BUG_ON(!engine); + + spin_lock(&engine->lock); + + dbg_tfr("Polling for shutdown completion\n"); + do { + engine_status_read(engine, 1, 0); + schedule(); + } while (engine->status & XDMA_STAT_BUSY); + + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + dbg_tfr("Engine has stopped\n"); + + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + + dbg_tfr("Shutdown completion polling done\n"); + spin_unlock(&engine->lock); + + return 0; +} + +static int cyclic_shutdown_interrupt(struct xdma_engine *engine) +{ + int rc; + + BUG_ON(!engine); + + rc = wait_event_interruptible_timeout(engine->shutdown_wq, + !engine->running, msecs_to_jiffies(10000)); + +#if 0 + if (rc) { + dbg_tfr("wait_event_interruptible=%d\n", rc); + return rc; + } +#endif + + if (engine->running) { + pr_info("%s still running?!, %d\n", engine->name, rc); + return -EINVAL; + } + + return rc; +} + +int xdma_cyclic_transfer_teardown(struct xdma_engine *engine) +{ + int rc; + struct xdma_dev *xdev = engine->xdev; + struct xdma_transfer *transfer; + unsigned long flags; + + transfer = engine_cyclic_stop(engine); + + spin_lock_irqsave(&engine->lock, flags); + if (transfer) { + dbg_tfr("%s: stop transfer 0x%p.\n", engine->name, transfer); + if (transfer != &engine->cyclic_req->xfer) { + pr_info("%s unexpected transfer 0x%p/0x%p\n", + engine->name, transfer, + &engine->cyclic_req->xfer); + } + } + /* allow engine to be serviced after stop request */ + spin_unlock_irqrestore(&engine->lock, flags); + + /* wait for engine to be no longer running */ + if (poll_mode) + rc = cyclic_shutdown_polled(engine); + else + rc = cyclic_shutdown_interrupt(engine); + + /* obtain spin lock to atomically remove resources */ + spin_lock_irqsave(&engine->lock, flags); + + if (engine->cyclic_req) { + xdma_request_free(engine->cyclic_req); + engine->cyclic_req = NULL; + } + + if (engine->cyclic_sgt.orig_nents) { + sgt_free_with_pages(&engine->cyclic_sgt, engine->dir, + xdev->pdev); + engine->cyclic_sgt.orig_nents = 0; + engine->cyclic_sgt.nents = 0; + engine->cyclic_sgt.sgl = NULL; + } + + spin_unlock_irqrestore(&engine->lock, flags); + + return 0; +} + +int engine_addrmode_set(struct xdma_engine *engine, unsigned long arg) +{ + int rv; + unsigned long dst; + u32 w = XDMA_CTRL_NON_INCR_ADDR; + + dbg_perf("IOCTL_XDMA_ADDRMODE_SET\n"); + rv = get_user(dst, (int __user *)arg); + + if (rv == 0) { + engine->non_incr_addr = !!dst; + if (engine->non_incr_addr) + write_register(w, &engine->regs->control_w1s, + (unsigned long)(&engine->regs->control_w1s) - + (unsigned long)(&engine->regs)); + else + write_register(w, &engine->regs->control_w1c, + (unsigned long)(&engine->regs->control_w1c) - + (unsigned long)(&engine->regs)); + } + engine_alignments(engine); + + return rv; +} + diff --git a/sources/xdma_driver/libxdma/libxdma.h b/sources/xdma_driver/libxdma/libxdma.h new file mode 100644 index 0000000..bbf1a45 --- /dev/null +++ b/sources/xdma_driver/libxdma/libxdma.h @@ -0,0 +1,601 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef XDMA_LIB_H +#define XDMA_LIB_H + +#include <linux/version.h> +#include <linux/types.h> +#include <linux/uaccess.h> +#include <linux/module.h> +#include <linux/dma-mapping.h> +#include <linux/init.h> +#include <linux/interrupt.h> +#include <linux/jiffies.h> +#include <linux/kernel.h> +#include <linux/pci.h> +#include <linux/workqueue.h> + +/* Switch debug printing on/off */ +#define XDMA_DEBUG 0 + +/* SECTION: Preprocessor macros/constants */ +#define XDMA_BAR_NUM (6) + +/* maximum amount of register space to map */ +#define XDMA_BAR_SIZE (0x8000UL) + +/* Use this definition to poll several times between calls to schedule */ +#define NUM_POLLS_PER_SCHED 100 + +#define XDMA_CHANNEL_NUM_MAX (4) +/* + * interrupts per engine, rad2_vul.sv:237 + * .REG_IRQ_OUT (reg_irq_from_ch[(channel*2) +: 2]), + */ +#define XDMA_ENG_IRQ_NUM (1) +#define MAX_EXTRA_ADJ (15) +#define RX_STATUS_EOP (1) + +/* Target internal components on XDMA control BAR */ +#define XDMA_OFS_INT_CTRL (0x2000UL) +#define XDMA_OFS_CONFIG (0x3000UL) + +/* maximum number of desc per transfer request */ +#define XDMA_TRANSFER_MAX_DESC (2048) + +/* maximum size of a single DMA transfer descriptor */ +#define XDMA_DESC_BLEN_BITS 28 +#define XDMA_DESC_BLEN_MAX ((1 << (XDMA_DESC_BLEN_BITS)) - 1) + +/* bits of the SG DMA control register */ +#define XDMA_CTRL_RUN_STOP (1UL << 0) +#define XDMA_CTRL_IE_DESC_STOPPED (1UL << 1) +#define XDMA_CTRL_IE_DESC_COMPLETED (1UL << 2) +#define XDMA_CTRL_IE_DESC_ALIGN_MISMATCH (1UL << 3) +#define XDMA_CTRL_IE_MAGIC_STOPPED (1UL << 4) +#define XDMA_CTRL_IE_IDLE_STOPPED (1UL << 6) +#define XDMA_CTRL_IE_READ_ERROR (0x1FUL << 9) +#define XDMA_CTRL_IE_DESC_ERROR (0x1FUL << 19) +#define XDMA_CTRL_NON_INCR_ADDR (1UL << 25) +#define XDMA_CTRL_POLL_MODE_WB (1UL << 26) + +/* bits of the SG DMA status register */ +#define XDMA_STAT_BUSY (1UL << 0) +#define XDMA_STAT_DESC_STOPPED (1UL << 1) +#define XDMA_STAT_DESC_COMPLETED (1UL << 2) +#define XDMA_STAT_ALIGN_MISMATCH (1UL << 3) +#define XDMA_STAT_MAGIC_STOPPED (1UL << 4) +#define XDMA_STAT_INVALID_LEN (1UL << 5) +#define XDMA_STAT_IDLE_STOPPED (1UL << 6) + +#define XDMA_STAT_COMMON_ERR_MASK \ + (XDMA_STAT_ALIGN_MISMATCH | XDMA_STAT_MAGIC_STOPPED | \ + XDMA_STAT_INVALID_LEN) + +/* desc_error, C2H & H2C */ +#define XDMA_STAT_DESC_UNSUPP_REQ (1UL << 19) +#define XDMA_STAT_DESC_COMPL_ABORT (1UL << 20) +#define XDMA_STAT_DESC_PARITY_ERR (1UL << 21) +#define XDMA_STAT_DESC_HEADER_EP (1UL << 22) +#define XDMA_STAT_DESC_UNEXP_COMPL (1UL << 23) + +#define XDMA_STAT_DESC_ERR_MASK \ + (XDMA_STAT_DESC_UNSUPP_REQ | XDMA_STAT_DESC_COMPL_ABORT | \ + XDMA_STAT_DESC_PARITY_ERR | XDMA_STAT_DESC_HEADER_EP | \ + XDMA_STAT_DESC_UNEXP_COMPL) + +/* read error: H2C */ +#define XDMA_STAT_H2C_R_UNSUPP_REQ (1UL << 9) +#define XDMA_STAT_H2C_R_COMPL_ABORT (1UL << 10) +#define XDMA_STAT_H2C_R_PARITY_ERR (1UL << 11) +#define XDMA_STAT_H2C_R_HEADER_EP (1UL << 12) +#define XDMA_STAT_H2C_R_UNEXP_COMPL (1UL << 13) + +#define XDMA_STAT_H2C_R_ERR_MASK \ + (XDMA_STAT_H2C_R_UNSUPP_REQ | XDMA_STAT_H2C_R_COMPL_ABORT | \ + XDMA_STAT_H2C_R_PARITY_ERR | XDMA_STAT_H2C_R_HEADER_EP | \ + XDMA_STAT_H2C_R_UNEXP_COMPL) + +/* write error, H2C only */ +#define XDMA_STAT_H2C_W_DECODE_ERR (1UL << 14) +#define XDMA_STAT_H2C_W_SLAVE_ERR (1UL << 15) + +#define XDMA_STAT_H2C_W_ERR_MASK \ + (XDMA_STAT_H2C_W_DECODE_ERR | XDMA_STAT_H2C_W_SLAVE_ERR) + +/* read error: C2H */ +#define XDMA_STAT_C2H_R_DECODE_ERR (1UL << 9) +#define XDMA_STAT_C2H_R_SLAVE_ERR (1UL << 10) + +#define XDMA_STAT_C2H_R_ERR_MASK \ + (XDMA_STAT_C2H_R_DECODE_ERR | XDMA_STAT_C2H_R_SLAVE_ERR) + +/* all combined */ +#define XDMA_STAT_H2C_ERR_MASK \ + (XDMA_STAT_COMMON_ERR_MASK | XDMA_STAT_DESC_ERR_MASK | \ + XDMA_STAT_H2C_R_ERR_MASK | XDMA_STAT_H2C_W_ERR_MASK) + +#define XDMA_STAT_C2H_ERR_MASK \ + (XDMA_STAT_COMMON_ERR_MASK | XDMA_STAT_DESC_ERR_MASK | \ + XDMA_STAT_C2H_R_ERR_MASK) + +/* bits of the SGDMA descriptor control field */ +#define XDMA_DESC_STOPPED (1UL << 0) +#define XDMA_DESC_COMPLETED (1UL << 1) +#define XDMA_DESC_EOP (1UL << 4) + +#define XDMA_PERF_RUN (1UL << 0) +#define XDMA_PERF_CLEAR (1UL << 1) +#define XDMA_PERF_AUTO (1UL << 2) + +#define MAGIC_ENGINE 0xEEEEEEEEUL +#define MAGIC_DEVICE 0xDDDDDDDDUL + +/* upper 16-bits of engine identifier register */ +#define XDMA_ID_H2C 0x1fc0U +#define XDMA_ID_C2H 0x1fc1U + +/* for C2H AXI-ST mode */ +#define CYCLIC_RX_PAGES_MAX 256 + +#define LS_BYTE_MASK 0x000000FFUL + +#define BLOCK_ID_MASK 0xFFF00000 +#define BLOCK_ID_HEAD 0x1FC00000 + +#define IRQ_BLOCK_ID 0x1fc20000UL +#define CONFIG_BLOCK_ID 0x1fc30000UL + +#define WB_COUNT_MASK 0x00ffffffUL +#define WB_ERR_MASK (1UL << 31) +#define POLL_TIMEOUT_SECONDS 10 + +#define MAX_USER_IRQ 16 + +#define MAX_DESC_BUS_ADDR (0xffffffffULL) + +#define DESC_MAGIC 0xAD4B0000UL + +#define C2H_WB 0x52B4UL + +#define MAX_NUM_ENGINES (XDMA_CHANNEL_NUM_MAX * 2) +#define H2C_CHANNEL_OFFSET 0x1000 +#define SGDMA_OFFSET_FROM_CHANNEL 0x4000 +#define CHANNEL_SPACING 0x100 +#define TARGET_SPACING 0x1000 + +#define BYPASS_MODE_SPACING 0x0100 + +/* obtain the 32 most significant (high) bits of a 32-bit or 64-bit address */ +#define PCI_DMA_H(addr) ((addr >> 16) >> 16) +/* obtain the 32 least significant (low) bits of a 32-bit or 64-bit address */ +#define PCI_DMA_L(addr) (addr & 0xffffffffUL) + +#ifndef VM_RESERVED + #define VMEM_FLAGS (VM_IO | VM_DONTEXPAND | VM_DONTDUMP) +#else + #define VMEM_FLAGS (VM_IO | VM_RESERVED) +#endif + +#ifdef __LIBXDMA_DEBUG__ +#define dbg_io pr_err +#define dbg_fops pr_err +#define dbg_perf pr_err +#define dbg_sg pr_err +#define dbg_tfr pr_err +#define dbg_irq pr_err +#define dbg_init pr_err +#define dbg_desc pr_err +#else +/* disable debugging */ +#define dbg_io(...) +#define dbg_fops(...) +#define dbg_perf(...) +#define dbg_sg(...) +#define dbg_tfr(...) +#define dbg_irq(...) +#define dbg_init(...) +#define dbg_desc(...) +#endif + +/* SECTION: Enum definitions */ +enum transfer_state { + TRANSFER_STATE_NEW = 0, + TRANSFER_STATE_SUBMITTED, + TRANSFER_STATE_COMPLETED, + TRANSFER_STATE_FAILED, + TRANSFER_STATE_ABORTED +}; + +enum shutdown_state { + ENGINE_SHUTDOWN_NONE = 0, /* No shutdown in progress */ + ENGINE_SHUTDOWN_REQUEST = 1, /* engine requested to shutdown */ + ENGINE_SHUTDOWN_IDLE = 2 /* engine has shutdown and is idle */ +}; + +enum dev_capabilities { + CAP_64BIT_DMA = 2, + CAP_64BIT_DESC = 4, + CAP_ENGINE_WRITE = 8, + CAP_ENGINE_READ = 16 +}; + +/* SECTION: Structure definitions */ + +struct config_regs { + u32 identifier; + u32 reserved_1[4]; + u32 msi_enable; +}; + +/** + * SG DMA Controller status and control registers + * + * These registers make the control interface for DMA transfers. + * + * It sits in End Point (FPGA) memory BAR[0] for 32-bit or BAR[0:1] for 64-bit. + * It references the first descriptor which exists in Root Complex (PC) memory. + * + * @note The registers must be accessed using 32-bit (PCI DWORD) read/writes, + * and their values are in little-endian byte ordering. + */ +struct engine_regs { + u32 identifier; + u32 control; + u32 control_w1s; + u32 control_w1c; + u32 reserved_1[12]; /* padding */ + + u32 status; + u32 status_rc; + u32 completed_desc_count; + u32 alignments; + u32 reserved_2[14]; /* padding */ + + u32 poll_mode_wb_lo; + u32 poll_mode_wb_hi; + u32 interrupt_enable_mask; + u32 interrupt_enable_mask_w1s; + u32 interrupt_enable_mask_w1c; + u32 reserved_3[9]; /* padding */ + + u32 perf_ctrl; + u32 perf_cyc_lo; + u32 perf_cyc_hi; + u32 perf_dat_lo; + u32 perf_dat_hi; + u32 perf_pnd_lo; + u32 perf_pnd_hi; +} __packed; + +struct engine_sgdma_regs { + u32 identifier; + u32 reserved_1[31]; /* padding */ + + /* bus address to first descriptor in Root Complex Memory */ + u32 first_desc_lo; + u32 first_desc_hi; + /* number of adjacent descriptors at first_desc */ + u32 first_desc_adjacent; + u32 credits; +} __packed; + +struct msix_vec_table_entry { + u32 msi_vec_addr_lo; + u32 msi_vec_addr_hi; + u32 msi_vec_data_lo; + u32 msi_vec_data_hi; +} __packed; + +struct msix_vec_table { + struct msix_vec_table_entry entry_list[32]; +} __packed; + +struct interrupt_regs { + u32 identifier; + u32 user_int_enable; + u32 user_int_enable_w1s; + u32 user_int_enable_w1c; + u32 channel_int_enable; + u32 channel_int_enable_w1s; + u32 channel_int_enable_w1c; + u32 reserved_1[9]; /* padding */ + + u32 user_int_request; + u32 channel_int_request; + u32 user_int_pending; + u32 channel_int_pending; + u32 reserved_2[12]; /* padding */ + + u32 user_msi_vector[8]; + u32 channel_msi_vector[8]; +} __packed; + +struct sgdma_common_regs { + u32 padding[8]; + u32 credit_mode_enable; + u32 credit_mode_enable_w1s; + u32 credit_mode_enable_w1c; +} __packed; + + +/* Structure for polled mode descriptor writeback */ +struct xdma_poll_wb { + u32 completed_desc_count; + u32 reserved_1[7]; +} __packed; + + +/** + * Descriptor for a single contiguous memory block transfer. + * + * Multiple descriptors are linked by means of the next pointer. An additional + * extra adjacent number gives the amount of extra contiguous descriptors. + * + * The descriptors are in root complex memory, and the bytes in the 32-bit + * words must be in little-endian byte ordering. + */ +struct xdma_desc { + u32 control; + u32 bytes; /* transfer length in bytes */ + u32 src_addr_lo; /* source address (low 32-bit) */ + u32 src_addr_hi; /* source address (high 32-bit) */ + u32 dst_addr_lo; /* destination address (low 32-bit) */ + u32 dst_addr_hi; /* destination address (high 32-bit) */ + /* + * next descriptor in the single-linked list of descriptors; + * this is the PCIe (bus) address of the next descriptor in the + * root complex memory + */ + u32 next_lo; /* next desc address (low 32-bit) */ + u32 next_hi; /* next desc address (high 32-bit) */ +} __packed; + +/* 32 bytes (four 32-bit words) or 64 bytes (eight 32-bit words) */ +struct xdma_result { + u32 status; + u32 length; + u32 reserved_1[6]; /* padding */ +} __packed; + +struct sw_desc { + dma_addr_t addr; + unsigned int len; +}; + +/* Describes a (SG DMA) single transfer for the engine */ +struct xdma_transfer { + struct list_head entry; /* queue of non-completed transfers */ + struct xdma_desc *desc_virt; /* virt addr of the 1st descriptor */ + dma_addr_t desc_bus; /* bus addr of the first descriptor */ + int desc_adjacent; /* adjacent descriptors at desc_bus */ + int desc_num; /* number of descriptors in transfer */ + enum dma_data_direction dir; + wait_queue_head_t wq; /* wait queue for transfer completion */ + + enum transfer_state state; /* state of the transfer */ + unsigned int flags; +#define XFER_FLAG_NEED_UNMAP 0x1 + int cyclic; /* flag if transfer is cyclic */ + int last_in_request; /* flag if last within request */ + unsigned int len; + struct sg_table *sgt; +}; + +struct xdma_request_cb { + struct sg_table *sgt; + unsigned int total_len; + u64 ep_addr; + + struct xdma_transfer xfer; + + unsigned int sw_desc_idx; + unsigned int sw_desc_cnt; + struct sw_desc sdesc[0]; +}; + +struct xdma_engine { + unsigned long magic; /* structure ID for sanity checks */ + struct xdma_dev *xdev; /* parent device */ + char name[5]; /* name of this engine */ + int version; /* version of this engine */ + //dev_t cdevno; /* character device major:minor */ + //struct cdev cdev; /* character device (embedded struct) */ + + /* HW register address offsets */ + struct engine_regs *regs; /* Control reg BAR offset */ + struct engine_sgdma_regs *sgdma_regs; /* SGDAM reg BAR offset */ + u32 bypass_offset; /* Bypass mode BAR offset */ + + /* Engine state, configuration and flags */ + enum shutdown_state shutdown; /* engine shutdown mode */ + enum dma_data_direction dir; + int device_open; /* flag if engine node open, ST mode only */ + int running; /* flag if the driver started engine */ + int non_incr_addr; /* flag if non-incremental addressing used */ + int streaming; + int addr_align; /* source/dest alignment in bytes */ + int len_granularity; /* transfer length multiple */ + int addr_bits; /* HW datapath address width */ + int channel; /* engine indices */ + int max_extra_adj; /* descriptor prefetch capability */ + int desc_dequeued; /* num descriptors of completed transfers */ + u32 status; /* last known status of device */ + u32 interrupt_enable_mask_value;/* only used for MSIX mode to store per-engine interrupt mask value */ + + /* Transfer list management */ + struct list_head transfer_list; /* queue of transfers */ + + /* Members applicable to AXI-ST C2H (cyclic) transfers */ + struct xdma_result *cyclic_result; + dma_addr_t cyclic_result_bus; /* bus addr for transfer */ + struct xdma_request_cb *cyclic_req; + struct sg_table cyclic_sgt; + u8 eop_found; /* used only for cyclic(rx:c2h) */ + + int rx_tail; /* follows the HW */ + int rx_head; /* where the SW reads from */ + int rx_overrun; /* flag if overrun occured */ + + /* for copy from cyclic buffer to user buffer */ + unsigned int user_buffer_index; + + /* Members associated with polled mode support */ + u8 *poll_mode_addr_virt; /* virt addr for descriptor writeback */ + dma_addr_t poll_mode_bus; /* bus addr for descriptor writeback */ + + /* Members associated with interrupt mode support */ + wait_queue_head_t shutdown_wq; /* wait queue for shutdown sync */ + spinlock_t lock; /* protects concurrent access */ + int prev_cpu; /* remember CPU# of (last) locker */ + int msix_irq_line; /* MSI-X vector for this engine */ + u32 irq_bitmask; /* IRQ bit mask for this engine */ + struct work_struct work; /* Work queue for interrupt handling */ + + spinlock_t desc_lock; /* protects concurrent access */ + dma_addr_t desc_bus; + struct xdma_desc *desc; + + /* for performance test support */ + struct xdma_performance_ioctl *xdma_perf; /* perf test control */ + wait_queue_head_t xdma_perf_wq; /* Perf test sync */ +}; + +struct xdma_user_irq { + struct xdma_dev *xdev; /* parent device */ + u8 user_idx; /* 0 ~ 15 */ + u8 events_irq; /* accumulated IRQs */ + spinlock_t events_lock; /* lock to safely update events_irq */ + wait_queue_head_t events_wq; /* wait queue to sync waiting threads */ + irq_handler_t handler; + + void *dev; +}; + +/* XDMA PCIe device specific book-keeping */ +#define XDEV_FLAG_OFFLINE 0x1 +struct xdma_dev { + struct list_head list_head; + struct list_head rcu_node; + + unsigned long magic; /* structure ID for sanity checks */ + struct pci_dev *pdev; /* pci device struct from probe() */ + int idx; /* dev index */ + + const char *mod_name; /* name of module owning the dev */ + + spinlock_t lock; /* protects concurrent access */ + unsigned int flags; + + /* PCIe BAR management */ + void *__iomem bar[XDMA_BAR_NUM]; /* addresses for mapped BARs */ + int user_bar_idx; /* BAR index of user logic */ + int config_bar_idx; /* BAR index of XDMA config logic */ + int bypass_bar_idx; /* BAR index of XDMA bypass logic */ + int regions_in_use; /* flag if dev was in use during probe() */ + int got_regions; /* flag if probe() obtained the regions */ + + int user_max; + int c2h_channel_max; + int h2c_channel_max; + + /* Interrupt management */ + int irq_count; /* interrupt counter */ + int irq_line; /* flag if irq allocated successfully */ + int msi_enabled; /* flag if msi was enabled for the device */ + int msix_enabled; /* flag if msi-x was enabled for the device */ +#if LINUX_VERSION_CODE < KERNEL_VERSION(4,12,0) + struct msix_entry entry[32]; /* msi-x vector/entry table */ +#endif + struct xdma_user_irq user_irq[16]; /* user IRQ management */ + unsigned int mask_irq_user; + + /* XDMA engine management */ + int engines_num; /* Total engine count */ + u32 mask_irq_h2c; + u32 mask_irq_c2h; + struct xdma_engine engine_h2c[XDMA_CHANNEL_NUM_MAX]; + struct xdma_engine engine_c2h[XDMA_CHANNEL_NUM_MAX]; + + /* SD_Accel specific */ + enum dev_capabilities capabilities; + u64 feature_id; +}; + +static inline int xdma_device_flag_check(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + if (xdev->flags & f) { + spin_unlock_irqrestore(&xdev->lock, flags); + return 1; + } + spin_unlock_irqrestore(&xdev->lock, flags); + return 0; +} + +static inline int xdma_device_flag_test_n_set(struct xdma_dev *xdev, + unsigned int f) +{ + unsigned long flags; + int rv = 0; + + spin_lock_irqsave(&xdev->lock, flags); + if (xdev->flags & f) { + spin_unlock_irqrestore(&xdev->lock, flags); + rv = 1; + } else + xdev->flags |= f; + spin_unlock_irqrestore(&xdev->lock, flags); + return rv; +} + +static inline void xdma_device_flag_set(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + xdev->flags |= f; + spin_unlock_irqrestore(&xdev->lock, flags); +} + +static inline void xdma_device_flag_clear(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + xdev->flags &= ~f; + spin_unlock_irqrestore(&xdev->lock, flags); +} + +void write_register(u32 value, void *iomem); +u32 read_register(void *iomem); + +struct xdma_dev *xdev_find_by_pdev(struct pci_dev *pdev); + +void xdma_device_offline(struct pci_dev *pdev, void *dev_handle); +void xdma_device_online(struct pci_dev *pdev, void *dev_handle); + +int xdma_performance_submit(struct xdma_dev *xdev, struct xdma_engine *engine); +struct xdma_transfer *engine_cyclic_stop(struct xdma_engine *engine); +void enable_perf(struct xdma_engine *engine); +void get_perf_stats(struct xdma_engine *engine); + +int xdma_cyclic_transfer_setup(struct xdma_engine *engine); +int xdma_cyclic_transfer_teardown(struct xdma_engine *engine); +ssize_t xdma_engine_read_cyclic(struct xdma_engine *, char __user *, size_t, + int); +int engine_addrmode_set(struct xdma_engine *engine, unsigned long arg); + +#endif /* XDMA_LIB_H */ diff --git a/sources/xdma_driver/libxdma/version.h b/sources/xdma_driver/libxdma/version.h new file mode 100644 index 0000000..703424d --- /dev/null +++ b/sources/xdma_driver/libxdma/version.h @@ -0,0 +1,28 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XDMA_VERSION_H__ +#define __XDMA_VERSION_H__ + +#define DRV_MOD_MAJOR 2017 +#define DRV_MOD_MINOR 1 +#define DRV_MOD_PATCHLEVEL 36 + +#define DRV_MODULE_VERSION \ + __stringify(DRV_MOD_MAJOR) "." \ + __stringify(DRV_MOD_MINOR) "." \ + __stringify(DRV_MOD_PATCHLEVEL) + +#define DRV_MOD_VERSION_NUMBER \ + ((DRV_MOD_MAJOR)*1000 + (DRV_MOD_MINOR)*100 + DRV_MOD_PATCHLEVEL) + +#endif /* ifndef __XDMA_VERSION_H__ */ diff --git a/sources/xdma_driver/load_driver.sh b/sources/xdma_driver/load_driver.sh new file mode 100755 index 0000000..9910136 --- /dev/null +++ b/sources/xdma_driver/load_driver.sh @@ -0,0 +1,56 @@ +#!/bin/bash + +# Make sure only root can run our script +if [[ $EUID -ne 0 ]]; then + echo "This script must be run as root" 1>&2 + exit 1 +fi + +cd xdma +make clean +make -j +if [ ! $? == 0 ]; then + echo "Error: Kernel module did not compile properly." + echo " FAILED" + exit 1 +fi +cd .. + + +# Remove the existing xdma kernel module +lsmod | grep xdma +if [ $? -eq 0 ]; then + rmmod xdma +fi +echo -n "Loading xdma driver..." +# Use the following command to Load the driver in the default +# or interrupt drive mode. This will allow the driver to use +# interrupts to signal when DMA transfers are completed. +insmod xdma/xdma.ko enable_credit_mp=1 +# Use the following command to Load the driver in Polling +# mode rather than than interrupt mode. This will allow the +# driver to use polling to determ when DMA transfers are +# completed. +#insmod ../xdma/xdma.ko poll_mode=1 + +if [ ! $? == 0 ]; then + echo "Error: Kernel module did not load properly." + echo " FAILED" + exit 1 +fi + +# Check to see if the xdma devices were recognized +echo "" +cat /proc/devices | grep xdma > /dev/null +returnVal=$? +if [ $returnVal == 0 ]; then + # Installed devices were recognized. + echo "The Kernel module installed correctly and the xmda devices were recognized." +else + # No devices were installed. + echo "Error: The Kernel module installed correctly, but no devices were recognized." + echo " FAILED" + exit 1 +fi + +echo " DONE" diff --git a/sources/xdma_driver/readme.txt b/sources/xdma_driver/readme.txt new file mode 100644 index 0000000..cbc80f0 --- /dev/null +++ b/sources/xdma_driver/readme.txt @@ -0,0 +1,140 @@ +Modifications to make the dirver compatible with newer Linux kernels: +===================================================================== +For kernel version >= 5.3.0: +- xdma/: + - cdev_ctrl.c: Interface changed for access_ok(): https://forums.xilinx.com/t5/PCIe-and-CPM/PCIe-DMA-driver-compilation-issues-in-Linux-Ubuntu-19-04/td-p/1022239 + + - cdev_xvc.c: + - libxdma.c: Does not need mmiowb() + +(This readme is less relevant as we've modified the file structure) + +The files in this directory provide Xilinx PCIe DMA drivers, example software, +and example test scripts that can be used to exercise the Xilinx PCIe DMA IP. + +This software can be used directly or referenced to create drivers and software +for your Xilinx FPGA hardware design. + +Directory and file description: +=============================== + - xdma/: This directory contains the Xilinx PCIe DMA kernel module + driver files. + + - libxdma/: This directory contains support files for the kernel driver module, + which interfaces directly with the XDMA IP. + + - include/: This directory contains all include files that are needed for + compiling driver. + + - etc/: This directory contains rules for the Xilinx PCIe DMA kernel module + and software. The files in this directory should be copied to the /etc/ + directory on your linux system. + + - tests/: This directory contains example application software to exercise the + provided kernel module driver and Xilinx PCIe DMA IP. This directory + also contains the following scripts and directories. + + - load_driver.sh: + This script loads the kernel module and creates the necissary + kernel nodes used by the provided software. + The The kernel device nodes will be created under /dev/xdma*. + Additional device nodes are created under /dev/xdma/card* to + more easily differentiate between multiple PCIe DMA enabled + cards. Root permissions will be required to run this script. + + - run_test.sh: + This script runs sample tests on a Xilinx PCIe DMA target and + returns a pass (0) or fail (1) result. + This script is intended for use with the PCIe DMA example + design. + + - perform_hwcount.sh: + This script runs hardware performance for XDMA for both Host to + Card (H2C) and Card to Host (C2H). The result are copied to + 'hw_log_h2c.txt' and hw_log_c2h.txt' text files. + For each direction the performance script loops from 64 bytes + to 4MBytes and generate performance numbers (byte size doubles + for each loop count). + You can grep for 'data rate' on those two files to see data + rate values. + Data rate values are in percentage of maximum throughput. + Maximum data rate for x8 Gen3 is 8Gbytes/s, so for a x8Gen3 + design value of 0.81 data rate is 0.81*8 = 6.48Gbytes/s. + Maximum data rate for x16 Gen3 is 16Gbytes/s, so for a x16Gen3 + design value of 0.78 data rate is 0.78*16 = 12.48Gbytes/s. + This program can be run on AXI-MM example design. + AXI-ST example design is a loopback design, both H2C and C2H + are connected. Running on AXI-ST example design will not + generate proper numbers. + If a AXI-ST design is independent of H2C and C2H, performance + number can be generated. + - data/: + This directory contains binary data files that are used for DMA + data transfers to the Xilinx FPGA PCIe endpoint device. + +Usage: + - Change directory to the driver directory. + cd xdma + - Compile and install the kernel module driver. + make install + - Change directory to the tools directory. + cd tools + - Compile the provided example test tools. + make + - Copy the provided driver rules from the etc directory to the /etc/ directory + on your system. + cp ../etc/udev/rules.d/* /etc/udev/rules.d/ + - Load the kernel module driver: + a. modprobe xdma + b. using the provided script. + cd tests + ./load_driver.sh + - Run the provided test script to generate basic DMA traffic. + ./run_test.sh + - Check driver Version number + modinfo xdma (or) + modinfo ../xdma/xdma.ko + +Updates and Backward Compaitiblity: + - The following features were added to the PCIe DMA IP and driver in Vivado + 2016.1. These features cannot be used with PCIe DMA IP if the IP was + generated using a Vivado build earlier than 2016.1. + - Poll Mode: Earlier versions of Vivado only support interrupt mode which + is the default behavior of the driver. + - Source/Destination Address: Earlier versions of Vivado PCIe DMA IP + required the low-order bits of the Source and Destination address to be + the same. + As of 2016.1 this restriction has been removed and the Source and + Destination addresses can be any arbitrary address that is valid for + your system. + +Frequently asked questions: + Q: How do I uninstall the kernel module driver? + A: Use the following commands to uninstall the driver. + - Uninstall the kernel module. + rmmod -s xdma + - Delete the dma rules that were added. + rm -f /etc/udev/rules.d/60-xdma.rules + rm -f /etc/udev/rules.d/xdma-udev-command.sh + + Q: How do I modify the PCIe Device IDs recognized by the kernel module driver? + A: The xdma/xdma_mod.c file constains the pci_device_id struct that identifies + the PCIe Device IDs that are recognized by the driver in the following + format: + { PCI_DEVICE(0x10ee, 0x8038), }, + Add, remove, or modify the PCIe Device IDs in this struct as desired. Then + uninstall the existing xdma kernel module, compile the driver again, and + re-install the driver using the load_driver.sh script. + + Q: By default the driver uses interupts to signal when DMA transfers are + completed. How do I modify the driver to use polling rather than + interrupts to determine when DMA transactions are completed? + A: The driver can be changed from being interrupt driven (default) to being + polling driven (poll mode) when the kernel module is inserted. To do this + modify the load_driver.sh file as follows: + Change: insmod xdma/xdma.ko + To: insmod xdma/xdma.ko poll_mode=1 + Note: Interrupt vs Poll mode will apply to all DMA channels. If desired the + driver can be modified such that some channels are interrupt driven while + others are polling driven. Refer to the poll mode section of PG195 for + additional information on using the PCIe DMA IP in poll mode. diff --git a/sources/xdma_driver/xdma/Makefile b/sources/xdma_driver/xdma/Makefile new file mode 100644 index 0000000..d1c07a8 --- /dev/null +++ b/sources/xdma_driver/xdma/Makefile @@ -0,0 +1,35 @@ +SHELL = /bin/bash +ifneq ($(xvc_bar_num),) + XVC_FLAGS += -D__XVC_BAR_NUM__=$(xvc_bar_num) +endif + +ifneq ($(xvc_bar_offset),) + XVC_FLAGS += -D__XVC_BAR_OFFSET__=$(xvc_bar_offset) +endif + +$(warning XVC_FLAGS: $(XVC_FLAGS).) + +topdir := $(shell cd $(src)/.. && pwd) + +TARGET_MODULE:=xdma + +EXTRA_CFLAGS := -I$(topdir)/include $(XVC_FLAGS) +#EXTRA_CFLAGS += -D__LIBXDMA_DEBUG__ +#EXTRA_CFLAGS += -DINTERNAL_TESTING + +ifneq ($(KERNELRELEASE),) + $(TARGET_MODULE)-objs := libxdma.o xdma_cdev.o cdev_ctrl.o cdev_events.o cdev_sgdma.o cdev_xvc.o cdev_bypass.o xdma_mod.o + obj-m := $(TARGET_MODULE).o +else + BUILDSYSTEM_DIR:=/lib/modules/$(shell uname -r)/build + PWD:=$(shell pwd) +all : + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) modules + +clean: + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) clean + +install: all + $(MAKE) -C $(BUILDSYSTEM_DIR) M=$(PWD) modules_install + +endif diff --git a/sources/xdma_driver/xdma/cdev_bypass.c b/sources/xdma_driver/xdma/cdev_bypass.c new file mode 100644 index 0000000..4b84526 --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_bypass.c @@ -0,0 +1,180 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#include "libxdma_api.h" +#include "xdma_cdev.h" + +#define write_register(v,mem,off) iowrite32(v, mem) + +static int copy_desc_data(struct xdma_transfer *transfer, char __user *buf, + size_t *buf_offset, size_t buf_size) +{ + int i; + int copy_err; + int rc = 0; + + BUG_ON(!buf); + BUG_ON(!buf_offset); + + /* Fill user buffer with descriptor data */ + for (i = 0; i < transfer->desc_num; i++) { + if (*buf_offset + sizeof(struct xdma_desc) <= buf_size) { + copy_err = copy_to_user(&buf[*buf_offset], + transfer->desc_virt + i, + sizeof(struct xdma_desc)); + + if (copy_err) { + dbg_sg("Copy to user buffer failed\n"); + *buf_offset = buf_size; + rc = -EINVAL; + } else { + *buf_offset += sizeof(struct xdma_desc); + } + } else { + rc = -ENOMEM; + } + } + + return rc; +} + +static ssize_t char_bypass_read(struct file *file, char __user *buf, + size_t count, loff_t *pos) +{ + struct xdma_dev *xdev; + struct xdma_engine *engine; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_transfer *transfer; + struct list_head *idx; + size_t buf_offset = 0; + int rc = 0; + + rc = xcdev_check(__func__, xcdev, 1); + if (rc < 0) + return rc; + xdev = xcdev->xdev; + engine = xcdev->engine; + + dbg_sg("In char_bypass_read()\n"); + + if (count & 3) { + dbg_sg("Buffer size must be a multiple of 4 bytes\n"); + return -EINVAL; + } + + if (!buf) { + dbg_sg("Caught NULL pointer\n"); + return -EINVAL; + } + + if (xdev->bypass_bar_idx < 0) { + dbg_sg("Bypass BAR not present - unsupported operation\n"); + return -ENODEV; + } + + spin_lock(&engine->lock); + + if (!list_empty(&engine->transfer_list)) { + list_for_each(idx, &engine->transfer_list) { + transfer = list_entry(idx, struct xdma_transfer, entry); + + rc = copy_desc_data(transfer, buf, &buf_offset, count); + } + } + + spin_unlock(&engine->lock); + + if (rc < 0) + return rc; + else + return buf_offset; +} + +static ssize_t char_bypass_write(struct file *file, const char __user *buf, + size_t count, loff_t *pos) +{ + struct xdma_dev *xdev; + struct xdma_engine *engine; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + + u32 desc_data; + u32 *bypass_addr; + size_t buf_offset = 0; + int rc = 0; + int copy_err; + + rc = xcdev_check(__func__, xcdev, 1); + if (rc < 0) + return rc; + xdev = xcdev->xdev; + engine = xcdev->engine; + + if (count & 3) { + dbg_sg("Buffer size must be a multiple of 4 bytes\n"); + return -EINVAL; + } + + if (!buf) { + dbg_sg("Caught NULL pointer\n"); + return -EINVAL; + } + + if (xdev->bypass_bar_idx < 0) { + dbg_sg("Bypass BAR not present - unsupported operation\n"); + return -ENODEV; + } + + dbg_sg("In char_bypass_write()\n"); + + spin_lock(&engine->lock); + + /* Write descriptor data to the bypass BAR */ + bypass_addr = (u32 *)xdev->bar[xdev->bypass_bar_idx]; + bypass_addr += engine->bypass_offset; + while (buf_offset < count) { + copy_err = copy_from_user(&desc_data, &buf[buf_offset], + sizeof(u32)); + if (!copy_err) { + write_register(desc_data, bypass_addr, bypass_addr - engine->bypass_offset); + buf_offset += sizeof(u32); + rc = buf_offset; + } else { + dbg_sg("Error reading data from userspace buffer\n"); + rc = -EINVAL; + break; + } + } + + spin_unlock(&engine->lock); + + + return rc; +} + + +/* + * character device file operations for bypass operation + */ + +static const struct file_operations bypass_fops = { + .owner = THIS_MODULE, + .open = char_open, + .release = char_close, + .read = char_bypass_read, + .write = char_bypass_write, + .mmap = bridge_mmap, +}; + +void cdev_bypass_init(struct xdma_cdev *xcdev) +{ + cdev_init(&xcdev->cdev, &bypass_fops); +} diff --git a/sources/xdma_driver/xdma/cdev_bypass.o.ur-safe b/sources/xdma_driver/xdma/cdev_bypass.o.ur-safe new file mode 100644 index 0000000..0cd6b8c --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_bypass.o.ur-safe @@ -0,0 +1,2 @@ +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/cdev_bypass.o-.text-23c +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/cdev_bypass.o-.text-e9 diff --git a/sources/xdma_driver/xdma/cdev_ctrl.c b/sources/xdma_driver/xdma/cdev_ctrl.c new file mode 100644 index 0000000..6699f5c --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_ctrl.c @@ -0,0 +1,267 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include <linux/ioctl.h> +#include "version.h" +#include "xdma_cdev.h" +#include "cdev_ctrl.h" + +/* + * character device file operations for control bus (through control bridge) + */ +static ssize_t char_ctrl_read(struct file *fp, char __user *buf, size_t count, + loff_t *pos) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)fp->private_data; + struct xdma_dev *xdev; + void *reg; + u32 w; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + + /* only 32-bit aligned and 32-bit multiples */ + if (*pos & 3) + return -EPROTO; + /* first address is BAR base plus file position offset */ + reg = xdev->bar[xcdev->bar] + *pos; + //w = read_register(reg); + w = ioread32(reg); + dbg_sg("char_ctrl_read(@%p, count=%ld, pos=%d) value = 0x%08x\n", reg, + (long)count, (int)*pos, w); + rv = copy_to_user(buf, &w, 4); + if (rv) + dbg_sg("Copy to userspace failed but continuing\n"); + + *pos += 4; + return 4; +} + +static ssize_t char_ctrl_write(struct file *file, const char __user *buf, + size_t count, loff_t *pos) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_dev *xdev; + void *reg; + u32 w; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + + /* only 32-bit aligned and 32-bit multiples */ + if (*pos & 3) + return -EPROTO; + + /* first address is BAR base plus file position offset */ + reg = xdev->bar[xcdev->bar] + *pos; + rv = copy_from_user(&w, buf, 4); + if (rv) { + pr_info("copy from user failed %d/4, but continuing.\n", rv); + } + + dbg_sg("char_ctrl_write(0x%08x @%p, count=%ld, pos=%d)\n", w, reg, + (long)count, (int)*pos); + //write_register(w, reg); + iowrite32(w, reg); + *pos += 4; + return 4; +} + +static long version_ioctl(struct xdma_cdev *xcdev, void __user *arg) +{ + struct xdma_ioc_info obj; + struct xdma_dev *xdev = xcdev->xdev; + int rv; + + rv = copy_from_user((void *)&obj, arg, sizeof(struct xdma_ioc_info)); + if (rv) { + pr_info("copy from user failed %d/%ld.\n", + rv, sizeof(struct xdma_ioc_info)); + return -EFAULT; + } + memset(&obj, 0, sizeof(obj)); + obj.vendor = xdev->pdev->vendor; + obj.device = xdev->pdev->device; + obj.subsystem_vendor = xdev->pdev->subsystem_vendor; + obj.subsystem_device = xdev->pdev->subsystem_device; + obj.feature_id = xdev->feature_id; + obj.driver_version = DRV_MOD_VERSION_NUMBER; + obj.domain = 0; + obj.bus = PCI_BUS_NUM(xdev->pdev->devfn); + obj.dev = PCI_SLOT(xdev->pdev->devfn); + obj.func = PCI_FUNC(xdev->pdev->devfn); + if (copy_to_user(arg, &obj, sizeof(struct xdma_ioc_info))) + return -EFAULT; + return 0; +} + +long char_ctrl_ioctl(struct file *filp, unsigned int cmd, unsigned long arg) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)filp->private_data; + struct xdma_dev *xdev; + struct xdma_ioc_base ioctl_obj; + long result = 0; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + + pr_info("cmd 0x%x, xdev 0x%p, pdev 0x%p.\n", cmd, xdev, xdev->pdev); + + if (_IOC_TYPE(cmd) != XDMA_IOC_MAGIC) { + pr_err("cmd %u, bad magic 0x%x/0x%x.\n", + cmd, _IOC_TYPE(cmd), XDMA_IOC_MAGIC); + return -ENOTTY; + } + + #if LINUX_VERSION_CODE < KERNEL_VERSION(5,3,0) + if (_IOC_DIR(cmd) & _IOC_READ) + result = !access_ok(VERIFY_WRITE, (void __user *)arg, + _IOC_SIZE(cmd)); + else if (_IOC_DIR(cmd) & _IOC_WRITE) + result = !access_ok(VERIFY_READ, (void __user *)arg, + _IOC_SIZE(cmd)); + #else + if (_IOC_DIR(cmd) & _IOC_READ) + result = !access_ok((void __user *)arg, + _IOC_SIZE(cmd)); + else if (_IOC_DIR(cmd) & _IOC_WRITE) + result = !access_ok((void __user *)arg, + _IOC_SIZE(cmd)); + #endif + + if (result) { + pr_err("bad access %ld.\n", result); + return -EFAULT; + } + + switch (cmd) { + case XDMA_IOCINFO: + if (copy_from_user((void *)&ioctl_obj, (void *) arg, + sizeof(struct xdma_ioc_base))) { + pr_err("copy_from_user failed.\n"); + return -EFAULT; + } + + if (ioctl_obj.magic != XDMA_XCL_MAGIC) { + pr_err("magic 0x%x != XDMA_XCL_MAGIC (0x%x).\n", + ioctl_obj.magic, XDMA_XCL_MAGIC); + return -ENOTTY; + } + + return version_ioctl(xcdev, (void __user *)arg); + case XDMA_IOCOFFLINE: + if (!xdev) { + pr_info("cmd %u, xdev NULL.\n", cmd); + return -EINVAL; + } + xdma_device_offline(xdev->pdev, xdev); + break; + case XDMA_IOCONLINE: + if (!xdev) { + pr_info("cmd %u, xdev NULL.\n", cmd); + return -EINVAL; + } + xdma_device_online(xdev->pdev, xdev); + break; + default: + pr_err("UNKNOWN ioctl cmd 0x%x.\n", cmd); + return -ENOTTY; + } + return 0; +} + +/* maps the PCIe BAR into user space for memory-like access using mmap() */ +int bridge_mmap(struct file *file, struct vm_area_struct *vma) +{ + struct xdma_dev *xdev; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + unsigned long off; + unsigned long phys; + unsigned long vsize; + unsigned long psize; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + + off = vma->vm_pgoff << PAGE_SHIFT; + /* BAR physical address */ + phys = pci_resource_start(xdev->pdev, xcdev->bar) + off; + vsize = vma->vm_end - vma->vm_start; + /* complete resource */ + psize = pci_resource_end(xdev->pdev, xcdev->bar) - + pci_resource_start(xdev->pdev, xcdev->bar) + 1 - off; + + dbg_sg("mmap(): xcdev = 0x%08lx\n", (unsigned long)xcdev); + dbg_sg("mmap(): cdev->bar = %d\n", xcdev->bar); + dbg_sg("mmap(): xdev = 0x%p\n", xdev); + dbg_sg("mmap(): pci_dev = 0x%08lx\n", (unsigned long)xdev->pdev); + + dbg_sg("off = 0x%lx\n", off); + dbg_sg("start = 0x%llx\n", + (unsigned long long)pci_resource_start(xdev->pdev, + xcdev->bar)); + dbg_sg("phys = 0x%lx\n", phys); + + if (vsize > psize) + return -EINVAL; + /* + * pages must not be cached as this would result in cache line sized + * accesses to the end point + */ + vma->vm_page_prot = pgprot_noncached(vma->vm_page_prot); + /* + * prevent touching the pages (byte access) for swap-in, + * and prevent the pages from being swapped out + */ + vma->vm_flags |= VMEM_FLAGS; + /* make MMIO accessible to user space */ + rv = io_remap_pfn_range(vma, vma->vm_start, phys >> PAGE_SHIFT, + vsize, vma->vm_page_prot); + dbg_sg("vma=0x%p, vma->vm_start=0x%lx, phys=0x%lx, size=%lu = %d\n", + vma, vma->vm_start, phys >> PAGE_SHIFT, vsize, rv); + + if (rv) + return -EAGAIN; + return 0; +} + +/* + * character device file operations for control bus (through control bridge) + */ +static const struct file_operations ctrl_fops = { + .owner = THIS_MODULE, + .open = char_open, + .release = char_close, + .read = char_ctrl_read, + .write = char_ctrl_write, + .mmap = bridge_mmap, + .unlocked_ioctl = char_ctrl_ioctl, +}; + +void cdev_ctrl_init(struct xdma_cdev *xcdev) +{ + cdev_init(&xcdev->cdev, &ctrl_fops); +} diff --git a/sources/xdma_driver/xdma/cdev_ctrl.h b/sources/xdma_driver/xdma/cdev_ctrl.h new file mode 100644 index 0000000..cb8ef6d --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_ctrl.h @@ -0,0 +1,80 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef _XDMA_IOCALLS_POSIX_H_ +#define _XDMA_IOCALLS_POSIX_H_ + +#include <linux/ioctl.h> + +/* Use 'x' as magic number */ +#define XDMA_IOC_MAGIC 'x' +/* XL OpenCL X->58(ASCII), L->6C(ASCII), O->0 C->C L->6C(ASCII); */ +#define XDMA_XCL_MAGIC 0X586C0C6C + +/* + * S means "Set" through a ptr, + * T means "Tell" directly with the argument value + * G means "Get": reply by setting through a pointer + * Q means "Query": response is on the return value + * X means "eXchange": switch G and S atomically + * H means "sHift": switch T and Q atomically + * + * _IO(type,nr) no arguments + * _IOR(type,nr,datatype) read data from driver + * _IOW(type,nr.datatype) write data to driver + * _IORW(type,nr,datatype) read/write data + * + * _IOC_DIR(nr) returns direction + * _IOC_TYPE(nr) returns magic + * _IOC_NR(nr) returns number + * _IOC_SIZE(nr) returns size + */ + +enum XDMA_IOC_TYPES { + XDMA_IOC_NOP, + XDMA_IOC_INFO, + XDMA_IOC_OFFLINE, + XDMA_IOC_ONLINE, + XDMA_IOC_MAX +}; + +struct xdma_ioc_base { + unsigned int magic; + unsigned int command; +}; + +struct xdma_ioc_info { + struct xdma_ioc_base base; + unsigned short vendor; + unsigned short device; + unsigned short subsystem_vendor; + unsigned short subsystem_device; + unsigned int dma_engine_version; + unsigned int driver_version; + unsigned long long feature_id; + unsigned short domain; + unsigned char bus; + unsigned char dev; + unsigned char func; +}; + +/* IOCTL codes */ +#define XDMA_IOCINFO _IOWR(XDMA_IOC_MAGIC, XDMA_IOC_INFO, \ + struct xdma_ioc_info) +#define XDMA_IOCOFFLINE _IO(XDMA_IOC_MAGIC, XDMA_IOC_OFFLINE) +#define XDMA_IOCONLINE _IO(XDMA_IOC_MAGIC, XDMA_IOC_ONLINE) + +#define IOCTL_XDMA_ADDRMODE_SET _IOW('q', 4, int) +#define IOCTL_XDMA_ADDRMODE_GET _IOR('q', 5, int) +#define IOCTL_XDMA_ALIGN_GET _IOR('q', 6, int) + +#endif /* _XDMA_IOCALLS_POSIX_H_ */ diff --git a/sources/xdma_driver/xdma/cdev_events.c b/sources/xdma_driver/xdma/cdev_events.c new file mode 100644 index 0000000..3ada7db --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_events.c @@ -0,0 +1,113 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include "xdma_cdev.h" + +/* + * character device file operations for events + */ +static ssize_t char_events_read(struct file *file, char __user *buf, + size_t count, loff_t *pos) +{ + int rv; + struct xdma_user_irq *user_irq; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + u32 events_user; + unsigned long flags; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + user_irq = xcdev->user_irq; + if (!user_irq) { + pr_info("xcdev 0x%p, user_irq NULL.\n", xcdev); + return -EINVAL; + } + + if (count != 4) + return -EPROTO; + + if (*pos & 3) + return -EPROTO; + + /* + * sleep until any interrupt events have occurred, + * or a signal arrived + */ + rv = wait_event_interruptible(user_irq->events_wq, + user_irq->events_irq != 0); + if (rv) + dbg_sg("wait_event_interruptible=%d\n", rv); + + /* wait_event_interruptible() was interrupted by a signal */ + if (rv == -ERESTARTSYS) + return -ERESTARTSYS; + + /* atomically decide which events are passed to the user */ + spin_lock_irqsave(&user_irq->events_lock, flags); + events_user = user_irq->events_irq; + user_irq->events_irq = 0; + spin_unlock_irqrestore(&user_irq->events_lock, flags); + + rv = copy_to_user(buf, &events_user, 4); + if (rv) + dbg_sg("Copy to user failed but continuing\n"); + + return 4; +} + +static unsigned int char_events_poll(struct file *file, poll_table *wait) +{ + struct xdma_user_irq *user_irq; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + unsigned long flags; + unsigned int mask = 0; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + user_irq = xcdev->user_irq; + if (!user_irq) { + pr_info("xcdev 0x%p, user_irq NULL.\n", xcdev); + return -EINVAL; + } + + poll_wait(file, &user_irq->events_wq, wait); + + spin_lock_irqsave(&user_irq->events_lock, flags); + if (user_irq->events_irq) + mask = POLLIN | POLLRDNORM; /* readable */ + + spin_unlock_irqrestore(&user_irq->events_lock, flags); + + return mask; +} + +/* + * character device file operations for the irq events + */ +static const struct file_operations events_fops = { + .owner = THIS_MODULE, + .open = char_open, + .release = char_close, + .read = char_events_read, + .poll = char_events_poll, +}; + +void cdev_event_init(struct xdma_cdev *xcdev) +{ + xcdev->user_irq = &(xcdev->xdev->user_irq[xcdev->bar]); + cdev_init(&xcdev->cdev, &events_fops); +} diff --git a/sources/xdma_driver/xdma/cdev_sgdma.c b/sources/xdma_driver/xdma/cdev_sgdma.c new file mode 100644 index 0000000..410ef45 --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_sgdma.c @@ -0,0 +1,552 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include <asm/cacheflush.h> +#include "libxdma_api.h" +#include "xdma_cdev.h" +#include "cdev_sgdma.h" + +/* Module Parameters */ +unsigned int sgdma_timeout = 10; +module_param(sgdma_timeout, uint, 0644); +MODULE_PARM_DESC(sgdma_timeout, "timeout in seconds for sgdma, default is 10 sec."); + +/* + * character device file operations for SG DMA engine + */ +static loff_t char_sgdma_llseek(struct file *file, loff_t off, int whence) +{ + loff_t newpos = 0; + + switch (whence) { + case 0: /* SEEK_SET */ + newpos = off; + break; + case 1: /* SEEK_CUR */ + newpos = file->f_pos + off; + break; + case 2: /* SEEK_END, @TODO should work from end of address space */ + newpos = UINT_MAX + off; + break; + default: /* can't happen */ + return -EINVAL; + } + if (newpos < 0) + return -EINVAL; + file->f_pos = newpos; + dbg_fops("char_sgdma_llseek: pos=%lld\n", (signed long long)newpos); + +#if 0 + pr_err("0x%p, off 0x%lld, whence %d -> pos %lld.\n", + file, (signed long long)off, whence, (signed long long)off); +#endif + + return newpos; +} + +/* char_sgdma_read_write() -- Read from or write to the device + * + * @buf userspace buffer + * @count number of bytes in the userspace buffer + * @pos byte-address in device + * @dir_to_device If !0, a write to the device is performed + * + * Iterate over the userspace buffer, taking at most 255 * PAGE_SIZE bytes for + * each DMA transfer. + * + * For each transfer, get the user pages, build a sglist, map, build a + * descriptor table. submit the transfer. wait for the interrupt handler + * to wake us on completion. + */ + +static int check_transfer_align(struct xdma_engine *engine, + const char __user *buf, size_t count, loff_t pos, int sync) +{ + BUG_ON(!engine); + + /* AXI ST or AXI MM non-incremental addressing mode? */ + if (engine->non_incr_addr) { + int buf_lsb = (int)((uintptr_t)buf) & (engine->addr_align - 1); + size_t len_lsb = count & ((size_t)engine->len_granularity - 1); + int pos_lsb = (int)pos & (engine->addr_align - 1); + + dbg_tfr("AXI ST or MM non-incremental\n"); + dbg_tfr("buf_lsb = %d, pos_lsb = %d, len_lsb = %ld\n", buf_lsb, + pos_lsb, len_lsb); + + if (buf_lsb != 0) { + dbg_tfr("FAIL: non-aligned buffer address %p\n", buf); + return -EINVAL; + } + + if ((pos_lsb != 0) && (sync)) { + dbg_tfr("FAIL: non-aligned AXI MM FPGA addr 0x%llx\n", + (unsigned long long)pos); + return -EINVAL; + } + + if (len_lsb != 0) { + dbg_tfr("FAIL: len %d is not a multiple of %d\n", + (int)count, + (int)engine->len_granularity); + return -EINVAL; + } + /* AXI MM incremental addressing mode */ + } else { + int buf_lsb = (int)((uintptr_t)buf) & (engine->addr_align - 1); + int pos_lsb = (int)pos & (engine->addr_align - 1); + + if (buf_lsb != pos_lsb) { + dbg_tfr("FAIL: Misalignment error\n"); + dbg_tfr("host addr %p, FPGA addr 0x%llx\n", buf, pos); + return -EINVAL; + } + } + + return 0; +} + +/* + * Map a user memory range into a scatterlist + * inspired by vhost_scsi_map_to_sgl() + * Returns the number of scatterlist entries used or -errno on error. + */ +static inline void xdma_io_cb_release(struct xdma_io_cb *cb) +{ + int i; + + for (i = 0; i < cb->pages_nr; i++) + put_page(cb->pages[i]); + + sg_free_table(&cb->sgt); + kfree(cb->pages); + + memset(cb, 0, sizeof(*cb)); +} + +static void char_sgdma_unmap_user_buf(struct xdma_io_cb *cb, bool write) +{ + int i; + + sg_free_table(&cb->sgt); + + if (!cb->pages || !cb->pages_nr) + return; + + for (i = 0; i < cb->pages_nr; i++) { + if (cb->pages[i]) { + if (!write) + set_page_dirty_lock(cb->pages[i]); + put_page(cb->pages[i]); + } else + break; + } + + if (i != cb->pages_nr) + pr_info("sgl pages %d/%u.\n", i, cb->pages_nr); + + kfree(cb->pages); + cb->pages = NULL; +} + +static int char_sgdma_map_user_buf_to_sgl(struct xdma_io_cb *cb, bool write) +{ + struct sg_table *sgt = &cb->sgt; + unsigned long len = cb->len; + char *buf = cb->buf; + struct scatterlist *sg; + unsigned int pages_nr = (((unsigned long)buf + len + PAGE_SIZE -1) - + ((unsigned long)buf & PAGE_MASK)) + >> PAGE_SHIFT; + int i; + int rv; + + if (pages_nr == 0) { + return -EINVAL; + } + + if (sg_alloc_table(sgt, pages_nr, GFP_KERNEL)) { + pr_err("sgl OOM.\n"); + return -ENOMEM; + } + + cb->pages = kcalloc(pages_nr, sizeof(struct page *), GFP_KERNEL); + if (!cb->pages) { + pr_err("pages OOM.\n"); + rv = -ENOMEM; + goto err_out; + } + + rv = get_user_pages_fast((unsigned long)buf, pages_nr, 1/* write */, + cb->pages); + /* No pages were pinned */ + if (rv < 0) { + pr_err("unable to pin down %u user pages, %d.\n", + pages_nr, rv); + goto err_out; + } + /* Less pages pinned than wanted */ + if (rv != pages_nr) { + pr_err("unable to pin down all %u user pages, %d.\n", + pages_nr, rv); + rv = -EFAULT; + cb->pages_nr = rv; + goto err_out; + } + + for (i = 1; i < pages_nr; i++) { + if (cb->pages[i - 1] == cb->pages[i]) { + pr_err("duplicate pages, %d, %d.\n", + i - 1, i); + rv = -EFAULT; + cb->pages_nr = pages_nr; + goto err_out; + } + } + + sg = sgt->sgl; + for (i = 0; i < pages_nr; i++, sg = sg_next(sg)) { + //unsigned int offset = (uintptr_t)buf & ~PAGE_MASK; + unsigned int offset = offset_in_page(buf); + unsigned int nbytes = min_t(unsigned int, PAGE_SIZE - offset, len); + + flush_dcache_page(cb->pages[i]); + sg_set_page(sg, cb->pages[i], nbytes, offset); + + buf += nbytes; + len -= nbytes; + } + + BUG_ON(len); + cb->pages_nr = pages_nr; + return 0; + +err_out: + char_sgdma_unmap_user_buf(cb, write); + + return rv; +} + +static ssize_t char_sgdma_read_write(struct file *file, char __user *buf, + size_t count, loff_t *pos, bool write) +{ + int rv; + ssize_t res = 0; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_dev *xdev; + struct xdma_engine *engine; + struct xdma_io_cb cb; + + rv = xcdev_check(__func__, xcdev, 1); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + engine = xcdev->engine; + + dbg_tfr("file 0x%p, priv 0x%p, buf 0x%p,%llu, pos %llu, W %d, %s.\n", + file, file->private_data, buf, (u64)count, (u64)*pos, write, + engine->name); + + if ((write && engine->dir != DMA_TO_DEVICE) || + (!write && engine->dir != DMA_FROM_DEVICE)) { + pr_err("r/w mismatch. W %d, dir %d.\n", + write, engine->dir); + return -EINVAL; + } + + rv = check_transfer_align(engine, buf, count, *pos, 1); + if (rv) { + pr_info("Invalid transfer alignment detected\n"); + return rv; + } + + memset(&cb, 0, sizeof(struct xdma_io_cb)); + cb.buf = buf; + cb.len = count; + rv = char_sgdma_map_user_buf_to_sgl(&cb, write); + if (rv < 0) + return rv; + + res = xdma_xfer_submit(xdev, engine->channel, write, *pos, &cb.sgt, + 0, sgdma_timeout * 1000); + //pr_err("xfer_submit return=%lld.\n", (s64)res); + + //interrupt_status(xdev); + + char_sgdma_unmap_user_buf(&cb, write); + + return res; +} + + +static ssize_t char_sgdma_write(struct file *file, const char __user *buf, + size_t count, loff_t *pos) +{ + return char_sgdma_read_write(file, (char *)buf, count, pos, 1); +} + +static ssize_t char_sgdma_read(struct file *file, char __user *buf, + size_t count, loff_t *pos) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_engine *engine; + int rv; + + rv = xcdev_check(__func__, xcdev, 1); + if (rv < 0) + return rv; + + engine = xcdev->engine; + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) { + rv = xdma_cyclic_transfer_setup(engine); + if (rv < 0 && rv != -EBUSY) + return rv; + /* 600 sec. timeout */ + return xdma_engine_read_cyclic(engine, buf, count, 600000); + } + + return char_sgdma_read_write(file, (char *)buf, count, pos, 0); +} + +static int ioctl_do_perf_start(struct xdma_engine *engine, unsigned long arg) +{ + int rv; + struct xdma_dev *xdev; + + BUG_ON(!engine); + xdev = engine->xdev; + BUG_ON(!xdev); + + /* performance measurement already running on this engine? */ + if (engine->xdma_perf) { + dbg_perf("IOCTL_XDMA_PERF_START failed!\n"); + dbg_perf("Perf measurement already seems to be running!\n"); + return -EBUSY; + } + engine->xdma_perf = kzalloc(sizeof(struct xdma_performance_ioctl), + GFP_KERNEL); + + if (!engine->xdma_perf) + return -ENOMEM; + + rv = copy_from_user(engine->xdma_perf, + (struct xdma_performance_ioctl *)arg, + sizeof(struct xdma_performance_ioctl)); + + if (rv < 0) { + dbg_perf("Failed to copy from user space 0x%lx\n", arg); + return -EINVAL; + } + if (engine->xdma_perf->version != IOCTL_XDMA_PERF_V1) { + dbg_perf("Unsupported IOCTL version %d\n", + engine->xdma_perf->version); + return -EINVAL; + } + + enable_perf(engine); + dbg_perf("transfer_size = %d\n", engine->xdma_perf->transfer_size); + /* initialize wait queue */ + init_waitqueue_head(&engine->xdma_perf_wq); + xdma_performance_submit(xdev, engine); + + return 0; +} + +static int ioctl_do_perf_stop(struct xdma_engine *engine, unsigned long arg) +{ + struct xdma_transfer *transfer = NULL; + int rv; + + dbg_perf("IOCTL_XDMA_PERF_STOP\n"); + + /* no performance measurement running on this engine? */ + if (!engine->xdma_perf) { + dbg_perf("No measurement in progress\n"); + return -EINVAL; + } + + /* stop measurement */ + transfer = engine_cyclic_stop(engine); + dbg_perf("Waiting for measurement to stop\n"); + + if (engine->xdma_perf) { + get_perf_stats(engine); + + rv = copy_to_user((void __user *)arg, engine->xdma_perf, + sizeof(struct xdma_performance_ioctl)); + if (rv) { + dbg_perf("Error copying result to user\n"); + return -EINVAL; + } + } else { + dbg_perf("engine->xdma_perf == NULL?\n"); + } + + kfree(engine->xdma_perf); + engine->xdma_perf = NULL; + + return 0; +} + +static int ioctl_do_perf_get(struct xdma_engine *engine, unsigned long arg) +{ + int rc; + + BUG_ON(!engine); + + dbg_perf("IOCTL_XDMA_PERF_GET\n"); + + if (engine->xdma_perf) { + get_perf_stats(engine); + + rc = copy_to_user((void __user *)arg, engine->xdma_perf, + sizeof(struct xdma_performance_ioctl)); + if (rc) { + dbg_perf("Error copying result to user\n"); + return -EINVAL; + } + } else { + dbg_perf("engine->xdma_perf == NULL?\n"); + return -EPROTO; + } + + return 0; +} + +static int ioctl_do_addrmode_set(struct xdma_engine *engine, unsigned long arg) +{ + return engine_addrmode_set(engine, arg); +} + +static int ioctl_do_addrmode_get(struct xdma_engine *engine, unsigned long arg) +{ + int rv; + unsigned long src; + + BUG_ON(!engine); + src = !!engine->non_incr_addr; + + dbg_perf("IOCTL_XDMA_ADDRMODE_GET\n"); + rv = put_user(src, (int __user *)arg); + + return rv; +} + +static int ioctl_do_align_get(struct xdma_engine *engine, unsigned long arg) +{ + BUG_ON(!engine); + + dbg_perf("IOCTL_XDMA_ALIGN_GET\n"); + return put_user(engine->addr_align, (int __user *)arg); +} + +static long char_sgdma_ioctl(struct file *file, unsigned int cmd, + unsigned long arg) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_dev *xdev; + struct xdma_engine *engine; + + int rv = 0; + + rv = xcdev_check(__func__, xcdev, 1); + if (rv < 0) + return rv; + + xdev = xcdev->xdev; + engine = xcdev->engine; + + switch (cmd) { + case IOCTL_XDMA_PERF_START: + rv = ioctl_do_perf_start(engine, arg); + break; + case IOCTL_XDMA_PERF_STOP: + rv = ioctl_do_perf_stop(engine, arg); + break; + case IOCTL_XDMA_PERF_GET: + rv = ioctl_do_perf_get(engine, arg); + break; + case IOCTL_XDMA_ADDRMODE_SET: + rv = ioctl_do_addrmode_set(engine, arg); + break; + case IOCTL_XDMA_ADDRMODE_GET: + rv = ioctl_do_addrmode_get(engine, arg); + break; + case IOCTL_XDMA_ALIGN_GET: + rv = ioctl_do_align_get(engine, arg); + break; + default: + dbg_perf("Unsupported operation\n"); + rv = -EINVAL; + break; + } + + return rv; +} + +static int char_sgdma_open(struct inode *inode, struct file *file) +{ + struct xdma_cdev *xcdev; + struct xdma_engine *engine; + + char_open(inode, file); + + xcdev = (struct xdma_cdev *)file->private_data; + engine = xcdev->engine; + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) { + if (engine->device_open == 1) + return -EBUSY; + else + engine->device_open = 1; + } + + return 0; +} + +static int char_sgdma_close(struct inode *inode, struct file *file) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + struct xdma_engine *engine; + int rv; + + rv = xcdev_check(__func__, xcdev, 1); + if (rv < 0) + return rv; + + engine = xcdev->engine; + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) { + engine->device_open = 0; + if (engine->cyclic_req) + return xdma_cyclic_transfer_teardown(engine); + } + + return 0; +} +static const struct file_operations sgdma_fops = { + .owner = THIS_MODULE, + .open = char_sgdma_open, + .release = char_sgdma_close, + .write = char_sgdma_write, + .read = char_sgdma_read, + .unlocked_ioctl = char_sgdma_ioctl, + .llseek = char_sgdma_llseek, +}; + +void cdev_sgdma_init(struct xdma_cdev *xcdev) +{ + cdev_init(&xcdev->cdev, &sgdma_fops); +} diff --git a/sources/xdma_driver/xdma/cdev_sgdma.h b/sources/xdma_driver/xdma/cdev_sgdma.h new file mode 100644 index 0000000..7781fae --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_sgdma.h @@ -0,0 +1,66 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef _XDMA_IOCALLS_POSIX_H_ +#define _XDMA_IOCALLS_POSIX_H_ + +#include <linux/ioctl.h> + + +#define IOCTL_XDMA_PERF_V1 (1) +#define XDMA_ADDRMODE_MEMORY (0) +#define XDMA_ADDRMODE_FIXED (1) + +/* + * S means "Set" through a ptr, + * T means "Tell" directly with the argument value + * G means "Get": reply by setting through a pointer + * Q means "Query": response is on the return value + * X means "eXchange": switch G and S atomically + * H means "sHift": switch T and Q atomically + * + * _IO(type,nr) no arguments + * _IOR(type,nr,datatype) read data from driver + * _IOW(type,nr.datatype) write data to driver + * _IORW(type,nr,datatype) read/write data + * + * _IOC_DIR(nr) returns direction + * _IOC_TYPE(nr) returns magic + * _IOC_NR(nr) returns number + * _IOC_SIZE(nr) returns size + */ + +struct xdma_performance_ioctl +{ + /* IOCTL_XDMA_IOCTL_Vx */ + uint32_t version; + uint32_t transfer_size; + /* measurement */ + uint32_t stopped; + uint32_t iterations; + uint64_t clock_cycle_count; + uint64_t data_cycle_count; + uint64_t pending_count; +}; + + + +/* IOCTL codes */ + +#define IOCTL_XDMA_PERF_START _IOW('q', 1, struct xdma_performance_ioctl *) +#define IOCTL_XDMA_PERF_STOP _IOW('q', 2, struct xdma_performance_ioctl *) +#define IOCTL_XDMA_PERF_GET _IOR('q', 3, struct xdma_performance_ioctl *) +#define IOCTL_XDMA_ADDRMODE_SET _IOW('q', 4, int) +#define IOCTL_XDMA_ADDRMODE_GET _IOR('q', 5, int) +#define IOCTL_XDMA_ALIGN_GET _IOR('q', 6, int) + +#endif /* _XDMA_IOCALLS_POSIX_H_ */ diff --git a/sources/xdma_driver/xdma/cdev_xvc.c b/sources/xdma_driver/xdma/cdev_xvc.c new file mode 100644 index 0000000..a077bb1 --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_xvc.c @@ -0,0 +1,238 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include "xdma_cdev.h" +#include "cdev_xvc.h" + +#define COMPLETION_LOOP_MAX 100 + +#define XVC_BAR_LENGTH_REG 0x0 +#define XVC_BAR_TMS_REG 0x4 +#define XVC_BAR_TDI_REG 0x8 +#define XVC_BAR_TDO_REG 0xC +#define XVC_BAR_CTRL_REG 0x10 + +#ifdef __REG_DEBUG__ +/* SECTION: Function definitions */ +inline void __write_register(const char *fn, u32 value, void *base, + unsigned int off) +{ + pr_info("%s: 0x%p, W reg 0x%lx, 0x%x.\n", fn, base, off, value); + iowrite32(value, base + off); +} + +inline u32 __read_register(const char *fn, void *base, unsigned int off) +{ + u32 v = ioread32(base + off); + + pr_info("%s: 0x%p, R reg 0x%lx, 0x%x.\n", fn, base, off, v); + return v; +} +#define write_register(v,base,off) __write_register(__func__, v, base, off) +#define read_register(base,off) __read_register(__func__, base, off) + +#else +#define write_register(v,base,off) iowrite32(v, (base) + (off)) +#define read_register(base,off) ioread32((base) + (off)) +#endif /* #ifdef __REG_DEBUG__ */ + + +static int xvc_shift_bits(void *base, u32 tms_bits, u32 tdi_bits, + u32 *tdo_bits) +{ + u32 control; + int count; + + /* set tms bit */ + write_register(tms_bits, base, XVC_BAR_TMS_REG); + /* set tdi bits and shift data out */ + write_register(tdi_bits, base, XVC_BAR_TDI_REG); + /* enable shift operation */ + write_register(0x1, base, XVC_BAR_CTRL_REG); + + /* poll for completion */ + count = COMPLETION_LOOP_MAX; + while (count) { + /* read control reg to check shift operation completion */ + control = read_register(base, XVC_BAR_CTRL_REG); + if ((control & 0x01) == 0) + break; + + count--; + } + + if (!count) { + pr_warn("XVC bar transaction timed out (0x%0X)\n", control); + return -ETIMEDOUT; + } + + /* read tdo bits back out */ + *tdo_bits = read_register(base, XVC_BAR_TDO_REG); + + return 0; +} + +static long xvc_ioctl(struct file *filp, unsigned int cmd, unsigned long arg) +{ + struct xdma_cdev *xcdev = (struct xdma_cdev *)filp->private_data; + struct xdma_dev *xdev; + struct xvc_ioc xvc_obj; + unsigned int opcode; + unsigned int total_bits; + unsigned int total_bytes; + unsigned char *buffer = NULL; + unsigned char *tms_buf = NULL; + unsigned char *tdi_buf = NULL; + unsigned char *tdo_buf = NULL; + unsigned int bits, bits_left; + void __iomem *iobase; + int rv; + + rv = xcdev_check(__func__, xcdev, 0); + if (rv < 0) + return rv; + xdev = xcdev->xdev; + + if (cmd != XDMA_IOCXVC) { + pr_info("ioctl 0x%x, UNKNOWN cmd.\n", cmd); + return -ENOIOCTLCMD; + } + + rv = copy_from_user((void *)&xvc_obj, (void __user *)arg, + sizeof(struct xvc_ioc)); + /* anything not copied ? */ + if (rv) { + pr_info("copy_from_user xvc_obj failed: %d.\n", rv); + goto cleanup; + } + + opcode = xvc_obj.opcode; + + /* Invalid operation type, no operation performed */ + if (opcode != 0x01 && opcode != 0x02) { + pr_info("UNKNOWN opcode 0x%x.\n", opcode); + return -EINVAL; + } + + total_bits = xvc_obj.length; + total_bytes = (total_bits + 7) >> 3; + + buffer = (char *)kmalloc(total_bytes * 3, GFP_KERNEL); + if (!buffer) { + pr_info("OOM %u, op 0x%x, len %u bits, %u bytes.\n", + 3 * total_bytes, opcode, total_bits, total_bytes); + rv = -ENOMEM; + goto cleanup; + } + tms_buf = buffer; + tdi_buf = tms_buf + total_bytes; + tdo_buf = tdi_buf + total_bytes; + + rv = copy_from_user((void *)tms_buf, xvc_obj.tms_buf, total_bytes); + if (rv) { + pr_info("copy tmfs_buf failed: %d/%u.\n", rv, total_bytes); + goto cleanup; + } + rv = copy_from_user((void *)tdi_buf, xvc_obj.tdi_buf, total_bytes); + if (rv) { + pr_info("copy tdi_buf failed: %d/%u.\n", rv, total_bytes); + goto cleanup; + } + + /* exclusive access */ + spin_lock(&xcdev->lock); + + iobase = xdev->bar[xcdev->bar] + xcdev->base; + /* set length register to 32 initially if more than one + * word-transaction is to be done */ + if (total_bits >= 32) + write_register(0x20, iobase, XVC_BAR_LENGTH_REG); + + for (bits = 0, bits_left = total_bits; bits < total_bits; bits += 32, + bits_left -= 32) { + unsigned int bytes = bits >> 3; + unsigned int shift_bytes = 4; + u32 tms_store = 0; + u32 tdi_store = 0; + u32 tdo_store = 0; + + if (bits_left < 32) { + /* set number of bits to shift out */ + write_register(bits_left, iobase, XVC_BAR_LENGTH_REG); + shift_bytes = (bits_left + 7) >> 3; + } + + memcpy(&tms_store, tms_buf + bytes, shift_bytes); + memcpy(&tdi_store, tdi_buf + bytes, shift_bytes); + + /* Shift data out and copy to output buffer */ + rv = xvc_shift_bits(iobase, tms_store, tdi_store, &tdo_store); + if (rv < 0) + goto cleanup; + + memcpy(tdo_buf + bytes, &tdo_store, shift_bytes); + } + + /* if testing bar access swap tdi and tdo bufferes to "loopback" */ + if (opcode == 0x2) { + char *tmp = tdo_buf; + + tdo_buf = tdi_buf; + tdi_buf = tmp; + } + + rv = copy_to_user((void *)xvc_obj.tdo_buf, tdo_buf, total_bytes); + if (rv) { + pr_info("copy back tdo_buf failed: %d/%u.\n", rv, total_bytes); + rv = -EFAULT; + goto cleanup; + } + +cleanup: + if (buffer) + kfree(buffer); + + #if LINUX_VERSION_CODE < KERNEL_VERSION(5,3,0) + mmiowb(); + #endif + + spin_unlock(&xcdev->lock); + + return rv; +} + +/* + * character device file operations for the XVC + */ +static const struct file_operations xvc_fops = { + .owner = THIS_MODULE, + .open = char_open, + .release = char_close, + .unlocked_ioctl = xvc_ioctl, +}; + +void cdev_xvc_init(struct xdma_cdev *xcdev) +{ +#ifdef __XVC_BAR_NUM__ + xcdev->bar = __XVC_BAR_NUM__; +#endif +#ifdef __XVC_BAR_OFFSET__ + xcdev->base = __XVC_BAR_OFFSET__; +#else + xcdev->base = XVC_BAR_OFFSET_DFLT; +#endif + pr_info("xcdev 0x%p, bar %u, offset 0x%lx.\n", + xcdev, xcdev->bar, xcdev->base); + cdev_init(&xcdev->cdev, &xvc_fops); +} diff --git a/sources/xdma_driver/xdma/cdev_xvc.h b/sources/xdma_driver/xdma/cdev_xvc.h new file mode 100644 index 0000000..e689706 --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_xvc.h @@ -0,0 +1,36 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XVC_IOCTL_H__ +#define __XVC_IOCTL_H__ + +#include <linux/ioctl.h> + +/* + * !!! TODO !!! + * need a better way set the bar offset dynamicly + */ +#define XVC_BAR_OFFSET_DFLT 0x40000 /* DSA 4.0 */ + +#define XVC_MAGIC 0x58564344 // "XVCD" + +struct xvc_ioc { + unsigned int opcode; + unsigned int length; + unsigned char *tms_buf; + unsigned char *tdi_buf; + unsigned char *tdo_buf; +}; + +#define XDMA_IOCXVC _IOWR(XVC_MAGIC, 1, struct xvc_ioc) + +#endif /* __XVC_IOCTL_H__ */ diff --git a/sources/xdma_driver/xdma/cdev_xvc.o.ur-safe b/sources/xdma_driver/xdma/cdev_xvc.o.ur-safe new file mode 100644 index 0000000..3a9a009 --- /dev/null +++ b/sources/xdma_driver/xdma/cdev_xvc.o.ur-safe @@ -0,0 +1 @@ +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/cdev_xvc.o-.text-315 diff --git a/sources/xdma_driver/xdma/libxdma.c b/sources/xdma_driver/xdma/libxdma.c new file mode 100644 index 0000000..1743780 --- /dev/null +++ b/sources/xdma_driver/xdma/libxdma.c @@ -0,0 +1,4445 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include <linux/module.h> +#include <linux/kernel.h> +#include <linux/string.h> +#include <linux/mm.h> +#include <linux/errno.h> +#include <linux/sched.h> +#include <linux/vmalloc.h> + +#include "libxdma.h" +#include "libxdma_api.h" +#include "cdev_sgdma.h" + +/* SECTION: Module licensing */ + +#ifdef __LIBXDMA_MOD__ +#include "version.h" +#define DRV_MODULE_NAME "libxdma" +#define DRV_MODULE_DESC "Xilinx XDMA Base Driver" +#define DRV_MODULE_RELDATE "Feb. 2018" + +static char version[] = + DRV_MODULE_DESC " " DRV_MODULE_NAME " v" DRV_MODULE_VERSION "\n"; + +MODULE_AUTHOR("Xilinx, Inc."); +MODULE_DESCRIPTION(DRV_MODULE_DESC); +MODULE_VERSION(DRV_MODULE_VERSION); +MODULE_LICENSE("Dual BSD/GPL"); +#endif + +/* Module Parameters */ +static unsigned int poll_mode; +module_param(poll_mode, uint, 0644); +MODULE_PARM_DESC(poll_mode, "Set 1 for hw polling, default is 0 (interrupts)"); + +static unsigned int interrupt_mode; +module_param(interrupt_mode, uint, 0644); +MODULE_PARM_DESC(interrupt_mode, "0 - MSI-x , 1 - MSI, 2 - Legacy"); + +static unsigned int enable_credit_mp; +module_param(enable_credit_mp, uint, 0644); +MODULE_PARM_DESC(enable_credit_mp, "Set 1 to enable creidt feature, default is 0 (no credit control)"); + +unsigned int desc_blen_max = XDMA_DESC_BLEN_MAX; +module_param(desc_blen_max, uint, 0644); +MODULE_PARM_DESC(desc_blen_max, "per descriptor max. buffer length, default is (1 << 28) - 1"); + +/* + * xdma device management + * maintains a list of the xdma devices + */ +static LIST_HEAD(xdev_list); +static DEFINE_MUTEX(xdev_mutex); + +static LIST_HEAD(xdev_rcu_list); +static DEFINE_SPINLOCK(xdev_rcu_lock); + +#ifndef list_last_entry +#define list_last_entry(ptr, type, member) \ + list_entry((ptr)->prev, type, member) +#endif + +static inline void xdev_list_add(struct xdma_dev *xdev) +{ + mutex_lock(&xdev_mutex); + if (list_empty(&xdev_list)) + xdev->idx = 0; + else { + struct xdma_dev *last; + + last = list_last_entry(&xdev_list, struct xdma_dev, list_head); + xdev->idx = last->idx + 1; + } + list_add_tail(&xdev->list_head, &xdev_list); + mutex_unlock(&xdev_mutex); + + dbg_init("dev %s, xdev 0x%p, xdma idx %d.\n", + dev_name(&xdev->pdev->dev), xdev, xdev->idx); + + spin_lock(&xdev_rcu_lock); + list_add_tail_rcu(&xdev->rcu_node, &xdev_rcu_list); + spin_unlock(&xdev_rcu_lock); +} + +#undef list_last_entry + +static inline void xdev_list_remove(struct xdma_dev *xdev) +{ + mutex_lock(&xdev_mutex); + list_del(&xdev->list_head); + mutex_unlock(&xdev_mutex); + + spin_lock(&xdev_rcu_lock); + list_del_rcu(&xdev->rcu_node); + spin_unlock(&xdev_rcu_lock); + synchronize_rcu(); +} + +struct xdma_dev *xdev_find_by_pdev(struct pci_dev *pdev) +{ + struct xdma_dev *xdev, *tmp; + + mutex_lock(&xdev_mutex); + list_for_each_entry_safe(xdev, tmp, &xdev_list, list_head) { + if (xdev->pdev == pdev) { + mutex_unlock(&xdev_mutex); + return xdev; + } + } + mutex_unlock(&xdev_mutex); + return NULL; +} +EXPORT_SYMBOL_GPL(xdev_find_by_pdev); + +static inline int debug_check_dev_hndl(const char *fname, struct pci_dev *pdev, + void *hndl) +{ + struct xdma_dev *xdev; + + if (!pdev) + return -EINVAL; + + xdev = xdev_find_by_pdev(pdev); + if (!xdev) { + pr_info("%s pdev 0x%p, hndl 0x%p, NO match found!\n", + fname, pdev, hndl); + return -EINVAL; + } + if (xdev != hndl) { + pr_err("%s pdev 0x%p, hndl 0x%p != 0x%p!\n", + fname, pdev, hndl, xdev); + return -EINVAL; + } + + return 0; +} + +#ifdef __LIBXDMA_DEBUG__ +/* SECTION: Function definitions */ +inline void __write_register(const char *fn, u32 value, void *iomem, unsigned long off) +{ + pr_err("%s: w reg 0x%lx(0x%p), 0x%x.\n", fn, off, iomem, value); + iowrite32(value, iomem); +} +#define write_register(v,mem,off) __write_register(__func__, v, mem, off) +#else +#define write_register(v,mem,off) iowrite32(v, mem) +#endif + +inline u32 read_register(void *iomem) +{ + return ioread32(iomem); +} + +static inline u32 build_u32(u32 hi, u32 lo) +{ + return ((hi & 0xFFFFUL) << 16) | (lo & 0xFFFFUL); +} + +static inline u64 build_u64(u64 hi, u64 lo) +{ + return ((hi & 0xFFFFFFFULL) << 32) | (lo & 0xFFFFFFFFULL); +} + +static void check_nonzero_interrupt_status(struct xdma_dev *xdev) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + u32 w; + + w = read_register(®->user_int_enable); + if (w) + pr_info("%s xdma%d user_int_enable = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->channel_int_enable); + if (w) + pr_info("%s xdma%d channel_int_enable = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->user_int_request); + if (w) + pr_info("%s xdma%d user_int_request = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + w = read_register(®->channel_int_request); + if (w) + pr_info("%s xdma%d channel_int_request = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + + w = read_register(®->user_int_pending); + if (w) + pr_info("%s xdma%d user_int_pending = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); + w = read_register(®->channel_int_pending); + if (w) + pr_info("%s xdma%d channel_int_pending = 0x%08x\n", + dev_name(&xdev->pdev->dev), xdev->idx, w); +} + +/* channel_interrupts_enable -- Enable interrupts we are interested in */ +static void channel_interrupts_enable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->channel_int_enable_w1s, XDMA_OFS_INT_CTRL); +} + +/* channel_interrupts_disable -- Disable interrupts we not interested in */ +static void channel_interrupts_disable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->channel_int_enable_w1c, XDMA_OFS_INT_CTRL); +} + +/* user_interrupts_enable -- Enable interrupts we are interested in */ +static void user_interrupts_enable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->user_int_enable_w1s, XDMA_OFS_INT_CTRL); +} + +/* user_interrupts_disable -- Disable interrupts we not interested in */ +static void user_interrupts_disable(struct xdma_dev *xdev, u32 mask) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + + write_register(mask, ®->user_int_enable_w1c, XDMA_OFS_INT_CTRL); +} + +/* read_interrupts -- Print the interrupt controller status */ +static u32 read_interrupts(struct xdma_dev *xdev) +{ + struct interrupt_regs *reg = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + XDMA_OFS_INT_CTRL); + u32 lo; + u32 hi; + + /* extra debugging; inspect complete engine set of registers */ + hi = read_register(®->user_int_request); + dbg_io("ioread32(0x%p) returned 0x%08x (user_int_request).\n", + ®->user_int_request, hi); + lo = read_register(®->channel_int_request); + dbg_io("ioread32(0x%p) returned 0x%08x (channel_int_request)\n", + ®->channel_int_request, lo); + + /* return interrupts: user in upper 16-bits, channel in lower 16-bits */ + return build_u32(hi, lo); +} + +void enable_perf(struct xdma_engine *engine) +{ + u32 w; + + w = XDMA_PERF_CLEAR; + write_register(w, &engine->regs->perf_ctrl, + (unsigned long)(&engine->regs->perf_ctrl) - + (unsigned long)(&engine->regs)); + read_register(&engine->regs->identifier); + w = XDMA_PERF_AUTO | XDMA_PERF_RUN; + write_register(w, &engine->regs->perf_ctrl, + (unsigned long)(&engine->regs->perf_ctrl) - + (unsigned long)(&engine->regs)); + read_register(&engine->regs->identifier); + + dbg_perf("IOCTL_XDMA_PERF_START\n"); + +} +EXPORT_SYMBOL_GPL(enable_perf); + +void get_perf_stats(struct xdma_engine *engine) +{ + u32 hi; + u32 lo; + + BUG_ON(!engine); + + if (!engine->xdma_perf) { + pr_info("%s perf struct not set up.\n", engine->name); + return; + } + + hi = 0; + lo = read_register(&engine->regs->completed_desc_count); + engine->xdma_perf->iterations = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_cyc_hi); + lo = read_register(&engine->regs->perf_cyc_lo); + + engine->xdma_perf->clock_cycle_count = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_dat_hi); + lo = read_register(&engine->regs->perf_dat_lo); + engine->xdma_perf->data_cycle_count = build_u64(hi, lo); + + hi = read_register(&engine->regs->perf_pnd_hi); + lo = read_register(&engine->regs->perf_pnd_lo); + engine->xdma_perf->pending_count = build_u64(hi, lo); +} +EXPORT_SYMBOL_GPL(get_perf_stats); + +static void engine_reg_dump(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + + w = read_register(&engine->regs->identifier); + pr_info("%s: ioread32(0x%p) = 0x%08x (id).\n", + engine->name, &engine->regs->identifier, w); + w &= BLOCK_ID_MASK; + if (w != BLOCK_ID_HEAD) { + pr_info("%s: engine id missing, 0x%08x exp. & 0x%x = 0x%x\n", + engine->name, w, BLOCK_ID_MASK, BLOCK_ID_HEAD); + return; + } + /* extra debugging; inspect complete engine set of registers */ + w = read_register(&engine->regs->status); + pr_info("%s: ioread32(0x%p) = 0x%08x (status).\n", + engine->name, &engine->regs->status, w); + w = read_register(&engine->regs->control); + pr_info("%s: ioread32(0x%p) = 0x%08x (control)\n", + engine->name, &engine->regs->control, w); + w = read_register(&engine->sgdma_regs->first_desc_lo); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_lo)\n", + engine->name, &engine->sgdma_regs->first_desc_lo, w); + w = read_register(&engine->sgdma_regs->first_desc_hi); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_hi)\n", + engine->name, &engine->sgdma_regs->first_desc_hi, w); + w = read_register(&engine->sgdma_regs->first_desc_adjacent); + pr_info("%s: ioread32(0x%p) = 0x%08x (first_desc_adjacent).\n", + engine->name, &engine->sgdma_regs->first_desc_adjacent, w); + w = read_register(&engine->regs->completed_desc_count); + pr_info("%s: ioread32(0x%p) = 0x%08x (completed_desc_count).\n", + engine->name, &engine->regs->completed_desc_count, w); + w = read_register(&engine->regs->interrupt_enable_mask); + pr_info("%s: ioread32(0x%p) = 0x%08x (interrupt_enable_mask)\n", + engine->name, &engine->regs->interrupt_enable_mask, w); +} + +/** + * engine_status_read() - read status of SG DMA engine (optionally reset) + * + * Stores status in engine->status. + * + * @return -1 on failure, status register otherwise + */ +static void engine_status_dump(struct xdma_engine *engine) +{ + u32 v = engine->status; + char buffer[256]; + char *buf = buffer; + int len = 0; + + len = sprintf(buf, "SG engine %s status: 0x%08x: ", engine->name, v); + + if ((v & XDMA_STAT_BUSY)) + len += sprintf(buf + len, "BUSY,"); + if ((v & XDMA_STAT_DESC_STOPPED)) + len += sprintf(buf + len, "DESC_STOPPED,"); + if ((v & XDMA_STAT_DESC_COMPLETED)) + len += sprintf(buf + len, "DESC_COMPL,"); + + /* common H2C & C2H */ + if ((v & XDMA_STAT_COMMON_ERR_MASK)) { + if ((v & XDMA_STAT_ALIGN_MISMATCH)) + len += sprintf(buf + len, "ALIGN_MISMATCH "); + if ((v & XDMA_STAT_MAGIC_STOPPED)) + len += sprintf(buf + len, "MAGIC_STOPPED "); + if ((v & XDMA_STAT_INVALID_LEN)) + len += sprintf(buf + len, "INVLIAD_LEN "); + if ((v & XDMA_STAT_IDLE_STOPPED)) + len += sprintf(buf + len, "IDLE_STOPPED "); + buf[len - 1] = ','; + } + + if ((engine->dir == DMA_TO_DEVICE)) { + /* H2C only */ + if ((v & XDMA_STAT_H2C_R_ERR_MASK)) { + len += sprintf(buf + len, "R:"); + if ((v & XDMA_STAT_H2C_R_UNSUPP_REQ)) + len += sprintf(buf + len, "UNSUPP_REQ "); + if ((v & XDMA_STAT_H2C_R_COMPL_ABORT)) + len += sprintf(buf + len, "COMPL_ABORT "); + if ((v & XDMA_STAT_H2C_R_PARITY_ERR)) + len += sprintf(buf + len, "PARITY "); + if ((v & XDMA_STAT_H2C_R_HEADER_EP)) + len += sprintf(buf + len, "HEADER_EP "); + if ((v & XDMA_STAT_H2C_R_UNEXP_COMPL)) + len += sprintf(buf + len, "UNEXP_COMPL "); + buf[len - 1] = ','; + } + + if ((v & XDMA_STAT_H2C_W_ERR_MASK)) { + len += sprintf(buf + len, "W:"); + if ((v & XDMA_STAT_H2C_W_DECODE_ERR)) + len += sprintf(buf + len, "DECODE_ERR "); + if ((v & XDMA_STAT_H2C_W_SLAVE_ERR)) + len += sprintf(buf + len, "SLAVE_ERR "); + buf[len - 1] = ','; + } + + } else { + /* C2H only */ + if ((v & XDMA_STAT_C2H_R_ERR_MASK)) { + len += sprintf(buf + len, "R:"); + if ((v & XDMA_STAT_C2H_R_DECODE_ERR)) + len += sprintf(buf + len, "DECODE_ERR "); + if ((v & XDMA_STAT_C2H_R_SLAVE_ERR)) + len += sprintf(buf + len, "SLAVE_ERR "); + buf[len - 1] = ','; + } + } + + /* common H2C & C2H */ + if ((v & XDMA_STAT_DESC_ERR_MASK)) { + len += sprintf(buf + len, "DESC_ERR:"); + if ((v & XDMA_STAT_DESC_UNSUPP_REQ)) + len += sprintf(buf + len, "UNSUPP_REQ "); + if ((v & XDMA_STAT_DESC_COMPL_ABORT)) + len += sprintf(buf + len, "COMPL_ABORT "); + if ((v & XDMA_STAT_DESC_PARITY_ERR)) + len += sprintf(buf + len, "PARITY "); + if ((v & XDMA_STAT_DESC_HEADER_EP)) + len += sprintf(buf + len, "HEADER_EP "); + if ((v & XDMA_STAT_DESC_UNEXP_COMPL)) + len += sprintf(buf + len, "UNEXP_COMPL "); + buf[len - 1] = ','; + } + + buf[len - 1] = '\0'; + pr_info("%s\n", buffer); +} + +static u32 engine_status_read(struct xdma_engine *engine, bool clear, bool dump) +{ + u32 value; + + BUG_ON(!engine); + + if (dump) + engine_reg_dump(engine); + + /* read status register */ + if (clear) + value = engine->status = + read_register(&engine->regs->status_rc); + else + value = engine->status = read_register(&engine->regs->status); + + if (dump) + engine_status_dump(engine); + + return value; +} + +/** + * xdma_engine_stop() - stop an SG DMA engine + * + */ +static void xdma_engine_stop(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + dbg_tfr("xdma_engine_stop(engine=%p)\n", engine); + + w = 0; + w |= (u32)XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= (u32)XDMA_CTRL_IE_MAGIC_STOPPED; + w |= (u32)XDMA_CTRL_IE_READ_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ERROR; + + if (poll_mode) { + w |= (u32) XDMA_CTRL_POLL_MODE_WB; + } else { + w |= (u32)XDMA_CTRL_IE_DESC_STOPPED; + w |= (u32)XDMA_CTRL_IE_DESC_COMPLETED; + + /* Disable IDLE STOPPED for MM */ + if ((engine->streaming && (engine->dir == DMA_FROM_DEVICE)) || + (engine->xdma_perf)) + w |= (u32)XDMA_CTRL_IE_IDLE_STOPPED; + } + + dbg_tfr("Stopping SG DMA %s engine; writing 0x%08x to 0x%p.\n", + engine->name, w, (u32 *)&engine->regs->control); + write_register(w, &engine->regs->control, + (unsigned long)(&engine->regs->control) - + (unsigned long)(&engine->regs)); + /* dummy read of status register to flush all previous writes */ + dbg_tfr("xdma_engine_stop(%s) done\n", engine->name); +} + +static void engine_start_mode_config(struct xdma_engine *engine) +{ + u32 w; + + BUG_ON(!engine); + + /* If a perf test is running, enable the engine interrupts */ + if (engine->xdma_perf) { + w = XDMA_CTRL_IE_DESC_STOPPED; + w |= XDMA_CTRL_IE_DESC_COMPLETED; + w |= XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= XDMA_CTRL_IE_MAGIC_STOPPED; + w |= XDMA_CTRL_IE_IDLE_STOPPED; + w |= XDMA_CTRL_IE_READ_ERROR; + w |= XDMA_CTRL_IE_DESC_ERROR; + + write_register(w, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + } + + /* write control register of SG DMA engine */ + w = (u32)XDMA_CTRL_RUN_STOP; + w |= (u32)XDMA_CTRL_IE_READ_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ERROR; + w |= (u32)XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + w |= (u32)XDMA_CTRL_IE_MAGIC_STOPPED; + + if (poll_mode) { + w |= (u32)XDMA_CTRL_POLL_MODE_WB; + } else { + w |= (u32)XDMA_CTRL_IE_DESC_STOPPED; + w |= (u32)XDMA_CTRL_IE_DESC_COMPLETED; + + if ((engine->streaming && (engine->dir == DMA_FROM_DEVICE)) || + (engine->xdma_perf)) + w |= (u32)XDMA_CTRL_IE_IDLE_STOPPED; + + /* set non-incremental addressing mode */ + if (engine->non_incr_addr) + w |= (u32)XDMA_CTRL_NON_INCR_ADDR; + } + + dbg_tfr("iowrite32(0x%08x to 0x%p) (control)\n", w, + (void *)&engine->regs->control); + /* start the engine */ + write_register(w, &engine->regs->control, + (unsigned long)(&engine->regs->control) - + (unsigned long)(&engine->regs)); + + /* dummy read of status register to flush all previous writes */ + w = read_register(&engine->regs->status); + dbg_tfr("ioread32(0x%p) = 0x%08x (dummy read flushes writes).\n", + &engine->regs->status, w); +} + +/** + * engine_start() - start an idle engine with its first transfer on queue + * + * The engine will run and process all transfers that are queued using + * transfer_queue() and thus have their descriptor lists chained. + * + * During the run, new transfers will be processed if transfer_queue() has + * chained the descriptors before the hardware fetches the last descriptor. + * A transfer that was chained too late will invoke a new run of the engine + * initiated from the engine_service() routine. + * + * The engine must be idle and at least one transfer must be queued. + * This function does not take locks; the engine spinlock must already be + * taken. + * + */ +static struct xdma_transfer *engine_start(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer; + u32 w; + int extra_adj = 0; + + /* engine must be idle */ + BUG_ON(engine->running); + /* engine transfer queue must not be empty */ + BUG_ON(list_empty(&engine->transfer_list)); + /* inspect first transfer queued on the engine */ + transfer = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + BUG_ON(!transfer); + + /* engine is no longer shutdown */ + engine->shutdown = ENGINE_SHUTDOWN_NONE; + + dbg_tfr("engine_start(%s): transfer=0x%p.\n", engine->name, transfer); + + /* initialize number of descriptors of dequeued transfers */ + engine->desc_dequeued = 0; + + /* write lower 32-bit of bus address of transfer first descriptor */ + w = cpu_to_le32(PCI_DMA_L(transfer->desc_bus)); + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_lo)\n", w, + (void *)&engine->sgdma_regs->first_desc_lo); + write_register(w, &engine->sgdma_regs->first_desc_lo, + (unsigned long)(&engine->sgdma_regs->first_desc_lo) - + (unsigned long)(&engine->sgdma_regs)); + /* write upper 32-bit of bus address of transfer first descriptor */ + w = cpu_to_le32(PCI_DMA_H(transfer->desc_bus)); + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_hi)\n", w, + (void *)&engine->sgdma_regs->first_desc_hi); + write_register(w, &engine->sgdma_regs->first_desc_hi, + (unsigned long)(&engine->sgdma_regs->first_desc_hi) - + (unsigned long)(&engine->sgdma_regs)); + + if (transfer->desc_adjacent > 0) { + extra_adj = transfer->desc_adjacent - 1; + if (extra_adj > MAX_EXTRA_ADJ) + extra_adj = MAX_EXTRA_ADJ; + } + dbg_tfr("iowrite32(0x%08x to 0x%p) (first_desc_adjacent)\n", + extra_adj, (void *)&engine->sgdma_regs->first_desc_adjacent); + write_register(extra_adj, &engine->sgdma_regs->first_desc_adjacent, + (unsigned long)(&engine->sgdma_regs->first_desc_adjacent) - + (unsigned long)(&engine->sgdma_regs)); + + dbg_tfr("ioread32(0x%p) (dummy read flushes writes).\n", + &engine->regs->status); + + #if LINUX_VERSION_CODE < KERNEL_VERSION(5,3,0) + mmiowb(); + #endif + + engine_start_mode_config(engine); + + engine_status_read(engine, 0, 0); + + dbg_tfr("%s engine 0x%p now running\n", engine->name, engine); + /* remember the engine is running */ + engine->running = 1; + return transfer; +} + +/** + * engine_service() - service an SG DMA engine + * + * must be called with engine->lock already acquired + * + * @engine pointer to struct xdma_engine + * + */ +static void engine_service_shutdown(struct xdma_engine *engine) +{ + /* if the engine stopped with RUN still asserted, de-assert RUN now */ + dbg_tfr("engine just went idle, resetting RUN_STOP.\n"); + xdma_engine_stop(engine); + engine->running = 0; + + /* awake task on engine's shutdown wait queue */ + wake_up_interruptible(&engine->shutdown_wq); +} + +struct xdma_transfer *engine_transfer_completion(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + BUG_ON(!engine); + + if (unlikely(!transfer)) { + pr_info("%s: xfer empty.\n", engine->name); + return NULL; + } + + /* synchronous I/O? */ + /* awake task on transfer's wait queue */ + wake_up_interruptible(&transfer->wq); + + return transfer; +} + +struct xdma_transfer *engine_service_transfer_list(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 *pdesc_completed) +{ + BUG_ON(!engine); + BUG_ON(!pdesc_completed); + + if (unlikely(!transfer)) { + pr_info("%s xfer empty, pdesc completed %u.\n", + engine->name, *pdesc_completed); + return NULL; + } + + /* + * iterate over all the transfers completed by the engine, + * except for the last (i.e. use > instead of >=). + */ + while (transfer && (!transfer->cyclic) && + (*pdesc_completed > transfer->desc_num)) { + /* remove this transfer from pdesc_completed */ + *pdesc_completed -= transfer->desc_num; + dbg_tfr("%s engine completed non-cyclic xfer 0x%p (%d desc)\n", + engine->name, transfer, transfer->desc_num); + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); + /* add to dequeued number of descriptors during this run */ + engine->desc_dequeued += transfer->desc_num; + /* mark transfer as succesfully completed */ + transfer->state = TRANSFER_STATE_COMPLETED; + + /* Complete transfer - sets transfer to NULL if an async + * transfer has completed */ + transfer = engine_transfer_completion(engine, transfer); + + /* if exists, get the next transfer on the list */ + if (!list_empty(&engine->transfer_list)) { + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + dbg_tfr("Non-completed transfer %p\n", transfer); + } else { + /* no further transfers? */ + transfer = NULL; + } + } + + return transfer; +} + +static void engine_err_handle(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 desc_completed) +{ + u32 value; + + /* + * The BUSY bit is expected to be clear now but older HW has a race + * condition which could cause it to be still set. If it's set, re-read + * and check again. If it's still set, log the issue. + */ + if (engine->status & XDMA_STAT_BUSY) { + value = read_register(&engine->regs->status); + if ((value & XDMA_STAT_BUSY) && printk_ratelimit()) + pr_info("%s has errors but is still BUSY\n", + engine->name); + } + + if (printk_ratelimit()) { + pr_info("%s, s 0x%x, aborted xfer 0x%p, cmpl %d/%d\n", + engine->name, engine->status, transfer, desc_completed, + transfer->desc_num); + } + + /* mark transfer as failed */ + transfer->state = TRANSFER_STATE_FAILED; + xdma_engine_stop(engine); +} + +struct xdma_transfer *engine_service_final_transfer(struct xdma_engine *engine, + struct xdma_transfer *transfer, u32 *pdesc_completed) +{ + BUG_ON(!engine); + BUG_ON(!pdesc_completed); + + /* inspect the current transfer */ + if (unlikely(!transfer)) { + pr_info("%s xfer empty, pdesc completed %u.\n", + engine->name, *pdesc_completed); + return NULL; + } else { + if (((engine->dir == DMA_FROM_DEVICE) && + (engine->status & XDMA_STAT_C2H_ERR_MASK)) || + ((engine->dir == DMA_TO_DEVICE) && + (engine->status & XDMA_STAT_H2C_ERR_MASK))) { + pr_info("engine %s, status error 0x%x.\n", + engine->name, engine->status); + engine_status_dump(engine); + engine_err_handle(engine, transfer, *pdesc_completed); + goto transfer_del; + } + + if (engine->status & XDMA_STAT_BUSY) + pr_debug("engine %s is unexpectedly busy - ignoring\n", + engine->name); + + /* the engine stopped on current transfer? */ + if (*pdesc_completed < transfer->desc_num) { + transfer->state = TRANSFER_STATE_FAILED; + pr_info("%s, xfer 0x%p, stopped half-way, %d/%d.\n", + engine->name, transfer, *pdesc_completed, + transfer->desc_num); + } else { + dbg_tfr("engine %s completed transfer\n", engine->name); + dbg_tfr("Completed transfer ID = 0x%p\n", transfer); + dbg_tfr("*pdesc_completed=%d, transfer->desc_num=%d", + *pdesc_completed, transfer->desc_num); + + if (!transfer->cyclic) { + /* + * if the engine stopped on this transfer, + * it should be the last + */ + WARN_ON(*pdesc_completed > transfer->desc_num); + } + /* mark transfer as succesfully completed */ + transfer->state = TRANSFER_STATE_COMPLETED; + } + +transfer_del: + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); + /* add to dequeued number of descriptors during this run */ + engine->desc_dequeued += transfer->desc_num; + + /* + * Complete transfer - sets transfer to NULL if an asynchronous + * transfer has completed + */ + transfer = engine_transfer_completion(engine, transfer); + } + + return transfer; +} + +static void engine_service_perf(struct xdma_engine *engine, u32 desc_completed) +{ + BUG_ON(!engine); + + /* performance measurement is running? */ + if (engine->xdma_perf) { + /* a descriptor was completed? */ + if (engine->status & XDMA_STAT_DESC_COMPLETED) { + engine->xdma_perf->iterations = desc_completed; + dbg_perf("transfer->xdma_perf->iterations=%d\n", + engine->xdma_perf->iterations); + } + + /* a descriptor stopped the engine? */ + if (engine->status & XDMA_STAT_DESC_STOPPED) { + engine->xdma_perf->stopped = 1; + /* + * wake any XDMA_PERF_IOCTL_STOP waiting for + * the performance run to finish + */ + wake_up_interruptible(&engine->xdma_perf_wq); + dbg_perf("transfer->xdma_perf stopped\n"); + } + } +} + +static void engine_transfer_dequeue(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer; + + BUG_ON(!engine); + + /* pick first transfer on the queue (was submitted to the engine) */ + transfer = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + if (!transfer || transfer != &engine->cyclic_req->xfer) { + pr_info("%s, xfer 0x%p != 0x%p.\n", + engine->name, transfer, &engine->cyclic_req->xfer); + return; + } + dbg_tfr("%s engine completed cyclic transfer 0x%p (%d desc).\n", + engine->name, transfer, transfer->desc_num); + /* remove completed transfer from list */ + list_del(engine->transfer_list.next); +} + +static int engine_ring_process(struct xdma_engine *engine) +{ + struct xdma_result *result; + int start; + int eop_count = 0; + + BUG_ON(!engine); + result = engine->cyclic_result; + BUG_ON(!result); + + /* where we start receiving in the ring buffer */ + start = engine->rx_tail; + + /* iterate through all newly received RX result descriptors */ + dbg_tfr("%s, result %d, 0x%x, len 0x%x.\n", + engine->name, engine->rx_tail, result[engine->rx_tail].status, + result[engine->rx_tail].length); + while (result[engine->rx_tail].status && !engine->rx_overrun) { + /* EOP bit set in result? */ + if (result[engine->rx_tail].status & RX_STATUS_EOP){ + eop_count++; + } + + /* increment tail pointer */ + engine->rx_tail = (engine->rx_tail + 1) % CYCLIC_RX_PAGES_MAX; + + dbg_tfr("%s, head %d, tail %d, 0x%x, len 0x%x.\n", + engine->name, engine->rx_head, engine->rx_tail, + result[engine->rx_tail].status, + result[engine->rx_tail].length); + + /* overrun? */ + if (engine->rx_tail == engine->rx_head) { + dbg_tfr("%s: overrun\n", engine->name); + /* flag to user space that overrun has occurred */ + engine->rx_overrun = 1; + } + } + + return eop_count; +} + +static int engine_service_cyclic_polled(struct xdma_engine *engine) +{ + int eop_count = 0; + int rc = 0; + struct xdma_poll_wb *writeback_data; + u32 sched_limit = 0; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + writeback_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + while (eop_count == 0) { + if (sched_limit != 0) { + if ((sched_limit % NUM_POLLS_PER_SCHED) == 0) + schedule(); + } + sched_limit++; + + /* Monitor descriptor writeback address for errors */ + if ((writeback_data->completed_desc_count) & WB_ERR_MASK) { + rc = -1; + break; + } + + eop_count = engine_ring_process(engine); + } + + if (eop_count == 0) { + engine_status_read(engine, 1, 0); + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + } + + return rc; +} + +static int engine_service_cyclic_interrupt(struct xdma_engine *engine) +{ + int eop_count = 0; + struct xdma_transfer *xfer; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + engine_status_read(engine, 1, 0); + + eop_count = engine_ring_process(engine); + /* + * wake any reader on EOP, as one or more packets are now in + * the RX buffer + */ + xfer = &engine->cyclic_req->xfer; + if(enable_credit_mp){ + if (eop_count > 0) { + //engine->eop_found = 1; + } + wake_up_interruptible(&xfer->wq); + }else{ + if (eop_count > 0) { + /* awake task on transfer's wait queue */ + dbg_tfr("wake_up_interruptible() due to %d EOP's\n", eop_count); + engine->eop_found = 1; + wake_up_interruptible(&xfer->wq); + } + } + + /* engine was running but is no longer busy? */ + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + + return 0; +} + +/* must be called with engine->lock already acquired */ +static int engine_service_cyclic(struct xdma_engine *engine) +{ + int rc = 0; + + dbg_tfr("engine_service_cyclic()"); + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + if (poll_mode) + rc = engine_service_cyclic_polled(engine); + else + rc = engine_service_cyclic_interrupt(engine); + + return rc; +} + + +static void engine_service_resume(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer_started; + + BUG_ON(!engine); + + /* engine stopped? */ + if (!engine->running) { + /* in the case of shutdown, let it finish what's in the Q */ + if (!list_empty(&engine->transfer_list)) { + /* (re)start engine */ + transfer_started = engine_start(engine); + pr_info("re-started %s engine with pending xfer 0x%p\n", + engine->name, transfer_started); + /* engine was requested to be shutdown? */ + } else if (engine->shutdown & ENGINE_SHUTDOWN_REQUEST) { + engine->shutdown |= ENGINE_SHUTDOWN_IDLE; + /* awake task on engine's shutdown wait queue */ + wake_up_interruptible(&engine->shutdown_wq); + } else { + dbg_tfr("no pending transfers, %s engine stays idle.\n", + engine->name); + } + } else { + /* engine is still running? */ + if (list_empty(&engine->transfer_list)) { + pr_warn("no queued transfers but %s engine running!\n", + engine->name); + WARN_ON(1); + } + } +} + +/** + * engine_service() - service an SG DMA engine + * + * must be called with engine->lock already acquired + * + * @engine pointer to struct xdma_engine + * + */ +static int engine_service(struct xdma_engine *engine, int desc_writeback) +{ + struct xdma_transfer *transfer = NULL; + u32 desc_count = desc_writeback & WB_COUNT_MASK; + u32 err_flag = desc_writeback & WB_ERR_MASK; + int rv = 0; + struct xdma_poll_wb *wb_data; + + BUG_ON(!engine); + + /* If polling detected an error, signal to the caller */ + if (err_flag) + rv = -1; + + /* Service the engine */ + if (!engine->running) { + dbg_tfr("Engine was not running!!! Clearing status\n"); + engine_status_read(engine, 1, 0); + return 0; + } + + /* + * If called by the ISR or polling detected an error, read and clear + * engine status. For polled mode descriptor completion, this read is + * unnecessary and is skipped to reduce latency + */ + if ((desc_count == 0) || (err_flag != 0)) + engine_status_read(engine, 1, 0); + + /* + * engine was running but is no longer busy, or writeback occurred, + * shut down + */ + if ((engine->running && !(engine->status & XDMA_STAT_BUSY)) || + (desc_count != 0)) + engine_service_shutdown(engine); + + /* + * If called from the ISR, or if an error occurred, the descriptor + * count will be zero. In this scenario, read the descriptor count + * from HW. In polled mode descriptor completion, this read is + * unnecessary and is skipped to reduce latency + */ + if (!desc_count) + desc_count = read_register(&engine->regs->completed_desc_count); + dbg_tfr("desc_count = %d\n", desc_count); + + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) { + /* pick first transfer on queue (was submitted to the engine) */ + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + + dbg_tfr("head of queue transfer 0x%p has %d descriptors\n", + transfer, (int)transfer->desc_num); + + dbg_tfr("Engine completed %d desc, %d not yet dequeued\n", + (int)desc_count, + (int)desc_count - engine->desc_dequeued); + + engine_service_perf(engine, desc_count); + } + + /* account for already dequeued transfers during this engine run */ + desc_count -= engine->desc_dequeued; + + /* Process all but the last transfer */ + transfer = engine_service_transfer_list(engine, transfer, &desc_count); + + /* + * Process final transfer - includes checks of number of descriptors to + * detect faulty completion + */ + transfer = engine_service_final_transfer(engine, transfer, &desc_count); + + /* Before starting engine again, clear the writeback data */ + if (poll_mode) { + wb_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + wb_data->completed_desc_count = 0; + } + + /* Restart the engine following the servicing */ + engine_service_resume(engine); + + return 0; +} + +/* engine_service_work */ +static void engine_service_work(struct work_struct *work) +{ + struct xdma_engine *engine; + unsigned long flags; + + engine = container_of(work, struct xdma_engine, work); + BUG_ON(engine->magic != MAGIC_ENGINE); + + /* lock the engine */ + spin_lock_irqsave(&engine->lock, flags); + + dbg_tfr("engine_service() for %s engine %p\n", + engine->name, engine); + if (engine->cyclic_req) + engine_service_cyclic(engine); + else + engine_service(engine, 0); + + /* re-enable interrupts for this engine */ + if (engine->xdev->msix_enabled){ + write_register(engine->interrupt_enable_mask_value, + &engine->regs->interrupt_enable_mask_w1s, + (unsigned long)(&engine->regs->interrupt_enable_mask_w1s) - + (unsigned long)(&engine->regs)); + } else + channel_interrupts_enable(engine->xdev, engine->irq_bitmask); + + /* unlock the engine */ + spin_unlock_irqrestore(&engine->lock, flags); +} + +static u32 engine_service_wb_monitor(struct xdma_engine *engine, + u32 expected_wb) +{ + struct xdma_poll_wb *wb_data; + u32 desc_wb = 0; + u32 sched_limit = 0; + unsigned long timeout; + + BUG_ON(!engine); + wb_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + /* + * Poll the writeback location for the expected number of + * descriptors / error events This loop is skipped for cyclic mode, + * where the expected_desc_count passed in is zero, since it cannot be + * determined before the function is called + */ + + timeout = jiffies + (POLL_TIMEOUT_SECONDS * HZ); + while (expected_wb != 0) { + desc_wb = wb_data->completed_desc_count; + + if (desc_wb & WB_ERR_MASK) + break; + else if (desc_wb == expected_wb) + break; + + /* RTO - prevent system from hanging in polled mode */ + if (time_after(jiffies, timeout)) { + dbg_tfr("Polling timeout occurred"); + dbg_tfr("desc_wb = 0x%08x, expected 0x%08x\n", desc_wb, + expected_wb); + if ((desc_wb & WB_COUNT_MASK) > expected_wb) + desc_wb = expected_wb | WB_ERR_MASK; + + break; + } + + /* + * Define NUM_POLLS_PER_SCHED to limit how much time is spent + * in the scheduler + */ + + if (sched_limit != 0) { + if ((sched_limit % NUM_POLLS_PER_SCHED) == 0) + schedule(); + } + sched_limit++; + } + + return desc_wb; +} + +static int engine_service_poll(struct xdma_engine *engine, + u32 expected_desc_count) +{ + struct xdma_poll_wb *writeback_data; + u32 desc_wb = 0; + unsigned long flags; + int rv = 0; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + writeback_data = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + + if ((expected_desc_count & WB_COUNT_MASK) != expected_desc_count) { + dbg_tfr("Queued descriptor count is larger than supported\n"); + return -1; + } + + /* + * Poll the writeback location for the expected number of + * descriptors / error events This loop is skipped for cyclic mode, + * where the expected_desc_count passed in is zero, since it cannot be + * determined before the function is called + */ + + desc_wb = engine_service_wb_monitor(engine, expected_desc_count); + + spin_lock_irqsave(&engine->lock, flags); + dbg_tfr("%s service.\n", engine->name); + if (engine->cyclic_req) { + rv = engine_service_cyclic(engine); + } else { + rv = engine_service(engine, desc_wb); + } + spin_unlock_irqrestore(&engine->lock, flags); + + return rv; +} + +static irqreturn_t user_irq_service(int irq, struct xdma_user_irq *user_irq) +{ + unsigned long flags; + + BUG_ON(!user_irq); + + if (user_irq->handler) + return user_irq->handler(user_irq->user_idx, user_irq->dev); + + spin_lock_irqsave(&(user_irq->events_lock), flags); + if (!user_irq->events_irq) { + user_irq->events_irq = 1; + wake_up_interruptible(&(user_irq->events_wq)); + } + spin_unlock_irqrestore(&(user_irq->events_lock), flags); + + return IRQ_HANDLED; +} + +/* + * xdma_isr() - Interrupt handler + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_isr(int irq, void *dev_id) +{ + u32 ch_irq; + u32 user_irq; + u32 mask; + struct xdma_dev *xdev; + struct interrupt_regs *irq_regs; + + dbg_irq("(irq=%d, dev 0x%p) <<<< ISR.\n", irq, dev_id); + BUG_ON(!dev_id); + xdev = (struct xdma_dev *)dev_id; + + if (!xdev) { + WARN_ON(!xdev); + dbg_irq("xdma_isr(irq=%d) xdev=%p ??\n", irq, xdev); + return IRQ_NONE; + } + + irq_regs = (struct interrupt_regs *)(xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + + /* read channel interrupt requests */ + ch_irq = read_register(&irq_regs->channel_int_request); + dbg_irq("ch_irq = 0x%08x\n", ch_irq); + + /* + * disable all interrupts that fired; these are re-enabled individually + * after the causing module has been fully serviced. + */ + if (ch_irq) + channel_interrupts_disable(xdev, ch_irq); + + /* read user interrupts - this read also flushes the above write */ + user_irq = read_register(&irq_regs->user_int_request); + dbg_irq("user_irq = 0x%08x\n", user_irq); + + if (user_irq) { + int user = 0; + u32 mask = 1; + int max = xdev->h2c_channel_max; + + for (; user < max && user_irq; user++, mask <<= 1) { + if (user_irq & mask) { + user_irq &= ~mask; + user_irq_service(irq, &xdev->user_irq[user]); + } + } + } + + mask = ch_irq & xdev->mask_irq_h2c; + if (mask) { + int channel = 0; + int max = xdev->h2c_channel_max; + + /* iterate over H2C (PCIe read) */ + for (channel = 0; channel < max && mask; channel++) { + struct xdma_engine *engine = &xdev->engine_h2c[channel]; + + /* engine present and its interrupt fired? */ + if((engine->irq_bitmask & mask) && + (engine->magic == MAGIC_ENGINE)) { + mask &= ~engine->irq_bitmask; + dbg_tfr("schedule_work, %s.\n", engine->name); + schedule_work(&engine->work); + } + } + } + + mask = ch_irq & xdev->mask_irq_c2h; + if (mask) { + int channel = 0; + int max = xdev->c2h_channel_max; + + /* iterate over C2H (PCIe write) */ + for (channel = 0; channel < max && mask; channel++) { + struct xdma_engine *engine = &xdev->engine_c2h[channel]; + + /* engine present and its interrupt fired? */ + if((engine->irq_bitmask & mask) && + (engine->magic == MAGIC_ENGINE)) { + mask &= ~engine->irq_bitmask; + dbg_tfr("schedule_work, %s.\n", engine->name); + schedule_work(&engine->work); + } + } + } + + xdev->irq_count++; + return IRQ_HANDLED; +} + +/* + * xdma_user_irq() - Interrupt handler for user interrupts in MSI-X mode + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_user_irq(int irq, void *dev_id) +{ + struct xdma_user_irq *user_irq; + + dbg_irq("(irq=%d) <<<< INTERRUPT SERVICE ROUTINE\n", irq); + + BUG_ON(!dev_id); + user_irq = (struct xdma_user_irq *)dev_id; + + return user_irq_service(irq, user_irq); +} + +/* + * xdma_channel_irq() - Interrupt handler for channel interrupts in MSI-X mode + * + * @dev_id pointer to xdma_dev + */ +static irqreturn_t xdma_channel_irq(int irq, void *dev_id) +{ + struct xdma_dev *xdev; + struct xdma_engine *engine; + struct interrupt_regs *irq_regs; + + dbg_irq("(irq=%d) <<<< INTERRUPT service ROUTINE\n", irq); + BUG_ON(!dev_id); + + engine = (struct xdma_engine *)dev_id; + xdev = engine->xdev; + + if (!xdev) { + WARN_ON(!xdev); + dbg_irq("xdma_channel_irq(irq=%d) xdev=%p ??\n", irq, xdev); + return IRQ_NONE; + } + + irq_regs = (struct interrupt_regs *)(xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + + /* Disable the interrupt for this engine */ + write_register(engine->interrupt_enable_mask_value, + &engine->regs->interrupt_enable_mask_w1c, + (unsigned long) + (&engine->regs->interrupt_enable_mask_w1c) - + (unsigned long)(&engine->regs)); + /* Dummy read to flush the above write */ + read_register(&irq_regs->channel_int_pending); + /* Schedule the bottom half */ + schedule_work(&engine->work); + + /* + * RTO - need to protect access here if multiple MSI-X are used for + * user interrupts + */ + xdev->irq_count++; + return IRQ_HANDLED; +} + +/* + * Unmap the BAR regions that had been mapped earlier using map_bars() + */ +static void unmap_bars(struct xdma_dev *xdev, struct pci_dev *dev) +{ + int i; + + for (i = 0; i < XDMA_BAR_NUM; i++) { + /* is this BAR mapped? */ + if (xdev->bar[i]) { + /* unmap BAR */ + pci_iounmap(dev, xdev->bar[i]); + /* mark as unmapped */ + xdev->bar[i] = NULL; + } + } +} + +static int map_single_bar(struct xdma_dev *xdev, struct pci_dev *dev, int idx) +{ + resource_size_t bar_start; + resource_size_t bar_len; + resource_size_t map_len; + + bar_start = pci_resource_start(dev, idx); + bar_len = pci_resource_len(dev, idx); + map_len = bar_len; + + xdev->bar[idx] = NULL; + + /* do not map BARs with length 0. Note that start MAY be 0! */ + if (!bar_len) { + //pr_info("BAR #%d is not present - skipping\n", idx); + return 0; + } + + /* BAR size exceeds maximum desired mapping? */ + if (bar_len > INT_MAX) { + pr_info("Limit BAR %d mapping from %llu to %d bytes\n", idx, + (u64)bar_len, INT_MAX); + map_len = (resource_size_t)INT_MAX; + } + /* + * map the full device memory or IO region into kernel virtual + * address space + */ + dbg_init("BAR%d: %llu bytes to be mapped.\n", idx, (u64)map_len); + xdev->bar[idx] = pci_iomap(dev, idx, map_len); + + if (!xdev->bar[idx]) { + pr_info("Could not map BAR %d.\n", idx); + return -1; + } + + pr_info("BAR%d at 0x%llx mapped at 0x%p, length=%llu(/%llu)\n", idx, + (u64)bar_start, xdev->bar[idx], (u64)map_len, (u64)bar_len); + + return (int)map_len; +} + +static int is_config_bar(struct xdma_dev *xdev, int idx) +{ + u32 irq_id = 0; + u32 cfg_id = 0; + int flag = 0; + u32 mask = 0xffff0000; /* Compare only XDMA ID's not Version number */ + struct interrupt_regs *irq_regs = + (struct interrupt_regs *) (xdev->bar[idx] + XDMA_OFS_INT_CTRL); + struct config_regs *cfg_regs = + (struct config_regs *)(xdev->bar[idx] + XDMA_OFS_CONFIG); + + irq_id = read_register(&irq_regs->identifier); + cfg_id = read_register(&cfg_regs->identifier); + + if (((irq_id & mask)== IRQ_BLOCK_ID) && + ((cfg_id & mask)== CONFIG_BLOCK_ID)) { + dbg_init("BAR %d is the XDMA config BAR\n", idx); + flag = 1; + } else { + dbg_init("BAR %d is NOT the XDMA config BAR: 0x%x, 0x%x.\n", + idx, irq_id, cfg_id); + flag = 0; + } + + return flag; +} + +static void identify_bars(struct xdma_dev *xdev, int *bar_id_list, int num_bars, + int config_bar_pos) +{ + /* + * The following logic identifies which BARs contain what functionality + * based on the position of the XDMA config BAR and the number of BARs + * detected. The rules are that the user logic and bypass logic BARs + * are optional. When both are present, the XDMA config BAR will be the + * 2nd BAR detected (config_bar_pos = 1), with the user logic being + * detected first and the bypass being detected last. When one is + * omitted, the type of BAR present can be identified by whether the + * XDMA config BAR is detected first or last. When both are omitted, + * only the XDMA config BAR is present. This somewhat convoluted + * approach is used instead of relying on BAR numbers in order to work + * correctly with both 32-bit and 64-bit BARs. + */ + + BUG_ON(!xdev); + BUG_ON(!bar_id_list); + + dbg_init("xdev 0x%p, bars %d, config at %d.\n", + xdev, num_bars, config_bar_pos); + + switch (num_bars) { + case 1: + /* Only one BAR present - no extra work necessary */ + break; + + case 2: + if (config_bar_pos == 0) { + xdev->bypass_bar_idx = bar_id_list[1]; + } else if (config_bar_pos == 1) { + xdev->user_bar_idx = bar_id_list[0]; + } else { + pr_info("2, XDMA config BAR unexpected %d.\n", + config_bar_pos); + } + break; + + case 3: + case 4: + if ((config_bar_pos == 1) || (config_bar_pos == 2)) { + /* user bar at bar #0 */ + xdev->user_bar_idx = bar_id_list[0]; + /* bypass bar at the last bar */ + xdev->bypass_bar_idx = bar_id_list[num_bars - 1]; + } else { + pr_info("3/4, XDMA config BAR unexpected %d.\n", + config_bar_pos); + } + break; + + default: + /* Should not occur - warn user but safe to continue */ + pr_info("Unexpected # BARs (%d), XDMA config BAR only.\n", + num_bars); + break; + + } + pr_info("%d BARs: config %d, user %d, bypass %d.\n", + num_bars, config_bar_pos, xdev->user_bar_idx, + xdev->bypass_bar_idx); +} + +/* map_bars() -- map device regions into kernel virtual address space + * + * Map the device memory regions into kernel virtual address space after + * verifying their sizes respect the minimum sizes needed + */ +static int map_bars(struct xdma_dev *xdev, struct pci_dev *dev) +{ + int rv; + int i; + int bar_id_list[XDMA_BAR_NUM]; + int bar_id_idx = 0; + int config_bar_pos = 0; + + /* iterate through all the BARs */ + for (i = 0; i < XDMA_BAR_NUM; i++) { + int bar_len; + + bar_len = map_single_bar(xdev, dev, i); + if (bar_len == 0) { + continue; + } else if (bar_len < 0) { + rv = -EINVAL; + goto fail; + } + + /* Try to identify BAR as XDMA control BAR */ + if ((bar_len >= XDMA_BAR_SIZE) && (xdev->config_bar_idx < 0)) { + + if (is_config_bar(xdev, i)) { + xdev->config_bar_idx = i; + config_bar_pos = bar_id_idx; + pr_info("config bar %d, pos %d.\n", + xdev->config_bar_idx, config_bar_pos); + } + } + + bar_id_list[bar_id_idx] = i; + bar_id_idx++; + } + + /* The XDMA config BAR must always be present */ + if (xdev->config_bar_idx < 0) { + pr_info("Failed to detect XDMA config BAR\n"); + rv = -EINVAL; + goto fail; + } + +#ifdef __LIBXDMA_CONFIG_BAR_ONLY__ + /* unmapped all other bars, except XDMA config. bar */ + for (i = 0; i < XDMA_BAR_NUM; i++) { + if (i == xdev->config_bar_idx) + continue; + + /* is this BAR mapped? */ + if (xdev->bar[i]) { + /* unmap BAR */ + pci_iounmap(dev, xdev->bar[i]); + /* mark as unmapped */ + xdev->bar[i] = NULL; + pr_info("unmap non-config bar %d.\n", i); + } + } +#else + identify_bars(xdev, bar_id_list, bar_id_idx, config_bar_pos); +#endif + + /* successfully mapped all required BAR regions */ + return 0; + +fail: + /* unwind; unmap any BARs that we did map */ + unmap_bars(xdev, dev); + return rv; +} + +/* + * MSI-X interrupt: + * <h2c+c2h channel_max> vectors, followed by <user_max> vectors + */ + +/* + * RTO - code to detect if MSI/MSI-X capability exists is derived + * from linux/pci/msi.c - pci_msi_check_device + */ + +#ifndef arch_msi_check_device +int arch_msi_check_device(struct pci_dev *dev, int nvec, int type) +{ + return 0; +} +#endif + +/* type = PCI_CAP_ID_MSI or PCI_CAP_ID_MSIX */ +static int msi_msix_capable(struct pci_dev *dev, int type) +{ + struct pci_bus *bus; + int ret; + + if (!dev || dev->no_msi) + return 0; + + for (bus = dev->bus; bus; bus = bus->parent) + if (bus->bus_flags & PCI_BUS_FLAGS_NO_MSI) + return 0; + + ret = arch_msi_check_device(dev, 1, type); + if (ret) + return 0; + + if (!pci_find_capability(dev, type)) + return 0; + + return 1; +} + +static void disable_msi_msix(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + if (xdev->msix_enabled) { + pci_disable_msix(pdev); + xdev->msix_enabled = 0; + } else if (xdev->msi_enabled) { + pci_disable_msi(pdev); + xdev->msi_enabled = 0; + } +} + +static int enable_msi_msix(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv = 0; + + BUG_ON(!xdev); + BUG_ON(!pdev); + + if (!interrupt_mode && msi_msix_capable(pdev, PCI_CAP_ID_MSIX)) { + int req_nvec = xdev->c2h_channel_max + xdev->h2c_channel_max + + xdev->user_max; + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + dbg_init("Enabling MSI-X\n"); + rv = pci_alloc_irq_vectors(pdev, req_nvec, req_nvec, + PCI_IRQ_MSIX); +#else + int i; + + dbg_init("Enabling MSI-X\n"); + for (i = 0; i < req_nvec; i++) + xdev->entry[i].entry = i; + + rv = pci_enable_msix(pdev, xdev->entry, req_nvec); +#endif + if (rv < 0) + dbg_init("Couldn't enable MSI-X mode: %d\n", rv); + + xdev->msix_enabled = 1; + + } else if (interrupt_mode == 1 && + msi_msix_capable(pdev, PCI_CAP_ID_MSI)) { + /* enable message signalled interrupts */ + dbg_init("pci_enable_msi()\n"); + rv = pci_enable_msi(pdev); + if (rv < 0) + dbg_init("Couldn't enable MSI mode: %d\n", rv); + xdev->msi_enabled = 1; + + } else { + dbg_init("MSI/MSI-X not detected - using legacy interrupts\n"); + } + + return rv; +} + +static void pci_check_intr_pend(struct pci_dev *pdev) +{ + u16 v; + + pci_read_config_word(pdev, PCI_STATUS, &v); + if (v & PCI_STATUS_INTERRUPT) { + pr_info("%s PCI STATUS Interrupt pending 0x%x.\n", + dev_name(&pdev->dev), v); + pci_write_config_word(pdev, PCI_STATUS, PCI_STATUS_INTERRUPT); + } +} + +static void pci_keep_intx_enabled(struct pci_dev *pdev) +{ + /* workaround to a h/w bug: + * when msix/msi become unavaile, default to legacy. + * However the legacy enable was not checked. + * If the legacy was disabled, no ack then everything stuck + */ + u16 pcmd, pcmd_new; + + pci_read_config_word(pdev, PCI_COMMAND, &pcmd); + pcmd_new = pcmd & ~PCI_COMMAND_INTX_DISABLE; + if (pcmd_new != pcmd) { + pr_info("%s: clear INTX_DISABLE, 0x%x -> 0x%x.\n", + dev_name(&pdev->dev), pcmd, pcmd_new); + pci_write_config_word(pdev, PCI_COMMAND, pcmd_new); + } +} + +static void prog_irq_msix_user(struct xdma_dev *xdev, bool clear) +{ + /* user */ + struct interrupt_regs *int_regs = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + u32 i = xdev->c2h_channel_max + xdev->h2c_channel_max; + u32 max = i + xdev->user_max; + int j; + + for (j = 0; i < max; j++) { + u32 val = 0; + int k; + int shift = 0; + + if (clear) + i += 4; + else + for (k = 0; k < 4 && i < max; i++, k++, shift += 8) + val |= (i & 0x1f) << shift; + + write_register(val, &int_regs->user_msi_vector[j], + XDMA_OFS_INT_CTRL + + ((unsigned long)&int_regs->user_msi_vector[j] - + (unsigned long)int_regs)); + + dbg_init("vector %d, 0x%x.\n", j, val); + } +} + +static void prog_irq_msix_channel(struct xdma_dev *xdev, bool clear) +{ + struct interrupt_regs *int_regs = (struct interrupt_regs *) + (xdev->bar[xdev->config_bar_idx] + + XDMA_OFS_INT_CTRL); + u32 max = xdev->c2h_channel_max + xdev->h2c_channel_max; + u32 i; + int j; + + /* engine */ + for (i = 0, j = 0; i < max; j++) { + u32 val = 0; + int k; + int shift = 0; + + if (clear) + i += 4; + else + for (k = 0; k < 4 && i < max; i++, k++, shift += 8) + val |= (i & 0x1f) << shift; + + write_register(val, &int_regs->channel_msi_vector[j], + XDMA_OFS_INT_CTRL + + ((unsigned long)&int_regs->channel_msi_vector[j] - + (unsigned long)int_regs)); + dbg_init("vector %d, 0x%x.\n", j, val); + } +} + +static void irq_msix_channel_teardown(struct xdma_dev *xdev) +{ + struct xdma_engine *engine; + int j = 0; + int i = 0; + + if (!xdev->msix_enabled) + return; + + prog_irq_msix_channel(xdev, 1); + + engine = xdev->engine_h2c; + for (i = 0; i < xdev->h2c_channel_max; i++, j++, engine++) { + if (!engine->msix_irq_line) + break; + dbg_sg("Release IRQ#%d for engine %p\n", engine->msix_irq_line, + engine); + free_irq(engine->msix_irq_line, engine); + } + + engine = xdev->engine_c2h; + for (i = 0; i < xdev->c2h_channel_max; i++, j++, engine++) { + if (!engine->msix_irq_line) + break; + dbg_sg("Release IRQ#%d for engine %p\n", engine->msix_irq_line, + engine); + free_irq(engine->msix_irq_line, engine); + } +} + +static int irq_msix_channel_setup(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max; + int rv = 0; + u32 vector; + struct xdma_engine *engine; + + BUG_ON(!xdev); + if (!xdev->msix_enabled) + return 0; + + engine = xdev->engine_h2c; + for (i = 0; i < xdev->h2c_channel_max; i++, engine++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + vector = pci_irq_vector(xdev->pdev, i); +#else + vector = xdev->entry[i].vector; +#endif + rv = request_irq(vector, xdma_channel_irq, 0, xdev->mod_name, + engine); + if (rv) { + pr_info("requesti irq#%d failed %d, engine %s.\n", + vector, rv, engine->name); + return rv; + } + pr_info("engine %s, irq#%d.\n", engine->name, vector); + engine->msix_irq_line = vector; + } + + engine = xdev->engine_c2h; + for (i = 0; i < xdev->c2h_channel_max; i++, j++, engine++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + vector = pci_irq_vector(xdev->pdev, j); +#else + vector = xdev->entry[j].vector; +#endif + rv = request_irq(vector, xdma_channel_irq, 0, xdev->mod_name, + engine); + if (rv) { + pr_info("requesti irq#%d failed %d, engine %s.\n", + vector, rv, engine->name); + return rv; + } + pr_info("engine %s, irq#%d.\n", engine->name, vector); + engine->msix_irq_line = vector; + } + + return 0; +} + +static void irq_msix_user_teardown(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max + xdev->c2h_channel_max; + + BUG_ON(!xdev); + + if (!xdev->msix_enabled) + return; + + prog_irq_msix_user(xdev, 1); + + for (i = 0; i < xdev->user_max; i++, j++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + dbg_init("user %d, releasing IRQ#%d\n", i, vector); + free_irq(vector, &xdev->user_irq[i]); + } +} + +static int irq_msix_user_setup(struct xdma_dev *xdev) +{ + int i; + int j = xdev->h2c_channel_max + xdev->c2h_channel_max; + int rv = 0; + + /* vectors set in probe_scan_for_msi() */ + for (i = 0; i < xdev->user_max; i++, j++) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + rv = request_irq(vector, xdma_user_irq, 0, xdev->mod_name, + &xdev->user_irq[i]); + if (rv) { + pr_info("user %d couldn't use IRQ#%d, %d\n", + i, vector, rv); + break; + } + pr_info("%d-USR-%d, IRQ#%d with 0x%p\n", xdev->idx, i, vector, + &xdev->user_irq[i]); + } + + /* If any errors occur, free IRQs that were successfully requested */ + if (rv) { + for (i--, j--; i >= 0; i--, j--) { +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,12,0) + u32 vector = pci_irq_vector(xdev->pdev, j); +#else + u32 vector = xdev->entry[j].vector; +#endif + free_irq(vector, &xdev->user_irq[i]); + } + } + + return rv; +} + +static int irq_msi_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv; + + xdev->irq_line = (int)pdev->irq; + rv = request_irq(pdev->irq, xdma_isr, 0, xdev->mod_name, xdev); + if (rv) + dbg_init("Couldn't use IRQ#%d, %d\n", pdev->irq, rv); + else + dbg_init("Using IRQ#%d with 0x%p\n", pdev->irq, xdev); + + return rv; +} + +static int irq_legacy_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + u32 w; + u8 val; + void *reg; + int rv; + + pci_read_config_byte(pdev, PCI_INTERRUPT_PIN, &val); + dbg_init("Legacy Interrupt register value = %d\n", val); + if (val > 1) { + val--; + w = (val<<24) | (val<<16) | (val<<8)| val; + /* Program IRQ Block Channel vactor and IRQ Block User vector + * with Legacy interrupt value */ + reg = xdev->bar[xdev->config_bar_idx] + 0x2080; // IRQ user + write_register(w, reg, 0x2080); + write_register(w, reg+0x4, 0x2084); + write_register(w, reg+0x8, 0x2088); + write_register(w, reg+0xC, 0x208C); + reg = xdev->bar[xdev->config_bar_idx] + 0x20A0; // IRQ Block + write_register(w, reg, 0x20A0); + write_register(w, reg+0x4, 0x20A4); + } + + xdev->irq_line = (int)pdev->irq; + rv = request_irq(pdev->irq, xdma_isr, IRQF_SHARED, xdev->mod_name, + xdev); + if (rv) + dbg_init("Couldn't use IRQ#%d, %d\n", pdev->irq, rv); + else + dbg_init("Using IRQ#%d with 0x%p\n", pdev->irq, xdev); + + return rv; +} + +static void irq_teardown(struct xdma_dev *xdev) +{ + if (xdev->msix_enabled) { + irq_msix_channel_teardown(xdev); + irq_msix_user_teardown(xdev); + } else if (xdev->irq_line != -1) { + dbg_init("Releasing IRQ#%d\n", xdev->irq_line); + free_irq(xdev->irq_line, xdev); + } +} + +static int irq_setup(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + pci_keep_intx_enabled(pdev); + + if (xdev->msix_enabled) { + int rv = irq_msix_channel_setup(xdev); + if (rv) + return rv; + rv = irq_msix_user_setup(xdev); + if (rv) + return rv; + prog_irq_msix_channel(xdev, 0); + prog_irq_msix_user(xdev, 0); + + return 0; + } else if (xdev->msi_enabled) + return irq_msi_setup(xdev, pdev); + + return irq_legacy_setup(xdev, pdev); +} + +#ifdef __LIBXDMA_DEBUG__ +static void dump_desc(struct xdma_desc *desc_virt) +{ + int j; + u32 *p = (u32 *)desc_virt; + static char * const field_name[] = { + "magic|extra_adjacent|control", "bytes", "src_addr_lo", + "src_addr_hi", "dst_addr_lo", "dst_addr_hi", "next_addr", + "next_addr_pad"}; + char *dummy; + + /* remove warning about unused variable when debug printing is off */ + dummy = field_name[0]; + + for (j = 0; j < 8; j += 1) { + pr_info("0x%08lx/0x%02lx: 0x%08x 0x%08x %s\n", + (uintptr_t)p, (uintptr_t)p & 15, (int)*p, + le32_to_cpu(*p), field_name[j]); + p++; + } + pr_info("\n"); +} + +static void transfer_dump(struct xdma_transfer *transfer) +{ + int i; + struct xdma_desc *desc_virt = transfer->desc_virt; + + pr_info("xfer 0x%p, state 0x%x, f 0x%x, dir %d, len %u, last %d.\n", + transfer, transfer->state, transfer->flags, transfer->dir, + transfer->len, transfer->last_in_request); + + pr_info("transfer 0x%p, desc %d, bus 0x%llx, adj %d.\n", + transfer, transfer->desc_num, (u64)transfer->desc_bus, + transfer->desc_adjacent); + for (i = 0; i < transfer->desc_num; i += 1) + dump_desc(desc_virt + i); +} +#endif /* __LIBXDMA_DEBUG__ */ + +/* xdma_desc_alloc() - Allocate cache-coherent array of N descriptors. + * + * Allocates an array of 'number' descriptors in contiguous PCI bus addressable + * memory. Chains the descriptors as a singly-linked list; the descriptor's + * next * pointer specifies the bus address of the next descriptor. + * + * + * @dev Pointer to pci_dev + * @number Number of descriptors to be allocated + * @desc_bus_p Pointer where to store the first descriptor bus address + * + * @return Virtual address of the first descriptor + * + */ +static void transfer_desc_init(struct xdma_transfer *transfer, int count) +{ + struct xdma_desc *desc_virt = transfer->desc_virt; + dma_addr_t desc_bus = transfer->desc_bus; + int i; + int adj = count - 1; + int extra_adj; + u32 temp_control; + + BUG_ON(count > XDMA_TRANSFER_MAX_DESC); + + /* create singly-linked list for SG DMA controller */ + for (i = 0; i < count - 1; i++) { + /* increment bus address to next in array */ + desc_bus += sizeof(struct xdma_desc); + + /* singly-linked list uses bus addresses */ + desc_virt[i].next_lo = cpu_to_le32(PCI_DMA_L(desc_bus)); + desc_virt[i].next_hi = cpu_to_le32(PCI_DMA_H(desc_bus)); + desc_virt[i].bytes = cpu_to_le32(0); + + /* any adjacent descriptors? */ + if (adj > 0) { + extra_adj = adj - 1; + if (extra_adj > MAX_EXTRA_ADJ) + extra_adj = MAX_EXTRA_ADJ; + + adj--; + } else { + extra_adj = 0; + } + + temp_control = DESC_MAGIC | (extra_adj << 8); + + desc_virt[i].control = cpu_to_le32(temp_control); + } + /* { i = number - 1 } */ + /* zero the last descriptor next pointer */ + desc_virt[i].next_lo = cpu_to_le32(0); + desc_virt[i].next_hi = cpu_to_le32(0); + desc_virt[i].bytes = cpu_to_le32(0); + + temp_control = DESC_MAGIC; + + desc_virt[i].control = cpu_to_le32(temp_control); +} + +/* xdma_desc_link() - Link two descriptors + * + * Link the first descriptor to a second descriptor, or terminate the first. + * + * @first first descriptor + * @second second descriptor, or NULL if first descriptor must be set as last. + * @second_bus bus address of second descriptor + */ +static void xdma_desc_link(struct xdma_desc *first, struct xdma_desc *second, + dma_addr_t second_bus) +{ + /* + * remember reserved control in first descriptor, but zero + * extra_adjacent! + */ + /* RTO - what's this about? Shouldn't it be 0x0000c0ffUL? */ + u32 control = le32_to_cpu(first->control) & 0x0000f0ffUL; + /* second descriptor given? */ + if (second) { + /* + * link last descriptor of 1st array to first descriptor of + * 2nd array + */ + first->next_lo = cpu_to_le32(PCI_DMA_L(second_bus)); + first->next_hi = cpu_to_le32(PCI_DMA_H(second_bus)); + WARN_ON(first->next_hi); + /* no second descriptor given */ + } else { + /* first descriptor is the last */ + first->next_lo = 0; + first->next_hi = 0; + } + /* merge magic, extra_adjacent and control field */ + control |= DESC_MAGIC; + + /* write bytes and next_num */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_adjacent -- Set how many descriptors are adjacent to this one */ +static void xdma_desc_adjacent(struct xdma_desc *desc, int next_adjacent) +{ + int extra_adj = 0; + /* remember reserved and control bits */ + u32 control = le32_to_cpu(desc->control) & 0x0000f0ffUL; + u32 max_adj_4k = 0; + + if (next_adjacent > 0) { + extra_adj = next_adjacent - 1; + if (extra_adj > MAX_EXTRA_ADJ){ + extra_adj = MAX_EXTRA_ADJ; + } + max_adj_4k = (0x1000 - ((le32_to_cpu(desc->next_lo))&0xFFF))/32 - 1; + if (extra_adj>max_adj_4k) { + extra_adj = max_adj_4k; + } + if(extra_adj<0){ + printk("Warning: extra_adj<0, converting it to 0\n"); + extra_adj = 0; + } + } + /* merge adjacent and control field */ + control |= 0xAD4B0000UL | (extra_adj << 8); + /* write control and next_adjacent */ + desc->control = cpu_to_le32(control); +} + +/* xdma_desc_control -- Set complete control field of a descriptor. */ +static void xdma_desc_control_set(struct xdma_desc *first, u32 control_field) +{ + /* remember magic and adjacent number */ + u32 control = le32_to_cpu(first->control) & ~(LS_BYTE_MASK); + + BUG_ON(control_field & ~(LS_BYTE_MASK)); + /* merge adjacent and control field */ + control |= control_field; + /* write control and next_adjacent */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_clear -- Clear bits in control field of a descriptor. */ +static void xdma_desc_control_clear(struct xdma_desc *first, u32 clear_mask) +{ + /* remember magic and adjacent number */ + u32 control = le32_to_cpu(first->control); + + BUG_ON(clear_mask & ~(LS_BYTE_MASK)); + + /* merge adjacent and control field */ + control &= (~clear_mask); + /* write control and next_adjacent */ + first->control = cpu_to_le32(control); +} + +/* xdma_desc_done - recycle cache-coherent linked list of descriptors. + * + * @dev Pointer to pci_dev + * @number Number of descriptors to be allocated + * @desc_virt Pointer to (i.e. virtual address of) first descriptor in list + * @desc_bus Bus address of first descriptor in list + */ +static inline void xdma_desc_done(struct xdma_desc *desc_virt) +{ + memset(desc_virt, 0, XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc)); +} + +/* xdma_desc() - Fill a descriptor with the transfer details + * + * @desc pointer to descriptor to be filled + * @addr root complex address + * @ep_addr end point address + * @len number of bytes, must be a (non-negative) multiple of 4. + * @dir, dma direction + * is the end point address. If zero, vice versa. + * + * Does not modify the next pointer + */ +static void xdma_desc_set(struct xdma_desc *desc, dma_addr_t rc_bus_addr, + u64 ep_addr, int len, int dir) +{ + /* transfer length */ + desc->bytes = cpu_to_le32(len); + if (dir == DMA_TO_DEVICE) { + /* read from root complex memory (source address) */ + desc->src_addr_lo = cpu_to_le32(PCI_DMA_L(rc_bus_addr)); + desc->src_addr_hi = cpu_to_le32(PCI_DMA_H(rc_bus_addr)); + /* write to end point address (destination address) */ + desc->dst_addr_lo = cpu_to_le32(PCI_DMA_L(ep_addr)); + desc->dst_addr_hi = cpu_to_le32(PCI_DMA_H(ep_addr)); + } else { + /* read from end point address (source address) */ + desc->src_addr_lo = cpu_to_le32(PCI_DMA_L(ep_addr)); + desc->src_addr_hi = cpu_to_le32(PCI_DMA_H(ep_addr)); + /* write to root complex memory (destination address) */ + desc->dst_addr_lo = cpu_to_le32(PCI_DMA_L(rc_bus_addr)); + desc->dst_addr_hi = cpu_to_le32(PCI_DMA_H(rc_bus_addr)); + } +} + +/* + * should hold the engine->lock; + */ +static void transfer_abort(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + struct xdma_transfer *head; + + BUG_ON(!engine); + BUG_ON(!transfer); + BUG_ON(transfer->desc_num == 0); + + pr_info("abort transfer 0x%p, desc %d, engine desc queued %d.\n", + transfer, transfer->desc_num, engine->desc_dequeued); + + head = list_entry(engine->transfer_list.next, struct xdma_transfer, + entry); + if (head == transfer) + list_del(engine->transfer_list.next); + else + pr_info("engine %s, transfer 0x%p NOT found, 0x%p.\n", + engine->name, transfer, head); + + if (transfer->state == TRANSFER_STATE_SUBMITTED) + transfer->state = TRANSFER_STATE_ABORTED; +} + +/* transfer_queue() - Queue a DMA transfer on the engine + * + * @engine DMA engine doing the transfer + * @transfer DMA transfer submitted to the engine + * + * Takes and releases the engine spinlock + */ +static int transfer_queue(struct xdma_engine *engine, + struct xdma_transfer *transfer) +{ + int rv = 0; + struct xdma_transfer *transfer_started; + struct xdma_dev *xdev; + unsigned long flags; + + BUG_ON(!engine); + BUG_ON(!engine->xdev); + BUG_ON(!transfer); + BUG_ON(transfer->desc_num == 0); + dbg_tfr("transfer_queue(transfer=0x%p).\n", transfer); + + xdev = engine->xdev; + if (xdma_device_flag_check(xdev, XDEV_FLAG_OFFLINE)) { + pr_info("dev 0x%p offline, transfer 0x%p not queued.\n", + xdev, transfer); + return -EBUSY; + } + + /* lock the engine state */ + spin_lock_irqsave(&engine->lock, flags); + + engine->prev_cpu = get_cpu(); + put_cpu(); + + /* engine is being shutdown; do not accept new transfers */ + if (engine->shutdown & ENGINE_SHUTDOWN_REQUEST) { + pr_info("engine %s offline, transfer 0x%p not queued.\n", + engine->name, transfer); + rv = -EBUSY; + goto shutdown; + } + + /* mark the transfer as submitted */ + transfer->state = TRANSFER_STATE_SUBMITTED; + /* add transfer to the tail of the engine transfer queue */ + list_add_tail(&transfer->entry, &engine->transfer_list); + + /* engine is idle? */ + if (!engine->running) { + /* start engine */ + dbg_tfr("transfer_queue(): starting %s engine.\n", + engine->name); + transfer_started = engine_start(engine); + dbg_tfr("transfer=0x%p started %s engine with transfer 0x%p.\n", + transfer, engine->name, transfer_started); + } else { + dbg_tfr("transfer=0x%p queued, with %s engine running.\n", + transfer, engine->name); + } + +shutdown: + /* unlock the engine state */ + dbg_tfr("engine->running = %d\n", engine->running); + spin_unlock_irqrestore(&engine->lock, flags); + return rv; +} + +static void engine_alignments(struct xdma_engine *engine) +{ + u32 w; + u32 align_bytes; + u32 granularity_bytes; + u32 address_bits; + + w = read_register(&engine->regs->alignments); + dbg_init("engine %p name %s alignments=0x%08x\n", engine, + engine->name, (int)w); + + /* RTO - add some macros to extract these fields */ + align_bytes = (w & 0x00ff0000U) >> 16; + granularity_bytes = (w & 0x0000ff00U) >> 8; + address_bits = (w & 0x000000ffU); + + dbg_init("align_bytes = %d\n", align_bytes); + dbg_init("granularity_bytes = %d\n", granularity_bytes); + dbg_init("address_bits = %d\n", address_bits); + + if (w) { + engine->addr_align = align_bytes; + engine->len_granularity = granularity_bytes; + engine->addr_bits = address_bits; + } else { + /* Some default values if alignments are unspecified */ + engine->addr_align = 1; + engine->len_granularity = 1; + engine->addr_bits = 64; + } +} + +static void engine_free_resource(struct xdma_engine *engine) +{ + struct xdma_dev *xdev = engine->xdev; + + /* Release memory use for descriptor writebacks */ + if (engine->poll_mode_addr_virt) { + dbg_sg("Releasing memory for descriptor writeback\n"); + dma_free_coherent(&xdev->pdev->dev, + sizeof(struct xdma_poll_wb), + engine->poll_mode_addr_virt, + engine->poll_mode_bus); + dbg_sg("Released memory for descriptor writeback\n"); + engine->poll_mode_addr_virt = NULL; + } + + if (engine->desc) { + dbg_init("device %s, engine %s pre-alloc desc 0x%p,0x%llx.\n", + dev_name(&xdev->pdev->dev), engine->name, + engine->desc, engine->desc_bus); + dma_free_coherent(&xdev->pdev->dev, + XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc), + engine->desc, engine->desc_bus); + engine->desc = NULL; + } + + if (engine->cyclic_result) { + dma_free_coherent(&xdev->pdev->dev, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result), + engine->cyclic_result, engine->cyclic_result_bus); + engine->cyclic_result = NULL; + } +} + +static void engine_destroy(struct xdma_dev *xdev, struct xdma_engine *engine) +{ + BUG_ON(!xdev); + BUG_ON(!engine); + + dbg_sg("Shutting down engine %s%d", engine->name, engine->channel); + + /* Disable interrupts to stop processing new events during shutdown */ + write_register(0x0, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + + if (enable_credit_mp && engine->streaming && + engine->dir == DMA_FROM_DEVICE) { + u32 reg_value = (0x1 << engine->channel) << 16; + struct sgdma_common_regs *reg = (struct sgdma_common_regs *) + (xdev->bar[xdev->config_bar_idx] + + (0x6*TARGET_SPACING)); + write_register(reg_value, ®->credit_mode_enable_w1c, 0); + } + + /* Release memory use for descriptor writebacks */ + engine_free_resource(engine); + + memset(engine, 0, sizeof(struct xdma_engine)); + /* Decrement the number of engines available */ + xdev->engines_num--; +} + +/** + *engine_cyclic_stop() - stop a cyclic transfer running on an SG DMA engine + * + *engine->lock must be taken + */ +struct xdma_transfer *engine_cyclic_stop(struct xdma_engine *engine) +{ + struct xdma_transfer *transfer = 0; + + /* transfers on queue? */ + if (!list_empty(&engine->transfer_list)) { + /* pick first transfer on the queue (was submitted to engine) */ + transfer = list_entry(engine->transfer_list.next, + struct xdma_transfer, entry); + BUG_ON(!transfer); + + xdma_engine_stop(engine); + + if (transfer->cyclic) { + if (engine->xdma_perf) + dbg_perf("Stopping perf transfer on %s\n", + engine->name); + else + dbg_perf("Stopping cyclic transfer on %s\n", + engine->name); + /* make sure the handler sees correct transfer state */ + transfer->cyclic = 1; + /* + * set STOP flag and interrupt on completion, on the + * last descriptor + */ + xdma_desc_control_set( + transfer->desc_virt + transfer->desc_num - 1, + XDMA_DESC_COMPLETED | XDMA_DESC_STOPPED); + } else { + dbg_sg("(engine=%p) running transfer is not cyclic\n", + engine); + } + } else { + dbg_sg("(engine=%p) found not running transfer.\n", engine); + } + return transfer; +} +EXPORT_SYMBOL_GPL(engine_cyclic_stop); + +static int engine_writeback_setup(struct xdma_engine *engine) +{ + u32 w; + struct xdma_dev *xdev; + struct xdma_poll_wb *writeback; + + BUG_ON(!engine); + xdev = engine->xdev; + BUG_ON(!xdev); + + /* + * RTO - doing the allocation per engine is wasteful since a full page + * is allocated each time - better to allocate one page for the whole + * device during probe() and set per-engine offsets here + */ + writeback = (struct xdma_poll_wb *)engine->poll_mode_addr_virt; + writeback->completed_desc_count = 0; + + dbg_init("Setting writeback location to 0x%llx for engine %p", + engine->poll_mode_bus, engine); + w = cpu_to_le32(PCI_DMA_L(engine->poll_mode_bus)); + write_register(w, &engine->regs->poll_mode_wb_lo, + (unsigned long)(&engine->regs->poll_mode_wb_lo) - + (unsigned long)(&engine->regs)); + w = cpu_to_le32(PCI_DMA_H(engine->poll_mode_bus)); + write_register(w, &engine->regs->poll_mode_wb_hi, + (unsigned long)(&engine->regs->poll_mode_wb_hi) - + (unsigned long)(&engine->regs)); + + return 0; +} + + +/* engine_create() - Create an SG DMA engine bookkeeping data structure + * + * An SG DMA engine consists of the resources for a single-direction transfer + * queue; the SG DMA hardware, the software queue and interrupt handling. + * + * @dev Pointer to pci_dev + * @offset byte address offset in BAR[xdev->config_bar_idx] resource for the + * SG DMA * controller registers. + * @dir: DMA_TO/FROM_DEVICE + * @streaming Whether the engine is attached to AXI ST (rather than MM) + */ +static int engine_init_regs(struct xdma_engine *engine) +{ + u32 reg_value; + int rv = 0; + + write_register(XDMA_CTRL_NON_INCR_ADDR, &engine->regs->control_w1c, + (unsigned long)(&engine->regs->control_w1c) - + (unsigned long)(&engine->regs)); + + engine_alignments(engine); + + /* Configure error interrupts by default */ + reg_value = XDMA_CTRL_IE_DESC_ALIGN_MISMATCH; + reg_value |= XDMA_CTRL_IE_MAGIC_STOPPED; + reg_value |= XDMA_CTRL_IE_MAGIC_STOPPED; + reg_value |= XDMA_CTRL_IE_READ_ERROR; + reg_value |= XDMA_CTRL_IE_DESC_ERROR; + + /* if using polled mode, configure writeback address */ + if (poll_mode) { + rv = engine_writeback_setup(engine); + if (rv) { + dbg_init("%s descr writeback setup failed.\n", + engine->name); + goto fail_wb; + } + } else { + /* enable the relevant completion interrupts */ + reg_value |= XDMA_CTRL_IE_DESC_STOPPED; + reg_value |= XDMA_CTRL_IE_DESC_COMPLETED; + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) + reg_value |= XDMA_CTRL_IE_IDLE_STOPPED; + } + + /* Apply engine configurations */ + write_register(reg_value, &engine->regs->interrupt_enable_mask, + (unsigned long)(&engine->regs->interrupt_enable_mask) - + (unsigned long)(&engine->regs)); + + engine->interrupt_enable_mask_value = reg_value; + + /* only enable credit mode for AXI-ST C2H */ + if (enable_credit_mp && engine->streaming && + engine->dir == DMA_FROM_DEVICE) { + + struct xdma_dev *xdev = engine->xdev; + u32 reg_value = (0x1 << engine->channel) << 16; + struct sgdma_common_regs *reg = (struct sgdma_common_regs *) + (xdev->bar[xdev->config_bar_idx] + + (0x6*TARGET_SPACING)); + + write_register(reg_value, ®->credit_mode_enable_w1s, 0); + } + + return 0; + +fail_wb: + return rv; +} + +static int engine_alloc_resource(struct xdma_engine *engine) +{ + struct xdma_dev *xdev = engine->xdev; + + engine->desc = dma_alloc_coherent(&xdev->pdev->dev, + XDMA_TRANSFER_MAX_DESC * sizeof(struct xdma_desc), + &engine->desc_bus, GFP_KERNEL); + if (!engine->desc) { + pr_warn("dev %s, %s pre-alloc desc OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + + if (poll_mode) { + engine->poll_mode_addr_virt = dma_alloc_coherent( + &xdev->pdev->dev, + sizeof(struct xdma_poll_wb), + &engine->poll_mode_bus, GFP_KERNEL); + if (!engine->poll_mode_addr_virt) { + pr_warn("%s, %s poll pre-alloc writeback OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + } + + if (engine->streaming && engine->dir == DMA_FROM_DEVICE) { + engine->cyclic_result = dma_alloc_coherent(&xdev->pdev->dev, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result), + &engine->cyclic_result_bus, GFP_KERNEL); + + if (!engine->cyclic_result) { + pr_warn("%s, %s pre-alloc result OOM.\n", + dev_name(&xdev->pdev->dev), engine->name); + goto err_out; + } + } + + return 0; + +err_out: + engine_free_resource(engine); + return -ENOMEM; +} + +static int engine_init(struct xdma_engine *engine, struct xdma_dev *xdev, + int offset, enum dma_data_direction dir, int channel) +{ + int rv; + u32 val; + + dbg_init("channel %d, offset 0x%x, dir %d.\n", channel, offset, dir); + + /* set magic */ + engine->magic = MAGIC_ENGINE; + + engine->channel = channel; + + /* engine interrupt request bit */ + engine->irq_bitmask = (1 << XDMA_ENG_IRQ_NUM) - 1; + engine->irq_bitmask <<= (xdev->engines_num * XDMA_ENG_IRQ_NUM); + engine->bypass_offset = xdev->engines_num * BYPASS_MODE_SPACING; + + /* parent */ + engine->xdev = xdev; + /* register address */ + engine->regs = (xdev->bar[xdev->config_bar_idx] + offset); + engine->sgdma_regs = xdev->bar[xdev->config_bar_idx] + offset + + SGDMA_OFFSET_FROM_CHANNEL; + val = read_register(&engine->regs->identifier); + if (val & 0x8000U) + engine->streaming = 1; + + /* remember SG DMA direction */ + engine->dir = dir; + sprintf(engine->name, "%d-%s%d-%s", xdev->idx, + (dir == DMA_TO_DEVICE) ? "H2C" : "C2H", channel, + engine->streaming ? "ST" : "MM"); + + dbg_init("engine %p name %s irq_bitmask=0x%08x\n", engine, engine->name, + (int)engine->irq_bitmask); + + /* initialize the deferred work for transfer completion */ + INIT_WORK(&engine->work, engine_service_work); + + if (dir == DMA_TO_DEVICE) + xdev->mask_irq_h2c |= engine->irq_bitmask; + else + xdev->mask_irq_c2h |= engine->irq_bitmask; + xdev->engines_num++; + + rv = engine_alloc_resource(engine); + if (rv) + return rv; + + rv = engine_init_regs(engine); + if (rv) + return rv; + + return 0; +} + +/* transfer_destroy() - free transfer */ +static void transfer_destroy(struct xdma_dev *xdev, struct xdma_transfer *xfer) +{ + /* free descriptors */ + xdma_desc_done(xfer->desc_virt); + + if (xfer->last_in_request && (xfer->flags & XFER_FLAG_NEED_UNMAP)) { + struct sg_table *sgt = xfer->sgt; + + if (sgt->nents) { + pci_unmap_sg(xdev->pdev, sgt->sgl, sgt->nents, + xfer->dir); + sgt->nents = 0; + } + } +} + +static int transfer_build(struct xdma_engine *engine, + struct xdma_request_cb *req, unsigned int desc_max) +{ + struct xdma_transfer *xfer = &req->xfer; + struct sw_desc *sdesc = &(req->sdesc[req->sw_desc_idx]); + int i = 0; + int j = 0; + + for (; i < desc_max; i++, j++, sdesc++) { + dbg_desc("sw desc %d/%u: 0x%llx, 0x%x, ep 0x%llx.\n", + i + req->sw_desc_idx, req->sw_desc_cnt, + sdesc->addr, sdesc->len, req->ep_addr); + + /* fill in descriptor entry j with transfer details */ + xdma_desc_set(xfer->desc_virt + j, sdesc->addr, req->ep_addr, + sdesc->len, xfer->dir); + xfer->len += sdesc->len; + + /* for non-inc-add mode don't increment ep_addr */ + if (!engine->non_incr_addr) + req->ep_addr += sdesc->len; + } + req->sw_desc_idx += desc_max; + return 0; +} + +static int transfer_init(struct xdma_engine *engine, struct xdma_request_cb *req) +{ + struct xdma_transfer *xfer = &req->xfer; + unsigned int desc_max = min_t(unsigned int, + req->sw_desc_cnt - req->sw_desc_idx, + XDMA_TRANSFER_MAX_DESC); + int i = 0; + int last = 0; + u32 control; + + memset(xfer, 0, sizeof(*xfer)); + + /* initialize wait queue */ + init_waitqueue_head(&xfer->wq); + + /* remember direction of transfer */ + xfer->dir = engine->dir; + + xfer->desc_virt = engine->desc; + xfer->desc_bus = engine->desc_bus; + + transfer_desc_init(xfer, desc_max); + + dbg_sg("transfer->desc_bus = 0x%llx.\n", (u64)xfer->desc_bus); + + transfer_build(engine, req, desc_max); + + /* terminate last descriptor */ + last = desc_max - 1; + xdma_desc_link(xfer->desc_virt + last, 0, 0); + /* stop engine, EOP for AXI ST, req IRQ on last descriptor */ + control = XDMA_DESC_STOPPED; + control |= XDMA_DESC_EOP; + control |= XDMA_DESC_COMPLETED; + xdma_desc_control_set(xfer->desc_virt + last, control); + + xfer->desc_num = xfer->desc_adjacent = desc_max; + + dbg_sg("transfer 0x%p has %d descriptors\n", xfer, xfer->desc_num); + /* fill in adjacent numbers */ + for (i = 0; i < xfer->desc_num; i++) + xdma_desc_adjacent(xfer->desc_virt + i, xfer->desc_num - i - 1); + + return 0; +} + +#ifdef __LIBXDMA_DEBUG__ +static void sgt_dump(struct sg_table *sgt) +{ + int i; + struct scatterlist *sg = sgt->sgl; + + pr_info("sgt 0x%p, sgl 0x%p, nents %u/%u.\n", + sgt, sgt->sgl, sgt->nents, sgt->orig_nents); + + for (i = 0; i < sgt->orig_nents; i++, sg = sg_next(sg)) + pr_info("%d, 0x%p, pg 0x%p,%u+%u, dma 0x%llx,%u.\n", + i, sg, sg_page(sg), sg->offset, sg->length, + sg_dma_address(sg), sg_dma_len(sg)); +} + +static void xdma_request_cb_dump(struct xdma_request_cb *req) +{ + int i; + + pr_info("request 0x%p, total %u, ep 0x%llx, sw_desc %u, sgt 0x%p.\n", + req, req->total_len, req->ep_addr, req->sw_desc_cnt, req->sgt); + sgt_dump(req->sgt); + for (i = 0; i < req->sw_desc_cnt; i++) + pr_info("%d/%u, 0x%llx, %u.\n", + i, req->sw_desc_cnt, req->sdesc[i].addr, + req->sdesc[i].len); +} +#endif + +static void xdma_request_free(struct xdma_request_cb *req) +{ + if (((unsigned long)req) >= VMALLOC_START && + ((unsigned long)req) < VMALLOC_END) + vfree(req); + else + kfree(req); +} + +static struct xdma_request_cb * xdma_request_alloc(unsigned int sdesc_nr) +{ + struct xdma_request_cb *req; + unsigned int size = sizeof(struct xdma_request_cb) + + sdesc_nr * sizeof(struct sw_desc); + + req = kzalloc(size, GFP_KERNEL); + if (!req) { + req = vmalloc(size); + if (req) + memset(req, 0, size); + } + if (!req) { + pr_info("OOM, %u sw_desc, %u.\n", sdesc_nr, size); + return NULL; + } + + return req; +} + +static struct xdma_request_cb * xdma_init_request(struct sg_table *sgt, + u64 ep_addr) +{ + struct xdma_request_cb *req; + struct scatterlist *sg = sgt->sgl; + int max = sgt->nents; + int extra = 0; + int i, j = 0; + + for (i = 0; i < max; i++, sg = sg_next(sg)) { + unsigned int len = sg_dma_len(sg); + + if (unlikely(len > desc_blen_max)) + extra += (len + desc_blen_max - 1) / desc_blen_max; + } + +//pr_info("ep 0x%llx, desc %u+%u.\n", ep_addr, max, extra); + + max += extra; + req = xdma_request_alloc(max); + if (!req) + return NULL; + + req->sgt = sgt; + req->ep_addr = ep_addr; + + for (i = 0, sg = sgt->sgl; i < sgt->nents; i++, sg = sg_next(sg)) { + unsigned int tlen = sg_dma_len(sg); + dma_addr_t addr = sg_dma_address(sg); + + req->total_len += tlen; + while (tlen) { + req->sdesc[j].addr = addr; + if (tlen > desc_blen_max) { + req->sdesc[j].len = desc_blen_max; + addr += desc_blen_max; + tlen -= desc_blen_max; + } else { + req->sdesc[j].len = tlen; + tlen = 0; + } + j++; + } + } + BUG_ON(j > max); + + req->sw_desc_cnt = j; +#ifdef __LIBXDMA_DEBUG__ + xdma_request_cb_dump(req); +#endif + return req; +} + +ssize_t xdma_xfer_submit(void *dev_hndl, int channel, bool write, u64 ep_addr, + struct sg_table *sgt, bool dma_mapped, int timeout_ms) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + int rv = 0; + ssize_t done = 0; + struct scatterlist *sg = sgt->sgl; + int nents; + enum dma_data_direction dir = write ? DMA_TO_DEVICE : DMA_FROM_DEVICE; + struct xdma_request_cb *req = NULL; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + if (write == 1) { + if (channel >= xdev->h2c_channel_max) { + pr_warn("H2C channel %d >= %d.\n", + channel, xdev->h2c_channel_max); + return -EINVAL; + } + engine = &xdev->engine_h2c[channel]; + } else if (write == 0) { + if (channel >= xdev->c2h_channel_max) { + pr_warn("C2H channel %d >= %d.\n", + channel, xdev->c2h_channel_max); + return -EINVAL; + } + engine = &xdev->engine_c2h[channel]; + } else { + pr_warn("write %d, exp. 0|1.\n", write); + return -EINVAL; + } + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + xdev = engine->xdev; + if (xdma_device_flag_check(xdev, XDEV_FLAG_OFFLINE)) { + pr_info("xdev 0x%p, offline.\n", xdev); + return -EBUSY; + } + + /* check the direction */ + if (engine->dir != dir) { + pr_info("0x%p, %s, %d, W %d, 0x%x/0x%x mismatch.\n", + engine, engine->name, channel, write, engine->dir, dir); + return -EINVAL; + } + + if (!dma_mapped) { + nents = pci_map_sg(xdev->pdev, sg, sgt->orig_nents, dir); + if (!nents) { + pr_info("map sgl failed, sgt 0x%p.\n", sgt); + return -EIO; + } + sgt->nents = nents; + } else { + BUG_ON(!sgt->nents); + } + + req = xdma_init_request(sgt, ep_addr); + if (!req) { + rv = -ENOMEM; + goto unmap_sgl; + } + + dbg_tfr("%s, len %u sg cnt %u.\n", + engine->name, req->total_len, req->sw_desc_cnt); + + sg = sgt->sgl; + nents = req->sw_desc_cnt; + while (nents) { + unsigned long flags; + struct xdma_transfer *xfer; + + /* one transfer at a time */ + spin_lock(&engine->desc_lock); + + /* build transfer */ + rv = transfer_init(engine, req); + if (rv < 0) { + spin_unlock(&engine->desc_lock); + goto unmap_sgl; + } + xfer = &req->xfer; + + if (!dma_mapped) + xfer->flags = XFER_FLAG_NEED_UNMAP; + + /* last transfer for the given request? */ + nents -= xfer->desc_num; + if (!nents) { + xfer->last_in_request = 1; + xfer->sgt = sgt; + } + + dbg_tfr("xfer, %u, ep 0x%llx, done %lu, sg %u/%u.\n", + xfer->len, req->ep_addr, done, req->sw_desc_idx, + req->sw_desc_cnt); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); +#endif + + rv = transfer_queue(engine, xfer); + if (rv < 0) { + spin_unlock(&engine->desc_lock); + pr_info("unable to submit %s, %d.\n", engine->name, rv); + goto unmap_sgl; + } + + /* + * When polling, determine how many descriptors have been queued * on the engine to determine the writeback value expected + */ + if (poll_mode) { + unsigned int desc_count; + + spin_lock_irqsave(&engine->lock, flags); + desc_count = xfer->desc_num; + spin_unlock_irqrestore(&engine->lock, flags); + + dbg_tfr("%s poll desc_count=%d\n", + engine->name, desc_count); + rv = engine_service_poll(engine, desc_count); + + } else { + rv = wait_event_interruptible_timeout(xfer->wq, + (xfer->state != TRANSFER_STATE_SUBMITTED), + msecs_to_jiffies(timeout_ms)); + } + + spin_lock_irqsave(&engine->lock, flags); + + switch(xfer->state) { + case TRANSFER_STATE_COMPLETED: + spin_unlock_irqrestore(&engine->lock, flags); + + dbg_tfr("transfer %p, %u, ep 0x%llx compl, +%lu.\n", + xfer, xfer->len, req->ep_addr - xfer->len, done); + done += xfer->len; + rv = 0; + break; + case TRANSFER_STATE_FAILED: + pr_info("xfer 0x%p,%u, failed, ep 0x%llx.\n", + xfer, xfer->len, req->ep_addr - xfer->len); + spin_unlock_irqrestore(&engine->lock, flags); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); + sgt_dump(sgt); +#endif + rv = -EIO; + break; + default: + /* transfer can still be in-flight */ + pr_info("xfer 0x%p,%u, s 0x%x timed out, ep 0x%llx.\n", + xfer, xfer->len, xfer->state, req->ep_addr); + engine_status_read(engine, 0, 1); + //engine_status_dump(engine); + transfer_abort(engine, xfer); + + xdma_engine_stop(engine); + spin_unlock_irqrestore(&engine->lock, flags); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); + sgt_dump(sgt); +#endif + rv = -ERESTARTSYS; + break; + } + + transfer_destroy(xdev, xfer); + spin_unlock(&engine->desc_lock); + + if (rv < 0) + goto unmap_sgl; + } /* while (sg) */ + +unmap_sgl: + if (!dma_mapped && sgt->nents) { + pci_unmap_sg(xdev->pdev, sgt->sgl, sgt->orig_nents, dir); + sgt->nents = 0; + } + + if (req) + xdma_request_free(req); + + if (rv < 0) + return rv; + + return done; +} +EXPORT_SYMBOL_GPL(xdma_xfer_submit); + +int xdma_performance_submit(struct xdma_dev *xdev, struct xdma_engine *engine) +{ + u8 *buffer_virt; + u32 max_consistent_size = 128 * 32 * 1024; /* 1024 pages, 4MB */ + dma_addr_t buffer_bus; /* bus address */ + struct xdma_transfer *transfer; + u64 ep_addr = 0; + int num_desc_in_a_loop = 128; + int size_in_desc = engine->xdma_perf->transfer_size; + int size = size_in_desc * num_desc_in_a_loop; + int i; + + BUG_ON(size_in_desc > max_consistent_size); + + if (size > max_consistent_size) { + size = max_consistent_size; + num_desc_in_a_loop = size / size_in_desc; + } + + buffer_virt = dma_alloc_coherent(&xdev->pdev->dev, size, + &buffer_bus, GFP_KERNEL); + + /* allocate transfer data structure */ + transfer = kzalloc(sizeof(struct xdma_transfer), GFP_KERNEL); + BUG_ON(!transfer); + + /* 0 = write engine (to_dev=0) , 1 = read engine (to_dev=1) */ + transfer->dir = engine->dir; + /* set number of descriptors */ + transfer->desc_num = num_desc_in_a_loop; + + /* allocate descriptor list */ + if (!engine->desc) { + engine->desc = dma_alloc_coherent(&xdev->pdev->dev, + num_desc_in_a_loop * sizeof(struct xdma_desc), + &engine->desc_bus, GFP_KERNEL); + BUG_ON(!engine->desc); + dbg_init("device %s, engine %s pre-alloc desc 0x%p,0x%llx.\n", + dev_name(&xdev->pdev->dev), engine->name, + engine->desc, engine->desc_bus); + } + transfer->desc_virt = engine->desc; + transfer->desc_bus = engine->desc_bus; + + transfer_desc_init(transfer, transfer->desc_num); + + dbg_sg("transfer->desc_bus = 0x%llx.\n", (u64)transfer->desc_bus); + + for (i = 0; i < transfer->desc_num; i++) { + struct xdma_desc *desc = transfer->desc_virt + i; + dma_addr_t rc_bus_addr = buffer_bus + size_in_desc * i; + + /* fill in descriptor entry with transfer details */ + xdma_desc_set(desc, rc_bus_addr, ep_addr, size_in_desc, + engine->dir); + } + + /* stop engine and request interrupt on last descriptor */ + xdma_desc_control_set(transfer->desc_virt, 0); + /* create a linked loop */ + xdma_desc_link(transfer->desc_virt + transfer->desc_num - 1, + transfer->desc_virt, transfer->desc_bus); + + transfer->cyclic = 1; + + /* initialize wait queue */ + init_waitqueue_head(&transfer->wq); + + //printk("=== Descriptor print for PERF \n"); + //transfer_dump(transfer); + + dbg_perf("Queueing XDMA I/O %s request for performance measurement.\n", + engine->dir ? "write (to dev)" : "read (from dev)"); + transfer_queue(engine, transfer); + return 0; + +} +EXPORT_SYMBOL_GPL(xdma_performance_submit); + +static struct xdma_dev *alloc_dev_instance(struct pci_dev *pdev) +{ + int i; + struct xdma_dev *xdev; + struct xdma_engine *engine; + + BUG_ON(!pdev); + + /* allocate zeroed device book keeping structure */ + xdev = kzalloc(sizeof(struct xdma_dev), GFP_KERNEL); + if (!xdev) { + pr_info("OOM, xdma_dev.\n"); + return NULL; + } + spin_lock_init(&xdev->lock); + + xdev->magic = MAGIC_DEVICE; + xdev->config_bar_idx = -1; + xdev->user_bar_idx = -1; + xdev->bypass_bar_idx = -1; + xdev->irq_line = -1; + + /* create a driver to device reference */ + xdev->pdev = pdev; + dbg_init("xdev = 0x%p\n", xdev); + + /* Set up data user IRQ data structures */ + for (i = 0; i < 16; i++) { + xdev->user_irq[i].xdev = xdev; + spin_lock_init(&xdev->user_irq[i].events_lock); + init_waitqueue_head(&xdev->user_irq[i].events_wq); + xdev->user_irq[i].handler = NULL; + xdev->user_irq[i].user_idx = i; /* 0 based */ + } + + engine = xdev->engine_h2c; + for (i = 0; i < XDMA_CHANNEL_NUM_MAX; i++, engine++) { + spin_lock_init(&engine->lock); + spin_lock_init(&engine->desc_lock); + INIT_LIST_HEAD(&engine->transfer_list); + init_waitqueue_head(&engine->shutdown_wq); + init_waitqueue_head(&engine->xdma_perf_wq); + } + + engine = xdev->engine_c2h; + for (i = 0; i < XDMA_CHANNEL_NUM_MAX; i++, engine++) { + spin_lock_init(&engine->lock); + spin_lock_init(&engine->desc_lock); + INIT_LIST_HEAD(&engine->transfer_list); + init_waitqueue_head(&engine->shutdown_wq); + init_waitqueue_head(&engine->xdma_perf_wq); + } + + return xdev; +} + +static int request_regions(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + int rv; + + BUG_ON(!xdev); + BUG_ON(!pdev); + + dbg_init("pci_request_regions()\n"); + rv = pci_request_regions(pdev, xdev->mod_name); + /* could not request all regions? */ + if (rv) { + dbg_init("pci_request_regions() = %d, device in use?\n", rv); + /* assume device is in use so do not disable it later */ + xdev->regions_in_use = 1; + } else { + xdev->got_regions = 1; + } + + return rv; +} + +static int set_dma_mask(struct pci_dev *pdev) +{ + BUG_ON(!pdev); + + dbg_init("sizeof(dma_addr_t) == %ld\n", sizeof(dma_addr_t)); + /* 64-bit addressing capability for XDMA? */ + if (!pci_set_dma_mask(pdev, DMA_BIT_MASK(64))) { + /* query for DMA transfer */ + /* @see Documentation/DMA-mapping.txt */ + dbg_init("pci_set_dma_mask()\n"); + /* use 64-bit DMA */ + dbg_init("Using a 64-bit DMA mask.\n"); + /* use 32-bit DMA for descriptors */ + pci_set_consistent_dma_mask(pdev, DMA_BIT_MASK(32)); + /* use 64-bit DMA, 32-bit for consistent */ + } else if (!pci_set_dma_mask(pdev, DMA_BIT_MASK(32))) { + dbg_init("Could not set 64-bit DMA mask.\n"); + pci_set_consistent_dma_mask(pdev, DMA_BIT_MASK(32)); + /* use 32-bit DMA */ + dbg_init("Using a 32-bit DMA mask.\n"); + } else { + dbg_init("No suitable DMA possible.\n"); + return -EINVAL; + } + + return 0; +} + +static u32 get_engine_channel_id(struct engine_regs *regs) +{ + u32 value; + + BUG_ON(!regs); + + value = read_register(®s->identifier); + + return (value & 0x00000f00U) >> 8; +} + +static u32 get_engine_id(struct engine_regs *regs) +{ + u32 value; + + BUG_ON(!regs); + + value = read_register(®s->identifier); + return (value & 0xffff0000U) >> 16; +} + +static void remove_engines(struct xdma_dev *xdev) +{ + struct xdma_engine *engine; + int i; + + BUG_ON(!xdev); + + /* iterate over channels */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + if (engine->magic == MAGIC_ENGINE) { + dbg_sg("Remove %s, %d", engine->name, i); + engine_destroy(xdev, engine); + dbg_sg("%s, %d removed", engine->name, i); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + dbg_sg("Remove %s, %d", engine->name, i); + engine_destroy(xdev, engine); + dbg_sg("%s, %d removed", engine->name, i); + } + } +} + +static int probe_for_engine(struct xdma_dev *xdev, enum dma_data_direction dir, + int channel) +{ + struct engine_regs *regs; + int offset = channel * CHANNEL_SPACING; + u32 engine_id; + u32 engine_id_expected; + u32 channel_id; + struct xdma_engine *engine; + int rv; + + /* register offset for the engine */ + /* read channels at 0x0000, write channels at 0x1000, + * channels at 0x100 interval */ + if (dir == DMA_TO_DEVICE) { + engine_id_expected = XDMA_ID_H2C; + engine = &xdev->engine_h2c[channel]; + } else { + offset += H2C_CHANNEL_OFFSET; + engine_id_expected = XDMA_ID_C2H; + engine = &xdev->engine_c2h[channel]; + } + + regs = xdev->bar[xdev->config_bar_idx] + offset; + engine_id = get_engine_id(regs); + channel_id = get_engine_channel_id(regs); + + if ((engine_id != engine_id_expected) || (channel_id != channel)) { + dbg_init("%s %d engine, reg off 0x%x, id mismatch 0x%x,0x%x," + "exp 0x%x,0x%x, SKIP.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", + channel, offset, engine_id, channel_id, + engine_id_expected, channel_id != channel); + return -EINVAL; + } + + dbg_init("found AXI %s %d engine, reg. off 0x%x, id 0x%x,0x%x.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", channel, + offset, engine_id, channel_id); + + /* allocate and initialize engine */ + rv = engine_init(engine, xdev, offset, dir, channel); + if (rv != 0) { + pr_info("failed to create AXI %s %d engine.\n", + dir == DMA_TO_DEVICE ? "H2C" : "C2H", + channel); + return rv; + } + + return 0; +} + +static int probe_engines(struct xdma_dev *xdev) +{ + int i; + int rv = 0; + + BUG_ON(!xdev); + + /* iterate over channels */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + rv = probe_for_engine(xdev, DMA_TO_DEVICE, i); + if (rv) + break; + } + xdev->h2c_channel_max = i; + + for (i = 0; i < xdev->c2h_channel_max; i++) { + rv = probe_for_engine(xdev, DMA_FROM_DEVICE, i); + if (rv) + break; + } + xdev->c2h_channel_max = i; + + return 0; +} + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(3,5,0) +static void pci_enable_relaxed_ordering(struct pci_dev *pdev) +{ + pcie_capability_set_word(pdev, PCI_EXP_DEVCTL, PCI_EXP_DEVCTL_RELAX_EN); +} +#else +static void pci_enable_relaxed_ordering(struct pci_dev *pdev) +{ + u16 v; + int pos; + + pos = pci_pcie_cap(pdev); + if (pos > 0) { + pci_read_config_word(pdev, pos + PCI_EXP_DEVCTL, &v); + v |= PCI_EXP_DEVCTL_RELAX_EN; + pci_write_config_word(pdev, pos + PCI_EXP_DEVCTL, v); + } +} +#endif + +static void pci_check_extended_tag(struct xdma_dev *xdev, struct pci_dev *pdev) +{ + u16 cap; + u32 v; + void *__iomem reg; + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(3,5,0) + pcie_capability_read_word(pdev, PCI_EXP_DEVCTL, &cap); +#else + int pos; + + pos = pci_pcie_cap(pdev); + if (pos > 0) + pci_read_config_word(pdev, pos + PCI_EXP_DEVCTL, &cap); + else { + pr_info("pdev 0x%p, unable to access pcie cap.\n", pdev); + return; + } +#endif + + if ((cap & PCI_EXP_DEVCTL_EXT_TAG)) + return; + + /* extended tag not enabled */ + pr_info("0x%p EXT_TAG disabled.\n", pdev); + + if (xdev->config_bar_idx < 0) { + pr_info("pdev 0x%p, xdev 0x%p, config bar UNKNOWN.\n", + pdev, xdev); + return; + } + + reg = xdev->bar[xdev->config_bar_idx] + XDMA_OFS_CONFIG + 0x4C; + v = read_register(reg); + v = (v & 0xFF) | (((u32)32) << 8); + write_register(v, reg, XDMA_OFS_CONFIG + 0x4C); +} + +void *xdma_device_open(const char *mname, struct pci_dev *pdev, int *user_max, + int *h2c_channel_max, int *c2h_channel_max) +{ + struct xdma_dev *xdev = NULL; + int rv = 0; + + pr_info("%s device %s, 0x%p.\n", mname, dev_name(&pdev->dev), pdev); + + /* allocate zeroed device book keeping structure */ + xdev = alloc_dev_instance(pdev); + if (!xdev) + return NULL; + xdev->mod_name = mname; + xdev->user_max = *user_max; + xdev->h2c_channel_max = *h2c_channel_max; + xdev->c2h_channel_max = *c2h_channel_max; + + xdma_device_flag_set(xdev, XDEV_FLAG_OFFLINE); + xdev_list_add(xdev); + + if (xdev->user_max == 0 || xdev->user_max > MAX_USER_IRQ) + xdev->user_max = MAX_USER_IRQ; + if (xdev->h2c_channel_max == 0 || + xdev->h2c_channel_max > XDMA_CHANNEL_NUM_MAX) + xdev->h2c_channel_max = XDMA_CHANNEL_NUM_MAX; + if (xdev->c2h_channel_max == 0 || + xdev->c2h_channel_max > XDMA_CHANNEL_NUM_MAX) + xdev->c2h_channel_max = XDMA_CHANNEL_NUM_MAX; + + rv = pci_enable_device(pdev); + if (rv) { + dbg_init("pci_enable_device() failed, %d.\n", rv); + goto err_enable; + } + + /* keep INTx enabled */ + pci_check_intr_pend(pdev); + + /* enable relaxed ordering */ + pci_enable_relaxed_ordering(pdev); + + pci_check_extended_tag(xdev, pdev); + + /* force MRRS to be 512 */ + rv = pcie_set_readrq(pdev, 512); + if (rv) + pr_info("device %s, error set PCI_EXP_DEVCTL_READRQ: %d.\n", + dev_name(&pdev->dev), rv); + + /* enable bus master capability */ + pci_set_master(pdev); + + rv = request_regions(xdev, pdev); + if (rv) + goto err_regions; + + rv = map_bars(xdev, pdev); + if (rv) + goto err_map; + + rv = set_dma_mask(pdev); + if (rv) + goto err_mask; + + check_nonzero_interrupt_status(xdev); + /* explicitely zero all interrupt enable masks */ + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + + rv = probe_engines(xdev); + if (rv) + goto err_engines; + + rv = enable_msi_msix(xdev, pdev); + if (rv < 0) + goto err_enable_msix; + + rv = irq_setup(xdev, pdev); + if (rv < 0) + goto err_interrupts; + + if (!poll_mode) + channel_interrupts_enable(xdev, ~0); + + /* Flush writes */ + read_interrupts(xdev); + + *user_max = xdev->user_max; + *h2c_channel_max = xdev->h2c_channel_max; + *c2h_channel_max = xdev->c2h_channel_max; + + xdma_device_flag_clear(xdev, XDEV_FLAG_OFFLINE); + return (void *)xdev; + +err_interrupts: + irq_teardown(xdev); +err_enable_msix: + disable_msi_msix(xdev, pdev); +err_engines: + remove_engines(xdev); +err_mask: + unmap_bars(xdev, pdev); +err_map: + if (xdev->got_regions) + pci_release_regions(pdev); +err_regions: + if (!xdev->regions_in_use) + pci_disable_device(pdev); +err_enable: + xdev_list_remove(xdev); + kfree(xdev); + return NULL; +} +EXPORT_SYMBOL_GPL(xdma_device_open); + +void xdma_device_close(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + dbg_init("pdev 0x%p, xdev 0x%p.\n", pdev, dev_hndl); + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + + dbg_sg("remove(dev = 0x%p) where pdev->dev.driver_data = 0x%p\n", + pdev, xdev); + if (xdev->pdev != pdev) { + dbg_sg("pci_dev(0x%lx) != pdev(0x%lx)\n", + (unsigned long)xdev->pdev, (unsigned long)pdev); + } + + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + + irq_teardown(xdev); + disable_msi_msix(xdev, pdev); + + remove_engines(xdev); + unmap_bars(xdev, pdev); + + if (xdev->got_regions) { + dbg_init("pci_release_regions 0x%p.\n", pdev); + pci_release_regions(pdev); + } + + if (!xdev->regions_in_use) { + dbg_init("pci_disable_device 0x%p.\n", pdev); + pci_disable_device(pdev); + } + + xdev_list_remove(xdev); + + kfree(xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_close); + +void xdma_device_offline(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + int i; + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + +pr_info("pdev 0x%p, xdev 0x%p.\n", pdev, xdev); + xdma_device_flag_set(xdev, XDEV_FLAG_OFFLINE); + + /* wait for all engines to be idle */ + for (i = 0; i < xdev->h2c_channel_max; i++) { + unsigned long flags; + + engine = &xdev->engine_h2c[i]; + + if (engine->magic == MAGIC_ENGINE) { + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown |= ENGINE_SHUTDOWN_REQUEST; + + xdma_engine_stop(engine); + engine->running = 0; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + unsigned long flags; + + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown |= ENGINE_SHUTDOWN_REQUEST; + + xdma_engine_stop(engine); + engine->running = 0; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + /* turn off interrupts */ + channel_interrupts_disable(xdev, ~0); + user_interrupts_disable(xdev, ~0); + read_interrupts(xdev); + irq_teardown(xdev); + + pr_info("xdev 0x%p, done.\n", xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_offline); + +void xdma_device_online(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + struct xdma_engine *engine; + unsigned long flags; + int i; + + if (!dev_hndl) + return; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return; + +pr_info("pdev 0x%p, xdev 0x%p.\n", pdev, xdev); + + for (i = 0; i < xdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + if (engine->magic == MAGIC_ENGINE) { + engine_init_regs(engine); + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown &= ~ENGINE_SHUTDOWN_REQUEST; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + for (i = 0; i < xdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + if (engine->magic == MAGIC_ENGINE) { + engine_init_regs(engine); + spin_lock_irqsave(&engine->lock, flags); + engine->shutdown &= ~ENGINE_SHUTDOWN_REQUEST; + spin_unlock_irqrestore(&engine->lock, flags); + } + } + + /* re-write the interrupt table */ + if (!poll_mode) { + irq_setup(xdev, pdev); + + channel_interrupts_enable(xdev, ~0); + user_interrupts_enable(xdev, xdev->mask_irq_user); + read_interrupts(xdev); + } + + xdma_device_flag_clear(xdev, XDEV_FLAG_OFFLINE); +pr_info("xdev 0x%p, done.\n", xdev); +} +EXPORT_SYMBOL_GPL(xdma_device_online); + +int xdma_device_restart(struct pci_dev *pdev, void *dev_hndl) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, pdev, dev_hndl) < 0) + return -EINVAL; + + pr_info("NOT implemented, 0x%p.\n", xdev); + return -EINVAL; +} +EXPORT_SYMBOL_GPL(xdma_device_restart); + +int xdma_user_isr_register(void *dev_hndl, unsigned int mask, + irq_handler_t handler, void *dev) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + int i; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + for (i = 0; i < xdev->user_max && mask; i++) { + unsigned int bit = (1 << i); + + if ((bit & mask) == 0) + continue; + + mask &= ~bit; + xdev->user_irq[i].handler = handler; + xdev->user_irq[i].dev = dev; + } + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_register); + +int xdma_user_isr_enable(void *dev_hndl, unsigned int mask) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + xdev->mask_irq_user |= mask; + /* enable user interrupts */ + user_interrupts_enable(xdev, mask); + read_interrupts(xdev); + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_enable); + +int xdma_user_isr_disable(void *dev_hndl, unsigned int mask) +{ + struct xdma_dev *xdev = (struct xdma_dev *)dev_hndl; + + if (!dev_hndl) + return -EINVAL; + + if (debug_check_dev_hndl(__func__, xdev->pdev, dev_hndl) < 0) + return -EINVAL; + + xdev->mask_irq_user &= ~mask; + user_interrupts_disable(xdev, mask); + read_interrupts(xdev); + + return 0; +} +EXPORT_SYMBOL_GPL(xdma_user_isr_disable); + +#ifdef __LIBXDMA_MOD__ +static int __init xdma_base_init(void) +{ + printk(KERN_INFO "%s", version); + return 0; +} + +static void __exit xdma_base_exit(void) +{ + return; +} + +module_init(xdma_base_init); +module_exit(xdma_base_exit); +#endif +/* makes an existing transfer cyclic */ +static void xdma_transfer_cyclic(struct xdma_transfer *transfer) +{ + /* link last descriptor to first descriptor */ + xdma_desc_link(transfer->desc_virt + transfer->desc_num - 1, + transfer->desc_virt, transfer->desc_bus); + /* remember transfer is cyclic */ + transfer->cyclic = 1; +} + +static int transfer_monitor_cyclic(struct xdma_engine *engine, + struct xdma_transfer *transfer, int timeout_ms) +{ + struct xdma_result *result; + int rc = 0; + + BUG_ON(!engine); + BUG_ON(!transfer); + + result = engine->cyclic_result; + BUG_ON(!result); + + if (poll_mode) { + int i ; + for (i = 0; i < 5; i++) { + rc = engine_service_poll(engine, 0); + if (rc) { + pr_info("%s service_poll failed %d.\n", + engine->name, rc); + rc = -ERESTARTSYS; + } + if (result[engine->rx_head].status) + return 0; + } + } else { + if (enable_credit_mp){ + dbg_tfr("%s: rx_head=%d,rx_tail=%d, wait ...\n", + engine->name, engine->rx_head, engine->rx_tail); + rc = wait_event_interruptible_timeout( transfer->wq, + (engine->rx_head!=engine->rx_tail || + engine->rx_overrun), + msecs_to_jiffies(timeout_ms)); + dbg_tfr("%s: wait returns %d, rx %d/%d, overrun %d.\n", + engine->name, rc, engine->rx_head, + engine->rx_tail, engine->rx_overrun); + } else { + rc = wait_event_interruptible_timeout( transfer->wq, + engine->eop_found, + msecs_to_jiffies(timeout_ms)); + dbg_tfr("%s: wait returns %d, eop_found %d.\n", + engine->name, rc, engine->eop_found); + } + } + + return 0; +} + +struct scatterlist *sglist_index(struct sg_table *sgt, unsigned int idx) +{ + struct scatterlist *sg = sgt->sgl; + int i; + + if (idx >= sgt->orig_nents) + return NULL; + + if (!idx) + return sg; + + for (i = 0; i < idx; i++, sg = sg_next(sg)) + ; + + return sg; +} + +static int copy_cyclic_to_user(struct xdma_engine *engine, int pkt_length, + int head, char __user *buf, size_t count) +{ + struct scatterlist *sg; + int more = pkt_length; + + BUG_ON(!engine); + BUG_ON(!buf); + + dbg_tfr("%s, pkt_len %d, head %d, user buf idx %u.\n", + engine->name, pkt_length, head, engine->user_buffer_index); + + sg = sglist_index(&engine->cyclic_sgt, head); + if (!sg) { + pr_info("%s, head %d OOR, sgl %u.\n", + engine->name, head, engine->cyclic_sgt.orig_nents); + return -EIO; + } + + /* EOP found? Transfer anything from head to EOP */ + while (more) { + unsigned int copy = more > PAGE_SIZE ? PAGE_SIZE : more; + unsigned int blen = count - engine->user_buffer_index; + int rv; + + if (copy > blen) + copy = blen; + + dbg_tfr("%s sg %d, 0x%p, copy %u to user %u.\n", + engine->name, head, sg, copy, + engine->user_buffer_index); + + rv = copy_to_user(&buf[engine->user_buffer_index], + page_address(sg_page(sg)), copy); + if (rv) { + pr_info("%s copy_to_user %u failed %d\n", + engine->name, copy, rv); + return -EIO; + } + + more -= copy; + engine->user_buffer_index += copy; + + if (engine->user_buffer_index == count) { + /* user buffer used up */ + break; + } + + head++; + if (head >= CYCLIC_RX_PAGES_MAX) { + head = 0; + sg = engine->cyclic_sgt.sgl; + } else + sg = sg_next(sg); + } + + return pkt_length; +} + +static int complete_cyclic(struct xdma_engine *engine, char __user *buf, + size_t count) +{ + struct xdma_result *result; + int pkt_length = 0; + int fault = 0; + int eop = 0; + int head; + int rc = 0; + int num_credit = 0; + unsigned long flags; + + BUG_ON(!engine); + result = engine->cyclic_result; + BUG_ON(!result); + + spin_lock_irqsave(&engine->lock, flags); + + /* where the host currently is in the ring buffer */ + head = engine->rx_head; + + /* iterate over newly received results */ + while (engine->rx_head != engine->rx_tail||engine->rx_overrun) { + + WARN_ON(result[engine->rx_head].status==0); + + dbg_tfr("%s, result[%d].status = 0x%x length = 0x%x.\n", + engine->name, engine->rx_head, + result[engine->rx_head].status, + result[engine->rx_head].length); + + if ((result[engine->rx_head].status >> 16) != C2H_WB) { + pr_info("%s, result[%d].status 0x%x, no magic.\n", + engine->name, engine->rx_head, + result[engine->rx_head].status); + fault = 1; + } else if (result[engine->rx_head].length > PAGE_SIZE) { + pr_info("%s, result[%d].len 0x%x, > PAGE_SIZE 0x%lx.\n", + engine->name, engine->rx_head, + result[engine->rx_head].length, PAGE_SIZE); + fault = 1; + } else if (result[engine->rx_head].length == 0) { + pr_info("%s, result[%d].length 0x%x.\n", + engine->name, engine->rx_head, + result[engine->rx_head].length); + fault = 1; + /* valid result */ + } else { + pkt_length += result[engine->rx_head].length; + num_credit++; + /* seen eop? */ + //if (result[engine->rx_head].status & RX_STATUS_EOP) + if (result[engine->rx_head].status & RX_STATUS_EOP){ + eop = 1; + engine->eop_found = 1; + } + + dbg_tfr("%s, pkt_length=%d (%s)\n", + engine->name, pkt_length, + eop ? "with EOP" : "no EOP yet"); + } + /* clear result */ + result[engine->rx_head].status = 0; + result[engine->rx_head].length = 0; + /* proceed head pointer so we make progress, even when fault */ + engine->rx_head = (engine->rx_head + 1) % CYCLIC_RX_PAGES_MAX; + + /* stop processing if a fault/eop was detected */ + if (fault || eop){ + break; + } + } + + spin_unlock_irqrestore(&engine->lock, flags); + + if (fault) + return -EIO; + + rc = copy_cyclic_to_user(engine, pkt_length, head, buf, count); + engine->rx_overrun = 0; + /* if copy is successful, release credits */ + if(rc > 0) + write_register(num_credit,&engine->sgdma_regs->credits, 0); + + return rc; +} + +ssize_t xdma_engine_read_cyclic(struct xdma_engine *engine, char __user *buf, + size_t count, int timeout_ms) +{ + int i = 0; + int rc = 0; + int rc_len = 0; + struct xdma_transfer *transfer; + + BUG_ON(!engine); + BUG_ON(engine->magic != MAGIC_ENGINE); + + transfer = &engine->cyclic_req->xfer; + BUG_ON(!transfer); + + engine->user_buffer_index = 0; + + do { + rc = transfer_monitor_cyclic(engine, transfer, timeout_ms); + if (rc < 0) + return rc; + rc = complete_cyclic(engine, buf, count); + if (rc < 0) + return rc; + rc_len += rc; + + i++; + if (i > 10) + break; + } while (!engine->eop_found); + + if(enable_credit_mp) + engine->eop_found = 0; + + return rc_len; +} + +static void sgt_free_with_pages(struct sg_table *sgt, int dir, + struct pci_dev *pdev) +{ + struct scatterlist *sg = sgt->sgl; + int npages = sgt->orig_nents; + int i; + + for (i = 0; i < npages; i++, sg = sg_next(sg)) { + struct page *pg = sg_page(sg); + dma_addr_t bus = sg_dma_address(sg); + + if (pg) { + if (pdev) + pci_unmap_page(pdev, bus, PAGE_SIZE, dir); + __free_page(pg); + } else + break; + } + sg_free_table(sgt); + memset(sgt, 0, sizeof(struct sg_table)); +} + +static int sgt_alloc_with_pages(struct sg_table *sgt, unsigned int npages, + int dir, struct pci_dev *pdev) +{ + struct scatterlist *sg; + int i; + + if (sg_alloc_table(sgt, npages, GFP_KERNEL)) { + pr_info("sgt OOM.\n"); + return -ENOMEM; + } + + sg = sgt->sgl; + for (i = 0; i < npages; i++, sg = sg_next(sg)) { + struct page *pg = alloc_page(GFP_KERNEL); + + if (!pg) { + pr_info("%d/%u, page OOM.\n", i, npages); + goto err_out; + } + + if (pdev) { + dma_addr_t bus = pci_map_page(pdev, pg, 0, PAGE_SIZE, + dir); + if (unlikely(pci_dma_mapping_error(pdev, bus))) { + pr_info("%d/%u, page 0x%p map err.\n", + i, npages, pg); + __free_page(pg); + goto err_out; + } + sg_dma_address(sg) = bus; + sg_dma_len(sg) = PAGE_SIZE; + } + sg_set_page(sg, pg, PAGE_SIZE, 0); + } + + sgt->orig_nents = sgt->nents = npages; + + return 0; + +err_out: + sgt_free_with_pages(sgt, dir, pdev); + return -ENOMEM; +} + +/* + * !NOTE! reference/demo purpose only + * xdma_cyclic_transfer_setup is used for streaming C2H transfers: + * - A list of buffers are pre-allocated for incoming streaming data + * - the ring of the buffers is allowed to wrap around + */ +int xdma_cyclic_transfer_setup(struct xdma_engine *engine) +{ + struct xdma_dev *xdev; + struct xdma_transfer *xfer; + dma_addr_t bus; + unsigned long flags; + int i; + int rc; + + BUG_ON(!engine); + xdev = engine->xdev; + BUG_ON(!xdev); + + if (engine->cyclic_req) { + //pr_info("%s: exclusive access already taken.\n", + //engine->name); + return -EBUSY; + } + + spin_lock_irqsave(&engine->lock, flags); + + engine->rx_tail = 0; + engine->rx_head = 0; + engine->rx_overrun = 0; + engine->eop_found = 0; + + rc = sgt_alloc_with_pages(&engine->cyclic_sgt, CYCLIC_RX_PAGES_MAX, + engine->dir, xdev->pdev); + if (rc < 0) { + pr_info("%s cyclic pages %u OOM.\n", + engine->name, CYCLIC_RX_PAGES_MAX); + goto err_out; + } + + engine->cyclic_req = xdma_init_request(&engine->cyclic_sgt, 0); + if (!engine->cyclic_req) { + pr_info("%s cyclic request OOM.\n", engine->name); + rc = -ENOMEM; + goto err_out; + } + +#ifdef __LIBXDMA_DEBUG__ + xdma_request_cb_dump(engine->cyclic_req); +#endif + + rc = transfer_init(engine, engine->cyclic_req); + if (rc < 0) + goto err_out; + + xfer = &engine->cyclic_req->xfer; + + /* replace source addresses with result write-back addresses */ + memset(engine->cyclic_result, 0, + CYCLIC_RX_PAGES_MAX * sizeof(struct xdma_result)); + bus = engine->cyclic_result_bus; + for (i = 0; i < xfer->desc_num; i++) { + xfer->desc_virt[i].src_addr_lo = cpu_to_le32(PCI_DMA_L(bus)); + xfer->desc_virt[i].src_addr_hi = cpu_to_le32(PCI_DMA_H(bus)); + bus += sizeof(struct xdma_result); + } + /* set control of all descriptors */ + for (i = 0; i < xfer->desc_num; i++) { + xdma_desc_control_clear(xfer->desc_virt + i, LS_BYTE_MASK); + xdma_desc_control_set(xfer->desc_virt + i, + XDMA_DESC_EOP | XDMA_DESC_COMPLETED); + } + + /* make this a cyclic transfer */ + xdma_transfer_cyclic(xfer); + +#ifdef __LIBXDMA_DEBUG__ + transfer_dump(xfer); +#endif + + if(enable_credit_mp){ + //write_register(RX_BUF_PAGES,&engine->sgdma_regs->credits); + write_register(128, &engine->sgdma_regs->credits, 0); + } + + spin_unlock_irqrestore(&engine->lock, flags); + + /* start cyclic transfer */ + transfer_queue(engine, xfer); + + return 0; + + /* unwind on errors */ +err_out: + if (engine->cyclic_req) { + xdma_request_free(engine->cyclic_req); + engine->cyclic_req = NULL; + } + + if (engine->cyclic_sgt.orig_nents) { + sgt_free_with_pages(&engine->cyclic_sgt, engine->dir, + xdev->pdev); + engine->cyclic_sgt.orig_nents = 0; + engine->cyclic_sgt.nents = 0; + engine->cyclic_sgt.sgl = NULL; + } + + spin_unlock_irqrestore(&engine->lock, flags); + + return rc; +} + + +static int cyclic_shutdown_polled(struct xdma_engine *engine) +{ + BUG_ON(!engine); + + spin_lock(&engine->lock); + + dbg_tfr("Polling for shutdown completion\n"); + do { + engine_status_read(engine, 1, 0); + schedule(); + } while (engine->status & XDMA_STAT_BUSY); + + if ((engine->running) && !(engine->status & XDMA_STAT_BUSY)) { + dbg_tfr("Engine has stopped\n"); + + if (!list_empty(&engine->transfer_list)) + engine_transfer_dequeue(engine); + + engine_service_shutdown(engine); + } + + dbg_tfr("Shutdown completion polling done\n"); + spin_unlock(&engine->lock); + + return 0; +} + +static int cyclic_shutdown_interrupt(struct xdma_engine *engine) +{ + int rc; + + BUG_ON(!engine); + + rc = wait_event_interruptible_timeout(engine->shutdown_wq, + !engine->running, msecs_to_jiffies(10000)); + +#if 0 + if (rc) { + dbg_tfr("wait_event_interruptible=%d\n", rc); + return rc; + } +#endif + + if (engine->running) { + pr_info("%s still running?!, %d\n", engine->name, rc); + return -EINVAL; + } + + return rc; +} + +int xdma_cyclic_transfer_teardown(struct xdma_engine *engine) +{ + int rc; + struct xdma_dev *xdev = engine->xdev; + struct xdma_transfer *transfer; + unsigned long flags; + + transfer = engine_cyclic_stop(engine); + + spin_lock_irqsave(&engine->lock, flags); + if (transfer) { + dbg_tfr("%s: stop transfer 0x%p.\n", engine->name, transfer); + if (transfer != &engine->cyclic_req->xfer) { + pr_info("%s unexpected transfer 0x%p/0x%p\n", + engine->name, transfer, + &engine->cyclic_req->xfer); + } + } + /* allow engine to be serviced after stop request */ + spin_unlock_irqrestore(&engine->lock, flags); + + /* wait for engine to be no longer running */ + if (poll_mode) + rc = cyclic_shutdown_polled(engine); + else + rc = cyclic_shutdown_interrupt(engine); + + /* obtain spin lock to atomically remove resources */ + spin_lock_irqsave(&engine->lock, flags); + + if (engine->cyclic_req) { + xdma_request_free(engine->cyclic_req); + engine->cyclic_req = NULL; + } + + if (engine->cyclic_sgt.orig_nents) { + sgt_free_with_pages(&engine->cyclic_sgt, engine->dir, + xdev->pdev); + engine->cyclic_sgt.orig_nents = 0; + engine->cyclic_sgt.nents = 0; + engine->cyclic_sgt.sgl = NULL; + } + + spin_unlock_irqrestore(&engine->lock, flags); + + return 0; +} + +int engine_addrmode_set(struct xdma_engine *engine, unsigned long arg) +{ + int rv; + unsigned long dst; + u32 w = XDMA_CTRL_NON_INCR_ADDR; + + dbg_perf("IOCTL_XDMA_ADDRMODE_SET\n"); + rv = get_user(dst, (int __user *)arg); + + if (rv == 0) { + engine->non_incr_addr = !!dst; + if (engine->non_incr_addr) + write_register(w, &engine->regs->control_w1s, + (unsigned long)(&engine->regs->control_w1s) - + (unsigned long)(&engine->regs)); + else + write_register(w, &engine->regs->control_w1c, + (unsigned long)(&engine->regs->control_w1c) - + (unsigned long)(&engine->regs)); + } + engine_alignments(engine); + + return rv; +} + diff --git a/sources/xdma_driver/xdma/libxdma.h b/sources/xdma_driver/xdma/libxdma.h new file mode 100644 index 0000000..bbf1a45 --- /dev/null +++ b/sources/xdma_driver/xdma/libxdma.h @@ -0,0 +1,601 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef XDMA_LIB_H +#define XDMA_LIB_H + +#include <linux/version.h> +#include <linux/types.h> +#include <linux/uaccess.h> +#include <linux/module.h> +#include <linux/dma-mapping.h> +#include <linux/init.h> +#include <linux/interrupt.h> +#include <linux/jiffies.h> +#include <linux/kernel.h> +#include <linux/pci.h> +#include <linux/workqueue.h> + +/* Switch debug printing on/off */ +#define XDMA_DEBUG 0 + +/* SECTION: Preprocessor macros/constants */ +#define XDMA_BAR_NUM (6) + +/* maximum amount of register space to map */ +#define XDMA_BAR_SIZE (0x8000UL) + +/* Use this definition to poll several times between calls to schedule */ +#define NUM_POLLS_PER_SCHED 100 + +#define XDMA_CHANNEL_NUM_MAX (4) +/* + * interrupts per engine, rad2_vul.sv:237 + * .REG_IRQ_OUT (reg_irq_from_ch[(channel*2) +: 2]), + */ +#define XDMA_ENG_IRQ_NUM (1) +#define MAX_EXTRA_ADJ (15) +#define RX_STATUS_EOP (1) + +/* Target internal components on XDMA control BAR */ +#define XDMA_OFS_INT_CTRL (0x2000UL) +#define XDMA_OFS_CONFIG (0x3000UL) + +/* maximum number of desc per transfer request */ +#define XDMA_TRANSFER_MAX_DESC (2048) + +/* maximum size of a single DMA transfer descriptor */ +#define XDMA_DESC_BLEN_BITS 28 +#define XDMA_DESC_BLEN_MAX ((1 << (XDMA_DESC_BLEN_BITS)) - 1) + +/* bits of the SG DMA control register */ +#define XDMA_CTRL_RUN_STOP (1UL << 0) +#define XDMA_CTRL_IE_DESC_STOPPED (1UL << 1) +#define XDMA_CTRL_IE_DESC_COMPLETED (1UL << 2) +#define XDMA_CTRL_IE_DESC_ALIGN_MISMATCH (1UL << 3) +#define XDMA_CTRL_IE_MAGIC_STOPPED (1UL << 4) +#define XDMA_CTRL_IE_IDLE_STOPPED (1UL << 6) +#define XDMA_CTRL_IE_READ_ERROR (0x1FUL << 9) +#define XDMA_CTRL_IE_DESC_ERROR (0x1FUL << 19) +#define XDMA_CTRL_NON_INCR_ADDR (1UL << 25) +#define XDMA_CTRL_POLL_MODE_WB (1UL << 26) + +/* bits of the SG DMA status register */ +#define XDMA_STAT_BUSY (1UL << 0) +#define XDMA_STAT_DESC_STOPPED (1UL << 1) +#define XDMA_STAT_DESC_COMPLETED (1UL << 2) +#define XDMA_STAT_ALIGN_MISMATCH (1UL << 3) +#define XDMA_STAT_MAGIC_STOPPED (1UL << 4) +#define XDMA_STAT_INVALID_LEN (1UL << 5) +#define XDMA_STAT_IDLE_STOPPED (1UL << 6) + +#define XDMA_STAT_COMMON_ERR_MASK \ + (XDMA_STAT_ALIGN_MISMATCH | XDMA_STAT_MAGIC_STOPPED | \ + XDMA_STAT_INVALID_LEN) + +/* desc_error, C2H & H2C */ +#define XDMA_STAT_DESC_UNSUPP_REQ (1UL << 19) +#define XDMA_STAT_DESC_COMPL_ABORT (1UL << 20) +#define XDMA_STAT_DESC_PARITY_ERR (1UL << 21) +#define XDMA_STAT_DESC_HEADER_EP (1UL << 22) +#define XDMA_STAT_DESC_UNEXP_COMPL (1UL << 23) + +#define XDMA_STAT_DESC_ERR_MASK \ + (XDMA_STAT_DESC_UNSUPP_REQ | XDMA_STAT_DESC_COMPL_ABORT | \ + XDMA_STAT_DESC_PARITY_ERR | XDMA_STAT_DESC_HEADER_EP | \ + XDMA_STAT_DESC_UNEXP_COMPL) + +/* read error: H2C */ +#define XDMA_STAT_H2C_R_UNSUPP_REQ (1UL << 9) +#define XDMA_STAT_H2C_R_COMPL_ABORT (1UL << 10) +#define XDMA_STAT_H2C_R_PARITY_ERR (1UL << 11) +#define XDMA_STAT_H2C_R_HEADER_EP (1UL << 12) +#define XDMA_STAT_H2C_R_UNEXP_COMPL (1UL << 13) + +#define XDMA_STAT_H2C_R_ERR_MASK \ + (XDMA_STAT_H2C_R_UNSUPP_REQ | XDMA_STAT_H2C_R_COMPL_ABORT | \ + XDMA_STAT_H2C_R_PARITY_ERR | XDMA_STAT_H2C_R_HEADER_EP | \ + XDMA_STAT_H2C_R_UNEXP_COMPL) + +/* write error, H2C only */ +#define XDMA_STAT_H2C_W_DECODE_ERR (1UL << 14) +#define XDMA_STAT_H2C_W_SLAVE_ERR (1UL << 15) + +#define XDMA_STAT_H2C_W_ERR_MASK \ + (XDMA_STAT_H2C_W_DECODE_ERR | XDMA_STAT_H2C_W_SLAVE_ERR) + +/* read error: C2H */ +#define XDMA_STAT_C2H_R_DECODE_ERR (1UL << 9) +#define XDMA_STAT_C2H_R_SLAVE_ERR (1UL << 10) + +#define XDMA_STAT_C2H_R_ERR_MASK \ + (XDMA_STAT_C2H_R_DECODE_ERR | XDMA_STAT_C2H_R_SLAVE_ERR) + +/* all combined */ +#define XDMA_STAT_H2C_ERR_MASK \ + (XDMA_STAT_COMMON_ERR_MASK | XDMA_STAT_DESC_ERR_MASK | \ + XDMA_STAT_H2C_R_ERR_MASK | XDMA_STAT_H2C_W_ERR_MASK) + +#define XDMA_STAT_C2H_ERR_MASK \ + (XDMA_STAT_COMMON_ERR_MASK | XDMA_STAT_DESC_ERR_MASK | \ + XDMA_STAT_C2H_R_ERR_MASK) + +/* bits of the SGDMA descriptor control field */ +#define XDMA_DESC_STOPPED (1UL << 0) +#define XDMA_DESC_COMPLETED (1UL << 1) +#define XDMA_DESC_EOP (1UL << 4) + +#define XDMA_PERF_RUN (1UL << 0) +#define XDMA_PERF_CLEAR (1UL << 1) +#define XDMA_PERF_AUTO (1UL << 2) + +#define MAGIC_ENGINE 0xEEEEEEEEUL +#define MAGIC_DEVICE 0xDDDDDDDDUL + +/* upper 16-bits of engine identifier register */ +#define XDMA_ID_H2C 0x1fc0U +#define XDMA_ID_C2H 0x1fc1U + +/* for C2H AXI-ST mode */ +#define CYCLIC_RX_PAGES_MAX 256 + +#define LS_BYTE_MASK 0x000000FFUL + +#define BLOCK_ID_MASK 0xFFF00000 +#define BLOCK_ID_HEAD 0x1FC00000 + +#define IRQ_BLOCK_ID 0x1fc20000UL +#define CONFIG_BLOCK_ID 0x1fc30000UL + +#define WB_COUNT_MASK 0x00ffffffUL +#define WB_ERR_MASK (1UL << 31) +#define POLL_TIMEOUT_SECONDS 10 + +#define MAX_USER_IRQ 16 + +#define MAX_DESC_BUS_ADDR (0xffffffffULL) + +#define DESC_MAGIC 0xAD4B0000UL + +#define C2H_WB 0x52B4UL + +#define MAX_NUM_ENGINES (XDMA_CHANNEL_NUM_MAX * 2) +#define H2C_CHANNEL_OFFSET 0x1000 +#define SGDMA_OFFSET_FROM_CHANNEL 0x4000 +#define CHANNEL_SPACING 0x100 +#define TARGET_SPACING 0x1000 + +#define BYPASS_MODE_SPACING 0x0100 + +/* obtain the 32 most significant (high) bits of a 32-bit or 64-bit address */ +#define PCI_DMA_H(addr) ((addr >> 16) >> 16) +/* obtain the 32 least significant (low) bits of a 32-bit or 64-bit address */ +#define PCI_DMA_L(addr) (addr & 0xffffffffUL) + +#ifndef VM_RESERVED + #define VMEM_FLAGS (VM_IO | VM_DONTEXPAND | VM_DONTDUMP) +#else + #define VMEM_FLAGS (VM_IO | VM_RESERVED) +#endif + +#ifdef __LIBXDMA_DEBUG__ +#define dbg_io pr_err +#define dbg_fops pr_err +#define dbg_perf pr_err +#define dbg_sg pr_err +#define dbg_tfr pr_err +#define dbg_irq pr_err +#define dbg_init pr_err +#define dbg_desc pr_err +#else +/* disable debugging */ +#define dbg_io(...) +#define dbg_fops(...) +#define dbg_perf(...) +#define dbg_sg(...) +#define dbg_tfr(...) +#define dbg_irq(...) +#define dbg_init(...) +#define dbg_desc(...) +#endif + +/* SECTION: Enum definitions */ +enum transfer_state { + TRANSFER_STATE_NEW = 0, + TRANSFER_STATE_SUBMITTED, + TRANSFER_STATE_COMPLETED, + TRANSFER_STATE_FAILED, + TRANSFER_STATE_ABORTED +}; + +enum shutdown_state { + ENGINE_SHUTDOWN_NONE = 0, /* No shutdown in progress */ + ENGINE_SHUTDOWN_REQUEST = 1, /* engine requested to shutdown */ + ENGINE_SHUTDOWN_IDLE = 2 /* engine has shutdown and is idle */ +}; + +enum dev_capabilities { + CAP_64BIT_DMA = 2, + CAP_64BIT_DESC = 4, + CAP_ENGINE_WRITE = 8, + CAP_ENGINE_READ = 16 +}; + +/* SECTION: Structure definitions */ + +struct config_regs { + u32 identifier; + u32 reserved_1[4]; + u32 msi_enable; +}; + +/** + * SG DMA Controller status and control registers + * + * These registers make the control interface for DMA transfers. + * + * It sits in End Point (FPGA) memory BAR[0] for 32-bit or BAR[0:1] for 64-bit. + * It references the first descriptor which exists in Root Complex (PC) memory. + * + * @note The registers must be accessed using 32-bit (PCI DWORD) read/writes, + * and their values are in little-endian byte ordering. + */ +struct engine_regs { + u32 identifier; + u32 control; + u32 control_w1s; + u32 control_w1c; + u32 reserved_1[12]; /* padding */ + + u32 status; + u32 status_rc; + u32 completed_desc_count; + u32 alignments; + u32 reserved_2[14]; /* padding */ + + u32 poll_mode_wb_lo; + u32 poll_mode_wb_hi; + u32 interrupt_enable_mask; + u32 interrupt_enable_mask_w1s; + u32 interrupt_enable_mask_w1c; + u32 reserved_3[9]; /* padding */ + + u32 perf_ctrl; + u32 perf_cyc_lo; + u32 perf_cyc_hi; + u32 perf_dat_lo; + u32 perf_dat_hi; + u32 perf_pnd_lo; + u32 perf_pnd_hi; +} __packed; + +struct engine_sgdma_regs { + u32 identifier; + u32 reserved_1[31]; /* padding */ + + /* bus address to first descriptor in Root Complex Memory */ + u32 first_desc_lo; + u32 first_desc_hi; + /* number of adjacent descriptors at first_desc */ + u32 first_desc_adjacent; + u32 credits; +} __packed; + +struct msix_vec_table_entry { + u32 msi_vec_addr_lo; + u32 msi_vec_addr_hi; + u32 msi_vec_data_lo; + u32 msi_vec_data_hi; +} __packed; + +struct msix_vec_table { + struct msix_vec_table_entry entry_list[32]; +} __packed; + +struct interrupt_regs { + u32 identifier; + u32 user_int_enable; + u32 user_int_enable_w1s; + u32 user_int_enable_w1c; + u32 channel_int_enable; + u32 channel_int_enable_w1s; + u32 channel_int_enable_w1c; + u32 reserved_1[9]; /* padding */ + + u32 user_int_request; + u32 channel_int_request; + u32 user_int_pending; + u32 channel_int_pending; + u32 reserved_2[12]; /* padding */ + + u32 user_msi_vector[8]; + u32 channel_msi_vector[8]; +} __packed; + +struct sgdma_common_regs { + u32 padding[8]; + u32 credit_mode_enable; + u32 credit_mode_enable_w1s; + u32 credit_mode_enable_w1c; +} __packed; + + +/* Structure for polled mode descriptor writeback */ +struct xdma_poll_wb { + u32 completed_desc_count; + u32 reserved_1[7]; +} __packed; + + +/** + * Descriptor for a single contiguous memory block transfer. + * + * Multiple descriptors are linked by means of the next pointer. An additional + * extra adjacent number gives the amount of extra contiguous descriptors. + * + * The descriptors are in root complex memory, and the bytes in the 32-bit + * words must be in little-endian byte ordering. + */ +struct xdma_desc { + u32 control; + u32 bytes; /* transfer length in bytes */ + u32 src_addr_lo; /* source address (low 32-bit) */ + u32 src_addr_hi; /* source address (high 32-bit) */ + u32 dst_addr_lo; /* destination address (low 32-bit) */ + u32 dst_addr_hi; /* destination address (high 32-bit) */ + /* + * next descriptor in the single-linked list of descriptors; + * this is the PCIe (bus) address of the next descriptor in the + * root complex memory + */ + u32 next_lo; /* next desc address (low 32-bit) */ + u32 next_hi; /* next desc address (high 32-bit) */ +} __packed; + +/* 32 bytes (four 32-bit words) or 64 bytes (eight 32-bit words) */ +struct xdma_result { + u32 status; + u32 length; + u32 reserved_1[6]; /* padding */ +} __packed; + +struct sw_desc { + dma_addr_t addr; + unsigned int len; +}; + +/* Describes a (SG DMA) single transfer for the engine */ +struct xdma_transfer { + struct list_head entry; /* queue of non-completed transfers */ + struct xdma_desc *desc_virt; /* virt addr of the 1st descriptor */ + dma_addr_t desc_bus; /* bus addr of the first descriptor */ + int desc_adjacent; /* adjacent descriptors at desc_bus */ + int desc_num; /* number of descriptors in transfer */ + enum dma_data_direction dir; + wait_queue_head_t wq; /* wait queue for transfer completion */ + + enum transfer_state state; /* state of the transfer */ + unsigned int flags; +#define XFER_FLAG_NEED_UNMAP 0x1 + int cyclic; /* flag if transfer is cyclic */ + int last_in_request; /* flag if last within request */ + unsigned int len; + struct sg_table *sgt; +}; + +struct xdma_request_cb { + struct sg_table *sgt; + unsigned int total_len; + u64 ep_addr; + + struct xdma_transfer xfer; + + unsigned int sw_desc_idx; + unsigned int sw_desc_cnt; + struct sw_desc sdesc[0]; +}; + +struct xdma_engine { + unsigned long magic; /* structure ID for sanity checks */ + struct xdma_dev *xdev; /* parent device */ + char name[5]; /* name of this engine */ + int version; /* version of this engine */ + //dev_t cdevno; /* character device major:minor */ + //struct cdev cdev; /* character device (embedded struct) */ + + /* HW register address offsets */ + struct engine_regs *regs; /* Control reg BAR offset */ + struct engine_sgdma_regs *sgdma_regs; /* SGDAM reg BAR offset */ + u32 bypass_offset; /* Bypass mode BAR offset */ + + /* Engine state, configuration and flags */ + enum shutdown_state shutdown; /* engine shutdown mode */ + enum dma_data_direction dir; + int device_open; /* flag if engine node open, ST mode only */ + int running; /* flag if the driver started engine */ + int non_incr_addr; /* flag if non-incremental addressing used */ + int streaming; + int addr_align; /* source/dest alignment in bytes */ + int len_granularity; /* transfer length multiple */ + int addr_bits; /* HW datapath address width */ + int channel; /* engine indices */ + int max_extra_adj; /* descriptor prefetch capability */ + int desc_dequeued; /* num descriptors of completed transfers */ + u32 status; /* last known status of device */ + u32 interrupt_enable_mask_value;/* only used for MSIX mode to store per-engine interrupt mask value */ + + /* Transfer list management */ + struct list_head transfer_list; /* queue of transfers */ + + /* Members applicable to AXI-ST C2H (cyclic) transfers */ + struct xdma_result *cyclic_result; + dma_addr_t cyclic_result_bus; /* bus addr for transfer */ + struct xdma_request_cb *cyclic_req; + struct sg_table cyclic_sgt; + u8 eop_found; /* used only for cyclic(rx:c2h) */ + + int rx_tail; /* follows the HW */ + int rx_head; /* where the SW reads from */ + int rx_overrun; /* flag if overrun occured */ + + /* for copy from cyclic buffer to user buffer */ + unsigned int user_buffer_index; + + /* Members associated with polled mode support */ + u8 *poll_mode_addr_virt; /* virt addr for descriptor writeback */ + dma_addr_t poll_mode_bus; /* bus addr for descriptor writeback */ + + /* Members associated with interrupt mode support */ + wait_queue_head_t shutdown_wq; /* wait queue for shutdown sync */ + spinlock_t lock; /* protects concurrent access */ + int prev_cpu; /* remember CPU# of (last) locker */ + int msix_irq_line; /* MSI-X vector for this engine */ + u32 irq_bitmask; /* IRQ bit mask for this engine */ + struct work_struct work; /* Work queue for interrupt handling */ + + spinlock_t desc_lock; /* protects concurrent access */ + dma_addr_t desc_bus; + struct xdma_desc *desc; + + /* for performance test support */ + struct xdma_performance_ioctl *xdma_perf; /* perf test control */ + wait_queue_head_t xdma_perf_wq; /* Perf test sync */ +}; + +struct xdma_user_irq { + struct xdma_dev *xdev; /* parent device */ + u8 user_idx; /* 0 ~ 15 */ + u8 events_irq; /* accumulated IRQs */ + spinlock_t events_lock; /* lock to safely update events_irq */ + wait_queue_head_t events_wq; /* wait queue to sync waiting threads */ + irq_handler_t handler; + + void *dev; +}; + +/* XDMA PCIe device specific book-keeping */ +#define XDEV_FLAG_OFFLINE 0x1 +struct xdma_dev { + struct list_head list_head; + struct list_head rcu_node; + + unsigned long magic; /* structure ID for sanity checks */ + struct pci_dev *pdev; /* pci device struct from probe() */ + int idx; /* dev index */ + + const char *mod_name; /* name of module owning the dev */ + + spinlock_t lock; /* protects concurrent access */ + unsigned int flags; + + /* PCIe BAR management */ + void *__iomem bar[XDMA_BAR_NUM]; /* addresses for mapped BARs */ + int user_bar_idx; /* BAR index of user logic */ + int config_bar_idx; /* BAR index of XDMA config logic */ + int bypass_bar_idx; /* BAR index of XDMA bypass logic */ + int regions_in_use; /* flag if dev was in use during probe() */ + int got_regions; /* flag if probe() obtained the regions */ + + int user_max; + int c2h_channel_max; + int h2c_channel_max; + + /* Interrupt management */ + int irq_count; /* interrupt counter */ + int irq_line; /* flag if irq allocated successfully */ + int msi_enabled; /* flag if msi was enabled for the device */ + int msix_enabled; /* flag if msi-x was enabled for the device */ +#if LINUX_VERSION_CODE < KERNEL_VERSION(4,12,0) + struct msix_entry entry[32]; /* msi-x vector/entry table */ +#endif + struct xdma_user_irq user_irq[16]; /* user IRQ management */ + unsigned int mask_irq_user; + + /* XDMA engine management */ + int engines_num; /* Total engine count */ + u32 mask_irq_h2c; + u32 mask_irq_c2h; + struct xdma_engine engine_h2c[XDMA_CHANNEL_NUM_MAX]; + struct xdma_engine engine_c2h[XDMA_CHANNEL_NUM_MAX]; + + /* SD_Accel specific */ + enum dev_capabilities capabilities; + u64 feature_id; +}; + +static inline int xdma_device_flag_check(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + if (xdev->flags & f) { + spin_unlock_irqrestore(&xdev->lock, flags); + return 1; + } + spin_unlock_irqrestore(&xdev->lock, flags); + return 0; +} + +static inline int xdma_device_flag_test_n_set(struct xdma_dev *xdev, + unsigned int f) +{ + unsigned long flags; + int rv = 0; + + spin_lock_irqsave(&xdev->lock, flags); + if (xdev->flags & f) { + spin_unlock_irqrestore(&xdev->lock, flags); + rv = 1; + } else + xdev->flags |= f; + spin_unlock_irqrestore(&xdev->lock, flags); + return rv; +} + +static inline void xdma_device_flag_set(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + xdev->flags |= f; + spin_unlock_irqrestore(&xdev->lock, flags); +} + +static inline void xdma_device_flag_clear(struct xdma_dev *xdev, unsigned int f) +{ + unsigned long flags; + + spin_lock_irqsave(&xdev->lock, flags); + xdev->flags &= ~f; + spin_unlock_irqrestore(&xdev->lock, flags); +} + +void write_register(u32 value, void *iomem); +u32 read_register(void *iomem); + +struct xdma_dev *xdev_find_by_pdev(struct pci_dev *pdev); + +void xdma_device_offline(struct pci_dev *pdev, void *dev_handle); +void xdma_device_online(struct pci_dev *pdev, void *dev_handle); + +int xdma_performance_submit(struct xdma_dev *xdev, struct xdma_engine *engine); +struct xdma_transfer *engine_cyclic_stop(struct xdma_engine *engine); +void enable_perf(struct xdma_engine *engine); +void get_perf_stats(struct xdma_engine *engine); + +int xdma_cyclic_transfer_setup(struct xdma_engine *engine); +int xdma_cyclic_transfer_teardown(struct xdma_engine *engine); +ssize_t xdma_engine_read_cyclic(struct xdma_engine *, char __user *, size_t, + int); +int engine_addrmode_set(struct xdma_engine *engine, unsigned long arg); + +#endif /* XDMA_LIB_H */ diff --git a/sources/xdma_driver/xdma/libxdma.o.ur-safe b/sources/xdma_driver/xdma/libxdma.o.ur-safe new file mode 100644 index 0000000..699074b --- /dev/null +++ b/sources/xdma_driver/xdma/libxdma.o.ur-safe @@ -0,0 +1,10 @@ +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-29fb +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-2a64 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-2ac5 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-2df4 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-39f5 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-3f12 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-4ee8 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-511e +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-5348 +/home/safari/aolgun/SoftMC_DDR4/sources/xdma_driver/xdma/libxdma.o-.text-61ea diff --git a/sources/xdma_driver/xdma/version.h b/sources/xdma_driver/xdma/version.h new file mode 100644 index 0000000..edd3abf --- /dev/null +++ b/sources/xdma_driver/xdma/version.h @@ -0,0 +1,28 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XDMA_VERSION_H__ +#define __XDMA_VERSION_H__ + +#define DRV_MOD_MAJOR 2017 +#define DRV_MOD_MINOR 1 +#define DRV_MOD_PATCHLEVEL 47 + +#define DRV_MODULE_VERSION \ + __stringify(DRV_MOD_MAJOR) "." \ + __stringify(DRV_MOD_MINOR) "." \ + __stringify(DRV_MOD_PATCHLEVEL) + +#define DRV_MOD_VERSION_NUMBER \ + ((DRV_MOD_MAJOR)*1000 + (DRV_MOD_MINOR)*100 + DRV_MOD_PATCHLEVEL) + +#endif /* ifndef __XDMA_VERSION_H__ */ diff --git a/sources/xdma_driver/xdma/xdma_cdev.c b/sources/xdma_driver/xdma/xdma_cdev.c new file mode 100644 index 0000000..b7981f6 --- /dev/null +++ b/sources/xdma_driver/xdma/xdma_cdev.c @@ -0,0 +1,539 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include "xdma_cdev.h" + +struct class *g_xdma_class; + +enum cdev_type { + CHAR_USER, + CHAR_CTRL, + CHAR_XVC, + CHAR_EVENTS, + CHAR_XDMA_H2C, + CHAR_XDMA_C2H, + CHAR_BYPASS_H2C, + CHAR_BYPASS_C2H, + CHAR_BYPASS, +}; + +static const char * const devnode_names[] = { + XDMA_NODE_NAME "%d_user", + XDMA_NODE_NAME "%d_control", + XDMA_NODE_NAME "%d_xvc", + XDMA_NODE_NAME "%d_events_%d", + XDMA_NODE_NAME "%d_h2c_%d", + XDMA_NODE_NAME "%d_c2h_%d", + XDMA_NODE_NAME "%d_bypass_h2c_%d", + XDMA_NODE_NAME "%d_bypass_c2h_%d", + XDMA_NODE_NAME "%d_bypass", +}; + +enum xpdev_flags_bits { + XDF_CDEV_USER, + XDF_CDEV_CTRL, + XDF_CDEV_XVC, + XDF_CDEV_EVENT, + XDF_CDEV_SG, + XDF_CDEV_BYPASS, +}; + +static inline void xpdev_flag_set(struct xdma_pci_dev *xpdev, + enum xpdev_flags_bits fbit) +{ + xpdev->flags |= 1 << fbit; +} + +static inline void xcdev_flag_clear(struct xdma_pci_dev *xpdev, + enum xpdev_flags_bits fbit) +{ + xpdev->flags &= ~(1 << fbit); +} + +static inline int xpdev_flag_test(struct xdma_pci_dev *xpdev, + enum xpdev_flags_bits fbit) +{ + return xpdev->flags & (1 << fbit); +} + +#ifdef __XDMA_SYSFS__ +ssize_t show_device_numbers(struct device *dev, struct device_attribute *attr, + char *buf) +{ + struct xdma_pci_dev *xpdev = (struct xdma_pci_dev *)dev_get_drvdata(dev); + + return snprintf(buf, PAGE_SIZE, "%d\t%d\n", + xpdev->major, xpdev->xdev->idx); +} + +static DEVICE_ATTR(xdma_dev_instance, S_IRUGO, show_device_numbers, NULL); +#endif + +static int config_kobject(struct xdma_cdev *xcdev, enum cdev_type type) +{ + int rv = -EINVAL; + struct xdma_dev *xdev = xcdev->xdev; + struct xdma_engine *engine = xcdev->engine; + + switch (type) { + case CHAR_XDMA_H2C: + case CHAR_XDMA_C2H: + case CHAR_BYPASS_H2C: + case CHAR_BYPASS_C2H: + BUG_ON(!engine); + rv = kobject_set_name(&xcdev->cdev.kobj, devnode_names[type], + xdev->idx, engine->channel); + break; + case CHAR_BYPASS: + case CHAR_USER: + case CHAR_CTRL: + case CHAR_XVC: + rv = kobject_set_name(&xcdev->cdev.kobj, devnode_names[type], + xdev->idx); + break; + case CHAR_EVENTS: + rv = kobject_set_name(&xcdev->cdev.kobj, devnode_names[type], + xdev->idx, xcdev->bar); + break; + default: + pr_warn("%s: UNKNOWN type 0x%x.\n", __func__, type); + break; + } + + if (rv) + pr_err("%s: type 0x%x, failed %d.\n", __func__, type, rv); + return rv; +} + +int xcdev_check(const char *fname, struct xdma_cdev *xcdev, bool check_engine) +{ + struct xdma_dev *xdev; + + if (!xcdev || xcdev->magic != MAGIC_CHAR) { + pr_info("%s, xcdev 0x%p, magic 0x%lx.\n", + fname, xcdev, xcdev ? xcdev->magic : 0xFFFFFFFF); + return -EINVAL; + } + + xdev = xcdev->xdev; + if (!xdev || xdev->magic != MAGIC_DEVICE) { + pr_info("%s, xdev 0x%p, magic 0x%lx.\n", + fname, xdev, xdev ? xdev->magic : 0xFFFFFFFF); + return -EINVAL; + } + + if (check_engine) { + struct xdma_engine *engine = xcdev->engine; + if (!engine || engine->magic != MAGIC_ENGINE) { + pr_info("%s, engine 0x%p, magic 0x%lx.\n", fname, + engine, engine ? engine->magic : 0xFFFFFFFF); + return -EINVAL; + } + } + + return 0; +} + +int char_open(struct inode *inode, struct file *file) +{ + struct xdma_cdev *xcdev; + + /* pointer to containing structure of the character device inode */ + xcdev = container_of(inode->i_cdev, struct xdma_cdev, cdev); + BUG_ON(xcdev->magic != MAGIC_CHAR); + /* create a reference to our char device in the opened file */ + file->private_data = xcdev; + + return 0; +} + +/* + * Called when the device goes from used to unused. + */ +int char_close(struct inode *inode, struct file *file) +{ + struct xdma_dev *xdev; + struct xdma_cdev *xcdev = (struct xdma_cdev *)file->private_data; + + BUG_ON(!xcdev); + BUG_ON(xcdev->magic != MAGIC_CHAR); + + /* fetch device specific data stored earlier during open */ + xdev = xcdev->xdev; + BUG_ON(!xdev); + BUG_ON(xdev->magic != MAGIC_DEVICE); + + return 0; +} + +/* create_xcdev() -- create a character device interface to data or control bus + * + * If at least one SG DMA engine is specified, the character device interface + * is coupled to the SG DMA file operations which operate on the data bus. If + * no engines are specified, the interface is coupled with the control bus. + */ + +static int create_sys_device(struct xdma_cdev *xcdev, enum cdev_type type) +{ + struct xdma_dev *xdev = xcdev->xdev; + struct xdma_engine *engine = xcdev->engine; + int last_param; + + if (type == CHAR_EVENTS) + last_param = xcdev->bar; + else + last_param = engine ? engine->channel : 0; + + xcdev->sys_device = device_create(g_xdma_class, &xdev->pdev->dev, + xcdev->cdevno, NULL, devnode_names[type], xdev->idx, + last_param); + + if (!xcdev->sys_device) { + pr_err("device_create(%s) failed\n", devnode_names[type]); + return -1; + } + + return 0; +} + +static int destroy_xcdev(struct xdma_cdev *cdev) +{ + if (!cdev) { + pr_warn("cdev NULL.\n"); + return 0; + } + if (cdev->magic != MAGIC_CHAR) { + pr_warn("cdev 0x%p magic mismatch 0x%lx\n", cdev, cdev->magic); + return 0; + } + BUG_ON(!cdev->xdev); + BUG_ON(!g_xdma_class); + BUG_ON(!cdev->sys_device); + + if (cdev->sys_device) + device_destroy(g_xdma_class, cdev->cdevno); + + cdev_del(&cdev->cdev); + + return 0; +} + +static int create_xcdev(struct xdma_pci_dev *xpdev, struct xdma_cdev *xcdev, + int bar, struct xdma_engine *engine, + enum cdev_type type) +{ + int rv; + int minor; + struct xdma_dev *xdev = xpdev->xdev; + dev_t dev; + + spin_lock_init(&xcdev->lock); + /* new instance? */ + if (!xpdev->major) { + /* allocate a dynamically allocated char device node */ + int rv = alloc_chrdev_region(&dev, XDMA_MINOR_BASE, + XDMA_MINOR_COUNT, XDMA_NODE_NAME); + + if (rv) { + pr_err("unable to allocate cdev region %d.\n", rv); + return rv; + } + xpdev->major = MAJOR(dev); + } + + /* + * do not register yet, create kobjects and name them, + */ + xcdev->magic = MAGIC_CHAR; + xcdev->cdev.owner = THIS_MODULE; + xcdev->xpdev = xpdev; + xcdev->xdev = xdev; + xcdev->engine = engine; + xcdev->bar = bar; + + rv = config_kobject(xcdev, type); + if (rv < 0) + return rv; + + switch (type) { + case CHAR_USER: + case CHAR_CTRL: + /* minor number is type index for non-SGDMA interfaces */ + minor = type; + cdev_ctrl_init(xcdev); + break; + case CHAR_XVC: + /* minor number is type index for non-SGDMA interfaces */ + minor = type; + cdev_xvc_init(xcdev); + break; + case CHAR_XDMA_H2C: + minor = 32 + engine->channel; + cdev_sgdma_init(xcdev); + break; + case CHAR_XDMA_C2H: + minor = 36 + engine->channel; + cdev_sgdma_init(xcdev); + break; + case CHAR_EVENTS: + minor = 10 + bar; + cdev_event_init(xcdev); + break; + case CHAR_BYPASS_H2C: + minor = 64 + engine->channel; + cdev_bypass_init(xcdev); + break; + case CHAR_BYPASS_C2H: + minor = 68 + engine->channel; + cdev_bypass_init(xcdev); + break; + case CHAR_BYPASS: + minor = 100; + cdev_bypass_init(xcdev); + break; + default: + pr_info("type 0x%x NOT supported.\n", type); + return -EINVAL; + } + xcdev->cdevno = MKDEV(xpdev->major, minor); + + /* bring character device live */ + rv = cdev_add(&xcdev->cdev, xcdev->cdevno, 1); + if (rv < 0) { + pr_err("cdev_add failed %d, type 0x%x.\n", rv, type); + goto unregister_region; + } + + dbg_init("xcdev 0x%p, %u:%u, %s, type 0x%x.\n", + xcdev, xpdev->major, minor, xcdev->cdev.kobj.name, type); + + /* create device on our class */ + if (g_xdma_class) { + rv = create_sys_device(xcdev, type); + if (rv < 0) + goto del_cdev; + } + + return 0; + +del_cdev: + cdev_del(&xcdev->cdev); +unregister_region: + unregister_chrdev_region(dev, XDMA_MINOR_COUNT); + return rv; +} + +void xpdev_destroy_interfaces(struct xdma_pci_dev *xpdev) +{ + int i; + +#ifdef __XDMA_SYSFS__ + device_remove_file(&xpdev->pdev->dev, &dev_attr_xdma_dev_instance); +#endif + + if (xpdev_flag_test(xpdev, XDF_CDEV_SG)) { + /* iterate over channels */ + for (i = 0; i < xpdev->h2c_channel_max; i++) + /* remove SG DMA character device */ + destroy_xcdev(&xpdev->sgdma_h2c_cdev[i]); + for (i = 0; i < xpdev->c2h_channel_max; i++) + destroy_xcdev(&xpdev->sgdma_c2h_cdev[i]); + } + + if (xpdev_flag_test(xpdev, XDF_CDEV_EVENT)) { + for (i = 0; i < xpdev->user_max; i++) + destroy_xcdev(&xpdev->events_cdev[i]); + } + + /* remove control character device */ + if (xpdev_flag_test(xpdev, XDF_CDEV_CTRL)) { + destroy_xcdev(&xpdev->ctrl_cdev); + } + + /* remove user character device */ + if (xpdev_flag_test(xpdev, XDF_CDEV_USER)) { + destroy_xcdev(&xpdev->user_cdev); + } + + if (xpdev_flag_test(xpdev, XDF_CDEV_XVC)) { + destroy_xcdev(&xpdev->xvc_cdev); + } + + if (xpdev_flag_test(xpdev, XDF_CDEV_BYPASS)) { + /* iterate over channels */ + for (i = 0; i < xpdev->h2c_channel_max; i++) + /* remove DMA Bypass character device */ + destroy_xcdev(&xpdev->bypass_h2c_cdev[i]); + for (i = 0; i < xpdev->c2h_channel_max; i++) + destroy_xcdev(&xpdev->bypass_c2h_cdev[i]); + destroy_xcdev(&xpdev->bypass_cdev_base); + } + + if (xpdev->major) + unregister_chrdev_region(MKDEV(xpdev->major, XDMA_MINOR_BASE), XDMA_MINOR_COUNT); +} + +int xpdev_create_interfaces(struct xdma_pci_dev *xpdev) +{ + struct xdma_dev *xdev = xpdev->xdev; + struct xdma_engine *engine; + int i; + int rv = 0; + + /* initialize control character device */ + rv = create_xcdev(xpdev, &xpdev->ctrl_cdev, xdev->config_bar_idx, + NULL, CHAR_CTRL); + if (rv < 0) { + pr_err("create_char(ctrl_cdev) failed\n"); + goto fail; + } + xpdev_flag_set(xpdev, XDF_CDEV_CTRL); + + /* initialize events character device */ + for (i = 0; i < xpdev->user_max; i++) { + rv = create_xcdev(xpdev, &xpdev->events_cdev[i], i, NULL, + CHAR_EVENTS); + if (rv < 0) { + pr_err("create char event %d failed, %d.\n", i, rv); + goto fail; + } + } + xpdev_flag_set(xpdev, XDF_CDEV_EVENT); + + /* iterate over channels */ + for (i = 0; i < xpdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + + if (engine->magic != MAGIC_ENGINE) + continue; + + rv = create_xcdev(xpdev, &xpdev->sgdma_h2c_cdev[i], i, engine, + CHAR_XDMA_H2C); + if (rv < 0) { + pr_err("create char h2c %d failed, %d.\n", i, rv); + goto fail; + } + } + + for (i = 0; i < xpdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + + if (engine->magic != MAGIC_ENGINE) + continue; + + rv = create_xcdev(xpdev, &xpdev->sgdma_c2h_cdev[i], i, engine, + CHAR_XDMA_C2H); + if (rv < 0) { + pr_err("create char c2h %d failed, %d.\n", i, rv); + goto fail; + } + } + xpdev_flag_set(xpdev, XDF_CDEV_SG); + + /* ??? Bypass */ + /* Initialize Bypass Character Device */ + if (xdev->bypass_bar_idx > 0){ + for (i = 0; i < xpdev->h2c_channel_max; i++) { + engine = &xdev->engine_h2c[i]; + + if (engine->magic != MAGIC_ENGINE) + continue; + + rv = create_xcdev(xpdev, &xpdev->bypass_h2c_cdev[i], i, + engine, CHAR_BYPASS_H2C); + if (rv < 0) { + pr_err("create h2c %d bypass I/F failed, %d.\n", + i, rv); + goto fail; + } + } + + for (i = 0; i < xpdev->c2h_channel_max; i++) { + engine = &xdev->engine_c2h[i]; + + if (engine->magic != MAGIC_ENGINE) + continue; + + rv = create_xcdev(xpdev, &xpdev->bypass_c2h_cdev[i], i, + engine, CHAR_BYPASS_C2H); + if (rv < 0) { + pr_err("create c2h %d bypass I/F failed, %d.\n", + i, rv); + goto fail; + } + } + + rv = create_xcdev(xpdev, &xpdev->bypass_cdev_base, + xdev->bypass_bar_idx, NULL, CHAR_BYPASS); + if (rv < 0) { + pr_err("create bypass failed %d.\n", rv); + goto fail; + } + xpdev_flag_set(xpdev, XDF_CDEV_BYPASS); + } + + /* initialize user character device */ + if (xdev->user_bar_idx >= 0) { + rv = create_xcdev(xpdev, &xpdev->user_cdev, xdev->user_bar_idx, + NULL, CHAR_USER); + if (rv < 0) { + pr_err("create_char(user_cdev) failed\n"); + goto fail; + } + xpdev_flag_set(xpdev, XDF_CDEV_USER); + + /* xvc */ + rv = create_xcdev(xpdev, &xpdev->xvc_cdev, xdev->user_bar_idx, + NULL, CHAR_XVC); + if (rv < 0) { + pr_err("create xvc failed, %d.\n", rv); + goto fail; + } + xpdev_flag_set(xpdev, XDF_CDEV_XVC); + } + +#ifdef __XDMA_SYSFS__ + /* sys file */ + rv = device_create_file(&xpdev->pdev->dev, + &dev_attr_xdma_dev_instance); + if (rv) { + pr_err("Failed to create device file \n"); + goto fail; + } +#endif + + return 0; + +fail: + rv = -1; + xpdev_destroy_interfaces(xpdev); + return rv; +} + +int xdma_cdev_init(void) +{ + g_xdma_class = class_create(THIS_MODULE, XDMA_NODE_NAME); + if (IS_ERR(g_xdma_class)) { + dbg_init(XDMA_NODE_NAME ": failed to create class"); + return -1; + } + + return 0; +} + +void xdma_cdev_cleanup(void) +{ + if (g_xdma_class) + class_destroy(g_xdma_class); +} diff --git a/sources/xdma_driver/xdma/xdma_cdev.h b/sources/xdma_driver/xdma/xdma_cdev.h new file mode 100644 index 0000000..9570066 --- /dev/null +++ b/sources/xdma_driver/xdma/xdma_cdev.h @@ -0,0 +1,44 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XDMA_CHRDEV_H__ +#define __XDMA_CHRDEV_H__ + +#include <linux/kernel.h> +#include <linux/types.h> +#include <linux/uaccess.h> +#include <linux/errno.h> +#include "xdma_mod.h" + +#define XDMA_NODE_NAME "xdma" +#define XDMA_MINOR_BASE (0) +#define XDMA_MINOR_COUNT (255) + +void xdma_cdev_cleanup(void); +int xdma_cdev_init(void); + +int char_open(struct inode *inode, struct file *file); +int char_close(struct inode *inode, struct file *file); +int xcdev_check(const char *, struct xdma_cdev *, bool); + +void cdev_ctrl_init(struct xdma_cdev *xcdev); +void cdev_xvc_init(struct xdma_cdev *xcdev); +void cdev_event_init(struct xdma_cdev *xcdev); +void cdev_sgdma_init(struct xdma_cdev *xcdev); +void cdev_bypass_init(struct xdma_cdev *xcdev); + +void xpdev_destroy_interfaces(struct xdma_pci_dev *xpdev); +int xpdev_create_interfaces(struct xdma_pci_dev *xpdev); + +int bridge_mmap(struct file *file, struct vm_area_struct *vma); + +#endif /* __XDMA_CHRDEV_H__ */ diff --git a/sources/xdma_driver/xdma/xdma_mod.c b/sources/xdma_driver/xdma/xdma_mod.c new file mode 100644 index 0000000..d6f724a --- /dev/null +++ b/sources/xdma_driver/xdma/xdma_mod.c @@ -0,0 +1,347 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#define pr_fmt(fmt) KBUILD_MODNAME ":%s: " fmt, __func__ + +#include <linux/ioctl.h> +#include <linux/types.h> +#include <linux/errno.h> +#include <linux/aer.h> +/* include early, to verify it depends only on the headers above */ +#include "libxdma_api.h" +#include "libxdma.h" +#include "xdma_mod.h" +#include "xdma_cdev.h" +#include "version.h" + +#define DRV_MODULE_NAME "xdma" +#define DRV_MODULE_DESC "Xilinx XDMA Reference Driver" +#define DRV_MODULE_RELDATE "Feb. 2018" + +static char version[] = + DRV_MODULE_DESC " " DRV_MODULE_NAME " v" DRV_MODULE_VERSION "\n"; + +MODULE_AUTHOR("Xilinx, Inc."); +MODULE_DESCRIPTION(DRV_MODULE_DESC); +MODULE_VERSION(DRV_MODULE_VERSION); +MODULE_LICENSE("Dual BSD/GPL"); + +/* SECTION: Module global variables */ +static int xpdev_cnt = 0; + +static const struct pci_device_id pci_ids[] = { + { PCI_DEVICE(0x10ee, 0x903f), }, + { PCI_DEVICE(0x10ee, 0x9038), }, + { PCI_DEVICE(0x10ee, 0x9028), }, + { PCI_DEVICE(0x10ee, 0x9018), }, + { PCI_DEVICE(0x10ee, 0x9034), }, + { PCI_DEVICE(0x10ee, 0x9024), }, + { PCI_DEVICE(0x10ee, 0x9014), }, + { PCI_DEVICE(0x10ee, 0x9032), }, + { PCI_DEVICE(0x10ee, 0x9022), }, + { PCI_DEVICE(0x10ee, 0x9012), }, + { PCI_DEVICE(0x10ee, 0x9031), }, + { PCI_DEVICE(0x10ee, 0x9021), }, + { PCI_DEVICE(0x10ee, 0x9011), }, + + { PCI_DEVICE(0x10ee, 0x8011), }, + { PCI_DEVICE(0x10ee, 0x8012), }, + { PCI_DEVICE(0x10ee, 0x8014), }, + { PCI_DEVICE(0x10ee, 0x8018), }, + { PCI_DEVICE(0x10ee, 0x8021), }, + { PCI_DEVICE(0x10ee, 0x8022), }, + { PCI_DEVICE(0x10ee, 0x8024), }, + { PCI_DEVICE(0x10ee, 0x8028), }, + { PCI_DEVICE(0x10ee, 0x8031), }, + { PCI_DEVICE(0x10ee, 0x8032), }, + { PCI_DEVICE(0x10ee, 0x8034), }, + { PCI_DEVICE(0x10ee, 0x8038), }, + + { PCI_DEVICE(0x10ee, 0x7011), }, + { PCI_DEVICE(0x10ee, 0x7012), }, + { PCI_DEVICE(0x10ee, 0x7014), }, + { PCI_DEVICE(0x10ee, 0x7018), }, + { PCI_DEVICE(0x10ee, 0x7021), }, + { PCI_DEVICE(0x10ee, 0x7022), }, + { PCI_DEVICE(0x10ee, 0x7024), }, + { PCI_DEVICE(0x10ee, 0x7028), }, + { PCI_DEVICE(0x10ee, 0x7031), }, + { PCI_DEVICE(0x10ee, 0x7032), }, + { PCI_DEVICE(0x10ee, 0x7034), }, + { PCI_DEVICE(0x10ee, 0x7038), }, + + { PCI_DEVICE(0x10ee, 0x6828), }, + { PCI_DEVICE(0x10ee, 0x6830), }, + { PCI_DEVICE(0x10ee, 0x6928), }, + { PCI_DEVICE(0x10ee, 0x6930), }, + { PCI_DEVICE(0x10ee, 0x6A28), }, + { PCI_DEVICE(0x10ee, 0x6A30), }, + { PCI_DEVICE(0x10ee, 0x6D30), }, + + { PCI_DEVICE(0x10ee, 0x4808), }, + { PCI_DEVICE(0x10ee, 0x4828), }, + { PCI_DEVICE(0x10ee, 0x4908), }, + { PCI_DEVICE(0x10ee, 0x4A28), }, + { PCI_DEVICE(0x10ee, 0x4B28), }, + + { PCI_DEVICE(0x10ee, 0x2808), }, + +#ifdef INTERNAL_TESTING + { PCI_DEVICE(0x1d0f, 0x1042), 0}, +#endif + {0,} +}; +MODULE_DEVICE_TABLE(pci, pci_ids); + +static void xpdev_free(struct xdma_pci_dev *xpdev) +{ + struct xdma_dev *xdev = xpdev->xdev; + + pr_info("xpdev 0x%p, destroy_interfaces, xdev 0x%p.\n", xpdev, xdev); + xpdev_destroy_interfaces(xpdev); + xpdev->xdev = NULL; + pr_info("xpdev 0x%p, xdev 0x%p xdma_device_close.\n", xpdev, xdev); + xdma_device_close(xpdev->pdev, xdev); + xpdev_cnt--; + + kfree(xpdev); +} + +static struct xdma_pci_dev *xpdev_alloc(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev = kmalloc(sizeof(*xpdev), GFP_KERNEL); + + if (!xpdev) + return NULL; + memset(xpdev, 0, sizeof(*xpdev)); + + xpdev->magic = MAGIC_DEVICE; + xpdev->pdev = pdev; + xpdev->user_max = MAX_USER_IRQ; + xpdev->h2c_channel_max = XDMA_CHANNEL_NUM_MAX; + xpdev->c2h_channel_max = XDMA_CHANNEL_NUM_MAX; + + xpdev_cnt++; + return xpdev; +} + +static int probe_one(struct pci_dev *pdev, const struct pci_device_id *id) +{ + int rv = 0; + struct xdma_pci_dev *xpdev = NULL; + struct xdma_dev *xdev; + void *hndl; + + xpdev = xpdev_alloc(pdev); + if (!xpdev) + return -ENOMEM; + + hndl = xdma_device_open(DRV_MODULE_NAME, pdev, &xpdev->user_max, + &xpdev->h2c_channel_max, &xpdev->c2h_channel_max); + if (!hndl) + return -EINVAL; + + BUG_ON(xpdev->user_max > MAX_USER_IRQ); + BUG_ON(xpdev->h2c_channel_max > XDMA_CHANNEL_NUM_MAX); + BUG_ON(xpdev->c2h_channel_max > XDMA_CHANNEL_NUM_MAX); + + if (!xpdev->h2c_channel_max && !xpdev->c2h_channel_max) + pr_warn("NO engine found!\n"); + + if (xpdev->user_max) { + u32 mask = (1 << (xpdev->user_max + 1)) - 1; + + rv = xdma_user_isr_enable(hndl, mask); + if (rv) + goto err_out; + } + + /* make sure no duplicate */ + xdev = xdev_find_by_pdev(pdev); + if (!xdev) { + pr_warn("NO xdev found!\n"); + return -EINVAL; + } + BUG_ON(hndl != xdev ); + + pr_info("%s xdma%d, pdev 0x%p, xdev 0x%p, 0x%p, usr %d, ch %d,%d.\n", + dev_name(&pdev->dev), xdev->idx, pdev, xpdev, xdev, + xpdev->user_max, xpdev->h2c_channel_max, + xpdev->c2h_channel_max); + + xpdev->xdev = hndl; + + rv = xpdev_create_interfaces(xpdev); + if (rv) + goto err_out; + + dev_set_drvdata(&pdev->dev, xpdev); + + return 0; + +err_out: + pr_err("pdev 0x%p, err %d.\n", pdev, rv); + xpdev_free(xpdev); + return rv; +} + +static void remove_one(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev; + + if (!pdev) + return; + + xpdev = dev_get_drvdata(&pdev->dev); + if (!xpdev) + return; + + pr_info("pdev 0x%p, xdev 0x%p, 0x%p.\n", + pdev, xpdev, xpdev->xdev); + xpdev_free(xpdev); + + dev_set_drvdata(&pdev->dev, NULL); +} + +static pci_ers_result_t xdma_error_detected(struct pci_dev *pdev, + pci_channel_state_t state) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + switch (state) { + case pci_channel_io_normal: + return PCI_ERS_RESULT_CAN_RECOVER; + case pci_channel_io_frozen: + pr_warn("dev 0x%p,0x%p, frozen state error, reset controller\n", + pdev, xpdev); + xdma_device_offline(pdev, xpdev->xdev); + pci_disable_device(pdev); + return PCI_ERS_RESULT_NEED_RESET; + case pci_channel_io_perm_failure: + pr_warn("dev 0x%p,0x%p, failure state error, req. disconnect\n", + pdev, xpdev); + return PCI_ERS_RESULT_DISCONNECT; + } + return PCI_ERS_RESULT_NEED_RESET; +} + +static pci_ers_result_t xdma_slot_reset(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + pr_info("0x%p restart after slot reset\n", xpdev); + if (pci_enable_device_mem(pdev)) { + pr_info("0x%p failed to renable after slot reset\n", xpdev); + return PCI_ERS_RESULT_DISCONNECT; + } + + pci_set_master(pdev); + pci_restore_state(pdev); + pci_save_state(pdev); + xdma_device_online(pdev, xpdev->xdev); + + return PCI_ERS_RESULT_RECOVERED; +} + +static void xdma_error_resume(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + pr_info("dev 0x%p,0x%p.\n", pdev, xpdev); +#if KERNEL_VERSION(5, 7, 0) <= LINUX_VERSION_CODE + pci_aer_clear_nonfatal_status(pdev); +#else + pci_cleanup_aer_uncorrect_error_status(pdev); +#endif +} + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,13,0) +static void xdma_reset_prepare(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + pr_info("dev 0x%p,0x%p.\n", pdev, xpdev); + xdma_device_offline(pdev, xpdev->xdev); +} + +static void xdma_reset_done(struct pci_dev *pdev) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + pr_info("dev 0x%p,0x%p.\n", pdev, xpdev); + xdma_device_online(pdev, xpdev->xdev); +} + +#elif LINUX_VERSION_CODE >= KERNEL_VERSION(3,16,0) +static void xdma_reset_notify(struct pci_dev *pdev, bool prepare) +{ + struct xdma_pci_dev *xpdev = dev_get_drvdata(&pdev->dev); + + pr_info("dev 0x%p,0x%p, prepare %d.\n", pdev, xpdev, prepare); + + if (prepare) + xdma_device_offline(pdev, xpdev->xdev); + else + xdma_device_online(pdev, xpdev->xdev); +} +#endif + +static const struct pci_error_handlers xdma_err_handler = { + .error_detected = xdma_error_detected, + .slot_reset = xdma_slot_reset, + .resume = xdma_error_resume, +#if LINUX_VERSION_CODE >= KERNEL_VERSION(4,13,0) + .reset_prepare = xdma_reset_prepare, + .reset_done = xdma_reset_done, +#elif LINUX_VERSION_CODE >= KERNEL_VERSION(3,16,0) + .reset_notify = xdma_reset_notify, +#endif +}; + +static struct pci_driver pci_driver = { + .name = DRV_MODULE_NAME, + .id_table = pci_ids, + .probe = probe_one, + .remove = remove_one, + .err_handler = &xdma_err_handler, +}; + +static int __init xdma_mod_init(void) +{ + int rv; + extern unsigned int desc_blen_max; + extern unsigned int sgdma_timeout; + + pr_info("%s", version); + + if (desc_blen_max > XDMA_DESC_BLEN_MAX) + desc_blen_max = XDMA_DESC_BLEN_MAX; + pr_info("desc_blen_max: 0x%x/%u, sgdma_timeout: %u sec.\n", + desc_blen_max, desc_blen_max, sgdma_timeout); + + rv = xdma_cdev_init(); + if (rv < 0) + return rv; + + return pci_register_driver(&pci_driver); +} + +static void __exit xdma_mod_exit(void) +{ + /* unregister this driver from the PCI bus driver */ + dbg_init("pci_unregister_driver.\n"); + pci_unregister_driver(&pci_driver); + xdma_cdev_cleanup(); +} + +module_init(xdma_mod_init); +module_exit(xdma_mod_exit); diff --git a/sources/xdma_driver/xdma/xdma_mod.h b/sources/xdma_driver/xdma/xdma_mod.h new file mode 100644 index 0000000..93202b1 --- /dev/null +++ b/sources/xdma_driver/xdma/xdma_mod.h @@ -0,0 +1,98 @@ +/* + * This file is part of the Xilinx DMA IP Core driver for Linux + * + * Copyright (c) 2016-present, Xilinx, Inc. + * All rights reserved. + * + * This source code is licensed under both the BSD-style license (found in the + * LICENSE file in the root directory of this source tree) and the GPLv2 (found + * in the COPYING file in the root directory of this source tree). + * You may select, at your option, one of the above-listed licenses. + */ + +#ifndef __XDMA_MODULE_H__ +#define __XDMA_MODULE_H__ + +#include <linux/types.h> +#include <linux/module.h> +#include <linux/cdev.h> +#include <linux/dma-mapping.h> +#include <linux/delay.h> +#include <linux/fb.h> +#include <linux/fs.h> +#include <linux/init.h> +#include <linux/interrupt.h> +#include <linux/io.h> +#include <linux/jiffies.h> +#include <linux/kernel.h> +#include <linux/mm.h> +#include <linux/mm_types.h> +#include <linux/poll.h> +#include <linux/pci.h> +#include <linux/sched.h> +#include <linux/slab.h> +#include <linux/vmalloc.h> +#include <linux/workqueue.h> +#include <linux/aio.h> +#include <linux/splice.h> +#include <linux/version.h> +#include <linux/uio.h> + +#include "libxdma.h" + +#define MAGIC_ENGINE 0xEEEEEEEEUL +#define MAGIC_DEVICE 0xDDDDDDDDUL +#define MAGIC_CHAR 0xCCCCCCCCUL +#define MAGIC_BITSTREAM 0xBBBBBBBBUL + +struct xdma_cdev { + unsigned long magic; /* structure ID for sanity checks */ + struct xdma_pci_dev *xpdev; + struct xdma_dev *xdev; + dev_t cdevno; /* character device major:minor */ + struct cdev cdev; /* character device embedded struct */ + int bar; /* PCIe BAR for HW access, if needed */ + unsigned long base; /* bar access offset */ + struct xdma_engine *engine; /* engine instance, if needed */ + struct xdma_user_irq *user_irq; /* IRQ value, if needed */ + struct device *sys_device; /* sysfs device */ + spinlock_t lock; +}; + +/* XDMA PCIe device specific book-keeping */ +struct xdma_pci_dev { + unsigned long magic; /* structure ID for sanity checks */ + struct pci_dev *pdev; /* pci device struct from probe() */ + struct xdma_dev *xdev; + int major; /* major number */ + int instance; /* instance number */ + int user_max; + int c2h_channel_max; + int h2c_channel_max; + + unsigned int flags; + /* character device structures */ + struct xdma_cdev ctrl_cdev; + struct xdma_cdev sgdma_c2h_cdev[XDMA_CHANNEL_NUM_MAX]; + struct xdma_cdev sgdma_h2c_cdev[XDMA_CHANNEL_NUM_MAX]; + struct xdma_cdev events_cdev[16]; + + struct xdma_cdev user_cdev; + struct xdma_cdev bypass_c2h_cdev[XDMA_CHANNEL_NUM_MAX]; + struct xdma_cdev bypass_h2c_cdev[XDMA_CHANNEL_NUM_MAX]; + struct xdma_cdev bypass_cdev_base; + + struct xdma_cdev xvc_cdev; + + void *data; +}; + +struct xdma_io_cb { + void __user *buf; + size_t len; + unsigned int pages_nr; + struct sg_table sgt; + struct page **pages; +}; + +#endif /* ifndef __XDMA_MODULE_H__ */ |
