diff --git a/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvKernel.maxj b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvKernel.maxj new file mode 100644 index 0000000..501d337 --- /dev/null +++ b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvKernel.maxj @@ -0,0 +1,346 @@ + LUTs FFs BRAMs DSPs : SpmvKernel.maxj + 105055 132726 614 192 : resources used by this file + 20.02% 12.65% 23.92% 9.78% : % of available + 58.95% 51.70% 47.30% 100.00% : % of total used + 98.74% 93.30% 99.68% 100.00% : % of user resources + + : import java.util.*; + : + : import com.maxeler.maxcompiler.v2.kernelcompiler.Kernel; + : import com.maxeler.maxcompiler.v2.kernelcompiler.KernelParameters; + : import com.maxeler.maxcompiler.v2.kernelcompiler.types.base.*; + : import com.maxeler.maxcompiler.v2.kernelcompiler.stdlib.core.Count.*; + : import com.maxeler.maxcompiler.v2.kernelcompiler.types.composite.DFEVectorType; + : import com.maxeler.maxcompiler.v2.kernelcompiler.types.composite.DFEVector; + : import com.maxeler.maxcompiler.v2.kernelcompiler.stdlib.memory.*; + : import com.maxeler.maxcompiler.v2.kernelcompiler.stdlib.KernelMath; + : import com.maxeler.maxcompiler.v2.utils.MathUtils; + : + : import com.custom_computing_ic.dfe_snippets.utils.Reductions; + : import com.custom_computing_ic.dfe_snippets.reductions.LogAddReduce; + : import com.custom_computing_ic.dfe_snippets.utils.FetchSubTuple; + : + : + : class SpmvKernel extends Kernel { + : + : // N x N matrix, nnzs nonzeros + : // + : // load data for N cycles into cache + : // compute for N cycles + : // -- in the mean time load data for N cycles in double buffer + : // repeat + : // + : // Sources of inefficiency: + : // -- empty rows + : // -- port sharing cannot be used + : + : private final DFEVectorType vtype, ivtype; + : private final int inputWidth; + : private final DFEType addressT; + : private final boolean dbg; + : + : private static final int fpL = 16; + : + : protected SpmvKernel( + : KernelParameters parameters, + : int inputWidth, + : int cacheSize, + : int indexWidth, + : int mantissaWidth, + : boolean dbg) { + : super(parameters); + : + : optimization.pushPipeliningFactor(0.5); + : optimization.pushDSPFactor(1); + : + : + : this.inputWidth = inputWidth; + : this.addressT = dfeUInt(MathUtils.bitsToAddress(cacheSize)); + : this.dbg = dbg; + : + : vtype = new DFEVectorType (dfeFloat(11, mantissaWidth), inputWidth); + : ivtype = new DFEVectorType (dfeUInt(indexWidth), inputWidth); + : + : // TODO: remove prefetch from this kernel + : + 1 2 0 0 : DFEVar vRomLoadEnable = io.input("loadEnabled", dfeBool()); + 1 0 0 0 : DFEVector matrixValues = io.input("matrixValues", vtype, ~vRomLoadEnable); + 1 1 0 0 : DFEVector vectorValues = io.input("vectorValues", vtype, ~vRomLoadEnable); + : + 65864 89061 127 192 : DFEVar result = Reductions.reduce(matrixValues * vectorValues); + : + : // --- Accumulation + 1 2 0 0 : DFEVar rowFinished = io.input("rowFinished", dfeBool()); + 1 1 0 0 : DFEVar rowLength = io.input("rowLength", dfeUInt(32)); + 1 1 0 0 : DFEVar cycleCounter = io.input("cycleCounter", dfeUInt(32)); + 1 0 0 0 : DFEVar firstReadPosition = io.input("firstReadPosition", dfeUInt(32)); + : + 32 32 0 0 : DFEVar runLength = firstReadPosition + rowLength; + 258 234 2 0 : DFEVar modulo = KernelMath.modulo(runLength, inputWidth); + 1240 988 10 0 : DFEVar quot = (runLength - modulo.cast(dfeUInt(32))) / inputWidth; + : + 33 51 0 0 : DFEVar totalCycles = quot + (modulo === 0 ? constant.var(dfeUInt(32), 0) : 1); + : + : DFEVar carriedSum = dfeFloat(11, mantissaWidth).newInstance(this); + 1103 1177 3 0 : DFEVar newSum = result + (cycleCounter < fpL ? 0 : carriedSum); + : carriedSum <== stream.offset(newSum, -fpL); + : + 41 65 0 0 : DFEVar firstValidPartialSum = (totalCycles > fpL)? (totalCycles - fpL) : 0; + 25 1 0 0 : DFEVar validPartialSums = (cycleCounter >= firstValidPartialSum); + 4654 5213 4 0 : LogAddReduce r = new LogAddReduce(this, + : validPartialSums, + : rowFinished, + : newSum, + : dfeFloat(11, mantissaWidth), + : fpL); + : + : // TODO still need reduction step + : // TODO: remove prefetch + 10 12 1 0 : DFEVar outputEnable = ~vRomLoadEnable & rowFinished; + 9 6 1 0 : io.output("output", r.getOutput(), dfeFloat(11, mantissaWidth), outputEnable); + : + : // --- Debug + : if (dbg) { + : Params rowCounterParams = control.count.makeParams(32) + : .withEnable(rowFinished); + : DFEVar rowCounter = control.count.makeCounter(rowCounterParams).getCount(); + : /* debug.simPrintf( + : rowLength !== 0, + : "Kernel -- row %d totalCycles %d, readenable %d, readmask %d rowFinished %d rowLength %d output: %f cycleCounter %d validPartialSums %d newSum %f firstRead %d", + : rowCounter, totalCycles, readEnable, readMask, rowFinished, rowLength, r.getOutput(), cycleCounter, validPartialSums, newSum, firstReadPosition); + : debug.simPrintf(rowLength !== 0 , " loadEnable %d loadAddress %d\n", vRomLoadEnable, loadAddress); + : */ + : //debug.simPrintf("\n"); + : //debug.simPrintf("Values: "); + : //for (int i = 0; i < inputWidth; i++) + : //debug.simPrintf("%f ", values[i]); + : //debug.simPrintf("Indices: "); + : //for (int i = 0; i < inputWidth; i++) + : //debug.simPrintf("%d ", colptr[i]); + : //debug.simPrintf("\n"); + : } + : } + : + : } + : + : // This implements cache as + : // - row of BRAMs + : // - SM, which helps to select BRAM values + : class SpmvCacheKernel extends Kernel { + : + : private final DFEVectorType vtype, ivtype; + : private final List> vroms; + : private final int inputWidth; + : private final DFEType addressT; + : + : protected SpmvCacheKernel(KernelParameters parameters, + : int inputWidth, + : int cacheSize, + : int indexWidth, + : int mantissaWidth) { + : super(parameters); + : + : this.inputWidth = inputWidth; + : this.addressT = dfeUInt(MathUtils.bitsToAddress(cacheSize)); + : + : // load entire vector or until cache is full + : int sizeBits = 32; // XXX may need to run for more cycles + : DFEVar vRomLoadEnable = io.input("loadEnabled_in", dfeBool()); + : + : DFEVar vectorLoadCycles = io.scalarInput("vectorLoadCycles", dfeUInt(32)); + : Params loadAddressParams = control.count.makeParams(sizeBits ) + : .withMax(vectorLoadCycles) + : .withEnable(vRomLoadEnable); + : DFEVar loadAddress = control.count.makeCounter(loadAddressParams).getCount(); + : + : + : vtype = new DFEVectorType (dfeFloat(11, mantissaWidth), inputWidth); + : ivtype = new DFEVectorType (dfeUInt(indexWidth), inputWidth); + : DFEVar vectorValue = io.input("vromLoad", dfeFloat(11, mantissaWidth), vRomLoadEnable); + : //--- Cache allocation and control + : vroms = new ArrayList>(); + : for (int i = 0; i < inputWidth; i++) { + : Memory vrom = mem.alloc(dfeFloat(11, mantissaWidth), cacheSize); + : vroms.add(vrom); + : vrom.write( + : loadAddress.cast(addressT), + : vectorValue, + : vRomLoadEnable); + : } + : + : // --- I/O + : DFEVar readEnable = io.input("readenable", dfeBool()) & ~vRomLoadEnable; + : DFEVar readMask = io.input("readmask", dfeUInt(inputWidth)); + : DFEVector matrixValues = selectValues( + : io.input("values", vtype, readEnable), + : readMask); + : DFEVector colptr = selectValues( + : io.input("indptr", ivtype, readEnable), + : readMask); + : DFEVector vectorValues = resolveVectorReads(colptr); + : + : io.output("loadEnabled_out", vRomLoadEnable, dfeBool()); + : io.output("matrixValues", matrixValues, vtype, ~vRomLoadEnable); + : io.output("vectorValues", vectorValues, vtype, ~vRomLoadEnable); + : } + : + : DFEVector selectValues( + : DFEVector in, + : DFEVar readMask) { + : DFEVector out = in.getType().newInstance(this); + : for (int i = 0; i < in.getSize(); i++) + : out[i] <== readMask.slice(i) === 0 ? 0 : in[i]; + : return out; + : } + : + : + : DFEVector resolveVectorReads(DFEVector reads) { + : DFEVector out = vtype.newInstance(this); + : for (int i = 0; i < vroms.size(); i++) + : out[i] <== vroms.get(i).read(reads[i].cast(addressT)); + : return out; + : } + : + : } + : + : // This implements cache as + : // - row of BRAMs + : // - FetchSubTuple + : // First, rough and pointless implementation. + : class SpmvCacheFSTKernel extends Kernel { + : + : private final DFEVectorType vtype, ivtype; + : private final List> vroms; + : private final int inputWidth; + : private final DFEType addressT; + : + : protected SpmvCacheFSTKernel(KernelParameters parameters, + : int inputWidth, + : int cacheSize, + : int indexWidth, + : int mantissaWidth) { + : super(parameters); + : + : this.inputWidth = inputWidth; + : this.addressT = dfeUInt(MathUtils.bitsToAddress(cacheSize)); + : + : // load entire vector or until cache is full + : int sizeBits = 32; // XXX may need to run for more cycles + 1 1 0 0 : DFEVar vRomLoadEnable = io.input("loadEnabled_in", dfeBool()); + : + : DFEVar vectorLoadCycles = io.scalarInput("vectorLoadCycles", dfeUInt(32)); + : Params loadAddressParams = control.count.makeParams(sizeBits ) + : .withMax(vectorLoadCycles) + : .withEnable(vRomLoadEnable); + 94 78 0 0 : DFEVar loadAddress = control.count.makeCounter(loadAddressParams).getCount(); + : + : + : vtype = new DFEVectorType (dfeFloat(11, mantissaWidth), inputWidth); + : ivtype = new DFEVectorType (dfeUInt(indexWidth), inputWidth); + 1 65 0 0 : DFEVar vectorValue = io.input("vromLoad", dfeFloat(11, mantissaWidth), vRomLoadEnable); + : //--- Cache allocation and control + : vroms = new ArrayList>(); + : for (int i = 0; i < inputWidth; i++) { + 1 1 336 0 : Memory vrom = mem.alloc(dfeFloat(11, mantissaWidth), cacheSize); + : vroms.add(vrom); + : vrom.write( + : loadAddress.cast(addressT), + : vectorValue, + : vRomLoadEnable); + : } + : + : // --- I/O + : // --- buffering matrix and colptr data in FST + : + : // assuming readEnable guards internal storage of FST from overflow: + : // + : // DFEVar dataRequestEnableLoop = dfeBool().newInstance(this); + : // DFEVar dataRequestEnable = control.count.pulse(1)? 0 + : // : stream.offset(dataRequestEnableLoop,-1); + : // DFEVar enable = readEnable & dataRequestEnable; + : + 2 3 0 0 : DFEVar readEnable = io.input("readenable", dfeBool()) & ~vRomLoadEnable; + 1 1 0 0 : DFEVar readMask = io.input("readmask", dfeUInt(inputWidth)); + : + 1 3803 0 0 : DFEVector matrix_in = io.input("values", vtype, readEnable); + 1 1 0 0 : DFEVector index_in = io.input("indptr", ivtype, readEnable); + : + : // big hack around: counting the number of 1s in readmask + : List readmask_bits = new ArrayList(); + : for (int i = 0; i < inputWidth; i++) + : { + 0 529 0 0 : readmask_bits.add(readMask[i].cast(dfeBool()).cast(dfeUInt(6))); // 2^6 = 64 > 48 = inputWidth + : } + 144 145 0 0 : DFEVar numEntriesToRead = Reductions.reduce(readmask_bits); + : + : // we don't care which PE receives each entry (assume matrices are + : // numerically nice enough to change the order of accumulation) + : boolean alignOutput = false; + : int matrixTypeWidth = 11 + mantissaWidth; + : FetchSubTuple matrixBuffer = new FetchSubTuple(this, "matrix", inputWidth, + : matrixTypeWidth, dfeFloat(11, mantissaWidth), + : alignOutput); + : FetchSubTuple indexBuffer = new FetchSubTuple(this, "index", inputWidth, + : indexWidth, dfeUInt(indexWidth), alignOutput); + : + 25403 25051 0 0 : DFEVector matrixValues = matrixBuffer.popPush(numEntriesToRead, readEnable, matrix_in); + 4753 4228 0 0 : DFEVector colptr = indexBuffer.popPush(numEntriesToRead, readEnable, index_in); + : + : //dataRequestEnableLoop <== matrixBuffer.nextPushEnable(); + : + : DFEVector vectorValues = resolveVectorReads(colptr); + : + : io.output("loadEnabled_out", vRomLoadEnable, dfeBool()); + 0 5 0 0 : io.output("matrixValues", matrixValues, vtype, ~vRomLoadEnable); + 1 2 0 0 : io.output("vectorValues", vectorValues, vtype, ~vRomLoadEnable); + : } + : + : DFEVector resolveVectorReads(DFEVector reads) { + : DFEVector out = vtype.newInstance(this); + : for (int i = 0; i < vroms.size(); i++) + : out[i] <== vroms.get(i).read(reads[i].cast(addressT)); + : return out; + : } + : + : } + : + : + : + : + : // XXX for now we assume the matrix is smaller then max rows + : // will have to implement an lmem design above this threshold + : // using lmem wrapped above this thresholed may be feasilbe, + : // see the DramAccumulator snippet in dfe-snippets + : class SpmvReductionKernel extends Kernel { + : + : protected SpmvReductionKernel(KernelParameters parameters, int fpl, int maxRows) { + : super(parameters); + : + 1 1 0 0 : DFEVar in = io.input("reductionIn", dfeFloat(11, 53)); + : DFEVar n = io.scalarInput("nRows", dfeUInt(32)); + : DFEVar totalCycles = io.scalarInput("totalCycles", dfeUInt(32)); + : + 69 120 0 0 : DFEVar cycles = control.count.simpleCounter(32); + : + : DFEVar sumCarried = dfeFloat(11, 53).newInstance(this); + 1116 1572 2 0 : DFEVar sum = in + (cycles < n ? 0 : sumCarried); + 44 48 128 0 : sumCarried <== stream.offset(sum, -(n.cast(dfeInt(32))), -maxRows, -(fpl + 2)); + : + : // output on the last n cycles + 55 33 0 0 : DFEVar outputEnable = totalCycles - cycles <= n; + 1 14 0 0 : io.output("reductionOut", sum, dfeFloat(11, 53), outputEnable); + : } + : } + : + : + : class PaddingKernel extends Kernel { + : protected PaddingKernel(KernelParameters parameters) { + : super(parameters); + : DFEVar nInputs = io.scalarInput("nInputs", dfeUInt(32)); + 69 107 0 0 : DFEVar cycles = control.count.simpleCounter(32); + 20 1 0 0 : DFEVar paddingCycles = cycles >= nInputs; + 1 1 0 0 : DFEVar input = io.input("paddingIn", dfeFloat(11, 53), ~paddingCycles); + 0 69 0 0 : DFEVar out = paddingCycles ? 0 : input; + : io.output("paddingOut", out, dfeFloat(11, 53)); + : } + : } diff --git a/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvManager.maxj b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvManager.maxj new file mode 100644 index 0000000..419d9c1 --- /dev/null +++ b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/SpmvManager.maxj @@ -0,0 +1,153 @@ + LUTs FFs BRAMs DSPs : SpmvManager.maxj + 105961 134189 614 192 : resources used by this file + 20.19% 12.78% 23.92% 9.78% : % of available + 59.46% 52.27% 47.30% 100.00% : % of total used + 99.59% 94.33% 99.68% 100.00% : % of user resources + + : import com.maxeler.maxcompiler.v2.managers.engine_interfaces.CPUTypes; + : import com.maxeler.maxcompiler.v2.managers.engine_interfaces.EngineInterface; + : import com.maxeler.maxcompiler.v2.managers.engine_interfaces.InterfaceParam; + : import com.maxeler.maxcompiler.v2.managers.custom.CustomManager; + : import com.maxeler.maxcompiler.v2.managers.custom.blocks.KernelBlock; + : import com.maxeler.maxcompiler.v2.build.EngineParameters; + : import com.maxeler.maxcompiler.v2.managers.custom.blocks.StateMachineBlock; + : import com.maxeler.maxcompiler.v2.statemachine.manager.ManagerStateMachine; + : import com.maxeler.maxcompiler.v2.managers.custom.stdlib.MemoryControlGroup; + : + : import com.custom_computing_ic.dfe_snippets.sparse.*; + : import com.custom_computing_ic.dfe_snippets.manager.*; + : + : public class SpmvManager extends CustomManager{ + : private static final String s_kernelName = "SpmvKernel"; + : private static final String s_reductionKernel = "SpmvReductionKernel"; + : private static final String s_paddingKernel = "SpmvPaddingKernel"; + : private static final String s_cacheKernel = "SpmvCacheKernel"; + : + : private static final int cacheSize = 1024 * 2; + : private static final int inputWidth = 48; + : private static final int MAX_ROWS = 30000; + : + : // parameters of CSR format used: float64 values, int32 index. + : private static final int mantissaWidth = 53; + : private static final int indexWidth = 32; + : + : private static final int FLOATING_POINT_LATENCY = 16; + : + : private static final boolean DBG_CSR_DECODER = false; + : private static final boolean DBG_PAR_CSR_CTL = false; + : private static final boolean DBG_SPMV_KERNEL = false; + : + : SpmvManager(EngineParameters ep) { + : super(ep); + : addMaxFileConstant("inputWidth", inputWidth); + : addMaxFileConstant("cacheSize", cacheSize); + : addMaxFileConstant("maxRows", MAX_ROWS); + : + : ManagerUtils.setDRAMMaxDeviceFrequency(this, ep); + : + : ManagerStateMachine csrDecoder = new CsrDecoder(this, DBG_CSR_DECODER); + : StateMachineBlock csrDecoderBlock = addStateMachine("csrDecoder", csrDecoder); + : csrDecoderBlock.getInput("colptr") <== addStreamFromCPU("colptr"); + : + : ManagerStateMachine readControl = new ParallelCsrReadControl(this, inputWidth, DBG_PAR_CSR_CTL); + : StateMachineBlock readControlBlock = addStateMachine("readControl", readControl); + : readControlBlock.getInput("length") <== csrDecoderBlock.getOutput("rowLength_out"); + : + 227 370 0 0 : KernelBlock cache = addKernel(new SpmvCacheFSTKernel( + : makeKernelParameters(s_cacheKernel), + : inputWidth, + : cacheSize, + : indexWidth, + 30403 33913 336 0 : mantissaWidth + : )); + : + 226 374 0 0 : KernelBlock k = addKernel(new SpmvKernel( + : makeKernelParameters(s_kernelName), + : inputWidth, + : cacheSize, + : indexWidth, + : mantissaWidth, + 73276 96847 148 192 : DBG_SPMV_KERNEL + : )); + : + : addStreamToOnCardMemory("cpu2lmem", MemoryControlGroup.MemoryAccessPattern.LINEAR_1D) <== addStreamFromCPU("fromcpu"); + : + : + : + : ManagerUtils.addLinearStreamFromLmemToKernel(this, cache, "indptr"); + : ManagerUtils.addLinearStreamFromLmemToKernel(this, cache, "values"); + : + : cache.getInput("vromLoad") <== addStreamFromCPU("vromLoad"); + : + : cache.getInput("readenable") <== readControlBlock.getOutput("readenable"); + : cache.getInput("readmask") <== readControlBlock.getOutput("readmask"); + : cache.getInput("loadEnabled_in") <== readControlBlock.getOutput("vectorLoad"); + : + : k.getInput("loadEnabled") <== cache.getOutput("loadEnabled_out"); + : k.getInput("vectorValues") <== cache.getOutput("vectorValues"); + : k.getInput("matrixValues") <== cache.getOutput("matrixValues"); + : + : k.getInput("rowLength") <== readControlBlock.getOutput("rowLength"); + : k.getInput("rowFinished") <== readControlBlock.getOutput("rowFinished"); + : k.getInput("cycleCounter") <== readControlBlock.getOutput("cycleCounter"); + : k.getInput("firstReadPosition") <== readControlBlock.getOutput("firstReadPosition"); + : + 227 358 0 0 : KernelBlock r = addKernel(new SpmvReductionKernel( + : makeKernelParameters(s_reductionKernel), + : FLOATING_POINT_LATENCY, + 1286 1788 130 0 : MAX_ROWS)); + : r.getInput("reductionIn") <== k.getOutput("output"); + : + 226 361 0 0 : KernelBlock p = addKernel(new PaddingKernel( + 90 178 0 0 : makeKernelParameters(s_paddingKernel))); + : + : p.getInput("paddingIn") <== r.getOutput("reductionOut"); + : addStreamToCPU("output") <== p.getOutput("paddingOut"); + : } + : + : private static EngineInterface interfaceDefault() { + : EngineInterface ei = new EngineInterface(); + : + : InterfaceParam n = ei.addParam("nrows", CPUTypes.INT); + : InterfaceParam vectorSize = ei.addParam("vectorSize", CPUTypes.INT); + : InterfaceParam vectorLoadCycles = ei.addParam("vectorLoadCycles", CPUTypes.INT); + : InterfaceParam totalCycles = ei.addParam("totalCycles", CPUTypes.INT); + : InterfaceParam paddingCycles = ei.addParam("paddingCycles", CPUTypes.INT); + : InterfaceParam nPartitions = ei.addParam("nPartitions", CPUTypes.INT); + : + : ei.setTicks(s_kernelName, totalCycles); + : + : ei.setTicks(s_cacheKernel, totalCycles); + : ei.setScalar(s_cacheKernel, "vectorLoadCycles", vectorLoadCycles); + : + : ei.setTicks(s_reductionKernel, (n * nPartitions)); + : ei.setScalar(s_reductionKernel, "nRows", n); + : ei.setScalar(s_reductionKernel, "totalCycles", n * nPartitions); + : + : ei.setTicks(s_paddingKernel, (n + paddingCycles)); + : ei.setScalar(s_paddingKernel, "nInputs", n); + : + : ei.setScalar("csrDecoder", "nrows", n); + : + : ei.setScalar("readControl", "nrows", n); + : ei.setScalar("readControl", "vectorLoadCycles", vectorLoadCycles); + : ei.setScalar("readControl", "nPartitions", nPartitions); + : + : ei.setStream("output", CPUTypes.DOUBLE, (n + paddingCycles) * CPUTypes.DOUBLE.sizeInBytes()); + : ei.setStream("vromLoad", CPUTypes.DOUBLE, vectorSize * CPUTypes.DOUBLE.sizeInBytes()); + : + : ei.ignoreLMem("cpu2lmem"); + : ei.ignoreStream("fromcpu"); + : return ei; + : } + : + : public static void main(String[] args) { + 105961 134189 614 192 : SpmvManager manager = new SpmvManager(new EngineParameters(args)); + : manager.createSLiCinterface(interfaceDefault()); + : // ManagerUtils.debug(manager); + : ManagerUtils.setFullBuild(manager, 2, 2); + : manager.createSLiCinterface(ManagerUtils.interfaceWrite( + : "write", "fromcpu", "cpu2lmem")); + : manager.build(); + : } + : } diff --git a/experimental-data/spmv_48p_2048c_0.5pf_full_fst/report.txt b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/report.txt new file mode 100644 index 0000000..3ad1fe8 --- /dev/null +++ b/experimental-data/spmv_48p_2048c_0.5pf_full_fst/report.txt @@ -0,0 +1,155 @@ + +Total resource usage +----------------------------------------------------------------- + LUTs FFs BRAMs DSPs + 524800 1049600 2567 1963 total available resources for FPGA + 178215 256726 1298 192 total resources used + 33.96% 24.46% 50.56% 9.78% % of available + 106398 142250 616 192 used by kernels + 20.27% 13.55% 24.00% 9.78% % of available + 70847 112634 671 0 used by manager + 13.50% 10.73% 26.14% 0.00% % of available + 127363 198534 1100 192 stray resources + 24.27% 18.92% 42.85% 9.78% % of available + +High level manager breakdown aggregated by type +----------------------------------------------------------------- + LUTs FFs BRAMs DSPs Type Occurrences + 538 1227 0 0 AddrGen 3 + 39 37 1 0 ChecksumMappedDRP 1 + 641 636 0 0 DualAspectMux 3 + 19 3138 0 0 DualAspectReg 2 + 1025 1028 199 0 Fifo 20 + 106398 142250 616 192 Kernel 4 + 140 215 0 0 MAX4CPLD 1 + 733 1020 2 0 MAX4PCIeSlaveInterface 1 + 21 51 0 0 MAXEvents 1 + 82 171 0 0 ManagerStateMachine_csrD 1 + 1655 536 0 0 ManagerStateMachine_read 1 + 469 84 0 0 MappedElementSwitch 1 + 443 997 5 0 MappedMemoriesController 1 + 160 130 0 0 MappedRegistersControlle 1 + 16013 48002 281 0 MemoryControllerPro 1 + 1630 847 4 0 PCIeBase 1 + 1497 1608 34 0 PCIeSlaveStreaming 1 + 269 393 0 0 PerfMonitor 1 + 16 23 0 0 ResetControl 2 + 88 186 0 0 SanityBlock 1 + 98 90 1 0 SignalForwardingAdapter 1 + 45269 52214 144 0 StratixVDDR3 6 + 2 1 0 0 StreamPullPushAdapter 1 + 0 0 0 0 Memory Controller -- + 0 0 0 0 Other InterFPGA -- + 1050 1297 8 0 Other MappedElements -- + 2903 3834 42 0 Other PCIe -- + +Kernel breakdown +----------------------------------------------------------------- + LUTs FFs BRAMs DSPs category + 106398 142250 616 192 total for all kernels + 20.27% 13.55% 24.00% 9.78% % of total available + +Totals for each kernel + LUTs FFs BRAMs DSPs Kernel name + 30747 40877 337 0 SpmvCacheKernel (total) + 5.86% 3.89% 13.13% 0.00% % of total available + 30630 4741 336 0 SpmvCacheKernel (user) + 5.84% 0.45% 13.09% 0.00% % of total available + 0 29542 0 0 SpmvCacheKernel (scheduling) + 0.00% 2.81% 0.00% 0.00% % of total available + 117 6594 1 0 SpmvCacheKernel (other Kernel resources) + 0.02% 0.63% 0.04% 0.00% % of total available + 73645 97694 149 192 SpmvKernel (total) + 14.03% 9.31% 5.80% 9.78% % of total available + 73442 96704 142 192 SpmvKernel (user) + 13.99% 9.21% 5.53% 9.78% % of total available + 60 517 6 0 SpmvKernel (scheduling) + 0.01% 0.05% 0.23% 0.00% % of total available + 143 473 1 0 SpmvKernel (other Kernel resources) + 0.03% 0.05% 0.04% 0.00% % of total available + 363 944 0 0 SpmvPaddingKernel (total) + 0.07% 0.09% 0.00% 0.00% % of total available + 316 534 0 0 SpmvPaddingKernel (user) + 0.06% 0.05% 0.00% 0.00% % of total available + 0 5 0 0 SpmvPaddingKernel (scheduling) + 0.00% 0.00% 0.00% 0.00% % of total available + 47 405 0 0 SpmvPaddingKernel (other Kernel resources) + 0.01% 0.04% 0.00% 0.00% % of total available + 1643 2735 130 0 SpmvReductionKernel (total) + 0.31% 0.26% 5.06% 0.00% % of total available + 1513 2132 130 0 SpmvReductionKernel (user) + 0.29% 0.20% 5.06% 0.00% % of total available + 0 14 0 0 SpmvReductionKernel (scheduling) + 0.00% 0.00% 0.00% 0.00% % of total available + 130 589 0 0 SpmvReductionKernel (other Kernel resources) + 0.02% 0.06% 0.00% 0.00% % of total available + + +Manager breakdown +----------------------------------------------------------------- + LUTs FFs BRAMs DSPs Type Instance + 30747 40877 337 0 Kernel SpmvCacheKernel + 73645 97694 149 192 Kernel SpmvKernel + 363 944 0 0 Kernel SpmvPaddingKernel + 1643 2735 130 0 Kernel SpmvReductionKernel + 17 3072 0 0 DualAspectReg Stream_14 + 533 531 0 0 DualAspectMux Stream_17 + 39 37 0 0 DualAspectMux Stream_1 + 69 68 0 0 DualAspectMux Stream_31 + 43 32 1 0 Fifo Stream_34 + 43 33 2 0 Fifo Stream_36 + 43 33 1 0 Fifo Stream_38 + 44 32 1 0 Fifo Stream_40 + 44 34 77 0 Fifo Stream_42 + 44 33 77 0 Fifo Stream_44 + 43 34 1 0 Fifo Stream_46 + 43 33 1 0 Fifo Stream_48 + 43 33 1 0 Fifo Stream_50 + 43 33 1 0 Fifo Stream_52 + 45 33 2 0 Fifo Stream_55 + 45 36 2 0 Fifo Stream_58 + 39 31 1 0 Fifo Stream_5 + 2 66 0 0 DualAspectReg Stream_61 + 43 31 4 0 Fifo Stream_64 + 77 119 1 0 Fifo Stream_66 + 75 125 4 0 Fifo Stream_68 + 2 1 0 0 StreamPullPushAdapter Stream_70 + 43 31 4 0 Fifo Stream_72 + 107 138 2 0 Fifo Stream_74 + 44 34 14 0 Fifo Stream_78 + 74 120 2 0 Fifo Stream_80 + 179 409 0 0 AddrGen addrgen_cmd_cpu2lmem + 180 412 0 0 AddrGen addrgen_cmd_indptr + 179 406 0 0 AddrGen addrgen_cmd_values + 82 171 0 0 ManagerStateMachine_csrD csrDecoder + 1655 536 0 0 ManagerStateMachine_read readControl + 733 1020 2 0 MAX4PCIeSlaveInterface MAX4PCIeSlaveInterface_i + 8 11 0 0 ResetControl control_streams_rst_ctl + 469 84 0 0 MappedElementSwitch MappedElementSwitch_i + 443 997 5 0 MappedMemoriesController MappedMemoriesController_i + 160 130 0 0 MappedRegistersControlle MappedRegistersController_i + 269 393 0 0 PerfMonitor perfm + 88 186 0 0 SanityBlock SanityBlock_i + 98 90 1 0 SignalForwardingAdapter SignalForwardingAdapter_i + 39 37 1 0 ChecksumMappedDRP checksum_mem_drp + 1497 1608 34 0 PCIeSlaveStreaming dynpcie + 8 12 0 0 ResetControl reset_controller + 16013 48002 281 0 MemoryControllerPro memctrlpro_maia_sodimms + 1630 847 4 0 PCIeBase PCIeBase_i + 140 215 0 0 MAX4CPLD cpld_io_ext_inst + 21 51 0 0 MAXEvents max_events + 7581 8726 24 0 StratixVDDR3 ddr3_core + 7474 8623 24 0 StratixVDDR3 ddr3_core + 7582 8738 24 0 StratixVDDR3 ddr3_core + 7472 8570 24 0 StratixVDDR3 ddr3_core + 7579 8787 24 0 StratixVDDR3 ddr3_core + 7581 8770 24 0 StratixVDDR3 ddr3_core + +Source files annotation report +----------------------------------------------------------------- + +% of total used for each file (note: multiple files may share the same resources) + LUTs FFs BRAMs DSPs filename + 58.95% 51.70% 47.30% 100.00% SpmvKernel.maxj + 59.46% 52.27% 47.30% 100.00% SpmvManager.maxj + 100.00% 87.96% 51.54% 100.00% [ missing source files ] diff --git a/src/spmv/src/Spmv.cpp b/src/spmv/src/Spmv.cpp index 18fda7b..2e18491 100644 --- a/src/spmv/src/Spmv.cpp +++ b/src/spmv/src/Spmv.cpp @@ -10,6 +10,20 @@ using EigenSparseMatrix = Eigen::SparseMatrix; +// how many cycles does it take to resolve the accesses with FST +int cycleCountFST(int32_t* v, int size) { + int cycles = 0; + int bufferWidth = Spmv_inputWidth; + for (int i = 0; i < size; i++) { + int toread = v[i] - (i > 0 ? v[i - 1] : 0); + do { + toread -= std::min(toread, bufferWidth); + cycles++; + } while (toread > 0); + } + return cycles; +} + // how many cycles does it take to resolve the accesses int cycleCount(int32_t* v, int size) { int cycles = 0; diff --git a/src/spmv/src/SpmvKernel.maxj b/src/spmv/src/SpmvKernel.maxj index 506643a..f77e76b 100644 --- a/src/spmv/src/SpmvKernel.maxj +++ b/src/spmv/src/SpmvKernel.maxj @@ -12,6 +12,8 @@ import com.maxeler.maxcompiler.v2.utils.MathUtils; import com.custom_computing_ic.dfe_snippets.utils.Reductions; import com.custom_computing_ic.dfe_snippets.reductions.LogAddReduce; +import com.custom_computing_ic.dfe_snippets.utils.FetchSubTuple; + class SpmvKernel extends Kernel { @@ -115,8 +117,9 @@ class SpmvKernel extends Kernel { } -// This implements cache -// +// This implements cache as +// - row of BRAMs +// - SM, which helps to select BRAM values class SpmvCacheKernel extends Kernel { private final DFEVectorType vtype, ivtype; @@ -194,6 +197,107 @@ class SpmvCacheKernel extends Kernel { } +// This implements cache as +// - row of BRAMs +// - FetchSubTuple +// First, rough and pointless implementation. +class SpmvCacheFSTKernel extends Kernel { + + private final DFEVectorType vtype, ivtype; + private final List> vroms; + private final int inputWidth; + private final DFEType addressT; + + protected SpmvCacheFSTKernel(KernelParameters parameters, + int inputWidth, + int cacheSize, + int indexWidth, + int mantissaWidth) { + super(parameters); + + this.inputWidth = inputWidth; + this.addressT = dfeUInt(MathUtils.bitsToAddress(cacheSize)); + + // load entire vector or until cache is full + int sizeBits = 32; // XXX may need to run for more cycles + DFEVar vRomLoadEnable = io.input("loadEnabled_in", dfeBool()); + + DFEVar vectorLoadCycles = io.scalarInput("vectorLoadCycles", dfeUInt(32)); + Params loadAddressParams = control.count.makeParams(sizeBits ) + .withMax(vectorLoadCycles) + .withEnable(vRomLoadEnable); + DFEVar loadAddress = control.count.makeCounter(loadAddressParams).getCount(); + + + vtype = new DFEVectorType (dfeFloat(11, mantissaWidth), inputWidth); + ivtype = new DFEVectorType (dfeUInt(indexWidth), inputWidth); + DFEVar vectorValue = io.input("vromLoad", dfeFloat(11, mantissaWidth), vRomLoadEnable); + //--- Cache allocation and control + vroms = new ArrayList>(); + for (int i = 0; i < inputWidth; i++) { + Memory vrom = mem.alloc(dfeFloat(11, mantissaWidth), cacheSize); + vroms.add(vrom); + vrom.write( + loadAddress.cast(addressT), + vectorValue, + vRomLoadEnable); + } + + // --- I/O + // --- buffering matrix and colptr data in FST + + // assuming readEnable guards internal storage of FST from overflow: + // + // DFEVar dataRequestEnableLoop = dfeBool().newInstance(this); + // DFEVar dataRequestEnable = control.count.pulse(1)? 0 + // : stream.offset(dataRequestEnableLoop,-1); + // DFEVar enable = readEnable & dataRequestEnable; + + DFEVar readEnable = io.input("readenable", dfeBool()) & ~vRomLoadEnable; + DFEVar readMask = io.input("readmask", dfeUInt(inputWidth)); + + DFEVector matrix_in = io.input("values", vtype, readEnable); + DFEVector index_in = io.input("indptr", ivtype, readEnable); + + // big hack around: counting the number of 1s in readmask + List readmask_bits = new ArrayList(); + for (int i = 0; i < inputWidth; i++) + { + readmask_bits.add(readMask[i].cast(dfeBool()).cast(dfeUInt(6))); // 2^6 = 64 > 48 = inputWidth + } + DFEVar numEntriesToRead = Reductions.reduce(readmask_bits); + + // we don't care which PE receives each entry (assume matrices are + // numerically nice enough to change the order of accumulation) + boolean alignOutput = false; + int matrixTypeWidth = 11 + mantissaWidth; + FetchSubTuple matrixBuffer = new FetchSubTuple(this, "matrix", inputWidth, + matrixTypeWidth, dfeFloat(11, mantissaWidth), + alignOutput); + FetchSubTuple indexBuffer = new FetchSubTuple(this, "index", inputWidth, + indexWidth, dfeUInt(indexWidth), alignOutput); + + DFEVector matrixValues = matrixBuffer.popPush(numEntriesToRead, readEnable, matrix_in); + DFEVector colptr = indexBuffer.popPush(numEntriesToRead, readEnable, index_in); + + //dataRequestEnableLoop <== matrixBuffer.nextPushEnable(); + + DFEVector vectorValues = resolveVectorReads(colptr); + + io.output("loadEnabled_out", vRomLoadEnable, dfeBool()); + io.output("matrixValues", matrixValues, vtype, ~vRomLoadEnable); + io.output("vectorValues", vectorValues, vtype, ~vRomLoadEnable); + } + + DFEVector resolveVectorReads(DFEVector reads) { + DFEVector out = vtype.newInstance(this); + for (int i = 0; i < vroms.size(); i++) + out[i] <== vroms.get(i).read(reads[i].cast(addressT)); + return out; + } + +} + diff --git a/src/spmv/src/SpmvManager.maxj b/src/spmv/src/SpmvManager.maxj index a662458..d078295 100644 --- a/src/spmv/src/SpmvManager.maxj +++ b/src/spmv/src/SpmvManager.maxj @@ -47,7 +47,7 @@ public class SpmvManager extends CustomManager{ StateMachineBlock readControlBlock = addStateMachine("readControl", readControl); readControlBlock.getInput("length") <== csrDecoderBlock.getOutput("rowLength_out"); - KernelBlock cache = addKernel(new SpmvCacheKernel( + KernelBlock cache = addKernel(new SpmvCacheFSTKernel( makeKernelParameters(s_cacheKernel), inputWidth, cacheSize,