From 3ce541fb5e30034f891152ea013b6a8708eaef1a Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Fri, 28 Aug 2015 18:50:00 +0200 Subject: [PATCH 01/63] Added geometry generators --- device-gen/2Dto3D/Makefile | 4 +- device-gen/2Dto3D/main.cpp | 79 ++++++----- device-gen/common/common.h | 67 +++++++++ device-gen/common/redistance.h | 169 ++++++++++++++++++++++ device-gen/post-process/h52ply.py | 37 +++++ device-gen/post-process/plyScale.py | 108 ++++++++++++++ device-gen/scripts/README | 26 +++- device-gen/scripts/files.txt | 12 -- device-gen/scripts/run-cylinder.py | 20 +++ device-gen/scripts/run-egg.py | 88 ++++++++++++ device-gen/scripts/run-parab.py | 55 ++++++++ device-gen/scripts/run.sh | 19 --- device-gen/sdf-collage/Makefile | 4 +- device-gen/sdf-collage/main.cpp | 209 ++++++++++------------------ device-gen/sdf-cylinder/Makefile | 7 + device-gen/sdf-cylinder/main.cpp | 61 ++++++++ device-gen/sdf-shift/Makefile | 7 + device-gen/sdf-shift/main.cpp | 58 ++++++++ device-gen/sdf-unit-egg/Makefile | 7 + device-gen/sdf-unit-egg/main.cpp | 125 +++++++++++++++++ device-gen/sdf-unit-par/Makefile | 4 +- device-gen/sdf-unit-par/main.cpp | 11 +- 22 files changed, 961 insertions(+), 216 deletions(-) create mode 100644 device-gen/common/common.h create mode 100644 device-gen/common/redistance.h create mode 100755 device-gen/post-process/h52ply.py create mode 100755 device-gen/post-process/plyScale.py delete mode 100644 device-gen/scripts/files.txt create mode 100755 device-gen/scripts/run-cylinder.py create mode 100755 device-gen/scripts/run-egg.py create mode 100755 device-gen/scripts/run-parab.py delete mode 100755 device-gen/scripts/run.sh create mode 100644 device-gen/sdf-cylinder/Makefile create mode 100644 device-gen/sdf-cylinder/main.cpp create mode 100644 device-gen/sdf-shift/Makefile create mode 100644 device-gen/sdf-shift/main.cpp create mode 100644 device-gen/sdf-unit-egg/Makefile create mode 100644 device-gen/sdf-unit-egg/main.cpp diff --git a/device-gen/2Dto3D/Makefile b/device-gen/2Dto3D/Makefile index b6d431f68..6024c3212 100644 --- a/device-gen/2Dto3D/Makefile +++ b/device-gen/2Dto3D/Makefile @@ -1,5 +1,5 @@ -2Dto3D: main.cpp - g++ main.cpp -O0 -g3 -fopenmp -o 2Dto3D +2Dto3D: main.cpp ../common/common.h ../common/redistance.h + g++-4.9 main.cpp -O3 -fopenmp -o 2Dto3D clean: rm 2Dto3D diff --git a/device-gen/2Dto3D/main.cpp b/device-gen/2Dto3D/main.cpp index 75e9b96bf..1642ebf1e 100644 --- a/device-gen/2Dto3D/main.cpp +++ b/device-gen/2Dto3D/main.cpp @@ -16,6 +16,7 @@ #include #include #include +#include "../common/common.h" using namespace std; @@ -35,56 +36,66 @@ int main(int argc, char ** argv) float xextent, yextent; vector slice; - - { - printf("Reading file %s...\n", argv[1]); - FILE * f = fopen(argv[1], "r"); - assert(f != 0); - float zextentOld; - int NZOld; - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextentOld); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZOld); - printf("Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextentOld, NX, NY,NZOld); - slice.resize(NX * NY, 0.0f); - fread(&slice[0], sizeof(float), slice.size(), f); - fclose(f); - } + int oldNZ; + float zextentOld; + readDAT(argv[1], slice, xextent, yextent, zextentOld, NX, NY, oldNZ); + assert(oldNZ == 1); printf("Generating data with extent [%f, %f, %f], dimensions [%d, %d, %d], zmargin %f\n", xextent, yextent, zextent + 2 * zmargin, NX, NY, NZ, zmargin); - vector volume(NX * NY * NZ, 0.0f); + + vector outputslice(NX * NY, 0.0f); const float z0 = -zextent * 0.5 - zmargin; const float dz = (zextent + 2 * zmargin) / (NZ - 1); -//#pragma omp parallel for + FILE * f = fopen(argv[5], "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); + fprintf(f, "%d %d %d\n", NY, NX, NZ); + for(int iz = 0; iz < NZ; ++iz) { const float z = z0 + iz * dz; + +#pragma omp parallel for for(int iy = 0; iy < NY; ++iy) for(int ix = 0; ix < NX; ++ix) { const float xysdf = slice[ix + NX * (NY - 1 - iy)]; // NY -1 to change Y-axis direction - const float zsdf = fabs(z) - zextent * 0.5; - float val; - if (xysdf < 0) - val = max(zsdf, xysdf); - else - val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); - - assert(iy + NY * (ix + NX * iz) < volume.size()); - assert(volume[iy + NY * (ix + NX * iz)] == 0.0f); - volume[iy + NY * (ix + NX * iz)] = val; + float val = xysdf; + if (zmargin != 0.0f) { + const float zsdf = fabs(z) - zextent * 0.5; + if (xysdf < 0) + val = max(zsdf, xysdf); + else + val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); + } + assert(iy + NY * (ix) < outputslice.size()); + + assert(fabs(val) < 1e3); // to check that the value has reasonable range + outputslice[iy + NY * ix] = val; } + + if (iz == 0) + { + unsigned char * ptr = (unsigned char *)&outputslice[0]; + if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) + { + ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; + printf("INFO: some symbols were changed while writing\n"); + } + } + + int result = fwrite(&outputslice.front(), sizeof(float), NX * NY, f); + + if (result != NX * NY) { + printf("ERROR: written less than expected"); + exit(3); + } + } - { - FILE * f = fopen(argv[5], "w"); - assert(f != 0); - fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); //exchange X and Y - fprintf(f, "%d %d %d\n", NY, NX, NZ); - fwrite(&volume[0], sizeof(float), volume.size(), f); - fclose(f); - } + fclose(f); } diff --git a/device-gen/common/common.h b/device-gen/common/common.h new file mode 100644 index 000000000..5a59c1cbb --- /dev/null +++ b/device-gen/common/common.h @@ -0,0 +1,67 @@ +/* + * common.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#pragma once +#include +#include +#include +#include +#include + +inline float sign(float x) { + return 1.0f - 2.0f*std::signbit(x); +} + +inline void readDAT(const std::string& fileName, std::vector& data, + float& xextent, float& yextent, float& zextent, int& NX, int& NY, int& NZ) +{ + FILE * f = fopen(fileName.c_str(), "r"); + assert(f != 0); + int result = fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); + assert(result == 3); + result = fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); + assert(result == 3); + printf("Read file %s. Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", + fileName.c_str(), xextent, yextent, zextent, NX, NY, NZ); + data.resize(NX * NY * NZ, 0.0f); + result = fread(&data[0], sizeof(float), NX * NY * NZ, f); + if (result != data.size()) { + printf("ERROR: read less than expected"); + exit(3); + } + fclose(f); +} + +inline void writeDAT(const std::string& fileName, std::vector& data, + float xextent, float yextent, float zextent, int NX, int NY, int NZ) +{ + FILE * f = fopen(fileName.c_str(), "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", xextent, yextent, zextent); + fprintf(f, "%d %d %d\n", NX, NY, NZ); + + unsigned char * ptr = (unsigned char *)&data[0]; + if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) + { + ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; + printf("INFO: some symbols were changed while writing\n"); + } + + int result = fwrite(&data[0], sizeof(float), (int)data.size(), f); + if (result != data.size()) { + printf("ERROR: written less than expected"); + exit(3); + } + + fclose(f); +} + diff --git a/device-gen/common/redistance.h b/device-gen/common/redistance.h new file mode 100644 index 000000000..1408cd3e3 --- /dev/null +++ b/device-gen/common/redistance.h @@ -0,0 +1,169 @@ +/* + * resistance.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include + +using namespace std; + +#define _ACCESS(f, x, y) f[(x) + xsize * (y)] + +namespace Redistancing +{ + int xsize, ysize; + float * phi0, * phi; + float dt, invdx, invdy; + float dls[9]; + + template + inline bool anycrossing_dir(int ix, int iy, const float sgn0) + { + const int dx = d == 0, dy = d == 1, dz = d == 2; + + const float fm1 = _ACCESS(phi0, ix - dx, iy - dy); + const float fp1 = _ACCESS(phi0, ix + dx, iy + dy); + + return (fm1 * sgn0 < 0 || fp1 * sgn0 < 0); + } + + inline bool anycrossing(int ix, int iy, const float sgn0) + { + return + anycrossing_dir<0>(ix, iy, sgn0) || + anycrossing_dir<1>(ix, iy, sgn0) ; + } + + float simple_scheme(int ix, int iy, float sgn0, float myphi0); + + float sussman_scheme(int ix, int iy, float sgn0) + { + const float phicenter = _ACCESS(phi, ix, iy); + + const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); + const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; + const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); + const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; + + if (sgn0 == 1) + { + const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; + const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; + + const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; + + return phicenter - dt * sgn0 * G0; + } + else + { + const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; + const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; + + const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; + + return phicenter - dt * sgn0 * G1; + } + } + + void redistancing(const int iterations, const float dt, const float dx, const float dy, + const int xsize, const int ysize, + float * field) + { + Redistancing::xsize = xsize; + Redistancing::ysize = ysize; + Redistancing::dt = dt; + Redistancing::invdx = 1. / dx; + Redistancing::invdy = 1. / dy; + + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const float deltax = dx * ((code % 3) - 1); + const float deltay = dy * ((code % 9) / 3 - 1); + + const float dl = sqrtf(deltax * deltax + deltay * deltay); + + Redistancing::dls[code] = dl; + } + + + Redistancing::phi0 = new float[xsize * ysize]; + memcpy(phi0, field, sizeof(float) * xsize * ysize); + Redistancing::phi = field; + + float * tmp = new float[xsize * ysize]; + for(int t = 0; t < iterations; ++t) + { + if (t % 100 == 0) + printf("t: %d, size: %d %d\n", t, xsize, ysize); + +//#pragma omp parallel for + for(int iy = 0; iy < ysize; ++iy) + for(int ix = 0; ix < xsize; ++ix) + { + const float myval0 = _ACCESS(phi0, ix, iy); + const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); + + const bool boundary = ( + ix == 0 || ix == xsize - 1 || + iy == 0 || iy == ysize - 1); + if (boundary) + tmp[ix + xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); + else + { + if (anycrossing(ix, iy, sgn0)) + tmp[ix + xsize * iy] = myval0; + else + tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); + } + assert(fabs(tmp[ix + xsize * iy]) < 1e7); + } + + memcpy(field, tmp, sizeof(float) * xsize * ysize); + } + + delete [] tmp; + delete [] phi0; + phi0 = NULL; + } + + inline float simple_scheme(int ix, int iy, float sgn0, float myphi0) + { + float mindistance = 1e6f; + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const int xneighbor = ix + (code % 3) - 1; + const int yneighbor = iy + (code % 9) / 3 - 1; + + if (xneighbor < 0 || xneighbor >= xsize) continue; + if (yneighbor < 0 || yneighbor >= ysize) continue; + + const float phi0_neighbor = _ACCESS(phi0, xneighbor, yneighbor); + const float phi_neighbor = _ACCESS(phi, xneighbor, yneighbor); + + const float dl = Redistancing::dls[code]; + + float distance = 0; + + if (sgn0 * phi0_neighbor < 0) + distance = - myphi0 * dl / (phi0_neighbor - myphi0); + else + distance = dl + abs(phi_neighbor); + + mindistance = min(mindistance, distance); + } + + return sgn0 * mindistance; + } +} diff --git a/device-gen/post-process/h52ply.py b/device-gen/post-process/h52ply.py new file mode 100755 index 000000000..c2a2dd31e --- /dev/null +++ b/device-gen/post-process/h52ply.py @@ -0,0 +1,37 @@ +#!/usr/bin/env /Applications/paraview.app/Contents/bin/pvpython + +''' + * Part of CTC/device-gen/post-processing/h52ply.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +import argparse +import os +from paraview.simple import * + +print("h52ply started") +parser = argparse.ArgumentParser(description='Transforms h5 which has xmf to ply using Paraview python lib.', + usage= './h52ply.py -i -o ') +parser.add_argument('-i','--inputFile', help='XMF', required=True) +parser.add_argument('-o','--outputFile', help='PLY', required=True) +args = vars(parser.parse_args()) + +# paraview wants to have absolute path +fullPath = os.path.dirname(os.path.abspath(args['inputFile'])) + '/' +print fullPath + +a13x59xmf = XDMFReader(FileNames=[fullPath + args['inputFile']]) +#a13x59xmf.GridStatus = ['Grid_26'] + +contour1 = Contour(Input=a13x59xmf) +contour1.Isosurfaces = [0.0] + +# save data +SaveData(fullPath + args['outputFile'], proxy=contour1) + diff --git a/device-gen/post-process/plyScale.py b/device-gen/post-process/plyScale.py new file mode 100755 index 000000000..06de0e408 --- /dev/null +++ b/device-gen/post-process/plyScale.py @@ -0,0 +1,108 @@ +#!/usr/bin/env python + +''' + * Part of CTC/device-gen/post-processing/plyScale.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +from plyfile import PlyData, PlyElement +import argparse +import copy +import numpy + +def computeExtent(vertices): + extentMax = [-10e6] * 3 + extentMin = [10e6] * 3 + for i in range(0, len(vertices)): + vi = vertices[i] + for dim in range(0, 3): + extentMax[dim] = max(extentMax[dim], vi[dim]) + extentMin[dim] = min(extentMin[dim], vi[dim]) + + origOrigin = [(extentMax[i] + extentMin[i])/2.0 for i in range(0, 3)] + origExtent = [extentMax[i] - extentMin[i] for i in range(0, 3)] + return (origOrigin, origExtent) + +parser = argparse.ArgumentParser(description='Scales ply file.\n Example: ./plyScale.py -f input.ply -o out.ply -x 150 -y 40 -z 48') +parser.add_argument('-f','--inputFile', help='Input file name', required=True) +parser.add_argument('-o','--outputFile', help='Output file name', required=True) +parser.add_argument('-x','--lx', help='', required=False, default="0") +parser.add_argument('-y','--ly', help='', required=False, default="0") +parser.add_argument('-z','--lz', help='', required=False, default="0") +parser.add_argument('-r','--order', help='values are 0-2', required=False, default="012") +parser.add_argument('-c','--cut', help='values are 0-2', required=False, default="none") +args = vars(parser.parse_args()) + +desiredBox = [float(args['lx']), float(args['ly']), float(args['lz'])] + +plydata = PlyData.read(args['inputFile']) +vertices = plydata['vertex'].data + +# swap coords +order = args['order'] +if (order != "012"): + idx = [int(order[i]) for i in range(0, len(order))] + assert(len(idx) == 3) + print "Swapping axis!" + for i in range(0, len(vertices)): + v = copy.deepcopy(vertices[i]) + for dim in range(0, 3): + vertices[i][dim] = v[ idx[dim] ] + + +# Current box +(origOrigin, origExtent) = computeExtent(vertices) +for dim in range(0, 3): + if desiredBox[dim] == 0: + desiredBox[dim] = origExtent[dim] +print ("Extent is (%f, %f, %f). Center is (%f, %f, %f)."%(origExtent[0], origExtent[1], origExtent[2], + origOrigin[0], origOrigin[1], origOrigin[2])) +for i in range(0, len(vertices)): + for dim in range(0, 3): + vertices[i][dim] -= origOrigin[dim] + vertices[i][dim] *= desiredBox[dim]/origExtent[dim] + +if (args['cut'] != "none"): + print "Cut it!" + toDelete = list() + dim = int(args['cut']) + newInx = [None]*len(vertices) + j = 0 + for i in range(0, len(vertices)): + if (vertices[i][dim] > 0.0): + toDelete.append(i) + else: + newInx[i] = j + j += 1 + + plydata['vertex'].data = numpy.delete(plydata['vertex'].data, toDelete, axis=0) + # remove polygons containing these vertices + setVertToDel = set(toDelete) + faces = plydata['face'].data + facesToDelete = list() + for i in range(0, len(faces)): + curr = set(faces[i][0]) + common = curr & setVertToDel + if (common): + facesToDelete.append(i) + plydata['face'].data = numpy.delete(plydata['face'].data, facesToDelete, axis=0) + + #update vertices in polygons + faces = plydata['face'].data + for i in range(0, len(faces)): + f = faces[i][0] + for i in range(0, len(f)): + f[i] = newInx[ f[i] ] + +(finalOrigin, finalExtent) = computeExtent(vertices) +print ("Extent is (%f, %f, %f). Center is (%f, %f, %f)."%(finalExtent[0], finalExtent[1], finalExtent[2], + finalOrigin[0], finalOrigin[1], finalOrigin[2])) + + +plydata.write(args['outputFile']) diff --git a/device-gen/scripts/README b/device-gen/scripts/README index 6d30e57db..6e58fb6f3 100644 --- a/device-gen/scripts/README +++ b/device-gen/scripts/README @@ -1,5 +1,25 @@ -Generates parabolic funnels. +# Generate microfluidic geometry + +Set of scripts to generate device geometries as Signed Distance Function in *.dat format +To clean the dependencies and then build them: +``` +./cleanall.sh ./makeall.sh -./run.sh +``` + +## Parabolic funnels +Geometry mimicing the microfluidic device by McFaul et al [Cell separation based on size and deformability using microfluidic funnel ratchets](http://www.ncbi.nlm.nih.gov/pubmed/22517056) +To generate a this geometry with 10 rows and 20 columns run: +``` +./run-parab.py -r 10 -c 20 +``` + +## Later displacement device +Geometry reproducing CTC-iChip1 module by Karabacak et al [Microfluidic, marker-free isolation of circulating tumor cells from blood samples](http://www.nature.com/nprot/journal/v9/n3/full/nprot.2014.044.html) + +To build geometry with 13 columns and 59 rows: +``` +./run-egg.py -r 59 -c 13 +``` +Note, the python script requre modifications if you want to change resolution, wall width and other parameters -The result is in the file sdf.dat diff --git a/device-gen/scripts/files.txt b/device-gen/scripts/files.txt deleted file mode 100644 index 217f12814..000000000 --- a/device-gen/scripts/files.txt +++ /dev/null @@ -1,12 +0,0 @@ -r4.dat -r5.dat -r6.dat -r7.dat -r8.dat -r9.dat -r10.dat -r11.dat -r12.dat -r13.dat -r14.dat -r15.dat \ No newline at end of file diff --git a/device-gen/scripts/run-cylinder.py b/device-gen/scripts/run-cylinder.py new file mode 100755 index 000000000..26b9fde90 --- /dev/null +++ b/device-gen/scripts/run-cylinder.py @@ -0,0 +1,20 @@ +#!/usr/bin/env python +''' + * Part of CTC/device-gen/scripts/run-cylinder.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +import os + +N = 32 +radius = 6.0 +length = 16.0 +os.system("../sdf-cylinder//sdf-unit %d %f unit.dat"%(N, radius)) +outFile = "cylinder%d.dat"%(radius) +os.system("../2Dto3D/2Dto3D unit.dat %f %f %d %s"%(length, 0.0, 32, outFile)) diff --git a/device-gen/scripts/run-egg.py b/device-gen/scripts/run-egg.py new file mode 100755 index 000000000..640747fb6 --- /dev/null +++ b/device-gen/scripts/run-egg.py @@ -0,0 +1,88 @@ +#!/usr/bin/env python +''' + * Part of CTC/device-gen/scripts/run-egg.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +import os +import math +import argparse + +parser = argparse.ArgumentParser(description='Generates eggs: ', + usage= './run-egg.py -r -c -d <0|1>') +parser.add_argument('-r','--nRows', help='', required=True) +parser.add_argument('-c','--nColumns', help='', required=True) +#parser.add_argument('-d','--draw', help='values: 0 | 1', required=False, default=1) +args = vars(parser.parse_args()) + +marginZ = 5.0 +marginY = 5.0 # responsible for side walls +ntimesInXdir = 2 # how many times to repeat +eggSize = [56, 32, 48 + 2*marginZ] + +resolution = 1.4 #0.7 +unitXRes = resolution*eggSize[0] +unitYRes = resolution*eggSize[1] +unitZRes = resolution*eggSize[2] +print("Unit grid size: %g %g %g\n"%(unitXRes, unitYRes, unitZRes)) + +nRows = int(args['nRows']) +nColumns = int(args['nColumns']) +#draw = int(args['draw']) == 1 + +angle = 1.7 * math.pi/180.0 + +os.system("../sdf-unit-egg/sdf-unit %d %d %g %g %s"%(unitXRes, unitYRes, eggSize[0], eggSize[1], "egg.dat")) + +nRowsPerShift = int(math.ceil(eggSize[0] / (eggSize[1] * math.tan(angle)))) +if (math.fabs(eggSize[0] / (eggSize[1] * math.tan(angle)) - nRowsPerShift) > 1e-1): + print("ERROR: Suggest changing the angle") + exit() + +padding = float(math.ceil(nRows * eggSize[1] * math.tan(angle))) + +nUniqueRows = nRows +if (nRows > nRowsPerShift): + nUniqueRows = nRowsPerShift + padding = float(round(nRowsPerShift * eggSize[1] * math.tan(angle), 0)) + +print("nRowsPershift = %d, nUniqueRows = %d, Padding = %f"%(nRowsPerShift, nUniqueRows, padding)) + +# workaround +if (padding < 32): + padding = 0 +if (padding == 57): + padding = 56 +padding = padding + 8 # 8 change by hands if needed + +print("Launching rows generation. Padding = %g"%(padding)) +for i in range(nUniqueRows-1, -1, -1): + os.system("../sdf-collage/sdf-collage %s %d %d %f %s"%("egg.dat", nColumns, 1, 0.0, "raw-row.dat")) + xshift = i * 32.0 * math.tan(angle) + print("Calling: ../sdf-shift/sdf-shift %s %g %g %s"%("raw-row.dat", xshift, padding, "row%d.dat"%(i))) + os.system("../sdf-shift/sdf-shift %s %g %g %s"%("raw-row.dat", xshift, padding, "row%d.dat"%(i))) + +with open("files.txt", 'w') as f: + for i in range(nRows-1, -1, -1): + j = i % nRowsPerShift + f.write("row%d.dat\n"%(j)) + +os.system("../sdf-collage/sdf-collage files.txt 1 1 %f %s "%(marginY, "collage.dat")) + +if (ntimesInXdir > 1): + os.system("mv collage.dat collage_temp.dat") + os.system("../sdf-collage/sdf-collage collage_temp.dat %d %d 0.0 collage.dat"%(1, ntimesInXdir)) + +#if (draw): +# os.system("../dat2hdf5/dat2hdf5 collage.dat 2d") +os.system("../2Dto3D/2Dto3D collage.dat %g %g %d %s"%(eggSize[2] - 2*marginZ, marginZ, unitZRes, "%dx%d.dat"%(nColumns, nRows))) +#if (draw): +# os.system("../dat2hdf5/dat2hdf5 %dx%d.dat %dx%d"%(nColumns, nRows, nColumns, nRows)) + +os.system("rm row*.dat") diff --git a/device-gen/scripts/run-parab.py b/device-gen/scripts/run-parab.py new file mode 100755 index 000000000..329872b89 --- /dev/null +++ b/device-gen/scripts/run-parab.py @@ -0,0 +1,55 @@ +#!/usr/bin/env python +''' + * Part of CTC/device-gen/scripts/run-parab.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +import os +import math +import argparse + +parser = argparse.ArgumentParser(description='Generates parabolic funnel obstacles: ', + usage= './run-parab.py -r ') +parser.add_argument('-r','--nRows', help='', required=True) +parser.add_argument('-c','--nColumns', help='', required=True) +parser.add_argument('-d','--draw', help='values: 0 | 1', required=False, default=1) +args = vars(parser.parse_args()) + +nRows = int(args['nRows']) +nColumns = int(args['nColumns']) +draw = int(args['draw']) == 1 + +with open("files.txt", 'w') as f: + for i in range(3, nRows + 3): + f.write("r%d.dat\n"%(i)) + +unitXRes=24/2 +unitYRes=96/2 +unitZRes=128/2 + +cleftWidth = 0.5 +zMargin = 4.0 +zSize = 128.0 + +for i in range(3, nRows+3): + os.system("../sdf-unit-par/sdf-unit %d %d %f %f %f gap%d.dat"%(unitXRes, unitYRes, 24.0, 96.0, cleftWidth*i, i)) + +for i in range(3, nRows+3): + os.system("../sdf-collage/sdf-collage gap%d.dat %d %d %f r%d.dat"%(i, nColumns, 1, 0.0, i)) + +os.system("../sdf-collage/sdf-collage files.txt 1 1 0.0 collage.dat") +if draw == 1: + os.system("../dat2hdf5/dat2hdf5 collage.dat 2d") +outFile = "%dx%d.dat"%(nRows, nColumns) +os.system("../2Dto3D/2Dto3D collage.dat %f %f %d %s"%(zSize - 2*zMargin, zMargin, unitZRes, outFile)) +if draw == 1: + os.system("../dat2hdf5/dat2hdf5 %s 3d"%(outFile)) + +os.system("rm row*.dat") +#os.system("tar -czvf %dx%d.tar.gz %s"%(nRows, nColumns, outFile)) diff --git a/device-gen/scripts/run.sh b/device-gen/scripts/run.sh deleted file mode 100755 index ea81e33be..000000000 --- a/device-gen/scripts/run.sh +++ /dev/null @@ -1,19 +0,0 @@ -#! /usr/local/bin/bash - -unitXRes=32 -unitYRes=128 -unitZRes=64 - -nColumns=2 -# nRows is defined by files.txt - -for i in `seq 3 15`; do - ../sdf-unit-par/sdf-unit $unitXRes $unitYRes 24 96 $i gap$i.dat -done - -for i in `seq 3 15`; do - ../sdf-collage/sdf-collage gap$i.dat $nColumns 1 r$i.dat -done - -../sdf-collage/sdf-collage files.txt 1 1 collage.dat -../2Dto3D/2Dto3D collage.dat 40.0 4.0 $unitZRes sdf.dat diff --git a/device-gen/sdf-collage/Makefile b/device-gen/sdf-collage/Makefile index 306311353..0b7a37ae4 100644 --- a/device-gen/sdf-collage/Makefile +++ b/device-gen/sdf-collage/Makefile @@ -1,5 +1,5 @@ -sdf-collage: main.cpp - g++ main.cpp -O0 -g3 -fopenmp -o sdf-collage +sdf-collage: main.cpp ../common/common.h ../common/redistance.h + g++-4.9 main.cpp -O3 -g -fopenmp -o sdf-collage clean: rm sdf-collage diff --git a/device-gen/sdf-collage/main.cpp b/device-gen/sdf-collage/main.cpp index 690159bc0..edf299824 100644 --- a/device-gen/sdf-collage/main.cpp +++ b/device-gen/sdf-collage/main.cpp @@ -9,8 +9,8 @@ * to employ the present software for their own publications * before getting a written permission from the author of this file. */ - #include +#include #include #include #include @@ -18,106 +18,12 @@ #include #include #include - +#include +#include "../common/redistance.h" +#include "../common/common.h" using namespace std; -#define _ACCESS(f, x, y) f[(x) + xsize * (y)] - -namespace Redistancing -{ - int xsize; - float * phi0, * phi; - float dt, invdx, invdy; - - template - inline bool anycrossing_dir(int ix, int iy, const float sgn0) - { - const int dx = d == 0, dy = d == 1, dz = d == 2; - - const float fm1 = _ACCESS(phi0, ix - dx, iy - dy); - const float fp1 = _ACCESS(phi0, ix + dx, iy + dy); - - return (fm1 * sgn0 < 0 || fp1 * sgn0 < 0); - } - - inline bool anycrossing(int ix, int iy, const float sgn0) - { - return - anycrossing_dir<0>(ix, iy, sgn0) || - anycrossing_dir<1>(ix, iy, sgn0) ; - } - - float sussman_scheme(int ix, int iy, float sgn0) - { - const float phicenter = _ACCESS(phi, ix, iy); - - const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); - const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; - const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); - const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; - - if (sgn0 == 1) - { - const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; - const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; - - const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; - - return phicenter - dt * sgn0 * G0; - } - else - { - const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; - const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; - - const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; - - return phicenter - dt * sgn0 * G1; - } - } - - void redistancing(const int iterations, const float dt, const float dx, const float dy, - const int xsize, const int ysize, - float * field) - { - Redistancing::xsize = xsize; - Redistancing::dt = dt; - Redistancing::invdx = 1. / dx; - Redistancing::invdy = 1. / dy; - - Redistancing::phi0 = new float[xsize * ysize]; - memcpy(phi0, field, sizeof(float) * xsize * ysize); - Redistancing::phi = field; - - float * tmp = new float[xsize * ysize]; - for(int t = 0; t < iterations; ++t) - { - if (t % 30 == 0) - printf("t: %d\n", t); - -#pragma omp parallel for - for(int iy = 0; iy < ysize; ++iy) - for(int ix = 0; ix < xsize; ++ix) - { - const float myval0 = _ACCESS(phi0, ix, iy); - const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); - - if (anycrossing(ix, iy, sgn0) || ix == 0 || ix == xsize - 1 || iy == 0 || iy == ysize - 1) - tmp[ix + xsize * iy] = myval0; - else - tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); - } - - memcpy(field, tmp, sizeof(float) * xsize * ysize); - } - - delete [] tmp; - delete [] phi0; - phi0 = NULL; - } -} - -void mergeSDF(int NX, int NY, vector< vector >& cookie, vector& cake) +void mergeSDF(int NX, int NY, const vector< vector >& cookie, vector& cake) { cake.resize(cookie.size() * cookie[0].size()); printf("SIZE: %d\n", cake.size()); @@ -127,15 +33,16 @@ void mergeSDF(int NX, int NY, vector< vector >& cookie, vector& ca { const int dst = ix + stride * iy; const int iobst = iy / NY; + assert(ix + NX * (iy - iobst*NY) < cookie[iobst].size()); cake[dst] = cookie[iobst][ix + NX * (iy - iobst*NY)]; } } int main(int argc, char ** argv) { - if (argc != 5) + if (argc != 6) { - printf("usage: ./sdf-collage \n"); + printf("usage: ./sdf-collage \n"); return -1; } @@ -150,16 +57,7 @@ int main(int argc, char ** argv) { cookie.resize(1); // for one file - FILE * f = fopen(argv[1], "r"); - assert(f != 0); - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); - printf("Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextent, NX, NY,NZ); - assert(NZ == 1); - cookie[0].resize(NX * NY * NZ, 0.0f); - fread(&cookie[0][0], sizeof(float), NX * NY * NZ, f); - fclose(f); - + readDAT(argv[1], cookie[0], xextent, yextent, zextent, NX, NY, NZ); printf("Populate %d * %d times\n", xtimes, ytimes); const int stride = xtimes * NX; cake.resize(xtimes * ytimes * NX * NY); @@ -196,32 +94,75 @@ int main(int argc, char ** argv) for (int i = files.size() - 1; i >= 0; --i) { printf("Reading file %s ...\n", files[i].c_str()); - FILE * f = fopen(files[i].c_str(), "r"); - assert(f != 0); - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); - printf("Extent: [%g, %g, %g]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextent, NX, NY,NZ); - assert(NZ == 1); - cookie[i].resize(NX * NY * NZ, 0.0f); - fread(&cookie[i][0], sizeof(float), NX * NY * NZ, f); - fclose(f); + readDAT(files[i].c_str(), cookie[i], xextent, yextent, zextent, NX, NY,NZ); + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + if (cookie[i][ix + NX * iy] > 1e3) + { + std::cout << "ERROR in file " << files[i].c_str() << std::endl; + exit(0); + } + } } - mergeSDF(NX, NY, cookie, cake); - } - - const float dx = xextent / NX; - const float dy = yextent / NY; - Redistancing::redistancing(240, 0.25 * min(dx, dy), dx, dy, xtimes * NX, ytimes * NY, &cake[0]); - - { - FILE * f = fopen(argv[4], "w"); - assert(f != 0); - fprintf(f, "%f %f %f\n", xtimes * xextent, ytimes * yextent, 1.0f); - fprintf(f, "%d %d %d\n", xtimes * NX, ytimes * NY, 1); - fwrite(&cake[0], sizeof(float), cake.size(), f); - fclose(f); + mergeSDF(NX, NY, cookie, cake); + + // add walls in Y directio + float wallWidth = atof(argv[4]); + if (wallWidth != 0.0f) + { + int cakeNX = xtimes * NX; + int cakeNY = ytimes * NY; + const float x0 = -xtimes * xextent * 0.5; + const float dx = xtimes * xextent / (cakeNX - 1); + + const float y0 = -ytimes * yextent * 0.5; + const float dy = ytimes * yextent / (cakeNY - 1); + + const float angle = (1.8/180.)*M_PI; + const float normal[] = {-cos(angle), sin(angle)}; + wallWidth = -2*y0*tan(angle); + + float ypick = 25.0f; //15 + float widthOfBufferZone = 8-wallWidth + 0.0*(48 - 2*wallWidth); + float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); + const float angle2 = atan(xpick/ypick); + std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; + const float normal2[] = {-cos(angle2), -sin(angle2)}; + + + const float linePoint[] = {-x0 - wallWidth, y0}; + const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; + + for(int iy = 0; iy < cakeNY; ++iy) + for(int ix = 0; ix < cakeNX; ++ix) + { + const float signX = sign(dx*ix + x0); + float p[] = {dx*ix + x0, dy*iy + y0}; + //float xsdf = std::numeric_limits::min(); + float padding = signbit(-p[0])*widthOfBufferZone; + float xsdf = -1e6; + + if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { + xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); + } else { + xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); + } + + cake[ix + cakeNX*iy] = std::max(cake[ix + cakeNX*iy], xsdf); + //} + //const float xsdf = fabs(dx*ix + x0) - (xextent * 0.5 - wallWidth); + //cake[ix + cakeNX*iy] = std::max(cake[ix + cakeNX*iy], xsdf); + } + } } + const float dx = xextent / (NX - 1); + const float dy = yextent / (NY - 1); + Redistancing::redistancing(1000, 0.25 * min(dx, dy), dx, dy, xtimes * NX, ytimes * NY, &cake[0]); + + writeDAT(argv[5], cake, xtimes * xextent, ytimes * yextent, 1.0f, xtimes * NX, ytimes * NY, 1); return 0; } + diff --git a/device-gen/sdf-cylinder/Makefile b/device-gen/sdf-cylinder/Makefile new file mode 100644 index 000000000..2d07290c1 --- /dev/null +++ b/device-gen/sdf-cylinder/Makefile @@ -0,0 +1,7 @@ +sdf-unit: main.cpp + g++-4.9 main.cpp -O0 -g3 -o sdf-unit + +clean: + rm sdf-unit + +.PHONY = clean diff --git a/device-gen/sdf-cylinder/main.cpp b/device-gen/sdf-cylinder/main.cpp new file mode 100644 index 000000000..1299aca7f --- /dev/null +++ b/device-gen/sdf-cylinder/main.cpp @@ -0,0 +1,61 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include "../common/common.h" +using namespace std; + +#define REAL float + +REAL distToSide(REAL x, REAL y, REAL radius) { + return sqrt(x*x + y*y) - radius; +} + +int main(int argc, char ** argv) +{ + if (argc != 4) + { + printf("usage: ./sdf-cylinder \n"); + return 1; + } + + const int N = atoi(argv[1]); + const REAL radius = atof(argv[2]);; + const REAL extent = 2.0f*radius + 4.0f; + + std:cout << "Will generate SDF with extent " << extent << " " << extent + << ". Grid size " << N << " x " << N << std::endl; + + const REAL xlb = -extent/2.0f; + const REAL ylb = -extent/2.0f; + + vector sdf(N * N, 0.0f); + const REAL dx = extent / (N-1); + const REAL dy = extent / (N-1); + + for(int iy = 0; iy < N; ++iy) + for(int ix = 0; ix < N; ++ix) + { + const REAL x = xlb + ix * dx; + const REAL y = ylb + iy * dy; + + sdf[ix + N * iy] = distToSide(x, y, radius); + } + + writeDAT(argv[3], sdf, extent, extent, REAL(1.0), N, N, 1); + + return 0; +} diff --git a/device-gen/sdf-shift/Makefile b/device-gen/sdf-shift/Makefile new file mode 100644 index 000000000..6bf9c3cae --- /dev/null +++ b/device-gen/sdf-shift/Makefile @@ -0,0 +1,7 @@ +sdf-shift: main.cpp ../common/common.h ../common/redistance.h + g++-4.9 main.cpp -O0 -g3 -std=c++0x -o sdf-shift + +clean: + rm sdf-shift + +.PHONY = clean diff --git a/device-gen/sdf-shift/main.cpp b/device-gen/sdf-shift/main.cpp new file mode 100644 index 000000000..32dd5fd84 --- /dev/null +++ b/device-gen/sdf-shift/main.cpp @@ -0,0 +1,58 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Kirill Lykov on 2015-03-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include "../common/common.h" +#include "../common/redistance.h" +#include +#include + +int main(int argc, char** argv) +{ + if (argc != 5) + { + printf("usage: ./sdf-shift \n"); + return -1; + } + + float xshift = atof(argv[2]); + float xpadding = atof(argv[3]); + + float xextent, yextent, zextent; + int NX, NY,NZ; + std::vector inputGrid; + readDAT(argv[1], inputGrid, xextent, yextent, zextent, NX, NY, NZ); + + float h = xextent / (NX - 1); + int ixshift = xshift / h; + int ipadding = xpadding / h + 1; + + float minVal = *std::min(inputGrid.begin(), inputGrid.end()); + std::vector outGrid((NX + ipadding) * NY * NZ, -1e6); + assert(NZ == 1); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX ; ++ix) + { + int newIx = (ix + ixshift) % (NX + ipadding); + assert(fabs(inputGrid[ix + NX * iy]) < 1e3); + outGrid[newIx + (NX + ipadding) * iy] = inputGrid[ix + NX * iy]; + } + + const float dx = (xextent + xpadding)/ (NX + ipadding - 1); + const float dy = yextent / (NY - 1); + Redistancing::redistancing(1000, 0.25 * min(dx, dy), dx, dy, NX + ipadding, NY, &outGrid[0]); + + writeDAT(argv[4], outGrid, xextent + xpadding, yextent, zextent, NX + ipadding, NY, NZ); + + + return 0; +} diff --git a/device-gen/sdf-unit-egg/Makefile b/device-gen/sdf-unit-egg/Makefile new file mode 100644 index 000000000..ea8da56e9 --- /dev/null +++ b/device-gen/sdf-unit-egg/Makefile @@ -0,0 +1,7 @@ +sdf-unit: main.cpp + g++-4.9 main.cpp -O0 -g3 -std=c++0x -o sdf-unit + +clean: + rm sdf-unit + +.PHONY = clean diff --git a/device-gen/sdf-unit-egg/main.cpp b/device-gen/sdf-unit-egg/main.cpp new file mode 100644 index 000000000..a1eccf68d --- /dev/null +++ b/device-gen/sdf-unit-egg/main.cpp @@ -0,0 +1,125 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Kirill Lykov on 2015-03-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include +#include "../common/common.h" +using namespace std; + +struct Egg +{ + float r1, r2, alpha; + + Egg() + : r1(12.0f), r2(8.5f), alpha(0.03f) + { + } + + float x2y(float x) const { + return sqrt(r2*r2 * exp(-alpha * x) * (1.0f - x*x/r1/r1)); + } + + void run(vector& vx, vector& vy) { + int N = 500; + float dx = 2.0f * r1 / (N - 1); + for (int i = 0; i < N; ++i) { + float x = i * dx - r1; + float y = x2y(x); + vx.push_back(x); + vy.push_back(y); + } + + auto vxRev = vx; + vx.insert(vx.end(), vxRev.rbegin(), vxRev.rend()); + + auto vyRev = vy; + for_each(vyRev.begin(), vyRev.end(), [](float& i) { i *= -1.0f; }); + vy.insert(vy.end(), vyRev.rbegin(), vyRev.rend()); + } +}; + +int main(int argc, char ** argv) +{ + if (argc != 6) + { + printf("usage: ./sdf-unit-egg \n"); + return 1; + } + + const int NX = atoi(argv[1]); + const int NY = atoi(argv[2]); + const float xextent = atof(argv[3]); + const float yextent = atof(argv[4]); + + vector xs, ys; + Egg egg; + egg.run(xs, ys); + + const float xlb = -xextent/2.0f; + const float ylb = -yextent/2.0f; + printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", + NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); + + vector sdf(NX * NY, 0.0f); + const float dx = xextent / NX; + const float dy = yextent / NY; + const int nsamples = xs.size(); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const float x = xlb + ix * dx; + const float y = ylb + iy * dy; + + float distance2 = 1e6; + int iclosest = 0; + for(int i = 0; i < nsamples ; ++i) + { + const float xd = xs[i] - x; + const float yd = ys[i] - y; + const float candidate = xd * xd + yd * yd; + + if (candidate < distance2) + { + iclosest = i; + distance2 = candidate; + } + } + + float s = -1; + + { + const float ycurve = egg.x2y(x); + if (x >= -egg.r1 && x <= egg.r1 && fabs(y) <= ycurve) + s = +1; + } + + + sdf[ix + NX * iy] = s * sqrt(distance2); + } + + writeDAT(argv[5], sdf, xextent, yextent, 1.0f, NX, NY, 1); + //FILE * f = fopen(argv[5], "w"); + //fprintf(f, "%f %f %f\n", xextent, yextent, 1.0f); + //fprintf(f, "%d %d %d\n", NX, NY, 1); + //fwrite(sdf, sizeof(float), NX * NY, f); + //fclose(f); + + //delete [] sdf; + + return 0; +} + diff --git a/device-gen/sdf-unit-par/Makefile b/device-gen/sdf-unit-par/Makefile index 8d0bef89c..eea0d33ce 100644 --- a/device-gen/sdf-unit-par/Makefile +++ b/device-gen/sdf-unit-par/Makefile @@ -1,7 +1,7 @@ sdf-unit: main.cpp - g++ main.cpp -o sdf-unit + g++-4.9 main.cpp -O3 -o sdf-unit clean: rm sdf-unit -.PHONY = clean \ No newline at end of file +.PHONY = clean diff --git a/device-gen/sdf-unit-par/main.cpp b/device-gen/sdf-unit-par/main.cpp index 0c7951959..6e23ef8ec 100644 --- a/device-gen/sdf-unit-par/main.cpp +++ b/device-gen/sdf-unit-par/main.cpp @@ -15,6 +15,7 @@ #include #include #include +#include "../common/common.h" using namespace std; struct Parabola @@ -80,7 +81,7 @@ int main(int argc, char ** argv) printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); - float * sdf = new float[NX * NY]; + vector sdf(NX * NY, 0.0f); const float dx = xextent / NX; const float dy = yextent / NY; const int nsamples = xs.size(); @@ -120,13 +121,7 @@ int main(int argc, char ** argv) sdf[ix + NX * iy] = s * sqrt(distance2); } - FILE * f = fopen(argv[6], "w"); - fprintf(f, "%f %f %f\n", xextent, yextent, 1.0f); - fprintf(f, "%d %d %d\n", NX, NY, 1); - fwrite(sdf, sizeof(float), NX * NY, f); - fclose(f); - - delete [] sdf; + writeDAT(argv[6], sdf, xextent, yextent, 1.0f, NX, NY, 1); return 0; } From 9768ed4f51243109455816ab312145f66ce96b10 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 15:44:35 +0200 Subject: [PATCH 02/63] Splitted redistance into header and source files --- device-gen/common/redistance.cpp | 128 +++++++++++++++++++++++++++++++ device-gen/common/redistance.h | 122 +---------------------------- 2 files changed, 130 insertions(+), 120 deletions(-) create mode 100644 device-gen/common/redistance.cpp diff --git a/device-gen/common/redistance.cpp b/device-gen/common/redistance.cpp new file mode 100644 index 000000000..ddc6b506a --- /dev/null +++ b/device-gen/common/redistance.cpp @@ -0,0 +1,128 @@ +#include "redistance.h" + +namespace Redistancing +{ + float sussman_scheme(int ix, int iy, float sgn0) + { + const float phicenter = _ACCESS(phi, ix, iy); + + const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); + const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; + const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); + const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; + + if (sgn0 == 1) + { + const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; + const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; + + const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; + + return phicenter - dt * sgn0 * G0; + } + else + { + const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; + const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; + + const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; + + return phicenter - dt * sgn0 * G1; + } + } + + void redistancing(const int iterations, const float dt, const float dx, const float dy, + const int xsize, const int ysize, + float * field) + { + Redistancing::xsize = xsize; + Redistancing::ysize = ysize; + Redistancing::dt = dt; + Redistancing::invdx = 1. / dx; + Redistancing::invdy = 1. / dy; + + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const float deltax = dx * ((code % 3) - 1); + const float deltay = dy * ((code % 9) / 3 - 1); + + const float dl = sqrtf(deltax * deltax + deltay * deltay); + + Redistancing::dls[code] = dl; + } + + + Redistancing::phi0 = new float[xsize * ysize]; + memcpy(phi0, field, sizeof(float) * xsize * ysize); + Redistancing::phi = field; + + float * tmp = new float[xsize * ysize]; + for(int t = 0; t < iterations; ++t) + { + if (t % 100 == 0) + printf("t: %d, size: %d %d\n", t, xsize, ysize); + +//#pragma omp parallel for + for(int iy = 0; iy < ysize; ++iy) + for(int ix = 0; ix < xsize; ++ix) + { + const float myval0 = _ACCESS(phi0, ix, iy); + const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); + + const bool boundary = ( + ix == 0 || ix == xsize - 1 || + iy == 0 || iy == ysize - 1); + if (boundary) + tmp[ix + xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); + else + { + if (anycrossing(ix, iy, sgn0)) + tmp[ix + xsize * iy] = myval0; + else + tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); + } + assert(fabs(tmp[ix + xsize * iy]) < 1e7); + } + + memcpy(field, tmp, sizeof(float) * xsize * ysize); + } + + delete [] tmp; + delete [] phi0; + phi0 = NULL; + } + + float simple_scheme(int ix, int iy, float sgn0, float myphi0) + { + float mindistance = 1e6f; + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const int xneighbor = ix + (code % 3) - 1; + const int yneighbor = iy + (code % 9) / 3 - 1; + + if (xneighbor < 0 || xneighbor >= xsize) continue; + if (yneighbor < 0 || yneighbor >= ysize) continue; + + const float phi0_neighbor = _ACCESS(phi0, xneighbor, yneighbor); + const float phi_neighbor = _ACCESS(phi, xneighbor, yneighbor); + + const float dl = Redistancing::dls[code]; + + float distance = 0; + + if (sgn0 * phi0_neighbor < 0) + distance = - myphi0 * dl / (phi0_neighbor - myphi0); + else + distance = dl + abs(phi_neighbor); + + mindistance = min(mindistance, distance); + } + + return sgn0 * mindistance; + } + +} diff --git a/device-gen/common/redistance.h b/device-gen/common/redistance.h index 1408cd3e3..2dd0e175e 100644 --- a/device-gen/common/redistance.h +++ b/device-gen/common/redistance.h @@ -44,126 +44,8 @@ namespace Redistancing float simple_scheme(int ix, int iy, float sgn0, float myphi0); - float sussman_scheme(int ix, int iy, float sgn0) - { - const float phicenter = _ACCESS(phi, ix, iy); - - const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); - const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; - const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); - const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; - - if (sgn0 == 1) - { - const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; - const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; - - const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; - - return phicenter - dt * sgn0 * G0; - } - else - { - const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; - const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; - - const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; - - return phicenter - dt * sgn0 * G1; - } - } + float sussman_scheme(int ix, int iy, float sgn0); void redistancing(const int iterations, const float dt, const float dx, const float dy, - const int xsize, const int ysize, - float * field) - { - Redistancing::xsize = xsize; - Redistancing::ysize = ysize; - Redistancing::dt = dt; - Redistancing::invdx = 1. / dx; - Redistancing::invdy = 1. / dy; - - for(int code = 0; code < 3 * 3; ++code) - { - if (code == 1 + 3) continue; - - const float deltax = dx * ((code % 3) - 1); - const float deltay = dy * ((code % 9) / 3 - 1); - - const float dl = sqrtf(deltax * deltax + deltay * deltay); - - Redistancing::dls[code] = dl; - } - - - Redistancing::phi0 = new float[xsize * ysize]; - memcpy(phi0, field, sizeof(float) * xsize * ysize); - Redistancing::phi = field; - - float * tmp = new float[xsize * ysize]; - for(int t = 0; t < iterations; ++t) - { - if (t % 100 == 0) - printf("t: %d, size: %d %d\n", t, xsize, ysize); - -//#pragma omp parallel for - for(int iy = 0; iy < ysize; ++iy) - for(int ix = 0; ix < xsize; ++ix) - { - const float myval0 = _ACCESS(phi0, ix, iy); - const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); - - const bool boundary = ( - ix == 0 || ix == xsize - 1 || - iy == 0 || iy == ysize - 1); - if (boundary) - tmp[ix + xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); - else - { - if (anycrossing(ix, iy, sgn0)) - tmp[ix + xsize * iy] = myval0; - else - tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); - } - assert(fabs(tmp[ix + xsize * iy]) < 1e7); - } - - memcpy(field, tmp, sizeof(float) * xsize * ysize); - } - - delete [] tmp; - delete [] phi0; - phi0 = NULL; - } - - inline float simple_scheme(int ix, int iy, float sgn0, float myphi0) - { - float mindistance = 1e6f; - for(int code = 0; code < 3 * 3; ++code) - { - if (code == 1 + 3) continue; - - const int xneighbor = ix + (code % 3) - 1; - const int yneighbor = iy + (code % 9) / 3 - 1; - - if (xneighbor < 0 || xneighbor >= xsize) continue; - if (yneighbor < 0 || yneighbor >= ysize) continue; - - const float phi0_neighbor = _ACCESS(phi0, xneighbor, yneighbor); - const float phi_neighbor = _ACCESS(phi, xneighbor, yneighbor); - - const float dl = Redistancing::dls[code]; - - float distance = 0; - - if (sgn0 * phi0_neighbor < 0) - distance = - myphi0 * dl / (phi0_neighbor - myphi0); - else - distance = dl + abs(phi_neighbor); - - mindistance = min(mindistance, distance); - } - - return sgn0 * mindistance; - } + const int xsize, const int ysize, float * field); } From 2aea886460857053b6503c5b10b79be50b3ef57b Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 19:17:46 +0200 Subject: [PATCH 03/63] Extracted common code for all devices --- device-gen/common/2Dto3D.cpp | 83 +++++++++++++++++++ device-gen/common/2Dto3D.h | 6 ++ device-gen/common/collage.cpp | 134 +++++++++++++++++++++++++++++++ device-gen/common/collage.h | 13 +++ device-gen/common/common.h | 4 +- device-gen/common/redistance.cpp | 103 +++++++++++------------- device-gen/common/redistance.h | 29 +++---- 7 files changed, 302 insertions(+), 70 deletions(-) create mode 100644 device-gen/common/2Dto3D.cpp create mode 100644 device-gen/common/2Dto3D.h create mode 100644 device-gen/common/collage.cpp create mode 100644 device-gen/common/collage.h diff --git a/device-gen/common/2Dto3D.cpp b/device-gen/common/2Dto3D.cpp new file mode 100644 index 000000000..9bd735c37 --- /dev/null +++ b/device-gen/common/2Dto3D.cpp @@ -0,0 +1,83 @@ +/* + * main.cpp + * Part of CTC/device-gen/2Dto3D/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#include "2Dto3D.h" +#include +#include +#include +#include +#include +#include +#include "common.h" + +using namespace std; + +void conver2Dto3D(const int NX, const int NY, const float xextent, const float yextent, const std::vector& slice, + const int NZ, const float zextent, const float zmargin, const std::string& fileName) +{ + printf("Generating data with extent [%f, %f, %f], dimensions [%d, %d, %d], zmargin %f\n", + xextent, yextent, zextent + 2 * zmargin, NX, NY, NZ, zmargin); + + vector outputslice(NX * NY, 0.0f); + + const float z0 = -zextent * 0.5 - zmargin; + const float dz = (zextent + 2 * zmargin) / (NZ - 1); + + FILE * f = fopen(fileName.c_str(), "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); + fprintf(f, "%d %d %d\n", NY, NX, NZ); + + for(int iz = 0; iz < NZ; ++iz) + { + const float z = z0 + iz * dz; + +#pragma omp parallel for + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const float xysdf = slice[ix + NX * (NY - 1 - iy)]; // NY -1 to change Y-axis direction + float val = xysdf; + if (zmargin != 0.0f) { + const float zsdf = fabs(z) - zextent * 0.5; + if (xysdf < 0) + val = max(zsdf, xysdf); + else + val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); + } + assert(iy + NY * (ix) < outputslice.size()); + + assert(fabs(val) < 1e3); // to check that the value has reasonable range + outputslice[iy + NY * ix] = val; + } + + if (iz == 0) + { + unsigned char * ptr = (unsigned char *)&outputslice[0]; + if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) + { + ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; + printf("INFO: some symbols were changed while writing\n"); + } + } + + int result = fwrite(&outputslice.front(), sizeof(float), NX * NY, f); + + if (result != NX * NY) { + printf("ERROR: written less than expected"); + exit(3); + } + + } + + fclose(f); +} + diff --git a/device-gen/common/2Dto3D.h b/device-gen/common/2Dto3D.h new file mode 100644 index 000000000..74d617476 --- /dev/null +++ b/device-gen/common/2Dto3D.h @@ -0,0 +1,6 @@ +#pragma once +#include +#include + +void conver2Dto3D(const int NX, const int NY, const float xextent, const float yextent, const std::vector& slice, + const int NZ, const float zextent, const float zmargin, const std::string& fileName); diff --git a/device-gen/common/collage.cpp b/device-gen/common/collage.cpp new file mode 100644 index 000000000..7ac227cda --- /dev/null +++ b/device-gen/common/collage.cpp @@ -0,0 +1,134 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-collage/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#include "collage.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "common.h" +using namespace std; + +static void mergeSDF(int NX, int NY, const vector< vector >& sampleSDF, vector& outputSDF) +{ + outputSDF.resize(sampleSDF.size() * sampleSDF[0].size()); + printf("SIZE: %d\n", outputSDF.size()); + const int stride = NX; + for(int iy = 0; iy < sampleSDF.size() * NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const int dst = ix + stride * iy; + const int iobst = iy / NY; + assert(ix + NX * (iy - iobst*NY) < sampleSDF[iobst].size()); + outputSDF[dst] = sampleSDF[iobst][ix + NX * (iy - iobst*NY)]; + } +} + +void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, + const int xtimes, const int ytimes, std::vector& outputSDF) +{ + printf("Populate %d * %d times\n", xtimes, ytimes); + const int stride = xtimes * NX; + outputSDF.resize(xtimes * ytimes * NX * NY); + + for(int ty = 0; ty < ytimes; ++ty) + for(int tx = 0; tx < xtimes; ++tx) { + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const int gx = ix + NX * tx; + const int gy = iy + NY * ty; + const int dst = gx + stride * gy; + + assert(dst < outputSDF.size()); + assert(ix + NX * iy < sampleSDF.size()); + outputSDF[dst] = sampleSDF[ix + NX * iy]; + } + } +} + +void collageSDF(const int NX, const int NY, const float xextent, const float yextent, + const vector< vector >& sampleSDF, const int ytimes, + bool wallInY, vector& outputSDF) +{ + mergeSDF(NX, NY, sampleSDF, outputSDF); + const int xtimes = 1; + if (wallInY) + { + int outputSDFNX = xtimes * NX; + int outputSDFNY = ytimes * NY; + const float x0 = -xtimes * xextent * 0.5; + const float dx = xtimes * xextent / (outputSDFNX - 1); + + const float y0 = -ytimes * yextent * 0.5; + const float dy = ytimes * yextent / (outputSDFNY - 1); + + const float angle = (1.8/180.)*M_PI; + const float normal[] = {-cos(angle), sin(angle)}; + const float wallWidth = -2*y0*tan(angle); + + float ypick = 25.0f; //15 + float widthOfBufferZone = 8-wallWidth + 0.0*(48 - 2*wallWidth); + float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); + const float angle2 = atan(xpick/ypick); + std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; + const float normal2[] = {-cos(angle2), -sin(angle2)}; + + + const float linePoint[] = {-x0 - wallWidth, y0}; + const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; + + for(int iy = 0; iy < outputSDFNY; ++iy) + for(int ix = 0; ix < outputSDFNX; ++ix) + { + const float signX = sign(dx*ix + x0); + float p[] = {dx*ix + x0, dy*iy + y0}; + float padding = signbit(-p[0])*widthOfBufferZone; + float xsdf = -1e6; + + if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { + xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); + } else { + xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); + } + + outputSDF[ix + outputSDFNX*iy] = std::max(outputSDF[ix + outputSDFNX*iy], xsdf); + } + } +} + +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, std::vector& inputGrid, + const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid) +{ + float h = xextent / (NX - 1); + int ixshift = xshift / h; + int ipadding = xpadding / h + 1; + newNX = NX + ipadding; + newXextent = xextent + xpadding; + + float minVal = *std::min(inputGrid.begin(), inputGrid.end()); + outGrid.resize(newNX * NY, -1e6); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX ; ++ix) + { + int newIx = (ix + ixshift) % newNX; + assert(fabs(inputGrid[ix + NX * iy]) < 1e3); + outGrid[newIx + newNX * iy] = inputGrid[ix + NX * iy]; + } +} + diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h new file mode 100644 index 000000000..daa42720d --- /dev/null +++ b/device-gen/common/collage.h @@ -0,0 +1,13 @@ +#pragma once + +#include + +void collageSDF(const int NX, const int NY, const float xextent, const float yextent, + const std::vector< std::vector >& sampleSDF, const int ytimes, + bool wallInY, std::vector& outputSDF); + +void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, + const int xtimes, const int ytimes, std::vector& outputSDF); + +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, std::vector& inputGrid, + const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid); diff --git a/device-gen/common/common.h b/device-gen/common/common.h index 5a59c1cbb..1bc90cfc3 100644 --- a/device-gen/common/common.h +++ b/device-gen/common/common.h @@ -22,7 +22,7 @@ inline float sign(float x) { } inline void readDAT(const std::string& fileName, std::vector& data, - float& xextent, float& yextent, float& zextent, int& NX, int& NY, int& NZ) + int& NX, int& NY, int& NZ, float& xextent, float& yextent, float& zextent) { FILE * f = fopen(fileName.c_str(), "r"); assert(f != 0); @@ -42,7 +42,7 @@ inline void readDAT(const std::string& fileName, std::vector& data, } inline void writeDAT(const std::string& fileName, std::vector& data, - float xextent, float yextent, float zextent, int NX, int NY, int NZ) + const int NX, const int NY, const int NZ, float xextent, float yextent, float zextent) { FILE * f = fopen(fileName.c_str(), "w"); assert(f != 0); diff --git a/device-gen/common/redistance.cpp b/device-gen/common/redistance.cpp index ddc6b506a..97f1feda9 100644 --- a/device-gen/common/redistance.cpp +++ b/device-gen/common/redistance.cpp @@ -1,100 +1,96 @@ #include "redistance.h" +#include +#include -namespace Redistancing -{ - float sussman_scheme(int ix, int iy, float sgn0) + float Redistance::sussman_scheme(int ix, int iy, float sgn0) { - const float phicenter = _ACCESS(phi, ix, iy); + const float phicenter = _ACCESS(m_phi, ix, iy); - const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); - const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; - const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); - const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; + const float dphidxm = phicenter - _ACCESS(m_phi, ix - 1, iy); + const float dphidxp = _ACCESS(m_phi, ix + 1, iy) - phicenter; + const float dphidym = phicenter - _ACCESS(m_phi, ix, iy - 1); + const float dphidyp = _ACCESS(m_phi, ix, iy + 1) - phicenter; if (sgn0 == 1) { - const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; - const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; + const float xgrad0 = std::max( max(0.0f, dphidxm), -std::min(0.0f, dphidxp)) * m_invdx; + const float ygrad0 = std::max( max(0.0f, dphidym), -min(0.0f, dphidyp)) * m_invdy; - const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; + const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1.0f; - return phicenter - dt * sgn0 * G0; + return phicenter - m_dt * sgn0 * G0; } else { - const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; - const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; + const float xgrad1 = std::max( -min(0.0f, dphidxm), std::max(0.0f, dphidxp)) * m_invdx; + const float ygrad1 = std::max( -min(0.0f, dphidym), std::max(0.0f, dphidyp)) * m_invdy; - const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; + const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1.0f; - return phicenter - dt * sgn0 * G1; + return phicenter - m_dt * sgn0 * G1; } } - void redistancing(const int iterations, const float dt, const float dx, const float dy, - const int xsize, const int ysize, - float * field) - { - Redistancing::xsize = xsize; - Redistancing::ysize = ysize; - Redistancing::dt = dt; - Redistancing::invdx = 1. / dx; - Redistancing::invdy = 1. / dy; +Redistance::Redistance(const float dt, const float dx, const float dy, + const int xsize, const int ysize) +: m_xsize(xsize), m_ysize(ysize), m_dt(dt), m_dx(dx), m_dy(dy), m_invdx(1.0f/dx), m_invdy(1.0f/dy), m_phi0(nullptr), m_phi(nullptr) +{} + void Redistance::run(const int iterations, float * field) + { for(int code = 0; code < 3 * 3; ++code) { if (code == 1 + 3) continue; - const float deltax = dx * ((code % 3) - 1); - const float deltay = dy * ((code % 9) / 3 - 1); + const float deltax = m_dx * ((code % 3) - 1); + const float deltay = m_dy * ((code % 9) / 3 - 1); const float dl = sqrtf(deltax * deltax + deltay * deltay); - Redistancing::dls[code] = dl; + m_dls[code] = dl; } + m_phi0 = new float[m_xsize * m_ysize]; + memcpy(m_phi0, field, sizeof(float) * m_xsize * m_ysize); + m_phi = field; - Redistancing::phi0 = new float[xsize * ysize]; - memcpy(phi0, field, sizeof(float) * xsize * ysize); - Redistancing::phi = field; - - float * tmp = new float[xsize * ysize]; + float * tmp = new float[m_xsize * m_ysize]; for(int t = 0; t < iterations; ++t) { if (t % 100 == 0) - printf("t: %d, size: %d %d\n", t, xsize, ysize); + printf("t: %d, size: %d %d\n", t, m_xsize, m_ysize); //#pragma omp parallel for - for(int iy = 0; iy < ysize; ++iy) - for(int ix = 0; ix < xsize; ++ix) + for(int iy = 0; iy < m_ysize; ++iy) + for(int ix = 0; ix < m_xsize; ++ix) { - const float myval0 = _ACCESS(phi0, ix, iy); + const float myval0 = _ACCESS(m_phi0, ix, iy); const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); const bool boundary = ( - ix == 0 || ix == xsize - 1 || - iy == 0 || iy == ysize - 1); + ix == 0 || ix == m_xsize - 1 || + iy == 0 || iy == m_ysize - 1); if (boundary) - tmp[ix + xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); + tmp[ix + m_xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); else { if (anycrossing(ix, iy, sgn0)) - tmp[ix + xsize * iy] = myval0; + tmp[ix + m_xsize * iy] = myval0; else - tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); + tmp[ix + m_xsize * iy] = sussman_scheme(ix, iy, sgn0); } - assert(fabs(tmp[ix + xsize * iy]) < 1e7); + assert(fabs(tmp[ix + m_xsize * iy]) < 1e7); } - memcpy(field, tmp, sizeof(float) * xsize * ysize); + memcpy(field, tmp, sizeof(float) * m_xsize * m_ysize); } delete [] tmp; - delete [] phi0; - phi0 = NULL; + delete [] m_phi0; + m_phi0 = nullptr; } - float simple_scheme(int ix, int iy, float sgn0, float myphi0) + float Redistance::simple_scheme(int ix, int iy, float sgn0, float myphi0) { float mindistance = 1e6f; for(int code = 0; code < 3 * 3; ++code) @@ -104,13 +100,13 @@ namespace Redistancing const int xneighbor = ix + (code % 3) - 1; const int yneighbor = iy + (code % 9) / 3 - 1; - if (xneighbor < 0 || xneighbor >= xsize) continue; - if (yneighbor < 0 || yneighbor >= ysize) continue; + if (xneighbor < 0 || xneighbor >= m_xsize) continue; + if (yneighbor < 0 || yneighbor >= m_ysize) continue; - const float phi0_neighbor = _ACCESS(phi0, xneighbor, yneighbor); - const float phi_neighbor = _ACCESS(phi, xneighbor, yneighbor); + const float phi0_neighbor = _ACCESS(m_phi0, xneighbor, yneighbor); + const float phi_neighbor = _ACCESS(m_phi, xneighbor, yneighbor); - const float dl = Redistancing::dls[code]; + const float dl = m_dls[code]; float distance = 0; @@ -119,10 +115,9 @@ namespace Redistancing else distance = dl + abs(phi_neighbor); - mindistance = min(mindistance, distance); + mindistance = std::min(mindistance, distance); } return sgn0 * mindistance; } -} diff --git a/device-gen/common/redistance.h b/device-gen/common/redistance.h index 2dd0e175e..6e815e960 100644 --- a/device-gen/common/redistance.h +++ b/device-gen/common/redistance.h @@ -15,22 +15,22 @@ using namespace std; -#define _ACCESS(f, x, y) f[(x) + xsize * (y)] +#define _ACCESS(f, x, y) f[(x) + m_xsize * (y)] -namespace Redistancing +class Redistance { - int xsize, ysize; - float * phi0, * phi; - float dt, invdx, invdy; - float dls[9]; + int m_xsize, m_ysize; + float * m_phi0, * m_phi; + float m_dt, m_dx, m_dy, m_invdx, m_invdy; + float m_dls[9]; template inline bool anycrossing_dir(int ix, int iy, const float sgn0) { const int dx = d == 0, dy = d == 1, dz = d == 2; - const float fm1 = _ACCESS(phi0, ix - dx, iy - dy); - const float fp1 = _ACCESS(phi0, ix + dx, iy + dy); + const float fm1 = _ACCESS(m_phi0, ix - dx, iy - dy); + const float fp1 = _ACCESS(m_phi0, ix + dx, iy + dy); return (fm1 * sgn0 < 0 || fp1 * sgn0 < 0); } @@ -39,13 +39,14 @@ namespace Redistancing { return anycrossing_dir<0>(ix, iy, sgn0) || - anycrossing_dir<1>(ix, iy, sgn0) ; + anycrossing_dir<1>(ix, iy, sgn0); } - + float simple_scheme(int ix, int iy, float sgn0, float myphi0); float sussman_scheme(int ix, int iy, float sgn0); - - void redistancing(const int iterations, const float dt, const float dx, const float dy, - const int xsize, const int ysize, float * field); -} +public: + Redistance(const float dt, const float dx, const float dy, + const int xsize, const int ysize); + void run(const int iterations, float * field); +}; From ea562a7895bf0b6b9801d80f99052f4550b1087c Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 19:18:25 +0200 Subject: [PATCH 04/63] Added ctc-ichip first version --- device-gen/ctc-ichip/Makefile | 19 ++++ device-gen/ctc-ichip/main.cpp | 198 ++++++++++++++++++++++++++++++++++ 2 files changed, 217 insertions(+) create mode 100644 device-gen/ctc-ichip/Makefile create mode 100644 device-gen/ctc-ichip/main.cpp diff --git a/device-gen/ctc-ichip/Makefile b/device-gen/ctc-ichip/Makefile new file mode 100644 index 000000000..558ff7e5e --- /dev/null +++ b/device-gen/ctc-ichip/Makefile @@ -0,0 +1,19 @@ +CXX = g++-4.9 +CXXFLAGS += -O0 -g3 -std=c++11 + +ctc-ichip: *.cpp collage.o redistance.o 2Dto3D.o + $(CXX) $(CXXFLAGS) collage.o redistance.o 2Dto3D.o main.cpp -o ctc-ichip + +collage.o: ../common/collage.h ../common/collage.cpp + $(CXX) $(CXXFLAGS) -c $^ + +redistance.o: ../common/redistance.h ../common/redistance.cpp + $(CXX) $(CXXFLAGS) -c $^ + +2Dto3D.o: ../common/2Dto3D.h ../common/2Dto3D.cpp + $(CXX) $(CXXFLAGS) -c $^ + +clean: + rm -f test *.o *.d *.h.gch + + diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp new file mode 100644 index 000000000..74a929c40 --- /dev/null +++ b/device-gen/ctc-ichip/main.cpp @@ -0,0 +1,198 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Kirill Lykov on 2015-03-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include +#include "../common/common.h" +#include "../common/collage.h" +#include "../common/redistance.h" +#include "../common/2Dto3D.h" + +using namespace std; + +struct Egg +{ + float r1, r2, alpha; + + Egg() + : r1(12.0f), r2(8.5f), alpha(0.03f) + { + } + + float x2y(float x) const { + return sqrt(r2*r2 * exp(-alpha * x) * (1.0f - x*x/r1/r1)); + } + + void run(vector& vx, vector& vy) { + int N = 500; + float dx = 2.0f * r1 / (N - 1); + for (int i = 0; i < N; ++i) { + float x = i * dx - r1; + float y = x2y(x); + vx.push_back(x); + vy.push_back(y); + } + + auto vxRev = vx; + vx.insert(vx.end(), vxRev.rbegin(), vxRev.rend()); + + auto vyRev = vy; + for_each(vyRev.begin(), vyRev.end(), [](float& i) { i *= -1.0f; }); + vy.insert(vy.end(), vyRev.rbegin(), vyRev.rend()); + } +}; + +void generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) +{ + vector xs, ys; + Egg egg; + egg.run(xs, ys); + + const float xlb = -xextent/2.0f; + const float ylb = -yextent/2.0f; + printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", + NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); + + sdf.resize(NX * NY, 0.0f); + const float dx = xextent / NX; + const float dy = yextent / NY; + const int nsamples = xs.size(); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const float x = xlb + ix * dx; + const float y = ylb + iy * dy; + + float distance2 = 1e6; + int iclosest = 0; + for(int i = 0; i < nsamples ; ++i) + { + const float xd = xs[i] - x; + const float yd = ys[i] - y; + const float candidate = xd * xd + yd * yd; + + if (candidate < distance2) + { + iclosest = i; + distance2 = candidate; + } + } + + float s = -1; + + { + const float ycurve = egg.x2y(x); + if (x >= -egg.r1 && x <= egg.r1 && fabs(y) <= ycurve) + s = +1; + } + + + sdf[ix + NX * iy] = s * sqrt(distance2); + } +} + +typedef vector SDF; + +int main(int argc, char ** argv) +{ + + int nColumns = 5; + int nRows = 57; + int nrepeat = 2; + + int zmargin = 5.0f; + + // 1 Create 2D SDF for 1 obstacle + const float eggSizeX = 56.0f; // size of the egg with the empty space aroung it + const float eggSizeY = 32.0f; + const float eggSizeZ = 58.0f; + const float resolution = 1.0f; // how many grid point per one micron + const int eggNX = static_cast(eggSizeX * resolution); + const int eggNY = static_cast(eggSizeY * resolution); + const int eggNZ = static_cast(eggSizeZ * resolution); + + SDF eggSdf; + generateEggSDF(eggNX, eggNY, eggSizeX, eggSizeY, eggSdf); + writeDAT("out.dat", eggSdf, eggNX, eggNY, 1, eggSizeX, eggSizeY, 1.0f); + + // 2 Create 1 row of obstacles + int rowNX = nColumns*eggNX; + int rowNY = eggNY; + int rowSizeX = nColumns*eggSizeX; + int rowSizeY = eggSizeY; + SDF rowObstacles; + populateSDF(eggNX, eggNY, eggSizeX, eggSizeY, eggSdf, nColumns, 1, rowObstacles); + + // 3 Shift rows + const float angle = 1.7f * M_PI / 180.0f; + const int nRowsPerShift = static_cast(ceil(eggSizeX / (eggSizeY * tan(angle)))); + if (fabs(eggSizeX / (eggSizeY * tan(angle)) - nRowsPerShift) > 1e-1) { + std::cout << "ERROR: Suggest changing the angle\n"; + return 1; + } + + float padding = float(ceil(nRows * eggSizeY * tan(angle))); + // TODO Do I need this nUniqueRows? + int nUniqueRows = nRows; + if (nRows > nRowsPerShift) { + nUniqueRows = nRowsPerShift; + padding = float(round(nRowsPerShift * eggSizeY * tan(angle))); + } + + // TODO fix this stupid workaround + if (padding < 32.0f) + padding = 0.0f; + if (padding == 57.0f) + padding = 56.0f; + padding = padding + 8; // adjust padding to have desired size + + std::cout << "Launching rows generation. Padding = "<< padding < shiftedRows(nUniqueRows); + int shiftedRowNX = 0; // they are all the same length + float shiftedRowSizeX = 0.0f; + //for (int i = nUniqueRows-1; i >= 0; --i) { + for (int i = 0; i < nUniqueRows; ++i) { + float xshift = (nUniqueRows - i -1 ) * 32.0f * tan(angle); + shiftSDF(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, xshift, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows[i]); + } + + // 4 Collage rows + SDF finalSDF; + collageSDF(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, nRows, true, finalSDF); + + // 5 Apply redistancing for the result + float finalExtent[] = {shiftedRowSizeX, nRows*rowSizeY}; + int finalN[] = {shiftedRowNX, nRows*rowNY}; + const float dx = finalExtent[0] / (finalN[0] - 1); + const float dy = finalExtent[1] / (finalN[1] - 1); + Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); + redistancer.run(1e2, &finalSDF[0]); + + // 6 Repeat this pattern + SDF finalSDF2; + populateSDF(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, 1, nrepeat, finalSDF2); + std::swap(finalSDF, finalSDF2); + + // 6 Write result to the file + writeDAT("2d.dat", finalSDF, finalN[0], nrepeat * finalN[1], 1, finalExtent[0], nrepeat*finalExtent[1], 1.0f); + + conver2Dto3D(finalN[0], nrepeat * finalN[1], finalExtent[0], nrepeat*finalExtent[1], finalSDF, + eggNZ, eggSizeZ - 2.0f*zmargin, zmargin, "3d.dat"); + + return 0; +} + From af7be21dd9971322e5c7fe61b286242e618ea984 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:18:05 +0200 Subject: [PATCH 05/63] Cleaned the code --- device-gen/ctc-ichip/main.cpp | 228 +++++++++++++++++++++++----------- 1 file changed, 158 insertions(+), 70 deletions(-) diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp index 74a929c40..72d7e9ead 100644 --- a/device-gen/ctc-ichip/main.cpp +++ b/device-gen/ctc-ichip/main.cpp @@ -1,8 +1,8 @@ /* * main.cpp - * Part of CTC/device-gen/sdf-unit-par/ + * Part of CTC/device-gen/ctc-ichip/ * - * Created and authored by Kirill Lykov on 2015-03-28. + * Created and authored by Kirill Lykov on 2015-09-7. * Copyright 2015. All rights reserved. * * Users are NOT authorized @@ -16,6 +16,7 @@ #include #include #include +#include #include "../common/common.h" #include "../common/collage.h" #include "../common/redistance.h" @@ -55,7 +56,127 @@ struct Egg } }; -void generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) +typedef vector SDF; + +class CTCiChip1Builder +{ + int m_ncolumns, m_nrows, m_nrepeat; + float m_resolution, m_zmargin; + float m_eggSizeX, m_eggSizeY, m_eggSizeZ; // size of the egg with the empty space aroung it + const float m_angle; + std::string m_outFileName2D, m_outFileName3D; +public: + CTCiChip1Builder() + : m_ncolumns(0), m_nrows(0), m_nrepeat(0), m_resolution(0), m_zmargin(0), + m_eggSizeX(56.0f), m_eggSizeY(32.0f), m_eggSizeZ(58.0f), + m_angle(1.7f * M_PI / 180.0f) + {} + + CTCiChip1Builder& setNColumns(int ncolumns) + { + m_ncolumns = ncolumns; + return *this; + } + + CTCiChip1Builder& setNRows(int nrows) + { + m_nrows = nrows; + return *this; + } + + CTCiChip1Builder& setRepeat(float nrepeat) + { + m_nrepeat = nrepeat; + return *this; + } + + CTCiChip1Builder& setResolution(float resolution) + { + m_resolution = resolution; + return *this; + } + + CTCiChip1Builder& setZWallWidth(float zmargin) + { + m_zmargin = zmargin; + return *this; + } + + CTCiChip1Builder& setFileNameFor2D(const std::string& outFileName2D) + { + m_outFileName2D = outFileName2D; + return *this; + } + + CTCiChip1Builder& setFileNameFor3D(const std::string& outFileName3D) + { + m_outFileName3D = outFileName3D; + return *this; + } + + void build() const; + +private: + void generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) const; + + void shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, + float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const; +}; + +void CTCiChip1Builder::build() const +{ + if (m_ncolumns * m_nrows * m_nrepeat * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) + throw std::runtime_error("Invalid parameters"); + // 1 Create 1 obstacle + const int eggNX = static_cast(m_eggSizeX * m_resolution); + const int eggNY = static_cast(m_eggSizeY * m_resolution); + const int eggNZ = static_cast(m_eggSizeZ * m_resolution); + + SDF eggSdf; + generateEggSDF(eggNX, eggNY, m_eggSizeX, m_eggSizeY, eggSdf); + + // 2 Create 1 row of obstacles + int rowNX = m_ncolumns*eggNX; + int rowNY = eggNY; + int rowSizeX = m_ncolumns * m_eggSizeX; + int rowSizeY = m_eggSizeY; + SDF rowObstacles; + populateSDF(eggNX, eggNY, m_eggSizeX, m_eggSizeY, eggSdf, m_ncolumns, 1, rowObstacles); + + // 3 Shift rows + float padding = 0.0f; + int shiftedRowNX = 0; // they are all the same length + float shiftedRowSizeX = 0.0f; + + std::vector shiftedRows; + shiftRows(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows); + + // 4 Collage rows + SDF finalSDF; + collageSDF(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, m_nrows, true, finalSDF); + + // 5 Apply redistancing for the result + float finalExtent[] = {shiftedRowSizeX, static_cast(m_nrows * rowSizeY)}; + int finalN[] = {shiftedRowNX, m_nrows*rowNY}; + const float dx = finalExtent[0] / (finalN[0] - 1); + const float dy = finalExtent[1] / (finalN[1] - 1); + Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); + redistancer.run(1e2, &finalSDF[0]); + + // 6 Repeat this pattern + SDF finalSDF2; + populateSDF(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, 1, m_nrepeat, finalSDF2); + std::swap(finalSDF, finalSDF2); + + // 6 Write result to the file + if (m_outFileName2D.length() != 0) + writeDAT(m_outFileName2D, finalSDF, finalN[0], m_nrepeat * finalN[1], 1, finalExtent[0], m_nrepeat * finalExtent[1], 1.0f); + + conver2Dto3D(finalN[0], m_nrepeat * finalN[1], finalExtent[0], m_nrepeat*finalExtent[1], finalSDF, + eggNZ, m_eggSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); +} + +void CTCiChip1Builder::generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) const { vector xs, ys; Egg egg; @@ -105,52 +226,20 @@ void generateEggSDF(const int NX, const int NY, const float xextent, const float } } -typedef vector SDF; - -int main(int argc, char ** argv) +void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, + float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const { - - int nColumns = 5; - int nRows = 57; - int nrepeat = 2; - - int zmargin = 5.0f; - - // 1 Create 2D SDF for 1 obstacle - const float eggSizeX = 56.0f; // size of the egg with the empty space aroung it - const float eggSizeY = 32.0f; - const float eggSizeZ = 58.0f; - const float resolution = 1.0f; // how many grid point per one micron - const int eggNX = static_cast(eggSizeX * resolution); - const int eggNY = static_cast(eggSizeY * resolution); - const int eggNZ = static_cast(eggSizeZ * resolution); - - SDF eggSdf; - generateEggSDF(eggNX, eggNY, eggSizeX, eggSizeY, eggSdf); - writeDAT("out.dat", eggSdf, eggNX, eggNY, 1, eggSizeX, eggSizeY, 1.0f); - - // 2 Create 1 row of obstacles - int rowNX = nColumns*eggNX; - int rowNY = eggNY; - int rowSizeX = nColumns*eggSizeX; - int rowSizeY = eggSizeY; - SDF rowObstacles; - populateSDF(eggNX, eggNY, eggSizeX, eggSizeY, eggSdf, nColumns, 1, rowObstacles); - - // 3 Shift rows - const float angle = 1.7f * M_PI / 180.0f; - const int nRowsPerShift = static_cast(ceil(eggSizeX / (eggSizeY * tan(angle)))); - if (fabs(eggSizeX / (eggSizeY * tan(angle)) - nRowsPerShift) > 1e-1) { - std::cout << "ERROR: Suggest changing the angle\n"; - return 1; + const int nRowsPerShift = static_cast(ceil(m_eggSizeX / (m_eggSizeY * tan(m_angle)))); + if (fabs(m_eggSizeX / (m_eggSizeY * tan(m_angle)) - nRowsPerShift) > 1e-1) { + throw std::runtime_error("Suggest changing the angle"); } - float padding = float(ceil(nRows * eggSizeY * tan(angle))); + padding = float(ceil(m_nrows * m_eggSizeY * tan(m_angle))); // TODO Do I need this nUniqueRows? - int nUniqueRows = nRows; - if (nRows > nRowsPerShift) { + int nUniqueRows = m_nrows; + if (m_nrows > nRowsPerShift) { nUniqueRows = nRowsPerShift; - padding = float(round(nRowsPerShift * eggSizeY * tan(angle))); + padding = float(round(nRowsPerShift * m_eggSizeY * tan(m_angle))); } // TODO fix this stupid workaround @@ -159,40 +248,39 @@ int main(int argc, char ** argv) if (padding == 57.0f) padding = 56.0f; padding = padding + 8; // adjust padding to have desired size - + std::cout << "Launching rows generation. Padding = "<< padding < shiftedRows(nUniqueRows); - int shiftedRowNX = 0; // they are all the same length - float shiftedRowSizeX = 0.0f; - //for (int i = nUniqueRows-1; i >= 0; --i) { + shiftedRows.resize(nUniqueRows); for (int i = 0; i < nUniqueRows; ++i) { - float xshift = (nUniqueRows - i -1 ) * 32.0f * tan(angle); - shiftSDF(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, xshift, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows[i]); + float xshift = (nUniqueRows - i -1 ) * 32.0f * tan(m_angle); + shiftSDF(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, xshift, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows[i]); } +} - // 4 Collage rows - SDF finalSDF; - collageSDF(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, nRows, true, finalSDF); - // 5 Apply redistancing for the result - float finalExtent[] = {shiftedRowSizeX, nRows*rowSizeY}; - int finalN[] = {shiftedRowNX, nRows*rowNY}; - const float dx = finalExtent[0] / (finalN[0] - 1); - const float dy = finalExtent[1] / (finalN[1] - 1); - Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); - redistancer.run(1e2, &finalSDF[0]); +int main(int argc, char ** argv) +{ + + int nColumns = 5; + int nRows = 57; + int nRepeat = 2; - // 6 Repeat this pattern - SDF finalSDF2; - populateSDF(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, 1, nrepeat, finalSDF2); - std::swap(finalSDF, finalSDF2); - - // 6 Write result to the file - writeDAT("2d.dat", finalSDF, finalN[0], nrepeat * finalN[1], 1, finalExtent[0], nrepeat*finalExtent[1], 1.0f); + int zMargin = 5.0f; - conver2Dto3D(finalN[0], nrepeat * finalN[1], finalExtent[0], nrepeat*finalExtent[1], finalSDF, - eggNZ, eggSizeZ - 2.0f*zmargin, zmargin, "3d.dat"); + std::string outFileName = "3d"; + CTCiChip1Builder builder; + try { + builder.setNColumns(nColumns) + .setNRows(nRows) + .setRepeat(nRepeat) + .setResolution(1.0f) + .setZWallWidth(zMargin) + .setFileNameFor3D(outFileName) + .build(); + } catch(const std::exception& ex) { + std::cout << "ERROR: " << ex.what() << std::endl; + } return 0; } From 8b965ecb32ee6c0f6744aca35b38f3216efa49d2 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:32:04 +0200 Subject: [PATCH 06/63] Added arguments parser --- device-gen/ctc-ichip/main.cpp | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp index 72d7e9ead..c0368cff0 100644 --- a/device-gen/ctc-ichip/main.cpp +++ b/device-gen/ctc-ichip/main.cpp @@ -17,6 +17,7 @@ #include #include #include +#include #include "../common/common.h" #include "../common/collage.h" #include "../common/redistance.h" @@ -260,14 +261,23 @@ void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float row int main(int argc, char ** argv) { - - int nColumns = 5; - int nRows = 57; - int nRepeat = 2; - int zMargin = 5.0f; + ArgumentParser argp(vector(argv, argv + argc)); + + int nColumns = argp("-nColumns").asInt(1); + int nRows = argp("-nRows").asInt(1); + int nRepeat = argp("-nRepeat").asInt(1); + float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); + + std::string outFileName = argp("-out").asString("3d"); + + //int nColumns = 5; + //int nRows = 57; + //int nRepeat = 2; + + //int zMargin = 5.0f; - std::string outFileName = "3d"; + //std::string outFileName = "3d"; CTCiChip1Builder builder; try { From c333e417a10c8ec8b270c5545df8b4de36296c82 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:34:57 +0200 Subject: [PATCH 07/63] Removed folder scripts --- device-gen/scripts/README | 25 --------- device-gen/scripts/cleanall.sh | 7 --- device-gen/scripts/makeall.sh | 7 --- device-gen/scripts/run-cylinder.py | 20 ------- device-gen/scripts/run-egg.py | 88 ------------------------------ device-gen/scripts/run-parab.py | 55 ------------------- 6 files changed, 202 deletions(-) delete mode 100644 device-gen/scripts/README delete mode 100755 device-gen/scripts/cleanall.sh delete mode 100755 device-gen/scripts/makeall.sh delete mode 100755 device-gen/scripts/run-cylinder.py delete mode 100755 device-gen/scripts/run-egg.py delete mode 100755 device-gen/scripts/run-parab.py diff --git a/device-gen/scripts/README b/device-gen/scripts/README deleted file mode 100644 index 6e58fb6f3..000000000 --- a/device-gen/scripts/README +++ /dev/null @@ -1,25 +0,0 @@ -# Generate microfluidic geometry - -Set of scripts to generate device geometries as Signed Distance Function in *.dat format -To clean the dependencies and then build them: -``` -./cleanall.sh -./makeall.sh -``` - -## Parabolic funnels -Geometry mimicing the microfluidic device by McFaul et al [Cell separation based on size and deformability using microfluidic funnel ratchets](http://www.ncbi.nlm.nih.gov/pubmed/22517056) -To generate a this geometry with 10 rows and 20 columns run: -``` -./run-parab.py -r 10 -c 20 -``` - -## Later displacement device -Geometry reproducing CTC-iChip1 module by Karabacak et al [Microfluidic, marker-free isolation of circulating tumor cells from blood samples](http://www.nature.com/nprot/journal/v9/n3/full/nprot.2014.044.html) - -To build geometry with 13 columns and 59 rows: -``` -./run-egg.py -r 59 -c 13 -``` -Note, the python script requre modifications if you want to change resolution, wall width and other parameters - diff --git a/device-gen/scripts/cleanall.sh b/device-gen/scripts/cleanall.sh deleted file mode 100755 index b9e3a1b26..000000000 --- a/device-gen/scripts/cleanall.sh +++ /dev/null @@ -1,7 +0,0 @@ -#! /bin/bash -cd ../ -for d in */ ; do - pushd $d - make clean - popd -done diff --git a/device-gen/scripts/makeall.sh b/device-gen/scripts/makeall.sh deleted file mode 100755 index 8c8723c2a..000000000 --- a/device-gen/scripts/makeall.sh +++ /dev/null @@ -1,7 +0,0 @@ -#! /bin/bash -cd ../ -for d in */ ; do - pushd $d - make - popd -done diff --git a/device-gen/scripts/run-cylinder.py b/device-gen/scripts/run-cylinder.py deleted file mode 100755 index 26b9fde90..000000000 --- a/device-gen/scripts/run-cylinder.py +++ /dev/null @@ -1,20 +0,0 @@ -#!/usr/bin/env python -''' - * Part of CTC/device-gen/scripts/run-cylinder.py - * - * Created and authored by Kirill Lykov on 2015-08-28. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. -''' - -import os - -N = 32 -radius = 6.0 -length = 16.0 -os.system("../sdf-cylinder//sdf-unit %d %f unit.dat"%(N, radius)) -outFile = "cylinder%d.dat"%(radius) -os.system("../2Dto3D/2Dto3D unit.dat %f %f %d %s"%(length, 0.0, 32, outFile)) diff --git a/device-gen/scripts/run-egg.py b/device-gen/scripts/run-egg.py deleted file mode 100755 index 640747fb6..000000000 --- a/device-gen/scripts/run-egg.py +++ /dev/null @@ -1,88 +0,0 @@ -#!/usr/bin/env python -''' - * Part of CTC/device-gen/scripts/run-egg.py - * - * Created and authored by Kirill Lykov on 2015-08-28. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. -''' - -import os -import math -import argparse - -parser = argparse.ArgumentParser(description='Generates eggs: ', - usage= './run-egg.py -r -c -d <0|1>') -parser.add_argument('-r','--nRows', help='', required=True) -parser.add_argument('-c','--nColumns', help='', required=True) -#parser.add_argument('-d','--draw', help='values: 0 | 1', required=False, default=1) -args = vars(parser.parse_args()) - -marginZ = 5.0 -marginY = 5.0 # responsible for side walls -ntimesInXdir = 2 # how many times to repeat -eggSize = [56, 32, 48 + 2*marginZ] - -resolution = 1.4 #0.7 -unitXRes = resolution*eggSize[0] -unitYRes = resolution*eggSize[1] -unitZRes = resolution*eggSize[2] -print("Unit grid size: %g %g %g\n"%(unitXRes, unitYRes, unitZRes)) - -nRows = int(args['nRows']) -nColumns = int(args['nColumns']) -#draw = int(args['draw']) == 1 - -angle = 1.7 * math.pi/180.0 - -os.system("../sdf-unit-egg/sdf-unit %d %d %g %g %s"%(unitXRes, unitYRes, eggSize[0], eggSize[1], "egg.dat")) - -nRowsPerShift = int(math.ceil(eggSize[0] / (eggSize[1] * math.tan(angle)))) -if (math.fabs(eggSize[0] / (eggSize[1] * math.tan(angle)) - nRowsPerShift) > 1e-1): - print("ERROR: Suggest changing the angle") - exit() - -padding = float(math.ceil(nRows * eggSize[1] * math.tan(angle))) - -nUniqueRows = nRows -if (nRows > nRowsPerShift): - nUniqueRows = nRowsPerShift - padding = float(round(nRowsPerShift * eggSize[1] * math.tan(angle), 0)) - -print("nRowsPershift = %d, nUniqueRows = %d, Padding = %f"%(nRowsPerShift, nUniqueRows, padding)) - -# workaround -if (padding < 32): - padding = 0 -if (padding == 57): - padding = 56 -padding = padding + 8 # 8 change by hands if needed - -print("Launching rows generation. Padding = %g"%(padding)) -for i in range(nUniqueRows-1, -1, -1): - os.system("../sdf-collage/sdf-collage %s %d %d %f %s"%("egg.dat", nColumns, 1, 0.0, "raw-row.dat")) - xshift = i * 32.0 * math.tan(angle) - print("Calling: ../sdf-shift/sdf-shift %s %g %g %s"%("raw-row.dat", xshift, padding, "row%d.dat"%(i))) - os.system("../sdf-shift/sdf-shift %s %g %g %s"%("raw-row.dat", xshift, padding, "row%d.dat"%(i))) - -with open("files.txt", 'w') as f: - for i in range(nRows-1, -1, -1): - j = i % nRowsPerShift - f.write("row%d.dat\n"%(j)) - -os.system("../sdf-collage/sdf-collage files.txt 1 1 %f %s "%(marginY, "collage.dat")) - -if (ntimesInXdir > 1): - os.system("mv collage.dat collage_temp.dat") - os.system("../sdf-collage/sdf-collage collage_temp.dat %d %d 0.0 collage.dat"%(1, ntimesInXdir)) - -#if (draw): -# os.system("../dat2hdf5/dat2hdf5 collage.dat 2d") -os.system("../2Dto3D/2Dto3D collage.dat %g %g %d %s"%(eggSize[2] - 2*marginZ, marginZ, unitZRes, "%dx%d.dat"%(nColumns, nRows))) -#if (draw): -# os.system("../dat2hdf5/dat2hdf5 %dx%d.dat %dx%d"%(nColumns, nRows, nColumns, nRows)) - -os.system("rm row*.dat") diff --git a/device-gen/scripts/run-parab.py b/device-gen/scripts/run-parab.py deleted file mode 100755 index 329872b89..000000000 --- a/device-gen/scripts/run-parab.py +++ /dev/null @@ -1,55 +0,0 @@ -#!/usr/bin/env python -''' - * Part of CTC/device-gen/scripts/run-parab.py - * - * Created and authored by Kirill Lykov on 2015-08-28. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. -''' - -import os -import math -import argparse - -parser = argparse.ArgumentParser(description='Generates parabolic funnel obstacles: ', - usage= './run-parab.py -r ') -parser.add_argument('-r','--nRows', help='', required=True) -parser.add_argument('-c','--nColumns', help='', required=True) -parser.add_argument('-d','--draw', help='values: 0 | 1', required=False, default=1) -args = vars(parser.parse_args()) - -nRows = int(args['nRows']) -nColumns = int(args['nColumns']) -draw = int(args['draw']) == 1 - -with open("files.txt", 'w') as f: - for i in range(3, nRows + 3): - f.write("r%d.dat\n"%(i)) - -unitXRes=24/2 -unitYRes=96/2 -unitZRes=128/2 - -cleftWidth = 0.5 -zMargin = 4.0 -zSize = 128.0 - -for i in range(3, nRows+3): - os.system("../sdf-unit-par/sdf-unit %d %d %f %f %f gap%d.dat"%(unitXRes, unitYRes, 24.0, 96.0, cleftWidth*i, i)) - -for i in range(3, nRows+3): - os.system("../sdf-collage/sdf-collage gap%d.dat %d %d %f r%d.dat"%(i, nColumns, 1, 0.0, i)) - -os.system("../sdf-collage/sdf-collage files.txt 1 1 0.0 collage.dat") -if draw == 1: - os.system("../dat2hdf5/dat2hdf5 collage.dat 2d") -outFile = "%dx%d.dat"%(nRows, nColumns) -os.system("../2Dto3D/2Dto3D collage.dat %f %f %d %s"%(zSize - 2*zMargin, zMargin, unitZRes, outFile)) -if draw == 1: - os.system("../dat2hdf5/dat2hdf5 %s 3d"%(outFile)) - -os.system("rm row*.dat") -#os.system("tar -czvf %dx%d.tar.gz %s"%(nRows, nColumns, outFile)) From cd223377423c868d29727f17c51cdeb33bc14750 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:35:34 +0200 Subject: [PATCH 08/63] Modified plyScale so it can cut ply with more options --- device-gen/post-process/plyScale.py | 30 +++++++++++++++++++---------- 1 file changed, 20 insertions(+), 10 deletions(-) diff --git a/device-gen/post-process/plyScale.py b/device-gen/post-process/plyScale.py index 06de0e408..82a06ea04 100755 --- a/device-gen/post-process/plyScale.py +++ b/device-gen/post-process/plyScale.py @@ -29,14 +29,17 @@ def computeExtent(vertices): origExtent = [extentMax[i] - extentMin[i] for i in range(0, 3)] return (origOrigin, origExtent) -parser = argparse.ArgumentParser(description='Scales ply file.\n Example: ./plyScale.py -f input.ply -o out.ply -x 150 -y 40 -z 48') +parser = argparse.ArgumentParser(description='Modifies ply file to be used for rendering.\n Example: ./plyScale.py -f input.ply -o out.ply -r 210 -cutX 10.0') parser.add_argument('-f','--inputFile', help='Input file name', required=True) parser.add_argument('-o','--outputFile', help='Output file name', required=True) -parser.add_argument('-x','--lx', help='', required=False, default="0") -parser.add_argument('-y','--ly', help='', required=False, default="0") -parser.add_argument('-z','--lz', help='', required=False, default="0") -parser.add_argument('-r','--order', help='values are 0-2', required=False, default="012") -parser.add_argument('-c','--cut', help='values are 0-2', required=False, default="none") +parser.add_argument('--lx', help='Desired size of bounding box (X axis)', required=False, default="0") +parser.add_argument('--ly', help='Desired size of bounding box (Y axis)', required=False, default="0") +parser.add_argument('--lz', help='Desired size of bounding box (Z axis)', required=False, default="0") +parser.add_argument('-r','--order', help='Reorder axis. By default 012, to swap x and z use 210', required=False, default="012") +helpStringForCut = 'Remove all the faces which are above specified value for %s. The origin is in the center of mass. If axis reordering was applied, axis are in new coordinates.' +parser.add_argument('--cutX', help=helpStringForCut%('X'), required=False, default="none") +parser.add_argument('--cutY', help=helpStringForCut%('Y'), required=False, default="none") +parser.add_argument('--cutZ', help=helpStringForCut%('Z'), required=False, default="none") args = vars(parser.parse_args()) desiredBox = [float(args['lx']), float(args['ly']), float(args['lz'])] @@ -68,14 +71,21 @@ def computeExtent(vertices): vertices[i][dim] -= origOrigin[dim] vertices[i][dim] *= desiredBox[dim]/origExtent[dim] -if (args['cut'] != "none"): +if (args['cutX'] != "none" or args['cutY'] != "none" or args['cutZ'] != "none"): print "Cut it!" + cut = [1e6]*3 + if args['cutX'] != "none": + cut[0] = float(args['cutX']) + if args['cutY'] != "none": + cut[1] = float(args['cutY']) + if args['cutZ'] != "none": + cut[2] = float(args['cutZ']) + toDelete = list() - dim = int(args['cut']) newInx = [None]*len(vertices) j = 0 for i in range(0, len(vertices)): - if (vertices[i][dim] > 0.0): + if (vertices[i][0] > cut[0] or vertices[i][1] > cut[1] or vertices[i][2] > cut[2]): toDelete.append(i) else: newInx[i] = j @@ -104,5 +114,5 @@ def computeExtent(vertices): print ("Extent is (%f, %f, %f). Center is (%f, %f, %f)."%(finalExtent[0], finalExtent[1], finalExtent[2], finalOrigin[0], finalOrigin[1], finalOrigin[2])) - plydata.write(args['outputFile']) + From d680c23b2473344ba17529414a19de2c87bf508a Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:36:00 +0200 Subject: [PATCH 09/63] Cleaned the folder according to the new geometry concept --- device-gen/2Dto3D/Makefile | 7 - device-gen/2Dto3D/main.cpp | 101 ----------- device-gen/common/collage.cpp | 2 +- device-gen/common/collage.h | 2 +- device-gen/ctc-ichip/Makefile | 2 +- device-gen/{sdf-unit-par => funnels}/Makefile | 0 device-gen/{sdf-unit-par => funnels}/main.cpp | 0 device-gen/sdf-collage/Makefile | 7 - device-gen/sdf-collage/main.cpp | 168 ------------------ device-gen/sdf-shift/Makefile | 7 - device-gen/sdf-shift/main.cpp | 58 ------ device-gen/sdf-unit-egg/Makefile | 7 - device-gen/sdf-unit-egg/main.cpp | 125 ------------- 13 files changed, 3 insertions(+), 483 deletions(-) delete mode 100644 device-gen/2Dto3D/Makefile delete mode 100644 device-gen/2Dto3D/main.cpp rename device-gen/{sdf-unit-par => funnels}/Makefile (100%) rename device-gen/{sdf-unit-par => funnels}/main.cpp (100%) delete mode 100644 device-gen/sdf-collage/Makefile delete mode 100644 device-gen/sdf-collage/main.cpp delete mode 100644 device-gen/sdf-shift/Makefile delete mode 100644 device-gen/sdf-shift/main.cpp delete mode 100644 device-gen/sdf-unit-egg/Makefile delete mode 100644 device-gen/sdf-unit-egg/main.cpp diff --git a/device-gen/2Dto3D/Makefile b/device-gen/2Dto3D/Makefile deleted file mode 100644 index 6024c3212..000000000 --- a/device-gen/2Dto3D/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -2Dto3D: main.cpp ../common/common.h ../common/redistance.h - g++-4.9 main.cpp -O3 -fopenmp -o 2Dto3D - -clean: - rm 2Dto3D - -.PHONY = clean diff --git a/device-gen/2Dto3D/main.cpp b/device-gen/2Dto3D/main.cpp deleted file mode 100644 index 1642ebf1e..000000000 --- a/device-gen/2Dto3D/main.cpp +++ /dev/null @@ -1,101 +0,0 @@ -/* - * main.cpp - * Part of CTC/device-gen/2Dto3D/ - * - * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include -#include -#include -#include -#include -#include -#include "../common/common.h" - -using namespace std; - -int main(int argc, char ** argv) -{ - if (argc != 6) - { - printf("usage: ./2to3 \n"); - return 1; - } - - const float zextent = atof(argv[2]); - const float zmargin = atof(argv[3]); - const int NZ = atoi(argv[4]); - - int NX, NY; - float xextent, yextent; - - vector slice; - int oldNZ; - float zextentOld; - readDAT(argv[1], slice, xextent, yextent, zextentOld, NX, NY, oldNZ); - assert(oldNZ == 1); - - printf("Generating data with extent [%f, %f, %f], dimensions [%d, %d, %d], zmargin %f\n", - xextent, yextent, zextent + 2 * zmargin, NX, NY, NZ, zmargin); - - vector outputslice(NX * NY, 0.0f); - - const float z0 = -zextent * 0.5 - zmargin; - const float dz = (zextent + 2 * zmargin) / (NZ - 1); - - FILE * f = fopen(argv[5], "w"); - assert(f != 0); - fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); - fprintf(f, "%d %d %d\n", NY, NX, NZ); - - for(int iz = 0; iz < NZ; ++iz) - { - const float z = z0 + iz * dz; - -#pragma omp parallel for - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const float xysdf = slice[ix + NX * (NY - 1 - iy)]; // NY -1 to change Y-axis direction - float val = xysdf; - if (zmargin != 0.0f) { - const float zsdf = fabs(z) - zextent * 0.5; - if (xysdf < 0) - val = max(zsdf, xysdf); - else - val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); - } - assert(iy + NY * (ix) < outputslice.size()); - - assert(fabs(val) < 1e3); // to check that the value has reasonable range - outputslice[iy + NY * ix] = val; - } - - if (iz == 0) - { - unsigned char * ptr = (unsigned char *)&outputslice[0]; - if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) - { - ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; - printf("INFO: some symbols were changed while writing\n"); - } - } - - int result = fwrite(&outputslice.front(), sizeof(float), NX * NY, f); - - if (result != NX * NY) { - printf("ERROR: written less than expected"); - exit(3); - } - - } - - fclose(f); -} - diff --git a/device-gen/common/collage.cpp b/device-gen/common/collage.cpp index 7ac227cda..9fe49aba7 100644 --- a/device-gen/common/collage.cpp +++ b/device-gen/common/collage.cpp @@ -111,7 +111,7 @@ void collageSDF(const int NX, const int NY, const float xextent, const float yex } } -void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, std::vector& inputGrid, +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& inputGrid, const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid) { float h = xextent / (NX - 1); diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h index daa42720d..3f1075fe5 100644 --- a/device-gen/common/collage.h +++ b/device-gen/common/collage.h @@ -9,5 +9,5 @@ void collageSDF(const int NX, const int NY, const float xextent, const float yex void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, const int xtimes, const int ytimes, std::vector& outputSDF); -void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, std::vector& inputGrid, +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& inputGrid, const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid); diff --git a/device-gen/ctc-ichip/Makefile b/device-gen/ctc-ichip/Makefile index 558ff7e5e..201248d34 100644 --- a/device-gen/ctc-ichip/Makefile +++ b/device-gen/ctc-ichip/Makefile @@ -2,7 +2,7 @@ CXX = g++-4.9 CXXFLAGS += -O0 -g3 -std=c++11 ctc-ichip: *.cpp collage.o redistance.o 2Dto3D.o - $(CXX) $(CXXFLAGS) collage.o redistance.o 2Dto3D.o main.cpp -o ctc-ichip + $(CXX) $(CXXFLAGS) -I../../mpi-dpd/ collage.o redistance.o 2Dto3D.o main.cpp -o ctc-ichip collage.o: ../common/collage.h ../common/collage.cpp $(CXX) $(CXXFLAGS) -c $^ diff --git a/device-gen/sdf-unit-par/Makefile b/device-gen/funnels/Makefile similarity index 100% rename from device-gen/sdf-unit-par/Makefile rename to device-gen/funnels/Makefile diff --git a/device-gen/sdf-unit-par/main.cpp b/device-gen/funnels/main.cpp similarity index 100% rename from device-gen/sdf-unit-par/main.cpp rename to device-gen/funnels/main.cpp diff --git a/device-gen/sdf-collage/Makefile b/device-gen/sdf-collage/Makefile deleted file mode 100644 index 0b7a37ae4..000000000 --- a/device-gen/sdf-collage/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -sdf-collage: main.cpp ../common/common.h ../common/redistance.h - g++-4.9 main.cpp -O3 -g -fopenmp -o sdf-collage - -clean: - rm sdf-collage - -.PHONY = clean diff --git a/device-gen/sdf-collage/main.cpp b/device-gen/sdf-collage/main.cpp deleted file mode 100644 index edf299824..000000000 --- a/device-gen/sdf-collage/main.cpp +++ /dev/null @@ -1,168 +0,0 @@ -/* - * main.cpp - * Part of CTC/device-gen/sdf-collage/ - * - * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include "../common/redistance.h" -#include "../common/common.h" -using namespace std; - -void mergeSDF(int NX, int NY, const vector< vector >& cookie, vector& cake) -{ - cake.resize(cookie.size() * cookie[0].size()); - printf("SIZE: %d\n", cake.size()); - const int stride = NX; - for(int iy = 0; iy < cookie.size() * NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const int dst = ix + stride * iy; - const int iobst = iy / NY; - assert(ix + NX * (iy - iobst*NY) < cookie[iobst].size()); - cake[dst] = cookie[iobst][ix + NX * (iy - iobst*NY)]; - } -} - -int main(int argc, char ** argv) -{ - if (argc != 6) - { - printf("usage: ./sdf-collage \n"); - return -1; - } - - const int xtimes = atoi(argv[2]); - int ytimes = atoi(argv[3]); - - float xextent, yextent, zextent; - int NX, NY,NZ; - vector< vector > cookie; - vector cake; - if (string(argv[1]) != "files.txt") - { - cookie.resize(1); - // for one file - readDAT(argv[1], cookie[0], xextent, yextent, zextent, NX, NY, NZ); - printf("Populate %d * %d times\n", xtimes, ytimes); - const int stride = xtimes * NX; - cake.resize(xtimes * ytimes * NX * NY); - - for(int ty = 0; ty < ytimes; ++ty) - for(int tx = 0; tx < xtimes; ++tx) { - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const int gx = ix + NX * tx; - const int gy = iy + NY * ty; - const int dst = gx + stride * gy; - - assert(dst < cake.size()); - assert(ix + NX * iy < cookie[0].size()); - cake[dst] = cookie[0][ix + NX * iy]; - } - } - } else { - vector files; - - FILE* fs = fopen(argv[1], "r"); - assert(fs != 0); - string buf(127, ' '); - while(fscanf(fs, "%s\n", &buf[0]) == 1) { - files.push_back(buf); - } - fclose(fs); - - ytimes = files.size(); - cookie.resize(ytimes); - assert(xtimes == 1); - - for (int i = files.size() - 1; i >= 0; --i) - { - printf("Reading file %s ...\n", files[i].c_str()); - readDAT(files[i].c_str(), cookie[i], xextent, yextent, zextent, NX, NY,NZ); - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - if (cookie[i][ix + NX * iy] > 1e3) - { - std::cout << "ERROR in file " << files[i].c_str() << std::endl; - exit(0); - } - } - } - - mergeSDF(NX, NY, cookie, cake); - - // add walls in Y directio - float wallWidth = atof(argv[4]); - if (wallWidth != 0.0f) - { - int cakeNX = xtimes * NX; - int cakeNY = ytimes * NY; - const float x0 = -xtimes * xextent * 0.5; - const float dx = xtimes * xextent / (cakeNX - 1); - - const float y0 = -ytimes * yextent * 0.5; - const float dy = ytimes * yextent / (cakeNY - 1); - - const float angle = (1.8/180.)*M_PI; - const float normal[] = {-cos(angle), sin(angle)}; - wallWidth = -2*y0*tan(angle); - - float ypick = 25.0f; //15 - float widthOfBufferZone = 8-wallWidth + 0.0*(48 - 2*wallWidth); - float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); - const float angle2 = atan(xpick/ypick); - std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; - const float normal2[] = {-cos(angle2), -sin(angle2)}; - - - const float linePoint[] = {-x0 - wallWidth, y0}; - const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; - - for(int iy = 0; iy < cakeNY; ++iy) - for(int ix = 0; ix < cakeNX; ++ix) - { - const float signX = sign(dx*ix + x0); - float p[] = {dx*ix + x0, dy*iy + y0}; - //float xsdf = std::numeric_limits::min(); - float padding = signbit(-p[0])*widthOfBufferZone; - float xsdf = -1e6; - - if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { - xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); - } else { - xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); - } - - cake[ix + cakeNX*iy] = std::max(cake[ix + cakeNX*iy], xsdf); - //} - //const float xsdf = fabs(dx*ix + x0) - (xextent * 0.5 - wallWidth); - //cake[ix + cakeNX*iy] = std::max(cake[ix + cakeNX*iy], xsdf); - } - } - } - - const float dx = xextent / (NX - 1); - const float dy = yextent / (NY - 1); - Redistancing::redistancing(1000, 0.25 * min(dx, dy), dx, dy, xtimes * NX, ytimes * NY, &cake[0]); - - writeDAT(argv[5], cake, xtimes * xextent, ytimes * yextent, 1.0f, xtimes * NX, ytimes * NY, 1); - return 0; -} - diff --git a/device-gen/sdf-shift/Makefile b/device-gen/sdf-shift/Makefile deleted file mode 100644 index 6bf9c3cae..000000000 --- a/device-gen/sdf-shift/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -sdf-shift: main.cpp ../common/common.h ../common/redistance.h - g++-4.9 main.cpp -O0 -g3 -std=c++0x -o sdf-shift - -clean: - rm sdf-shift - -.PHONY = clean diff --git a/device-gen/sdf-shift/main.cpp b/device-gen/sdf-shift/main.cpp deleted file mode 100644 index 32dd5fd84..000000000 --- a/device-gen/sdf-shift/main.cpp +++ /dev/null @@ -1,58 +0,0 @@ -/* - * main.cpp - * Part of CTC/device-gen/sdf-unit-par/ - * - * Created and authored by Kirill Lykov on 2015-03-28. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include "../common/common.h" -#include "../common/redistance.h" -#include -#include - -int main(int argc, char** argv) -{ - if (argc != 5) - { - printf("usage: ./sdf-shift \n"); - return -1; - } - - float xshift = atof(argv[2]); - float xpadding = atof(argv[3]); - - float xextent, yextent, zextent; - int NX, NY,NZ; - std::vector inputGrid; - readDAT(argv[1], inputGrid, xextent, yextent, zextent, NX, NY, NZ); - - float h = xextent / (NX - 1); - int ixshift = xshift / h; - int ipadding = xpadding / h + 1; - - float minVal = *std::min(inputGrid.begin(), inputGrid.end()); - std::vector outGrid((NX + ipadding) * NY * NZ, -1e6); - assert(NZ == 1); - - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX ; ++ix) - { - int newIx = (ix + ixshift) % (NX + ipadding); - assert(fabs(inputGrid[ix + NX * iy]) < 1e3); - outGrid[newIx + (NX + ipadding) * iy] = inputGrid[ix + NX * iy]; - } - - const float dx = (xextent + xpadding)/ (NX + ipadding - 1); - const float dy = yextent / (NY - 1); - Redistancing::redistancing(1000, 0.25 * min(dx, dy), dx, dy, NX + ipadding, NY, &outGrid[0]); - - writeDAT(argv[4], outGrid, xextent + xpadding, yextent, zextent, NX + ipadding, NY, NZ); - - - return 0; -} diff --git a/device-gen/sdf-unit-egg/Makefile b/device-gen/sdf-unit-egg/Makefile deleted file mode 100644 index ea8da56e9..000000000 --- a/device-gen/sdf-unit-egg/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -sdf-unit: main.cpp - g++-4.9 main.cpp -O0 -g3 -std=c++0x -o sdf-unit - -clean: - rm sdf-unit - -.PHONY = clean diff --git a/device-gen/sdf-unit-egg/main.cpp b/device-gen/sdf-unit-egg/main.cpp deleted file mode 100644 index a1eccf68d..000000000 --- a/device-gen/sdf-unit-egg/main.cpp +++ /dev/null @@ -1,125 +0,0 @@ -/* - * main.cpp - * Part of CTC/device-gen/sdf-unit-par/ - * - * Created and authored by Kirill Lykov on 2015-03-28. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include -#include -#include -#include -#include -#include -#include "../common/common.h" -using namespace std; - -struct Egg -{ - float r1, r2, alpha; - - Egg() - : r1(12.0f), r2(8.5f), alpha(0.03f) - { - } - - float x2y(float x) const { - return sqrt(r2*r2 * exp(-alpha * x) * (1.0f - x*x/r1/r1)); - } - - void run(vector& vx, vector& vy) { - int N = 500; - float dx = 2.0f * r1 / (N - 1); - for (int i = 0; i < N; ++i) { - float x = i * dx - r1; - float y = x2y(x); - vx.push_back(x); - vy.push_back(y); - } - - auto vxRev = vx; - vx.insert(vx.end(), vxRev.rbegin(), vxRev.rend()); - - auto vyRev = vy; - for_each(vyRev.begin(), vyRev.end(), [](float& i) { i *= -1.0f; }); - vy.insert(vy.end(), vyRev.rbegin(), vyRev.rend()); - } -}; - -int main(int argc, char ** argv) -{ - if (argc != 6) - { - printf("usage: ./sdf-unit-egg \n"); - return 1; - } - - const int NX = atoi(argv[1]); - const int NY = atoi(argv[2]); - const float xextent = atof(argv[3]); - const float yextent = atof(argv[4]); - - vector xs, ys; - Egg egg; - egg.run(xs, ys); - - const float xlb = -xextent/2.0f; - const float ylb = -yextent/2.0f; - printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", - NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); - - vector sdf(NX * NY, 0.0f); - const float dx = xextent / NX; - const float dy = yextent / NY; - const int nsamples = xs.size(); - - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const float x = xlb + ix * dx; - const float y = ylb + iy * dy; - - float distance2 = 1e6; - int iclosest = 0; - for(int i = 0; i < nsamples ; ++i) - { - const float xd = xs[i] - x; - const float yd = ys[i] - y; - const float candidate = xd * xd + yd * yd; - - if (candidate < distance2) - { - iclosest = i; - distance2 = candidate; - } - } - - float s = -1; - - { - const float ycurve = egg.x2y(x); - if (x >= -egg.r1 && x <= egg.r1 && fabs(y) <= ycurve) - s = +1; - } - - - sdf[ix + NX * iy] = s * sqrt(distance2); - } - - writeDAT(argv[5], sdf, xextent, yextent, 1.0f, NX, NY, 1); - //FILE * f = fopen(argv[5], "w"); - //fprintf(f, "%f %f %f\n", xextent, yextent, 1.0f); - //fprintf(f, "%d %d %d\n", NX, NY, 1); - //fwrite(sdf, sizeof(float), NX * NY, f); - //fclose(f); - - //delete [] sdf; - - return 0; -} - From 0644362612fb5de1895eb1efc1fca8918170e067 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Tue, 8 Sep 2015 20:47:22 +0200 Subject: [PATCH 10/63] Added some missing license headers and minor code changes --- device-gen/common/2Dto3D.cpp | 4 ++-- device-gen/common/2Dto3D.h | 11 +++++++++++ device-gen/common/collage.cpp | 2 +- device-gen/common/collage.h | 11 +++++++++++ device-gen/common/redistance.cpp | 11 +++++++++++ device-gen/ctc-ichip/main.cpp | 15 ++++----------- 6 files changed, 40 insertions(+), 14 deletions(-) diff --git a/device-gen/common/2Dto3D.cpp b/device-gen/common/2Dto3D.cpp index 9bd735c37..c1bab4dcc 100644 --- a/device-gen/common/2Dto3D.cpp +++ b/device-gen/common/2Dto3D.cpp @@ -1,6 +1,6 @@ /* - * main.cpp - * Part of CTC/device-gen/2Dto3D/ + * 2Dto3D.cpp + * Part of CTC/device-gen/common/ * * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. * Copyright 2015. All rights reserved. diff --git a/device-gen/common/2Dto3D.h b/device-gen/common/2Dto3D.h index 74d617476..482f7a36b 100644 --- a/device-gen/common/2Dto3D.h +++ b/device-gen/common/2Dto3D.h @@ -1,3 +1,14 @@ +/* + * 2Dto3D.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ #pragma once #include #include diff --git a/device-gen/common/collage.cpp b/device-gen/common/collage.cpp index 9fe49aba7..b3c3ca511 100644 --- a/device-gen/common/collage.cpp +++ b/device-gen/common/collage.cpp @@ -1,5 +1,5 @@ /* - * main.cpp + * collage.cpp * Part of CTC/device-gen/sdf-collage/ * * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h index 3f1075fe5..8f003ce0d 100644 --- a/device-gen/common/collage.h +++ b/device-gen/common/collage.h @@ -1,3 +1,14 @@ +/* + * collage.cpp + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ #pragma once #include diff --git a/device-gen/common/redistance.cpp b/device-gen/common/redistance.cpp index 97f1feda9..64b7c1aed 100644 --- a/device-gen/common/redistance.cpp +++ b/device-gen/common/redistance.cpp @@ -1,3 +1,14 @@ +/* + * resistance.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ #include "redistance.h" #include #include diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp index c0368cff0..09eca2c78 100644 --- a/device-gen/ctc-ichip/main.cpp +++ b/device-gen/ctc-ichip/main.cpp @@ -65,12 +65,13 @@ class CTCiChip1Builder float m_resolution, m_zmargin; float m_eggSizeX, m_eggSizeY, m_eggSizeZ; // size of the egg with the empty space aroung it const float m_angle; - std::string m_outFileName2D, m_outFileName3D; + std::string m_outFileName2D, m_outFileName3D; + int m_niterRedistance; public: CTCiChip1Builder() : m_ncolumns(0), m_nrows(0), m_nrepeat(0), m_resolution(0), m_zmargin(0), m_eggSizeX(56.0f), m_eggSizeY(32.0f), m_eggSizeZ(58.0f), - m_angle(1.7f * M_PI / 180.0f) + m_angle(1.7f * M_PI / 180.0f), m_niterRedistance(1e3) {} CTCiChip1Builder& setNColumns(int ncolumns) @@ -162,7 +163,7 @@ void CTCiChip1Builder::build() const const float dx = finalExtent[0] / (finalN[0] - 1); const float dy = finalExtent[1] / (finalN[1] - 1); Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); - redistancer.run(1e2, &finalSDF[0]); + redistancer.run(m_niterRedistance, &finalSDF[0]); // 6 Repeat this pattern SDF finalSDF2; @@ -270,14 +271,6 @@ int main(int argc, char ** argv) float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); std::string outFileName = argp("-out").asString("3d"); - - //int nColumns = 5; - //int nRows = 57; - //int nRepeat = 2; - - //int zMargin = 5.0f; - - //std::string outFileName = "3d"; CTCiChip1Builder builder; try { From a8e3a0812273215b7763a225f3d7c5528932f725 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 9 Sep 2015 16:35:24 +0200 Subject: [PATCH 11/63] Added builder for funnel and enabled openmp in makefile and redistancing --- device-gen/common/redistance.cpp | 2 +- device-gen/ctc-ichip/Makefile | 4 +- device-gen/funnels/Makefile | 24 ++++-- device-gen/funnels/main.cpp | 141 ++++++++++++++++++++++++++++--- 4 files changed, 148 insertions(+), 23 deletions(-) diff --git a/device-gen/common/redistance.cpp b/device-gen/common/redistance.cpp index 64b7c1aed..ef32a05af 100644 --- a/device-gen/common/redistance.cpp +++ b/device-gen/common/redistance.cpp @@ -71,7 +71,7 @@ Redistance::Redistance(const float dt, const float dx, const float dy, if (t % 100 == 0) printf("t: %d, size: %d %d\n", t, m_xsize, m_ysize); -//#pragma omp parallel for +#pragma omp parallel for for(int iy = 0; iy < m_ysize; ++iy) for(int ix = 0; ix < m_xsize; ++ix) { diff --git a/device-gen/ctc-ichip/Makefile b/device-gen/ctc-ichip/Makefile index 201248d34..93dccdc3e 100644 --- a/device-gen/ctc-ichip/Makefile +++ b/device-gen/ctc-ichip/Makefile @@ -1,5 +1,5 @@ CXX = g++-4.9 -CXXFLAGS += -O0 -g3 -std=c++11 +CXXFLAGS += -O0 -g3 -std=c++11 -fopenmp ctc-ichip: *.cpp collage.o redistance.o 2Dto3D.o $(CXX) $(CXXFLAGS) -I../../mpi-dpd/ collage.o redistance.o 2Dto3D.o main.cpp -o ctc-ichip @@ -14,6 +14,6 @@ redistance.o: ../common/redistance.h ../common/redistance.cpp $(CXX) $(CXXFLAGS) -c $^ clean: - rm -f test *.o *.d *.h.gch + rm -f ctc-ichip *.o *.d *.h.gch diff --git a/device-gen/funnels/Makefile b/device-gen/funnels/Makefile index eea0d33ce..5c7799168 100644 --- a/device-gen/funnels/Makefile +++ b/device-gen/funnels/Makefile @@ -1,7 +1,19 @@ -sdf-unit: main.cpp - g++-4.9 main.cpp -O3 -o sdf-unit - +CXX = g++-4.9 +CXXFLAGS += -O0 -g3 -std=c++11 -fopenmp + +funnel: *.cpp collage.o redistance.o 2Dto3D.o + $(CXX) $(CXXFLAGS) -I../../mpi-dpd/ collage.o redistance.o 2Dto3D.o main.cpp -o funnel + +collage.o: ../common/collage.h ../common/collage.cpp + $(CXX) $(CXXFLAGS) -c $^ + +redistance.o: ../common/redistance.h ../common/redistance.cpp + $(CXX) $(CXXFLAGS) -c $^ + +2Dto3D.o: ../common/2Dto3D.h ../common/2Dto3D.cpp + $(CXX) $(CXXFLAGS) -c $^ + clean: - rm sdf-unit - -.PHONY = clean + rm -f test *.o *.d *.h.gch + + diff --git a/device-gen/funnels/main.cpp b/device-gen/funnels/main.cpp index 6e23ef8ec..496d5427f 100644 --- a/device-gen/funnels/main.cpp +++ b/device-gen/funnels/main.cpp @@ -15,7 +15,12 @@ #include #include #include +#include #include "../common/common.h" +#include "../common/collage.h" +#include "../common/redistance.h" +#include "../common/2Dto3D.h" + using namespace std; struct Parabola @@ -57,20 +62,65 @@ struct Parabola } }; -int main(int argc, char ** argv) +class FunnelsBuilder { - if (argc != 7) + typedef vector SDF; + int m_ncolumns, m_nrows; + float m_resolution, m_zmargin; + float m_unitSizeX, m_unitSizeY, m_unitSizeZ; // size of the egg with the empty space aroung it + std::string m_outFileName2D, m_outFileName3D; + int m_niterRedistance; +public: + FunnelsBuilder() + : m_ncolumns(0), m_nrows(0), m_resolution(0), m_zmargin(0), + m_unitSizeX(24.0f), m_unitSizeY(96.0f), m_unitSizeZ(58.0f), + m_niterRedistance(1e3) + {} + + FunnelsBuilder& setNColumns(int ncolumns) { - printf("usage: ./sdf-unit \n"); - return 1; + m_ncolumns = ncolumns; + return *this; } - - const int NX = atoi(argv[1]); - const int NY = atoi(argv[2]); - const float xextent = atof(argv[3]); - const float yextent = atof(argv[4]); - const float gap = atof(argv[5]); - + + FunnelsBuilder& setNRows(int nrows) + { + m_nrows = nrows; + return *this; + } + + FunnelsBuilder& setResolution(float resolution) + { + m_resolution = resolution; + return *this; + } + + FunnelsBuilder& setZWallWidth(float zmargin) + { + m_zmargin = zmargin; + return *this; + } + + FunnelsBuilder& setFileNameFor2D(const std::string& outFileName2D) + { + m_outFileName2D = outFileName2D; + return *this; + } + + FunnelsBuilder& setFileNameFor3D(const std::string& outFileName3D) + { + m_outFileName3D = outFileName3D; + return *this; + } + + void build() const; + +private: + void generateUnitSDF(const int NX, const int NY, const float xextent, const float yextent, float gap, vector& sdf) const; +}; +// TODO reduce number of parameters +void FunnelsBuilder::generateUnitSDF(const int NX, const int NY, const float xextent, const float yextent, float gap, vector& sdf) const +{ vector xs, ys; Parabola par(gap, xextent); par.line1(xs, ys); @@ -81,8 +131,8 @@ int main(int argc, char ** argv) printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); - vector sdf(NX * NY, 0.0f); - const float dx = xextent / NX; + sdf.resize(NX * NY, 0.0f); + const float dx = xextent / NX; //TODO NX-1 const float dy = yextent / NY; const int nsamples = xs.size(); @@ -120,8 +170,71 @@ int main(int argc, char ** argv) sdf[ix + NX * iy] = s * sqrt(distance2); } +} + +void FunnelsBuilder::build() const +{ + if (m_ncolumns * m_nrows * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) + throw std::runtime_error("Invalid parameters"); + + // 1 Create obstacles with different gaps + const float m_gapSpace = 1.0; + + const int unitNX = static_cast(m_unitSizeX * m_resolution); + const int unitNY = static_cast(m_unitSizeY * m_resolution); + const int unitNZ = static_cast(m_unitSizeZ * m_resolution); + + std::vector unitSDF(m_nrows); + for (int i = 0; i < m_nrows; ++i) { + float gap = m_gapSpace * (i + 3); + generateUnitSDF(unitNX, unitNY, m_unitSizeX, m_unitSizeY, gap, unitSDF[i]); + } + + // 2 Create rows of obstacles + std::vector rows(m_nrows); + for (int i = 0; i < m_nrows; ++i) { + populateSDF(unitNX, unitNY, m_unitSizeX, m_unitSizeY, unitSDF[i], m_ncolumns, 1, rows[i]); + } + + // 3 Collage rows + SDF finalSDF; + collageSDF(unitNX * m_ncolumns, unitNY, m_unitSizeX * m_ncolumns, m_unitSizeY, rows, m_nrows, false, finalSDF); + + // 4 Apply redistancing for the result + float finalExtent[] = {m_unitSizeX * m_ncolumns, m_unitSizeY * m_nrows}; + int finalN[] = {unitNX * m_ncolumns, unitNY * m_nrows}; + const float dx = finalExtent[0] / (finalN[0] - 1); + const float dy = finalExtent[1] / (finalN[1] - 1); + Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); + redistancer.run(m_niterRedistance, &finalSDF[0]); - writeDAT(argv[6], sdf, xextent, yextent, 1.0f, NX, NY, 1); + if (m_outFileName2D.length() != 0) + writeDAT(m_outFileName2D.c_str(), finalSDF, finalExtent[0], finalExtent[1], 1.0f, finalN[0], finalN[1], 1); + conver2Dto3D(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); +} + +int main(int argc, char ** argv) +{ + //ArgumentParser argp(vector(argv, argv + argc)); + + int nColumns = 4; //argp("-nColumns").asInt(1); + int nRows = 2; //argp("-nRows").asInt(1); + float zMargin = 5.0f; //static_cast(argp("-zMargin").asDouble(5.0)); + + std::string outFileName = "3d.dat";//argp("-out").asString("3d"); + + FunnelsBuilder builder; + try { + builder.setNColumns(nColumns) + .setNRows(nRows) + .setResolution(1.0f) + .setZWallWidth(zMargin) + .setFileNameFor2D("2d.dat") + .setFileNameFor3D(outFileName) + .build(); + } catch(const std::exception& ex) { + std::cout << "ERROR: " << ex.what() << std::endl; + } return 0; } From 2daaf55e99552d73e8122733bc9eae7063dad5d5 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 9 Sep 2015 17:04:39 +0200 Subject: [PATCH 12/63] Extracted base class --- device-gen/funnels/main.cpp | 86 +++++++++++++++++++++---------------- 1 file changed, 50 insertions(+), 36 deletions(-) diff --git a/device-gen/funnels/main.cpp b/device-gen/funnels/main.cpp index 496d5427f..c54a732b7 100644 --- a/device-gen/funnels/main.cpp +++ b/device-gen/funnels/main.cpp @@ -62,21 +62,36 @@ struct Parabola } }; -class FunnelsBuilder +class DeviceBuilder { +protected: typedef vector SDF; int m_ncolumns, m_nrows; float m_resolution, m_zmargin; float m_unitSizeX, m_unitSizeY, m_unitSizeZ; // size of the egg with the empty space aroung it std::string m_outFileName2D, m_outFileName3D; int m_niterRedistance; + int m_unitNX, m_unitNY, m_unitNZ; public: - FunnelsBuilder() + DeviceBuilder(float unitSizeX, float unitSizeY, float unitSizeZ) : m_ncolumns(0), m_nrows(0), m_resolution(0), m_zmargin(0), - m_unitSizeX(24.0f), m_unitSizeY(96.0f), m_unitSizeZ(58.0f), - m_niterRedistance(1e3) + m_unitSizeX(unitSizeX), m_unitSizeY(unitSizeY), m_unitSizeZ(unitSizeZ), + m_niterRedistance(1e3), m_unitNX(0), m_unitNY(0), m_unitNZ(0) {} + virtual void build() = 0; + + virtual ~DeviceBuilder() {} +}; + +class FunnelsBuilder : public DeviceBuilder +{ + float m_gapSpace; // unit gap between obstacles +public: + FunnelsBuilder() + : DeviceBuilder(24.0f, 96.0f, 58.0f), m_gapSpace(1.0f) + {} + FunnelsBuilder& setNColumns(int ncolumns) { m_ncolumns = ncolumns; @@ -113,31 +128,30 @@ class FunnelsBuilder return *this; } - void build() const; + void build(); private: - void generateUnitSDF(const int NX, const int NY, const float xextent, const float yextent, float gap, vector& sdf) const; + void generateUnitSDF(float gap, vector& sdf) const; }; -// TODO reduce number of parameters -void FunnelsBuilder::generateUnitSDF(const int NX, const int NY, const float xextent, const float yextent, float gap, vector& sdf) const + +void FunnelsBuilder::generateUnitSDF(float gap, vector& sdf) const { + assert(m_unitNX * m_unitNY * m_unitNZ != 0); vector xs, ys; - Parabola par(gap, xextent); + Parabola par(gap, m_unitSizeX); par.line1(xs, ys); par.line2(xs, ys); - const float xlb = -xextent/2.0f; - const float ylb = -(yextent - par.ymax)/2.0f; - printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", - NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); + const float xlb = -m_unitSizeX/2.0f; + const float ylb = -(m_unitSizeY - par.ymax)/2.0f; - sdf.resize(NX * NY, 0.0f); - const float dx = xextent / NX; //TODO NX-1 - const float dy = yextent / NY; + sdf.resize(m_unitNX * m_unitNY, 0.0f); + const float dx = m_unitSizeX / (m_unitNX - 1); //TODO NX-1 + const float dy = m_unitSizeY / (m_unitNY - 1); const int nsamples = xs.size(); - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) + for(int iy = 0; iy < m_unitNY; ++iy) + for(int ix = 0; ix < m_unitNX; ++ix) { const float x = xlb + ix * dx; const float y = ylb + iy * dy; @@ -168,41 +182,39 @@ void FunnelsBuilder::generateUnitSDF(const int NX, const int NY, const float xex } - sdf[ix + NX * iy] = s * sqrt(distance2); + sdf[ix + m_unitNX * iy] = s * sqrt(distance2); } } -void FunnelsBuilder::build() const +void FunnelsBuilder::build() { if (m_ncolumns * m_nrows * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) throw std::runtime_error("Invalid parameters"); // 1 Create obstacles with different gaps - const float m_gapSpace = 1.0; - - const int unitNX = static_cast(m_unitSizeX * m_resolution); - const int unitNY = static_cast(m_unitSizeY * m_resolution); - const int unitNZ = static_cast(m_unitSizeZ * m_resolution); + m_unitNX = static_cast(m_unitSizeX * m_resolution); + m_unitNY = static_cast(m_unitSizeY * m_resolution); + m_unitNZ = static_cast(m_unitSizeZ * m_resolution); std::vector unitSDF(m_nrows); for (int i = 0; i < m_nrows; ++i) { float gap = m_gapSpace * (i + 3); - generateUnitSDF(unitNX, unitNY, m_unitSizeX, m_unitSizeY, gap, unitSDF[i]); + generateUnitSDF(gap, unitSDF[i]); } // 2 Create rows of obstacles std::vector rows(m_nrows); for (int i = 0; i < m_nrows; ++i) { - populateSDF(unitNX, unitNY, m_unitSizeX, m_unitSizeY, unitSDF[i], m_ncolumns, 1, rows[i]); + populateSDF(m_unitNX, m_unitNY, m_unitSizeX, m_unitSizeY, unitSDF[i], m_ncolumns, 1, rows[i]); } // 3 Collage rows SDF finalSDF; - collageSDF(unitNX * m_ncolumns, unitNY, m_unitSizeX * m_ncolumns, m_unitSizeY, rows, m_nrows, false, finalSDF); + collageSDF(m_unitNX * m_ncolumns, m_unitNY, m_unitSizeX * m_ncolumns, m_unitSizeY, rows, m_nrows, false, finalSDF); // 4 Apply redistancing for the result float finalExtent[] = {m_unitSizeX * m_ncolumns, m_unitSizeY * m_nrows}; - int finalN[] = {unitNX * m_ncolumns, unitNY * m_nrows}; + int finalN[] = {m_unitNX * m_ncolumns, m_unitNY * m_nrows}; const float dx = finalExtent[0] / (finalN[0] - 1); const float dy = finalExtent[1] / (finalN[1] - 1); Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); @@ -211,18 +223,19 @@ void FunnelsBuilder::build() const if (m_outFileName2D.length() != 0) writeDAT(m_outFileName2D.c_str(), finalSDF, finalExtent[0], finalExtent[1], 1.0f, finalN[0], finalN[1], 1); - conver2Dto3D(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); + conver2Dto3D(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, m_unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); } int main(int argc, char ** argv) { - //ArgumentParser argp(vector(argv, argv + argc)); + ArgumentParser argp(vector(argv, argv + argc)); - int nColumns = 4; //argp("-nColumns").asInt(1); - int nRows = 2; //argp("-nRows").asInt(1); - float zMargin = 5.0f; //static_cast(argp("-zMargin").asDouble(5.0)); + int nColumns = argp("-nColumns").asInt(1); + int nRows = argp("-nRows").asInt(1); + float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); + float resolution = static_cast(argp("-zResolution").asDouble(1.0)); - std::string outFileName = "3d.dat";//argp("-out").asString("3d"); + std::string outFileName = argp("-out").asString("3d"); FunnelsBuilder builder; try { @@ -230,11 +243,12 @@ int main(int argc, char ** argv) .setNRows(nRows) .setResolution(1.0f) .setZWallWidth(zMargin) - .setFileNameFor2D("2d.dat") .setFileNameFor3D(outFileName) + .setResolution(resolution) .build(); } catch(const std::exception& ex) { std::cout << "ERROR: " << ex.what() << std::endl; + return 1; } return 0; } From 3c9af4dee42ec97965e2d042c1014e828bafe133 Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 9 Sep 2015 17:44:26 +0200 Subject: [PATCH 13/63] Extracted base class from CtciChipDevice. Moved DeviceBuilder base class to the common folder. --- device-gen/common/collage.h | 2 +- device-gen/common/device-builder.h | 36 ++++++++++++++ device-gen/ctc-ichip/main.cpp | 76 +++++++++++++----------------- device-gen/funnels/main.cpp | 23 +-------- 4 files changed, 72 insertions(+), 65 deletions(-) create mode 100644 device-gen/common/device-builder.h diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h index 8f003ce0d..f9f577f87 100644 --- a/device-gen/common/collage.h +++ b/device-gen/common/collage.h @@ -1,5 +1,5 @@ /* - * collage.cpp + * collage.h * Part of CTC/device-gen/common/ * * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. diff --git a/device-gen/common/device-builder.h b/device-gen/common/device-builder.h new file mode 100644 index 000000000..736c9e01e --- /dev/null +++ b/device-gen/common/device-builder.h @@ -0,0 +1,36 @@ + +/* + * device-builder.h + * Part of CTC/device-gen/ctc-ichip/ + * + * Created and authored by Kirill Lykov on 2015-09-7. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#pragma once + +class DeviceBuilder +{ +protected: + typedef vector SDF; + int m_ncolumns, m_nrows; + float m_resolution, m_zmargin; + float m_unitSizeX, m_unitSizeY, m_unitSizeZ; // size of the egg with the empty space aroung it + std::string m_outFileName2D, m_outFileName3D; + int m_niterRedistance; + int m_unitNX, m_unitNY, m_unitNZ; +public: + DeviceBuilder(float unitSizeX, float unitSizeY, float unitSizeZ) + : m_ncolumns(0), m_nrows(0), m_resolution(0), m_zmargin(0), + m_unitSizeX(unitSizeX), m_unitSizeY(unitSizeY), m_unitSizeZ(unitSizeZ), + m_niterRedistance(1e3), m_unitNX(0), m_unitNY(0), m_unitNZ(0) + {} + + virtual void build() = 0; + + virtual ~DeviceBuilder() {} +}; diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp index 09eca2c78..b5fa0f093 100644 --- a/device-gen/ctc-ichip/main.cpp +++ b/device-gen/ctc-ichip/main.cpp @@ -18,6 +18,7 @@ #include #include #include +#include "../common/device-builder.h" #include "../common/common.h" #include "../common/collage.h" #include "../common/redistance.h" @@ -57,21 +58,14 @@ struct Egg } }; -typedef vector SDF; - -class CTCiChip1Builder +class CTCiChip1Builder : public DeviceBuilder { - int m_ncolumns, m_nrows, m_nrepeat; - float m_resolution, m_zmargin; - float m_eggSizeX, m_eggSizeY, m_eggSizeZ; // size of the egg with the empty space aroung it + int m_nrepeat; const float m_angle; - std::string m_outFileName2D, m_outFileName3D; - int m_niterRedistance; public: CTCiChip1Builder() - : m_ncolumns(0), m_nrows(0), m_nrepeat(0), m_resolution(0), m_zmargin(0), - m_eggSizeX(56.0f), m_eggSizeY(32.0f), m_eggSizeZ(58.0f), - m_angle(1.7f * M_PI / 180.0f), m_niterRedistance(1e3) + : DeviceBuilder(56.0f, 32.0f, 58.0f), + m_nrepeat(0), m_angle(1.7f * M_PI / 180.0f) {} CTCiChip1Builder& setNColumns(int ncolumns) @@ -116,34 +110,35 @@ class CTCiChip1Builder return *this; } - void build() const; + void build(); private: - void generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) const; + void generateUnitSDF(vector& sdf) const; void shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const; }; -void CTCiChip1Builder::build() const +void CTCiChip1Builder::build() { if (m_ncolumns * m_nrows * m_nrepeat * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) throw std::runtime_error("Invalid parameters"); + // 1 Create 1 obstacle - const int eggNX = static_cast(m_eggSizeX * m_resolution); - const int eggNY = static_cast(m_eggSizeY * m_resolution); - const int eggNZ = static_cast(m_eggSizeZ * m_resolution); + m_unitNX = static_cast(m_unitSizeX * m_resolution); + m_unitNY = static_cast(m_unitSizeY * m_resolution); + m_unitNZ = static_cast(m_unitSizeZ * m_resolution); SDF eggSdf; - generateEggSDF(eggNX, eggNY, m_eggSizeX, m_eggSizeY, eggSdf); + generateUnitSDF(eggSdf); // 2 Create 1 row of obstacles - int rowNX = m_ncolumns*eggNX; - int rowNY = eggNY; - int rowSizeX = m_ncolumns * m_eggSizeX; - int rowSizeY = m_eggSizeY; + int rowNX = m_ncolumns*m_unitNX; + int rowNY = m_unitNY; + int rowSizeX = m_ncolumns * m_unitSizeX; + int rowSizeY = m_unitSizeY; SDF rowObstacles; - populateSDF(eggNX, eggNY, m_eggSizeX, m_eggSizeY, eggSdf, m_ncolumns, 1, rowObstacles); + populateSDF(m_unitNX, m_unitNY, m_unitSizeX, m_unitSizeY, eggSdf, m_ncolumns, 1, rowObstacles); // 3 Shift rows float padding = 0.0f; @@ -175,27 +170,25 @@ void CTCiChip1Builder::build() const writeDAT(m_outFileName2D, finalSDF, finalN[0], m_nrepeat * finalN[1], 1, finalExtent[0], m_nrepeat * finalExtent[1], 1.0f); conver2Dto3D(finalN[0], m_nrepeat * finalN[1], finalExtent[0], m_nrepeat*finalExtent[1], finalSDF, - eggNZ, m_eggSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); + m_unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); } -void CTCiChip1Builder::generateEggSDF(const int NX, const int NY, const float xextent, const float yextent, vector& sdf) const +void CTCiChip1Builder::generateUnitSDF(vector& sdf) const { vector xs, ys; Egg egg; egg.run(xs, ys); - const float xlb = -xextent/2.0f; - const float ylb = -yextent/2.0f; - printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", - NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); + const float xlb = -m_unitSizeX/2.0f; + const float ylb = -m_unitSizeY/2.0f; - sdf.resize(NX * NY, 0.0f); - const float dx = xextent / NX; - const float dy = yextent / NY; + sdf.resize(m_unitNX * m_unitNY, 0.0f); + const float dx = m_unitSizeX / (m_unitNX - 1); + const float dy = m_unitSizeY / (m_unitNY - 1); const int nsamples = xs.size(); - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) + for(int iy = 0; iy < m_unitNY; ++iy) + for(int ix = 0; ix < m_unitNX; ++ix) { const float x = xlb + ix * dx; const float y = ylb + iy * dy; @@ -224,24 +217,24 @@ void CTCiChip1Builder::generateEggSDF(const int NX, const int NY, const float xe } - sdf[ix + NX * iy] = s * sqrt(distance2); + sdf[ix + m_unitNX * iy] = s * sqrt(distance2); } } void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const { - const int nRowsPerShift = static_cast(ceil(m_eggSizeX / (m_eggSizeY * tan(m_angle)))); - if (fabs(m_eggSizeX / (m_eggSizeY * tan(m_angle)) - nRowsPerShift) > 1e-1) { + const int nRowsPerShift = static_cast(ceil(m_unitSizeX / (m_unitSizeY * tan(m_angle)))); + if (fabs(m_unitSizeX / (m_unitSizeY * tan(m_angle)) - nRowsPerShift) > 1e-1) { throw std::runtime_error("Suggest changing the angle"); } - padding = float(ceil(m_nrows * m_eggSizeY * tan(m_angle))); + padding = float(ceil(m_nrows * m_unitSizeY * tan(m_angle))); // TODO Do I need this nUniqueRows? int nUniqueRows = m_nrows; if (m_nrows > nRowsPerShift) { nUniqueRows = nRowsPerShift; - padding = float(round(nRowsPerShift * m_eggSizeY * tan(m_angle))); + padding = float(round(nRowsPerShift * m_unitSizeY * tan(m_angle))); } // TODO fix this stupid workaround @@ -262,14 +255,13 @@ void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float row int main(int argc, char ** argv) { - ArgumentParser argp(vector(argv, argv + argc)); int nColumns = argp("-nColumns").asInt(1); int nRows = argp("-nRows").asInt(1); int nRepeat = argp("-nRepeat").asInt(1); float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); - + float resolution = static_cast(argp("-zResolution").asDouble(1.0)); std::string outFileName = argp("-out").asString("3d"); CTCiChip1Builder builder; @@ -277,7 +269,7 @@ int main(int argc, char ** argv) builder.setNColumns(nColumns) .setNRows(nRows) .setRepeat(nRepeat) - .setResolution(1.0f) + .setResolution(resolution) .setZWallWidth(zMargin) .setFileNameFor3D(outFileName) .build(); diff --git a/device-gen/funnels/main.cpp b/device-gen/funnels/main.cpp index c54a732b7..46caebd14 100644 --- a/device-gen/funnels/main.cpp +++ b/device-gen/funnels/main.cpp @@ -16,6 +16,7 @@ #include #include #include +#include "../common/device-builder.h" #include "../common/common.h" #include "../common/collage.h" #include "../common/redistance.h" @@ -62,28 +63,6 @@ struct Parabola } }; -class DeviceBuilder -{ -protected: - typedef vector SDF; - int m_ncolumns, m_nrows; - float m_resolution, m_zmargin; - float m_unitSizeX, m_unitSizeY, m_unitSizeZ; // size of the egg with the empty space aroung it - std::string m_outFileName2D, m_outFileName3D; - int m_niterRedistance; - int m_unitNX, m_unitNY, m_unitNZ; -public: - DeviceBuilder(float unitSizeX, float unitSizeY, float unitSizeZ) - : m_ncolumns(0), m_nrows(0), m_resolution(0), m_zmargin(0), - m_unitSizeX(unitSizeX), m_unitSizeY(unitSizeY), m_unitSizeZ(unitSizeZ), - m_niterRedistance(1e3), m_unitNX(0), m_unitNY(0), m_unitNZ(0) - {} - - virtual void build() = 0; - - virtual ~DeviceBuilder() {} -}; - class FunnelsBuilder : public DeviceBuilder { float m_gapSpace; // unit gap between obstacles From 6211f35bc937bda7ed2ec1ac2e1264f0c11ff40c Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 9 Sep 2015 17:46:15 +0200 Subject: [PATCH 14/63] Remamed filder for the cylindrical pipe --- device-gen/{sdf-cylinder => pipe}/Makefile | 0 device-gen/{sdf-cylinder => pipe}/main.cpp | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename device-gen/{sdf-cylinder => pipe}/Makefile (100%) rename device-gen/{sdf-cylinder => pipe}/main.cpp (100%) diff --git a/device-gen/sdf-cylinder/Makefile b/device-gen/pipe/Makefile similarity index 100% rename from device-gen/sdf-cylinder/Makefile rename to device-gen/pipe/Makefile diff --git a/device-gen/sdf-cylinder/main.cpp b/device-gen/pipe/main.cpp similarity index 100% rename from device-gen/sdf-cylinder/main.cpp rename to device-gen/pipe/main.cpp From 7b3c800de349081e954fe8740e26f46e4613c35e Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Tue, 15 Sep 2015 14:51:53 +0200 Subject: [PATCH 15/63] - added tss - added rbcs histograms and ctc tracking - limited max contact force - fsi forces w/o repulsion - bugfix in datadump --- cell-placement/main.cpp | 519 +++++++++++---------- cuda-ctc/ctc-cuda.cu | 514 ++++++++++----------- cuda-rbc/rbc-cuda.cu | 2 +- mpi-dpd/common.h | 18 +- mpi-dpd/contact.cu | 4 +- mpi-dpd/containers.cu | 10 +- mpi-dpd/containers.h | 8 +- mpi-dpd/fsi.cu | 2 +- mpi-dpd/main.cu | 4 +- mpi-dpd/simulation.cu | 989 ++++++++++++++++++++++++---------------- mpi-dpd/simulation.h | 5 +- mpi-dpd/wall.cu | 4 +- mpi-dpd/wall.h | 2 +- 13 files changed, 1167 insertions(+), 914 deletions(-) diff --git a/cell-placement/main.cpp b/cell-placement/main.cpp index cf4679c06..b0b883d8a 100644 --- a/cell-placement/main.cpp +++ b/cell-placement/main.cpp @@ -1,6 +1,6 @@ /* * main.cpp - * Part of uDeviceX/cell-placement/ + * Part of CTC/cell-placement/ * * Created and authored by Diego Rossinelli on 2014-12-18. * Further edited by Dmitry Alexeev on 2014-03-25. @@ -34,11 +34,11 @@ Extent compute_extent(const char * const path) string line; if (in.good()) - cout << "Reading file " << path << endl; + cout << "Reading file " << path << endl; else { - cout << path << ": no such file" << endl; - exit(1); + cout << path << ": no such file" << endl; + exit(1); } int nparticles, nbonds, ntriang, ndihedrals; @@ -46,36 +46,52 @@ Extent compute_extent(const char * const path) in >> nparticles >> nbonds >> ntriang >> ndihedrals; if (in.good()) - cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang - << " triangles and " << ndihedrals << " dihedrals" << endl; + cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang + << " triangles and " << ndihedrals << " dihedrals" << endl; else { - cout << "Couldn't parse the file" << endl; - exit(1); + cout << "Couldn't parse the file" << endl; + exit(1); } vector xs(nparticles), ys(nparticles), zs(nparticles); for(int i = 0; i < nparticles; ++i) { - int dummy; - in >> dummy >> dummy >> dummy >> xs[i] >> ys[i] >> zs[i]; + int dummy; + in >> dummy >> dummy >> dummy >> xs[i] >> ys[i] >> zs[i]; + } + + Extent retval1 = { + *min_element(xs.begin(), xs.end()), + *min_element(ys.begin(), ys.end()), + *min_element(zs.begin(), zs.end()), + *max_element(xs.begin(), xs.end()), + *max_element(ys.begin(), ys.end()), + *max_element(zs.begin(), zs.end()) + }; + + for (int i=0; i= tol) - return false; - } + if (s[c] - e[c] >= tol) + return false; + } - return true; - } + return true; + } }; void verify(string path2ic) @@ -224,23 +240,23 @@ void verify(string path2ic) while(isgood) { - float tmp[19]; - for(int c = 0; c < 19; ++c) - { - int retval = fscanf(f, "%f", tmp + c); + float tmp[19]; + for(int c = 0; c < 19; ++c) + { + int retval = fscanf(f, "%f", tmp + c); - isgood &= retval == 1; - } + isgood &= retval == 1; + } - if (isgood) - { - printf("reading: "); + if (isgood) + { + printf("reading: "); - for(int c = 0; c < 19; ++c) - printf("%f ", tmp[c]); + for(int c = 0; c < 19; ++c) + printf("%f ", tmp[c]); - printf("\n"); - } + printf("\n"); + } } fclose(f); @@ -259,169 +275,174 @@ class Checker public: Checker(float hh, int dext[3], const float safetymargin): - safetymargin(safetymargin) - { - h[0] = h[1] = h[2] = hh; + safetymargin(safetymargin) +{ + h[0] = h[1] = h[2] = hh; - for (int d = 0; d < 3; ++d) - n[d] = (int)ceil((double)dext[d] / h[d]) + 2; + for (int d = 0; d < 3; ++d) + n[d] = (int)ceil((double)dext[d] / h[d]) + 2; - ntot = n[0] * n[1] * n[2]; + ntot = n[0] * n[1] * n[2]; - data.resize(ntot); - } + data.resize(ntot); +} bool check(TransformedExtent& ex) - { - int imin[3], imax[3]; + { + int imin[3], imax[3]; - for (int d=0; d<3; d++) - { - imin[d] = floor(ex.xmin[d] / h[d]) + 1; - imax[d] = floor(ex.xmax[d] / h[d]) + 1; - } + for (int d=0; d<3; d++) + { + imin[d] = floor(ex.xmin[d] / h[d]) + 1; + imax[d] = floor(ex.xmax[d] / h[d]) + 1; + } - for (int i=imin[0]; i<=imax[0]; i++) - for (int j=imin[1]; j<=imax[1]; j++) - for (int k=imin[2]; k<=imax[2]; k++) - { - const int icell = i * n[1] * n[2] + j * n[2] + k; + for (int i=imin[0]; i<=imax[0]; i++) + for (int j=imin[1]; j<=imax[1]; j++) + for (int k=imin[2]; k<=imax[2]; k++) + { + const int icell = i * n[1] * n[2] + j * n[2] + k; - bool good = true; + bool good = true; - for (auto rival : data[icell]) - good &= !ex.collides(rival, safetymargin); + for (auto rival : data[icell]) + good &= !ex.collides(rival, safetymargin); - if (!good) - return false; - } + if (!good) + return false; + } - return true; - } + return true; + } void add(TransformedExtent& ex) - { - int imin[3], imax[3]; - - for (int d=0; d<3; ++d) - { - imin[d] = floor(ex.xmin[d] / h[d]) + 1; - imax[d] = floor(ex.xmax[d] / h[d]) + 1; - } - - bool good = true; - - for (int i=imin[0]; i<=imax[0]; i++) - for (int j=imin[1]; j<=imax[1]; j++) - for (int k=imin[2]; k<=imax[2]; k++) - { - const int icell = i * n[1]*n[2] + j * n[2] + k; - data[icell].push_back(ex); - } - } + { + int imin[3], imax[3]; + + for (int d=0; d<3; ++d) + { + imin[d] = floor(ex.xmin[d] / h[d]) + 1; + imax[d] = floor(ex.xmax[d] / h[d]) + 1; + } + + bool good = true; + + for (int i=imin[0]; i<=imax[0]; i++) + for (int j=imin[1]; j<=imax[1]; j++) + for (int k=imin[2]; k<=imax[2]; k++) + { + const int icell = i * n[1]*n[2] + j * n[2] + k; + data[icell].push_back(ex); + } + } }; int main(int argc, const char ** argv) { - if ((argc != 4)&&(argc != 7)) + if (argc != 4) { - printf("usage-1: ./cell-placement \n"); - printf("usage-2: ./cell-placement \n"); - exit(-1); + printf("usage: ./cell-placement \n"); + exit(-1); } int domainextent[3]; - if (argc == 4) - { - for(int i = 0; i < 3; ++i) - domainextent[i] = atoi(argv[1 + i]); - } - if (argc == 7) - { - int ldomainextent[3]; - int ranki[3]; - for(int i = 0; i < 3; ++i) - { - ldomainextent[i] = atoi(argv[1 + i]); - ranki[i] = atoi(argv[4 + i]); - domainextent[i] = ldomainextent[i]*ranki[i]; - } - } + for(int i = 0; i < 3; ++i) + domainextent[i] = atoi(argv[1 + i]); printf("domain extent: %d %d %d\n", - domainextent[0], domainextent[1], domainextent[2]); + domainextent[0], domainextent[1], domainextent[2]); Extent extents[2] = { - compute_extent("../cuda-rbc/rbc2.atom_parsed"), - compute_extent("../cuda-ctc/sphere.dat") + compute_extent("../cuda-rbc/rbc2.atom_parsed"), + compute_extent("../cuda-ctc/sphere.dat") }; bool failed = false; vector results[2]; - const float tol = 0.7; - - Checker checker(8, domainextent, tol); + const float tol = 0.1; + Checker checker(8+tol, domainextent, tol); int tot = 0; + + TransformedExtent onectc(extents[1], domainextent); + + for (int i=0; i<4; i++) + for (int j=0; j<4; j++) + onectc.transform[i][j] = (i == j) ? 1 : 0; + onectc.transform[0][3] = 30; + onectc.transform[1][3] = 1120; + onectc.transform[2][3] = 33; + + onectc.xmin[0] = extents[1].xmin + onectc.transform[0][3]; + onectc.xmin[1] = extents[1].ymin + onectc.transform[1][3]; + onectc.xmin[2] = extents[1].zmin + onectc.transform[2][3]; + onectc.xmax[0] = extents[1].xmax + onectc.transform[0][3]; + onectc.xmax[1] = extents[1].ymax + onectc.transform[1][3]; + onectc.xmax[2] = extents[1].zmax + onectc.transform[2][3]; + + checker.add(onectc); + results[1].push_back(onectc); + ++tot; + while(!failed) { - const int maxattempts = 100000; + const int maxattempts = 100000; - int attempt = 0; - for(; attempt < maxattempts; ++attempt) - { - const int type = 0;//(int)(drand48() >= 0.25); + int attempt = 0; + for(; attempt < maxattempts; ++attempt) + { + const int type = 0;//(int)(drand48() >= 0.25); - TransformedExtent t(extents[type], domainextent); + TransformedExtent t(extents[type], domainextent); - bool noncolliding = true; + bool noncolliding = true; #if 0 //original code - for(int i = 0; i < 2; ++i) - for(int j = 0; j < results[i].size() && noncolliding; ++j) - noncolliding &= !t.collides(results[i][j], tol); + for(int i = 0; i < 2; ++i) + for(int j = 0; j < results[i].size() && noncolliding; ++j) + noncolliding &= !t.collides(results[i][j], tol); #else noncolliding = checker.check(t); #endif if (noncolliding) - { + { checker.add(t); - results[type].push_back(t); + results[type].push_back(t); ++tot; - break; - } - } + break; + } + } if (tot % 1000 == 0) - printf("Done with %d cells...\n", tot); + printf("Done with %d cells...\n", tot); - failed |= attempt == maxattempts; + failed |= attempt == maxattempts; } string output_names[2] = { "rbcs-ic.txt", "ctcs-ic.txt" }; for(int idtype = 0; idtype < 2; ++idtype) { - FILE * f = fopen(output_names[idtype].c_str(), "w"); + FILE * f = fopen(output_names[idtype].c_str(), "w"); - for(vector::iterator it = results[idtype].begin(); it != results[idtype].end(); ++it) - { - for(int c = 0; c < 3; ++c) - fprintf(f, "%f ", 0.5 * (it->xmin[c] + it->xmax[c])); + for(vector::iterator it = results[idtype].begin(); it != results[idtype].end(); ++it) + { + for(int c = 0; c < 3; ++c) + fprintf(f, "%f ", 0.5 * (it->xmin[c] + it->xmax[c])); - for(int i = 0; i < 4; ++i) - for(int j = 0; j < 4; ++j) - fprintf(f, "%f ", it->transform[i][j]); + for(int i = 0; i < 4; ++i) + for(int j = 0; j < 4; ++j) + fprintf(f, "%f ", it->transform[i][j]); - fprintf(f, "\n"); - } + fprintf(f, "\n"); + } - fclose(f); + fclose(f); } printf("Generated %d RBCs, %d CTCs\n", (int)results[0].size(), (int)results[1].size()); diff --git a/cuda-ctc/ctc-cuda.cu b/cuda-ctc/ctc-cuda.cu index 888972fec..5da4be086 100644 --- a/cuda-ctc/ctc-cuda.cu +++ b/cuda-ctc/ctc-cuda.cu @@ -28,26 +28,26 @@ using namespace std; namespace CudaCTC { -int nparticles; -int ntriang; -int nbonds; -int ndihedrals; + int nparticles; + int ntriang; + int nbonds; + int ndihedrals; -int *triangles; -int *dihedrals; + int *triangles; + int *dihedrals; int *triangles_host; int *triplets; -// Helper pointers + // Helper pointers int maxCells; __constant__ real *totA_V; real *host_av; -// Original configuration -real* orig_xyzuvw; + // Original configuration + real* orig_xyzuvw; -map bufmap; -__constant__ float A[4][4]; + map bufmap; + __constant__ float A[4][4]; Extent* dummy; @@ -59,143 +59,143 @@ __constant__ float A[4][4]; float totArea0, float totVolume0, float lunit, float tunit, int ndens, bool prn); void setup(int& nvertices, Extent& host_extent) -{ + { const float scale=1; - const bool report = false; + const bool report = false; // 0.0945, 0.00141, 1.642599, // 1, 1.8, a, v, a/m.ntriang, 945, 0, 472.5, // 90, 30, sin(phi), cos(phi), 6.048 const char* fname = "../cuda-ctc/sphere20.dat"; - ifstream in(fname); - string line; - - if (report) - if (in.good()) - { - cout << "Reading file " << fname << endl; - } - else - { - cout << fname << ": no such file" << endl; - exit(1); - } - - in >> nparticles >> nbonds >> ntriang >> ndihedrals; - - if (report) - if (in.good()) - { - cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang << " triangles and " << ndihedrals << " dihedrals" << endl; - } - else - { - cout << "Couldn't parse the file" << endl; - exit(1); - } - - // Atoms section - real *xyzuvw_host = new real[6*nparticles]; - - int cur = 0; - int tmp1, tmp2, aid; - while (in.good() && cur < nparticles) - { - in >> tmp1 >> tmp2 >> aid >> xyzuvw_host[6*cur+0] >> xyzuvw_host[6*cur+1] >> xyzuvw_host[6*cur+2]; - xyzuvw_host[6*cur+3] = xyzuvw_host[6*cur+4] = xyzuvw_host[6*cur+5] = 0; + ifstream in(fname); + string line; + + if (report) + if (in.good()) + { + cout << "Reading file " << fname << endl; + } + else + { + cout << fname << ": no such file" << endl; + exit(1); + } + + in >> nparticles >> nbonds >> ntriang >> ndihedrals; + + if (report) + if (in.good()) + { + cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang << " triangles and " << ndihedrals << " dihedrals" << endl; + } + else + { + cout << "Couldn't parse the file" << endl; + exit(1); + } + + // Atoms section + real *xyzuvw_host = new real[6*nparticles]; + + int cur = 0; + int tmp1, tmp2, aid; + while (in.good() && cur < nparticles) + { + in >> tmp1 >> tmp2 >> aid >> xyzuvw_host[6*cur+0] >> xyzuvw_host[6*cur+1] >> xyzuvw_host[6*cur+2]; + xyzuvw_host[6*cur+3] = xyzuvw_host[6*cur+4] = xyzuvw_host[6*cur+5] = 0; // Scale in dpd units xyzuvw_host[6*cur+0] *= scale; xyzuvw_host[6*cur+1] *= scale; xyzuvw_host[6*cur+2] *= scale; - if (aid != 1) break; - cur++; - } + if (aid != 1) break; + cur++; + } - // Shift the origin of "zeroth" rbc to 0,0,0 - float xmin[3] = { 1e10, 1e10, 1e10}; - float xmax[3] = {-1e10, -1e10, -1e10}; + // Shift the origin of "zeroth" rbc to 0,0,0 + float xmin[3] = { 1e10, 1e10, 1e10}; + float xmax[3] = {-1e10, -1e10, -1e10}; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1; - id0--; id1--; - bonds_host[2*i + 0] = id0; - bonds_host[2*i + 1] = id1; - } + int *bonds_host = new int[nbonds * 2]; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1; + id0--; id1--; + bonds_host[2*i + 0] = id0; + bonds_host[2*i + 1] = id1; + } - // Angles section --> triangles + // Angles section --> triangles triangles_host = new int[4*ntriang]; triplets = new int[3*ntriang]; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2; - id0--; id1--; id2--; + id0--; id1--; id2--; triangles_host[4*i + 0] = triplets[3*i + 0] = id0; triangles_host[4*i + 1] = triplets[3*i + 1] = id1; triangles_host[4*i + 2] = triplets[3*i + 2] = id2; - } + } - // Dihedrals section + // Dihedrals section - int *dihedrals_host = new int[4*ndihedrals]; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2 >> id3; - id0--; id1--; id2--; id3--; + int *dihedrals_host = new int[4*ndihedrals]; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2 >> id3; + id0--; id1--; id2--; id3--; - dihedrals_host[4*i + 0] = id0; - dihedrals_host[4*i + 1] = id1; - dihedrals_host[4*i + 2] = id2; - dihedrals_host[4*i + 3] = id3; - } + dihedrals_host[4*i + 0] = id0; + dihedrals_host[4*i + 1] = id1; + dihedrals_host[4*i + 2] = id2; + dihedrals_host[4*i + 3] = id3; + } - in.close(); + in.close(); - gpuErrchk( cudaMalloc(&orig_xyzuvw, nparticles * 6 * sizeof(float)) ); + gpuErrchk( cudaMalloc(&orig_xyzuvw, nparticles * 6 * sizeof(float)) ); gpuErrchk( cudaMalloc(&triangles, ntriang * 4 * sizeof(int)) ); - gpuErrchk( cudaMalloc(&dihedrals, ndihedrals * 4 * sizeof(int)) ); + gpuErrchk( cudaMalloc(&dihedrals, ndihedrals * 4 * sizeof(int)) ); - gpuErrchk( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nparticles * 6 * sizeof(float), cudaMemcpyHostToDevice) ); + gpuErrchk( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nparticles * 6 * sizeof(float), cudaMemcpyHostToDevice) ); gpuErrchk( cudaMemcpy(triangles, triangles_host, ntriang * 4 * sizeof(int), cudaMemcpyHostToDevice) ); - gpuErrchk( cudaMemcpy(dihedrals, dihedrals_host, ndihedrals * 4 * sizeof(int), cudaMemcpyHostToDevice) ); + gpuErrchk( cudaMemcpy(dihedrals, dihedrals_host, ndihedrals * 4 * sizeof(int), cudaMemcpyHostToDevice) ); - delete[] xyzuvw_host; - delete[] dihedrals_host; + delete[] xyzuvw_host; + delete[] dihedrals_host; - nvertices = nparticles; - host_extent.xmin = xmin[0] - origin[0]; - host_extent.ymin = xmin[1] - origin[1]; - host_extent.zmin = xmin[2] - origin[2]; + nvertices = nparticles; + host_extent.xmin = xmin[0] - origin[0]; + host_extent.ymin = xmin[1] - origin[1]; + host_extent.zmin = xmin[2] - origin[2]; - host_extent.xmax = xmax[0] - origin[0]; - host_extent.ymax = xmax[1] - origin[1]; - host_extent.zmax = xmax[2] - origin[2]; + host_extent.xmax = xmax[0] - origin[0]; + host_extent.ymax = xmax[1] - origin[1]; + host_extent.zmax = xmax[2] - origin[2]; maxCells = 5; gpuErrchk( cudaMalloc(&host_av, maxCells * 2 * sizeof(float)) ); @@ -225,7 +225,7 @@ __constant__ float A[4][4]; dummy = new Extent[maxCells]; - unitsSetup(1.64, 0.00141, 19.0476, 120, 12000, 12000, 0, 1256, 4189, 1e-6/ scale, 2.4295e-6, 4, false); + unitsSetup(1.64, 0.00141, 19.0476, 120, 40000, 40000, 0, 660, 1596, 1e-6/ scale, 2.4295e-6, 4, false); } void unitsSetup(float lmax, float p, float cq, float kb, float ka, float kv, float gammaC, @@ -243,10 +243,10 @@ __constant__ float A[4][4]; params.kbT = 580 * 250 * pow(ll, -2.0) * pow(tt, 2.0); params.p = p / ll; params.lmax = lmax / ll; - params.q = 1; + params.q = 1; params.Cq = cq * params.kbT * pow(ll, -2.0); params.totArea0 = totArea0 * pow(ll, -2.0); - params.area0 = params.totArea0 / (float)ntriang; + params.area0 = params.totArea0 / (float)ntriang; params.totVolume0 = totVolume0 * pow(ll, -3.0); params.ka = params.kbT * ka / (l0*l0); params.kd = params.kbT * 0.0 / (l0*l0); @@ -254,15 +254,15 @@ __constant__ float A[4][4]; params.gammaC = gammaC * 580 * pow(tt, 1.0); params.gammaT = 3.0 * params.gammaC; - params.rc = 0.5; - params.aij = 100; + params.rc = 0.5; + params.aij = 100; params.gamma = 15; - params.sigma = sqrt(2 * params.gamma * params.kbT); + params.sigma = sqrt(2 * params.gamma * params.kbT); // params.dt = dt; float phi = 2.7 / 180.0*M_PI; //float phi = 3.1 / 180.0*M_PI; - params.sinTheta0 = sin(phi); - params.cosTheta0 = cos(phi); + params.sinTheta0 = sin(phi); + params.cosTheta0 = cos(phi); params.kb = kb * params.kbT; params.mass = 1.1 / 0.995 * params.totVolume0 * ndens / nparticles; @@ -316,12 +316,12 @@ __constant__ float A[4][4]; printf("\t area %12.5f (%12.5f)\n", totArea0, params.totArea0); printf("\t volume %12.5f (%12.5f)\n", totVolume0, params.totVolume0); printf("************* **************** *************\n\n"); -} + } } int get_nvertices() { - return nparticles; + return nparticles; } Params& get_params() @@ -329,90 +329,90 @@ __constant__ float A[4][4]; return params; } -__global__ void transformKernel(float* xyzuvw, int n) -{ - int i = blockIdx.x * blockDim.x + threadIdx.x; - if (i >= n) return; + __global__ void transformKernel(float* xyzuvw, int n) + { + int i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= n) return; - float x = xyzuvw[6*i + 0]; - float y = xyzuvw[6*i + 1]; - float z = xyzuvw[6*i + 2]; + float x = xyzuvw[6*i + 0]; + float y = xyzuvw[6*i + 1]; + float z = xyzuvw[6*i + 2]; - xyzuvw[6*i + 0] = A[0][0]*x + A[0][1]*y + A[0][2]*z + A[0][3]; - xyzuvw[6*i + 1] = A[1][0]*x + A[1][1]*y + A[1][2]*z + A[1][3]; - xyzuvw[6*i + 2] = A[2][0]*x + A[2][1]*y + A[2][2]*z + A[2][3]; -} + xyzuvw[6*i + 0] = A[0][0]*x + A[0][1]*y + A[0][2]*z + A[0][3]; + xyzuvw[6*i + 1] = A[1][0]*x + A[1][1]*y + A[1][2]*z + A[1][3]; + xyzuvw[6*i + 2] = A[2][0]*x + A[2][1]*y + A[2][2]*z + A[2][3]; + } -void initialize(float *device_xyzuvw, const float (*transform)[4]) -{ - const int threads = 128; - const int blocks = (nparticles + threads - 1) / threads; + void initialize(float *device_xyzuvw, const float (*transform)[4]) + { + const int threads = 128; + const int blocks = (nparticles + threads - 1) / threads; - gpuErrchk( cudaMemcpyToSymbol(A, transform, 16 * sizeof(float)) ); - gpuErrchk( cudaMemcpy(device_xyzuvw, orig_xyzuvw, 6*nparticles * sizeof(float), cudaMemcpyDeviceToDevice) ); - transformKernel<<>>(device_xyzuvw, nparticles); -} + gpuErrchk( cudaMemcpyToSymbol(A, transform, 16 * sizeof(float)) ); + gpuErrchk( cudaMemcpy(device_xyzuvw, orig_xyzuvw, 6*nparticles * sizeof(float), cudaMemcpyDeviceToDevice) ); + transformKernel<<>>(device_xyzuvw, nparticles); + } -__inline__ __host__ __device__ float3 fminf(float3 a, float3 b) -{ - return make_float3(min(a.x,b.x), min(a.y,b.y), min(a.z,b.z)); -} + __inline__ __host__ __device__ float3 fminf(float3 a, float3 b) + { + return make_float3(min(a.x,b.x), min(a.y,b.y), min(a.z,b.z)); + } -__inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) -{ - return make_float3(max(a.x,b.x), max(a.y,b.y), max(a.z,b.z)); -} + __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) + { + return make_float3(max(a.x,b.x), max(a.y,b.y), max(a.z,b.z)); + } __device__ __inline__ float atomicMin(float *addr, float value) -{ - float old = *addr, assumed; - if(old <= value) return old; + { + float old = *addr, assumed; + if(old <= value) return old; - do - { - assumed = old; - old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(min(value, assumed))) ); - }while(old!=assumed); + do + { + assumed = old; + old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(min(value, assumed))) ); + }while(old!=assumed); - return old; -} + return old; + } __device__ __inline__ float atomicMax(float *addr, float value) -{ - float old = *addr, assumed; - if(old >= value) return old; + { + float old = *addr, assumed; + if(old >= value) return old; - do - { - assumed = old; - old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(max(value, assumed))) ); - }while(old!=assumed); + do + { + assumed = old; + old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(max(value, assumed))) ); + }while(old!=assumed); - return old; -} + return old; + } __global__ void extentKernel(const float* const __restrict__ xyzuvw, Extent* extent, int npart) -{ - float3 loBound = make_float3( 1e10f, 1e10f, 1e10f); - float3 hiBound = make_float3(-1e10f, -1e10f, -1e10f); + { + float3 loBound = make_float3( 1e10f, 1e10f, 1e10f); + float3 hiBound = make_float3(-1e10f, -1e10f, -1e10f); const int cid = blockIdx.y; - for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < npart; i += blockDim.x * gridDim.x) - { + for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < npart; i += blockDim.x * gridDim.x) + { const float* addr = xyzuvw + 6 * (devParams.nparticles*cid + i); float3 v = make_float3(addr[0], addr[1], addr[2]); - loBound = fminf(loBound, v); - hiBound = fmaxf(hiBound, v); - } + loBound = fminf(loBound, v); + hiBound = fmaxf(hiBound, v); + } - loBound = warpReduceMin(loBound); - __syncthreads(); - hiBound = warpReduceMax(hiBound); + loBound = warpReduceMin(loBound); + __syncthreads(); + hiBound = warpReduceMax(hiBound); - if ((threadIdx.x & (warpSize - 1)) == 0) - { + if ((threadIdx.x & (warpSize - 1)) == 0) + { atomicMin(&extent[cid].xmin, loBound.x); atomicMin(&extent[cid].ymin, loBound.y); atomicMin(&extent[cid].zmin, loBound.z); @@ -420,11 +420,11 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) atomicMax(&extent[cid].xmax, hiBound.x); atomicMax(&extent[cid].ymax, hiBound.y); atomicMax(&extent[cid].zmax, hiBound.z); - } -} + } + } void extent_nohost(cudaStream_t stream, int ncells, const float * const xyzuvw, Extent * device_extent, int n) -{ + { if (ncells == 0) return; dim3 threads(32*3, 1); @@ -449,10 +449,10 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) gpuErrchk( cudaMemcpy(device_extent, dummy, ncells * sizeof(Extent), cudaMemcpyHostToDevice) ); - if (n == -1) n = nparticles; - extentKernel<<>>(xyzuvw, device_extent, n); + if (n == -1) n = nparticles; + extentKernel<<>>(xyzuvw, device_extent, n); gpuErrchk( cudaPeekAtLastError() ); -} + } __device__ __inline__ vec3 tex2vec(int id) { @@ -462,29 +462,29 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) } __global__ void areaAndVolumeKernel() -{ - float2 a_v = make_float2(0.0f, 0.0f); + { + float2 a_v = make_float2(0.0f, 0.0f); const int cid = blockIdx.y; for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < devParams.ntriang; i += blockDim.x * gridDim.x) - { + { int4 ids = tex1Dfetch(texTriangles4, i); vec3 v0( tex2vec(6*(ids.x+cid*devParams.nparticles)) ); vec3 v1( tex2vec(6*(ids.y+cid*devParams.nparticles)) ); vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); - a_v.x += 0.5f * norm(cross(v1 - v0, v2 - v0)); - a_v.y += 0.1666666667f * (- v0.z*v1.y*v2.x + v0.z*v1.x*v2.y + v0.y*v1.z*v2.x - - v0.x*v1.z*v2.y - v0.y*v1.x*v2.z + v0.x*v1.y*v2.z); - } + a_v.x += 0.5f * norm(cross(v1 - v0, v2 - v0)); + a_v.y += 0.1666666667f * (- v0.z*v1.y*v2.x + v0.z*v1.x*v2.y + v0.y*v1.z*v2.x + - v0.x*v1.z*v2.y - v0.y*v1.x*v2.z + v0.x*v1.y*v2.z); + } - a_v = warpReduceSum(a_v); - if ((threadIdx.x & (warpSize - 1)) == 0) - { + a_v = warpReduceSum(a_v); + if ((threadIdx.x & (warpSize - 1)) == 0) + { atomicAdd(&totA_V[2*cid+0], a_v.x); atomicAdd(&totA_V[2*cid+1], a_v.y); - } -} + } + } __global__ void perTriangle(float* fxfyfz) { @@ -500,44 +500,44 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) vec3 v1( tex2vec(6*(ids.y+cid*devParams.nparticles)) ); vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); - vec3 ksi = cross(v1 - v0, v2 - v0); - float area = 0.5f * norm(ksi); + vec3 ksi = cross(v1 - v0, v2 - v0); + float area = 0.5f * norm(ksi); - // in-plane + // in-plane float alpha = 0.25f * devParams.q*devParams.Cq / powf(area, devParams.q+2.0f); - // area conservation + // area conservation float beta_a = -0.25f * ( devParams.ka*(totArea - devParams.totArea0) / (devParams.totArea0*area) + devParams.kd * (area - devParams.area0) / (devParams.area0 * area) ); - alpha += beta_a; - vec3 f0, f1, f2; + alpha += beta_a; + vec3 f0, f1, f2; - f0 = cross(ksi, v2-v1)*alpha; - f1 = cross(ksi, v0-v2)*alpha; - f2 = cross(ksi, v1-v0)*alpha; + f0 = cross(ksi, v2-v1)*alpha; + f1 = cross(ksi, v0-v2)*alpha; + f2 = cross(ksi, v1-v0)*alpha; - // volume conservation + // volume conservation // "-" here is because the normals look inside - vec3 ksi_3 = ksi*0.333333333f; - vec3 t_c = (v0 + v1 + v2) * 0.333333333f; + vec3 ksi_3 = ksi*0.333333333f; + vec3 t_c = (v0 + v1 + v2) * 0.333333333f; float beta_v = -0.1666666667f * devParams.kv * (totVolume - devParams.totVolume0) / (devParams.totVolume0); - f0 += (ksi_3 + cross(t_c, v2-v1)) * beta_v; - f1 += (ksi_3 + cross(t_c, v0-v2)) * beta_v; - f2 += (ksi_3 + cross(t_c, v1-v0)) * beta_v; + f0 += (ksi_3 + cross(t_c, v2-v1)) * beta_v; + f1 += (ksi_3 + cross(t_c, v0-v2)) * beta_v; + f2 += (ksi_3 + cross(t_c, v1-v0)) * beta_v; float* addr = fxfyfz + 3*cid*devParams.nparticles; #pragma unroll for (int d = 0; d<3; d++) - { + { atomicAdd(addr + 3*ids.x + d, f0[d]); atomicAdd(addr + 3*ids.y + d, f1[d]); atomicAdd(addr + 3*ids.z + d, f2[d]); - } -} + } + } __global__ void perDihedral(float* fxfyfz) -{ + { const int i = blockIdx.x * blockDim.x + threadIdx.x; const int cid = blockIdx.y; if (i >= devParams.ndihedrals) return; @@ -548,67 +548,67 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); vec3 v3( tex2vec(6*(ids.w+cid*devParams.nparticles)) ); - vec3 f0, f1, f2, f3; + vec3 f0, f1, f2, f3; - vec3 d21 = v2 - v1; - float r = norm(d21); - if (r < 0.0001) r = 0.0001; + vec3 d21 = v2 - v1; + float r = norm(d21); + if (r < 0.0001) r = 0.0001; float xx = r/devParams.lmax; float IbforceI = devParams.kbT / devParams.p * ( 0.25f/((1.0f-xx)*(1.0f-xx)) - 0.25f + xx ) / r; // TODO: minus?? - vec3 bforce = d21*IbforceI; - f1 += bforce; - f2 -= bforce; + vec3 bforce = d21*IbforceI; + f1 += bforce; + f2 -= bforce; - // Friction force + // Friction force vec3 u1( tex2vec(6*(ids.y+cid*devParams.nparticles) + 3) ); vec3 u2( tex2vec(6*(ids.z+cid*devParams.nparticles) + 3) ); - vec3 du21 = u2 - u1; + vec3 du21 = u2 - u1; vec3 dforce = du21*devParams.gammaT + d21 * devParams.gammaC * dot(du21, d21) / (r*r); - f1 += dforce; - f2 -= dforce; - //printf("%f %f %f\n", dforce.x, dforce.y, dforce.z); + f1 += dforce; + f2 -= dforce; + //printf("%f %f %f\n", dforce.x, dforce.y, dforce.z); - vec3 ksi = cross(v0 - v1, v0 - v2); - vec3 dzeta = cross(v2 - v3, v1 - v3); - vec3 t_c0 = (v0 + v1 + v2) * 0.3333333333f; - vec3 t_c1 = (v1 + v2 + v3) * 0.3333333333f; + vec3 ksi = cross(v0 - v1, v0 - v2); + vec3 dzeta = cross(v2 - v3, v1 - v3); + vec3 t_c0 = (v0 + v1 + v2) * 0.3333333333f; + vec3 t_c1 = (v1 + v2 + v3) * 0.3333333333f; - float IksiI = norm(ksi); - float IdzetaI = norm(dzeta); - float cosTheta = dot(ksi, dzeta) / (IksiI * IdzetaI); + float IksiI = norm(ksi); + float IdzetaI = norm(dzeta); + float cosTheta = dot(ksi, dzeta) / (IksiI * IdzetaI); - float IsinThetaI = sqrt(fabs(1.0f - cosTheta*cosTheta)); // TODO use copysign - if (fabs(IsinThetaI) < 0.001f) IsinThetaI = 0.001f; + float IsinThetaI = sqrt(fabs(1.0f - cosTheta*cosTheta)); // TODO use copysign + if (fabs(IsinThetaI) < 0.001f) IsinThetaI = 0.001f; - float sinTheta = IsinThetaI; - if (dot(ksi - dzeta, t_c0 - t_c1) > 0.0f) sinTheta = -sinTheta; // ">" because the normals look inside + float sinTheta = IsinThetaI; + if (dot(ksi - dzeta, t_c0 - t_c1) > 0.0f) sinTheta = -sinTheta; // ">" because the normals look inside float beta_b = devParams.kb * (sinTheta * devParams.cosTheta0 - cosTheta * devParams.sinTheta0) / sinTheta; - float b11 = -beta_b * cosTheta / (IksiI*IksiI); - float b12 = beta_b / (IksiI*IdzetaI); - float b22 = -beta_b * cosTheta / (IdzetaI*IdzetaI); + float b11 = -beta_b * cosTheta / (IksiI*IksiI); + float b12 = beta_b / (IksiI*IdzetaI); + float b22 = -beta_b * cosTheta / (IdzetaI*IdzetaI); - f0 += cross(ksi, v2 - v1)*b11 + cross(dzeta, v2 - v1)*b12; - f1 += cross(ksi, v0 - v2)*b11 + ( cross(ksi, v2 - v3) + cross(dzeta, v0 - v2) )*b12 + cross(dzeta, v2 - v3)*b22; - f2 += cross(ksi, v1 - v0)*b11 + ( cross(ksi, v3 - v1) + cross(dzeta, v1 - v0) )*b12 + cross(dzeta, v3 - v1)*b22; - f3 += cross(ksi, v1 - v2)*b12 + cross(dzeta, v1 - v2)*b22; + f0 += cross(ksi, v2 - v1)*b11 + cross(dzeta, v2 - v1)*b12; + f1 += cross(ksi, v0 - v2)*b11 + ( cross(ksi, v2 - v3) + cross(dzeta, v0 - v2) )*b12 + cross(dzeta, v2 - v3)*b22; + f2 += cross(ksi, v1 - v0)*b11 + ( cross(ksi, v3 - v1) + cross(dzeta, v1 - v0) )*b12 + cross(dzeta, v3 - v1)*b22; + f3 += cross(ksi, v1 - v2)*b12 + cross(dzeta, v1 - v2)*b22; float* addr = fxfyfz + 3*cid*devParams.nparticles; #pragma unroll for (int d = 0; d<3; d++) - { + { atomicAdd(addr + 3*ids.x + d, f0[d]); atomicAdd(addr + 3*ids.y + d, f1[d]); atomicAdd(addr + 3*ids.z + d, f2[d]); atomicAdd(addr + 3*ids.w + d, f3[d]); - } -} + } + } void forces_nohost(cudaStream_t stream, int ncells, const float * const device_xyzuvw, float * const device_axayaz) -{ + { if (ncells == 0) return; if (ncells > maxCells) @@ -620,7 +620,7 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) delete[] dummy; dummy = new Extent[maxCells]; - } + } size_t textureoffset; gpuErrchk( cudaBindTexture(&textureoffset, &texParticles, device_xyzuvw, &texParticles.channelDesc, ncells * nparticles * 6 * sizeof(float)) ); @@ -646,17 +646,17 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) perTriangle<<>>(device_axayaz); gpuErrchk( cudaPeekAtLastError() ); gpuErrchk( cudaUnbindTexture(texParticles) ); - } + } -void get_triangle_indexing(int (*&host_triplets_ptr)[3], int& ntriangles) -{ + void get_triangle_indexing(int (*&host_triplets_ptr)[3], int& ntriangles) + { host_triplets_ptr = (int(*)[3])triplets; - ntriangles = ntriang; -} + ntriangles = ntriang; + } float* get_orig_xyzuvw() - { + { return orig_xyzuvw; - } + } } diff --git a/cuda-rbc/rbc-cuda.cu b/cuda-rbc/rbc-cuda.cu index 340547181..6c1dd364e 100644 --- a/cuda-rbc/rbc-cuda.cu +++ b/cuda-rbc/rbc-cuda.cu @@ -309,7 +309,7 @@ namespace CudaRBC maxCells = 0; CUDA_CHECK( cudaMalloc(&host_av, 1 * 2 * sizeof(float)) ); - unitsSetup(1.64, 0.001412, 19.0476, 35, 2500, 3500, 50, 135, 91, 1e-6, 2.4295e-6, 4, report); + unitsSetup(1.194170681, 0.003092250212, 20.49568481, 39.2254922344138, 13223.5137655706, 7710.76185113627, 18.14524310, 135, 94, 1e-6, 2.4295e-6, 4, report); CUDA_CHECK( cudaFuncSetCacheConfig(fall_kernel<498>, cudaFuncCachePreferL1) ); } diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index 481592ee6..bf776c1e2 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -23,23 +23,23 @@ enum XSIZE_SUBDOMAIN = 48, YSIZE_SUBDOMAIN = 48, ZSIZE_SUBDOMAIN = 48, - XMARGIN_WALL = 6, - YMARGIN_WALL = 6, - ZMARGIN_WALL = 6, + XMARGIN_WALL = 10, + YMARGIN_WALL = 10, + ZMARGIN_WALL = 10, }; const int numberdensity = 4; -const float dt = 0.001; -const float kBT = 0.0945; -const float gammadpd = 45; +const float dt = 0.0025; +const float kBT = 1.0; +const float gammadpd = 20; const float sigma = sqrt(2 * gammadpd * kBT); const float sigmaf = sigma / sqrt(dt); -const float aij = 25; -const float hydrostatic_a = 0.05; +const float aij = 50; +const float hydrostatic_a = 0.018; extern float tend; extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces; -extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; +extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop, nsubsteps; #include #include diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 6c3689c44..1ade2ba22 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -336,7 +336,7 @@ namespace KernelsContact const float t2 = 0.0625f * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)); + const float lj = min(max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)), 1000.0f); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); @@ -545,7 +545,7 @@ namespace KernelsContact const float t2 = 0.0625f * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)); + const float lj = min(max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)), 1000.0f); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); diff --git a/mpi-dpd/containers.cu b/mpi-dpd/containers.cu index dfd96c564..75466846e 100644 --- a/mpi-dpd/containers.cu +++ b/mpi-dpd/containers.cu @@ -227,18 +227,18 @@ namespace ParticleKernels } } -void ParticleArray::update_stage1(const float driving_acceleration, cudaStream_t stream) +void ParticleArray::update_stage1(const float driving_acceleration, cudaStream_t stream, const float timestep) { if (size) ParticleKernels::update_stage1<<<(xyzuvw.size + 127) / 128, 128, 0, stream>>>( - xyzuvw.data, axayaz.data, xyzuvw.size, dt, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille, false); + xyzuvw.data, axayaz.data, xyzuvw.size, timestep, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille, false); } -void ParticleArray::update_stage2_and_1(const float driving_acceleration, cudaStream_t stream) +void ParticleArray::update_stage2_and_1(const float driving_acceleration, cudaStream_t stream, const float timestep) { if (size) ParticleKernels::update_stage2_and_1<<<(xyzuvw.size + 127) / 128, 128, 0, stream>>> - ((float2 *)xyzuvw.data, (float *)axayaz.data, xyzuvw.size, dt, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille); + ((float2 *)xyzuvw.data, (float *)axayaz.data, xyzuvw.size, timestep, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille); } void ParticleArray::resize(int n) @@ -281,6 +281,7 @@ void CollectionRBC::resize(const int count) ncells = count; ParticleArray::resize(count * get_nvertices()); + fsi_axayaz.resize(count * get_nvertices()); } void CollectionRBC::preserve_resize(const int count) @@ -288,6 +289,7 @@ void CollectionRBC::preserve_resize(const int count) ncells = count; ParticleArray::preserve_resize(count * get_nvertices()); + fsi_axayaz.preserve_resize(count * get_nvertices()); } struct TransformedExtent diff --git a/mpi-dpd/containers.h b/mpi-dpd/containers.h index 0bd850d44..592afcc0a 100644 --- a/mpi-dpd/containers.h +++ b/mpi-dpd/containers.h @@ -28,8 +28,8 @@ struct ParticleArray void resize(int n); void preserve_resize(int n); - void update_stage1(const float driving_acceleration, cudaStream_t stream); - void update_stage2_and_1(const float driving_acceleration, cudaStream_t stream); + void update_stage1(const float driving_acceleration, cudaStream_t stream, const float timestep = dt); + void update_stage2_and_1(const float driving_acceleration, cudaStream_t stream, const float timestep = dt); void clear_velocity(); void clear_acc(cudaStream_t stream) @@ -43,12 +43,13 @@ class CollectionRBC : public ParticleArray static int (*indices)[3]; static int ntriangles; static int nvertices; + SimpleDeviceBuffer fsi_axayaz; protected: MPI_Comm cartcomm; - int ncells, myrank, dims[3], periods[3], coords[3]; + int ncells, myrank, dims[3], periods[3], coords[3]; virtual int _ntriangles() const { return ntriangles; } @@ -70,6 +71,7 @@ class CollectionRBC : public ParticleArray Particle * data() { return xyzuvw.data; } Acceleration * acc() { return axayaz.data; } + Acceleration * fsiacc() { return fsi_axayaz.data; } void remove(const int * const entries, const int nentries); void resize(const int rbcs_count); void preserve_resize(int n); diff --git a/mpi-dpd/fsi.cu b/mpi-dpd/fsi.cu index 6f66540f7..0335c87b4 100644 --- a/mpi-dpd/fsi.cu +++ b/mpi-dpd/fsi.cu @@ -31,7 +31,7 @@ ComputeFSI::ComputeFSI(MPI_Comm comm) //TODO: use CUDA_CHECK(cudaEventCreateWithFlags(&evuploaded, cudaEventDisableTiming)); - KernelsFSI::Params params = {12.5 , gammadpd, sigmaf}; + KernelsFSI::Params params = {0.0f, gammadpd, sigmaf}; CUDA_CHECK(cudaMemcpyToSymbol(KernelsFSI::params, ¶ms, sizeof(params))); diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 685f13578..4c5acd799 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -25,7 +25,7 @@ bool currently_profiling = false; float tend; bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces; -int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; +int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop, nsubsteps; LocalComm localcomm; @@ -76,6 +76,7 @@ int main(int argc, char ** argv) rbcs = argp("-rbcs").asBool(false); ctcs = argp("-ctcs").asBool(false); xyz_dumps = argp("-xyz_dumps").asBool(false); + hdf5field_dumps = argp("-hdf5field_dumps").asBool(false); steps_per_report = argp("-steps_per_report").asInt(1000); steps_per_dump = argp("-steps_per_dump").asInt(1000); wall_creation_stepid = argp("-wall_creation_stepid").asInt(5000); @@ -83,6 +84,7 @@ int main(int argc, char ** argv) nvtxstop = argp("-nvtxstop").asInt(10500); adjust_message_sizes = argp("-adjust_message_sizes").asBool(false); contactforces = argp("-contactforces").asBool(false); + nsubsteps = argp("-nsubsteps").asInt(0); #ifndef _NO_DUMPS_ const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(true); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 0896c8f44..2d128d53f 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -25,25 +25,25 @@ __global__ void make_texture( float4 * __restrict xyzouvwo, ushort4 * __restrict const float2 * base = ( float2* )( xyzuvw + i * 6 ); #pragma unroll 3 for( uint j = lane; j < 96; j += 32 ) { - float2 u = base[j]; - // NVCC bug: no operator = between volatile float2 and float2 - asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); + float2 u = base[j]; + // NVCC bug: no operator = between volatile float2 and float2 + asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); } // SMEM: XYZUVW XYZUVW ... uint pid = lane / 2; const uint x_or_v = ( lane % 2 ) * 3; xyzouvwo[ i * 2 + lane ] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); pid += 16; xyzouvwo[ i * 2 + lane + 32] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); xyzo_half[i + lane] = make_ushort4( __float2half_rn( smem[ warpid * 192 + lane * 6 + 0 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); -// } + __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), + __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); + // } } void Simulation::_update_helper_arrays() @@ -69,19 +69,19 @@ std::vector Simulation::_ic() const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int iz = 0; iz < L[2]; iz++) - for(int iy = 0; iy < L[1]; iy++) - for(int ix = 0; ix < L[0]; ix++) - for(int l = 0; l < numberdensity; ++l) - { - const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); - - ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); - ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); - ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); - ic[p].u[0] = 0; - ic[p].u[1] = 0; - ic[p].u[2] = 0; - } + for(int iy = 0; iy < L[1]; iy++) + for(int ix = 0; ix < L[0]; ix++) + for(int l = 0; l < numberdensity; ++l) + { + const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); + + ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); + ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); + ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); + ic[p].u[0] = 0; + ic[p].u[1] = 0; + ic[p].u[2] = 0; + } /* use this to check robustness for(int i = 0; i < ic.size(); ++i) @@ -90,7 +90,7 @@ std::vector Simulation::_ic() ic[i].x[c] = -L[c] * 0.5 + drand48() * L[c]; ic[i].u[c] = 0; } - */ + */ return ic; } @@ -104,18 +104,18 @@ void Simulation::_redistribute() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.send(); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.bulk(particles->size, cells.start, cells.count, mainstream); @@ -125,17 +125,17 @@ void Simulation::_redistribute() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -148,10 +148,10 @@ void Simulation::_redistribute() swap(particles, newparticles); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -165,61 +165,61 @@ void Simulation::_report(const bool verbose, const int idtimestep) report_host_memory_usage(activecomm, stdout); { - static double t0 = MPI_Wtime(), t1; + static double t0 = MPI_Wtime(), t1; - t1 = MPI_Wtime(); + t1 = MPI_Wtime(); - float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; + float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; - host_busy_time *= 1e3 / steps_per_report; + host_busy_time *= 1e3 / steps_per_report; - float sumval, maxval, minval; - MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); + float sumval, maxval, minval; + MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); - int commsize; - MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); + int commsize; + MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); - const double imbalance = 100 * (maxval / sumval * commsize - 1); + const double imbalance = 100 * (maxval / sumval * commsize - 1); - if (verbose && imbalance >= 0) - printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", - imbalance , minval, sumval / commsize, maxval); + if (verbose && imbalance >= 0) + printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", + imbalance , minval, sumval / commsize, maxval); - localcomm.print_particles(particles->size); + localcomm.print_particles(particles->size); - host_idle_time = 0; - t0 = t1; + host_idle_time = 0; + t0 = t1; } { - static double t0 = MPI_Wtime(), t1; - - t1 = MPI_Wtime(); - - if (verbose) - { - printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); - printf("in more details, per time step:\n"); - double tt = 0; - for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) - { - printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); - tt += it->second; - it->second = 0; - } - printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); - } - - t0 = t1; + static double t0 = MPI_Wtime(), t1; + + t1 = MPI_Wtime(); + + if (verbose) + { + printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); + printf("in more details, per time step:\n"); + double tt = 0; + for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) + { + printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); + tt += it->second; + it->second = 0; + } + printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); + } + + t0 = t1; } } void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) { if (!coll || !coll->count()) - return; + return; SimpleDeviceBuffer marks(coll->pcount()); @@ -234,13 +234,13 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) std::vector tokill; for(int i = 0; i < nbodies; ++i) { - bool valid = true; + bool valid = true; - for(int j = 0; j < nvertices && valid; ++j) - valid &= 0 == tmp[j + nvertices * i]; + for(int j = 0; j < nvertices && valid; ++j) + valid &= 0 == tmp[j + nvertices * i]; - if (!valid) - tokill.push_back(i); + if (!valid) + tokill.push_back(i); } coll->remove(&tokill.front(), tokill.size()); @@ -252,7 +252,7 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) void Simulation::_create_walls(const bool verbose, bool & termination_request) { if (verbose) - printf("creation of the walls...\n"); + printf("creation of the walls...\n"); int nsurvived = 0; ExpectedMessageSizes new_sizes; @@ -260,30 +260,30 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) //adjust the message sizes if we're pushing the flow in x { - const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; - const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; - const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; - - for(int code = 0; code < 27; ++code) - { - const int d[3] = { - (code % 3) - 1, - ((code / 3) % 3) - 1, - ((code / 9) % 3) - 1 - }; - - const double IudotnI = - fabs(d[0] * xvelavg) + - fabs(d[1] * yvelavg) + - fabs(d[2] * zvelavg) ; - - const float factor = 1 + IudotnI * dt * 10 * numberdensity; - - //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", - //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); - - new_sizes.msgsizes[code] *= factor; - } + const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; + const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; + const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; + + for(int code = 0; code < 27; ++code) + { + const int d[3] = { + (code % 3) - 1, + ((code / 3) % 3) - 1, + ((code / 9) % 3) - 1 + }; + + const double IudotnI = + fabs(d[0] * xvelavg) + + fabs(d[1] * yvelavg) + + fabs(d[2] * zvelavg) ; + + const float factor = 1 + IudotnI * dt * 10 * numberdensity; + + //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", + //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); + + new_sizes.msgsizes[code] *= factor; + } } //MPI_CHECK(MPI_Barrier(activecomm)); @@ -315,7 +315,7 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) return; } } - */ + */ particles->resize(nsurvived); particles->clear_velocity(); @@ -330,18 +330,18 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) _remove_bodies_from_wall(ctcscoll); { - H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); - Particle * p = new Particle[particles->size]; + H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); + Particle * p = new Particle[particles->size]; - CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); + CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); - sd.dump(p, particles->size); + sd.dump(p, particles->size); - delete [] p; + delete [] p; } } -void Simulation::_forces() +void Simulation::_forces(bool firsttime) { double tstart = MPI_Wtime(); @@ -350,10 +350,10 @@ void Simulation::_forces() std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -362,10 +362,10 @@ void Simulation::_forces() particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); @@ -374,10 +374,10 @@ void Simulation::_forces() CUDA_CHECK(cudaPeekAtLastError()); if (contactforces) - contact.build_cells(wsolutes, mainstream); + contact.build_cells(wsolutes, mainstream); dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + cells.start, cells.count, mainstream); dpd.post(particles->xyzuvw.data, particles->size, mainstream, downloadstream); @@ -386,14 +386,14 @@ void Simulation::_forces() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -408,15 +408,18 @@ void Simulation::_forces() fsi.bulk(wsolutes, mainstream); if (contactforces) - contact.bulk(wsolutes, mainstream); + contact.bulk(wsolutes, mainstream); CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (nsubsteps == 0) + { + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + } CUDA_CHECK(cudaPeekAtLastError()); @@ -424,11 +427,173 @@ void Simulation::_forces() solutex.recv_a(mainstream); + if (nsubsteps) + { // TSS + if (rbcscoll) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->fsiacc(), rbcscoll->acc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + if (ctcscoll) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->fsiacc(), ctcscoll->acc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + for (int sstep = 0; sstep < nsubsteps; sstep++) + { + // Start with acc induced by solvent + if (rbcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->acc(), rbcscoll->fsiacc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + + if (firsttime) + rbcscoll->update_stage1(0.0, mainstream, (dt / nsubsteps)); + else + rbcscoll->update_stage2_and_1(0.0, mainstream, (dt / nsubsteps)); + + if (wall) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + + if (ctcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->acc(), ctcscoll->fsiacc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + + if (firsttime) + ctcscoll->update_stage2_and_1(0.0, mainstream, dt / (nsubsteps)); + else + ctcscoll->update_stage1(0.0, mainstream, dt / (nsubsteps)); + + if (wall) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + } + } + timings["interactions"] += MPI_Wtime() - tstart; CUDA_CHECK(cudaPeekAtLastError()); } +void Simulation::_qoi(Particle* rbcs, Particle * ctcs, const float tm) +{ + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + + const int subdomain[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + const int nbins = 15; // number of rows + + vector locRBChisto(nbins, 0); + vector locCTChisto(nbins, 0); + float totcom[3] = {0, 0, 0}; + + const float rwidth = 56; + const float offset = 40; + const float tan_a = tan(1.7 / 180.0 * M_PI); + + if (rbcscoll) + { + for (int p = 0; p < rbcscoll->count(); p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = rbcs + p * rbcscoll->get_nvertices(); + + for (int i=0; i < rbcscoll->get_nvertices(); i++) + for (int d = 0; d<3; d++) + { + totcom[d] += cur[i].x[d] + (coords[d] + 0.5) * subdomain[d]; + com[d] += cur[i].x[d]; + } + + for (int d = 0; d<3; d++) + com[d] = com[d] / rbcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locRBChisto[irow]++; + } + } + + if (ctcscoll) + for (int p = 0; p < ctcscoll->count(); p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = ctcs + p * ctcscoll->get_nvertices(); + + for (int i=0; i < ctcscoll->get_nvertices(); i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / ctcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locCTChisto[irow]++; + } + + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, cartcomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, cartcomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, cartcomm) ); + + + if (ctcscoll && ctcscoll->count() > 0) + { + float com[3] = {0, 0, 0}; + + Particle * cur = ctcs; + for (int i=0; i < ctcscoll->get_nvertices(); i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / ctcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; + + FILE* f = fopen("ctccom.txt", qoiid == 0 ? "w" : "a"); + fprintf(f, "%f %e %e %e\n", tm, com[0], com[1], com[2]); + fclose(f); + } + + if (rank == 0) + { + if (rbcscoll) + { + FILE* fout = fopen("rbchisto.dat", qoiid == 0 ? "w" : "a"); + fprintf(fout, "\n %f\n", tm); + for (int i=0; iget_nvertices(); + + FILE* fout = fopen("rbccom.txt", qoiid == 0 ? "w" : "a"); + fprintf(fout, "%f %e %e %e\n", tm, totcom[0] / totrbcs, totcom[1] / totrbcs, totcom[2] / totrbcs); + fclose(fout); + } + } + qoiid++; +} + + void Simulation::_datadump(const int idtimestep) { double tstart = MPI_Wtime(); @@ -436,38 +601,45 @@ void Simulation::_datadump(const int idtimestep) pthread_mutex_lock(&mutex_datadump); while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); + pthread_cond_wait(&done_datadump, &mutex_datadump); int n = particles->size; if (rbcscoll) - n += rbcscoll->pcount(); + n += rbcscoll->pcount(); if (ctcscoll) - n += ctcscoll->pcount(); + n += ctcscoll->pcount(); particles_datadump.resize(n); accelerations_datadump.resize(n); CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost,0)); CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data, particles->axayaz.data, sizeof(Acceleration) * particles->size, cudaMemcpyDeviceToHost,0)); + if (nsubsteps > 0) + { + CUDA_CHECK( cudaStreamSynchronize(0) ); + for (int i=0; isize; i++) + for (int c=0; c<3; c++) + particles_datadump.data[i].u[c] += dt * accelerations_datadump.data[i].a[c]; + } int start = particles->size; if (rbcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += rbcscoll->pcount(); + start += rbcscoll->pcount(); } if (ctcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += ctcscoll->pcount(); + start += ctcscoll->pcount(); } assert(start == n); @@ -483,7 +655,7 @@ void Simulation::_datadump(const int idtimestep) pthread_cond_signal(&request_datadump); #if defined(_SYNC_DUMPS_) while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); + pthread_cond_wait(&done_datadump, &mutex_datadump); #endif pthread_mutex_unlock(&mutex_datadump); @@ -512,117 +684,119 @@ void Simulation::_datadump_async() MPI_CHECK(MPI_Comm_rank(myactivecomm, &rank)); if (rank == 0) - mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); MPI_CHECK(MPI_Barrier(myactivecomm)); while (true) { - pthread_mutex_lock(&mutex_datadump); - async_thread_initialized = 1; + pthread_mutex_lock(&mutex_datadump); + async_thread_initialized = 1; + + while (!datadump_pending) + pthread_cond_wait(&request_datadump, &mutex_datadump); - while (!datadump_pending) - pthread_cond_wait(&request_datadump, &mutex_datadump); + pthread_mutex_unlock(&mutex_datadump); - pthread_mutex_unlock(&mutex_datadump); + if (curr_idtimestep == datadump_idtimestep) + if (simulation_is_done) + break; - if (curr_idtimestep == datadump_idtimestep) - if (simulation_is_done) - break; + CUDA_CHECK(cudaEventSynchronize(evdownloaded)); - CUDA_CHECK(cudaEventSynchronize(evdownloaded)); + const int n = particles_datadump.size; + Particle * p = particles_datadump.data; + Acceleration * a = accelerations_datadump.data; - const int n = particles_datadump.size; - Particle * p = particles_datadump.data; - Acceleration * a = accelerations_datadump.data; + { + NVTX_RANGE("diagnostics", NVTX_C1); + diagnostics(myactivecomm, mycartcomm, p, n, dt, datadump_idtimestep, a); + } - { - NVTX_RANGE("diagnostics", NVTX_C1); - diagnostics(myactivecomm, mycartcomm, p, n, dt, datadump_idtimestep, a); - } + if (xyz_dumps) + { + NVTX_RANGE("xyz dump", NVTX_C2); - if (xyz_dumps) - { - NVTX_RANGE("xyz dump", NVTX_C2); + if (walls && datadump_idtimestep >= wall_creation_stepid && !wallcreated) + { + if (rank == 0) + { + if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); - if (walls && datadump_idtimestep >= wall_creation_stepid && !wallcreated) - { - if (rank == 0) - { - if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); + if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); - if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); + if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); + } - if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); - } + MPI_CHECK(MPI_Barrier(myactivecomm)); - MPI_CHECK(MPI_Barrier(myactivecomm)); + wallcreated = true; + } - wallcreated = true; - } + xyz_dump(myactivecomm, mycartcomm, "xyz/particles->xyz", "all-particles", p, n, datadump_idtimestep > 0); + } - xyz_dump(myactivecomm, mycartcomm, "xyz/particles->xyz", "all-particles", p, n, datadump_idtimestep > 0); - } + if (hdf5part_dumps) + { + NVTX_RANGE("h5part dump", NVTX_C3); - if (hdf5part_dumps) - { - NVTX_RANGE("h5part dump", NVTX_C3); + if (!dump_part_solvent && walls && datadump_idtimestep >= wall_creation_stepid) + { + dump_part.close(); - if (!dump_part_solvent && walls && datadump_idtimestep >= wall_creation_stepid) - { - dump_part.close(); + dump_part_solvent = new H5PartDump("solvent-particles->h5part", activecomm, cartcomm); + } - dump_part_solvent = new H5PartDump("solvent-particles->h5part", activecomm, cartcomm); - } + if (dump_part_solvent) + dump_part_solvent->dump(p, n); + else + dump_part.dump(p, n); + } - if (dump_part_solvent) - dump_part_solvent->dump(p, n); - else - dump_part.dump(p, n); - } + if (hdf5field_dumps) + { + NVTX_RANGE("hdf5 field dump", NVTX_C4); - if (hdf5field_dumps) - { - NVTX_RANGE("hdf5 field dump", NVTX_C4); + dump_field.dump(activecomm, p, datadump_nsolvent, datadump_idtimestep); + } - dump_field.dump(activecomm, p, particles->size, datadump_idtimestep); - } + { + NVTX_RANGE("ply dump", NVTX_C5); - { - NVTX_RANGE("ply dump", NVTX_C5); + if (rbcscoll) + CollectionRBC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent, a + datadump_nsolvent, datadump_nrbcs, iddatadump); - if (rbcscoll) - CollectionRBC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent, a + datadump_nsolvent, datadump_nrbcs, iddatadump); + if (ctcscoll) + CollectionCTC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent + datadump_nrbcs, + a + datadump_nsolvent + datadump_nrbcs, datadump_nctcs, iddatadump); + } - if (ctcscoll) - CollectionCTC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent + datadump_nrbcs, - a + datadump_nsolvent + datadump_nrbcs, datadump_nctcs, iddatadump); - } + _qoi(p + datadump_nsolvent, p + datadump_nsolvent + datadump_nrbcs, curr_idtimestep * dt); - curr_idtimestep = datadump_idtimestep; + curr_idtimestep = datadump_idtimestep; - pthread_mutex_lock(&mutex_datadump); + pthread_mutex_lock(&mutex_datadump); - if (simulation_is_done) - { - pthread_mutex_unlock(&mutex_datadump); - break; - } + if (simulation_is_done) + { + pthread_mutex_unlock(&mutex_datadump); + break; + } - datadump_pending = false; + datadump_pending = false; - pthread_cond_signal(&done_datadump); + pthread_cond_signal(&done_datadump); - pthread_mutex_unlock(&mutex_datadump); + pthread_mutex_unlock(&mutex_datadump); - ++iddatadump; + ++iddatadump; } if (dump_part_solvent) - delete dump_part_solvent; + delete dump_part_solvent; CUDA_CHECK(cudaEventDestroy(evdownloaded)); } @@ -634,42 +808,49 @@ void Simulation::_update_and_bounce() CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - rbcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage2_and_1(0.0f, mainstream); - CUDA_CHECK(cudaPeekAtLastError()); + CUDA_CHECK(cudaPeekAtLastError()); - if (ctcscoll) - ctcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage2_and_1(0.0f, mainstream); + } timings["update"] += MPI_Wtime() - tstart; if (wall) { - tstart = MPI_Wtime(); - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + tstart = MPI_Wtime(); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); - if (rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + } - timings["bounce-walls"] += MPI_Wtime() - tstart; + timings["bounce-walls"] += MPI_Wtime() - tstart; } CUDA_CHECK(cudaPeekAtLastError()); } Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_termination)()) : - cartcomm(cartcomm), activecomm(activecomm), - /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), - rbcscoll(NULL), ctcscoll(NULL), wall(NULL), - redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), - dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), - check_termination(check_termination), - driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), - datadump_pending(false), simulation_is_done(false) + cartcomm(cartcomm), activecomm(activecomm), + /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), + rbcscoll(NULL), ctcscoll(NULL), wall(NULL), + redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), + dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), + check_termination(check_termination), + driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), + datadump_pending(false), simulation_is_done(false), + qoiid(0) { MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); @@ -677,36 +858,36 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term solutex.attach_halocomputation(fsi); if (contactforces) - solutex.attach_halocomputation(contact); + solutex.attach_halocomputation(contact); //localcomm.initialize(activecomm); int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); { - particles = &particles_pingpong[0]; - newparticles = &particles_pingpong[1]; + particles = &particles_pingpong[0]; + newparticles = &particles_pingpong[1]; - vector ic = _ic(); + vector ic = _ic(); - for(int c = 0; c < 2; ++c) - { - particles_pingpong[c].resize(ic.size()); + for(int c = 0; c < 2; ++c) + { + particles_pingpong[c].resize(ic.size()); - particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, - (0.5 + coords[1]) * YSIZE_SUBDOMAIN, - (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); + particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, + (0.5 + coords[1]) * YSIZE_SUBDOMAIN, + (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); - particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, - dims[1] * YSIZE_SUBDOMAIN, - dims[2] * ZSIZE_SUBDOMAIN); - } + particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, + dims[1] * YSIZE_SUBDOMAIN, + dims[2] * ZSIZE_SUBDOMAIN); + } - CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); + CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); - cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); + cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); - _update_helper_arrays(); + _update_helper_arrays(); } CUDA_CHECK(cudaStreamCreate(&mainstream)); @@ -715,45 +896,45 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term if (rbcs) { - rbcscoll = new CollectionRBC(cartcomm); - rbcscoll->setup("rbcs-ic.txt"); + rbcscoll = new CollectionRBC(cartcomm); + rbcscoll->setup("rbcs-ic.txt"); } if (ctcs) { - ctcscoll = new CollectionCTC(cartcomm); - ctcscoll->setup("ctcs-ic.txt"); + ctcscoll = new CollectionCTC(cartcomm); + ctcscoll->setup("ctcs-ic.txt"); } #ifndef _NO_DUMPS_ //setting up the asynchronous data dumps { - CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); - - particles_datadump.resize(particles->size * 1.5); - accelerations_datadump.resize(particles->size * 1.5); - - int rc = pthread_mutex_init(&mutex_datadump, NULL); - rc |= pthread_cond_init(&done_datadump, NULL); - rc |= pthread_cond_init(&request_datadump, NULL); - async_thread_initialized = 0; - rc |= pthread_create(&thread_datadump, NULL, datadump_trampoline, this); - - while (1) - { - pthread_mutex_lock(&mutex_datadump); - int done = async_thread_initialized; - pthread_mutex_unlock(&mutex_datadump); - - if (done) - break; - } - - if (rc) - { - printf("ERROR; return code from pthread_create() is %d\n", rc); - exit(-1); - } + CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); + + particles_datadump.resize(particles->size * 1.5); + accelerations_datadump.resize(particles->size * 1.5); + + int rc = pthread_mutex_init(&mutex_datadump, NULL); + rc |= pthread_cond_init(&done_datadump, NULL); + rc |= pthread_cond_init(&request_datadump, NULL); + async_thread_initialized = 0; + rc |= pthread_create(&thread_datadump, NULL, datadump_trampoline, this); + + while (1) + { + pthread_mutex_lock(&mutex_datadump); + int done = async_thread_initialized; + pthread_mutex_unlock(&mutex_datadump); + + if (done) + break; + } + + if (rc) + { + printf("ERROR; return code from pthread_create() is %d\n", rc); + exit(-1); + } } #endif } @@ -767,10 +948,10 @@ void Simulation::_lockstep() std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -779,20 +960,20 @@ void Simulation::_lockstep() particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); solutex.pack_p(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + cells.start, cells.count, mainstream); if (contactforces) - contact.build_cells(wsolutes, mainstream); + contact.build_cells(wsolutes, mainstream); solutex.post_p(mainstream, downloadstream); @@ -801,8 +982,8 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -817,16 +998,18 @@ void Simulation::_lockstep() fsi.bulk(wsolutes, mainstream); if (contactforces) - contact.bulk(wsolutes, mainstream); + contact.bulk(wsolutes, mainstream); CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (nsubsteps == 0) + { + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + } CUDA_CHECK(cudaPeekAtLastError()); solutex.post_a(); @@ -834,7 +1017,7 @@ void Simulation::_lockstep() particles->update_stage2_and_1(driving_acceleration, mainstream); if (wall) - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -847,42 +1030,81 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); CUDA_CHECK(cudaPeekAtLastError()); solutex.recv_a(mainstream); - if (rbcscoll) - rbcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage2_and_1(0.0f, mainstream); - if (ctcscoll) - ctcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage2_and_1(0.0f, mainstream); - if (wall && rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (wall && rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (wall && ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (wall && ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + } + else + { // TSS + if (rbcscoll) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->fsiacc(), rbcscoll->acc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + if (ctcscoll) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->fsiacc(), ctcscoll->acc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + for (int sstep = 0; sstep < nsubsteps; sstep++) + { + // Start with acc induced by solvent + if (rbcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->acc(), rbcscoll->fsiacc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + + rbcscoll->update_stage2_and_1(0.0, mainstream, (dt / nsubsteps)); + + if (wall) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + + if (ctcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->acc(), ctcscoll->fsiacc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + + ctcscoll->update_stage1(0.0, mainstream, dt / (nsubsteps)); + + if (wall) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + } + } const int newnp = redistribute.recv_count(mainstream, host_idle_time); CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -896,25 +1118,25 @@ void Simulation::_lockstep() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -925,112 +1147,115 @@ void Simulation::_lockstep() void Simulation::run() { if (rank == 0 && !walls) - printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); + printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); double time_simulation_start = MPI_Wtime(); _redistribute(); - _forces(); + _forces(nsubsteps > 0); if (!walls && pushtheflow) - driving_acceleration = hydrostatic_a; + driving_acceleration = hydrostatic_a; particles->update_stage1(driving_acceleration, mainstream); - if (rbcscoll) - rbcscoll->update_stage1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage1(0.0f, mainstream); - if (ctcscoll) - ctcscoll->update_stage1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage1(0.0f, mainstream); + } int it; for(it = 0; it < nsteps; ++it) { - const bool verbose = it > 0 && rank == 0; + const bool verbose = it > 0 && rank == 0; #ifdef _USE_NVTX_ - if (it == nvtxstart) - { - NvtxTracer::currently_profiling = true; - CUDA_CHECK(cudaProfilerStart()); - } - else if (it == nvtxstop) - { - CUDA_CHECK(cudaProfilerStop()); - NvtxTracer::currently_profiling = false; - CUDA_CHECK(cudaDeviceSynchronize()); - - if (rank == 0) - printf("profiling session ended. terminating the simulation now...\n"); - - break; - } + if (it == nvtxstart) + { + NvtxTracer::currently_profiling = true; + CUDA_CHECK(cudaProfilerStart()); + } + else if (it == nvtxstop) + { + CUDA_CHECK(cudaProfilerStop()); + NvtxTracer::currently_profiling = false; + CUDA_CHECK(cudaDeviceSynchronize()); + + if (rank == 0) + printf("profiling session ended. terminating the simulation now...\n"); + + break; + } #endif - if (it % steps_per_report == 0) - { - CUDA_CHECK(cudaStreamSynchronize(mainstream)); + if (it % steps_per_report == 0) + { + CUDA_CHECK(cudaStreamSynchronize(mainstream)); - if (simulation_is_done = check_termination()) - break; + if (simulation_is_done = check_termination()) + break; - _report(verbose, it); - } + _report(verbose, it); + } - _redistribute(); + _redistribute(); #if 1 - lockstep_check: + lockstep_check: - const bool lockstep_OK = - !(walls && it >= wall_creation_stepid && wall == NULL) && - !(it % steps_per_dump == 0) && - !(it + 1 == nvtxstart) && - !(it + 1 == nvtxstop) && - !((it + 1) % steps_per_report == 0) && - !(it + 1 == nsteps); + const bool lockstep_OK = + !(walls && it >= wall_creation_stepid && wall == NULL) && + !(it % steps_per_dump == 0) && + !(it + 1 == nvtxstart) && + !(it + 1 == nvtxstop) && + !((it + 1) % steps_per_report == 0) && + !(it + 1 == nsteps); - if (lockstep_OK) - { - _lockstep(); + if (lockstep_OK) + { + _lockstep(); - ++it; + ++it; - goto lockstep_check; - } + goto lockstep_check; + } #endif - if (walls && it >= wall_creation_stepid && wall == NULL) - { - CUDA_CHECK(cudaDeviceSynchronize()); + if (walls && it >= wall_creation_stepid && wall == NULL) + { + CUDA_CHECK(cudaDeviceSynchronize()); - bool termination_request = false; + bool termination_request = false; - _create_walls(verbose, termination_request); + _create_walls(verbose, termination_request); - _redistribute(); + _redistribute(); - if (termination_request) - break; + if (termination_request) + break; - time_simulation_start = MPI_Wtime(); + time_simulation_start = MPI_Wtime(); - if (pushtheflow) - driving_acceleration = hydrostatic_a; + if (pushtheflow) + driving_acceleration = hydrostatic_a; - if (rank == 0) - printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); - } + if (rank == 0) + printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); + } - _forces(); + _forces(); #ifndef _NO_DUMPS_ - if (it % steps_per_dump == 0) - _datadump(it); + if (it % steps_per_dump == 0) + _datadump(it); #endif - _update_and_bounce(); + _update_and_bounce(); } const double time_simulation_stop = MPI_Wtime(); @@ -1039,12 +1264,12 @@ void Simulation::run() simulation_is_done = true; if (rank == 0) - if (it == nsteps) - printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", - telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); - else - if (it != wall_creation_stepid) - printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); + if (it == nsteps) + printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", + telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); + else + if (it != wall_creation_stepid) + printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); fflush(stdout); } @@ -1067,11 +1292,11 @@ Simulation::~Simulation() CUDA_CHECK(cudaStreamDestroy(downloadstream)); if (wall) - delete wall; + delete wall; if (rbcscoll) - delete rbcscoll; + delete rbcscoll; if (ctcscoll) - delete ctcscoll; + delete ctcscoll; } diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 5a097d522..edc2453e0 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -70,7 +70,7 @@ class Simulation const size_t nsteps; float driving_acceleration; float host_idle_time; - int nranks, rank; + int nranks, rank, qoiid; std::vector _ic(); void _update_helper_arrays(); @@ -79,7 +79,8 @@ class Simulation void _report(const bool verbose, const int idtimestep); void _create_walls(const bool verbose, bool & termination_request); void _remove_bodies_from_wall(CollectionRBC * coll); - void _forces(); + void _forces(bool firsttime = false); + void _qoi(Particle* rbcs, Particle * ctcs, const float tm); void _datadump(const int idtimestep); void _update_and_bounce(); void _lockstep(); diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index ed32af0b6..531ecfd7a 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -1086,12 +1086,12 @@ ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& CUDA_CHECK(cudaPeekAtLastError()); } -void ComputeWall::bounce(Particle * const p, const int n, cudaStream_t stream) +void ComputeWall::bounce(Particle * const p, const int n, cudaStream_t stream, const float deltat) { NVTX_RANGE("WALL/bounce", NVTX_C3) if (n > 0) - SolidWallsKernel::bounce<<< (n + 127) / 128, 128, 0, stream>>>((float2 *)p, n, myrank, dt); + SolidWallsKernel::bounce<<< (n + 127) / 128, 128, 0, stream>>>((float2 *)p, n, myrank, deltat); CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/wall.h b/mpi-dpd/wall.h index 466c54977..45909ab73 100644 --- a/mpi-dpd/wall.h +++ b/mpi-dpd/wall.h @@ -42,7 +42,7 @@ class ComputeWall ~ComputeWall(); - void bounce(Particle * const p, const int n, cudaStream_t stream); + void bounce(Particle * const p, const int n, cudaStream_t stream, const float deltat = dt); void interactions(const Particle * const p, const int n, Acceleration * const acc, const int * const cellsstart, const int * const cellscount, cudaStream_t stream); From 95486324dd43bd46f8639006dfe707fce483de6d Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 16 Sep 2015 13:21:49 +0200 Subject: [PATCH 16/63] Added desired domain size parameter to ctc-ichip builder Renamed collageSDF and mergeSDF --- device-gen/common/collage.cpp | 76 +++++++++++++++++------------------ device-gen/common/collage.h | 6 ++- device-gen/ctc-ichip/main.cpp | 34 +++++++++++----- device-gen/funnels/main.cpp | 3 +- 4 files changed, 67 insertions(+), 52 deletions(-) diff --git a/device-gen/common/collage.cpp b/device-gen/common/collage.cpp index b3c3ca511..6d2730d07 100644 --- a/device-gen/common/collage.cpp +++ b/device-gen/common/collage.cpp @@ -23,7 +23,7 @@ #include "common.h" using namespace std; -static void mergeSDF(int NX, int NY, const vector< vector >& sampleSDF, vector& outputSDF) +void collageSDF(int NX, int NY, const vector< vector >& sampleSDF, vector& outputSDF) { outputSDF.resize(sampleSDF.size() * sampleSDF[0].size()); printf("SIZE: %d\n", outputSDF.size()); @@ -61,54 +61,50 @@ void populateSDF(const int NX, const int NY, const float xextent, const float ye } } -void collageSDF(const int NX, const int NY, const float xextent, const float yextent, +void collageSDFWithWall(const int NX, const int NY, const float xextent, const float yextent, const vector< vector >& sampleSDF, const int ytimes, - bool wallInY, vector& outputSDF) + const float paddingAdd, vector& outputSDF) { - mergeSDF(NX, NY, sampleSDF, outputSDF); + collageSDF(NX, NY, sampleSDF, outputSDF); const int xtimes = 1; - if (wallInY) - { - int outputSDFNX = xtimes * NX; - int outputSDFNY = ytimes * NY; - const float x0 = -xtimes * xextent * 0.5; - const float dx = xtimes * xextent / (outputSDFNX - 1); + int outputSDFNX = xtimes * NX; + int outputSDFNY = ytimes * NY; + const float x0 = -xtimes * xextent * 0.5; + const float dx = xtimes * xextent / (outputSDFNX - 1); - const float y0 = -ytimes * yextent * 0.5; - const float dy = ytimes * yextent / (outputSDFNY - 1); + const float y0 = -ytimes * yextent * 0.5; + const float dy = ytimes * yextent / (outputSDFNY - 1); - const float angle = (1.8/180.)*M_PI; - const float normal[] = {-cos(angle), sin(angle)}; - const float wallWidth = -2*y0*tan(angle); + const float angle = (1.8/180.)*M_PI; + const float normal[] = {-cos(angle), sin(angle)}; + const float wallWidth = -2*y0*tan(angle); - float ypick = 25.0f; //15 - float widthOfBufferZone = 8-wallWidth + 0.0*(48 - 2*wallWidth); - float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); - const float angle2 = atan(xpick/ypick); - std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; - const float normal2[] = {-cos(angle2), -sin(angle2)}; - - - const float linePoint[] = {-x0 - wallWidth, y0}; - const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; + float ypick = 25.0f; //15 + float widthOfBufferZone = paddingAdd - wallWidth; + float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); + const float angle2 = atan(xpick/ypick); + std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; + const float normal2[] = {-cos(angle2), -sin(angle2)}; + + const float linePoint[] = {-x0 - wallWidth, y0}; + const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; - for(int iy = 0; iy < outputSDFNY; ++iy) - for(int ix = 0; ix < outputSDFNX; ++ix) - { - const float signX = sign(dx*ix + x0); - float p[] = {dx*ix + x0, dy*iy + y0}; - float padding = signbit(-p[0])*widthOfBufferZone; - float xsdf = -1e6; + for (int iy = 0; iy < outputSDFNY; ++iy) + for (int ix = 0; ix < outputSDFNX; ++ix) + { + const float signX = sign(dx*ix + x0); + float p[] = {dx*ix + x0, dy*iy + y0}; + float padding = signbit(-p[0])*widthOfBufferZone; + float xsdf = -1e6; - if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { - xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); - } else { - xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); - } - - outputSDF[ix + outputSDFNX*iy] = std::max(outputSDF[ix + outputSDFNX*iy], xsdf); + if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { + xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); + } else { + xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); } - } + + outputSDF[ix + outputSDFNX*iy] = std::max(outputSDF[ix + outputSDFNX*iy], xsdf); + } } void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& inputGrid, diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h index f9f577f87..ffe304c19 100644 --- a/device-gen/common/collage.h +++ b/device-gen/common/collage.h @@ -13,9 +13,11 @@ #include -void collageSDF(const int NX, const int NY, const float xextent, const float yextent, +void collageSDF(int NX, int NY, const std::vector< std::vector >& sampleSDF, std::vector& outputSDF); + +void collageSDFWithWall(const int NX, const int NY, const float xextent, const float yextent, const std::vector< std::vector >& sampleSDF, const int ytimes, - bool wallInY, std::vector& outputSDF); + const float paddingAdd, std::vector& outputSDF); void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, const int xtimes, const int ytimes, std::vector& outputSDF); diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp index b5fa0f093..2fa4f12ae 100644 --- a/device-gen/ctc-ichip/main.cpp +++ b/device-gen/ctc-ichip/main.cpp @@ -62,9 +62,10 @@ class CTCiChip1Builder : public DeviceBuilder { int m_nrepeat; const float m_angle; + float m_desiredSubdomainSzX; public: CTCiChip1Builder() - : DeviceBuilder(56.0f, 32.0f, 58.0f), + : DeviceBuilder(56.0f, 32.0f, 128.0f), m_nrepeat(0), m_angle(1.7f * M_PI / 180.0f) {} @@ -98,6 +99,12 @@ class CTCiChip1Builder : public DeviceBuilder return *this; } + CTCiChip1Builder& setDiseredSubdomainX(float x) + { + m_desiredSubdomainSzX = x; + return *this; + } + CTCiChip1Builder& setFileNameFor2D(const std::string& outFileName2D) { m_outFileName2D = outFileName2D; @@ -116,7 +123,7 @@ class CTCiChip1Builder : public DeviceBuilder void generateUnitSDF(vector& sdf) const; void shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, - float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const; + float& padding, float& addPadding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const; }; void CTCiChip1Builder::build() @@ -142,15 +149,16 @@ void CTCiChip1Builder::build() // 3 Shift rows float padding = 0.0f; + float addPadding = 0.0f; int shiftedRowNX = 0; // they are all the same length float shiftedRowSizeX = 0.0f; std::vector shiftedRows; - shiftRows(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows); + shiftRows(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, padding, addPadding, shiftedRowNX, shiftedRowSizeX, shiftedRows); // 4 Collage rows SDF finalSDF; - collageSDF(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, m_nrows, true, finalSDF); + collageSDFWithWall(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, m_nrows, addPadding, finalSDF); // 5 Apply redistancing for the result float finalExtent[] = {shiftedRowSizeX, static_cast(m_nrows * rowSizeY)}; @@ -222,7 +230,7 @@ void CTCiChip1Builder::generateUnitSDF(vector& sdf) const } void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, - float& padding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const + float& padding, float& addPadding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const { const int nRowsPerShift = static_cast(ceil(m_unitSizeX / (m_unitSizeY * tan(m_angle)))); if (fabs(m_unitSizeX / (m_unitSizeY * tan(m_angle)) - nRowsPerShift) > 1e-1) { @@ -241,13 +249,19 @@ void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float row if (padding < 32.0f) padding = 0.0f; if (padding == 57.0f) - padding = 56.0f; - padding = padding + 8; // adjust padding to have desired size + padding = m_unitSizeX; + + // additional hack to have domain size in X direction to be devisible by desiredSubdomainSzX + { + float origSzX = m_ncolumns*m_unitSizeX + padding; + addPadding = (int(origSzX/m_desiredSubdomainSzX) + 1)*m_desiredSubdomainSzX - origSzX; + padding = padding + addPadding; // adjust padding to have desired size + } - std::cout << "Launching rows generation. Padding = "<< padding < Date: Wed, 16 Sep 2015 15:15:49 +0200 Subject: [PATCH 17/63] preparing for a run --- cell-placement/main.cpp | 2 +- mpi-dpd/common.h | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/cell-placement/main.cpp b/cell-placement/main.cpp index b0b883d8a..c8dc96b9e 100644 --- a/cell-placement/main.cpp +++ b/cell-placement/main.cpp @@ -373,7 +373,7 @@ int main(int argc, const char ** argv) for (int j=0; j<4; j++) onectc.transform[i][j] = (i == j) ? 1 : 0; onectc.transform[0][3] = 30; - onectc.transform[1][3] = 1120; + onectc.transform[1][3] = 328; onectc.transform[2][3] = 33; onectc.xmin[0] = extents[1].xmin + onectc.transform[0][3]; diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index bf776c1e2..2b86bf3d1 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -20,9 +20,9 @@ enum { - XSIZE_SUBDOMAIN = 48, - YSIZE_SUBDOMAIN = 48, - ZSIZE_SUBDOMAIN = 48, + XSIZE_SUBDOMAIN = 64, + YSIZE_SUBDOMAIN = 64, + ZSIZE_SUBDOMAIN = 64, XMARGIN_WALL = 10, YMARGIN_WALL = 10, ZMARGIN_WALL = 10, From 46f093ddf3bdcb51d5b404dd08f9f07bd388747c Mon Sep 17 00:00:00 2001 From: kirilllykov Date: Wed, 16 Sep 2015 15:18:42 +0200 Subject: [PATCH 18/63] Added readme for the geometry generators --- device-gen/README.md | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) create mode 100644 device-gen/README.md diff --git a/device-gen/README.md b/device-gen/README.md new file mode 100644 index 000000000..2ca1a05d2 --- /dev/null +++ b/device-gen/README.md @@ -0,0 +1,26 @@ +# Generate microfluidic geometry + +Set of scripts to generate device geometries as Signed Distance Function in \*.dat format. +The dat format consists of header and the binary float data: + + + +## Parabolic funnels +Geometry mimicing the microfluidic device by McFaul et al [Cell separation based on size and deformability using microfluidic funnel ratchets](http://www.ncbi.nlm.nih.gov/pubmed/22517056) +To generate a this geometry with 10 rows and 20 columns and with walls in z direction of width 4: +``` +cd funnels +make +./funnel -nColumns=20 -nRows=10 -zMargin=4 -out=geom.dat +``` + +## Later displacement device +Geometry reproducing CTC-iChip1 module by Karabacak et al [Microfluidic, marker-free isolation of circulating tumor cells from blood samples](http://www.nature.com/nprot/journal/v9/n3/full/nprot.2014.044.html) + +To build geometry constisting of 13 columns and 59 rows repeated twice, with wall widht 2 and such grid resolution that 0.5 grid points correspond to 1 unit of length: +``` +cd ctc-ichip +make +./ctc-ichip -nColumns=13 -nRows=59 -nRepeat=2 -zMargin=2.0 -out=13x59x2-05.dat -zResolution=0.5 +``` + From b78708d797f8661e71aa4e1544627faeec4c1b7d Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Wed, 16 Sep 2015 15:20:53 +0200 Subject: [PATCH 19/63] preparing for a run ...2 --- cell-placement/main.cpp | 12 ++++++------ mpi-dpd/common.h | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/cell-placement/main.cpp b/cell-placement/main.cpp index c8dc96b9e..bc3e9aeb8 100644 --- a/cell-placement/main.cpp +++ b/cell-placement/main.cpp @@ -149,9 +149,9 @@ struct TransformedExtent transform[i][3] = - 0.5 * (local_xmin[i] + local_xmax[i]); const float angles[3] = { - (float)(0.02 * (drand48() - 0.5) * 2 * M_PI), - (float)(M_PI * 0.5 + 0.02 * (drand48() * 2 - 1) * M_PI), - (float)(0.02 * (drand48() - 0.5) * 2 * M_PI) + (float)(0.01 * (drand48() - 0.5) * 2 * M_PI), + (float)(M_PI * 0.5 + 0.01 * (drand48() * 2 - 1) * M_PI), + (float)(0.01 * (drand48() - 0.5) * 2 * M_PI) }; for(int d = 0; d < 3; ++d) @@ -373,8 +373,8 @@ int main(int argc, const char ** argv) for (int j=0; j<4; j++) onectc.transform[i][j] = (i == j) ? 1 : 0; onectc.transform[0][3] = 30; - onectc.transform[1][3] = 328; - onectc.transform[2][3] = 33; + onectc.transform[1][3] = 275; + onectc.transform[2][3] = 64; onectc.xmin[0] = extents[1].xmin + onectc.transform[0][3]; onectc.xmin[1] = extents[1].ymin + onectc.transform[1][3]; @@ -389,7 +389,7 @@ int main(int argc, const char ** argv) while(!failed) { - const int maxattempts = 100000; + const int maxattempts = 10000; int attempt = 0; for(; attempt < maxattempts; ++attempt) diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index 2b86bf3d1..c0b58a51a 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -35,7 +35,7 @@ const float gammadpd = 20; const float sigma = sqrt(2 * gammadpd * kBT); const float sigmaf = sigma / sqrt(dt); const float aij = 50; -const float hydrostatic_a = 0.018; +const float hydrostatic_a = 0.02; extern float tend; extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces; From 4aad9847a5621340a4337483f7c93e3dac47449d Mon Sep 17 00:00:00 2001 From: Kirill Lykov Date: Wed, 16 Sep 2015 15:26:06 +0200 Subject: [PATCH 20/63] fixed mistakes in readme --- device-gen/README.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/device-gen/README.md b/device-gen/README.md index 2ca1a05d2..105e34647 100644 --- a/device-gen/README.md +++ b/device-gen/README.md @@ -2,8 +2,12 @@ Set of scripts to generate device geometries as Signed Distance Function in \*.dat format. The dat format consists of header and the binary float data: +``` - + + +``` +Where size is the geometry length units (typically microns), grid size defines how many grid points are there. ## Parabolic funnels Geometry mimicing the microfluidic device by McFaul et al [Cell separation based on size and deformability using microfluidic funnel ratchets](http://www.ncbi.nlm.nih.gov/pubmed/22517056) From 8fe5ea014b225bc00b83da60a1a45011f8642960 Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Thu, 17 Sep 2015 13:50:31 +0200 Subject: [PATCH 21/63] fixing bugs with 0 particles --- mpi-dpd/dpd.cu | 1 + mpi-dpd/redistribute-particles.cu | 9 ++++++++- mpi-dpd/redistribute-particles.h | 2 +- mpi-dpd/simulation.cu | 1 + mpi-dpd/solute-exchange.cu | 10 +++++++++- mpi-dpd/solvent-exchange.cu | 5 ++++- 6 files changed, 24 insertions(+), 4 deletions(-) diff --git a/mpi-dpd/dpd.cu b/mpi-dpd/dpd.cu index 8bec41b51..c573642b8 100644 --- a/mpi-dpd/dpd.cu +++ b/mpi-dpd/dpd.cu @@ -320,6 +320,7 @@ namespace BipsBatch CUDA_CHECK(cudaStreamWaitEvent(computestream, evhalodone, 0)); + if (nthreads) interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); CUDA_CHECK(cudaPeekAtLastError()); diff --git a/mpi-dpd/redistribute-particles.cu b/mpi-dpd/redistribute-particles.cu index fe8442248..e31ec5d3b 100644 --- a/mpi-dpd/redistribute-particles.cu +++ b/mpi-dpd/redistribute-particles.cu @@ -473,7 +473,7 @@ subindices_remote(1.5 * numberdensity * (XSIZE_SUBDOMAIN * YSIZE_SUBDOMAIN * ZSI CUDA_CHECK(cudaMalloc(&packbuffers[i].scattered_indices, sizeof(int) * estimate)); - if (i) + if (i && estimate) { CUDA_CHECK(cudaHostAlloc(&pinnedhost_sendbufs[i], sizeof(float) * 6 * estimate, cudaHostAllocMapped)); CUDA_CHECK(cudaHostGetDevicePointer(&packbuffers[i].buffer, pinnedhost_sendbufs[i], 0)); @@ -613,10 +613,12 @@ void RedistributeParticles::pack(const Particle * const particles, const int npa _post_recv(); size_t textureoffset; + if (nparticles) CUDA_CHECK(cudaBindTexture(&textureoffset, &RedistributeParticlesKernels::texAllParticles, particles, &RedistributeParticlesKernels::texAllParticles.channelDesc, sizeof(float) * 6 * nparticles)); + if (nparticles) CUDA_CHECK(cudaBindTexture(&textureoffset, &RedistributeParticlesKernels::texAllParticlesFloat2, particles, &RedistributeParticlesKernels::texAllParticlesFloat2.channelDesc, sizeof(float) * 6 * nparticles)); @@ -740,6 +742,8 @@ void RedistributeParticles::bulk(const int nparticles, int * const cellstarts, i subindices.resize(nparticles); */ subindices.resize(nparticles); + + if (nparticles) subindex_local<<< (nparticles + 127) / 128, 128, 0, mystream>>> (nparticles, RedistributeParticlesKernels::texparticledata, cellcounts, subindices.data); /* @@ -850,6 +854,7 @@ void RedistributeParticles::recv_unpack(Particle * const particles, float4 * con RedistributeParticlesKernels::subindex_remote<<< (nhalo_padded + 127) / 128, 128, 0, mystream >>> (nhalo_padded, nhalo, cellcounts, (float2 *)remote_particles.data, subindices_remote.data); + if (compressed_cellcounts.size) compress_counts<<< (compressed_cellcounts.size + 127) / 128, 128, 0, mystream >>> (compressed_cellcounts.size, (int4 *)cellcounts, (uchar4 *)compressed_cellcounts.data); @@ -859,6 +864,7 @@ void RedistributeParticles::recv_unpack(Particle * const particles, float4 * con CUDA_CHECK(cudaMemset(scattered_indices.data, 0xff, sizeof(int) * scattered_indices.size)); #endif + if (subindices.size) RedistributeParticlesKernels::scatter_indices<<< (subindices.size + 127) / 128, 128, 0, mystream>>> (false, subindices.data, subindices.size, cellstarts, scattered_indices.data, scattered_indices.size); @@ -868,6 +874,7 @@ void RedistributeParticles::recv_unpack(Particle * const particles, float4 * con assert(scattered_indices.size == nparticles); + if (nparticles) RedistributeParticlesKernels::gather_particles<<< (nparticles + 127) / 128, 128, 0, mystream>>> (scattered_indices.data, (float2 *)remote_particles.data, nhalo, RedistributeParticlesKernels::ntexparticles, nparticles, (float2 *)particles, xyzouvwo, xyzo_half); diff --git a/mpi-dpd/redistribute-particles.h b/mpi-dpd/redistribute-particles.h index f7e281b41..48f5b01ce 100644 --- a/mpi-dpd/redistribute-particles.h +++ b/mpi-dpd/redistribute-particles.h @@ -79,7 +79,7 @@ class RedistributeParticles const double tstart = MPI_Wtime(); MPI_Status statuses[n]; - MPI_CHECK( MPI_Waitall(n, reqs, statuses) ); + MPI_CHECK( MPI_Waitall(n, reqs, statuses) ); return MPI_Wtime() - tstart; } diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 2d128d53f..efb25f508 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -55,6 +55,7 @@ void Simulation::_update_helper_arrays() xyzouvwo.resize(2 * np); xyzo_half.resize(np); + if (np) make_texture <<< (np + 1023) / 1024, 1024, 1024 * 6 * sizeof( float )>>>(xyzouvwo.data, xyzo_half.data, (float *)particles->xyzuvw.data, np ); CUDA_CHECK(cudaPeekAtLastError()); diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index 5c980e95a..2f97eb497 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -289,9 +289,15 @@ void SoluteExchange::_pack_attempt(cudaStream_t stream) CUDA_CHECK(cudaMemsetAsync(local[i].result.data, 0xff, sizeof(Acceleration) * local[i].result.capacity, stream)); } #endif + CUDA_CHECK(cudaPeekAtLastError()); + if (packscount.size) CUDA_CHECK(cudaMemsetAsync(packscount.data, 0, sizeof(int) * packscount.size, stream)); + + if (packsoffset.size) CUDA_CHECK(cudaMemsetAsync(packsoffset.data, 0, sizeof(int) * packsoffset.size, stream)); + + if (packsstart.size) CUDA_CHECK(cudaMemsetAsync(packsstart.data, 0, sizeof(int) * packsstart.size, stream)); SolutePUP::init<<< 1, 1, 0, stream >>>(); @@ -300,7 +306,7 @@ void SoluteExchange::_pack_attempt(cudaStream_t stream) { const ParticlesWrap it = wsolutes[i]; - if (it.n) + if (it.n) { CUDA_CHECK(cudaMemcpyToSymbolAsync(SolutePUP::coffsets, packsoffset.data + 26 * i, sizeof(int) * 26, 0, cudaMemcpyDeviceToDevice, stream)); @@ -361,6 +367,8 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) if (wsolutes.size() == 0) return; + CUDA_CHECK(cudaPeekAtLastError()); + //consolidate the packing { NVTX_RANGE("FSI/consolidate", NVTX_C5); diff --git a/mpi-dpd/solvent-exchange.cu b/mpi-dpd/solvent-exchange.cu index 411016f2d..3fdd19b5b 100644 --- a/mpi-dpd/solvent-exchange.cu +++ b/mpi-dpd/solvent-exchange.cu @@ -384,6 +384,7 @@ void SolventExchange::_pack_all(const Particle * const p, const int n, const boo CUDA_CHECK(cudaMemcpyToSymbolAsync(PackingHalo::baginfos, baginfos, sizeof(baginfos), 0, cudaMemcpyHostToDevice, stream)); // peh: added stream } + if (PackingHalo::ncells) PackingHalo::fill_all<<< (PackingHalo::ncells + 1) / 2, 32, 0, stream>>>(p, n, required_send_bag_size); CUDA_CHECK(cudaEventRecord(evfillall, stream)); @@ -428,6 +429,7 @@ void SolventExchange::pack(const Particle * const p, const int n, const int * co } } + if (PackingHalo::ncells) PackingHalo::count_all<<<(PackingHalo::ncells + 127) / 128, 128, 0, stream>>>(cellsstart, cellscount, PackingHalo::ncells); PackingHalo::scan_diego< 32 ><<< 26, 32 * 32, 0, stream>>>(); @@ -475,10 +477,11 @@ void SolventExchange::pack(const Particle * const p, const int n, const int * co } } + if (PackingHalo::ncells) PackingHalo::copycells<0><<< (PackingHalo::ncells + 127) / 128, 128, 0, stream>>>(PackingHalo::ncells); _pack_all(p, n, firstpost, stream); - + CUDA_CHECK(cudaPeekAtLastError()); } void SolventExchange::post(const Particle * const p, const int n, cudaStream_t stream, cudaStream_t downloadstream) From dc988b961e8242225cc23102ceb38cc5a994358d Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Fri, 18 Sep 2015 12:02:25 +0200 Subject: [PATCH 22/63] different comm for qoi --- mpi-dpd/simulation.cu | 8 +++++--- mpi-dpd/simulation.h | 2 +- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index efb25f508..eef2bb615 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -537,9 +537,9 @@ void Simulation::_qoi(Particle* rbcs, Particle * ctcs, const float tm) locCTChisto[irow]++; } - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, cartcomm) ); - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, cartcomm) ); - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, cartcomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, qoicomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, qoicomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, qoicomm) ); if (ctcscoll && ctcscoll->count() > 0) @@ -856,6 +856,8 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); + MPI_CHECK( MPI_Comm_dup(activecomm, &qoicomm) ); + solutex.attach_halocomputation(fsi); if (contactforces) diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index edc2453e0..22e3f8c0e 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -60,7 +60,7 @@ class Simulation bool (*check_termination)(); bool simulation_is_done; - MPI_Comm activecomm, cartcomm; + MPI_Comm activecomm, cartcomm, qoicomm; //LocalComm localcomm; cudaStream_t mainstream, uploadstream, downloadstream; From 65dcc14c1a0ad9e7248ca360f7b26b9df2b1e1f7 Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Fri, 25 Sep 2015 17:45:09 +0200 Subject: [PATCH 23/63] - dumbcray bug fixed - periodic sdf sampling - simple mpi-async datadump - mpi-io hitns for datadump --- cuda-ctc/ctc-cuda.cu | 6 + cuda-rbc/rbc-cuda.cu | 6 + mpi-dpd/Makefile | 2 +- mpi-dpd/common.h | 2 +- mpi-dpd/dumper.cu | 253 ++++++++++++++++++ mpi-dpd/dumper.h | 33 +++ mpi-dpd/io.cu | 487 ++++++++++++++++++++++++++--------- mpi-dpd/io.h | 1 - mpi-dpd/main.cu | 211 ++++++++------- mpi-dpd/redistribute-rbcs.cu | 14 +- mpi-dpd/simulation.cu | 334 ++---------------------- mpi-dpd/simulation.h | 7 +- mpi-dpd/solute-exchange.cu | 2 +- mpi-dpd/wall.cu | 7 +- 14 files changed, 807 insertions(+), 558 deletions(-) create mode 100644 mpi-dpd/dumper.cu create mode 100644 mpi-dpd/dumper.h diff --git a/cuda-ctc/ctc-cuda.cu b/cuda-ctc/ctc-cuda.cu index 5da4be086..cd4b9e7a9 100644 --- a/cuda-ctc/ctc-cuda.cu +++ b/cuda-ctc/ctc-cuda.cu @@ -177,6 +177,12 @@ namespace CudaCTC in.close(); + int *dummyiii; + if ( cudaMalloc(&dummyiii, sizeof(int)) == cudaErrorDevicesUnavailable ) + return; + else + gpuErrchk(cudaFree(dummyiii)); + gpuErrchk( cudaMalloc(&orig_xyzuvw, nparticles * 6 * sizeof(float)) ); gpuErrchk( cudaMalloc(&triangles, ntriang * 4 * sizeof(int)) ); gpuErrchk( cudaMalloc(&dihedrals, ndihedrals * 4 * sizeof(int)) ); diff --git a/cuda-rbc/rbc-cuda.cu b/cuda-rbc/rbc-cuda.cu index 6c1dd364e..f0555df51 100644 --- a/cuda-rbc/rbc-cuda.cu +++ b/cuda-rbc/rbc-cuda.cu @@ -273,6 +273,12 @@ namespace CudaRBC xyzuvw_host[6*i+5] = 0; } + int *dummy; + if ( cudaMalloc(&dummy, sizeof(int)) == cudaErrorDevicesUnavailable ) + return; + else + CUDA_CHECK(cudaFree(dummy)); + CUDA_CHECK( cudaMalloc(&orig_xyzuvw, nvertices * 6 * sizeof(float)) ); CUDA_CHECK( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nvertices * 6 * sizeof(float), cudaMemcpyHostToDevice) ); delete[] xyzuvw_host; diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index 178d49f5d..2f4a6c3a4 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -16,7 +16,7 @@ OBJS = dpd.o wall.o fsi.o contact.o \ solvent-exchange.o solute-exchange.o \ common.o containers.o io.o \ scan.o minmax.o redistancing.o \ - simulation.o main.o + simulation.o main.o dumper.o LIBS = -lcuda-dpd -lcuda-rbc -lcuda-ctc -lcudart -ldl -lz diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index c0b58a51a..6b672f029 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -35,7 +35,7 @@ const float gammadpd = 20; const float sigma = sqrt(2 * gammadpd * kBT); const float sigmaf = sigma / sqrt(dt); const float aij = 50; -const float hydrostatic_a = 0.02; +const float hydrostatic_a = 0.022; extern float tend; extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces; diff --git a/mpi-dpd/dumper.cu b/mpi-dpd/dumper.cu new file mode 100644 index 000000000..275cc506e --- /dev/null +++ b/mpi-dpd/dumper.cu @@ -0,0 +1,253 @@ +/* + * dumper.cu + * ctc daint + * + * Created by Dmitry Alexeev on Sep 24, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +#include +#include + +#include "dumper.h" +#include "containers.h" +#include "ctc.h" +#include "io.h" + +using namespace std; + +Dumper::Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm) : iocomm(iocomm), iocartcomm(iocartcomm), intercomm(intercomm) +{ + CollectionRBC *rdummy = new CollectionRBC(iocartcomm); + CollectionCTC *cdummy = new CollectionCTC(iocartcomm); + nrbcverts = rdummy->get_nvertices(); + nctcverts = cdummy->get_nvertices(); + + MPI_CHECK(MPI_Comm_rank(iocomm, &rank)); +} + +void Dumper::qoi(Particle* rbcs, Particle * ctcs, int nrbcparts, int nctcparts, const float tm) +{ + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(iocartcomm, 3, dims, periods, coords) ); + + const int subdomain[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + const int nbins = 15; // number of rows + + vector locRBChisto(nbins, 0); + vector locCTChisto(nbins, 0); + float totcom[3] = {0, 0, 0}; + + const float rwidth = 56; + const float offset = -40; + const float tan_a = tan(1.7 / 180.0 * M_PI); + + const int nrbcs = nrbcparts / nrbcverts; + const int nctcs = nctcparts / nctcverts; + + if (nrbcs) + { + for (int p = 0; p < nrbcs; p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = rbcs + p * nrbcverts; + + for (int i=0; i < nrbcverts; i++) + for (int d = 0; d<3; d++) + { + totcom[d] += cur[i].x[d] + (coords[d] + 0.5) * subdomain[d]; + com[d] += cur[i].x[d]; + } + + for (int d = 0; d<3; d++) + com[d] = com[d] / nrbcverts + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locRBChisto[irow]++; + } + } + + if (nctcs) + for (int p = 0; p < nctcs; p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = ctcs + p * nctcverts; + + for (int i=0; i < nctcverts; i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / nctcverts + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locCTChisto[irow]++; + } + + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, iocomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, iocomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, iocomm) ); + + + if (nctcs) + { + float com[3] = {0, 0, 0}; + + Particle * cur = ctcs; + for (int i=0; i < nctcverts; i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / nctcverts + (coords[d] + 0.5) * subdomain[d]; + + FILE* f = fopen("ctccom.txt", qoiid == 0 ? "w" : "a"); + fprintf(f, "%f %e %e %e\n", tm, com[0], com[1], com[2]); + fclose(f); + } + + if (rank == 0) + { + if (nrbcs) + { + FILE* fout = fopen("rbchisto.dat", qoiid == 0 ? "w" : "a"); + fprintf(fout, "\n %f\n", tm); + for (int i=0; i particles.size()) particles.resize(1.1*n); + if (n > accelerations.size()) accelerations.resize(1.1*n); + + Particle* p = &particles[0]; + Acceleration* a = &accelerations[0]; + MPI_CHECK( MPI_Recv(p, n, Particle::datatype(), rank, 0, intercomm, &status) ); + MPI_CHECK( MPI_Recv(a, n, Acceleration::datatype(), rank, 0, intercomm, &status) ); + + H5PartDump dump_part("allparticles->h5part", iocomm, iocartcomm), *dump_part_solvent = NULL; + H5FieldDump dump_field(iocartcomm); + + + if (rank == 0) + mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + + MPI_CHECK(MPI_Barrier(iocomm)); + + { + NVTX_RANGE("diagnostics", NVTX_C1); + diagnostics(iocomm, iocartcomm, p, n, dt, iddatadump, a); + } + + if (xyz_dumps) + { + NVTX_RANGE("xyz dump", NVTX_C2); + + if (walls && iddatadump >= wall_creation_stepid && !wallcreated) + { + if (rank == 0) + { + if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); + + if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); + + if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); + } + + MPI_CHECK(MPI_Barrier(iocomm)); + + wallcreated = true; + } + + xyz_dump(iocomm, iocartcomm, "xyz/particles->xyz", "all-particles", p, n, iddatadump > 0); + } + + if (hdf5part_dumps) + { + if (!dump_part_solvent && walls && iddatadump >= wall_creation_stepid) + { + dump_part.close(); + + dump_part_solvent = new H5PartDump("solvent-particles->h5part", iocomm, iocartcomm); + } + + if (dump_part_solvent) + dump_part_solvent->dump(p, n); + else + dump_part.dump(p, n); + } + + if (hdf5field_dumps) + { + dump_field.dump(iocomm, p, nparticles, iddatadump); + } + + { + if (rbcs) + CollectionRBC::dump(iocomm, iocartcomm, p + nparticles, a + nparticles, nrbcparts, iddatadump); + + if (ctcs) + CollectionCTC::dump(iocomm, iocartcomm, p + nparticles + nrbcparts, a + nparticles + nrbcparts, nctcparts, iddatadump); + } + + qoi(p + nparticles, p + nparticles + nrbcparts, nrbcparts, nctcparts, iddatadump * dt); + + ++iddatadump; + } +} + diff --git a/mpi-dpd/dumper.h b/mpi-dpd/dumper.h new file mode 100644 index 000000000..67cdf0b0c --- /dev/null +++ b/mpi-dpd/dumper.h @@ -0,0 +1,33 @@ +/* + * dumper.h + * ctc daint + * + * Created by Dmitry Alexeev on Sep 24, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +#pragma once + +#include +#include + +#include "common.h" + +using namespace std; + +class Dumper +{ + MPI_Comm iocomm, intercomm, iocartcomm; + vector particles; + vector accelerations; + + int nrbcverts, nctcverts, qoiid, rank; + + void qoi(Particle* rbcs, Particle * ctcs, int nrbcparts, int nctcparts, const float tm); + +public: + Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm); + void do_dump(); +}; + diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 11b963c29..2f427ff59 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -1,6 +1,6 @@ /* * io.cu - * Part of uDeviceX/mpi-dpd/ + * Part of CTC/mpi-dpd/ * * Created and authored by Diego Rossinelli on 2015-01-30. * Major bug in H5 dump fixed by Panotelli on 2015-03-24. @@ -27,6 +27,9 @@ #include #include +//#define USE_POSIX_IO +#include + #include "io.h" using namespace std; @@ -44,7 +47,7 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha bool filenotthere; if (rank == 0) - filenotthere = access(filename, F_OK ) == -1; + filenotthere = access(filename, F_OK ) == -1; MPI_CHECK( MPI_Bcast(&filenotthere, 1, MPI_INT, 0, comm) ); @@ -54,7 +57,7 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), MPI_INFO_NULL, &f) ); if (!append) - MPI_CHECK( MPI_File_set_size (f, 0)); + MPI_CHECK( MPI_File_set_size (f, 0)); MPI_Offset base; MPI_CHECK( MPI_File_get_position(f, &base)); @@ -63,17 +66,17 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha if (rank == 0) { - ss << n << "\n"; - ss << particlename << "\n"; + ss << n << "\n"; + ss << particlename << "\n"; - printf("xyz dump <%s>: total number of particles: %d\n", filename, n); + printf("xyz dump <%s>: total number of particles: %d\n", filename, n); } for(int i = 0; i < nlocal; ++i) - ss << rank << " " - << (particles[i].x[0] + XSIZE_SUBDOMAIN / 2 + coords[0] * XSIZE_SUBDOMAIN) << " " - << (particles[i].x[1] + YSIZE_SUBDOMAIN / 2 + coords[1] * YSIZE_SUBDOMAIN) << " " - << (particles[i].x[2] + ZSIZE_SUBDOMAIN / 2 + coords[2] * ZSIZE_SUBDOMAIN) << "\n"; + ss << rank << " " + << (particles[i].x[0] + XSIZE_SUBDOMAIN / 2 + coords[0] * XSIZE_SUBDOMAIN) << " " + << (particles[i].x[1] + YSIZE_SUBDOMAIN / 2 + coords[1] * YSIZE_SUBDOMAIN) << " " + << (particles[i].x[2] + ZSIZE_SUBDOMAIN / 2 + coords[2] * ZSIZE_SUBDOMAIN) << "\n"; string content = ss.str(); @@ -83,102 +86,337 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha MPI_Status status; - MPI_CHECK( MPI_File_write_at_all(f, base + offset, const_cast(content.c_str()), len, MPI_CHAR, &status)); + MPI_CHECK( MPI_File_write_at(f, base + offset, const_cast(content.c_str()), len, MPI_CHAR, &status)); MPI_CHECK( MPI_File_close(&f)); } -void _write_bytes(const void * const ptr, const int nbytes32, MPI_File f, MPI_Comm comm) +void _write_bytes_mpi(const void * const ptr, const int nbytes32, MPI_File f, MPI_Offset base0, MPI_Offset offset0, MPI_Comm comm, int rank) { - MPI_Offset base; - MPI_CHECK( MPI_File_get_position(f, &base)); - - MPI_Offset offset = 0, nbytes = nbytes32; - MPI_CHECK( MPI_Exscan(&nbytes, &offset, 1, MPI_OFFSET, MPI_SUM, comm)); + MPI_Offset base = base0; + MPI_Offset offset = offset0; + MPI_Offset nbytes = nbytes32; MPI_Status status; MPI_CHECK( MPI_File_write_at_all(f, base + offset, ptr, nbytes, MPI_CHAR, &status)); +} + +void _write_bytes_posix(const void * const ptr, const int nbytes32, int f, MPI_Offset base0, MPI_Offset offset0, MPI_Comm comm) +{ + MPI_Offset base = base0; + MPI_Offset offset = offset0; + MPI_Offset nbytes = nbytes32; - MPI_Offset ntotal = 0; - MPI_CHECK( MPI_Allreduce(&nbytes, &ntotal, 1, MPI_OFFSET, MPI_SUM, comm) ); + MPI_Offset rc; + rc = lseek64(f, base + offset, SEEK_SET); + if (rc == -1) + { + printf("lseek64 failed!\n"); + MPI_Abort(MPI_COMM_WORLD, rc); + } - MPI_CHECK( MPI_File_seek(f, ntotal, MPI_SEEK_CUR)); + char * ptr0 = (char *)ptr; + MPI_Offset remaining = nbytes; + while (remaining > 0) + { + rc = write(f, ptr0, nbytes); + if (rc == -1) + { + printf("write failed!\n"); + MPI_Abort(MPI_COMM_WORLD, rc); + } + if (rc < remaining) + { + } + remaining -= rc; + ptr0 += rc; + } } -void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, - int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, - Particle * _particles, int nvertices_per_instance, bool append) +void ply_dump_mpi(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) { + double t0 = MPI_Wtime(); std::vector particles(_particles, _particles + ninstances * nvertices_per_instance); - int rank; + int rank, size; MPI_CHECK( MPI_Comm_rank(comm, &rank) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + /* Part II: particles */ int NPOINTS = 0; const int n = particles.size(); - MPI_CHECK( MPI_Reduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, 0, comm) ); + MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, comm) ); + /* Part III: triangles */ const int ntriangles = ntriangles_per_instance * ninstances; int NTRIANGLES = 0; - MPI_CHECK( MPI_Reduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, 0, comm) ); + MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, comm) ); + + /* Part I: header */ + std::stringstream ss; + if (rank == 0) + { + ss << "ply\n"; + ss << "format binary_little_endian 1.0\n"; + ss << "element vertex " << NPOINTS << "\n"; + ss << "property float x\nproperty float y\nproperty float z\n"; + ss << "property float u\nproperty float v\nproperty float w\n"; + //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; + ss << "element face " << NTRIANGLES << "\n"; + ss << "property list int int vertex_index\n"; + ss << "end_header\n"; + } + string content = ss.str(); + + const int headersize = content.size(); + int HEADERSIZE = 0; + MPI_CHECK( MPI_Allreduce(&headersize, &HEADERSIZE, 1, MPI_INT, MPI_SUM, comm) ); + + unsigned long size0 = HEADERSIZE; + unsigned long size1 = NPOINTS*sizeof(Particle); + // unsigned long size2 = NTRIANGLES*4*sizeof(int); + + MPI_Offset base0 = 0; + MPI_Offset base1 = base0 + size0; + MPI_Offset base2 = base1 + size1; + + int ioffset0 = 0; + int ioffset1 = 0; + int ioffset2 = 0; + MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_INTEGER, MPI_SUM, comm)); + + MPI_Offset poffset0 = ioffset0*sizeof(char); + MPI_Offset poffset1 = ioffset1*sizeof(Particle); + MPI_Offset poffset2 = ioffset2*4*sizeof(int); MPI_File f; - MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), MPI_INFO_NULL, &f) ); + double t1 = MPI_Wtime(); - if (!append) - MPI_CHECK( MPI_File_set_size (f, 0)); + int cb = 1; + while (cb*2 <= size) cb *= 2; - std::stringstream ss; + cb = min(cb, 128); + char cbstr[100]; + sprintf(cbstr, "%d", cb); + + MPI_Info info; + MPI_Info_create(&info); + MPI_Info_set(info, "cb_nodes", cbstr); + MPI_Info_set(info, "romio_cb_write", "enable"); + MPI_Info_set(info, "romio_ds_write", "disable"); + MPI_Info_set(info, "striping_factor", cbstr); + + MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), info, &f) ); + double t2 = MPI_Wtime(); + +#if 0 + if (!append) { + // MPI_CHECK( MPI_File_set_size (f, 0)); + MPI_CHECK (MPI_File_set_size (f, size0 + size1 + size2)); + } +#endif + + _write_bytes_mpi(content.c_str(), content.size(), f, base0, poffset0, comm, rank); + const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; + + for(int i = 0; i < n; ++i) + for(int c = 0; c < 3; ++c) + particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; + + _write_bytes_mpi(&particles.front(), sizeof(Particle) * n, f, base1, poffset1, comm, rank); + + int poffset = ioffset1; + std::vector buf; + + for(int j = 0; j < ninstances; ++j) + for(int i = 0; i < ntriangles_per_instance; ++i) + { + int primitive[4] = { 3, + poffset + nvertices_per_instance * j + mesh_indices[i][0], + poffset + nvertices_per_instance * j + mesh_indices[i][1], + poffset + nvertices_per_instance * j + mesh_indices[i][2] }; + + buf.insert(buf.end(), primitive, primitive + 4); + } + + _write_bytes_mpi(&buf.front(), sizeof(int) * buf.size(), f, base2, poffset2, comm, rank); + + MPI_Barrier(comm); + + double t3 = MPI_Wtime(); + MPI_CHECK( MPI_File_close(&f)); + double t4 = MPI_Wtime(); + + double d0 = 1e3*(t1 - t0); + double d1 = 1e3*(t2 - t1); + double d2 = 1e3*(t3 - t2); + double d3 = 1e3*(t4 - t3); + + MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d0, &d0, 1, MPI_DOUBLE, MPI_MAX, 0, comm); + MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d1, &d1, 1, MPI_DOUBLE, MPI_MAX, 0, comm); + MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d2, &d2, 1, MPI_DOUBLE, MPI_MAX, 0, comm); + MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d3, &d3, 1, MPI_DOUBLE, MPI_MAX, 0, comm); + + if (!rank) printf("ply_dump_mpi:\t %.2f ms; \t prep: %.2f, open %.2f, write %.2f, close %.2f\n", d0+d1+d2+d3, d0, d1, d2, d3); +} + +void ply_dump_posix(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) +{ + double t0 = MPI_Wtime(); + std::vector particles(_particles, _particles + ninstances * nvertices_per_instance); + + int rank; + MPI_CHECK( MPI_Comm_rank(comm, &rank) ); + + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + + /* Part II: particles */ + int NPOINTS = 0; + const int n = particles.size(); + MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, comm) ); + + /* Part III: triangles */ + const int ntriangles = ntriangles_per_instance * ninstances; + int NTRIANGLES = 0; + MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, comm) ); + /* Part I: header */ + std::stringstream ss; if (rank == 0) { - ss << "ply\n"; - ss << "format binary_little_endian 1.0\n"; - ss << "element vertex " << NPOINTS << "\n"; - ss << "property float x\nproperty float y\nproperty float z\n"; - ss << "property float u\nproperty float v\nproperty float w\n"; - //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; - ss << "element face " << NTRIANGLES << "\n"; - ss << "property list int int vertex_index\n"; - ss << "end_header\n"; + ss << "ply\n"; + ss << "format binary_little_endian 1.0\n"; + ss << "element vertex " << NPOINTS << "\n"; + ss << "property float x\nproperty float y\nproperty float z\n"; + ss << "property float u\nproperty float v\nproperty float w\n"; + //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; + ss << "element face " << NTRIANGLES << "\n"; + ss << "property list int int vertex_index\n"; + ss << "end_header\n"; } - string content = ss.str(); - _write_bytes(content.c_str(), content.size(), f, comm); + const int headersize = content.size(); + int HEADERSIZE = 0; + MPI_CHECK( MPI_Allreduce(&headersize, &HEADERSIZE, 1, MPI_INT, MPI_SUM, comm) ); - const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; + unsigned long size0 = HEADERSIZE; + unsigned long size1 = NPOINTS*sizeof(Particle); + // unsigned long size2 = NTRIANGLES*4*sizeof(int); - for(int i = 0; i < n; ++i) - for(int c = 0; c < 3; ++c) - particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; + MPI_Offset base0 = 0; + MPI_Offset base1 = base0 + size0; + MPI_Offset base2 = base1 + size1; + + int ioffset0 = 0; + int ioffset1 = 0; + int ioffset2 = 0; + MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_INTEGER, MPI_SUM, comm)); + + MPI_Offset poffset0 = ioffset0*sizeof(char); + MPI_Offset poffset1 = ioffset1*sizeof(Particle); + MPI_Offset poffset2 = ioffset2*4*sizeof(int); + + int f; + if (rank == 0) + { + f = open((char *)filename, O_CREAT | O_WRONLY | O_DIRECT, 0664); + if (f <= 0) + { + printf("File creation failed!\n"); + MPI_Abort(MPI_COMM_WORLD, f); + } + //ftruncate(f, size0+size1+size2); + close(f); + } + +#if 1 + MPI_CHECK(MPI_Barrier(comm)); +#else + int flag = 0; + MPI_Status status; + MPI_Request req; + MPI_Ibarrier(comm, &req); + while (1) + { + MPI_Test(&req, &flag, &status); + if (flag == 1) + break; + else + usleep(100); + } +#endif + f = open((char *)filename, O_WRONLY, 0664); + if (f <= 0) + { + printf("File opening failed!\n"); + MPI_Abort(MPI_COMM_WORLD, f); + } - _write_bytes(&particles.front(), sizeof(Particle) * n, f, comm); + _write_bytes_posix(content.c_str(), content.size(), f, base0, poffset0, comm); + const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; - int poffset = 0; + for(int i = 0; i < n; ++i) + for(int c = 0; c < 3; ++c) + particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; - MPI_CHECK( MPI_Exscan(&n, &poffset, 1, MPI_INTEGER, MPI_SUM, comm)); + _write_bytes_posix(&particles.front(), sizeof(Particle) * n, f, base1, poffset1, comm); + int poffset = ioffset1; std::vector buf; for(int j = 0; j < ninstances; ++j) - for(int i = 0; i < ntriangles_per_instance; ++i) - { - int primitive[4] = { 3, - poffset + nvertices_per_instance * j + mesh_indices[i][0], - poffset + nvertices_per_instance * j + mesh_indices[i][1], - poffset + nvertices_per_instance * j + mesh_indices[i][2] }; + for(int i = 0; i < ntriangles_per_instance; ++i) + { + int primitive[4] = { 3, + poffset + nvertices_per_instance * j + mesh_indices[i][0], + poffset + nvertices_per_instance * j + mesh_indices[i][1], + poffset + nvertices_per_instance * j + mesh_indices[i][2] }; - buf.insert(buf.end(), primitive, primitive + 4); - } + buf.insert(buf.end(), primitive, primitive + 4); + } - _write_bytes(&buf.front(), sizeof(int) * buf.size(), f, comm); + _write_bytes_posix(&buf.front(), sizeof(int) * buf.size(), f, base2, poffset2, comm); + + close(f); + + double t1 = MPI_Wtime(); + if (!rank && (t1-t0 > 0.05)) printf("ply_dump_posix:\t %f ms\n", 1e3*(t1-t0)); +} + +void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) +{ +#ifdef USE_POSIX_IO + ply_dump_posix(comm, cartcomm, filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); +#else + ply_dump_mpi(comm, cartcomm, filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); + +#if 0 // only for debugging + char new_filename[256]; + strcpy(new_filename, filename); + strcat(new_filename, ".posix"); + ply_dump_posix(comm, cartcomm, new_filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); +#endif +#endif - MPI_CHECK( MPI_File_close(&f)); } H5PartDump::H5PartDump(const string fname, MPI_Comm comm, MPI_Comm cartcomm): tstamp(0), disposed(false) @@ -195,7 +433,7 @@ void H5PartDump::_initialize(const std::string filename, MPI_Comm comm, MPI_Comm const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int c = 0; c < 3; ++c) - origin[c] = L[c] / 2 + coords[c] * L[c]; + origin[c] = L[c] / 2 + coords[c] * L[c]; mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); @@ -215,7 +453,7 @@ void H5PartDump::dump(Particle * host_particles, int n) { #ifndef NO_H5PART if (disposed) - return; + return; H5PartFile * f = (H5PartFile *)handler; @@ -227,12 +465,12 @@ void H5PartDump::dump(Particle * host_particles, int n) for(int c = 0; c < 3; ++c) { - vector data(n); + vector data(n); - for(int i = 0; i < n; ++i) - data[i] = host_particles[i].x[c] + origin[c]; + for(int i = 0; i < n; ++i) + data[i] = host_particles[i].x[c] + origin[c]; - H5PartWriteDataFloat32(f, labels[c].c_str(), &data.front()); + H5PartWriteDataFloat32(f, labels[c].c_str(), &data.front()); } tstamp++; @@ -244,20 +482,20 @@ void H5PartDump::_dispose() #ifndef NO_H5PART if (!disposed) { - H5PartFile * f = (H5PartFile *)handler; + H5PartFile * f = (H5PartFile *)handler; - H5PartCloseFile(f); + H5PartCloseFile(f); - disposed = true; + disposed = true; - handler = NULL; + handler = NULL; } #endif } H5PartDump::~H5PartDump() { - _dispose(); + _dispose(); } void H5FieldDump::_xdmf_header(FILE * xmf) @@ -269,12 +507,12 @@ void H5FieldDump::_xdmf_header(FILE * xmf) } void H5FieldDump::_xdmf_grid(FILE * xmf, float time, - const char * const h5path, const char * const * channelnames, int nchannels) + const char * const h5path, const char * const * channelnames, int nchannels) { fprintf(xmf, " \n"); fprintf(xmf, " \n"); - } +} void H5FieldDump::_xdmf_epilogue(FILE * xmf) { @@ -309,8 +547,8 @@ void H5FieldDump::_xdmf_epilogue(FILE * xmf) } void H5FieldDump::_write_fields(const char * const path2h5, - const float * const channeldata[], const char * const * const channelnames, const int nchannels, - MPI_Comm comm, const float time) + const float * const channeldata[], const char * const * const channelnames, const int nchannels, + MPI_Comm comm, const float time) { #ifndef NO_H5 int nranks[3], periods[3], myrank[3]; @@ -328,23 +566,23 @@ void H5FieldDump::_write_fields(const char * const path2h5, for(int ichannel = 0; ichannel < nchannels; ++ichannel) { - hid_t dset_id = H5Dcreate(file_id, channelnames[ichannel], H5T_NATIVE_FLOAT, filespace_simple, H5P_DEFAULT, H5P_DEFAULT, H5P_DEFAULT); - id_t plist_id = H5Pcreate(H5P_DATASET_XFER); + hid_t dset_id = H5Dcreate(file_id, channelnames[ichannel], H5T_NATIVE_FLOAT, filespace_simple, H5P_DEFAULT, H5P_DEFAULT, H5P_DEFAULT); + id_t plist_id = H5Pcreate(H5P_DATASET_XFER); - H5Pset_dxpl_mpio(plist_id, H5FD_MPIO_COLLECTIVE); + H5Pset_dxpl_mpio(plist_id, H5FD_MPIO_COLLECTIVE); - hsize_t start[4] = { myrank[2] * L[2], myrank[1] * L[1], myrank[0] * L[0], 0}; - hsize_t extent[4] = { L[2], L[1], L[0], 1}; - hid_t filespace = H5Dget_space(dset_id); - H5Sselect_hyperslab(filespace, H5S_SELECT_SET, start, NULL, extent, NULL); + hsize_t start[4] = { myrank[2] * L[2], myrank[1] * L[1], myrank[0] * L[0], 0}; + hsize_t extent[4] = { L[2], L[1], L[0], 1}; + hid_t filespace = H5Dget_space(dset_id); + H5Sselect_hyperslab(filespace, H5S_SELECT_SET, start, NULL, extent, NULL); - hid_t memspace = H5Screate_simple(4, extent, NULL); - herr_t status = H5Dwrite(dset_id, H5T_NATIVE_FLOAT, memspace, filespace, plist_id, channeldata[ichannel]); + hid_t memspace = H5Screate_simple(4, extent, NULL); + herr_t status = H5Dwrite(dset_id, H5T_NATIVE_FLOAT, memspace, filespace, plist_id, channeldata[ichannel]); - H5Sclose(memspace); - H5Sclose(filespace); - H5Pclose(plist_id); - H5Dclose(dset_id); + H5Sclose(memspace); + H5Sclose(filespace); + H5Pclose(plist_id); + H5Dclose(dset_id); } H5Sclose(filespace_simple); @@ -355,17 +593,17 @@ void H5FieldDump::_write_fields(const char * const path2h5, if (!rankscalar) { - char wrapper[256]; - sprintf(wrapper, "%s.xmf", string(path2h5).substr(0, string(path2h5).find_last_of(".h5") - 2).data()); + char wrapper[256]; + sprintf(wrapper, "%s.xmf", string(path2h5).substr(0, string(path2h5).find_last_of(".h5") - 2).data()); - FILE * xmf = fopen(wrapper, "w"); - assert(xmf); + FILE * xmf = fopen(wrapper, "w"); + assert(xmf); - _xdmf_header(xmf); - _xdmf_grid(xmf, time, string(path2h5).substr(string(path2h5).find_last_of("/") + 1).c_str(), channelnames, nchannels); - _xdmf_epilogue(xmf); + _xdmf_header(xmf); + _xdmf_grid(xmf, time, string(path2h5).substr(string(path2h5).find_last_of("/") + 1).c_str(), channelnames, nchannels); + _xdmf_epilogue(xmf); - fclose(xmf); + fclose(xmf); } #endif // NO_H5 } @@ -378,7 +616,7 @@ H5FieldDump::H5FieldDump(MPI_Comm cartcomm): cartcomm(cartcomm), last_idtimestep const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int c = 0; c < 3; ++c) - globalsize[c] = L[c] * dims[c]; + globalsize[c] = L[c] * dims[c]; } void H5FieldDump::dump_scalarfield(MPI_Comm comm, const float * const data, const char * channelname) @@ -399,41 +637,41 @@ void H5FieldDump::dump(MPI_Comm comm, const Particle * const p, const int n, int vector rho(ncells), u[3]; for(int c = 0; c < 3; ++c) - u[c].resize(ncells); + u[c].resize(ncells); for(int i = 0; i < n; ++i) { - const int cellindex[3] = { - max(0, min(XSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[0])) + XSIZE_SUBDOMAIN / 2)), - max(0, min(YSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[1])) + YSIZE_SUBDOMAIN / 2)), - max(0, min(ZSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[2])) + ZSIZE_SUBDOMAIN / 2)) + const int cellindex[3] = { + max(0, min(XSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[0])) + XSIZE_SUBDOMAIN / 2)), + max(0, min(YSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[1])) + YSIZE_SUBDOMAIN / 2)), + max(0, min(ZSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[2])) + ZSIZE_SUBDOMAIN / 2)) }; - const int entry = cellindex[0] + XSIZE_SUBDOMAIN * (cellindex[1] + YSIZE_SUBDOMAIN * cellindex[2]); + const int entry = cellindex[0] + XSIZE_SUBDOMAIN * (cellindex[1] + YSIZE_SUBDOMAIN * cellindex[2]); - rho[entry] += 1; + rho[entry] += 1; - for(int c = 0; c < 3; ++c) - u[c][entry] += p[i].u[c]; + for(int c = 0; c < 3; ++c) + u[c][entry] += p[i].u[c]; } for(int c = 0; c < 3; ++c) - for(int i = 0; i < ncells; ++i) - u[c][i] = rho[i] ? u[c][i] / rho[i] : 0; + for(int i = 0; i < ncells; ++i) + u[c][i] = rho[i] ? u[c][i] / rho[i] : 0; const char * names[] = { "density", "u", "v", "w" }; if (!directory_exists) { - int rank; - MPI_CHECK(MPI_Comm_rank(comm, &rank)); - - if (rank == 0) - mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + int rank; + MPI_CHECK(MPI_Comm_rank(comm, &rank)); + + if (rank == 0) + mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); - directory_exists = true; + directory_exists = true; - MPI_CHECK(MPI_Barrier(comm)); + MPI_CHECK(MPI_Barrier(comm)); } char filepath[512]; @@ -450,7 +688,7 @@ H5FieldDump::~H5FieldDump() { #ifndef NO_H5 if (last_idtimestep == 0) - return; + return; FILE * xmf = fopen("h5/flowfields-sequence.xmf", "w"); @@ -463,10 +701,10 @@ H5FieldDump::~H5FieldDump() const char * channelnames[] = { "density", "u", "v", "w" }; for(int it = 0; it <= last_idtimestep; it += steps_per_dump) { - char filepath[512]; - sprintf(filepath, "h5/flowfields-%04d.h5", it / steps_per_dump); + char filepath[512]; + sprintf(filepath, "h5/flowfields-%04d.h5", it / steps_per_dump); - _xdmf_grid(xmf, it * dt, string(filepath).substr(string(filepath).find_last_of("/") + 1).c_str(), channelnames, 4); + _xdmf_grid(xmf, it * dt, string(filepath).substr(string(filepath).find_last_of("/") + 1).c_str(), channelnames, 4); } fprintf(xmf, " \n"); @@ -479,3 +717,4 @@ H5FieldDump::~H5FieldDump() } bool H5FieldDump::directory_exists = false; + diff --git a/mpi-dpd/io.h b/mpi-dpd/io.h index a970f5983..579c6421e 100644 --- a/mpi-dpd/io.h +++ b/mpi-dpd/io.h @@ -16,7 +16,6 @@ #include "common.h" - void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const char * particlename, Particle * particles, int n, bool append); void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 4c5acd799..6b4f2b9ed 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -21,6 +21,7 @@ #include "argument-parser.h" #include "simulation.h" +#include "dumper.h" bool currently_profiling = false; float tend; @@ -35,21 +36,21 @@ namespace SignalHandling void signal_handler(int signum) { - graceful_exit = 1; - graceful_signum = signum; + graceful_exit = 1; + graceful_signum = signum; } void setup() { - struct sigaction action; - memset(&action, 0, sizeof(struct sigaction)); - action.sa_handler = signal_handler; - sigaction(SIGUSR1, &action, NULL); + struct sigaction action; + memset(&action, 0, sizeof(struct sigaction)); + action.sa_handler = signal_handler; + sigaction(SIGUSR1, &action, NULL); } bool check_termination_request() { - return graceful_exit; + return graceful_exit; } } @@ -60,12 +61,12 @@ int main(int argc, char ** argv) //parsing of the positional arguments if (argc < 4) { - printf("usage: ./mpi-dpd \n"); - exit(-1); + printf("usage: ./mpi-dpd \n"); + exit(-1); } else - for(int i = 0; i < 3; ++i) - ranks[i] = atoi(argv[1 + i]); + for(int i = 0; i < 3; ++i) + ranks[i] = atoi(argv[1 + i]); ArgumentParser argp(vector(argv + 4, argv + argc)); @@ -103,55 +104,42 @@ int main(int argc, char ** argv) CUDA_CHECK(cudaDeviceReset()); { - is_mps_enabled = false; + is_mps_enabled = false; - const char * mps_variables[] = { - "CRAY_CUDA_MPS", - "CUDA_MPS", - "CRAY_CUDA_PROXY", - "CUDA_PROXY" - }; + const char * mps_variables[] = { + "CRAY_CUDA_MPS", + "CUDA_MPS", + "CRAY_CUDA_PROXY", + "CUDA_PROXY" + }; - for(int i = 0; i < 4; ++i) - is_mps_enabled |= getenv(mps_variables[i])!= NULL && atoi(getenv(mps_variables[i])) != 0; + for(int i = 0; i < 4; ++i) + is_mps_enabled |= getenv(mps_variables[i])!= NULL && atoi(getenv(mps_variables[i])) != 0; } int nranks, rank; - if (mpi_thread_safe) - { - //needed for the asynchronous data dumps - setenv("MPICH_MAX_THREAD_SAFETY", "multiple", 0); - - int provided_safety_level; - MPI_CHECK( MPI_Init_thread(&argc, &argv, MPI_THREAD_MULTIPLE, &provided_safety_level)); - - MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); - MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); - - if (provided_safety_level != MPI_THREAD_MULTIPLE) - { - if (rank == 0) - printf("ooooooooops MPI thread safety level is just %d. Aborting now.\n", provided_safety_level); - abort(); - } - else - if (rank == 0) - printf("I have set MPICH_MAX_THREAD_SAFETY=multiple\n"); - } + + MPI_CHECK(MPI_Init(&argc, &argv)); + MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); + MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); + + MPI_Comm iocomm, activecomm, intercomm, splitcomm; + + assert(nranks & 0x1 == 0); + int computeTask = (rank+1) % 2; + MPI_CHECK( MPI_Comm_split(MPI_COMM_WORLD, computeTask, rank, &splitcomm) ); + if (computeTask) + MPI_CHECK( MPI_Comm_dup(splitcomm, &activecomm) ); else - { - MPI_CHECK(MPI_Init(&argc, &argv)); - MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); - MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); + MPI_CHECK( MPI_Comm_dup(splitcomm, &iocomm) ); - const char * env_thread_safety = getenv("MPICH_MAX_THREAD_SAFETY"); + if (computeTask) + MPI_CHECK( MPI_Intercomm_create(activecomm, 0, MPI_COMM_WORLD, 1, 0, &intercomm) ); + else + MPI_CHECK( MPI_Intercomm_create(iocomm, 0, MPI_COMM_WORLD, 0, 0, &intercomm) ); - if (rank == 0 && env_thread_safety) - printf("I read MPICH_MAX_THREAD_SAFETY=%s", env_thread_safety); - } - MPI_Comm activecomm = MPI_COMM_WORLD; #if defined(CUSTOM_REORDERING) activecomm = setup_reorder_comm(MPI_COMM_WORLD, rank, nranks); #endif @@ -161,92 +149,115 @@ int main(int argc, char ** argv) const char * env_reorder = getenv("MPICH_RANK_REORDER_METHOD"); //reordering of the ranks according to the computational domain and environment variables - if (atoi(env_reorder ? env_reorder : "-1") == atoi("3")) + if (computeTask && atoi(env_reorder ? env_reorder : "-1") == atoi("3")) { - reordering = false; + reordering = false; - const bool usefulrank = rank < ranks[0] * ranks[1] * ranks[2]; + const bool usefulrank = rank < ranks[0] * ranks[1] * ranks[2]; - MPI_CHECK(MPI_Comm_split(MPI_COMM_WORLD, usefulrank, rank, &activecomm)) ; + MPI_CHECK(MPI_Comm_split(MPI_COMM_WORLD, usefulrank, rank, &activecomm)) ; - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(activecomm)); - if (!usefulrank) - { - printf("rank %d has been thrown away\n", rank); - fflush(stdout); + if (!usefulrank) + { + printf("rank %d has been thrown away\n", rank); + fflush(stdout); - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(activecomm)); - MPI_Finalize(); + MPI_Finalize(); - return 0; - } + return 0; + } - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(MPI_COMM_WORLD)); } - MPI_Comm cartcomm; + MPI_Comm cartcomm, iocartcomm; int periods[] = {1, 1, 1}; - MPI_CHECK( MPI_Cart_create(activecomm, 3, ranks, periods, (int)reordering, &cartcomm) ); + if (computeTask) + MPI_CHECK( MPI_Cart_create(activecomm, 3, ranks, periods, (int)reordering, &cartcomm) ); + else + MPI_CHECK( MPI_Cart_create(iocomm, 3, ranks, periods, (int)reordering, &iocartcomm) ); activecomm = cartcomm; //print the rank-to-node mapping + if (computeTask) { - char name[1024]; - int len; - MPI_CHECK(MPI_Get_processor_name(name, &len)); + char name[1024]; + int len; + MPI_CHECK(MPI_Get_processor_name(name, &len)); - int dims[3], periods[3], coords[3]; - MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(activecomm)); #if defined(REPORT_TOPOLOGY) - int nid; - int rc = PMI_Get_nid(rank, &nid); - pmi_mesh_coord_t xyz; - PMI_Get_meshcoord((uint16_t) nid, &xyz); - printf("RANK %d: (%d, %d, %d) -> %s (%d, %d, %d)\n", rank, coords[0], coords[1], coords[2], name, xyz.mesh_x, xyz.mesh_y, xyz.mesh_z); + int nid; + int rc = PMI_Get_nid(rank, &nid); + pmi_mesh_coord_t xyz; + PMI_Get_meshcoord((uint16_t) nid, &xyz); + printf("RANK %d: (%d, %d, %d) -> %s (%d, %d, %d)\n", rank, coords[0], coords[1], coords[2], name, xyz.mesh_x, xyz.mesh_y, xyz.mesh_z); #else - printf("RANK %d: (%d, %d, %d) -> %s\n", rank, coords[0], coords[1], coords[2], name); + printf("RANK %d: (%d, %d, %d) -> %s\n", rank, coords[0], coords[1], coords[2], name); #endif - fflush(stdout); + fflush(stdout); - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(activecomm)); } //RAII { - MPI_CHECK(MPI_Barrier(activecomm)); - - if (rank == 0) - { - argp.print_arguments(); - fflush(stdout); - } - - localcomm.initialize(activecomm); - - MPI_CHECK(MPI_Barrier(activecomm)); - - Simulation simulation(cartcomm, activecomm, SignalHandling::check_termination_request); - - simulation.run(); + if (computeTask) + { + MPI_CHECK(MPI_Barrier(activecomm)); + + if (rank == 0) + { + argp.print_arguments(); + fflush(stdout); + } + + localcomm.initialize(activecomm); + + MPI_CHECK(MPI_Barrier(activecomm)); + + Simulation simulation(cartcomm, activecomm, intercomm, SignalHandling::check_termination_request); + simulation.run(); + } + else + { + Dumper dumper(iocomm, iocartcomm, intercomm); + dumper.do_dump(); + } } - if (activecomm != cartcomm) - MPI_CHECK(MPI_Comm_free(&activecomm)); + if (computeTask) + { + if (activecomm != cartcomm) + MPI_CHECK(MPI_Comm_free(&activecomm)); - MPI_CHECK(MPI_Comm_free(&cartcomm)); + MPI_CHECK(MPI_Comm_free(&cartcomm)); + MPI_CHECK(MPI_Comm_free(&intercomm)); + } + else + { + MPI_CHECK(MPI_Comm_free(&iocomm)); + MPI_CHECK(MPI_Comm_free(&intercomm)); + } MPI_CHECK(MPI_Finalize()); - CUDA_CHECK(cudaDeviceSynchronize()); + if (computeTask) + { + CUDA_CHECK(cudaDeviceSynchronize()); - CUDA_CHECK(cudaDeviceReset()); + CUDA_CHECK(cudaDeviceReset()); + } return 0; } diff --git a/mpi-dpd/redistribute-rbcs.cu b/mpi-dpd/redistribute-rbcs.cu index cb3998844..2129112d5 100644 --- a/mpi-dpd/redistribute-rbcs.cu +++ b/mpi-dpd/redistribute-rbcs.cu @@ -109,8 +109,8 @@ namespace ReorderingRBC ddestinations[idrbc][offset] = val; } - SimpleDeviceBuffer _ddestinations; - SimpleDeviceBuffer _dsources; + SimpleDeviceBuffer *_ddestinations = new SimpleDeviceBuffer; + SimpleDeviceBuffer *_dsources = new SimpleDeviceBuffer; void pack_all(cudaStream_t stream, const int nrbcs, const int nvertices, const float ** const sources, float ** const destinations) { @@ -128,13 +128,13 @@ namespace ReorderingRBC } else { - _ddestinations.resize(nrbcs); - _dsources.resize(nrbcs); + _ddestinations->resize(nrbcs); + _dsources->resize(nrbcs); - CUDA_CHECK(cudaMemcpyAsync(_ddestinations.data, destinations, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); - CUDA_CHECK(cudaMemcpyAsync(_dsources.data, sources, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemcpyAsync(_ddestinations->data, destinations, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemcpyAsync(_dsources->data, sources, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); - pack_all_kernel<<<(nthreads + 127) / 128, 128, 0, stream>>>(nrbcs, nvertices, _dsources.data, _ddestinations.data); + pack_all_kernel<<<(nthreads + 127) / 128, 128, 0, stream>>>(nrbcs, nvertices, _dsources->data, _ddestinations->data); } CUDA_CHECK(cudaPeekAtLastError()); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index eef2bb615..b98ea4c39 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -476,134 +476,10 @@ void Simulation::_forces(bool firsttime) CUDA_CHECK(cudaPeekAtLastError()); } -void Simulation::_qoi(Particle* rbcs, Particle * ctcs, const float tm) -{ - int dims[3], periods[3], coords[3]; - MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); - - const int subdomain[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; - const int nbins = 15; // number of rows - - vector locRBChisto(nbins, 0); - vector locCTChisto(nbins, 0); - float totcom[3] = {0, 0, 0}; - - const float rwidth = 56; - const float offset = 40; - const float tan_a = tan(1.7 / 180.0 * M_PI); - - if (rbcscoll) - { - for (int p = 0; p < rbcscoll->count(); p++) - { - float com[3] = {0, 0, 0}; - Particle * cur = rbcs + p * rbcscoll->get_nvertices(); - - for (int i=0; i < rbcscoll->get_nvertices(); i++) - for (int d = 0; d<3; d++) - { - totcom[d] += cur[i].x[d] + (coords[d] + 0.5) * subdomain[d]; - com[d] += cur[i].x[d]; - } - - for (int d = 0; d<3; d++) - com[d] = com[d] / rbcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; - - int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); - if (irow >= nbins) irow = nbins - 1; - if (irow < 0) irow = 0; - - locRBChisto[irow]++; - } - } - - if (ctcscoll) - for (int p = 0; p < ctcscoll->count(); p++) - { - float com[3] = {0, 0, 0}; - Particle * cur = ctcs + p * ctcscoll->get_nvertices(); - - for (int i=0; i < ctcscoll->get_nvertices(); i++) - for (int d = 0; d<3; d++) - com[d] += cur[i].x[d]; - - for (int d = 0; d<3; d++) - com[d] = com[d] / ctcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; - - int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); - if (irow >= nbins) irow = nbins - 1; - if (irow < 0) irow = 0; - - locCTChisto[irow]++; - } - - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, qoicomm) ); - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, qoicomm) ); - MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, qoicomm) ); - - - if (ctcscoll && ctcscoll->count() > 0) - { - float com[3] = {0, 0, 0}; - - Particle * cur = ctcs; - for (int i=0; i < ctcscoll->get_nvertices(); i++) - for (int d = 0; d<3; d++) - com[d] += cur[i].x[d]; - - for (int d = 0; d<3; d++) - com[d] = com[d] / ctcscoll->get_nvertices() + (coords[d] + 0.5) * subdomain[d]; - - FILE* f = fopen("ctccom.txt", qoiid == 0 ? "w" : "a"); - fprintf(f, "%f %e %e %e\n", tm, com[0], com[1], com[2]); - fclose(f); - } - - if (rank == 0) - { - if (rbcscoll) - { - FILE* fout = fopen("rbchisto.dat", qoiid == 0 ? "w" : "a"); - fprintf(fout, "\n %f\n", tm); - for (int i=0; iget_nvertices(); - - FILE* fout = fopen("rbccom.txt", qoiid == 0 ? "w" : "a"); - fprintf(fout, "%f %e %e %e\n", tm, totcom[0] / totrbcs, totcom[1] / totrbcs, totcom[2] / totrbcs); - fclose(fout); - } - } - qoiid++; -} - - void Simulation::_datadump(const int idtimestep) { double tstart = MPI_Wtime(); - pthread_mutex_lock(&mutex_datadump); - - while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); - int n = particles->size; if (rbcscoll) @@ -644,162 +520,22 @@ void Simulation::_datadump(const int idtimestep) } assert(start == n); - CUDA_CHECK(cudaEventRecord(evdownloaded, 0)); - datadump_idtimestep = idtimestep; datadump_nsolvent = particles->size; datadump_nrbcs = rbcscoll ? rbcscoll->pcount() : 0; datadump_nctcs = ctcscoll ? ctcscoll->pcount() : 0; - datadump_pending = true; - pthread_cond_signal(&request_datadump); -#if defined(_SYNC_DUMPS_) - while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); -#endif + MPI_CHECK( MPI_Send(&datadump_nsolvent, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nrbcs, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); - pthread_mutex_unlock(&mutex_datadump); + CUDA_CHECK( cudaEventSynchronize(evdownloaded) ); - timings["data-dump"] += MPI_Wtime() - tstart; -} - -void Simulation::_datadump_async() -{ -#ifdef _USE_NVTX_ - nvtxNameOsThread(pthread_self(), "DATADUMP_THREAD"); -#endif - - int iddatadump = 0, rank; - int curr_idtimestep = -1; - bool wallcreated = false; - - MPI_Comm myactivecomm, mycartcomm; - - MPI_CHECK(MPI_Comm_dup(activecomm, &myactivecomm) ); - MPI_CHECK(MPI_Comm_dup(cartcomm, &mycartcomm) ); - - H5PartDump dump_part("allparticles->h5part", activecomm, cartcomm), *dump_part_solvent = NULL; - H5FieldDump dump_field(cartcomm); + MPI_CHECK( MPI_Send(particles_datadump.data, n, Particle::datatype(), rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(accelerations_datadump.data, n, Acceleration::datatype(), rank, 0, intercomm) ); - MPI_CHECK(MPI_Comm_rank(myactivecomm, &rank)); - - if (rank == 0) - mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); - - MPI_CHECK(MPI_Barrier(myactivecomm)); - - while (true) - { - pthread_mutex_lock(&mutex_datadump); - async_thread_initialized = 1; - - while (!datadump_pending) - pthread_cond_wait(&request_datadump, &mutex_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - if (curr_idtimestep == datadump_idtimestep) - if (simulation_is_done) - break; - - CUDA_CHECK(cudaEventSynchronize(evdownloaded)); - - const int n = particles_datadump.size; - Particle * p = particles_datadump.data; - Acceleration * a = accelerations_datadump.data; - - { - NVTX_RANGE("diagnostics", NVTX_C1); - diagnostics(myactivecomm, mycartcomm, p, n, dt, datadump_idtimestep, a); - } - - if (xyz_dumps) - { - NVTX_RANGE("xyz dump", NVTX_C2); - - if (walls && datadump_idtimestep >= wall_creation_stepid && !wallcreated) - { - if (rank == 0) - { - if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); - - if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); - - if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); - } - - MPI_CHECK(MPI_Barrier(myactivecomm)); - - wallcreated = true; - } - - xyz_dump(myactivecomm, mycartcomm, "xyz/particles->xyz", "all-particles", p, n, datadump_idtimestep > 0); - } - - if (hdf5part_dumps) - { - NVTX_RANGE("h5part dump", NVTX_C3); - - if (!dump_part_solvent && walls && datadump_idtimestep >= wall_creation_stepid) - { - dump_part.close(); - - dump_part_solvent = new H5PartDump("solvent-particles->h5part", activecomm, cartcomm); - } - - if (dump_part_solvent) - dump_part_solvent->dump(p, n); - else - dump_part.dump(p, n); - } - - if (hdf5field_dumps) - { - NVTX_RANGE("hdf5 field dump", NVTX_C4); - - dump_field.dump(activecomm, p, datadump_nsolvent, datadump_idtimestep); - } - - { - NVTX_RANGE("ply dump", NVTX_C5); - - if (rbcscoll) - CollectionRBC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent, a + datadump_nsolvent, datadump_nrbcs, iddatadump); - - if (ctcscoll) - CollectionCTC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent + datadump_nrbcs, - a + datadump_nsolvent + datadump_nrbcs, datadump_nctcs, iddatadump); - } - - _qoi(p + datadump_nsolvent, p + datadump_nsolvent + datadump_nrbcs, curr_idtimestep * dt); - - curr_idtimestep = datadump_idtimestep; - - pthread_mutex_lock(&mutex_datadump); - - if (simulation_is_done) - { - pthread_mutex_unlock(&mutex_datadump); - break; - } - - datadump_pending = false; - - pthread_cond_signal(&done_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - ++iddatadump; - } - - if (dump_part_solvent) - delete dump_part_solvent; - - CUDA_CHECK(cudaEventDestroy(evdownloaded)); + timings["data-dump"] += MPI_Wtime() - tstart; } void Simulation::_update_and_bounce() @@ -842,8 +578,8 @@ void Simulation::_update_and_bounce() CUDA_CHECK(cudaPeekAtLastError()); } -Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_termination)()) : - cartcomm(cartcomm), activecomm(activecomm), +Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) : + cartcomm(cartcomm), activecomm(activecomm), intercomm(intercomm), /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), rbcscoll(NULL), ctcscoll(NULL), wall(NULL), redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), @@ -856,8 +592,6 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); - MPI_CHECK( MPI_Comm_dup(activecomm, &qoicomm) ); - solutex.attach_halocomputation(fsi); if (contactforces) @@ -909,37 +643,9 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term ctcscoll->setup("ctcs-ic.txt"); } -#ifndef _NO_DUMPS_ - //setting up the asynchronous data dumps - { - CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); - - particles_datadump.resize(particles->size * 1.5); - accelerations_datadump.resize(particles->size * 1.5); - - int rc = pthread_mutex_init(&mutex_datadump, NULL); - rc |= pthread_cond_init(&done_datadump, NULL); - rc |= pthread_cond_init(&request_datadump, NULL); - async_thread_initialized = 0; - rc |= pthread_create(&thread_datadump, NULL, datadump_trampoline, this); - - while (1) - { - pthread_mutex_lock(&mutex_datadump); - int done = async_thread_initialized; - pthread_mutex_unlock(&mutex_datadump); - - if (done) - break; - } - - if (rc) - { - printf("ERROR; return code from pthread_create() is %d\n", rc); - exit(-1); - } - } -#endif + CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); + particles_datadump.resize(particles->size * 1.5); + accelerations_datadump.resize(particles->size * 1.5); } void Simulation::_lockstep() @@ -1266,6 +972,11 @@ void Simulation::run() simulation_is_done = true; + datadump_nsolvent = datadump_nrbcs = datadump_nctcs = -1; + MPI_CHECK( MPI_Send(&datadump_nsolvent, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nrbcs, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); + if (rank == 0) if (it == nsteps) printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", @@ -1279,17 +990,6 @@ void Simulation::run() Simulation::~Simulation() { -#ifndef _NO_DUMPS_ - pthread_mutex_lock(&mutex_datadump); - - datadump_pending = true; - pthread_cond_signal(&request_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - pthread_join(thread_datadump, NULL); -#endif - CUDA_CHECK(cudaStreamDestroy(mainstream)); CUDA_CHECK(cudaStreamDestroy(uploadstream)); CUDA_CHECK(cudaStreamDestroy(downloadstream)); diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 22e3f8c0e..6ca0aa9c0 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -60,7 +60,7 @@ class Simulation bool (*check_termination)(); bool simulation_is_done; - MPI_Comm activecomm, cartcomm, qoicomm; + MPI_Comm activecomm, cartcomm, intercomm; //LocalComm localcomm; cudaStream_t mainstream, uploadstream, downloadstream; @@ -80,7 +80,6 @@ class Simulation void _create_walls(const bool verbose, bool & termination_request); void _remove_bodies_from_wall(CollectionRBC * coll); void _forces(bool firsttime = false); - void _qoi(Particle* rbcs, Particle * ctcs, const float tm); void _datadump(const int idtimestep); void _update_and_bounce(); void _lockstep(); @@ -101,11 +100,9 @@ class Simulation public: - Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_termination)()) ; + Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) ; void run(); ~Simulation(); - - static void * datadump_trampoline(void * x) { ((Simulation *)x)->_datadump_async(); return NULL; } }; diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index 2f97eb497..f7d0454db 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -516,9 +516,9 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) if (count > expected) MPI_CHECK( MPI_Recv(&remote[i].pmessage.front() + expected, (count - expected) * 6, MPI_FLOAT, dstranks[i], TAGBASE_P2 + recv_tags[i], cartcomm, &status) ); -#endif memcpy(remote[i].hstate.data, &remote[i].pmessage.front(), sizeof(Particle) * count); +#endif _not_nan((float*)remote[i].hstate.data, count * 6); } diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index 531ecfd7a..81ed3114a 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -246,6 +246,9 @@ namespace SolidWallsKernel //this is the worst case - it means that old position was bad already //we need to search and rescue the particle + cuda_printf("Warning rank %d sdf: %f (%.4f %.4f %.4f), from: sdf %f (%.4f %.4f %.4f)... ", + rank, currsdf, x, y, z, sdf(xold, yold, zold), xold, yold, zold); + const float3 mygrad = grad_sdf(x, y, z); const float mysdf = currsdf; @@ -261,6 +264,8 @@ namespace SolidWallsKernel v = -v; w = -w; + cuda_printf("rescued in %d steps\n", 9-l); + return; } @@ -718,7 +723,7 @@ struct FieldSampler int g[3]; for(int c = 0; c < 3; ++c) - g[c] = max(0, min(N[c] - 1, l[c] - 1 + anchor[c])); + g[c] = (l[c] - 1 + anchor[c] + N[c]) % N[c]; s += w[0][sx] * data[g[0] + N[0] * (g[1] + N[1] * g[2])]; } From 27518c590faee723c94a882501665a346d52bfe9 Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Fri, 25 Sep 2015 17:54:50 +0200 Subject: [PATCH 24/63] hints for hdf5 --- mpi-dpd/io.cu | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 2f427ff59..bf0e01066 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -555,7 +555,22 @@ void H5FieldDump::_write_fields(const char * const path2h5, MPI_CHECK( MPI_Cart_get(cartcomm, 3, nranks, periods, myrank) ); id_t plist_id_access = H5Pcreate(H5P_FILE_ACCESS); - H5Pset_fapl_mpio(plist_id_access, comm, MPI_INFO_NULL); + + int cb = 1; + while (cb*2 <= size) cb *= 2; + + cb = min(cb, 128); + char cbstr[100]; + sprintf(cbstr, "%d", cb); + + MPI_Info info; + MPI_Info_create(&info); + MPI_Info_set(info, "cb_nodes", cbstr); + MPI_Info_set(info, "romio_cb_write", "enable"); + MPI_Info_set(info, "romio_ds_write", "disable"); + MPI_Info_set(info, "striping_factor", cbstr); + + H5Pset_fapl_mpio(plist_id_access, comm, info); hid_t file_id = H5Fcreate(path2h5, H5F_ACC_TRUNC, H5P_DEFAULT, plist_id_access); H5Pclose(plist_id_access); From 00d522182ad007924623cd04a2264483eeb295a3 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Wed, 30 Sep 2015 16:32:51 +0200 Subject: [PATCH 25/63] stress computation and data dump in raw format --- cuda-dpd/dpd/Makefile | 7 +++-- cuda-dpd/dpd/cuda-dpd.h | 14 +++++++-- mpi-dpd/common.h | 2 +- mpi-dpd/dpd.cu | 68 ++++++++++++++++++++++++++++++++++------- mpi-dpd/dpd.h | 35 ++++++++++++++++++--- mpi-dpd/io.cu | 43 ++++++++++++++++++++++++++ mpi-dpd/io.h | 5 +++ mpi-dpd/main.cu | 4 ++- mpi-dpd/simulation.cu | 54 ++++++++++++++++++++++++++++++++ mpi-dpd/simulation.h | 2 ++ 10 files changed, 212 insertions(+), 22 deletions(-) diff --git a/cuda-dpd/dpd/Makefile b/cuda-dpd/dpd/Makefile index 81c92431f..3a8d0cc06 100644 --- a/cuda-dpd/dpd/Makefile +++ b/cuda-dpd/dpd/Makefile @@ -35,8 +35,8 @@ NVCCFLAGS += -DVISCOSITY_S_LEVEL=$(slevel) -lineinfo -Xptxas -v test-dpd: main.cpp libcuda-dpd.so $(CXX) $(CXXFLAGS) $^ -lcudart -lcurand -o test-dpd -libcuda-dpd.a: cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o celllists - ar rcs $@ cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o +libcuda-dpd.a: cuda-dpd.o cuda-dpd-bipartite.o stress.o ../profiler-dpd.o celllists + ar rcs $@ cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o stress.o libcuda-dpd.so: cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o celllists $(CXX) $(CXXFLAGS) -shared cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o -o libcuda-dpd.so -lcudart -lcurand @@ -48,6 +48,9 @@ cuda-dpd.o: $(CUDADPD) cuda-dpd.h cuda-dpd-bipartite.o: $(CUDADPDBIP) cuda-dpd.h $(NVCC) $(NVCCFLAGS) -c $(CUDADPDBIP) -o $@ +stress.o: stress.cu cuda-dpd.h + $(NVCC) $(NVCCFLAGS) -c $< -o $@ + ../%.o: make -C ../ $(@:../%=%) CXX="$(CXX)" NVCC="$(NVCC)" diff --git a/cuda-dpd/dpd/cuda-dpd.h b/cuda-dpd/dpd/cuda-dpd.h index dfeca681e..174870b52 100644 --- a/cuda-dpd/dpd/cuda-dpd.h +++ b/cuda-dpd/dpd/cuda-dpd.h @@ -29,7 +29,7 @@ template<> inline __device__ float viscosity_function<0>(float x){ return x; } void forces_dpd_cuda_nohost(const float * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, float * const _axayaz, const int np, - const int * const cellsstart, const int * const cellscount, + const int * const cellsstart, const int * const cellscount, const float rc, const float XL, const float YL, const float ZL, const float aij, @@ -37,7 +37,7 @@ void forces_dpd_cuda_nohost(const float * const xyzuvw, const float4 * const xyz const float sigma, const float invsqrtdt, const float seed1, - cudaStream_t stream); + cudaStream_t stream); void forces_dpd_cuda(const float * const xp, const float * const yp, const float * const zp, const float * const xv, const float * const yv, const float * const zv, @@ -62,3 +62,13 @@ void forces_dpd_cuda_bipartite_nohost(cudaStream_t stream, const float2 * const const int3 halo_ncells, const float aij, const float gamma, const float sigmaf, const float seed, const int mask, float * const axayaz); + +void compute_stress(const float * const xyzuvw, + const int np, + const int * const cellsstart, const int * const cellscount, + const int XL, const int YL, const int ZL, + const float aij, const float gamma, const float sigmaf, const float seed, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, + float * const sigma_yy, float * const sigma_yz, float * const sigma_zz, + float * const axayaz, + cudaStream_t stream); diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index 481592ee6..51c629846 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -38,7 +38,7 @@ const float aij = 25; const float hydrostatic_a = 0.05; extern float tend; -extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces; +extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces, stress; extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; #include diff --git a/mpi-dpd/dpd.cu b/mpi-dpd/dpd.cu index c573642b8..b8edcfe03 100644 --- a/mpi-dpd/dpd.cu +++ b/mpi-dpd/dpd.cu @@ -20,7 +20,9 @@ using namespace std; -ComputeDPD::ComputeDPD(MPI_Comm cartcomm): SolventExchange(cartcomm, 0), local_trunk(0, 0, 0, 0) +ComputeDPD::ComputeDPD(MPI_Comm cartcomm): +SolventExchange(cartcomm, 0), local_trunk(0, 0, 0, 0), +sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), sigma_zz(NULL) { int myrank; MPI_CHECK(MPI_Comm_rank(cartcomm, &myrank)); @@ -73,16 +75,25 @@ ComputeDPD::ComputeDPD(MPI_Comm cartcomm): SolventExchange(cartcomm, 0), local_t } } + + void ComputeDPD::local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, Acceleration * const a, const int * const cellsstart, const int * const cellscount, cudaStream_t stream) { NVTX_RANGE("DPD/local", NVTX_C5); if (n > 0) + { forces_dpd_cuda_nohost((float*)xyzuvw, xyzouvwo, xyzo_half, (float *)a, n, cellsstart, cellscount, 1, XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN, aij, gammadpd, - sigma, 1. / sqrt(dt), local_trunk.get_float(), stream); + sigma, 1. / sqrt(dt), current_lseed = local_trunk.get_float(), stream); + + if (sigma_xx) + compute_stress((float *)xyzuvw, n, cellsstart, cellscount, XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN, + aij, gammadpd, sigmaf, current_lseed, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, (float *)a, stream); + } } namespace BipsBatch @@ -102,9 +113,17 @@ namespace BipsBatch __constant__ BatchInfo batchinfos[26]; - __global__ void - interaction_kernel(const float aij, const float gamma, const float sigmaf, - const int ndstall, float * const adst, const int sizeadst) + struct StressInfo + { + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz; + }; + + __constant__ StressInfo stressinfo; + + + template < bool computestresses > __global__ + void interaction_kernel(const float aij, const float gamma, const float sigmaf, + const int ndstall, float * const adst, const int sizeadst) { #if !defined(__CUDA_ARCH__) #warning __CUDA_ARCH__ not defined! assuming 350 @@ -151,7 +170,8 @@ namespace BipsBatch const float vp = info.xdst[4 + dpid * 6]; const float wp = info.xdst[5 + dpid * 6]; - const int dstbase = 3 * info.scattered_entries[dpid]; + const int dstentry = info.scattered_entries[dpid]; + const int dstbase = 3 * dstentry; assert(dstbase < sizeadst * 3); uint scan1, scan2, ncandidates, spidbase; @@ -281,6 +301,16 @@ namespace BipsBatch xforce += strength * xr; yforce += strength * yr; zforce += strength * zr; + + if (computestresses) + { + atomicAdd(stressinfo.sigma_xx + dstentry, strength * xr * _xr); + atomicAdd(stressinfo.sigma_xy + dstentry, strength * xr * _yr); + atomicAdd(stressinfo.sigma_xz + dstentry, strength * xr * _zr); + atomicAdd(stressinfo.sigma_yy + dstentry, strength * yr * _yr); + atomicAdd(stressinfo.sigma_yz + dstentry, strength * yr * _zr); + atomicAdd(stressinfo.sigma_zz + dstentry, strength * zr * _zr); + } } atomicAdd(adst + dstbase + 0, xforce); @@ -295,12 +325,15 @@ namespace BipsBatch cudaEvent_t evhalodone; void interactions(const float aij, const float gamma, const float sigma, const float invsqrtdt, - const BatchInfo infos[20], cudaStream_t computestream, cudaStream_t uploadstream, float * const acc, const int n) + const BatchInfo infos[20], cudaStream_t computestream, cudaStream_t uploadstream, float * const acc, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, float * const sigma_yy, + float * const sigma_yz, float * const sigma_zz, const int n) { if (firstcall) { CUDA_CHECK(cudaEventCreate(&evhalodone, cudaEventDisableTiming)); - CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); firstcall = false; } @@ -316,12 +349,24 @@ namespace BipsBatch const int nthreads = 2 * hstart_padded[26]; + if (sigma_xx) + { + StressInfo strinfo = {sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, uploadstream)); + } + CUDA_CHECK(cudaEventRecord(evhalodone, uploadstream)); CUDA_CHECK(cudaStreamWaitEvent(computestream, evhalodone, 0)); if (nthreads) - interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + { + if (sigma_xx) + interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + else + interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + } CUDA_CHECK(cudaPeekAtLastError()); } @@ -346,7 +391,7 @@ void ComputeDPD::remote_interactions(const Particle * const p, const int n, Acce const int m2 = 0 == dz; BipsBatch::BatchInfo entry = { - (float *)sendhalos[i].dbuf.data, (float2 *)recvhalos[i].dbuf.data, interrank_trunks[i].get_float(), + (float *)sendhalos[i].dbuf.data, (float2 *)recvhalos[i].dbuf.data, current_rseeds[i] = interrank_trunks[i].get_float(), sendhalos[i].dbuf.size, recvhalos[i].dbuf.size, interrank_masks[i], recvhalos[i].dcellstarts.data, sendhalos[i].scattered_entries.data, dx, dy, dz, @@ -357,7 +402,8 @@ void ComputeDPD::remote_interactions(const Particle * const p, const int n, Acce infos[i] = entry; } - BipsBatch::interactions(aij, gammadpd, sigma, 1. / sqrt(dt), infos, stream, uploadstream, (float *)a, n); + BipsBatch::interactions(aij, gammadpd, sigma, 1. / sqrt(dt), infos, stream, uploadstream, (float *)a, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, n); CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/dpd.h b/mpi-dpd/dpd.h index 290fc7c49..059ac0999 100644 --- a/mpi-dpd/dpd.h +++ b/mpi-dpd/dpd.h @@ -25,18 +25,43 @@ //see the vanilla version of this code for details about how this class operates class ComputeDPD : public SolventExchange -{ +{ Logistic::KISS local_trunk; Logistic::KISS interrank_trunks[26]; + float current_lseed, current_rseeds[26], + * sigma_xx, * sigma_xy, * sigma_xz, * sigma_yy, + * sigma_yz, * sigma_zz; + bool interrank_masks[26]; - + public: - + ComputeDPD(MPI_Comm cartcomm); + void set_stress_buffers(float * const stress_xx, float * const stress_xy, float * const stress_xz, float * const stress_yy, + float * const stress_yz, float * const stress_zz) + { + sigma_xx = stress_xx; + sigma_xy = stress_xy; + sigma_xz = stress_xz; + sigma_yy = stress_yy; + sigma_yz = stress_yz; + sigma_zz = stress_zz; + } + + void clr_stress_buffers() + { + sigma_xx = NULL; + sigma_xy = NULL; + sigma_xz = NULL; + sigma_yy = NULL; + sigma_yz = NULL; + sigma_zz = NULL; + } + void remote_interactions(const Particle * const p, const int n, Acceleration * const a, cudaStream_t stream, cudaStream_t uploadstream); - void local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, Acceleration * const a, - const int * const cellsstart, const int * const cellscount, cudaStream_t stream); + void local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, + Acceleration * const a, const int * const cellsstart, const int * const cellscount, cudaStream_t stream); }; diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 11b963c29..24e659a9e 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -181,6 +181,49 @@ void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, MPI_CHECK( MPI_File_close(&f)); } +void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, + const Particle * const particles, + const float * const stress_xx, const float * const stress_xy, const float * const stress_xz, + const float * const stress_yy, const float * const stress_yz, const float * const stress_zz) +{ + std::vector buf(nparticles * 9); + + int rank; + MPI_CHECK( MPI_Comm_rank(cartcomm, &rank) ); + + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + + int NALL = 0; + const int n = nparticles; + MPI_CHECK( MPI_Reduce(&n, &NALL, 1, MPI_INT, MPI_SUM, 0, cartcomm) ); + + MPI_File f; + MPI_CHECK( MPI_File_open(cartcomm, filename , MPI_MODE_WRONLY | MPI_MODE_CREATE, MPI_INFO_NULL, &f) ); + + MPI_CHECK( MPI_File_set_size (f, sizeof(float) * 9 * NALL )); + + const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; + + for(int i = 0; i < n; ++i) + { + for(int c = 0; c < 3; ++c) + buf[9 * i + c] = particles[i].x[c] + L[c] / 2 + coords[c] * L[c]; + + buf[9 * i + 3] = stress_xx[i]; + buf[9 * i + 4] = stress_xy[i]; + buf[9 * i + 5] = stress_xz[i]; + buf[9 * i + 6] = stress_yy[i]; + buf[9 * i + 7] = stress_yz[i]; + buf[9 * i + 8] = stress_zz[i]; + } + + _write_bytes(buf.data(), sizeof(float) * 9 * n, f, cartcomm); + + MPI_CHECK( MPI_File_close(&f)); +} + + H5PartDump::H5PartDump(const string fname, MPI_Comm comm, MPI_Comm cartcomm): tstamp(0), disposed(false) { _initialize(fname, comm, cartcomm); diff --git a/mpi-dpd/io.h b/mpi-dpd/io.h index a970f5983..b103b2743 100644 --- a/mpi-dpd/io.h +++ b/mpi-dpd/io.h @@ -23,6 +23,11 @@ void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, Particle * _particles, int nvertices_per_instance, bool append); +void stress_dump(MPI_Comm comm, const char * filename, const int nparticles, + const Particle * const particles, + const float * const stress_xx, const float * const stress_xy, const float * const stress_xz, + const float * const stress_yy, const float * const stress_yz, const float * const stress_zz); + class H5PartDump { float origin[3]; diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 3e5a7e5f4..7c74c688a 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -24,7 +24,8 @@ bool currently_profiling = false; float tend; -bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces; +bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, + hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces, stress; int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; LocalComm localcomm; @@ -84,6 +85,7 @@ int main(int argc, char ** argv) nvtxstop = argp("-nvtxstop").asInt(10500); adjust_message_sizes = argp("-adjust_message_sizes").asBool(false); contactforces = argp("-contactforces").asBool(false); + stress = argp("-stress").asBool(false); #ifndef _NO_DUMPS_ const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(true); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 4ecf0f4a5..1ad47a74b 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -441,6 +441,15 @@ void Simulation::_datadump(const int idtimestep) int n = particles->size; + if (stress) + { + for(int c = 0; c < 6; ++c) + stresses_datadump[c].resize(n); + + for(int c = 0; c < 6; ++c) + CUDA_CHECK(cudaMemcpyAsync(stresses_datadump[c].data, stresses[c].data, sizeof(float) * n, cudaMemcpyDeviceToHost,0)); + } + if (rbcscoll) n += rbcscoll->pcount(); @@ -568,6 +577,40 @@ void Simulation::_datadump_async() xyz_dump(myactivecomm, mycartcomm, "xyz/particles->xyz", "all-particles", p, n, datadump_idtimestep > 0); } + if (stress) + { + char filename[1024]; + sprintf(filename, "stress/stresses-%05d.data", iddatadump); + + if(rank == 0 && iddatadump == 0) + mkdir("stress", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + + const int nsolvent = stresses_datadump[0].size; + + stress_dump(mycartcomm, filename, stresses_datadump[0].size, p, + stresses_datadump[0].data, stresses_datadump[1].data, stresses_datadump[2].data, + stresses_datadump[3].data, stresses_datadump[4].data, stresses_datadump[5].data); + + //lets quickly compute the average stress in the system + float avgstress[6] = {0, 0, 0, 0, 0, 0}; + + for(int c = 0; c < 6; ++c) + for(int i = 0; i < nsolvent; ++i) + avgstress[c] += stresses_datadump[c].data[i]; + + int ntotsolvent; + MPI_CHECK( MPI_Reduce(&nsolvent, &ntotsolvent, 1, MPI_INT, MPI_SUM, 0, myactivecomm)); + + float totavgstress[6]; + MPI_CHECK( MPI_Reduce(avgstress, totavgstress, 6, MPI_FLOAT, MPI_SUM, 0, myactivecomm)); + + for(int c = 0; c < 6; ++c) + totavgstress[c] /= ntotsolvent; + + printf("average stress: sxx:%.3e sxy:%.3e sxz:%.3e syy:%.3e syz:%.3e szz:%.3e\n", + totavgstress[0], totavgstress[1], totavgstress[2], totavgstress[3], totavgstress[4], totavgstress[5]); + } + if (hdf5part_dumps) { NVTX_RANGE("h5part dump", NVTX_C3); @@ -1025,8 +1068,19 @@ void Simulation::run() printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); } + if(stress && it % steps_per_dump == 0) + { + for(int c = 0; c < 6; ++c) + stresses[c].resize(particles->size); + + dpd.set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + } + _forces(); + if (stress && it % steps_per_dump == 0 ) + dpd.clr_stress_buffers(); + #ifndef _NO_DUMPS_ if (it % steps_per_dump == 0) _datadump(it); diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 5a097d522..135ad26c8 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -41,6 +41,7 @@ class Simulation ParticleArray * particles, * newparticles; SimpleDeviceBuffer xyzouvwo; SimpleDeviceBuffer xyzo_half; + SimpleDeviceBuffer stresses[6]; CellLists cells; CollectionRBC * rbcscoll; @@ -93,6 +94,7 @@ class Simulation PinnedHostBuffer particles_datadump; PinnedHostBuffer accelerations_datadump; + PinnedHostBuffer stresses_datadump[6]; cudaEvent_t evdownloaded; From 1ba53748102df21cbda4bd6b303b56dd74a29742 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Wed, 30 Sep 2015 16:37:39 +0200 Subject: [PATCH 26/63] forgot to add stress.cu --- cuda-dpd/dpd/stress.cu | 264 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 264 insertions(+) create mode 100644 cuda-dpd/dpd/stress.cu diff --git a/cuda-dpd/dpd/stress.cu b/cuda-dpd/dpd/stress.cu new file mode 100644 index 000000000..f4160e071 --- /dev/null +++ b/cuda-dpd/dpd/stress.cu @@ -0,0 +1,264 @@ +/* + * stress.cu + * Part of uDeviceX/cuda-dpd-sem/dpd/ + * + * Created and authored by Diego Rossinelli on 2015-09-29. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include + +#include "cuda-dpd.h" +#include "../dpd-rng.h" +#include "../hacks.h" + +namespace StressKernels +{ + struct InfoStress + { + int3 ncells; + float aij, gamma, sigmaf; + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz, *axayaz; + float seed; + }; + + __constant__ InfoStress info; + + texture texParticles; + texture texStart, texCount; + +#define _XCPB_ 2 +#define _YCPB_ 2 +#define _ZCPB_ 1 +#define CPB (_XCPB_ * _YCPB_ * _ZCPB_) + + __device__ float3 _dpd_interaction(const int dpid, const float3 xdest, const float3 udest, const int spid, const float2 stmp0, const float2 stmp1) + { + const int sentry = 3 * spid; + const float2 stmp2 = tex1Dfetch(texParticles, sentry + 2); + + const float _xr = xdest.x - stmp0.x; + const float _yr = xdest.y - stmp0.y; + const float _zr = xdest.z - stmp1.x; + const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; + assert(rij2 < 1); + + const float invrij = rsqrtf(rij2); + const float rij = rij2 * invrij; + const float argwr = 1 - rij; + const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(argwr); + + const float xr = _xr * invrij; + const float yr = _yr * invrij; + const float zr = _zr * invrij; + + const float rdotv = + xr * (udest.x - stmp1.y) + + yr * (udest.y - stmp2.x) + + zr * (udest.z - stmp2.y); + + const float myrandnr = Logistic::mean0var1(info.seed, min(spid, dpid), max(spid, dpid)); + + const float strength = info.aij * argwr - (info.gamma * wr * rdotv + info.sigmaf * myrandnr) * wr; + + return make_float3(strength * xr, strength * yr, strength * zr); + } + +#define __IMOD(x,y) ((x)-((x)/(y))*(y)) + + template + __global__ void stress_kernel() + { + int mycount = 0, myscan = 0; + + __shared__ int volatile starts[CPB][16], scan[CPB][16]; + + if (threadIdx.x < 14) + { + const int cbase = blockIdx.x * blockDim.y + threadIdx.y; + + int dx, dy, dz; + dx = dy = dz = threadIdx.x / 3; + dx = threadIdx.x - dx * 3 - 1; + dy = __IMOD(dy, 3) - 1; + dz = __IMOD(dz / 3, 3) - 1; + + int cid = cbase + dz * info.ncells.x * info.ncells.y + dy * info.ncells.x + dx; + + const bool valid_cid = (cid >= 0) && (cid < info.ncells.x * info.ncells.y * info.ncells.z); + + starts[threadIdx.y][threadIdx.x] = (valid_cid) ? tex1Dfetch(texStart, cid) : 0; + + myscan = mycount = (valid_cid) ? tex1Dfetch(texCount, cid) : 0; + } + +#pragma unroll + for(int L = 1; L < 16; L <<= 1) + myscan += (threadIdx.x >= L) * __shfl_up(myscan, L); + + if (threadIdx.x < 15) + scan[threadIdx.y][threadIdx.x] = myscan - mycount; + + const int subtid = threadIdx.x % COLS; + const int slot = threadIdx.x / COLS; + + const int dststart = starts[threadIdx.y][13]; + const int lastdst = dststart + scan[threadIdx.y][14] - scan[threadIdx.y][13]; + + const int nsrc = scan[threadIdx.y][14]; + const int nsrcext = scan[threadIdx.y][13]; + + for(int pid = subtid; pid < nsrc; pid += COLS) + { + const int key9 = 9 * (pid >= scan[threadIdx.y][9]); + + int key3 = 3 * (pid >= scan[threadIdx.y][key9 + 3]); + key3 += (key9 < 9) ? 3 * (pid >= scan[threadIdx.y][key9 + 6]) : 0; + + int spid = pid - scan[threadIdx.y][key3 + key9] + starts[threadIdx.y][key3 + key9]; + + const int sentry = 3 * spid; + const float2 stmp0 = tex1Dfetch(texParticles, sentry); + const float2 stmp1 = tex1Dfetch(texParticles, sentry + 1); + + for(int dpid = dststart + slot; dpid < lastdst; dpid += ROWS) + { + float3 xdest, udest; + + float2 dtmp0 = tex1Dfetch(texParticles, 3 * dpid); + xdest.x = dtmp0.x; + xdest.y = dtmp0.y; + + dtmp0 = tex1Dfetch(texParticles, 3 * dpid + 1); + xdest.z = dtmp0.x; + udest.x = dtmp0.y; + + dtmp0 = tex1Dfetch(texParticles, 3 * dpid + 2); + udest.y = dtmp0.x; + udest.z = dtmp0.y; + + const float rx = xdest.x - stmp0.x; + const float ry = xdest.y - stmp0.y; + const float rz = xdest.z - stmp1.x; + + const float d2 = rx * rx + ry * ry + rz * rz; + + if ((dpid != spid) && (d2 < 1.0f)) + { + const float3 f = _dpd_interaction(dpid, xdest, udest, spid, stmp0, stmp1); + + atomicAdd(info.sigma_xx + dpid, f.x * rx); + atomicAdd(info.sigma_xy + dpid, f.x * ry); + atomicAdd(info.sigma_xz + dpid, f.x * rz); + atomicAdd(info.sigma_yy + dpid, f.y * ry); + atomicAdd(info.sigma_yz + dpid, f.y * rz); + atomicAdd(info.sigma_zz + dpid, f.z * rz); + + if (info.axayaz) + { + atomicAdd(info.axayaz + 3 * dpid , f.x); + atomicAdd(info.axayaz + 3 * dpid + 1, f.y); + atomicAdd(info.axayaz + 3 * dpid + 2, f.z); + } + + if (pid < nsrcext) + { + atomicAdd(info.sigma_xx + spid, f.x * rx); + atomicAdd(info.sigma_xy + spid, f.x * ry); + atomicAdd(info.sigma_xz + spid, f.x * rz); + atomicAdd(info.sigma_yy + spid, f.y * ry); + atomicAdd(info.sigma_yz + spid, f.y * rz); + atomicAdd(info.sigma_zz + spid, f.z * rz); + + if (info.axayaz) + { + atomicAdd(info.axayaz + 3*spid , -f.x); + atomicAdd(info.axayaz + 3*spid + 1, -f.y); + atomicAdd(info.axayaz + 3*spid + 2, -f.z); + } + } + } + } + } + } + + bool computestress_init = false; +} + +using namespace StressKernels; + +void compute_stress(const float * const xyzuvw, + const int np, + const int * const cellsstart, const int * const cellscount, + const int XL, const int YL, const int ZL, + const float aij, const float gamma, const float sigmaf, const float seed, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, + float * const sigma_yy, float * const sigma_yz, float * const sigma_zz, + float * const axayaz, + cudaStream_t stream) +{ + if (np == 0) + { + printf("WARNING: stress_nohost called with np = %d\n", np); + return; + } + + if (!computestress_init) + { + texStart.channelDesc = cudaCreateChannelDesc(); + texStart.filterMode = cudaFilterModePoint; + texStart.mipmapFilterMode = cudaFilterModePoint; + texStart.normalized = 0; + + texCount.channelDesc = cudaCreateChannelDesc(); + texCount.filterMode = cudaFilterModePoint; + texCount.mipmapFilterMode = cudaFilterModePoint; + texCount.normalized = 0; + + texParticles.channelDesc = cudaCreateChannelDesc(); + texParticles.filterMode = cudaFilterModePoint; + texParticles.mipmapFilterMode = cudaFilterModePoint; + texParticles.normalized = 0; + + CUDA_CHECK(cudaFuncSetCacheConfig(stress_kernel<32, 1>, cudaFuncCachePreferL1)); + + computestress_init = true; + } + + size_t textureoffset; + CUDA_CHECK(cudaBindTexture(&textureoffset, &texParticles, xyzuvw, &texParticles.channelDesc, sizeof(float) * 6 * np)); + assert(textureoffset == 0); + + const int ncells = XL * YL * ZL; + + CUDA_CHECK(cudaBindTexture(&textureoffset, &texStart, cellsstart, &texStart.channelDesc, sizeof(int) * ncells)); + assert(textureoffset == 0); + CUDA_CHECK(cudaBindTexture(&textureoffset, &texCount, cellscount, &texCount.channelDesc, sizeof(int) * ncells)); + assert(textureoffset == 0); + + { + static InfoStress c = { make_int3(XL, YL, ZL), aij, gamma, sigmaf, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, + axayaz, seed }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(info, &c, sizeof(c), 0, cudaMemcpyHostToDevice, stream)); + } + + if (axayaz) + CUDA_CHECK(cudaMemsetAsync(axayaz, 0, sizeof(float) * 3 * np, stream)); + + float * const ptrs[] = { sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + for(int c = 0; c < 6; ++c) + CUDA_CHECK(cudaMemsetAsync(ptrs[c], 0, sizeof(float) * np, stream)); + + stress_kernel<32, 1><<<(ncells + CPB - 1) / CPB, dim3(32, CPB), 0, stream>>>(); + + CUDA_CHECK(cudaPeekAtLastError()); +} + From ee2009c8b3888f27dd99b3f9e7ae6d309419a84e Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Wed, 30 Sep 2015 16:48:26 +0200 Subject: [PATCH 27/63] MPI bugfix on daint --- mpi-dpd/io.cu | 2 +- mpi-dpd/simulation.cu | 5 +++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 24e659a9e..4afea962e 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -196,7 +196,7 @@ void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, int NALL = 0; const int n = nparticles; - MPI_CHECK( MPI_Reduce(&n, &NALL, 1, MPI_INT, MPI_SUM, 0, cartcomm) ); + MPI_CHECK( MPI_Allreduce(&n, &NALL, 1, MPI_INT, MPI_SUM, cartcomm) ); MPI_File f; MPI_CHECK( MPI_File_open(cartcomm, filename , MPI_MODE_WRONLY | MPI_MODE_CREATE, MPI_INFO_NULL, &f) ); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 1ad47a74b..d0e65c77c 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -607,8 +607,9 @@ void Simulation::_datadump_async() for(int c = 0; c < 6; ++c) totavgstress[c] /= ntotsolvent; - printf("average stress: sxx:%.3e sxy:%.3e sxz:%.3e syy:%.3e syz:%.3e szz:%.3e\n", - totavgstress[0], totavgstress[1], totavgstress[2], totavgstress[3], totavgstress[4], totavgstress[5]); + if (rank == 0) + printf("average stress: sxx:%.3e sxy:%.3e sxz:%.3e syy:%.3e syz:%.3e szz:%.3e\n", + totavgstress[0], totavgstress[1], totavgstress[2], totavgstress[3], totavgstress[4], totavgstress[5]); } if (hdf5part_dumps) From 965d35176883b21354abb61881253d891badc7bd Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Wed, 30 Sep 2015 17:50:28 +0200 Subject: [PATCH 28/63] stress from the walls --- mpi-dpd/simulation.cu | 8 ++++++++ mpi-dpd/wall.cu | 31 ++++++++++++++++++++++++++++++- mpi-dpd/wall.h | 23 +++++++++++++++++++++++ 3 files changed, 61 insertions(+), 1 deletion(-) diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index d0e65c77c..fa3e4d3fa 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -1075,13 +1075,21 @@ void Simulation::run() stresses[c].resize(particles->size); dpd.set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + + if (wall) + wall->set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); } _forces(); if (stress && it % steps_per_dump == 0 ) + { dpd.clr_stress_buffers(); + if (wall) + wall->clr_stress_buffers(); + } + #ifndef _NO_DUMPS_ if (it % steps_per_dump == 0) _datadump(it); diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index aea985207..0b3a61489 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -377,6 +377,14 @@ namespace SolidWallsKernel } } + struct StressInfo + { + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz; + }; + + __constant__ StressInfo stressinfo; + + template __global__ __launch_bounds__(128, 16) void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, float * const acc, const float seed, const float sigmaf) { @@ -500,6 +508,16 @@ namespace SolidWallsKernel xforce += strength * xr; yforce += strength * yr; zforce += strength * zr; + + if (computestresses) + { + atomicAdd(stressinfo.sigma_xx + pid, strength * xr * _xr); + atomicAdd(stressinfo.sigma_xy + pid, strength * xr * _yr); + atomicAdd(stressinfo.sigma_xz + pid, strength * xr * _zr); + atomicAdd(stressinfo.sigma_yy + pid, strength * yr * _yr); + atomicAdd(stressinfo.sigma_yz + pid, strength * yr * _zr); + atomicAdd(stressinfo.sigma_zz + pid, strength * zr * _zr); + } } atomicAdd(acc + 3 * pid + 0, xforce); @@ -1121,8 +1139,19 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati &SolidWallsKernel::texWallCellCount.channelDesc, sizeof(int) * cells.ncells)); assert(textureoffset == 0); - SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> + if (sigma_xx) + { + SolidWallsKernel::StressInfo strinfo = { sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(SolidWallsKernel::stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, stream)); + + SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf); + } + else + SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf); + CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallParticles)); CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallCellStart)); diff --git a/mpi-dpd/wall.h b/mpi-dpd/wall.h index 466c54977..dded752d3 100644 --- a/mpi-dpd/wall.h +++ b/mpi-dpd/wall.h @@ -31,6 +31,8 @@ class ComputeWall int solid_size; float4 * solid4; + float * sigma_xx, * sigma_xy, * sigma_xz, * sigma_yy, + * sigma_yz, * sigma_zz; cudaArray * arrSDF; @@ -44,6 +46,27 @@ class ComputeWall void bounce(Particle * const p, const int n, cudaStream_t stream); + void set_stress_buffers(float * const stress_xx, float * const stress_xy, float * const stress_xz, float * const stress_yy, + float * const stress_yz, float * const stress_zz) + { + sigma_xx = stress_xx; + sigma_xy = stress_xy; + sigma_xz = stress_xz; + sigma_yy = stress_yy; + sigma_yz = stress_yz; + sigma_zz = stress_zz; + } + + void clr_stress_buffers() + { + sigma_xx = NULL; + sigma_xy = NULL; + sigma_xz = NULL; + sigma_yy = NULL; + sigma_yz = NULL; + sigma_zz = NULL; + } + void interactions(const Particle * const p, const int n, Acceleration * const acc, const int * const cellsstart, const int * const cellscount, cudaStream_t stream); }; From 2d16c7490b25a3f99a8dc93d5382216acd877c04 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 09:10:13 +0200 Subject: [PATCH 29/63] fixed a linking problem inside wall.cu --- mpi-dpd/wall.cu | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index 0b3a61489..9efa547be 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -57,8 +57,10 @@ namespace SolidWallsKernel texture texWallParticles; texture texWallCellStart, texWallCellCount; + template __global__ void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, - float * const acc, const float seed, const float sigmaf); + float * const acc, const float seed, const float sigmaf); + void setup() { texSDF.normalized = 0; @@ -83,7 +85,8 @@ namespace SolidWallsKernel texWallCellCount.mipmapFilterMode = cudaFilterModePoint; texWallCellCount.normalized = 0; - CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); } __device__ float sdf(float x, float y, float z) @@ -201,8 +204,6 @@ namespace SolidWallsKernel return make_float3(xmygrad, ymygrad, zmygrad); } - - __global__ void fill_keys(const Particle * const particles, const int n, int * const key) { assert(blockDim.x * gridDim.x >= n); From 5bdb0bcd046475bc357f91bf4821d98c38a50f93 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 15:07:21 +0200 Subject: [PATCH 30/63] stress post-processing --- postprocessing/stress/Makefile | 9 ++ postprocessing/stress/main.cpp | 171 +++++++++++++++++++++++++++++++++ 2 files changed, 180 insertions(+) create mode 100644 postprocessing/stress/Makefile create mode 100644 postprocessing/stress/main.cpp diff --git a/postprocessing/stress/Makefile b/postprocessing/stress/Makefile new file mode 100644 index 000000000..472f3a068 --- /dev/null +++ b/postprocessing/stress/Makefile @@ -0,0 +1,9 @@ +CXX = g++ + +test: main.cpp + $(CXX) main.cpp -I../ -g -o test + +clean: + rm -f test + +.PHONY = clean diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp new file mode 100644 index 000000000..163d92e03 --- /dev/null +++ b/postprocessing/stress/main.cpp @@ -0,0 +1,171 @@ +#include + +#include + +int main() +{ + printf("hello\n"); + + const float origin[3] = {0, 0, 0}; + const float extent[3] = {48, 48, 48}; + const bool project[3] = {true, true, true}; + const int noutputchannels = 6; + + const size_t chunksize = (1 << 29) / 9 / sizeof(float); + + float * const pbuf = new float[9 * chunksize]; + + float binsize[3]; + for(int c = 0; c < 3; ++c) + binsize[c] = project[c] ? extent[c] : 1; + + int nbins[3]; + for(int c = 0; c < 3; ++c) + nbins[c] = extent[c] / binsize[c]; + + const int ntotbins = nbins[0] * nbins[1] * nbins[2]; + + int * const bincount = new int[ntotbins]; + memset(bincount, 0, sizeof(int) * ntotbins); + + const int noutput = noutputchannels * ntotbins; + + float * const bindata = new float[noutput]; + memset(bindata, 0, sizeof(float) * noutput); + + FILE * fin = fopen("../../mpi-dpd/stress/stresses-00019.data", "r"); + assert(fin); + + printf("reading...\n"); + fseek(fin, 0, SEEK_END); + const size_t filesize = ftell(fin); + fseek(fin, 0, SEEK_SET); + + const size_t nparticles = filesize / 9 / sizeof(float); + assert(filesize % (9 * sizeof(float)) == 0); + + const int nhotparticles = min(nparticles, chunksize); + fread(pbuf, sizeof(float) * 9, nhotparticles, fin); + + printf("i have found %d particles\n", nparticles); + printf("particle chunk %d\n", chunksize); + + size_t nvalid = 0; + + { + float avgs[9]; + for(int i = 0; i < 9; ++i) + avgs[i] = 0; + + for(int i = 0; i < nhotparticles; ++i) + for(int c = 0; c < 9; ++c) + avgs[c] += pbuf[9 * i + c]; + + for(int i = 0; i < 9; ++i) + printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); + } + + for(int i = 0; i < nhotparticles; ++i) + { + int index[3]; + for(int c = 0; c < 3; ++c) + index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); + + bool valid = true; + for(int c = 0; c < 3; ++c) + valid &= index[c] >= 0 && index[c] < nbins[c]; + + if (!valid) + continue; + + const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); + ++bincount[binid]; + + const int base = noutputchannels * binid; + + for(int c = 0; c < 6; ++c) + bindata[base + c] += pbuf[9 * i + 3 + c]; + + ++nvalid; + } + + const bool avg = true; + if (avg) + for(int i = 0; i < ntotbins; ++i) + { + printf("bincount %03d\n", bincount[i]); + + for(int c = 0; c < 6; ++c) + bindata[9 * i + c] /= bincount[i]; + } + + + int nprojections = 0; + for(int c = 0; c < 3; ++c) + nprojections += project[c]; + + if (nprojections == 3) //six numbers + { + assert(noutput == noutputchannels); + printf("result: "); + + for(int c = 0; c < noutputchannels; ++c) + printf("%+.3e\t", bindata[c]); + + printf("\n"); + } + else if (nprojections == 2) + { + int ctr = 0; + for(int iz = 0; iz < nbins[2]; ++iz) + for(int iy = 0; iy < nbins[1]; ++iy) + for(int ix = 0; ix < nbins[0]; ++ix) + { + printf("%03d ", ctr); + + for(int c = 0; c < noutputchannels; ++c) + printf("%+.3e ", bindata[noutputchannels * ctr + c]); + + printf("\n"); + + ++ctr; + } + } + else if (nprojections == 1) + { + int nx; + for(int c = 0; c < 3; ++c) + if (nbins[c] > 1) + { + nx = nbins[c]; + break; + } + + for(int c = 0; c < noutputchannels; ++c) + { + int ctr = 0; + + for(int iz = 0; iz < nbins[2]; ++iz) + for(int iy = 0; iy < nbins[1]; ++iy) + for(int ix = 0; ix < nbins[0]; ++ix) + { + printf("%+.3e ", bindata[noutputchannels * ctr + c]); + + ++ctr; + + if (ctr % nx == 0) + printf("\n"); + } + } + } + + printf("valid: %d\n", nvalid); + + fclose(fin); + + delete [] pbuf; + delete [] bincount; + delete [] bindata; + + return 0; +} From ad9595b720190c0d01b2ce87fb5dde8e52470ead Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 15:39:39 +0200 Subject: [PATCH 31/63] refactoring stress post processor --- postprocessing/argument-parser.h | 243 +++++++++++++++++++++++++++++++ postprocessing/stress/Makefile | 2 +- postprocessing/stress/main.cpp | 151 ++++++++++--------- 3 files changed, 329 insertions(+), 67 deletions(-) create mode 100644 postprocessing/argument-parser.h diff --git a/postprocessing/argument-parser.h b/postprocessing/argument-parser.h new file mode 100644 index 000000000..d01b9f87d --- /dev/null +++ b/postprocessing/argument-parser.h @@ -0,0 +1,243 @@ +/* + * ArgumentParser.h + * Cubism + * + *This argument parser assumes that all arguments are optional ie, each of the argument names is preceded by a '-' + *all arguments are however NOT optional to avoid a mess with default values and returned values when not found! + * + *More converter could be required: + *add as needed + *TypeName as{TypeName}() in Value + * + * Created by Christian Conti on 6/7/10. That is a long time ago. + * Modified by Diego Rossinelli several times after his dreadlocks hair cut. + * Copyright 2010 ETH Zurich. All rights reserved. + * + */ + +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace std; + +class Value +{ +private: + string content; + +public: + +Value() : content("") {} + +Value(string content_) : content(content_) { /*printf("%s\n",content.c_str());*/ } + + double asDouble(double def=0) const + { + if (content == "") return def; + return (double) atof(content.c_str()); + } + + int asInt(int def=0) const + { + if (content == "") return def; + return atoi(content.c_str()); + } + + bool asBool(bool def=false) const + { + if (content == "") return def; + if (content == "0") return false; + if (content == "false") return false; + + return true; + } + + string asString(string def="") const + { + if (content == "") return def; + + return content; + } + + vector asVecFloat(const int musthave_size = -1) const + { + //printf("mycontent is %s\n", content.c_str()); + std::stringstream ss(content); + //assert(ss.good()); + vector retval; + double e; + + while (ss >> e) + { + retval.push_back(e); + // printf("reading %f\n", e); + if (ss.peek() == ',') + ss.ignore(); + } + + if (musthave_size > 0) + assert(musthave_size == (int)retval.size()); + + return retval; + } +}; + +class ArgumentParser +{ +private: + + map mapArguments; + + const int iArgC; + const char** vArgV; + bool bStrictMode, bVerbose; + + const char delimiter; +public: + + Value operator()(const string arg) + { + map::const_iterator it = mapArguments.find(arg); + + if (bStrictMode) + { + if (it == mapArguments.end()) + { + printf("Runtime option NOT SPECIFIED! ABORTING! name: %s\n",arg.data()); + abort(); + } + } + + if (bVerbose) + printf("%s is %s\n", arg.data(), mapArguments[arg].asString().data()); + + if (it != mapArguments.end()) + return mapArguments[arg]; + else + return Value(); + } + + bool check(const string arg) const + { + return mapArguments.find(arg) != mapArguments.end(); + } + +ArgumentParser(const int argc, const char ** argv, bool bVerbose = false, const char delimiter = '=') : + mapArguments(), iArgC(argc), vArgV(argv), bStrictMode(false), bVerbose(bVerbose), delimiter(delimiter) + { + for (int i = 1; i args, bool bVerbose = false, const char delimiter = '='): + mapArguments(), iArgC(args.size()), vArgV(NULL), bStrictMode(false), bVerbose(bVerbose), delimiter(delimiter) + { + for(vector::iterator it = args.begin(); it != args.end(); ++it) + { + const char * arg = it->c_str(); + + int sep = 0; + while(arg[sep] != '\0' && arg[sep] != delimiter) + ++sep; + + string value; + + if (arg[sep] != '\0') + value = string(arg + sep + 1); + else + value = "1"; + + mapArguments[string(arg, sep)] = Value(value); + } + + mute(); + } + + int getargc() const { return iArgC; } + + const char** getargv() const { return vArgV; } + + void set_strict_mode() + { + bStrictMode = true; + } + + void unset_strict_mode() + { + bStrictMode = false; + } + + void mute() + { + bVerbose = false; + } + + void loud() + { + bVerbose = true; + } + + void print_arguments(FILE * f = stdout) + { + printf("PRINTOUT OF THE RUNTIME OPTIONS\n"); + + for(map::const_iterator it=mapArguments.begin(); it!=mapArguments.end(); it++) + fprintf(f, "%s: <%s>\n", it->first.c_str(), it->second.asString().c_str()); + + printf("END OF THE PRINTOUT.\n"); + } + + void print_arguments(string path2log) + { + FILE * f = fopen(path2log.c_str(), "w"); + + if (f == NULL) + { + printf("could not save the log to <%s>. Exiting now\n", path2log.c_str()); + exit(-1); + } + + print_arguments(f); + + fclose(f); + } + + vector find(string name) + { + map::iterator itb = mapArguments.lower_bound(name); + map::iterator ite = mapArguments.end(); + + vector retval; + for(map::iterator it = itb; it != ite; ++it) + { + if (it->first.find(name) == string::npos) + break; + + retval.push_back(it->first); + } + return retval; + } +}; diff --git a/postprocessing/stress/Makefile b/postprocessing/stress/Makefile index 472f3a068..53ca76639 100644 --- a/postprocessing/stress/Makefile +++ b/postprocessing/stress/Makefile @@ -1,7 +1,7 @@ CXX = g++ test: main.cpp - $(CXX) main.cpp -I../ -g -o test + $(CXX) main.cpp -I../ -g -O3 -o test clean: rm -f test diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index 163d92e03..05b7ed7a5 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -1,18 +1,32 @@ #include +#include #include -int main() +using namespace std; + +int main(const int argc, const char ** argv) { - printf("hello\n"); + const bool verbose = false; - const float origin[3] = {0, 0, 0}; - const float extent[3] = {48, 48, 48}; - const bool project[3] = {true, true, true}; - const int noutputchannels = 6; + ArgumentParser argp(argc, argv);//("-lightpos").asVecFloat(4) + const bool avg = argp("-average").asBool(true); + vector origin = argp("-origin").asVecFloat(3); + vector extent = argp("-extent").asVecFloat(3); + vector projectf = argp("-project").asVecFloat(3); + + bool project[3]; + for(int c = 0; c < 3; ++c) + project[c] = projectf[c] != 0; + + int nprojections = 0; + for(int c = 0; c < 3; ++c) + nprojections += project[c]; + + const int noutputchannels = 6; const size_t chunksize = (1 << 29) / 9 / sizeof(float); - + float * const pbuf = new float[9 * chunksize]; float binsize[3]; @@ -27,16 +41,18 @@ int main() int * const bincount = new int[ntotbins]; memset(bincount, 0, sizeof(int) * ntotbins); - + const int noutput = noutputchannels * ntotbins; - + float * const bindata = new float[noutput]; memset(bindata, 0, sizeof(float) * noutput); - + FILE * fin = fopen("../../mpi-dpd/stress/stresses-00019.data", "r"); assert(fin); - printf("reading...\n"); + if (verbose) \ + printf("reading...\n"); + fseek(fin, 0, SEEK_END); const size_t filesize = ftell(fin); fseek(fin, 0, SEEK_SET); @@ -44,70 +60,66 @@ int main() const size_t nparticles = filesize / 9 / sizeof(float); assert(filesize % (9 * sizeof(float)) == 0); - const int nhotparticles = min(nparticles, chunksize); - fread(pbuf, sizeof(float) * 9, nhotparticles, fin); - - printf("i have found %d particles\n", nparticles); - printf("particle chunk %d\n", chunksize); - - size_t nvalid = 0; + if (verbose) + { + printf("i have found %d particles\n", nparticles); + printf("particle chunk %d\n", chunksize); + } + for(size_t base = 0; base < nparticles; base += chunksize) { - float avgs[9]; - for(int i = 0; i < 9; ++i) - avgs[i] = 0; + const int nhotparticles = min(nparticles - base, chunksize); + fread(pbuf, sizeof(float) * 9, nhotparticles, fin); - for(int i = 0; i < nhotparticles; ++i) - for(int c = 0; c < 9; ++c) - avgs[c] += pbuf[9 * i + c]; +#ifndef NDEBUG + if (verbose) + { + float avgs[9]; + for(int i = 0; i < 9; ++i) + avgs[i] = 0; - for(int i = 0; i < 9; ++i) - printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); - } - - for(int i = 0; i < nhotparticles; ++i) - { - int index[3]; - for(int c = 0; c < 3; ++c) - index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); + for(int i = 0; i < nhotparticles; ++i) + for(int c = 0; c < 9; ++c) + avgs[c] += pbuf[9 * i + c]; - bool valid = true; - for(int c = 0; c < 3; ++c) - valid &= index[c] >= 0 && index[c] < nbins[c]; + for(int i = 0; i < 9; ++i) + printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); + } +#endif - if (!valid) - continue; + for(int i = 0; i < nhotparticles; ++i) + { + int index[3]; + for(int c = 0; c < 3; ++c) + index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); - const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); - ++bincount[binid]; - - const int base = noutputchannels * binid; - - for(int c = 0; c < 6; ++c) - bindata[base + c] += pbuf[9 * i + 3 + c]; + bool valid = true; + for(int c = 0; c < 3; ++c) + valid &= index[c] >= 0 && index[c] < nbins[c]; - ++nvalid; - } + if (!valid) + continue; - const bool avg = true; - if (avg) - for(int i = 0; i < ntotbins; ++i) - { - printf("bincount %03d\n", bincount[i]); + const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); + ++bincount[binid]; + + const int base = noutputchannels * binid; for(int c = 0; c < 6; ++c) - bindata[9 * i + c] /= bincount[i]; + bindata[base + c] += pbuf[9 * i + 3 + c]; } + } + fclose(fin); - int nprojections = 0; - for(int c = 0; c < 3; ++c) - nprojections += project[c]; + if (avg) + for(int i = 0; i < ntotbins; ++i) + for(int c = 0; c < noutputchannels; ++c) + bindata[noutputchannels * i + c] /= bincount[i]; - if (nprojections == 3) //six numbers + if (nprojections == 3) { assert(noutput == noutputchannels); - printf("result: "); for(int c = 0; c < noutputchannels; ++c) printf("%+.3e\t", bindata[c]); @@ -116,13 +128,14 @@ int main() } else if (nprojections == 2) { + int ctr = 0; for(int iz = 0; iz < nbins[2]; ++iz) for(int iy = 0; iy < nbins[1]; ++iy) for(int ix = 0; ix < nbins[0]; ++ix) { printf("%03d ", ctr); - + for(int c = 0; c < noutputchannels; ++c) printf("%+.3e ", bindata[noutputchannels * ctr + c]); @@ -132,7 +145,7 @@ int main() } } else if (nprojections == 1) - { + { int nx; for(int c = 0; c < 3; ++c) if (nbins[c] > 1) @@ -140,11 +153,12 @@ int main() nx = nbins[c]; break; } - + for(int c = 0; c < noutputchannels; ++c) { + //printf("OUTPUT %d:\n", c); int ctr = 0; - + //printf("start\n"); for(int iz = 0; iz < nbins[2]; ++iz) for(int iy = 0; iy < nbins[1]; ++iy) for(int ix = 0; ix < nbins[0]; ++ix) @@ -156,12 +170,17 @@ int main() if (ctr % nx == 0) printf("\n"); } + //printf("stop\n"); + + if (c < noutputchannels - 1) + printf("END OUTPUT\n"); } } - - printf("valid: %d\n", nvalid); - - fclose(fin); + else + { + printf("woops invalid number of projections. Exiting now...\n"); + exit(-1); + } delete [] pbuf; delete [] bincount; From d7eb1277efcbb2bed9dec2228dec9633049f8bf7 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 15:58:39 +0200 Subject: [PATCH 32/63] refactoring --- postprocessing/stress/Makefile | 6 +- postprocessing/stress/main.cpp | 123 +++++++++++++++++++-------------- 2 files changed, 76 insertions(+), 53 deletions(-) diff --git a/postprocessing/stress/Makefile b/postprocessing/stress/Makefile index 53ca76639..8fcb7d184 100644 --- a/postprocessing/stress/Makefile +++ b/postprocessing/stress/Makefile @@ -1,9 +1,9 @@ CXX = g++ -test: main.cpp - $(CXX) main.cpp -I../ -g -O3 -o test +stress: main.cpp + $(CXX) main.cpp -I../ -g -O3 -Wno-deprecated-declarations -o stress clean: - rm -f test + rm -f stress .PHONY = clean diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index 05b7ed7a5..f426164a1 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -9,7 +9,7 @@ int main(const int argc, const char ** argv) { const bool verbose = false; - ArgumentParser argp(argc, argv);//("-lightpos").asVecFloat(4) + ArgumentParser argp(argc, argv); const bool avg = argp("-average").asBool(true); vector origin = argp("-origin").asVecFloat(3); @@ -47,71 +47,94 @@ int main(const int argc, const char ** argv) float * const bindata = new float[noutput]; memset(bindata, 0, sizeof(float) * noutput); - FILE * fin = fopen("../../mpi-dpd/stress/stresses-00019.data", "r"); - assert(fin); + int numfiles = 0; + while (!feof(stdin)) + { + char path[2048]; + + gets(path); + //fscanf(stdin, "%2048s\n", + fprintf(stderr, "Working on <%s>\n", path); - if (verbose) \ - printf("reading...\n"); + FILE * fin = fopen(path, "r"); - fseek(fin, 0, SEEK_END); - const size_t filesize = ftell(fin); - fseek(fin, 0, SEEK_SET); + if (!fin) + { + printf("can't access <%s> , exiting now.\n"); + exit(-1); + } - const size_t nparticles = filesize / 9 / sizeof(float); - assert(filesize % (9 * sizeof(float)) == 0); + if (verbose) + printf("reading...\n"); - if (verbose) - { - printf("i have found %d particles\n", nparticles); - printf("particle chunk %d\n", chunksize); - } + fseek(fin, 0, SEEK_END); + const size_t filesize = ftell(fin); + fseek(fin, 0, SEEK_SET); - for(size_t base = 0; base < nparticles; base += chunksize) - { - const int nhotparticles = min(nparticles - base, chunksize); - fread(pbuf, sizeof(float) * 9, nhotparticles, fin); + const size_t nparticles = filesize / 9 / sizeof(float); + assert(filesize % (9 * sizeof(float)) == 0); -#ifndef NDEBUG if (verbose) { - float avgs[9]; - for(int i = 0; i < 9; ++i) - avgs[i] = 0; + printf("i have found %d particles\n", nparticles); + printf("particle chunk %d\n", chunksize); + } - for(int i = 0; i < nhotparticles; ++i) - for(int c = 0; c < 9; ++c) - avgs[c] += pbuf[9 * i + c]; + for(size_t base = 0; base < nparticles; base += chunksize) + { + const int nhotparticles = min(nparticles - base, chunksize); + fread(pbuf, sizeof(float) * 9, nhotparticles, fin); - for(int i = 0; i < 9; ++i) - printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); - } +#ifndef NDEBUG + if (verbose) + { + float avgs[9]; + for(int i = 0; i < 9; ++i) + avgs[i] = 0; + + for(int i = 0; i < nhotparticles; ++i) + for(int c = 0; c < 9; ++c) + avgs[c] += pbuf[9 * i + c]; + + for(int i = 0; i < 9; ++i) + printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); + } #endif - for(int i = 0; i < nhotparticles; ++i) - { - int index[3]; - for(int c = 0; c < 3; ++c) - index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); + for(int i = 0; i < nhotparticles; ++i) + { + int index[3]; + for(int c = 0; c < 3; ++c) + index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); - bool valid = true; - for(int c = 0; c < 3; ++c) - valid &= index[c] >= 0 && index[c] < nbins[c]; + bool valid = true; + for(int c = 0; c < 3; ++c) + valid &= index[c] >= 0 && index[c] < nbins[c]; - if (!valid) - continue; + if (!valid) + continue; - const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); - ++bincount[binid]; + const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); + ++bincount[binid]; - const int base = noutputchannels * binid; + const int base = noutputchannels * binid; - for(int c = 0; c < 6; ++c) - bindata[base + c] += pbuf[9 * i + 3 + c]; + for(int c = 0; c < 6; ++c) + bindata[base + c] += pbuf[9 * i + 3 + c]; + } } - } - fclose(fin); + fclose(fin); + ++numfiles; + } + + if (!numfiles) + { + printf("ooops zero files were read. Exiting now.\n"); + exit(-1); + } + if (avg) for(int i = 0; i < ntotbins; ++i) for(int c = 0; c < noutputchannels; ++c) @@ -156,9 +179,8 @@ int main(const int argc, const char ** argv) for(int c = 0; c < noutputchannels; ++c) { - //printf("OUTPUT %d:\n", c); int ctr = 0; - //printf("start\n"); + for(int iz = 0; iz < nbins[2]; ++iz) for(int iy = 0; iy < nbins[1]; ++iy) for(int ix = 0; ix < nbins[0]; ++ix) @@ -170,10 +192,9 @@ int main(const int argc, const char ** argv) if (ctr % nx == 0) printf("\n"); } - //printf("stop\n"); if (c < noutputchannels - 1) - printf("END OUTPUT\n"); + printf("SEPARATION\n"); } } else @@ -186,5 +207,7 @@ int main(const int argc, const char ** argv) delete [] bincount; delete [] bindata; + perror("all is done. ciao.\n"); + return 0; } From 5de25c7a160f5be9fc56cd26759ecb2effbea995 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 16:50:00 +0200 Subject: [PATCH 33/63] example to run as "sh example.run" --- postprocessing/stress/example.run | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 postprocessing/stress/example.run diff --git a/postprocessing/stress/example.run b/postprocessing/stress/example.run new file mode 100644 index 000000000..f848c7efa --- /dev/null +++ b/postprocessing/stress/example.run @@ -0,0 +1,24 @@ + +echo EXAMPLE1: REDUCE ALL +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=1,1,1 + +echo EXAMPLE2: 1D PROFILE + GNUPLOT +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=1,0,1 > profile.txt +gnuplot -persist <<- END_GNUPLOT + plot "profile.txt" u 1:2 w lp title "sigma_xx", "profile.txt" u 1:3 w lp title "sigma_xy", "profile.txt" u 1:4 w lp title "sigma_xz", "profile.txt" u 1:5 w lp title "sigma_yy", "profile.txt" u 1:6 w lp title "sigma_yz", "profile.txt" u 1:7 w lp title "sigma_zz" +END_GNUPLOT + +echo EXAMPLE3: HEIGHT FIELD + MATPLOTLIB +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=0,1,0 | csplit - %SEPARATION% {*} -f channel. +python <<- END_PYTHON +import numpy as np +import matplotlib.pyplot as plt +import matplotlib.cm as cm + +path = "channel.01" +ncols, nrows = 48, 48 +temp = np.loadtxt('channel.04').T +grid = temp.reshape((nrows, ncols)) +plt.imshow(grid, interpolation='nearest', cmap=cm.gist_rainbow) +plt.show() +END_PYTHON \ No newline at end of file From 49d753909cbe075a80c98c5fd5b6f038775750fa Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 17:28:04 +0200 Subject: [PATCH 34/63] buffix in stress postprocessing --- postprocessing/stress/example.run | 16 ++++++++++------ postprocessing/stress/main.cpp | 8 +++----- 2 files changed, 13 insertions(+), 11 deletions(-) diff --git a/postprocessing/stress/example.run b/postprocessing/stress/example.run index f848c7efa..9b23e1abf 100644 --- a/postprocessing/stress/example.run +++ b/postprocessing/stress/example.run @@ -1,3 +1,4 @@ +make echo EXAMPLE1: REDUCE ALL find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=1,1,1 @@ -9,16 +10,19 @@ gnuplot -persist <<- END_GNUPLOT END_GNUPLOT echo EXAMPLE3: HEIGHT FIELD + MATPLOTLIB -find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=0,1,0 | csplit - %SEPARATION% {*} -f channel. +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=0,1,0 | csplit --suppress-matched - '/^$/' {*} -f channel. python <<- END_PYTHON import numpy as np import matplotlib.pyplot as plt import matplotlib.cm as cm -path = "channel.01" -ncols, nrows = 48, 48 -temp = np.loadtxt('channel.04').T -grid = temp.reshape((nrows, ncols)) -plt.imshow(grid, interpolation='nearest', cmap=cm.gist_rainbow) +for path in ["channel.00", "channel.01", "channel.02", "channel.03", "channel.04", "channel.05"]: + ncols, nrows = 48, 48 + temp = np.loadtxt(path).T + grid = temp.reshape((nrows, ncols)) + plt.figure(path) + plt.imshow(grid, interpolation='nearest', cmap=cm.gist_rainbow) + plt.show() + END_PYTHON \ No newline at end of file diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index f426164a1..92450df61 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -53,7 +53,6 @@ int main(const int argc, const char ** argv) char path[2048]; gets(path); - //fscanf(stdin, "%2048s\n", fprintf(stderr, "Working on <%s>\n", path); FILE * fin = fopen(path, "r"); @@ -151,7 +150,6 @@ int main(const int argc, const char ** argv) } else if (nprojections == 2) { - int ctr = 0; for(int iz = 0; iz < nbins[2]; ++iz) for(int iy = 0; iy < nbins[1]; ++iy) @@ -160,7 +158,7 @@ int main(const int argc, const char ** argv) printf("%03d ", ctr); for(int c = 0; c < noutputchannels; ++c) - printf("%+.3e ", bindata[noutputchannels * ctr + c]); + printf("%+.4e ", bindata[noutputchannels * ctr + c]); printf("\n"); @@ -185,7 +183,7 @@ int main(const int argc, const char ** argv) for(int iy = 0; iy < nbins[1]; ++iy) for(int ix = 0; ix < nbins[0]; ++ix) { - printf("%+.3e ", bindata[noutputchannels * ctr + c]); + printf("%+.5e ", bindata[noutputchannels * ctr + c]); ++ctr; @@ -194,7 +192,7 @@ int main(const int argc, const char ** argv) } if (c < noutputchannels - 1) - printf("SEPARATION\n"); + printf("\n"); } } else From 05536edecf601b0bf17bd967eee4a7bc2f2a555e Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 17:50:34 +0200 Subject: [PATCH 35/63] added ply2vtk-points utility --- postprocessing/ply2vtkpts/Makefile | 17 ++ postprocessing/ply2vtkpts/daint-ply2vtkpts.sh | 45 ++++ postprocessing/ply2vtkpts/main.cpp | 223 ++++++++++++++++++ 3 files changed, 285 insertions(+) create mode 100644 postprocessing/ply2vtkpts/Makefile create mode 100755 postprocessing/ply2vtkpts/daint-ply2vtkpts.sh create mode 100644 postprocessing/ply2vtkpts/main.cpp diff --git a/postprocessing/ply2vtkpts/Makefile b/postprocessing/ply2vtkpts/Makefile new file mode 100644 index 000000000..ec76ea0e7 --- /dev/null +++ b/postprocessing/ply2vtkpts/Makefile @@ -0,0 +1,17 @@ +CXX ?= CC + +ply2vtk: main.cpp + $(CXX) -Ofast -std=c++11 -fopenmp main.cpp \ + -I/apps/daint/VTK/6.2/gnu_491/include/vtk-6.2 -L/apps/daint/VTK/6.2/gnu_491/lib \ + -lvtkIOImage-6.2 -lvtkCommonDataModel-6.2 -lvtkpng-6.2 -lvtktiff-6.2 \ + -lvtkmetaio-6.2 -lvtkDICOMParser-6.2 -lvtkzlib-6.2 -lvtksys-6.2 \ + -lvtkIOXMLParser-6.2 -lvtkCommonExecutionModel-6.2 -lvtkCommonTransforms-6.2 \ + -lvtkCommonCore-6.2 -lvtkIOXML-6.2 -lvtkexpat-6.2 -lvtkjpeg-6.2 -lvtkIOCore-6.2 \ + -lvtkCommonSystem-6.2 -lvtkCommonTransforms-6.2 -lvtkCommonMath-6.2 \ + -lvtkCommonMisc-6.2 \ + -o ply2vtk + +clean: + rm -f ply2vtk + +.PHONY = clean diff --git a/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh b/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh new file mode 100755 index 000000000..b5c9c0ade --- /dev/null +++ b/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh @@ -0,0 +1,45 @@ +module swap PrgEnv-cray PrgEnv-gnu +module load cray-hdf5-parallel +module unload cray-mpich/7.0.4 +module load cray-mpich/7.1.1 +module unload gcc +module load gcc/4.9.1 +module load vtk + +convert_some() +{ + MYFOLDER=$1 + SRCPATH=$2 + SRCPATTERN=$3 + NVERTPERCELLS=$4 + + mkdir -p $MYFOLDER + + echo `date` "convert_some: $*" >> ${MYFOLDER}/log.txt + + find "$SRCPATH" -name "$SRCPATTERN" > /tmp/asd.txt + + for F in $(cat /tmp/asd.txt) + do + SRC=`basename $F` + + DST=${MYFOLDER}/${SRC%.ply}.vtp + + aprun ./ply2vtk $NVERTPERCELLS $F $DST + done +} + +if (( $# != 3)) +then + echo "usage ./convert-all.sh " + + exit 1 +fi + + +MYFOLDER=$1 #for example "ichip31" +NVERTRBC=$2 #for example 498 +NVERTCTC=$3 #for example 5220 + +convert_some "$MYFOLDER" /scratch/daint/alexeedm/ctc/"$MYFOLDER"/ply/ "rbcs-*.ply" $NVERTRBC +convert_some "$MYFOLDER" /scratch/daint/alexeedm/ctc/"$MYFOLDER"/ply/ "ctcs-*.ply" $NVERTCTC \ No newline at end of file diff --git a/postprocessing/ply2vtkpts/main.cpp b/postprocessing/ply2vtkpts/main.cpp new file mode 100644 index 000000000..c1465f927 --- /dev/null +++ b/postprocessing/ply2vtkpts/main.cpp @@ -0,0 +1,223 @@ +#include +#include + +#include +#include +#include +#include + + +#define MPI_CHECK(ans) do { mpiAssert((ans), __FILE__, __LINE__); } while(0) + +inline void mpiAssert(int code, const char *file, int line, bool abort=true) +{ + if (code != MPI_SUCCESS) + { + char error_string[2048]; + int length_of_error_string = sizeof(error_string); + MPI_Error_string(code, error_string, &length_of_error_string); + + printf("mpiAssert: %s %d %s\n", file, line, error_string); + + MPI_Abort(MPI_COMM_WORLD, code); + } +} + +using namespace std; + +#include +#include +#include +#include +#include +#include + +int dump_vtk_points(const char * dstpath, const int nrbcs, const float * xs, const float * ys, const float * zs ) +{ + vtkSmartPointer points = + vtkSmartPointer::New(); + + for ( unsigned int i = 0; i < nrbcs; ++i ) + points->InsertNextPoint ( xs[i], ys[i], zs[i] ); + + // Create a polydata object and add the points to it. + vtkSmartPointer polydata = + vtkSmartPointer::New(); + polydata->SetPoints(points); + + // Write the file + vtkSmartPointer writer = + vtkSmartPointer::New(); + writer->SetFileName(dstpath); +#if VTK_MAJOR_VERSION <= 5 + writer->SetInput(polydata); +#else + writer->SetInputData(polydata); +#endif + + writer->Write(); + + return EXIT_SUCCESS; +} + +int main(int argc, char ** argv) +{ + MPI_CHECK(MPI_Init(&argc, &argv)); + + int nranks, rank; + MPI_CHECK(MPI_Comm_size(MPI_COMM_WORLD, &nranks)); + MPI_CHECK(MPI_Comm_rank(MPI_COMM_WORLD, &rank)); + + const bool verbose = false; + + if (argc != 4) + { + if (rank == 0) + printf("usage: test \n"); + + exit(EXIT_FAILURE); + } + + const int nvpc = atoi(argv[1]); + const char * path = argv[2]; + const char * dstpath = argv[3]; + + if (rank == 0) + printf("reading at location <%s>\n", path); + + int nallvertices, nallrbcs, headersize; + + const double tstart = omp_get_wtime(); + + if (rank == 0) + { + FILE * f = fopen(path, "r"); + assert(f); + char line[2048]; + + auto eat_line = [&] () + { + fgets(line, 2048, f); + + if (verbose) + printf("reading <%s>\n", line); + }; + + for(int i = 0; i < 3; ++i) + eat_line(); + + int retval = sscanf(line, "element vertex %d\n", &nallvertices); + assert(retval == 1); + + nallrbcs = nallvertices / nvpc; + + if (verbose) + printf("*** nvertices: %d\n", nallvertices); + + for(int i = 0; i < 7; ++i) + eat_line(); + + int nfaces = -1; + retval = sscanf(line, "element face %d\n", &nfaces); + assert(retval == 1); + + if (verbose) + printf("*** nfaces: %d\n", nfaces); + + for(int i = 0; i < 2; ++i) + eat_line(); + + headersize = ftell(f); + + fclose(f); + } + + MPI_CHECK(MPI_Bcast(&nallvertices, 1, MPI_INT, 0, MPI_COMM_WORLD)); + MPI_CHECK(MPI_Bcast(&nallrbcs, 1, MPI_INT, 0, MPI_COMM_WORLD)); + MPI_CHECK(MPI_Bcast(&headersize, 1, MPI_INT, 0, MPI_COMM_WORLD)); + + const double theader = omp_get_wtime(); + + const int myrbcs_size = nallrbcs / nranks + (int)(rank < (nallrbcs % nranks)); + const int myrbcs_start = nallrbcs / nranks * rank + min(rank, nallrbcs % nranks); + + float * data = new float[6 * myrbcs_size * nvpc]; + + { + MPI_File filehandle; + MPI_CHECK( MPI_File_open(MPI_COMM_WORLD, path, MPI_MODE_RDONLY, MPI_INFO_NULL, &filehandle) ); + + MPI_Status status; + MPI_CHECK( MPI_File_read_at(filehandle, headersize + myrbcs_start * nvpc * 6 * sizeof(float), + data, myrbcs_size * nvpc * 6, MPI_FLOAT, &status)); + + MPI_CHECK( MPI_File_close(&filehandle)); + } + + vector coords[3]; + + for(int i = 0; i < 3; ++i) + coords[i].resize(myrbcs_size); + +#pragma omp parallel for + for(int r = 0; r < myrbcs_size; ++r) + { + float com[3] = {0, 0, 0}; + + for(int v = 0; v < nvpc; ++v) + for(int c = 0; c < 3; ++c) + com[c] += data[c + 6 * (v + nvpc * r)]; + + for(int i = 0; i < 3; ++i) + coords[i][r] = com[i] /nvpc; + } + + delete [] data; + + vector allcoords[3]; + + if (rank == 0) + for(int i = 0; i < 3; ++i) + allcoords[i].resize(nranks * ((nallrbcs + nranks - 1) / nranks)); + + for(int i = 0; i < 3; ++i) + { + MPI_CHECK( MPI_Gather(&coords[i].front(), nallrbcs / nranks, MPI_FLOAT, + &allcoords[i].front(), nallrbcs / nranks, MPI_FLOAT, + 0, MPI_COMM_WORLD) ); + + MPI_CHECK( MPI_Gather(&coords[i].back(), 1, MPI_FLOAT, + (&allcoords[i].front()) + nranks * (nallrbcs / nranks), 1, MPI_FLOAT, + 0, MPI_COMM_WORLD) ); + } + + const double tthroughput = omp_get_wtime(); + + if (rank == 0) + dump_vtk_points(dstpath, nallrbcs, &allcoords[0].front(), &allcoords[1].front(), &allcoords[2].front()); + + const double tvtk = omp_get_wtime(); + + if (rank == 0) + { + const double ttotal = tvtk - tstart; + + printf("TOTAL TIME: %.2f\n", ttotal); + + printf("TDISTRIBUTION: HEADER:%.1f%%\tI/O+REDUCE:%.1f%%\tVTK:%.1f%%\t\n", + 100 / ttotal * (theader - tstart), + 100 / ttotal * (tthroughput - theader), + 100 / ttotal * (tvtk - tthroughput)); + + const double memfp_ply = 6. * nallvertices * sizeof(float) / pow(1024., 3); + const double memfp_vtk = 3. * nallrbcs * sizeof(float) / pow(1024., 3); + + printf("THROUGHPUT: %.1f GB/s\n", (memfp_ply + memfp_vtk) / (tthroughput - theader)); + printf("VTK DUMP: %.1f GB/s\n", memfp_vtk / (tvtk - tthroughput)); + } + + MPI_CHECK(MPI_Finalize()); + + return 0; +} + From abbf641bec24facd23f37342aa45408210ce811f Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 1 Oct 2015 17:58:53 +0200 Subject: [PATCH 36/63] chmod on ply2vtk-points script --- postprocessing/ply2vtkpts/daint-ply2vtkpts.sh | 0 1 file changed, 0 insertions(+), 0 deletions(-) mode change 100755 => 100644 postprocessing/ply2vtkpts/daint-ply2vtkpts.sh diff --git a/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh b/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh old mode 100755 new mode 100644 From 468015da2981267d3475b050b5ffb88bcc363839 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 2 Oct 2015 17:04:02 +0200 Subject: [PATCH 37/63] postprocessing is now distributed --- postprocessing/stress/Makefile | 4 +- postprocessing/stress/example.run | 4 +- postprocessing/stress/main.cpp | 90 +++++++++++++++++++++++-------- 3 files changed, 71 insertions(+), 27 deletions(-) diff --git a/postprocessing/stress/Makefile b/postprocessing/stress/Makefile index 8fcb7d184..1707fd490 100644 --- a/postprocessing/stress/Makefile +++ b/postprocessing/stress/Makefile @@ -1,7 +1,7 @@ -CXX = g++ +CXX = mpicxx stress: main.cpp - $(CXX) main.cpp -I../ -g -O3 -Wno-deprecated-declarations -o stress + $(CXX) main.cpp -I../ -g -O3 -Wno-deprecated-declarations -Wno-unused-result -o stress clean: rm -f stress diff --git a/postprocessing/stress/example.run b/postprocessing/stress/example.run index 9b23e1abf..8123d29ce 100644 --- a/postprocessing/stress/example.run +++ b/postprocessing/stress/example.run @@ -10,14 +10,14 @@ gnuplot -persist <<- END_GNUPLOT END_GNUPLOT echo EXAMPLE3: HEIGHT FIELD + MATPLOTLIB -find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=0,1,0 | csplit --suppress-matched - '/^$/' {*} -f channel. +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,32 -project=0,1,0 | csplit --suppress-matched - '/^$/' {*} -f channel. python <<- END_PYTHON import numpy as np import matplotlib.pyplot as plt import matplotlib.cm as cm for path in ["channel.00", "channel.01", "channel.02", "channel.03", "channel.04", "channel.05"]: - ncols, nrows = 48, 48 + ncols, nrows = 48, 32 temp = np.loadtxt(path).T grid = temp.reshape((nrows, ncols)) plt.figure(path) diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index 92450df61..e0884b93c 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -1,16 +1,26 @@ +#include + #include +#include #include +#include #include +#include using namespace std; -int main(const int argc, const char ** argv) +int main(int argc, const char ** argv) { - const bool verbose = false; + MPI_CHECK( MPI_Init(&argc, (char ***)&argv) ); + + int nranks, rank; + MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank)); + MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks)); ArgumentParser argp(argc, argv); + const bool verbose = argp("-verbose").asBool(false); const bool avg = argp("-average").asBool(true); vector origin = argp("-origin").asVecFloat(3); vector extent = argp("-extent").asVecFloat(3); @@ -44,27 +54,52 @@ int main(const int argc, const char ** argv) const int noutput = noutputchannels * ntotbins; - float * const bindata = new float[noutput]; - memset(bindata, 0, sizeof(float) * noutput); + double * const bindata = new double[noutput]; + memset(bindata, 0, sizeof(double) * noutput); + + vector paths; - int numfiles = 0; - while (!feof(stdin)) { - char path[2048]; - - gets(path); - fprintf(stderr, "Working on <%s>\n", path); + string myinput; + + if (rank == 0) + for (string line; getline(cin, line);) + myinput += line + "\n"; + + int inputsize = myinput.size(); + MPI_CHECK( MPI_Bcast(&inputsize, 1, MPI_INTEGER, 0, MPI_COMM_WORLD)); + + myinput.resize(inputsize); + + MPI_CHECK( MPI_Bcast(&myinput[0], inputsize, MPI_CHAR, 0, MPI_COMM_WORLD)); + + int c = 0; + istringstream iss(myinput); + + for (string line; getline(iss, line); ++c) + if (c % nranks == rank) + paths.push_back(line); + } + + int numfiles = paths.size(); + + for(int ipath = 0; ipath < paths.size(); ++ipath) + { + const char * const path = paths[ipath].c_str(); + + if (verbose) + fprintf(stderr, "working on <%s>\n", path); FILE * fin = fopen(path, "r"); if (!fin) { - printf("can't access <%s> , exiting now.\n"); + fprintf(stderr, "can't access <%s> , exiting now.\n", path); exit(-1); } if (verbose) - printf("reading...\n"); + perror("reading...\n"); fseek(fin, 0, SEEK_END); const size_t filesize = ftell(fin); @@ -75,8 +110,8 @@ int main(const int argc, const char ** argv) if (verbose) { - printf("i have found %d particles\n", nparticles); - printf("particle chunk %d\n", chunksize); + fprintf(stderr, "i have found %d particles\n", (int)nparticles); + fprintf(stderr, "particle chunk %d\n", (int)chunksize); } for(size_t base = 0; base < nparticles; base += chunksize) @@ -124,16 +159,20 @@ int main(const int argc, const char ** argv) } fclose(fin); - - ++numfiles; } if (!numfiles) { - printf("ooops zero files were read. Exiting now.\n"); + perror("ooops zero files were read. Exiting now.\n"); exit(-1); } - + + MPI_CHECK( MPI_Reduce(rank ? bincount : MPI_IN_PLACE, bincount, ntotbins, MPI_INT, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? bindata : MPI_IN_PLACE, bindata, noutput, MPI_DOUBLE, MPI_SUM, 0, MPI_COMM_WORLD) ); + + if (rank) + goto finalize; + if (avg) for(int i = 0; i < ntotbins; ++i) for(int c = 0; c < noutputchannels; ++c) @@ -167,7 +206,7 @@ int main(const int argc, const char ** argv) } else if (nprojections == 1) { - int nx; + int nx = 0; for(int c = 0; c < 3; ++c) if (nbins[c] > 1) { @@ -178,7 +217,7 @@ int main(const int argc, const char ** argv) for(int c = 0; c < noutputchannels; ++c) { int ctr = 0; - + for(int iz = 0; iz < nbins[2]; ++iz) for(int iy = 0; iy < nbins[1]; ++iy) for(int ix = 0; ix < nbins[0]; ++ix) @@ -190,22 +229,27 @@ int main(const int argc, const char ** argv) if (ctr % nx == 0) printf("\n"); } - + if (c < noutputchannels - 1) printf("\n"); } } else { - printf("woops invalid number of projections. Exiting now...\n"); + perror("woops invalid number of projections. Exiting now...\n"); exit(-1); } + if (verbose) + perror("all is done. ciao.\n"); + +finalize: + delete [] pbuf; delete [] bincount; delete [] bindata; - perror("all is done. ciao.\n"); + MPI_CHECK( MPI_Finalize() ); return 0; } From fede315f9b7813a538624762a1bef6d4de5cfa08 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 2 Oct 2015 17:37:43 +0200 Subject: [PATCH 38/63] posix I/O instead of std lib --- postprocessing/stress/main.cpp | 79 ++++++++++++++++++++-------------- 1 file changed, 46 insertions(+), 33 deletions(-) diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index e0884b93c..fbc3152d2 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -1,4 +1,6 @@ #include +#include +#include #include #include @@ -90,9 +92,9 @@ int main(int argc, const char ** argv) if (verbose) fprintf(stderr, "working on <%s>\n", path); - FILE * fin = fopen(path, "r"); + int fdin = open(path, O_RDONLY); - if (!fin) + if (!fdin) { fprintf(stderr, "can't access <%s> , exiting now.\n", path); exit(-1); @@ -101,9 +103,9 @@ int main(int argc, const char ** argv) if (verbose) perror("reading...\n"); - fseek(fin, 0, SEEK_END); - const size_t filesize = ftell(fin); - fseek(fin, 0, SEEK_SET); + const size_t filesize = lseek(fdin, 0, SEEK_END); + + lseek(fdin, 0, SEEK_SET); const size_t nparticles = filesize / 9 / sizeof(float); assert(filesize % (9 * sizeof(float)) == 0); @@ -117,48 +119,59 @@ int main(int argc, const char ** argv) for(size_t base = 0; base < nparticles; base += chunksize) { const int nhotparticles = min(nparticles - base, chunksize); - fread(pbuf, sizeof(float) * 9, nhotparticles, fin); + const size_t nhotbytes = nhotparticles * sizeof(float) * 9; -#ifndef NDEBUG - if (verbose) + size_t nreadbytes = 0; + int start = 0; + + while(start < nhotparticles) { - float avgs[9]; - for(int i = 0; i < 9; ++i) - avgs[i] = 0; + nreadbytes += read(fdin, pbuf, nhotbytes - nreadbytes); + const int stop = nreadbytes / sizeof(float) / 9; + +#ifndef NDEBUG + if (verbose) + { + float avgs[9]; + for(int i = 0; i < 9; ++i) + avgs[i] = 0; - for(int i = 0; i < nhotparticles; ++i) - for(int c = 0; c < 9; ++c) - avgs[c] += pbuf[9 * i + c]; + for(int i = 0; i < nhotparticles; ++i) + for(int c = 0; c < 9; ++c) + avgs[c] += pbuf[9 * i + c]; - for(int i = 0; i < 9; ++i) - printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); - } + for(int i = 0; i < 9; ++i) + printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); + } #endif - for(int i = 0; i < nhotparticles; ++i) - { - int index[3]; - for(int c = 0; c < 3; ++c) - index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); + for(int i = start; i < stop; ++i) + { + int index[3]; + for(int c = 0; c < 3; ++c) + index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); - bool valid = true; - for(int c = 0; c < 3; ++c) - valid &= index[c] >= 0 && index[c] < nbins[c]; + bool valid = true; + for(int c = 0; c < 3; ++c) + valid &= index[c] >= 0 && index[c] < nbins[c]; - if (!valid) - continue; + if (!valid) + continue; - const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); - ++bincount[binid]; + const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); + ++bincount[binid]; - const int base = noutputchannels * binid; + const int base = noutputchannels * binid; + + for(int c = 0; c < 6; ++c) + bindata[base + c] += pbuf[9 * i + 3 + c]; + } - for(int c = 0; c < 6; ++c) - bindata[base + c] += pbuf[9 * i + 3 + c]; + start = stop; } } - fclose(fin); + close(fdin); } if (!numfiles) From 8d06238a4a3bd26fc2f302aba9bf47bbd60cce61 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 2 Oct 2015 17:56:30 +0200 Subject: [PATCH 39/63] perf info on stress --- postprocessing/stress/main.cpp | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index fbc3152d2..a4f2ee444 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -85,7 +85,10 @@ int main(int argc, const char ** argv) int numfiles = paths.size(); - for(int ipath = 0; ipath < paths.size(); ++ipath) + size_t totalfootprint = 0; + double timeIO = 0; + + for(int ipath = 0; ipath < (int)paths.size(); ++ipath) { const char * const path = paths[ipath].c_str(); @@ -104,6 +107,8 @@ int main(int argc, const char ** argv) perror("reading...\n"); const size_t filesize = lseek(fdin, 0, SEEK_END); + + totalfootprint += filesize; lseek(fdin, 0, SEEK_SET); @@ -126,7 +131,10 @@ int main(int argc, const char ** argv) while(start < nhotparticles) { + const double tstart = MPI_Wtime(); nreadbytes += read(fdin, pbuf, nhotbytes - nreadbytes); + timeIO += MPI_Wtime() - tstart; + const int stop = nreadbytes / sizeof(float) / 9; #ifndef NDEBUG @@ -182,6 +190,8 @@ int main(int argc, const char ** argv) MPI_CHECK( MPI_Reduce(rank ? bincount : MPI_IN_PLACE, bincount, ntotbins, MPI_INT, MPI_SUM, 0, MPI_COMM_WORLD) ); MPI_CHECK( MPI_Reduce(rank ? bindata : MPI_IN_PLACE, bindata, noutput, MPI_DOUBLE, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? &timeIO : MPI_IN_PLACE, &timeIO, 1, MPI_DOUBLE, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? &totalfootprint : MPI_IN_PLACE, &totalfootprint, 1, MPI_OFFSET, MPI_SUM, 0, MPI_COMM_WORLD) ); if (rank) goto finalize; @@ -256,6 +266,9 @@ int main(int argc, const char ** argv) if (verbose) perror("all is done. ciao.\n"); + fprintf(stderr, "total footprint: %.3f MB, I/O time: %.3f ms\n", totalfootprint * 1. / 1024 / 1024, timeIO * 1e3); + fprintf(stderr, "read throughput: %.3f GB/s\n", totalfootprint /( 1024 * 1024) / timeIO / 1024); + finalize: delete [] pbuf; From 82edfcc185b229821d2fce69780d2122846f5ab9 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 2 Oct 2015 17:58:18 +0200 Subject: [PATCH 40/63] mpi-check.h in postprocessing/ --- postprocessing/mpi-check.h | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 postprocessing/mpi-check.h diff --git a/postprocessing/mpi-check.h b/postprocessing/mpi-check.h new file mode 100644 index 000000000..31900eaab --- /dev/null +++ b/postprocessing/mpi-check.h @@ -0,0 +1,19 @@ +#include + +#include + +#define MPI_CHECK(ans) do { mpiAssert((ans), __FILE__, __LINE__); } while(0) + +inline void mpiAssert(int code, const char *file, int line, bool abort=true) +{ + if (code != MPI_SUCCESS) + { + char error_string[2048]; + int length_of_error_string = sizeof(error_string); + MPI_Error_string(code, error_string, &length_of_error_string); + + printf("mpiAssert: %s %d %s\n", file, line, error_string); + + MPI_Abort(MPI_COMM_WORLD, code); + } +} From b31111226118bfc008b02d8d216d24a50d3bcf10 Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Mon, 5 Oct 2015 11:37:13 +0200 Subject: [PATCH 41/63] tweaked and corrected --- .gitignore | 3 +++ cuda-ctc/ctc-cuda.cu | 26 +++++++++--------- mpi-dpd/Makefile | 10 ------- mpi-dpd/contact.cu | 19 +++++++------ mpi-dpd/dumper.cu | 11 +++++++- mpi-dpd/fsi.cu | 5 +++- mpi-dpd/io.cu | 63 ++++++++++++++++++++++--------------------- mpi-dpd/main.cu | 6 ----- mpi-dpd/simulation.cu | 8 +++--- mpi-dpd/simulation.h | 2 +- 10 files changed, 78 insertions(+), 75 deletions(-) diff --git a/.gitignore b/.gitignore index 28cbebe2b..8d4ed82e0 100644 --- a/.gitignore +++ b/.gitignore @@ -9,3 +9,6 @@ .DS* *.orig *.xyz +*.ply +*.png +*.dcu diff --git a/cuda-ctc/ctc-cuda.cu b/cuda-ctc/ctc-cuda.cu index cd4b9e7a9..284f2f321 100644 --- a/cuda-ctc/ctc-cuda.cu +++ b/cuda-ctc/ctc-cuda.cu @@ -27,7 +27,6 @@ using namespace std; namespace CudaCTC { - int nparticles; int ntriang; int nbonds; @@ -85,16 +84,19 @@ namespace CudaCTC in >> nparticles >> nbonds >> ntriang >> ndihedrals; - if (report) - if (in.good()) - { + + if (in.good()) + { + if (report || nparticles <= 0 || ntriang <= 0) cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang << " triangles and " << ndihedrals << " dihedrals" << endl; - } - else - { - cout << "Couldn't parse the file" << endl; - exit(1); - } + } + else + { + cout << "Couldn't parse the file" << endl; + exit(1); + } + + if (nparticles <= 0 || ntriang <= 0) abort(); // Atoms section real *xyzuvw_host = new real[6*nparticles]; @@ -177,6 +179,7 @@ namespace CudaCTC in.close(); + nvertices = nparticles; int *dummyiii; if ( cudaMalloc(&dummyiii, sizeof(int)) == cudaErrorDevicesUnavailable ) return; @@ -194,7 +197,6 @@ namespace CudaCTC delete[] xyzuvw_host; delete[] dihedrals_host; - nvertices = nparticles; host_extent.xmin = xmin[0] - origin[0]; host_extent.ymin = xmin[1] - origin[1]; host_extent.zmin = xmin[2] - origin[2]; @@ -638,7 +640,7 @@ namespace CudaCTC dim3 dihThreads(32*3, 1); dim3 dihBlocks( (ndihedrals + dihThreads.x - 1) / dihThreads.x, ncells ); - gpuErrchk( cudaMemset(host_av, 0, ncells * 2 * sizeof(float)) ); + gpuErrchk( cudaMemsetAsync(host_av, 0, ncells * 2 * sizeof(float), stream) ); areaAndVolumeKernel<<>>(); gpuErrchk( cudaPeekAtLastError() ); diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index 2f4a6c3a4..fe3102bac 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -50,16 +50,6 @@ slevel ?= 0 flops ?= 0 datadump ?= 1 -ifeq "$(datadump)" "0" -NVCCFLAGS += -D_NO_DUMPS_ -endif -ifeq "$(datadump)" "1" -NVCCFLAGS += -D_SYNC_DUMPS_ -endif -ifeq "$(datadump)" "2" -NVCCFLAGS += -D_ASYNC_DUMPS_ -endif - NVCCFLAGS += -DVISCOSITY_S_LEVEL=$(slevel) inquire: $(bash [ `cat slevel.txt` == "$(slevel)" ] || { echo "cleanall" ; } ) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 1ade2ba22..085662625 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -11,6 +11,8 @@ */ static const int maxsolutes = 32; +static const float ljsigma = 0.5; +static const float ljsigma2 = ljsigma * ljsigma; #include <../dpd-rng.h> @@ -220,7 +222,8 @@ void ComputeContact::build_cells(std::vector wsolutes, cudaStream namespace KernelsContact { - __global__ void bulk_3tpp(const float2 * const particles, + __global__ __launch_bounds__(128, 10) + void bulk_3tpp(const float2 * const particles, const int np, const int ncellentries, const int nsolutes, float * const acc, const float seed, const int mysoluteid) { @@ -241,7 +244,7 @@ namespace KernelsContact int deltaspid1, deltaspid2; { - const int xcenter = XOFFSET + (int)floorf(dst0.x); + const int xcenter = min(XCELLS - 1, max(0, XOFFSET + (int)floorf(dst0.x))); const int xstart = max(0, xcenter - 1); const int xcount = min(XCELLS, xcenter + 2) - xstart; @@ -250,9 +253,9 @@ namespace KernelsContact assert(xcount >= 0); - const int ycenter = YOFFSET + (int)floorf(dst0.y); + const int ycenter = min(YCELLS - 1, max(0, YOFFSET + (int)floorf(dst0.y))); - const int zcenter = ZOFFSET + (int)floorf(dst1.x); + const int zcenter = min(ZCELLS - 1, max(0, ZOFFSET + (int)floorf(dst1.x))); const int zmy = zcenter - 1 + zplane; const bool zvalid = zmy >= 0 && zmy < ZCELLS; @@ -333,10 +336,10 @@ namespace KernelsContact continue; const float invr2 = invrij * invrij; - const float t2 = 0.0625f * invr2; + const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)), 1000.0f); + const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); @@ -542,10 +545,10 @@ namespace KernelsContact continue; const float invr2 = invrij * invrij; - const float t2 = 0.0625f * invr2; + const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(max(0.f, -24.f * invr2 * t6 * (2.f * t6 - 1.f)), 1000.0f); + const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); diff --git a/mpi-dpd/dumper.cu b/mpi-dpd/dumper.cu index 275cc506e..3e0151db5 100644 --- a/mpi-dpd/dumper.cu +++ b/mpi-dpd/dumper.cu @@ -24,6 +24,8 @@ Dumper::Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm) : iocom nrbcverts = rdummy->get_nvertices(); nctcverts = cdummy->get_nvertices(); + if (nrbcverts < 0 || nctcverts < 0) abort(); + MPI_CHECK(MPI_Comm_rank(iocomm, &rank)); } @@ -177,6 +179,8 @@ void Dumper::do_dump() MPI_CHECK( MPI_Recv(p, n, Particle::datatype(), rank, 0, intercomm, &status) ); MPI_CHECK( MPI_Recv(a, n, Acceleration::datatype(), rank, 0, intercomm, &status) ); + double t0 = MPI_Wtime(); + H5PartDump dump_part("allparticles->h5part", iocomm, iocartcomm), *dump_part_solvent = NULL; H5FieldDump dump_field(iocartcomm); @@ -234,7 +238,7 @@ void Dumper::do_dump() if (hdf5field_dumps) { - dump_field.dump(iocomm, p, nparticles, iddatadump); + dump_field.dump(iocomm, p, nparticles, iddatadump * steps_per_dump); } { @@ -247,6 +251,11 @@ void Dumper::do_dump() qoi(p + nparticles, p + nparticles + nrbcparts, nrbcparts, nctcparts, iddatadump * dt); + double t1 = MPI_Wtime(); + double d0 = 1e3*(t1 - t0); + MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d0, &d0, 1, MPI_DOUBLE, MPI_MAX, 0, iocomm); + if (!rank) printf(" \e[35mStep: %d, datadump time: %.2f ms\e[0m\n", iddatadump, d0); + ++iddatadump; } } diff --git a/mpi-dpd/fsi.cu b/mpi-dpd/fsi.cu index 0335c87b4..03e7a8b10 100644 --- a/mpi-dpd/fsi.cu +++ b/mpi-dpd/fsi.cu @@ -261,6 +261,8 @@ void ComputeFSI::bulk(std::vector wsolutes, cudaStream_t stream) KernelsFSI::setup(wsolvent.p, wsolvent.n, wsolvent.cellsstart, wsolvent.cellscount); + CUDA_CHECK(cudaPeekAtLastError()); + for(std::vector::iterator it = wsolutes.begin(); it != wsolutes.end(); ++it) if (it->n) KernelsFSI::interactions_3tpp<<< (3 * it->n + 127) / 128, 128, 0, stream >>> @@ -456,6 +458,8 @@ void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) KernelsFSI::setup(wsolvent.p, wsolvent.n, wsolvent.cellsstart, wsolvent.cellscount); + CUDA_CHECK(cudaPeekAtLastError()); + int nremote_padded = 0; { @@ -503,4 +507,3 @@ void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) CUDA_CHECK(cudaPeekAtLastError()); } - diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index bf0e01066..8b21b6888 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -1,6 +1,6 @@ /* * io.cu - * Part of CTC/mpi-dpd/ + * Part of uDeviceX/mpi-dpd/ * * Created and authored by Diego Rossinelli on 2015-01-30. * Major bug in H5 dump fixed by Panotelli on 2015-03-24. @@ -149,14 +149,14 @@ void ply_dump_mpi(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); /* Part II: particles */ - int NPOINTS = 0; - const int n = particles.size(); - MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, comm) ); + int64_t NPOINTS = 0; + const int64_t n = particles.size(); + MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_LONG_LONG, MPI_SUM, comm) ); /* Part III: triangles */ - const int ntriangles = ntriangles_per_instance * ninstances; - int NTRIANGLES = 0; - MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, comm) ); + const int64_t ntriangles = ntriangles_per_instance * ninstances; + int64_t NTRIANGLES = 0; + MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_LONG_LONG, MPI_SUM, comm) ); /* Part I: header */ std::stringstream ss; @@ -178,20 +178,20 @@ void ply_dump_mpi(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, int HEADERSIZE = 0; MPI_CHECK( MPI_Allreduce(&headersize, &HEADERSIZE, 1, MPI_INT, MPI_SUM, comm) ); - unsigned long size0 = HEADERSIZE; - unsigned long size1 = NPOINTS*sizeof(Particle); + int64_t size0 = HEADERSIZE; + int64_t size1 = NPOINTS*sizeof(Particle); // unsigned long size2 = NTRIANGLES*4*sizeof(int); MPI_Offset base0 = 0; MPI_Offset base1 = base0 + size0; MPI_Offset base2 = base1 + size1; - int ioffset0 = 0; - int ioffset1 = 0; - int ioffset2 = 0; - MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_INTEGER, MPI_SUM, comm)); - MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_INTEGER, MPI_SUM, comm)); - MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_INTEGER, MPI_SUM, comm)); + int64_t ioffset0 = 0; + int64_t ioffset1 = 0; + int64_t ioffset2 = 0; + MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_LONG_LONG, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_LONG_LONG, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_LONG_LONG, MPI_SUM, comm)); MPI_Offset poffset0 = ioffset0*sizeof(char); MPI_Offset poffset1 = ioffset1*sizeof(Particle); @@ -251,21 +251,21 @@ void ply_dump_mpi(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, MPI_Barrier(comm); - double t3 = MPI_Wtime(); +// double t3 = MPI_Wtime(); MPI_CHECK( MPI_File_close(&f)); - double t4 = MPI_Wtime(); - - double d0 = 1e3*(t1 - t0); - double d1 = 1e3*(t2 - t1); - double d2 = 1e3*(t3 - t2); - double d3 = 1e3*(t4 - t3); - - MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d0, &d0, 1, MPI_DOUBLE, MPI_MAX, 0, comm); - MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d1, &d1, 1, MPI_DOUBLE, MPI_MAX, 0, comm); - MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d2, &d2, 1, MPI_DOUBLE, MPI_MAX, 0, comm); - MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d3, &d3, 1, MPI_DOUBLE, MPI_MAX, 0, comm); - - if (!rank) printf("ply_dump_mpi:\t %.2f ms; \t prep: %.2f, open %.2f, write %.2f, close %.2f\n", d0+d1+d2+d3, d0, d1, d2, d3); +// double t4 = MPI_Wtime(); +// +// double d0 = 1e3*(t1 - t0); +// double d1 = 1e3*(t2 - t1); +// double d2 = 1e3*(t3 - t2); +// double d3 = 1e3*(t4 - t3); +// +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d0, &d0, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d1, &d1, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d2, &d2, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d3, &d3, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// +// if (!rank) printf("ply_dump_mpi:\t %.2f ms; \t prep: %.2f, open %.2f, write %.2f, close %.2f\n", d0+d1+d2+d3, d0, d1, d2, d3); } void ply_dump_posix(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, @@ -551,8 +551,9 @@ void H5FieldDump::_write_fields(const char * const path2h5, MPI_Comm comm, const float time) { #ifndef NO_H5 - int nranks[3], periods[3], myrank[3]; + int nranks[3], periods[3], myrank[3], size; MPI_CHECK( MPI_Cart_get(cartcomm, 3, nranks, periods, myrank) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); id_t plist_id_access = H5Pcreate(H5P_FILE_ACCESS); @@ -570,7 +571,7 @@ void H5FieldDump::_write_fields(const char * const path2h5, MPI_Info_set(info, "romio_ds_write", "disable"); MPI_Info_set(info, "striping_factor", cbstr); - H5Pset_fapl_mpio(plist_id_access, comm, info); + H5Pset_fapl_mpio(plist_id_access, comm, MPI_INFO_NULL); hid_t file_id = H5Fcreate(path2h5, H5F_ACC_TRUNC, H5P_DEFAULT, plist_id_access); H5Pclose(plist_id_access); diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 6b4f2b9ed..05eb299bd 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -87,12 +87,6 @@ int main(int argc, char ** argv) contactforces = argp("-contactforces").asBool(false); nsubsteps = argp("-nsubsteps").asInt(0); -#ifndef _NO_DUMPS_ - const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(true); -#else - const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(false); -#endif - SignalHandling::setup(); #ifdef _USE_NVTX_ diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index b98ea4c39..32852e2db 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -586,8 +586,7 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercom dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), check_termination(check_termination), driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), - datadump_pending(false), simulation_is_done(false), - qoiid(0) + datadump_pending(false), simulation_is_done(false) { MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); @@ -960,11 +959,10 @@ void Simulation::run() _forces(); -#ifndef _NO_DUMPS_ if (it % steps_per_dump == 0) _datadump(it); -#endif - _update_and_bounce(); + + _update_and_bounce(); } const double time_simulation_stop = MPI_Wtime(); diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 6ca0aa9c0..f7a0917eb 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -70,7 +70,7 @@ class Simulation const size_t nsteps; float driving_acceleration; float host_idle_time; - int nranks, rank, qoiid; + int nranks, rank; std::vector _ic(); void _update_helper_arrays(); From 6e95f629ac92635d70139cdb2a97feda3a9cf27f Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Mon, 5 Oct 2015 13:32:58 +0200 Subject: [PATCH 42/63] velocity contributes to stress --- mpi-dpd/io.cu | 27 +++++++++++-------- mpi-dpd/simulation.cu | 8 +++--- postprocessing/stress/main.cpp | 49 +++++++++++++++++++++------------- 3 files changed, 50 insertions(+), 34 deletions(-) diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 4afea962e..ff877381d 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -186,7 +186,7 @@ void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, const float * const stress_xx, const float * const stress_xy, const float * const stress_xz, const float * const stress_yy, const float * const stress_yz, const float * const stress_zz) { - std::vector buf(nparticles * 9); + std::vector buf(nparticles * 12); int rank; MPI_CHECK( MPI_Comm_rank(cartcomm, &rank) ); @@ -201,24 +201,29 @@ void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, MPI_File f; MPI_CHECK( MPI_File_open(cartcomm, filename , MPI_MODE_WRONLY | MPI_MODE_CREATE, MPI_INFO_NULL, &f) ); - MPI_CHECK( MPI_File_set_size (f, sizeof(float) * 9 * NALL )); + MPI_CHECK( MPI_File_set_size (f, sizeof(float) * 12 * NALL )); const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int i = 0; i < n; ++i) { + const int base = 12 * i; + + for(int c = 0; c < 3; ++c) + buf[base + c] = particles[i].x[c] + L[c] / 2 + coords[c] * L[c]; + for(int c = 0; c < 3; ++c) - buf[9 * i + c] = particles[i].x[c] + L[c] / 2 + coords[c] * L[c]; - - buf[9 * i + 3] = stress_xx[i]; - buf[9 * i + 4] = stress_xy[i]; - buf[9 * i + 5] = stress_xz[i]; - buf[9 * i + 6] = stress_yy[i]; - buf[9 * i + 7] = stress_yz[i]; - buf[9 * i + 8] = stress_zz[i]; + buf[base + 3 + c] = particles[i].u[c]; + + buf[base + 6] = stress_xx[i]; + buf[base + 7] = stress_xy[i]; + buf[base + 8] = stress_xz[i]; + buf[base + 9] = stress_yy[i]; + buf[base + 10] = stress_yz[i]; + buf[base + 11] = stress_zz[i]; } - _write_bytes(buf.data(), sizeof(float) * 9 * n, f, cartcomm); + _write_bytes(buf.data(), sizeof(float) * 12 * n, f, cartcomm); MPI_CHECK( MPI_File_close(&f)); } diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index fa3e4d3fa..b9ca694f7 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -592,11 +592,13 @@ void Simulation::_datadump_async() stresses_datadump[3].data, stresses_datadump[4].data, stresses_datadump[5].data); //lets quickly compute the average stress in the system - float avgstress[6] = {0, 0, 0, 0, 0, 0}; + const int v1[6] = {0, 0, 0, 1, 1, 2}; + const int v2[6] = {0, 1, 2, 1, 2, 2}; + float avgstress[6] = {0, 0, 0, 0, 0, 0}; for(int c = 0; c < 6; ++c) for(int i = 0; i < nsolvent; ++i) - avgstress[c] += stresses_datadump[c].data[i]; + avgstress[c] += stresses_datadump[c].data[i] + p[i].u[v1[c]] * p[i].u[v2[c]]; int ntotsolvent; MPI_CHECK( MPI_Reduce(&nsolvent, &ntotsolvent, 1, MPI_INT, MPI_SUM, 0, myactivecomm)); @@ -723,7 +725,6 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term if (contactforces) solutex.attach_halocomputation(contact); - //localcomm.initialize(activecomm); int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); @@ -990,7 +991,6 @@ void Simulation::run() int it; - for(it = 0; it < nsteps; ++it) { const bool verbose = it > 0 && rank == 0; diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index a4f2ee444..2427f8d13 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -27,6 +27,10 @@ int main(int argc, const char ** argv) vector origin = argp("-origin").asVecFloat(3); vector extent = argp("-extent").asVecFloat(3); vector projectf = argp("-project").asVecFloat(3); + string contributions = argp("-contributions").asString("uf"); + + const double ufactor = contributions.find("u") != string::npos; + const double ffactor = contributions.find("f") != string::npos; bool project[3]; for(int c = 0; c < 3; ++c) @@ -37,9 +41,9 @@ int main(int argc, const char ** argv) nprojections += project[c]; const int noutputchannels = 6; - const size_t chunksize = (1 << 29) / 9 / sizeof(float); + const size_t chunksize = (1 << 29) / 12 / sizeof(float); - float * const pbuf = new float[9 * chunksize]; + float * const pbuf = new float[12 * chunksize]; float binsize[3]; for(int c = 0; c < 3; ++c) @@ -87,7 +91,7 @@ int main(int argc, const char ** argv) size_t totalfootprint = 0; double timeIO = 0; - + for(int ipath = 0; ipath < (int)paths.size(); ++ipath) { const char * const path = paths[ipath].c_str(); @@ -107,13 +111,13 @@ int main(int argc, const char ** argv) perror("reading...\n"); const size_t filesize = lseek(fdin, 0, SEEK_END); - + totalfootprint += filesize; lseek(fdin, 0, SEEK_SET); - const size_t nparticles = filesize / 9 / sizeof(float); - assert(filesize % (9 * sizeof(float)) == 0); + const size_t nparticles = filesize / 12 / sizeof(float); + assert(filesize % (12 * sizeof(float)) == 0); if (verbose) { @@ -124,40 +128,42 @@ int main(int argc, const char ** argv) for(size_t base = 0; base < nparticles; base += chunksize) { const int nhotparticles = min(nparticles - base, chunksize); - const size_t nhotbytes = nhotparticles * sizeof(float) * 9; + const size_t nhotbytes = nhotparticles * sizeof(float) * 12; size_t nreadbytes = 0; int start = 0; - + while(start < nhotparticles) { const double tstart = MPI_Wtime(); nreadbytes += read(fdin, pbuf, nhotbytes - nreadbytes); timeIO += MPI_Wtime() - tstart; - - const int stop = nreadbytes / sizeof(float) / 9; - + + const int stop = nreadbytes / sizeof(float) / 12; + #ifndef NDEBUG if (verbose) { - float avgs[9]; - for(int i = 0; i < 9; ++i) + float avgs[12]; + for(int i = 0; i < 12; ++i) avgs[i] = 0; for(int i = 0; i < nhotparticles; ++i) - for(int c = 0; c < 9; ++c) - avgs[c] += pbuf[9 * i + c]; + for(int c = 0; c < 12; ++c) + avgs[c] += pbuf[12 * i + c]; - for(int i = 0; i < 9; ++i) + for(int i = 0; i < 12; ++i) printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); } #endif for(int i = start; i < stop; ++i) { + const int srcbase = 12 * i; + int index[3]; for(int c = 0; c < 3; ++c) - index[c] = (int)((pbuf[9 * i + c] - origin[c]) / binsize[c]); + index[c] = (int)((pbuf[srcbase + c] - origin[c]) / binsize[c]); bool valid = true; for(int c = 0; c < 3; ++c) @@ -169,10 +175,15 @@ int main(int argc, const char ** argv) const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); ++bincount[binid]; - const int base = noutputchannels * binid; + const int dstbase = noutputchannels * binid; + + const int v1[6] = {0, 0, 0, 1, 1, 2}; + const int v2[6] = {0, 1, 2, 1, 2, 2}; for(int c = 0; c < 6; ++c) - bindata[base + c] += pbuf[9 * i + 3 + c]; + bindata[dstbase + c] += + ffactor * pbuf[srcbase + 6 + c] + + ufactor * pbuf[srcbase + 3 + v1[c]] * pbuf[srcbase + 3 + v2[c]]; } start = stop; From b1225851de03690cd4cce8aa37c1231842bf3269 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Mon, 5 Oct 2015 17:06:33 +0200 Subject: [PATCH 43/63] plates geometry generation --- device-gen/plates/Makefile | 7 ++++++ device-gen/plates/plates.cpp | 47 ++++++++++++++++++++++++++++++++++++ mpi-dpd/common.h | 2 +- mpi-dpd/main.cu | 3 ++- mpi-dpd/simulation.cu | 2 +- mpi-dpd/wall.cu | 24 +++++++++++------- mpi-dpd/wall.h | 4 +-- 7 files changed, 75 insertions(+), 14 deletions(-) create mode 100644 device-gen/plates/Makefile create mode 100644 device-gen/plates/plates.cpp diff --git a/device-gen/plates/Makefile b/device-gen/plates/Makefile new file mode 100644 index 000000000..83f041649 --- /dev/null +++ b/device-gen/plates/Makefile @@ -0,0 +1,7 @@ +plates: plates.cpp + $(CXX) $(CXXFLAGS) $< -o plates + +clean: + rm -f plates + +.PHONY = clean diff --git a/device-gen/plates/plates.cpp b/device-gen/plates/plates.cpp new file mode 100644 index 000000000..0f8f6ad8f --- /dev/null +++ b/device-gen/plates/plates.cpp @@ -0,0 +1,47 @@ +#include +#include +#include +#include + +int main(const int argc, const char * argv[]) +{ + if (argc != 5) + { + printf("usage: ./plates \n"); + return 1; + } + + const int NX = atoi(argv[1]); + const int NY = atoi(argv[2]); + const int NZ = atoi(argv[3]); + const int zmargin = atoi(argv[4]); + + float * data = new float[NX * NY * NZ]; + + for(int iz = 0; iz < NZ; ++iz) + { + const float zval = fabs(iz + 0.5 - NZ * 0.5) - (NZ / 2 - zmargin); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + data[ix + NX * (iy + NY * iz)] = zval; + } + + FILE * f = fopen("sdf.dat", "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", (float)NX, (float)NY, (float)NZ); + fprintf(f, "%d %d %d\n", NY, NX, NZ); + fwrite(data, sizeof(float), NX * NY * NZ, f); + fclose(f); + + { + FILE * f = fopen("sdf.raw", "w"); + assert(f != 0); + fwrite(data, sizeof(float), NX * NY * NZ, f); + fclose(f); + } + + delete [] data; + + return 0; +} diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index 51c629846..d2009b4ad 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -37,7 +37,7 @@ const float sigmaf = sigma / sqrt(dt); const float aij = 25; const float hydrostatic_a = 0.05; -extern float tend; +extern float tend, couette; extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces, stress; extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 7c74c688a..4ef4b1680 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -23,7 +23,7 @@ #include "simulation.h" bool currently_profiling = false; -float tend; +float tend, couette; bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces, stress; int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; @@ -86,6 +86,7 @@ int main(int argc, char ** argv) adjust_message_sizes = argp("-adjust_message_sizes").asBool(false); contactforces = argp("-contactforces").asBool(false); stress = argp("-stress").asBool(false); + couette = argp("-couette").asDouble(0); #ifndef _NO_DUMPS_ const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(true); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index b9ca694f7..12f309863 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -257,7 +257,7 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) int nsurvived = 0; ExpectedMessageSizes new_sizes; - wall = new ComputeWall(cartcomm, particles->xyzuvw.data, particles->size, nsurvived, new_sizes, verbose); + wall = new ComputeWall(cartcomm, particles->xyzuvw.data, particles->size, nsurvived, new_sizes, couette); //adjust the message sizes if we're pushing the flow in x { diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index 9efa547be..3f7dda267 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -59,7 +59,7 @@ namespace SolidWallsKernel template __global__ void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, - float * const acc, const float seed, const float sigmaf); + float * const acc, const float seed, const float sigmaf, const float xvel, const float y0); void setup() { @@ -385,9 +385,11 @@ namespace SolidWallsKernel __constant__ StressInfo stressinfo; + template __global__ __launch_bounds__(128, 16) void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, - float * const acc, const float seed, const float sigmaf) + float * const acc, const float seed, const float sigmaf, + const float xvelocity_wall, const float y0) { assert(blockDim.x * gridDim.x >= np * 3); @@ -496,9 +498,11 @@ namespace SolidWallsKernel const float xr = _xr * invrij; const float yr = _yr * invrij; const float zr = _zr * invrij; - + + const float xvel = yq > y0 ? xvelocity_wall : 0; + const float rdotv = - xr * (dst1.y - 0) + + xr * (dst1.y - xvel) + yr * (dst2.x - 0) + zr * (dst2.y - 0); @@ -775,8 +779,8 @@ struct FieldSampler }; ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, - ExpectedMessageSizes& new_sizes, const bool verbose): - cartcomm(cartcomm), arrSDF(NULL), solid4(NULL), solid_size(0), + ExpectedMessageSizes& new_sizes, const float xvelocity): + cartcomm(cartcomm), arrSDF(NULL), solid4(NULL), solid_size(0), xvelocity(xvelocity), cells(XSIZE_SUBDOMAIN + 2 * XMARGIN_WALL, YSIZE_SUBDOMAIN + 2 * YMARGIN_WALL, ZSIZE_SUBDOMAIN + 2 * ZMARGIN_WALL) { MPI_CHECK( MPI_Comm_rank(cartcomm, &myrank)); @@ -785,6 +789,7 @@ ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& float * field = new float[ XTEXTURESIZE * YTEXTURESIZE * ZTEXTURESIZE]; + static const bool verbose = false; FieldSampler sampler("sdf.dat", cartcomm, verbose); const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; @@ -1123,7 +1128,6 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati const int * const cellsstart, const int * const cellscount, cudaStream_t stream) { NVTX_RANGE("WALL/interactions", NVTX_C3); - //cellsstart and cellscount IGNORED for now if (n > 0 && solid_size > 0) { @@ -1140,6 +1144,8 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati &SolidWallsKernel::texWallCellCount.channelDesc, sizeof(int) * cells.ncells)); assert(textureoffset == 0); + const float y0 = (dims[1] - 1 - 2 * coords[1]) * YSIZE_SUBDOMAIN / 2; + if (sigma_xx) { SolidWallsKernel::StressInfo strinfo = { sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; @@ -1147,11 +1153,11 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati CUDA_CHECK(cudaMemcpyToSymbolAsync(SolidWallsKernel::stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, stream)); SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> - ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf); + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, y0); } else SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> - ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf); + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, y0); CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallParticles)); diff --git a/mpi-dpd/wall.h b/mpi-dpd/wall.h index dded752d3..3c28c7672 100644 --- a/mpi-dpd/wall.h +++ b/mpi-dpd/wall.h @@ -32,7 +32,7 @@ class ComputeWall int solid_size; float4 * solid4; float * sigma_xx, * sigma_xy, * sigma_xz, * sigma_yy, - * sigma_yz, * sigma_zz; + * sigma_yz, * sigma_zz, xvelocity; cudaArray * arrSDF; @@ -40,7 +40,7 @@ class ComputeWall public: - ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, ExpectedMessageSizes& new_sizes, const bool verbose); + ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, ExpectedMessageSizes& new_sizes, const float xvelocity); ~ComputeWall(); From d9a5175813a3177a538d0388f5d7bf058b0c965a Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Mon, 5 Oct 2015 17:37:37 +0200 Subject: [PATCH 44/63] release version of plates generation --- device-gen/plates/Makefile | 2 ++ device-gen/plates/plates.cpp | 2 ++ 2 files changed, 4 insertions(+) diff --git a/device-gen/plates/Makefile b/device-gen/plates/Makefile index 83f041649..4e67bbfb1 100644 --- a/device-gen/plates/Makefile +++ b/device-gen/plates/Makefile @@ -1,3 +1,5 @@ +CXXFLAGS += -O3 -DNDEBUG + plates: plates.cpp $(CXX) $(CXXFLAGS) $< -o plates diff --git a/device-gen/plates/plates.cpp b/device-gen/plates/plates.cpp index 0f8f6ad8f..6e3a4546f 100644 --- a/device-gen/plates/plates.cpp +++ b/device-gen/plates/plates.cpp @@ -34,12 +34,14 @@ int main(const int argc, const char * argv[]) fwrite(data, sizeof(float), NX * NY * NZ, f); fclose(f); +#ifndef NDEBUG { FILE * f = fopen("sdf.raw", "w"); assert(f != 0); fwrite(data, sizeof(float), NX * NY * NZ, f); fclose(f); } +#endif delete [] data; From 943eefb3cea08d4596a1694d4acafed44281afc8 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Tue, 6 Oct 2015 10:17:24 +0200 Subject: [PATCH 45/63] bugfix in wall.cu, incomplete initialization list --- mpi-dpd/wall.cu | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index 3f7dda267..e30d1e693 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -780,7 +780,9 @@ struct FieldSampler ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, ExpectedMessageSizes& new_sizes, const float xvelocity): + sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), sigma_zz(NULL), cartcomm(cartcomm), arrSDF(NULL), solid4(NULL), solid_size(0), xvelocity(xvelocity), + cells(XSIZE_SUBDOMAIN + 2 * XMARGIN_WALL, YSIZE_SUBDOMAIN + 2 * YMARGIN_WALL, ZSIZE_SUBDOMAIN + 2 * ZMARGIN_WALL) { MPI_CHECK( MPI_Comm_rank(cartcomm, &myrank)); From 0b936ed7f6685659f9b42387f5f0dc7cd8256bb9 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Tue, 6 Oct 2015 17:05:13 +0200 Subject: [PATCH 46/63] bugfixes inside wall.cu, ppstress velocity profile --- mpi-dpd/wall.cu | 10 +++++----- postprocessing/stress/main.cpp | 7 +++++-- 2 files changed, 10 insertions(+), 7 deletions(-) diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index e30d1e693..e9205e79f 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -389,7 +389,7 @@ namespace SolidWallsKernel template __global__ __launch_bounds__(128, 16) void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, float * const acc, const float seed, const float sigmaf, - const float xvelocity_wall, const float y0) + const float xvelocity_wall, const float z0) { assert(blockDim.x * gridDim.x >= np * 3); @@ -499,7 +499,7 @@ namespace SolidWallsKernel const float yr = _yr * invrij; const float zr = _zr * invrij; - const float xvel = yq > y0 ? xvelocity_wall : 0; + const float xvel = zq > z0 ? xvelocity_wall : 0; const float rdotv = xr * (dst1.y - xvel) + @@ -1146,7 +1146,7 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati &SolidWallsKernel::texWallCellCount.channelDesc, sizeof(int) * cells.ncells)); assert(textureoffset == 0); - const float y0 = (dims[1] - 1 - 2 * coords[1]) * YSIZE_SUBDOMAIN / 2; + const float z0 = (dims[2] - 1 - 2 * coords[2]) * ZSIZE_SUBDOMAIN / 2; if (sigma_xx) { @@ -1155,11 +1155,11 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati CUDA_CHECK(cudaMemcpyToSymbolAsync(SolidWallsKernel::stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, stream)); SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> - ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, y0); + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, z0); } else SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> - ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, y0); + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, z0); CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallParticles)); diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp index 2427f8d13..c90aeb02f 100644 --- a/postprocessing/stress/main.cpp +++ b/postprocessing/stress/main.cpp @@ -40,7 +40,7 @@ int main(int argc, const char ** argv) for(int c = 0; c < 3; ++c) nprojections += project[c]; - const int noutputchannels = 6; + const int noutputchannels = 9; const size_t chunksize = (1 << 29) / 12 / sizeof(float); float * const pbuf = new float[12 * chunksize]; @@ -184,6 +184,9 @@ int main(int argc, const char ** argv) bindata[dstbase + c] += ffactor * pbuf[srcbase + 6 + c] + ufactor * pbuf[srcbase + 3 + v1[c]] * pbuf[srcbase + 3 + v2[c]]; + + for(int c = 0; c < 3; ++c) + bindata[dstbase + 6 + c] += pbuf[srcbase + 3 + c]; } start = stop; @@ -193,7 +196,7 @@ int main(int argc, const char ** argv) close(fdin); } - if (!numfiles) + if (rank == 0 && !numfiles) { perror("ooops zero files were read. Exiting now.\n"); exit(-1); From a3f7edd4be928e1e7ef3937c4e8e0119b70858a8 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Wed, 7 Oct 2015 16:08:40 +0200 Subject: [PATCH 47/63] 1/2 missing in the stress computation --- mpi-dpd/io.cu | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index ff877381d..122c20973 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -215,12 +215,12 @@ void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, for(int c = 0; c < 3; ++c) buf[base + 3 + c] = particles[i].u[c]; - buf[base + 6] = stress_xx[i]; - buf[base + 7] = stress_xy[i]; - buf[base + 8] = stress_xz[i]; - buf[base + 9] = stress_yy[i]; - buf[base + 10] = stress_yz[i]; - buf[base + 11] = stress_zz[i]; + buf[base + 6] = stress_xx[i] / 2; + buf[base + 7] = stress_xy[i] / 2; + buf[base + 8] = stress_xz[i] / 2; + buf[base + 9] = stress_yy[i] / 2; + buf[base + 10] = stress_yz[i] / 2; + buf[base + 11] = stress_zz[i] / 2; } _write_bytes(buf.data(), sizeof(float) * 12 * n, f, cartcomm); From 9104a5d2c0c9b44eb65d73f3590ebc5dbd27b181 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Thu, 15 Oct 2015 23:37:27 +0200 Subject: [PATCH 48/63] shotgun debugging --- mpi-dpd/contact.cu | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 42059759f..46f443027 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -460,7 +460,9 @@ namespace KernelsContact int deltaspid1, deltaspid2; { - const int xcenter = XOFFSET + (int)floorf(dst0.x); + int xcenter = XOFFSET + (int)floorf(dst0.x); + xcenter = max(-1, min(XCELLS, xcenter)); + const int xstart = max(0, xcenter - 1); const int xcount = min(XCELLS, xcenter + 2) - xstart; @@ -469,9 +471,12 @@ namespace KernelsContact assert(xcount >= 0); - const int ycenter = YOFFSET + (int)floorf(dst0.y); + int ycenter = YOFFSET + (int)floorf(dst0.y); + ycenter = max(-1, min(YCELLS, ycenter)); + + int zcenter = ZOFFSET + (int)floorf(dst1.x); + zcenter = max(-1, min(ZCELLS, zcenter)); - const int zcenter = ZOFFSET + (int)floorf(dst1.x); const int zmy = zcenter - 1 + zplane; const bool zvalid = zmy >= 0 && zmy < ZCELLS; From 4cfe09e3e5e44bad56aa5f00a0276fbe90934ba8 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 16 Oct 2015 09:25:25 +0200 Subject: [PATCH 49/63] contact hardbound set to 1e4 --- mpi-dpd/contact.cu | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 46f443027..1aecd204d 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -339,7 +339,7 @@ namespace KernelsContact const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + const float lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); @@ -553,7 +553,7 @@ namespace KernelsContact const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + const float lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); From e46891cbbdcd08716047cb6f4374852b18a1baa9 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Tue, 27 Oct 2015 05:49:34 +0100 Subject: [PATCH 50/63] bugfix overlapping cells --- mpi-dpd/contact.cu | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 1aecd204d..ea660a04f 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -141,7 +141,7 @@ namespace KernelsContact const int ncells = XSIZE_SUBDOMAIN * YSIZE_SUBDOMAIN * ZSIZE_SUBDOMAIN; - CUDA_CHECK(cudaBindTexture(&textureoffset, &texCellsStart, cellsstart, &texCellsStart.channelDesc, sizeof(int) * ncells)); + CUDA_CHECK(cudaBindTexture(&textureoffset, &texCellsStart, cellsstart, &texCellsStart.channelDesc, sizeof(int) * (1 + ncells))); assert(textureoffset == 0); const int n = wsolutes.size(); @@ -535,6 +535,14 @@ namespace KernelsContact const float2 stmp1 = _ACCESS(csolutes[soluteid] + sentry + 1); const float2 stmp2 = _ACCESS(csolutes[soluteid] + sentry + 2); + const bool invalidsrc = + stmp0.x < -XOFFSET || + stmp0.y < -YOFFSET || + stmp1.x < -ZOFFSET; + + if (invalidsrc) + continue; + const float _xr = dst0.x - stmp0.x; const float _yr = dst0.y - stmp0.y; const float _zr = dst1.x - stmp1.x; From 2a48b9b5f93091e36d27e9f9232105b458569bd9 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 30 Oct 2015 14:29:59 +0100 Subject: [PATCH 51/63] quickfix: contact forces no cell overlap --- mpi-dpd/contact.cu | 6 +++--- mpi-dpd/solute-exchange.cu | 12 +++++++----- 2 files changed, 10 insertions(+), 8 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index ea660a04f..dfc2b3234 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -535,14 +535,14 @@ namespace KernelsContact const float2 stmp1 = _ACCESS(csolutes[soluteid] + sentry + 1); const float2 stmp2 = _ACCESS(csolutes[soluteid] + sentry + 2); - const bool invalidsrc = +/* const bool invalidsrc = stmp0.x < -XOFFSET || stmp0.y < -YOFFSET || stmp1.x < -ZOFFSET; if (invalidsrc) - continue; - + continue; +*/ const float _xr = dst0.x - stmp0.x; const float _yr = dst0.y - stmp0.y; const float _zr = dst1.x - stmp1.x; diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index e2d63c03d..92910c0b8 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -72,7 +72,7 @@ namespace SolutePUP __constant__ int coffsets[26]; - __global__ void scatter_indices(const float2 * const particles, const int nparticles, int * const counts) + __global__ void scatter_indices(const float2 * const particles, const int nparticles, int * const counts, const bool contactforces) { assert(blockDim.x * gridDim.x >= nparticles && blockDim.x == 128); @@ -104,11 +104,13 @@ namespace SolutePUP assert(fabs(s2.x) < 1e4); assert(fabs(s2.y) < 1e4); + const int mymargin = contactforces ? 15 : 1; + const int halocode[3] = { - -1 + (int)(s0.x >= -HXSIZE + 1) + (int)(s0.x >= HXSIZE - 1), - -1 + (int)(s0.y >= -HYSIZE + 1) + (int)(s0.y >= HYSIZE - 1), - -1 + (int)(s1.x >= -HZSIZE + 1) + (int)(s1.x >= HZSIZE - 1) + -1 + (int)(s0.x >= -HXSIZE + mymargin) + (int)(s0.x >= HXSIZE - mymargin), + -1 + (int)(s0.y >= -HYSIZE + mymargin) + (int)(s0.y >= HYSIZE - mymargin), + -1 + (int)(s1.x >= -HZSIZE + mymargin) + (int)(s1.x >= HZSIZE - mymargin) }; if (halocode[0] == 0 && halocode[1] == 0 && halocode[2] == 0) @@ -310,7 +312,7 @@ void SoluteExchange::_pack_attempt(cudaStream_t stream) { CUDA_CHECK(cudaMemcpyToSymbolAsync(SolutePUP::coffsets, packsoffset.data + 26 * i, sizeof(int) * 26, 0, cudaMemcpyDeviceToDevice, stream)); - SolutePUP::scatter_indices<<< (it.n + 127) / 128, 128, 0, stream >>>((float2 *)it.p, it.n, packscount.data + i * 26); + SolutePUP::scatter_indices<<< (it.n + 127) / 128, 128, 0, stream >>>((float2 *)it.p, it.n, packscount.data + i * 26, contactforces); } SolutePUP::tiny_scan<<< 1, 32, 0, stream >>>(packscount.data + i * 26, packsoffset.data + 26 * i, From ea029f7a204afc5e7e68fcc6234d430db6736cb9 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 11:49:05 +0100 Subject: [PATCH 52/63] rewrite of contact.cu --- mpi-dpd/common-kernels.h | 24 +-- mpi-dpd/contact.cu | 312 ++++++++++++------------------ mpi-dpd/contact.h | 14 +- mpi-dpd/redistribute-particles.cu | 2 +- mpi-dpd/simulation.cu | 10 +- mpi-dpd/solute-exchange.cu | 12 +- 6 files changed, 143 insertions(+), 231 deletions(-) diff --git a/mpi-dpd/common-kernels.h b/mpi-dpd/common-kernels.h index a2b8a1572..59c165675 100644 --- a/mpi-dpd/common-kernels.h +++ b/mpi-dpd/common-kernels.h @@ -170,8 +170,7 @@ void write_AOS3f(float * const data, const int nparticles, float& s0, float& s1, data[laneid + 64] = s2; } -template -__global__ void subindex_local(const int nparticles, const float2 * particles, int * const partials, +__global__ static void subindex_local(const int nparticles, const float2 * particles, int * const partials, uchar4 * const subindices) { assert(blockDim.x == 128 && blockDim.x * gridDim.x >= nparticles); @@ -192,29 +191,18 @@ __global__ void subindex_local(const int nparticles, const float2 * particles, read_AOS6f(particles + 3 * base, nsrc, data0, data1, data2); - const bool inside = project || + const bool inside = (data0.x >= -XSIZE_SUBDOMAIN / 2 && data0.x < XSIZE_SUBDOMAIN / 2 && data0.y >= -YSIZE_SUBDOMAIN / 2 && data0.y < YSIZE_SUBDOMAIN / 2 && data1.x >= -ZSIZE_SUBDOMAIN / 2 && data1.x < ZSIZE_SUBDOMAIN / 2 ); if (lane < nsrc && inside) { - if (project) - { - const int xcid = min(XSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2))); - const int ycid = min(YSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2))); - const int zcid = min(ZSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2))); + const int xcid = (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2); + const int ycid = (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2); + const int zcid = (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2); - cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); - } - else - { - const int xcid = (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2); - const int ycid = (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2); - const int zcid = (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2); - - cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); - } + cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); } } diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index dfc2b3234..846bd3b45 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -42,11 +42,6 @@ namespace KernelsContact texture texCellsStart, texCellEntries; - __global__ void bulk_3tpp(const float2 * const particles, const int np, const int ncellentries, const int nsolutes, - float * const acc, const float seed, const int mysoluteid); - - __global__ void halo(const int nparticles_padded, const int ncellentries, const int nsolutes, const float seed); - void setup() { texCellsStart.channelDesc = cudaCreateChannelDesc(); @@ -58,14 +53,11 @@ namespace KernelsContact texCellEntries.filterMode = cudaFilterModePoint; texCellEntries.mipmapFilterMode = cudaFilterModePoint; texCellEntries.normalized = 0; - - CUDA_CHECK(cudaFuncSetCacheConfig(bulk_3tpp, cudaFuncCachePreferL1)); - CUDA_CHECK(cudaFuncSetCacheConfig(halo, cudaFuncCachePreferL1)); } } ComputeContact::ComputeContact(MPI_Comm comm): -cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16), compressed_cellscount(KernelsContact::NCELLS + 16) +cellsstart(KernelsContact::NCELLS), cellscount(KernelsContact::NCELLS), compressed_cellscount(KernelsContact::NCELLS) { int myrank; MPI_CHECK( MPI_Comm_rank(comm, &myrank)); @@ -81,7 +73,7 @@ cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16) namespace KernelsContact { - __global__ void populate(const uchar4 * const subindices, const int * const cellstart, + __global__ void populate(const uchar4 * const subindices, const int * const cellstart, const int nparticles, const int soluteid, const int ntotalparticles, CellEntry * const entrycells) { @@ -161,84 +153,34 @@ namespace KernelsContact CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutes, ps, sizeof(float2 *) * n, 0, cudaMemcpyHostToDevice, stream)); CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutesacc, as, sizeof(float *) * n, 0, cudaMemcpyHostToDevice, stream)); } -} - -void ComputeContact::build_cells(std::vector wsolutes, cudaStream_t stream) -{ - this->nsolutes = wsolutes.size(); - - int ntotal = 0; - - for(int i = 0; i < wsolutes.size(); ++i) - ntotal += wsolutes[i].n; - subindices.resize(ntotal); - cellsentries.resize(ntotal); - - CUDA_CHECK(cudaMemsetAsync(cellscount.data, 0, sizeof(int) * cellscount.size, stream)); - -#ifndef NDEBUG - CUDA_CHECK(cudaMemsetAsync(cellsentries.data, 0xff, sizeof(int) * cellsentries.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(subindices.data, 0xff, sizeof(int) * subindices.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(compressed_cellscount.data, 0xff, sizeof(unsigned char) * compressed_cellscount.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(cellsstart.data, 0xff, sizeof(int) * cellsstart.capacity, stream)); -#endif - - CUDA_CHECK(cudaPeekAtLastError()); - - int ctr = 0; - for(int i = 0; i < wsolutes.size(); ++i) - { - const ParticlesWrap it = wsolutes[i]; - - if (it.n) - subindex_local<<< (it.n + 127) / 128, 128, 0, stream >>> - (it.n, (float2 *)it.p, cellscount.data, subindices.data + ctr); - - ctr += it.n; - } - - compress_counts<<< (compressed_cellscount.size + 127) / 128, 128, 0, stream >>> - (compressed_cellscount.size, (int4 *)cellscount.data, (uchar4 *)compressed_cellscount.data); - - scan(compressed_cellscount.data, compressed_cellscount.size, stream, (uint *)cellsstart.data); - - ctr = 0; - for(int i = 0; i < wsolutes.size(); ++i) - { - const ParticlesWrap it = wsolutes[i]; - - if (it.n) - KernelsContact::populate<<< (it.n + 127) / 128, 128, 0, stream >>> - (subindices.data + ctr, cellsstart.data, it.n, i, ntotal, (KernelsContact::CellEntry *)cellsentries.data); - - ctr += it.n; - } - - CUDA_CHECK(cudaPeekAtLastError()); - - KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); -} - -namespace KernelsContact -{ - __global__ __launch_bounds__(128, 10) - void bulk_3tpp(const float2 * const particles, - const int np, const int ncellentries, const int nsolutes, - float * const acc, const float seed, const int mysoluteid) + __global__ void bulk_3tpp(const int np, const int nsolutes, const float seed) { assert(blockDim.x * gridDim.x >= np * 3); const int gid = threadIdx.x + blockDim.x * blockIdx.x; - const int pid = gid / 3; + const int myslot = gid / 3; const int zplane = gid % 3; - if (pid >= np) + if (myslot >= np) return; - const float2 dst0 = _ACCESS(particles + 3 * pid + 0); - const float2 dst1 = _ACCESS(particles + 3 * pid + 1); - const float2 dst2 = _ACCESS(particles + 3 * pid + 2); + float2 dst0, dst1, dst2; + int soluteid, actualpid; + + { + CellEntry ce; + ce.pid = tex1Dfetch(texCellEntries, myslot); + + soluteid = ce.code.w; + + ce.code.w = 0; + actualpid = ce.pid; + + dst0 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 0); + dst1 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 1); + dst2 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 2); + } int scan1, scan2, ncandidates, spidbase; int deltaspid1, deltaspid2; @@ -295,7 +237,6 @@ namespace KernelsContact float xforce = 0, yforce = 0, zforce = 0; -#pragma unroll 3 for(int i = 0; i < ncandidates; ++i) { const int m1 = (int)(i >= scan1); @@ -303,6 +244,9 @@ namespace KernelsContact const int slot = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); assert(slot >= 0 && slot < ncellentries); + if (slot >= myslot) + continue; + CellEntry ce; ce.pid = tex1Dfetch(texCellEntries, slot); const int soluteid = ce.code.w; @@ -313,9 +257,6 @@ namespace KernelsContact const int spid = ce.pid; assert(spid >= 0 && spid < cnsolutes[soluteid]); - if (mysoluteid < soluteid || mysoluteid == soluteid && pid <= spid) - continue; - const int sentry = 3 * spid; const float2 stmp0 = _ACCESS(csolutes[soluteid] + sentry ); const float2 stmp1 = _ACCESS(csolutes[soluteid] + sentry + 1); @@ -352,7 +293,7 @@ namespace KernelsContact yr * (dst2.x - stmp2.x) + zr * (dst2.y - stmp2.y); - const float myrandnr = Logistic::mean0var1(seed, pid, spid); + const float myrandnr = Logistic::mean0var1(seed, myslot, slot); const float strength = lj + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; @@ -377,82 +318,36 @@ namespace KernelsContact atomicAdd(csolutesacc[soluteid] + sentry + 2, -zinteraction); } - atomicAdd(acc + 3 * pid + 0, xforce); - atomicAdd(acc + 3 * pid + 1, yforce); - atomicAdd(acc + 3 * pid + 2, zforce); + atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 0, xforce); + atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); + atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); for(int c = 0; c < 3; ++c) assert(!isnan(acc[3 * pid + c])); } -} - -void ComputeContact::bulk(std::vector wsolutes, cudaStream_t stream) -{ - NVTX_RANGE("Contact/bulk", NVTX_C6); - - if (wsolutes.size() == 0) - return; - - for(int i = 0; i < wsolutes.size(); ++i) - { - ParticlesWrap it = wsolutes[i]; - - if (it.n) - KernelsContact::bulk_3tpp<<< (3 * it.n + 127) / 128, 128, 0, stream >>> - ((float2 *)it.p, it.n, cellsentries.size, wsolutes.size(), (float *)it.a, local_trunk.get_float(), i); - - CUDA_CHECK(cudaPeekAtLastError()); - } -} - -namespace KernelsContact -{ - __constant__ int packstarts_padded[27], packcount[26]; - __constant__ Particle * packstates[26]; - __constant__ Acceleration * packresults[26]; - __global__ void halo(const int nparticles_padded, const int ncellentries, const int nsolutes, const float seed) + __global__ void halo(const float2 * halo, const int nhalo, const int nsolutes, const float seed, float * const acc) { - assert(blockDim.x * gridDim.x >= nparticles_padded); + assert(blockDim.x * gridDim.x >= nhalo); const int laneid = threadIdx.x & 0x1f; const int warpid = threadIdx.x >> 5; - const int localbase = 32 * (warpid + 4 * blockIdx.x); - const int pid = localbase + laneid; + const int unpackbase = 32 * (warpid + 4 * blockIdx.x); + const int nunpack = min(32, nhalo - unpackbase); - if (localbase >= nparticles_padded) - return; - - int nunpack; float2 dst0, dst1, dst2; - float * dst = NULL; - - { - const uint key9 = 9 * (localbase >= packstarts_padded[9]) + 9 * (localbase >= packstarts_padded[18]); - const uint key3 = 3 * (localbase >= packstarts_padded[key9 + 3]) + 3 * (localbase >= packstarts_padded[key9 + 6]); - const uint key1 = (localbase >= packstarts_padded[key9 + key3 + 1]) + (localbase >= packstarts_padded[key9 + key3 + 2]); - const int code = key9 + key3 + key1; - assert(code >= 0 && code < 26); - assert(localbase >= packstarts_padded[code] && localbase < packstarts_padded[code + 1]); - - const int unpackbase = localbase - packstarts_padded[code]; - assert (unpackbase >= 0); - assert(unpackbase < packcount[code]); - - nunpack = min(32, packcount[code] - unpackbase); - - if (nunpack == 0) - return; - - read_AOS6f((float2 *)(packstates[code] + unpackbase), nunpack, dst0, dst1, dst2); - - dst = (float*)(packresults[code] + unpackbase); - } + read_AOS6f((float2 *)(halo + unpackbase), nunpack, dst0, dst1, dst2); float xforce, yforce, zforce; - read_AOS3f(dst, nunpack, xforce, yforce, zforce); + read_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); + + const bool valid = + laneid < nunpack && + dst0.x >= -XOFFSET && + dst0.y >= -YOFFSET && + dst1.x >= -ZOFFSET ; - const int nzplanes = laneid < nunpack ? 3 : 0; + const int nzplanes = valid ? 3 : 0; for(int zplane = 0; zplane < nzplanes; ++zplane) { @@ -460,9 +355,7 @@ namespace KernelsContact int deltaspid1, deltaspid2; { - int xcenter = XOFFSET + (int)floorf(dst0.x); - xcenter = max(-1, min(XCELLS, xcenter)); - + const int xcenter = XOFFSET + (int)floorf(dst0.x); const int xstart = max(0, xcenter - 1); const int xcount = min(XCELLS, xcenter + 2) - xstart; @@ -471,11 +364,8 @@ namespace KernelsContact assert(xcount >= 0); - int ycenter = YOFFSET + (int)floorf(dst0.y); - ycenter = max(-1, min(YCELLS, ycenter)); - - int zcenter = ZOFFSET + (int)floorf(dst1.x); - zcenter = max(-1, min(ZCELLS, zcenter)); + const int ycenter = YOFFSET + (int)floorf(dst0.y); + const int zcenter = ZOFFSET + (int)floorf(dst1.x); const int zmy = zcenter - 1 + zplane; const bool zvalid = zmy >= 0 && zmy < ZCELLS; @@ -535,14 +425,6 @@ namespace KernelsContact const float2 stmp1 = _ACCESS(csolutes[soluteid] + sentry + 1); const float2 stmp2 = _ACCESS(csolutes[soluteid] + sentry + 2); -/* const bool invalidsrc = - stmp0.x < -XOFFSET || - stmp0.y < -YOFFSET || - stmp1.x < -ZOFFSET; - - if (invalidsrc) - continue; -*/ const float _xr = dst0.x - stmp0.x; const float _yr = dst0.y - stmp0.y; const float _zr = dst1.x - stmp1.x; @@ -574,7 +456,7 @@ namespace KernelsContact yr * (dst2.x - stmp2.x) + zr * (dst2.y - stmp2.y); - const float myrandnr = Logistic::mean0var1(seed, pid, spid); + const float myrandnr = Logistic::mean0var1(seed, unpackbase + laneid, spid); const float strength = lj + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; @@ -600,7 +482,7 @@ namespace KernelsContact } } - write_AOS3f(dst, nunpack, xforce, yforce, zforce); + write_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); } } @@ -608,46 +490,98 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) { NVTX_RANGE("Contact/halo", NVTX_C7); - int nremote_padded = 0; - + //collate halos { - int recvpackcount[26], recvpackstarts_padded[27]; + int c = 0; + for(int i = 0; i < 26; ++i) + c += halos[i].n; + + allhalos.resize(c); + allhalosacc.resize(c); + c = 0; for(int i = 0; i < 26; ++i) - recvpackcount[i] = halos[i].n; + { + CUDA_CHECK(cudaMemcpyAsync(allhalos.data + c, halos[i].p, sizeof(Particle) * halos[i].n, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemcpyAsync(allhalosacc.data + c, halos[i].a, sizeof(Acceleration) * halos[i].n, cudaMemcpyHostToDevice, stream)); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packcount, recvpackcount, - sizeof(recvpackcount), 0, cudaMemcpyHostToDevice, stream)); + c += halos[i].n; + } + } - recvpackstarts_padded[0] = 0; - for(int i = 0, s = 0; i < 26; ++i) - recvpackstarts_padded[i + 1] = (s += 32 * ((halos[i].n + 31) / 32)); + CUDA_CHECK(cudaPeekAtLastError()); - nremote_padded = recvpackstarts_padded[26]; + ParticlesWrap halowrap(allhalos.data, allhalos.size, allhalosacc.data); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packstarts_padded, recvpackstarts_padded, - sizeof(recvpackstarts_padded), 0, cudaMemcpyHostToDevice, stream)); + wsolutes.push_back(halowrap); - const Particle * recvpackstates[26]; + int ntotal = 0; - for(int i = 0; i < 26; ++i) - recvpackstates[i] = halos[i].p; + for(int i = 0; i < wsolutes.size(); ++i) + ntotal += wsolutes[i].n; - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packstates, recvpackstates, - sizeof(recvpackstates), 0, cudaMemcpyHostToDevice, stream)); + subindices.resize(ntotal); + cellsentries.resize(ntotal); - Acceleration * packresults[26]; + CUDA_CHECK(cudaMemsetAsync(cellscount.data, 0, sizeof(int) * cellscount.size, stream)); - for(int i = 0; i < 26; ++i) - packresults[i] = halos[i].a; +#ifndef NDEBUG + CUDA_CHECK(cudaMemsetAsync(cellsentries.data, 0xff, sizeof(int) * cellsentries.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(subindices.data, 0xff, sizeof(int) * subindices.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(compressed_cellscount.data, 0xff, sizeof(unsigned char) * compressed_cellscount.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(cellsstart.data, 0xff, sizeof(int) * cellsstart.capacity, stream)); +#endif + + CUDA_CHECK(cudaPeekAtLastError()); + + int ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; + + if (it.n) + subindex_local<<< (it.n + 127) / 128, 128, 0, stream >>> + (it.n, (float2 *)it.p, cellscount.data, subindices.data + ctr); + + ctr += it.n; + } + + compress_counts<<< (compressed_cellscount.size + 127) / 128, 128, 0, stream >>> + (compressed_cellscount.size, (int4 *)cellscount.data, (uchar4 *)compressed_cellscount.data); + + scan(compressed_cellscount.data, compressed_cellscount.size, stream, (uint *)cellsstart.data); + + ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; + + if (it.n) + KernelsContact::populate<<< (it.n + 127) / 128, 128, 0, stream >>> + (subindices.data + ctr, cellsstart.data, it.n, i, ntotal, (KernelsContact::CellEntry *)cellsentries.data); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packresults, packresults, - sizeof(packresults), 0, cudaMemcpyHostToDevice, stream)); + ctr += it.n; } - if(nremote_padded) - KernelsContact::halo<<< (nremote_padded + 127) / 128, 128, 0, stream>>> - (nremote_padded, cellsentries.size, nsolutes, local_trunk.get_float()); + CUDA_CHECK(cudaPeekAtLastError()); + + KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); + + KernelsContact::bulk_3tpp<<< (3 * cellsentries.size + 127) / 128, 128, 0, stream >>> + (cellsentries.size, wsolutes.size(), local_trunk.get_float()); + + KernelsContact::halo<<< (allhalos.size + 127) / 128, 128, 0, stream>>> + ((float2 *)allhalos.data, allhalos.size, wsolutes.size(), local_trunk.get_float(), (float *)allhalosacc.data); + + //split back halos + { + int c = 0; + for(int i = 0; i < 26; ++i) + { + CUDA_CHECK(cudaMemcpyAsync(halos[i].a, allhalosacc.data + c, sizeof(Acceleration) * halos[i].n, cudaMemcpyDeviceToHost, stream)); + c += halos[i].n; + } + } CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/contact.h b/mpi-dpd/contact.h index d02aa692e..5b5f44d66 100644 --- a/mpi-dpd/contact.h +++ b/mpi-dpd/contact.h @@ -21,23 +21,21 @@ class ComputeContact : public SoluteExchange::Visitor { - //cudaEvent_t evuploaded; - - int nsolutes; - + std::vector wsolutes; + SimpleDeviceBuffer subindices; SimpleDeviceBuffer compressed_cellscount; SimpleDeviceBuffer cellsentries, cellsstart, cellscount; + SimpleDeviceBuffer allhalos; + SimpleDeviceBuffer allhalosacc; Logistic::KISS local_trunk; - + public: ComputeContact(MPI_Comm comm); - void build_cells(std::vector wsolutes, cudaStream_t stream); - - void bulk(std::vector wsolutes, cudaStream_t stream); + void attach_bulk(std::vector wsolutes) { this->wsolutes = wsolutes; } /*override of SoluteExchange::Visitor::halo*/ void halo(ParticlesWrap solutes[26], cudaStream_t stream); diff --git a/mpi-dpd/redistribute-particles.cu b/mpi-dpd/redistribute-particles.cu index e31ec5d3b..e4a88480c 100644 --- a/mpi-dpd/redistribute-particles.cu +++ b/mpi-dpd/redistribute-particles.cu @@ -744,7 +744,7 @@ void RedistributeParticles::bulk(const int nparticles, int * const cellstarts, i subindices.resize(nparticles); if (nparticles) - subindex_local<<< (nparticles + 127) / 128, 128, 0, mystream>>> + subindex_local<<< (nparticles + 127) / 128, 128, 0, mystream>>> (nparticles, RedistributeParticlesKernels::texparticledata, cellcounts, subindices.data); /* #ifndef NDEBUG diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 12f309863..095077b70 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -374,9 +374,6 @@ void Simulation::_forces() CUDA_CHECK(cudaPeekAtLastError()); - if (contactforces) - contact.build_cells(wsolutes, mainstream); - dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, cells.start, cells.count, mainstream); @@ -409,7 +406,7 @@ void Simulation::_forces() fsi.bulk(wsolutes, mainstream); if (contactforces) - contact.bulk(wsolutes, mainstream); + contact.attach_bulk(wsolutes); CUDA_CHECK(cudaPeekAtLastError()); @@ -837,9 +834,6 @@ void Simulation::_lockstep() dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, cells.start, cells.count, mainstream); - if (contactforces) - contact.build_cells(wsolutes, mainstream); - solutex.post_p(mainstream, downloadstream); dpd.post(particles->xyzuvw.data, particles->size, mainstream, downloadstream); @@ -863,7 +857,7 @@ void Simulation::_lockstep() fsi.bulk(wsolutes, mainstream); if (contactforces) - contact.bulk(wsolutes, mainstream); + contact.attach_bulk(wsolutes); CUDA_CHECK(cudaPeekAtLastError()); diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index 92910c0b8..e2d63c03d 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -72,7 +72,7 @@ namespace SolutePUP __constant__ int coffsets[26]; - __global__ void scatter_indices(const float2 * const particles, const int nparticles, int * const counts, const bool contactforces) + __global__ void scatter_indices(const float2 * const particles, const int nparticles, int * const counts) { assert(blockDim.x * gridDim.x >= nparticles && blockDim.x == 128); @@ -104,13 +104,11 @@ namespace SolutePUP assert(fabs(s2.x) < 1e4); assert(fabs(s2.y) < 1e4); - const int mymargin = contactforces ? 15 : 1; - const int halocode[3] = { - -1 + (int)(s0.x >= -HXSIZE + mymargin) + (int)(s0.x >= HXSIZE - mymargin), - -1 + (int)(s0.y >= -HYSIZE + mymargin) + (int)(s0.y >= HYSIZE - mymargin), - -1 + (int)(s1.x >= -HZSIZE + mymargin) + (int)(s1.x >= HZSIZE - mymargin) + -1 + (int)(s0.x >= -HXSIZE + 1) + (int)(s0.x >= HXSIZE - 1), + -1 + (int)(s0.y >= -HYSIZE + 1) + (int)(s0.y >= HYSIZE - 1), + -1 + (int)(s1.x >= -HZSIZE + 1) + (int)(s1.x >= HZSIZE - 1) }; if (halocode[0] == 0 && halocode[1] == 0 && halocode[2] == 0) @@ -312,7 +310,7 @@ void SoluteExchange::_pack_attempt(cudaStream_t stream) { CUDA_CHECK(cudaMemcpyToSymbolAsync(SolutePUP::coffsets, packsoffset.data + 26 * i, sizeof(int) * 26, 0, cudaMemcpyDeviceToDevice, stream)); - SolutePUP::scatter_indices<<< (it.n + 127) / 128, 128, 0, stream >>>((float2 *)it.p, it.n, packscount.data + i * 26, contactforces); + SolutePUP::scatter_indices<<< (it.n + 127) / 128, 128, 0, stream >>>((float2 *)it.p, it.n, packscount.data + i * 26); } SolutePUP::tiny_scan<<< 1, 32, 0, stream >>>(packscount.data + i * 26, packsoffset.data + 26 * i, From 08ae3da039d7e825356cbb62b31a21f8dabe0c3a Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 15:01:50 +0100 Subject: [PATCH 53/63] bugfix on the new contact force --- mpi-dpd/contact.cu | 57 ++++++++++++++++++++++++++++++++-------------- 1 file changed, 40 insertions(+), 17 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 846bd3b45..217cae7e1 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -57,7 +57,7 @@ namespace KernelsContact } ComputeContact::ComputeContact(MPI_Comm comm): -cellsstart(KernelsContact::NCELLS), cellscount(KernelsContact::NCELLS), compressed_cellscount(KernelsContact::NCELLS) +cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16), compressed_cellscount(KernelsContact::NCELLS + 16) { int myrank; MPI_CHECK( MPI_Comm_rank(comm, &myrank)); @@ -154,8 +154,10 @@ namespace KernelsContact CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutesacc, as, sizeof(float *) * n, 0, cudaMemcpyHostToDevice, stream)); } - __global__ void bulk_3tpp(const int np, const int nsolutes, const float seed) + __global__ void bulk_3tpp(const int nsolutes, const float seed) { + const int np = tex1Dfetch(texCellsStart, XCELLS * YCELLS * ZCELLS); + assert(blockDim.x * gridDim.x >= np * 3); const int gid = threadIdx.x + blockDim.x * blockIdx.x; @@ -177,9 +179,16 @@ namespace KernelsContact ce.code.w = 0; actualpid = ce.pid; + assert(soluteid < nsolutes); + assert(actualpid >= 0 && actualpid < cnsolutes[soluteid]); + dst0 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 0); dst1 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 1); dst2 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 2); + + assert(dst0.x >= -XOFFSET && dst0.x < XOFFSET); + assert(dst0.y >= -YOFFSET && dst0.y < YOFFSET); + assert(dst1.x >= -ZOFFSET && dst1.x < ZOFFSET); } int scan1, scan2, ncandidates, spidbase; @@ -242,7 +251,7 @@ namespace KernelsContact const int m1 = (int)(i >= scan1); const int m2 = (int)(i >= scan2); const int slot = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - assert(slot >= 0 && slot < ncellentries); + assert(slot >= 0 && slot < np); if (slot >= myslot) continue; @@ -318,16 +327,19 @@ namespace KernelsContact atomicAdd(csolutesacc[soluteid] + sentry + 2, -zinteraction); } - atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 0, xforce); - atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); - atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); + const float xacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 0, xforce); + const float yacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); + const float zacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); - for(int c = 0; c < 3; ++c) - assert(!isnan(acc[3 * pid + c])); + assert(!isnan(xacc)); + assert(!isnan(yacc)); + assert(!isnan(zacc)); } __global__ void halo(const float2 * halo, const int nhalo, const int nsolutes, const float seed, float * const acc) { + const int nbulk = tex1Dfetch(texCellsStart, XCELLS * YCELLS * ZCELLS); + assert(blockDim.x * gridDim.x >= nhalo); const int laneid = threadIdx.x & 0x1f; @@ -341,11 +353,12 @@ namespace KernelsContact float xforce, yforce, zforce; read_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); - const bool valid = - laneid < nunpack && - dst0.x >= -XOFFSET && - dst0.y >= -YOFFSET && - dst1.x >= -ZOFFSET ; + const bool outside_plus = + dst0.x >= XOFFSET || + dst0.y >= YOFFSET || + dst1.x >= ZOFFSET ; + + const bool valid = laneid < nunpack && outside_plus; const int nzplanes = valid ? 3 : 0; @@ -410,7 +423,7 @@ namespace KernelsContact const int m2 = (int)(i >= scan2); const int slot = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - assert(slot >= 0 && slot < ncellentries); + assert(slot >= 0 && slot < nbulk); CellEntry ce; ce.pid = tex1Dfetch(texCellEntries, slot); const int soluteid = ce.code.w; @@ -568,10 +581,20 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); KernelsContact::bulk_3tpp<<< (3 * cellsentries.size + 127) / 128, 128, 0, stream >>> - (cellsentries.size, wsolutes.size(), local_trunk.get_float()); + (wsolutes.size(), local_trunk.get_float()); - KernelsContact::halo<<< (allhalos.size + 127) / 128, 128, 0, stream>>> - ((float2 *)allhalos.data, allhalos.size, wsolutes.size(), local_trunk.get_float(), (float *)allhalosacc.data); + ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; + + if (it.n) + KernelsContact::halo<<< (it.n + 127) / 128, 128, 0, stream>>> + ((float2 *)it.p, it.n, wsolutes.size(), local_trunk.get_float(), (float *)it.a); + + + ctr += it.n; + } //split back halos { From 2bd919b4eea4fb6f184f9b5cf2c0ff250247cbcd Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 15:18:08 +0100 Subject: [PATCH 54/63] bugfix pointed out by sergey --- mpi-dpd/dpd.cu | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mpi-dpd/dpd.cu b/mpi-dpd/dpd.cu index b8edcfe03..3eafd9145 100644 --- a/mpi-dpd/dpd.cu +++ b/mpi-dpd/dpd.cu @@ -27,6 +27,8 @@ sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), int myrank; MPI_CHECK(MPI_Comm_rank(cartcomm, &myrank)); + local_trunk = Logistic::KISS(9078 - 2 * myrank, 321 - myrank, 552, 456); + for(int i = 0; i < 26; ++i) { int d[3] = { (i + 2) % 3 - 1, (i / 3 + 2) % 3 - 1, (i / 9 + 2) % 3 - 1 }; @@ -75,8 +77,6 @@ sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), } } - - void ComputeDPD::local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, Acceleration * const a, const int * const cellsstart, const int * const cellscount, cudaStream_t stream) { From 5f83c1eba1bb6827d4b67c11b5c7795810bc5ccc Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 15:31:28 +0100 Subject: [PATCH 55/63] moving halo-bench to tests/ --- {halo-bench => tests/halo-bench}/Makefile | 0 {halo-bench => tests/halo-bench}/byte_latency.cpp | 0 {halo-bench => tests/halo-bench}/halo-exchanger.cu | 0 {halo-bench => tests/halo-bench}/halo_bench.cpp | 0 {halo-bench => tests/halo-bench}/hpm.cpp | 0 {halo-bench => tests/halo-bench}/mesh_distances.cpp | 0 {halo-bench => tests/halo-bench}/mesh_topo.cpp | 0 {halo-bench => tests/halo-bench}/osu_latency.c | 0 {halo-bench => tests/halo-bench}/osu_latency_rdp.cpp | 0 {halo-bench => tests/halo-bench}/scripts/env.sh | 0 {halo-bench => tests/halo-bench}/scripts/exp_12x12x2.sh | 0 {halo-bench => tests/halo-bench}/scripts/exp_14x14x4.sh | 0 {halo-bench => tests/halo-bench}/scripts/exp_3x3x3.sh | 0 {halo-bench => tests/halo-bench}/scripts/exp_3x3x3_new.sh | 0 {halo-bench => tests/halo-bench}/scripts/modules.sh | 0 {halo-bench => tests/halo-bench}/scripts/unsetenv.sh | 0 16 files changed, 0 insertions(+), 0 deletions(-) rename {halo-bench => tests/halo-bench}/Makefile (100%) rename {halo-bench => tests/halo-bench}/byte_latency.cpp (100%) rename {halo-bench => tests/halo-bench}/halo-exchanger.cu (100%) rename {halo-bench => tests/halo-bench}/halo_bench.cpp (100%) rename {halo-bench => tests/halo-bench}/hpm.cpp (100%) rename {halo-bench => tests/halo-bench}/mesh_distances.cpp (100%) rename {halo-bench => tests/halo-bench}/mesh_topo.cpp (100%) rename {halo-bench => tests/halo-bench}/osu_latency.c (100%) rename {halo-bench => tests/halo-bench}/osu_latency_rdp.cpp (100%) rename {halo-bench => tests/halo-bench}/scripts/env.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/exp_12x12x2.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/exp_14x14x4.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/exp_3x3x3.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/exp_3x3x3_new.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/modules.sh (100%) rename {halo-bench => tests/halo-bench}/scripts/unsetenv.sh (100%) diff --git a/halo-bench/Makefile b/tests/halo-bench/Makefile similarity index 100% rename from halo-bench/Makefile rename to tests/halo-bench/Makefile diff --git a/halo-bench/byte_latency.cpp b/tests/halo-bench/byte_latency.cpp similarity index 100% rename from halo-bench/byte_latency.cpp rename to tests/halo-bench/byte_latency.cpp diff --git a/halo-bench/halo-exchanger.cu b/tests/halo-bench/halo-exchanger.cu similarity index 100% rename from halo-bench/halo-exchanger.cu rename to tests/halo-bench/halo-exchanger.cu diff --git a/halo-bench/halo_bench.cpp b/tests/halo-bench/halo_bench.cpp similarity index 100% rename from halo-bench/halo_bench.cpp rename to tests/halo-bench/halo_bench.cpp diff --git a/halo-bench/hpm.cpp b/tests/halo-bench/hpm.cpp similarity index 100% rename from halo-bench/hpm.cpp rename to tests/halo-bench/hpm.cpp diff --git a/halo-bench/mesh_distances.cpp b/tests/halo-bench/mesh_distances.cpp similarity index 100% rename from halo-bench/mesh_distances.cpp rename to tests/halo-bench/mesh_distances.cpp diff --git a/halo-bench/mesh_topo.cpp b/tests/halo-bench/mesh_topo.cpp similarity index 100% rename from halo-bench/mesh_topo.cpp rename to tests/halo-bench/mesh_topo.cpp diff --git a/halo-bench/osu_latency.c b/tests/halo-bench/osu_latency.c similarity index 100% rename from halo-bench/osu_latency.c rename to tests/halo-bench/osu_latency.c diff --git a/halo-bench/osu_latency_rdp.cpp b/tests/halo-bench/osu_latency_rdp.cpp similarity index 100% rename from halo-bench/osu_latency_rdp.cpp rename to tests/halo-bench/osu_latency_rdp.cpp diff --git a/halo-bench/scripts/env.sh b/tests/halo-bench/scripts/env.sh similarity index 100% rename from halo-bench/scripts/env.sh rename to tests/halo-bench/scripts/env.sh diff --git a/halo-bench/scripts/exp_12x12x2.sh b/tests/halo-bench/scripts/exp_12x12x2.sh similarity index 100% rename from halo-bench/scripts/exp_12x12x2.sh rename to tests/halo-bench/scripts/exp_12x12x2.sh diff --git a/halo-bench/scripts/exp_14x14x4.sh b/tests/halo-bench/scripts/exp_14x14x4.sh similarity index 100% rename from halo-bench/scripts/exp_14x14x4.sh rename to tests/halo-bench/scripts/exp_14x14x4.sh diff --git a/halo-bench/scripts/exp_3x3x3.sh b/tests/halo-bench/scripts/exp_3x3x3.sh similarity index 100% rename from halo-bench/scripts/exp_3x3x3.sh rename to tests/halo-bench/scripts/exp_3x3x3.sh diff --git a/halo-bench/scripts/exp_3x3x3_new.sh b/tests/halo-bench/scripts/exp_3x3x3_new.sh similarity index 100% rename from halo-bench/scripts/exp_3x3x3_new.sh rename to tests/halo-bench/scripts/exp_3x3x3_new.sh diff --git a/halo-bench/scripts/modules.sh b/tests/halo-bench/scripts/modules.sh similarity index 100% rename from halo-bench/scripts/modules.sh rename to tests/halo-bench/scripts/modules.sh diff --git a/halo-bench/scripts/unsetenv.sh b/tests/halo-bench/scripts/unsetenv.sh similarity index 100% rename from halo-bench/scripts/unsetenv.sh rename to tests/halo-bench/scripts/unsetenv.sh From de16a6268316112b1a202fc33f12b2dcee5b4357 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 21:56:36 +0100 Subject: [PATCH 56/63] debugging with tests/contact --- mpi-dpd/Makefile | 6 +- mpi-dpd/contact.cu | 34 ++-- mpi-dpd/solute-exchange.cu | 2 +- tests/contact/Makefile | 26 +++ tests/contact/testcontact.cu | 301 +++++++++++++++++++++++++++++++++++ 5 files changed, 357 insertions(+), 12 deletions(-) create mode 100644 tests/contact/Makefile create mode 100644 tests/contact/testcontact.cu diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index 178d49f5d..cb86ec357 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -5,8 +5,10 @@ ARCH_VAL ?= compute_35 CODE_VAL ?= sm_35 NVCCFLAGS += -I$(HDF5_DIR)/include -Xcudafe "--diag_suppress=unrecognized_gcc_pragma" -NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g -DNDEBUG -CXXFLAGS += -L../cuda-dpd/dpd -L../cuda-rbc/ -L../cuda-ctc/ -O3 -g -std=c++11 -DNDEBUG +NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g +#-DNDEBUG +CXXFLAGS += -L../cuda-dpd/dpd -L../cuda-rbc/ -L../cuda-ctc/ -O3 -g -std=c++11 +#-DNDEBUG NVCCFLAGS += -I../cuda-dpd/dpd -I../cuda-rbc/ -I../cuda-ctc #NVCCFLAGS += -DREPORT_TOPOLOGY $(CRAY_PMI_INCLUDE_OPTS) #NVCCFLAGS += -DCUSTOM_REORDERING=1 diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 217cae7e1..fb69a2ab4 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -64,7 +64,7 @@ cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16) local_trunk = Logistic::KISS(7119 - myrank, 187 + myrank, 18278, 15674); - KernelsContact::Params params = { gammadpd, sigmaf, 1}; + KernelsContact::Params params = { 0*gammadpd, 0*sigmaf, 1}; CUDA_CHECK(cudaMemcpyToSymbol(KernelsContact::params, ¶ms, sizeof(params))); @@ -331,6 +331,9 @@ namespace KernelsContact const float yacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); const float zacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); + if (isnan(xacc)) + printf("ooops xacc %f for soluteid %d pid %d\n", xacc, soluteid, actualpid); + assert(!isnan(xacc)); assert(!isnan(yacc)); assert(!isnan(zacc)); @@ -348,21 +351,27 @@ namespace KernelsContact const int nunpack = min(32, nhalo - unpackbase); float2 dst0, dst1, dst2; - read_AOS6f((float2 *)(halo + unpackbase), nunpack, dst0, dst1, dst2); + read_AOS6f((float2 *)(halo + 3 * unpackbase), nunpack, dst0, dst1, dst2); float xforce, yforce, zforce; - read_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); + read_AOS3f(acc + 3 * unpackbase, nunpack, xforce, yforce, zforce); const bool outside_plus = dst0.x >= XOFFSET || dst0.y >= YOFFSET || dst1.x >= ZOFFSET ; - const bool valid = laneid < nunpack && outside_plus; + const bool inside_outerhalo = + dst0.x < XOFFSET + 1 && + dst0.y < YOFFSET + 1 && + dst1.x < ZOFFSET + 1 ; + + const bool valid = laneid < nunpack && outside_plus && inside_outerhalo; - const int nzplanes = valid ? 3 : 0; + if (!valid) + return; - for(int zplane = 0; zplane < nzplanes; ++zplane) + for(int zplane = 0; zplane < 3; ++zplane) { int scan1, scan2, ncandidates, spidbase; int deltaspid1, deltaspid2; @@ -495,7 +504,14 @@ namespace KernelsContact } } - write_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); + //write_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); + //if (valid) + { + assert(valid); + acc[3 * (unpackbase + laneid) + 0] = xforce; + acc[3 * (unpackbase + laneid) + 1] = yforce; + acc[3 * (unpackbase + laneid) + 2] = zforce; + } } } @@ -581,8 +597,8 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); KernelsContact::bulk_3tpp<<< (3 * cellsentries.size + 127) / 128, 128, 0, stream >>> - (wsolutes.size(), local_trunk.get_float()); - + (wsolutes.size(), local_trunk.get_float()); + ctr = 0; for(int i = 0; i < wsolutes.size(); ++i) { diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index e2d63c03d..e5690d178 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -505,7 +505,7 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) #ifndef NDEBUG CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); - CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0xff, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); + CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0 /*0xff*/, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); #endif MPI_Status status; diff --git a/tests/contact/Makefile b/tests/contact/Makefile new file mode 100644 index 000000000..e14c7da00 --- /dev/null +++ b/tests/contact/Makefile @@ -0,0 +1,26 @@ +-include ../../mpi-dpd/.cache.Makefile + +NVCC ?= nvcc -ccbin $(CXX) +ARCH_VAL ?= compute_35 +CODE_VAL ?= sm_35 + + +NVCCFLAGS += -I$(HDF5_DIR)/include -Xcudafe "--diag_suppress=unrecognized_gcc_pragma" +NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g -DNDEBUG -Xcompiler "-fopenmp" +CXXFLAGS += -L../../cuda-dpd/dpd -L../../cuda-rbc/ -L../../cuda-ctc/ -O3 -g -std=c++11 -DNDEBUG -fopenmp +NVCCFLAGS += -I../../cuda-dpd/dpd -I../../cuda-dpd/ -I../../cuda-rbc/ -I../../cuda-ctc -I../../mpi-dpd +LIBS = -lcuda-dpd -lcuda-rbc -lcuda-ctc -lcudart -ldl -lz -fopenmp + +testcontact: ../../mpi-dpd/test testcontact.o + make -C ../../mpi-dpd + rm -f ../../mpi-dpd/main.o + $(CXX) $(CXXFLAGS) testcontact.o ../../mpi-dpd/*.o $(LIBS) -o testcontact + +../test: + make -C ../ + cp ../*.o . + +testcontact.o: testcontact.cu ../../mpi-dpd/argument-parser.h ../../mpi-dpd/common.h ../../mpi-dpd/containers.h ../../mpi-dpd/contact.h ../../cuda-dpd/dpd-rng.h + $(NVCC) $(NVCCFLAGS) testcontact.cu -c -o testcontact.o + +.PHONY: ../test diff --git a/tests/contact/testcontact.cu b/tests/contact/testcontact.cu new file mode 100644 index 000000000..0cb8d29ac --- /dev/null +++ b/tests/contact/testcontact.cu @@ -0,0 +1,301 @@ +/* + * main.cu + * ctc PANDA + * + * Created by Dmitry Alexeev on Oct 20, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +/* + * main.cu + * Part of uDeviceX/mpi-dpd/ + * + * Created and authored by Diego Rossinelli on 2014-11-14. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + enum + { + XCELLS = XSIZE_SUBDOMAIN, + YCELLS = YSIZE_SUBDOMAIN, + ZCELLS = ZSIZE_SUBDOMAIN, + XOFFSET = XCELLS / 2, + YOFFSET = YCELLS / 2, + ZOFFSET = ZCELLS / 2 + }; +using namespace std; + +float tend, couette; +bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces, stress; +int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; + +LocalComm localcomm; + +static const float ljsigma = 0.5; +static const float ljsigma2 = ljsigma * ljsigma; + +template +inline float _viscosity_function(float x) +{ + return sqrtf(viscosity_function(x)); +} + +template<> inline float _viscosity_function<1>(float x) { return sqrtf(x); } +template<> inline float _viscosity_function<0>(float x){ return x; } + +int main(int argc, char ** argv) +{ + CUDA_CHECK(cudaSetDevice(0)); + CUDA_CHECK(cudaDeviceReset()); + + { + is_mps_enabled = false; + + const char * mps_variables[] = { + "CRAY_CUDA_MPS", + "CUDA_MPS", + "CRAY_CUDA_PROXY", + "CUDA_PROXY" + }; + + for(int i = 0; i < 4; ++i) + is_mps_enabled |= getenv(mps_variables[i])!= NULL && atoi(getenv(mps_variables[i])) != 0; + } + + int nranks, rank; + MPI_CHECK(MPI_Init(&argc, &argv)); + MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); + MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); + MPI_Comm activecomm = MPI_COMM_WORLD; + + bool reordering = true; + const char * env_reorder = getenv("MPICH_RANK_REORDER_METHOD"); + + MPI_Comm cartcomm; + int periods[] = {1, 1, 1}; + int ranks[] = {1, 1, 1}; + + + MPI_CHECK( MPI_Cart_create(activecomm, 3, ranks, periods, (int)reordering, &cartcomm) ); + activecomm = cartcomm; + + { + MPI_CHECK(MPI_Barrier(activecomm)); + localcomm.initialize(activecomm); + + MPI_CHECK(MPI_Barrier(activecomm)); + + // test here + const size_t myseed = 0x563d00cf;//time(NULL); + srand48(myseed); + printf("myseed: 0x%x\n", myseed); + //srand48(0); + + int n = 25e3; //l*l*l*dens; + vector ic(n); + vector acc(n); + for (int i=0; i gpuacc(n); + + Logistic::KISS local_trunk = Logistic::KISS(7119 - rank, 187 + rank, 18278, 15674); + + const double center[3] = { XSIZE_SUBDOMAIN/2, YSIZE_SUBDOMAIN/2, 0*ZSIZE_SUBDOMAIN/2} ;//YSIZE_SUBDOMAIN/2 -}; + const double halfwidth[3] = {XSIZE_SUBDOMAIN/10., YSIZE_SUBDOMAIN/10., ZSIZE_SUBDOMAIN / 10.}; + + for(int i = 0; i < n; ++i) + { + ic[i].x[0] = center[0] + halfwidth[0] * 2 * (drand48() - 0.5); + ic[i].x[1] = center[1] + halfwidth[1] * 2 * (drand48() - 0.5); + ic[i].x[2] = center[2] + halfwidth[2] * 2 * (drand48() - 0.5); + ic[i].u[0] = 0.5 - drand48(); + ic[i].u[1] = 0.5 - drand48(); + ic[i].u[2] = 0.5 - drand48(); + } + + if (false)//if (true) + { + ic.resize(2); + acc.resize(2); + gpuacc.resize(2); + n = 2; + //ic[0].x[0] = 24.413; ic[0].x[1] = +14.924; ic[0].x[2] = +7.326; + //ic[1].x[0] = +23.895; ic[1].x[1] = +14.887; ic[1].x[2] = +7.455 ; + + ic[0].x[0] = +23.670 ; ic[0].x[1] =+23.494; ic[0].x[2] =-18.980; + ic[1].x[0] = -23.851 ; ic[1].x[1] =+23.696; ic[1].x[2] =-18.790; + } + + float seed = local_trunk.get_float(); + +#pragma omp parallel for + for (int i=0; i= 1) + continue; + + const double invr2 = invrij * invrij; + const double t2 = ljsigma2 * invr2; + const double t4 = t2 * t2; + const double t6 = t4 * t2; + const double lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + + const double wr = _viscosity_function<0>(1.f - rij); + + const double xr = _xr * invrij; + const double yr = _yr * invrij; + const double zr = _zr * invrij; + + const double strength = lj; + + const double xinteraction = strength * xr; + const double yinteraction = strength * yr; + const double zinteraction = strength * zr; + + acc[i].a[0] += xinteraction; + acc[i].a[1] += yinteraction; + acc[i].a[2] += zinteraction; + } + + ParticleArray p; + p.resize(n); + + CUDA_CHECK( cudaMemcpy(p.xyzuvw.data, &ic[0], n * sizeof(Particle), cudaMemcpyHostToDevice) ); + CUDA_CHECK( cudaMemset(p.axayaz.data, 0, n * sizeof(Acceleration)) ); + std::vector wsolutes; + wsolutes.push_back(ParticlesWrap(p.xyzuvw.data, n, p.axayaz.data)); + + ComputeContact contact(cartcomm); + SoluteExchange solutex(cartcomm); + + solutex.attach_halocomputation(contact); + contact.attach_bulk(wsolutes); + + solutex.bind_solutes(wsolutes); + solutex.pack_p(0); + solutex.post_p(0, 0); + solutex.recv_p(0); + solutex.halo(0, 0); + solutex.post_a(); + solutex.recv_a(0); + + CUDA_CHECK( cudaMemcpy(&gpuacc[0], p.axayaz.data, n * sizeof(Acceleration), cudaMemcpyDeviceToHost) ); + + { + double fx = 0, fy = 0, fz = 0; + double hfx = 0, hfy = 0, hfz = 0; + + for (int i=0; i= tol && fabs(err) >= tol; + + if (failed) + printf("p %d c %d: %e ref: %e -> %e %e\n", i / 3, i % 3, res[i], ref[i], err, relerr); + + if (i % 3 == 2 && failed) + { + const int pid = i/3; + + const bool inside = + ic[i].x[0] >= -XOFFSET && ic[pid].x[0] < XOFFSET && + ic[i].x[1] >= -YOFFSET && ic[pid].x[1] < YOFFSET && + ic[i].x[2] >= -ZOFFSET && ic[pid].x[2] < ZOFFSET ; + + printf("%d: CPU [%+.3f %+.3f %+.3f] GPU [%+.3f %+.3f %+.3f] -> p %+.3f %+.3f %+.3f -> inside: %d\n", + i, acc[pid].a[0], acc[pid].a[1], acc[pid].a[2], + gpuacc[pid].a[0], gpuacc[pid].a[1], gpuacc[pid].a[2], + ic[pid].x[0], ic[pid].x[1], ic[pid].x[2], inside); + + failed = false; + } + + assert(fabs(relerr) < tol || fabs(err) < tol); + + l1 += fabs(err); + l1_rel += fabs(relerr); + + linf = std::max(linf, fabs(err)); + linf_rel = std::max(linf_rel, fabs(relerr)); + } + + printf("l-infinity errors: %.03e (absolute) %.03e (relative)\n", linf, linf_rel); + printf(" l-1 errors: %.03e (absolute) %.03e (relative)\n", l1, l1_rel); + } + } + + if (activecomm != cartcomm) + MPI_CHECK(MPI_Comm_free(&activecomm)); + + MPI_CHECK(MPI_Comm_free(&cartcomm)); + + MPI_CHECK(MPI_Finalize()); + + CUDA_CHECK(cudaDeviceSynchronize()); + + CUDA_CHECK(cudaDeviceReset()); + + return 0; +} \ No newline at end of file From cee1cc4c26cf32115c776e0e16e8936eb8549f59 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 22:22:59 +0100 Subject: [PATCH 57/63] subtle bugfix --- mpi-dpd/contact.cu | 12 ++++++------ tests/contact/testcontact.cu | 27 +++++++++++++-------------- 2 files changed, 19 insertions(+), 20 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index fb69a2ab4..1702ee9e8 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -333,7 +333,7 @@ namespace KernelsContact if (isnan(xacc)) printf("ooops xacc %f for soluteid %d pid %d\n", xacc, soluteid, actualpid); - + assert(!isnan(xacc)); assert(!isnan(yacc)); assert(!isnan(zacc)); @@ -358,14 +358,14 @@ namespace KernelsContact const bool outside_plus = dst0.x >= XOFFSET || - dst0.y >= YOFFSET || - dst1.x >= ZOFFSET ; + dst0.x >= -XOFFSET && dst0.y >= YOFFSET || + dst0.x >= -XOFFSET && dst0.y >= -YOFFSET && dst1.x >= ZOFFSET; - const bool inside_outerhalo = + const bool inside_outerhalo = dst0.x < XOFFSET + 1 && dst0.y < YOFFSET + 1 && dst1.x < ZOFFSET + 1 ; - + const bool valid = laneid < nunpack && outside_plus && inside_outerhalo; if (!valid) @@ -598,7 +598,7 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) KernelsContact::bulk_3tpp<<< (3 * cellsentries.size + 127) / 128, 128, 0, stream >>> (wsolutes.size(), local_trunk.get_float()); - + ctr = 0; for(int i = 0; i < wsolutes.size(); ++i) { diff --git a/tests/contact/testcontact.cu b/tests/contact/testcontact.cu index 0cb8d29ac..d221b35de 100644 --- a/tests/contact/testcontact.cu +++ b/tests/contact/testcontact.cu @@ -107,9 +107,8 @@ int main(int argc, char ** argv) const size_t myseed = 0x563d00cf;//time(NULL); srand48(myseed); printf("myseed: 0x%x\n", myseed); - //srand48(0); - int n = 25e3; //l*l*l*dens; + int n = 25e3; vector ic(n); vector acc(n); for (int i=0; i= tol && fabs(err) >= tol; + failed |= fabs(relerr) >= tol && fabs(err) >= tol; if (failed) printf("p %d c %d: %e ref: %e -> %e %e\n", i / 3, i % 3, res[i], ref[i], err, relerr); - + if (i % 3 == 2 && failed) { const int pid = i/3; const bool inside = - ic[i].x[0] >= -XOFFSET && ic[pid].x[0] < XOFFSET && - ic[i].x[1] >= -YOFFSET && ic[pid].x[1] < YOFFSET && + ic[i].x[0] >= -XOFFSET && ic[pid].x[0] < XOFFSET && + ic[i].x[1] >= -YOFFSET && ic[pid].x[1] < YOFFSET && ic[i].x[2] >= -ZOFFSET && ic[pid].x[2] < ZOFFSET ; - + printf("%d: CPU [%+.3f %+.3f %+.3f] GPU [%+.3f %+.3f %+.3f] -> p %+.3f %+.3f %+.3f -> inside: %d\n", i, acc[pid].a[0], acc[pid].a[1], acc[pid].a[2], gpuacc[pid].a[0], gpuacc[pid].a[1], gpuacc[pid].a[2], @@ -271,7 +270,7 @@ int main(int argc, char ** argv) failed = false; } - + assert(fabs(relerr) < tol || fabs(err) < tol); l1 += fabs(err); @@ -280,7 +279,7 @@ int main(int argc, char ** argv) linf = std::max(linf, fabs(err)); linf_rel = std::max(linf_rel, fabs(relerr)); } - + printf("l-infinity errors: %.03e (absolute) %.03e (relative)\n", linf, linf_rel); printf(" l-1 errors: %.03e (absolute) %.03e (relative)\n", l1, l1_rel); } From ede6b5d423e68d44e5e4f0ca9e49a335008bf64a Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Fri, 6 Nov 2015 22:26:48 +0100 Subject: [PATCH 58/63] back to simulations (before: tests/contact) --- mpi-dpd/Makefile | 6 ++---- mpi-dpd/contact.cu | 5 +---- 2 files changed, 3 insertions(+), 8 deletions(-) diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index cb86ec357..178d49f5d 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -5,10 +5,8 @@ ARCH_VAL ?= compute_35 CODE_VAL ?= sm_35 NVCCFLAGS += -I$(HDF5_DIR)/include -Xcudafe "--diag_suppress=unrecognized_gcc_pragma" -NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g -#-DNDEBUG -CXXFLAGS += -L../cuda-dpd/dpd -L../cuda-rbc/ -L../cuda-ctc/ -O3 -g -std=c++11 -#-DNDEBUG +NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g -DNDEBUG +CXXFLAGS += -L../cuda-dpd/dpd -L../cuda-rbc/ -L../cuda-ctc/ -O3 -g -std=c++11 -DNDEBUG NVCCFLAGS += -I../cuda-dpd/dpd -I../cuda-rbc/ -I../cuda-ctc #NVCCFLAGS += -DREPORT_TOPOLOGY $(CRAY_PMI_INCLUDE_OPTS) #NVCCFLAGS += -DCUSTOM_REORDERING=1 diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 1702ee9e8..ebee707d0 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -64,7 +64,7 @@ cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16) local_trunk = Logistic::KISS(7119 - myrank, 187 + myrank, 18278, 15674); - KernelsContact::Params params = { 0*gammadpd, 0*sigmaf, 1}; + KernelsContact::Params params = { gammadpd, sigmaf, 1}; CUDA_CHECK(cudaMemcpyToSymbol(KernelsContact::params, ¶ms, sizeof(params))); @@ -331,9 +331,6 @@ namespace KernelsContact const float yacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); const float zacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); - if (isnan(xacc)) - printf("ooops xacc %f for soluteid %d pid %d\n", xacc, soluteid, actualpid); - assert(!isnan(xacc)); assert(!isnan(yacc)); assert(!isnan(zacc)); From 82bce54a87b11f0a9eb538bcdfbf66d64f875415 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Sat, 7 Nov 2015 14:35:57 +0100 Subject: [PATCH 59/63] new contact: daint checks --- mpi-dpd/contact.cu | 41 +++++++++++++++++++++++--------------- mpi-dpd/simulation.cu | 12 +++++------ mpi-dpd/solute-exchange.cu | 2 +- 3 files changed, 32 insertions(+), 23 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index ebee707d0..b9b1453e0 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -54,6 +54,8 @@ namespace KernelsContact texCellEntries.mipmapFilterMode = cudaFilterModePoint; texCellEntries.normalized = 0; } + + __global__ void bulk_3tpp(const int nsolutes, const float seed); } ComputeContact::ComputeContact(MPI_Comm comm): @@ -69,6 +71,8 @@ cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16) CUDA_CHECK(cudaMemcpyToSymbol(KernelsContact::params, ¶ms, sizeof(params))); CUDA_CHECK(cudaPeekAtLastError()); + + CUDA_CHECK(cudaFuncSetCacheConfig(KernelsContact::bulk_3tpp , cudaFuncCachePreferL1)); } namespace KernelsContact @@ -168,23 +172,23 @@ namespace KernelsContact return; float2 dst0, dst1, dst2; - int soluteid, actualpid; + int mysoluteid, actualpid; { CellEntry ce; ce.pid = tex1Dfetch(texCellEntries, myslot); - soluteid = ce.code.w; + mysoluteid = ce.code.w; ce.code.w = 0; actualpid = ce.pid; - assert(soluteid < nsolutes); - assert(actualpid >= 0 && actualpid < cnsolutes[soluteid]); + assert(mysoluteid < nsolutes); + assert(actualpid >= 0 && actualpid < cnsolutes[mysoluteid]); - dst0 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 0); - dst1 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 1); - dst2 = _ACCESS(csolutes[soluteid] + 3 * actualpid + 2); + dst0 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 0); + dst1 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 1); + dst2 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 2); assert(dst0.x >= -XOFFSET && dst0.x < XOFFSET); assert(dst0.y >= -YOFFSET && dst0.y < YOFFSET); @@ -318,18 +322,18 @@ namespace KernelsContact assert(!isnan(yinteraction)); assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); + assert(fabs(xinteraction) < 1e5); + assert(fabs(yinteraction) < 1e5); + assert(fabs(zinteraction) < 1e5); atomicAdd(csolutesacc[soluteid] + sentry , -xinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 1, -yinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 2, -zinteraction); } - const float xacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 0, xforce); - const float yacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 1, yforce); - const float zacc = atomicAdd(csolutesacc[soluteid] + 3 * actualpid + 2, zforce); + const float xacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 0, xforce); + const float yacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 1, yforce); + const float zacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 2, zforce); assert(!isnan(xacc)); assert(!isnan(yacc)); @@ -491,9 +495,9 @@ namespace KernelsContact assert(!isnan(yinteraction)); assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); + assert(fabs(xinteraction) < 1e5); + assert(fabs(yinteraction) < 1e5); + assert(fabs(zinteraction) < 1e5); atomicAdd(csolutesacc[soluteid] + sentry , -xinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 1, -yinteraction); @@ -525,6 +529,11 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) allhalos.resize(c); allhalosacc.resize(c); +#ifndef NDEBUG + CUDA_CHECK(cudaMemsetAsync(allhalos.data, 0xff, sizeof(Particle) * allhalos.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(allhalosacc.data, 0xff, sizeof(Acceleration) * allhalosacc.capacity, stream)); +#endif + c = 0; for(int i = 0; i < 26; ++i) { diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 095077b70..e0d69eaed 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -399,15 +399,15 @@ void Simulation::_forces() solutex.recv_p(uploadstream); + if (contactforces) + contact.attach_bulk(wsolutes); + solutex.halo(uploadstream, mainstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); fsi.bulk(wsolutes, mainstream); - if (contactforces) - contact.attach_bulk(wsolutes); - CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) @@ -850,15 +850,15 @@ void Simulation::_lockstep() solutex.recv_p(uploadstream); + if (contactforces) + contact.attach_bulk(wsolutes); + solutex.halo(uploadstream, mainstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); fsi.bulk(wsolutes, mainstream); - if (contactforces) - contact.attach_bulk(wsolutes); - CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index e5690d178..e2d63c03d 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -505,7 +505,7 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) #ifndef NDEBUG CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); - CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0 /*0xff*/, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); + CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0xff, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); #endif MPI_Status status; From 51b404c6dc486f99282ebfb11f56942d4ee1dcba Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Sun, 8 Nov 2015 15:46:22 +0100 Subject: [PATCH 60/63] better cpu-gpu coordination for solutex --- mpi-dpd/contact.cu | 39 +------------------ mpi-dpd/contact.h | 4 +- mpi-dpd/fsi.cu | 24 +++++++++--- mpi-dpd/fsi.h | 2 +- mpi-dpd/simulation.cu | 8 ++-- mpi-dpd/solute-exchange.cu | 78 +++++++++++++++++++++++++++++--------- mpi-dpd/solute-exchange.h | 17 +++++---- 7 files changed, 96 insertions(+), 76 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index b9b1453e0..750bbcf23 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -516,38 +516,12 @@ namespace KernelsContact } } -void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) +void ComputeContact::halo(ParticlesWrap halowrap, cudaStream_t stream) { NVTX_RANGE("Contact/halo", NVTX_C7); - //collate halos - { - int c = 0; - for(int i = 0; i < 26; ++i) - c += halos[i].n; - - allhalos.resize(c); - allhalosacc.resize(c); - -#ifndef NDEBUG - CUDA_CHECK(cudaMemsetAsync(allhalos.data, 0xff, sizeof(Particle) * allhalos.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(allhalosacc.data, 0xff, sizeof(Acceleration) * allhalosacc.capacity, stream)); -#endif - - c = 0; - for(int i = 0; i < 26; ++i) - { - CUDA_CHECK(cudaMemcpyAsync(allhalos.data + c, halos[i].p, sizeof(Particle) * halos[i].n, cudaMemcpyHostToDevice, stream)); - CUDA_CHECK(cudaMemcpyAsync(allhalosacc.data + c, halos[i].a, sizeof(Acceleration) * halos[i].n, cudaMemcpyHostToDevice, stream)); - - c += halos[i].n; - } - } - CUDA_CHECK(cudaPeekAtLastError()); - ParticlesWrap halowrap(allhalos.data, allhalos.size, allhalosacc.data); - wsolutes.push_back(halowrap); int ntotal = 0; @@ -614,19 +588,8 @@ void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) KernelsContact::halo<<< (it.n + 127) / 128, 128, 0, stream>>> ((float2 *)it.p, it.n, wsolutes.size(), local_trunk.get_float(), (float *)it.a); - ctr += it.n; } - //split back halos - { - int c = 0; - for(int i = 0; i < 26; ++i) - { - CUDA_CHECK(cudaMemcpyAsync(halos[i].a, allhalosacc.data + c, sizeof(Acceleration) * halos[i].n, cudaMemcpyDeviceToHost, stream)); - c += halos[i].n; - } - } - CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/contact.h b/mpi-dpd/contact.h index 5b5f44d66..3242a72c9 100644 --- a/mpi-dpd/contact.h +++ b/mpi-dpd/contact.h @@ -26,8 +26,6 @@ class ComputeContact : public SoluteExchange::Visitor SimpleDeviceBuffer subindices; SimpleDeviceBuffer compressed_cellscount; SimpleDeviceBuffer cellsentries, cellsstart, cellscount; - SimpleDeviceBuffer allhalos; - SimpleDeviceBuffer allhalosacc; Logistic::KISS local_trunk; @@ -38,5 +36,5 @@ class ComputeContact : public SoluteExchange::Visitor void attach_bulk(std::vector wsolutes) { this->wsolutes = wsolutes; } /*override of SoluteExchange::Visitor::halo*/ - void halo(ParticlesWrap solutes[26], cudaStream_t stream); + void halo(ParticlesWrap allhalos, cudaStream_t stream); }; diff --git a/mpi-dpd/fsi.cu b/mpi-dpd/fsi.cu index 4c8cdbec8..a127c8c86 100644 --- a/mpi-dpd/fsi.cu +++ b/mpi-dpd/fsi.cu @@ -158,10 +158,14 @@ namespace KernelsFSI const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; + if (!(rij2 > 0)) + printf("oopsa rij2 %f : src = %f %f %f dst= %f %f %f\n", rij2, stmp0.x, stmp0.y, stmp1.x, dst0.x, dst0.y, dst1.x); + assert(rij2 > 0); + const float invrij = rsqrtf(rij2); const float rij = rij2 * invrij; - + if (rij2 >= 1) continue; @@ -270,7 +274,7 @@ void ComputeFSI::bulk(std::vector wsolutes, cudaStream_t stream) CUDA_CHECK(cudaPeekAtLastError()); } - +/* namespace KernelsFSI { __constant__ int packstarts_padded[27], packcount[26]; @@ -450,16 +454,16 @@ namespace KernelsFSI write_AOS3f(dst, nunpack, xforce, yforce, zforce); } -} + }*/ -void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) +void ComputeFSI::halo(ParticlesWrap halowrap, cudaStream_t stream) { NVTX_RANGE("FSI/halo", NVTX_C7); KernelsFSI::setup(wsolvent.p, wsolvent.n, wsolvent.cellsstart, wsolvent.cellscount); CUDA_CHECK(cudaPeekAtLastError()); - +/* int nremote_padded = 0; { @@ -504,6 +508,14 @@ void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) if(nremote_padded) KernelsFSI::interactions_halo<<< (nremote_padded + 127) / 128, 128, 0, stream>>> (nremote_padded, wsolvent.n, (float *)wsolvent.a, local_trunk.get_float()); - +*/ + //printf("before halo fsi\n"); + + if (halowrap.n) + KernelsFSI::interactions_3tpp<<< (3 * halowrap.n + 127) / 128, 128, 0, stream >>> + ((float2 *)halowrap.p, halowrap.n, wsolvent.n, (float *)halowrap.a, (float *)wsolvent.a, local_trunk.get_float()); + + /*CUDA_CHECK(cudaDeviceSynchronize()); + printf("after halo fsi\n");*/ CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/fsi.h b/mpi-dpd/fsi.h index 130e753da..735f6031b 100644 --- a/mpi-dpd/fsi.h +++ b/mpi-dpd/fsi.h @@ -36,5 +36,5 @@ class ComputeFSI : public SoluteExchange::Visitor void bulk(std::vector wsolutes, cudaStream_t stream); /*override of SoluteExchange::Visitor::halo*/ - void halo(ParticlesWrap solutes[26], cudaStream_t stream); + void halo(ParticlesWrap halowrap, cudaStream_t stream); }; diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index e0d69eaed..79507bbd3 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -397,12 +397,12 @@ void Simulation::_forces() dpd.recv(mainstream, uploadstream); - solutex.recv_p(uploadstream); + solutex.recv_p(uploadstream, mainstream); if (contactforces) contact.attach_bulk(wsolutes); - solutex.halo(uploadstream, mainstream); + solutex.halo(uploadstream, mainstream, downloadstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); @@ -848,12 +848,12 @@ void Simulation::_lockstep() dpd.recv(mainstream, uploadstream); - solutex.recv_p(uploadstream); + solutex.recv_p(uploadstream, mainstream); if (contactforces) contact.attach_bulk(wsolutes); - solutex.halo(uploadstream, mainstream); + solutex.halo(uploadstream, mainstream, downloadstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index e2d63c03d..cd485fc32 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -59,7 +59,8 @@ iterationcount(-1), packstotalstart(27), host_packstotalstart(27), host_packstot _adjust_packbuffers(); CUDA_CHECK(cudaEventCreateWithFlags(&evPpacked, cudaEventDisableTiming | cudaEventBlockingSync)); - CUDA_CHECK(cudaEventCreateWithFlags(&evAcomputed, cudaEventDisableTiming | cudaEventBlockingSync)); + CUDA_CHECK(cudaEventCreateWithFlags(&evAcomputed, cudaEventDisableTiming)); + CUDA_CHECK(cudaEventCreateWithFlags(&evAdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); CUDA_CHECK(cudaPeekAtLastError()); } @@ -351,7 +352,7 @@ void SoluteExchange::pack_p(cudaStream_t stream) if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/pack", NVTX_C4); + NVTX_RANGE("SOLUTEX/pack", NVTX_C4); ++iterationcount; @@ -371,7 +372,7 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) //consolidate the packing { - NVTX_RANGE("FSI/consolidate", NVTX_C5); + NVTX_RANGE("SOLUTEX/consolidate", NVTX_C5); CUDA_CHECK(cudaEventSynchronize(evPpacked)); @@ -446,7 +447,7 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) //post the sending of the packs { - NVTX_RANGE("FSI/send", NVTX_C6); + NVTX_RANGE("SOLUTEX/send", NVTX_C6); reqsendC.resize(26); @@ -485,12 +486,12 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) } } -void SoluteExchange::recv_p(cudaStream_t uploadstream) +void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestream) { if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/recv-p", NVTX_C7); + NVTX_RANGE("SOLUTEX/recv-p", NVTX_C7); _wait(reqrecvC); _wait(reqrecvP); @@ -504,7 +505,7 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) remote[i].preserve_resize(count); #ifndef NDEBUG - CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); + //CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0xff, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); #endif @@ -525,14 +526,40 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) _postrecvC(); - for(int i = 0; i < 26; ++i) + /*for(int i = 0; i < 26; ++i) CUDA_CHECK(cudaMemcpyAsync(remote[i].dstate.data, remote[i].hstate.data, sizeof(Particle) * remote[i].hstate.size, - cudaMemcpyHostToDevice, uploadstream)); + cudaMemcpyHostToDevice, uploadstream));*/ + + + //collate halos + { + int c = 0; + for(int i = 0; i < 26; ++i) + c += remote[i].hstate.size; + +#ifndef NDEBUG + CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0xff, sizeof(Acceleration) * allremotehalosacc.capacity, computestream)); + CUDA_CHECK(cudaMemsetAsync(allremotehalos.data, 0xff, sizeof(Particle) * allremotehalos.capacity, uploadstream)); +#endif + + allremotehalos.resize(c); + allremotehalosacc.resize(c); + + CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0, sizeof(Acceleration) * allremotehalosacc.size, computestream)); + + c = 0; + for(int i = 0; i < 26; ++i) + { + CUDA_CHECK(cudaMemcpyAsync(allremotehalos.data + c, remote[i].hstate.data, sizeof(Particle) * remote[i].hstate.size, cudaMemcpyHostToDevice, uploadstream)); + + c += remote[i].hstate.size; + } + } } -void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t stream) +void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t computestream, cudaStream_t downloadstream) { - NVTX_RANGE("FSI/halo", NVTX_C7); + NVTX_RANGE("SOLUTEX/halo", NVTX_C7); if (wsolutes.size() == 0) return; @@ -540,19 +567,35 @@ void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t stream) if (iterationcount) _wait(reqsendA); - ParticlesWrap halos[26]; + /*ParticlesWrap halos[26]; for(int i = 0; i < 26; ++i) halos[i] = ParticlesWrap(remote[i].dstate.data, remote[i].dstate.size, remote[i].result.devptr); + */ + ParticlesWrap halowrap(allremotehalos.data, allremotehalos.size, allremotehalosacc.data); CUDA_CHECK(cudaStreamSynchronize(uploadstream)); for(int i = 0; i < visitors.size(); ++i) - visitors[i]->halo(halos, stream); + visitors[i]->halo(halowrap, computestream); CUDA_CHECK(cudaPeekAtLastError()); - CUDA_CHECK(cudaEventRecord(evAcomputed, stream)); + CUDA_CHECK(cudaEventRecord(evAcomputed, computestream)); + + CUDA_CHECK(cudaStreamWaitEvent(downloadstream, evAcomputed, 0)); + + //split back halos + { + int c = 0; + for(int i = 0; i < 26; ++i) + { + CUDA_CHECK(cudaMemcpyAsync(remote[i].result.data, allremotehalosacc.data + c, sizeof(Acceleration) * remote[i].hstate.size, cudaMemcpyDeviceToHost, downloadstream)); + c += remote[i].hstate.size; + } + } + + CUDA_CHECK(cudaEventRecord(evAdownloaded, downloadstream)); for(int i = 0; i < 26; ++i) local[i].update(); @@ -567,9 +610,9 @@ void SoluteExchange::post_a() if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/send-a", NVTX_C1); + NVTX_RANGE("SOLUTEX/send-a", NVTX_C1); - CUDA_CHECK(cudaEventSynchronize(evAcomputed)); + CUDA_CHECK(cudaEventSynchronize(evAdownloaded)); reqsendA.resize(26); for(int i = 0; i < 26; ++i) @@ -632,7 +675,7 @@ void SoluteExchange::recv_a(cudaStream_t stream) if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/merge", NVTX_C2); + NVTX_RANGE("SOLUTEX/merge", NVTX_C2); { float * recvbags[26]; @@ -667,4 +710,5 @@ SoluteExchange::~SoluteExchange() CUDA_CHECK(cudaEventDestroy(evPpacked)); CUDA_CHECK(cudaEventDestroy(evAcomputed)); + CUDA_CHECK(cudaEventDestroy(evAdownloaded)); } diff --git a/mpi-dpd/solute-exchange.h b/mpi-dpd/solute-exchange.h index bf315c628..5776efe66 100644 --- a/mpi-dpd/solute-exchange.h +++ b/mpi-dpd/solute-exchange.h @@ -22,7 +22,7 @@ class SoluteExchange public: - struct Visitor { virtual void halo(ParticlesWrap solutehalos[26], cudaStream_t stream) = 0; }; + struct Visitor { virtual void halo(ParticlesWrap allhalos, cudaStream_t stream) = 0; }; protected: @@ -34,7 +34,7 @@ class SoluteExchange dims[3], periods[3], coords[3], myrank, recv_tags[26], recv_counts[26], send_counts[26]; - cudaEvent_t evPpacked, evAcomputed; + cudaEvent_t evPpacked, evAcomputed, evAdownloaded; SimpleDeviceBuffer packscount, packsstart, packsoffset, packstotalstart; PinnedHostBuffer host_packstotalstart, host_packstotalcount; @@ -77,14 +77,14 @@ class SoluteExchange public: - SimpleDeviceBuffer dstate; + //SimpleDeviceBuffer dstate; PinnedHostBuffer hstate; PinnedHostBuffer result; std::vector pmessage; void preserve_resize(int n) { - dstate.resize(n); + //dstate.resize(n); hstate.preserve_resize(n); result.resize(n); history.update(n); @@ -92,10 +92,13 @@ class SoluteExchange int expected() const { return (int)ceil(history.max() * 1.1); } - int capacity() const { assert(hstate.capacity == dstate.capacity); return dstate.capacity; } + int capacity() const { /*assert(hstate.capacity == dstate.capacity);*/ return hstate.capacity; } } remote[26]; + SimpleDeviceBuffer allremotehalos; + SimpleDeviceBuffer allremotehalosacc; + class LocalHalo { TimeSeriesWindow history; @@ -213,9 +216,9 @@ class SoluteExchange void post_p(cudaStream_t stream, cudaStream_t downloadstream); - void recv_p(cudaStream_t uploadstream); + void recv_p(cudaStream_t uploadstream, cudaStream_t computestream); - void halo(cudaStream_t uploadstream, cudaStream_t stream); + void halo(cudaStream_t uploadstream, cudaStream_t computestream, cudaStream_t downloadstream); void post_a(); From 1c59f89a158e17db4aac61f6a4154f04850df309 Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Sun, 8 Nov 2015 20:18:33 +0100 Subject: [PATCH 61/63] cleanup --- mpi-dpd/contact.cu | 13 +- mpi-dpd/fsi.cu | 237 +------------------------------------ mpi-dpd/solute-exchange.cu | 35 ++---- mpi-dpd/solute-exchange.h | 22 ++-- 4 files changed, 28 insertions(+), 279 deletions(-) diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 750bbcf23..d91d80e60 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -158,7 +158,7 @@ namespace KernelsContact CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutesacc, as, sizeof(float *) * n, 0, cudaMemcpyHostToDevice, stream)); } - __global__ void bulk_3tpp(const int nsolutes, const float seed) + __global__ __launch_bounds__(128, 10) void bulk_3tpp(const int nsolutes, const float seed) { const int np = tex1Dfetch(texCellsStart, XCELLS * YCELLS * ZCELLS); @@ -505,14 +505,9 @@ namespace KernelsContact } } - //write_AOS3f(acc + unpackbase, nunpack, xforce, yforce, zforce); - //if (valid) - { - assert(valid); - acc[3 * (unpackbase + laneid) + 0] = xforce; - acc[3 * (unpackbase + laneid) + 1] = yforce; - acc[3 * (unpackbase + laneid) + 2] = zforce; - } + acc[3 * (unpackbase + laneid) + 0] = xforce; + acc[3 * (unpackbase + laneid) + 1] = yforce; + acc[3 * (unpackbase + laneid) + 2] = zforce; } } diff --git a/mpi-dpd/fsi.cu b/mpi-dpd/fsi.cu index a127c8c86..cb907000e 100644 --- a/mpi-dpd/fsi.cu +++ b/mpi-dpd/fsi.cu @@ -157,15 +157,12 @@ namespace KernelsFSI const float _zr = dst1.x - stmp1.x; const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; - - if (!(rij2 > 0)) - printf("oopsa rij2 %f : src = %f %f %f dst= %f %f %f\n", rij2, stmp0.x, stmp0.y, stmp1.x, dst0.x, dst0.y, dst1.x); assert(rij2 > 0); - + const float invrij = rsqrtf(rij2); const float rij = rij2 * invrij; - + if (rij2 >= 1) continue; @@ -274,187 +271,6 @@ void ComputeFSI::bulk(std::vector wsolutes, cudaStream_t stream) CUDA_CHECK(cudaPeekAtLastError()); } -/* -namespace KernelsFSI -{ - __constant__ int packstarts_padded[27], packcount[26]; - __constant__ Particle * packstates[26]; - __constant__ Acceleration * packresults[26]; - - __global__ void interactions_halo(const int nparticles_padded, const int nsolvent, float * const accsolvent, const float seed) - { - assert(blockDim.x * gridDim.x >= nparticles_padded); - - const int laneid = threadIdx.x & 0x1f; - const int warpid = threadIdx.x >> 5; - const int localbase = 32 * (warpid + 4 * blockIdx.x); - const int pid = localbase + laneid; - - if (localbase >= nparticles_padded) - return; - - int nunpack; - float2 dst0, dst1, dst2; - float * dst = NULL; - - { - const uint key9 = 9 * (localbase >= packstarts_padded[9]) + 9 * (localbase >= packstarts_padded[18]); - const uint key3 = 3 * (localbase >= packstarts_padded[key9 + 3]) + 3 * (localbase >= packstarts_padded[key9 + 6]); - const uint key1 = (localbase >= packstarts_padded[key9 + key3 + 1]) + (localbase >= packstarts_padded[key9 + key3 + 2]); - const int code = key9 + key3 + key1; - assert(code >= 0 && code < 26); - assert(localbase >= packstarts_padded[code] && localbase < packstarts_padded[code + 1]); - - const int unpackbase = localbase - packstarts_padded[code]; - assert (unpackbase >= 0); - assert(unpackbase < packcount[code]); - - nunpack = min(32, packcount[code] - unpackbase); - - if (nunpack == 0) - return; - - read_AOS6f((float2 *)(packstates[code] + unpackbase), nunpack, dst0, dst1, dst2); - - dst = (float*)(packresults[code] + unpackbase); - } - - float xforce = 0, yforce = 0, zforce = 0; - - const int nzplanes = laneid < nunpack ? 3 : 0; - - for(int zplane = 0; zplane < nzplanes; ++zplane) - { - int scan1, scan2, ncandidates, spidbase; - int deltaspid1, deltaspid2; - - { - enum - { - XCELLS = XSIZE_SUBDOMAIN, - YCELLS = YSIZE_SUBDOMAIN, - ZCELLS = ZSIZE_SUBDOMAIN, - XOFFSET = XCELLS / 2, - YOFFSET = YCELLS / 2, - ZOFFSET = ZCELLS / 2 - }; - - const int NCELLS = XSIZE_SUBDOMAIN * YSIZE_SUBDOMAIN * ZSIZE_SUBDOMAIN; - const int xcenter = XOFFSET + (int)floorf(dst0.x); - const int xstart = max(0, xcenter - 1); - const int xcount = min(XCELLS, xcenter + 2) - xstart; - - if (xcenter - 1 >= XCELLS || xcenter + 2 <= 0) - continue; - - assert(xcount >= 0); - - const int ycenter = YOFFSET + (int)floorf(dst0.y); - - const int zcenter = ZOFFSET + (int)floorf(dst1.x); - const int zmy = zcenter - 1 + zplane; - const bool zvalid = zmy >= 0 && zmy < ZCELLS; - - int count0 = 0, count1 = 0, count2 = 0; - - if (zvalid && ycenter - 1 >= 0 && ycenter - 1 < YCELLS) - { - const int cid0 = xstart + XCELLS * (ycenter - 1 + YCELLS * zmy); - assert(cid0 >= 0 && cid0 + xcount <= NCELLS); - spidbase = tex1Dfetch(texCellsStart, cid0); - count0 = ((cid0 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid0 + xcount)) - spidbase; - } - - if (zvalid && ycenter >= 0 && ycenter < YCELLS) - { - const int cid1 = xstart + XCELLS * (ycenter + YCELLS * zmy); - assert(cid1 >= 0 && cid1 + xcount <= NCELLS); - deltaspid1 = tex1Dfetch(texCellsStart, cid1); - count1 = ((cid1 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid1 + xcount)) - deltaspid1; - } - - if (zvalid && ycenter + 1 >= 0 && ycenter + 1 < YCELLS) - { - const int cid2 = xstart + XCELLS * (ycenter + 1 + YCELLS * zmy); - deltaspid2 = tex1Dfetch(texCellsStart, cid2); - assert(cid2 >= 0 && cid2 + xcount <= NCELLS); - count2 = ((cid2 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid2 + xcount)) - deltaspid2; - } - - scan1 = count0; - scan2 = count0 + count1; - ncandidates = scan2 + count2; - - deltaspid1 -= scan1; - deltaspid2 -= scan2; - } - - for(int i = 0; i < ncandidates; ++i) - { - const int m1 = (int)(i >= scan1); - const int m2 = (int)(i >= scan2); - const int spid = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - - assert(spid >= 0 && spid < nsolvent); - - const int sentry = 3 * spid; - const float2 stmp0 = tex1Dfetch(texSolventParticles, sentry ); - const float2 stmp1 = tex1Dfetch(texSolventParticles, sentry + 1); - const float2 stmp2 = tex1Dfetch(texSolventParticles, sentry + 2); - - const float _xr = dst0.x - stmp0.x; - const float _yr = dst0.y - stmp0.y; - const float _zr = dst1.x - stmp1.x; - - const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; - - const float invrij = rsqrtf(rij2); - - const float rij = rij2 * invrij; - - if (rij2 >= 1) - continue; - - const float argwr = 1.f - rij; - const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(argwr); - - const float xr = _xr * invrij; - const float yr = _yr * invrij; - const float zr = _zr * invrij; - - const float rdotv = - xr * (dst1.y - stmp1.y) + - yr * (dst2.x - stmp2.x) + - zr * (dst2.y - stmp2.y); - - const float myrandnr = Logistic::mean0var1(seed, pid, spid); - - const float strength = params.aij * argwr + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; - - const float xinteraction = strength * xr; - const float yinteraction = strength * yr; - const float zinteraction = strength * zr; - - xforce += xinteraction; - yforce += yinteraction; - zforce += zinteraction; - - assert(!isnan(xinteraction)); - assert(!isnan(yinteraction)); - assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); - - atomicAdd(accsolvent + sentry , -xinteraction); - atomicAdd(accsolvent + sentry + 1, -yinteraction); - atomicAdd(accsolvent + sentry + 2, -zinteraction); - } - } - - write_AOS3f(dst, nunpack, xforce, yforce, zforce); - } - }*/ void ComputeFSI::halo(ParticlesWrap halowrap, cudaStream_t stream) { @@ -463,59 +279,10 @@ void ComputeFSI::halo(ParticlesWrap halowrap, cudaStream_t stream) KernelsFSI::setup(wsolvent.p, wsolvent.n, wsolvent.cellsstart, wsolvent.cellscount); CUDA_CHECK(cudaPeekAtLastError()); -/* - int nremote_padded = 0; - - { - int recvpackcount[26], recvpackstarts_padded[27]; - - for(int i = 0; i < 26; ++i) - recvpackcount[i] = halos[i].n; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packcount, recvpackcount, - sizeof(recvpackcount), 0, cudaMemcpyHostToDevice, stream)); - - recvpackstarts_padded[0] = 0; - for(int i = 0, s = 0; i < 26; ++i) - recvpackstarts_padded[i + 1] = (s += 32 * ((halos[i].n + 31) / 32)); - - nremote_padded = recvpackstarts_padded[26]; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packstarts_padded, recvpackstarts_padded, - sizeof(recvpackstarts_padded), 0, cudaMemcpyHostToDevice, stream)); - } - - { - const Particle * recvpackstates[26]; - - for(int i = 0; i < 26; ++i) - recvpackstates[i] = halos[i].p; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packstates, recvpackstates, - sizeof(recvpackstates), 0, cudaMemcpyHostToDevice, stream)); - } - - { - Acceleration * packresults[26]; - - for(int i = 0; i < 26; ++i) - packresults[i] = halos[i].a; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packresults, packresults, - sizeof(packresults), 0, cudaMemcpyHostToDevice, stream)); - } - if(nremote_padded) - KernelsFSI::interactions_halo<<< (nremote_padded + 127) / 128, 128, 0, stream>>> - (nremote_padded, wsolvent.n, (float *)wsolvent.a, local_trunk.get_float()); -*/ - //printf("before halo fsi\n"); - if (halowrap.n) KernelsFSI::interactions_3tpp<<< (3 * halowrap.n + 127) / 128, 128, 0, stream >>> ((float2 *)halowrap.p, halowrap.n, wsolvent.n, (float *)halowrap.a, (float *)wsolvent.a, local_trunk.get_float()); - /*CUDA_CHECK(cudaDeviceSynchronize()); - printf("after halo fsi\n");*/ CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index cd485fc32..0625698e9 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -492,7 +492,7 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestrea return; NVTX_RANGE("SOLUTEX/recv-p", NVTX_C7); - + _wait(reqrecvC); _wait(reqrecvP); @@ -505,7 +505,6 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestrea remote[i].preserve_resize(count); #ifndef NDEBUG - //CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0xff, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); #endif @@ -525,12 +524,7 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestrea } _postrecvC(); - - /*for(int i = 0; i < 26; ++i) - CUDA_CHECK(cudaMemcpyAsync(remote[i].dstate.data, remote[i].hstate.data, sizeof(Particle) * remote[i].hstate.size, - cudaMemcpyHostToDevice, uploadstream));*/ - //collate halos { int c = 0; @@ -541,12 +535,12 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestrea CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0xff, sizeof(Acceleration) * allremotehalosacc.capacity, computestream)); CUDA_CHECK(cudaMemsetAsync(allremotehalos.data, 0xff, sizeof(Particle) * allremotehalos.capacity, uploadstream)); #endif - + allremotehalos.resize(c); allremotehalosacc.resize(c); CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0, sizeof(Acceleration) * allremotehalosacc.size, computestream)); - + c = 0; for(int i = 0; i < 26; ++i) { @@ -563,28 +557,23 @@ void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t computestream, if (wsolutes.size() == 0) return; - + if (iterationcount) _wait(reqsendA); - - /*ParticlesWrap halos[26]; - - for(int i = 0; i < 26; ++i) - halos[i] = ParticlesWrap(remote[i].dstate.data, remote[i].dstate.size, remote[i].result.devptr); - */ + ParticlesWrap halowrap(allremotehalos.data, allremotehalos.size, allremotehalosacc.data); - + CUDA_CHECK(cudaStreamSynchronize(uploadstream)); - + for(int i = 0; i < visitors.size(); ++i) visitors[i]->halo(halowrap, computestream); - + CUDA_CHECK(cudaPeekAtLastError()); - + CUDA_CHECK(cudaEventRecord(evAcomputed, computestream)); CUDA_CHECK(cudaStreamWaitEvent(downloadstream, evAcomputed, 0)); - + //split back halos { int c = 0; @@ -596,10 +585,10 @@ void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t computestream, } CUDA_CHECK(cudaEventRecord(evAdownloaded, downloadstream)); - + for(int i = 0; i < 26; ++i) local[i].update(); - + #ifndef _DUMBCRAY_ _postrecvP(); #endif diff --git a/mpi-dpd/solute-exchange.h b/mpi-dpd/solute-exchange.h index 5776efe66..a522172a7 100644 --- a/mpi-dpd/solute-exchange.h +++ b/mpi-dpd/solute-exchange.h @@ -19,11 +19,11 @@ class SoluteExchange { enum { TAGBASE_C = 113, TAGBASE_P = 365, TAGBASE_A = 668, TAGBASE_P2 = 1055, TAGBASE_A2 = 1501 }; - + public: - + struct Visitor { virtual void halo(ParticlesWrap allhalos, cudaStream_t stream) = 0; }; - + protected: MPI_Comm cartcomm; @@ -38,16 +38,16 @@ class SoluteExchange SimpleDeviceBuffer packscount, packsstart, packsoffset, packstotalstart; PinnedHostBuffer host_packstotalstart, host_packstotalcount; - + SimpleDeviceBuffer packbuf; PinnedHostBuffer host_packbuf; - + std::vector wsolutes; - + std::vector reqsendC, reqrecvC, reqsendP, reqrecvP, reqsendA, reqrecvA; std::vector visitors; - + class TimeSeriesWindow { static const int N = 200; @@ -77,14 +77,12 @@ class SoluteExchange public: - //SimpleDeviceBuffer dstate; PinnedHostBuffer hstate; PinnedHostBuffer result; std::vector pmessage; void preserve_resize(int n) { - //dstate.resize(n); hstate.preserve_resize(n); result.resize(n); history.update(n); @@ -92,7 +90,7 @@ class SoluteExchange int expected() const { return (int)ceil(history.max() * 1.1); } - int capacity() const { /*assert(hstate.capacity == dstate.capacity);*/ return hstate.capacity; } + int capacity() const { return hstate.capacity; } } remote[26]; @@ -205,7 +203,7 @@ class SoluteExchange void _pack_attempt(cudaStream_t stream); public: - + SoluteExchange(MPI_Comm cartcomm); void bind_solutes(std::vector wsolutes) { this->wsolutes = wsolutes; } @@ -217,7 +215,7 @@ class SoluteExchange void post_p(cudaStream_t stream, cudaStream_t downloadstream); void recv_p(cudaStream_t uploadstream, cudaStream_t computestream); - + void halo(cudaStream_t uploadstream, cudaStream_t computestream, cudaStream_t downloadstream); void post_a(); From 25d65afad4579f23b53863ad42dcde427dc3dc7a Mon Sep 17 00:00:00 2001 From: Diego Rossinelli Date: Mon, 9 Nov 2015 10:54:32 +0100 Subject: [PATCH 62/63] some documenting sentences --- README.md | 19 +++++++++--------- mpi-dpd/README.md | 51 +++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 61 insertions(+), 9 deletions(-) create mode 100644 mpi-dpd/README.md diff --git a/README.md b/README.md index 09588d229..54d26c1fc 100644 --- a/README.md +++ b/README.md @@ -1,13 +1,14 @@ uDeviceX comes with the GPLv2 LICENSE. === The uDeviceX folders is organized into: -balaprep: preprocessing tool for assigning ranks to compute nodes. -cell-placement: preprocessing tool for generating the initial RBC/CTC displacement -cuda-ctc: code for the CTC model -cuda-dpd: code for the DPD interactions -cuda-rbc: code for the RBC model -device-gen: preprocessing tool to generate new device geometries -halo-bench: OSU-like benchmark to measure latency and bandwidth across the MPI ranks. -mpi-dpd: the simulation code -proof-of-concept: tests and hacks. +* balaprep: preprocessing tool for assigning ranks to compute nodes. +* cell-placement: preprocessing tool for generating the initial RBC/CTC displacement +* cuda-ctc: code for the CTC model +* cuda-dpd: code for the DPD interactions +* cuda-rbc: code for the RBC model +* device-gen: preprocessing tool to generate new device geometries +* mpi-dpd: the simulation code +* postprocessing: auxiliary tools to extract quantities from simulations the output +* proof-of-concept: tests and hacks +* tests: accuracy and performance tests diff --git a/mpi-dpd/README.md b/mpi-dpd/README.md new file mode 100644 index 000000000..146275372 --- /dev/null +++ b/mpi-dpd/README.md @@ -0,0 +1,51 @@ +The main folder: mpi-dpd +======= + +This is the "main" folder containing the object files orchestrating the various kernels. +The generated object files are the following ones: + +* common.o: global simulation parameters, common datastructures like device arrays, cell lists etc. +* contact.o: computation of the contact/lubrication force across "touching" solute particles +* containers.o: particle arrays, collections encapsulating the data of RBCs and CTCs +* dpd.o: code coordinating the computation of cuda-dpd +* fsi.o: computation of the "flow-structure interaction" force between solvent and solute particles +* io.o: data dumps in XYZ, PLY (for cells), H5Part, HDF5 structured grids +* main.o: home sweet home +* minmax.o: computation of the extent of an array of RBCs or CTCs +* redistancing.o: computation of the distance transform for the implicit description of the wall geometry +* redistribute-particles.o: redistribution of the solvent across the MPI ranks once particles have moved +* redistribute-rbcs.o: redistribution of RBCs across MPI ranks once they have moved +* scan.o: computation of the prefix sum for the solvent cell lists count +* simulation.o: simulation "driver" coordinating the other object files, except for main.cu +* solute-exchange.o: exchange the "halo" solute particles across the MPI ranks close by to compute FSI and contact forces. +* solvent-exchange.o: exchange the "halo" solvent particles across the MPI ranks to compute the DPD interactions +* wall.o: computation of the particles interacting with the no-slip boundary conditions of the wall + +Compiling uDeviceX +------------- +The makefile will check for a .cache.Makefile, that can be optionally put in this folder. +For example my .cache.Makefile on Piz Daint is + +`h5part = 0 +NVCC = nvcc -I$(CRAY_MPICH2_DIR)/include -L$(CRAY_MPICH2_DIR)/lib -I/scratch/daint/diegor/h5part/include -I$(HDF5_DIR)/include -I/users/diegor/vtk/install/include/vtk-6.2 -I/users/diegor/h5part/include/ +CXX = CC $(CRAY_CUDATOOLKIT_POST_LINK_OPTS) $(CRAY_CUDATOOLKIT_INCLUDE_OPTS) -L/users/diegor/h5part/lib -L$(HDF5_DIR)/lib -L/users/diegor/vtk/install/lib` + +To clean uDeviceX entirely (mpi-dpd, cuda-dpd, cuda-rbc, cuda-ctc): +`make cleanall` + +To just cleanup the mpi-dpd folder: +`make clean` + +To compile: +`make -j` + +Running uDeviceX +----------- +Running uDeviceX consists of these steps: + +1. Generation of the geometry file (optional), the file should always be named `sdf.dat` (as Signed Distance Function). +See the folder `device-gen`. +2. Generation of the initial positioning of the RBCs and CTCs (optional). The IC files should be called +`rbcs-ic.txt and ctcs-ic.txt` See the folder `cell-placement`. +3. Execution of uDeviceX, for example `mpirun ./test 4 4 2 -walls -couette=1 -tend=5e4 -steps_per_dump=1000 -rbcs -contactforces` +4. Post processing of the simulation data (optional), for example `ls ./stress/* -rt1 | tail -n 50 | mpirun -n 32 -N 1 ../postprocessing/stress/stress -origin=0,0,5 -extent=192,192,85 -project=1,1,0 > stress-profile.txt` From 874938d9c16bdc093ad808644a6b0febb3d5954e Mon Sep 17 00:00:00 2001 From: Dmitry Alexeev Date: Tue, 1 Mar 2016 16:52:45 +0100 Subject: [PATCH 63/63] control and sample --- cuda-rbc/rbc-cuda.cu | 23 +- mpi-dpd/Makefile | 3 +- mpi-dpd/dumper.cu | 24 +- mpi-dpd/dumper.h | 1 + mpi-dpd/io.h | 8 +- mpi-dpd/simulation.cu | 615 ++++++++++++++++++++------------------- mpi-dpd/simulation.h | 5 + mpi-dpd/velcontroller.cu | 185 ++++++++++++ mpi-dpd/velcontroller.h | 51 ++++ mpi-dpd/velsampler.cu | 97 ++++++ mpi-dpd/velsampler.h | 49 ++++ 11 files changed, 746 insertions(+), 315 deletions(-) create mode 100644 mpi-dpd/velcontroller.cu create mode 100644 mpi-dpd/velcontroller.h create mode 100644 mpi-dpd/velsampler.cu create mode 100644 mpi-dpd/velsampler.h diff --git a/cuda-rbc/rbc-cuda.cu b/cuda-rbc/rbc-cuda.cu index f0555df51..bca181b94 100644 --- a/cuda-rbc/rbc-cuda.cu +++ b/cuda-rbc/rbc-cuda.cu @@ -273,12 +273,6 @@ namespace CudaRBC xyzuvw_host[6*i+5] = 0; } - int *dummy; - if ( cudaMalloc(&dummy, sizeof(int)) == cudaErrorDevicesUnavailable ) - return; - else - CUDA_CHECK(cudaFree(dummy)); - CUDA_CHECK( cudaMalloc(&orig_xyzuvw, nvertices * 6 * sizeof(float)) ); CUDA_CHECK( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nvertices * 6 * sizeof(float), cudaMemcpyHostToDevice) ); delete[] xyzuvw_host; @@ -315,7 +309,7 @@ namespace CudaRBC maxCells = 0; CUDA_CHECK( cudaMalloc(&host_av, 1 * 2 * sizeof(float)) ); - unitsSetup(1.194170681, 0.003092250212, 20.49568481, 39.2254922344138, 13223.5137655706, 7710.76185113627, 18.14524310, 135, 94, 1e-6, 2.4295e-6, 4, report); + unitsSetup(1.233, 0.00198, 30, 13*8.8, 5000, 10000, 30, 135, 95, 1.0e-6, 8.7e-3, 4, report); CUDA_CHECK( cudaFuncSetCacheConfig(fall_kernel<498>, cudaFuncCachePreferL1) ); } @@ -323,14 +317,15 @@ namespace CudaRBC void unitsSetup(float lmax, float p, float cq, float kb, float ka, float kv, float gammaC, float totArea0, float totVolume0, float lunit, float tunit, int ndens, bool prn) { - const float lrbc = 1.000000e-06; - const float trbc = 3.009441e-03; + const float lrbc = 1.0e-6; + const float trbc = 8.7e-3;//3.009441e-03; //const float mrbc = 3.811958e-13; float ll = lunit / lrbc; float tt = tunit / trbc; + float EE = pow(ll, -2.0) * pow(tt, 2.0); - params.kbT = 580 * 250 * pow(ll, -2.0) * pow(tt, 2.0); + params.kbT = 0.05 * EE; params.p = p / ll; params.lmax = lmax / ll; params.q = 1; @@ -338,12 +333,12 @@ namespace CudaRBC params.totArea0 = totArea0 * pow(ll, -2.0); params.totVolume0 = totVolume0 * pow(ll, -3.0); params.l0 = sqrt(params.totArea0 / (2.0*params.nvertices - 4.) * 4.0/sqrt(3.0)); - params.ka = ka * params.kbT / (params.totArea0 * params.l0 * params.l0); - params.kv = kv * params.kbT / (6 * params.totVolume0 * powf(params.l0, 3)); - params.gammaC = gammaC * 580 * pow(tt, 1.0); + params.ka = ka * EE / (params.totArea0 * params.l0 * params.l0); + params.kv = kv * EE / (6 * params.totVolume0 * powf(params.l0, 3)); + params.gammaC = gammaC * pow(tt, 1.0); params.gammaT = 3.0 * params.gammaC; - float phi = 6.97 / 180.0*M_PI; + float phi = 0*6.97 / 180.0*M_PI; params.sinTheta0 = sin(phi); params.cosTheta0 = cos(phi); params.kb = kb * params.kbT; diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index fe3102bac..02cabaa9b 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -16,7 +16,8 @@ OBJS = dpd.o wall.o fsi.o contact.o \ solvent-exchange.o solute-exchange.o \ common.o containers.o io.o \ scan.o minmax.o redistancing.o \ - simulation.o main.o dumper.o + simulation.o main.o dumper.o \ + velcontroller.o velsampler.o LIBS = -lcuda-dpd -lcuda-rbc -lcuda-ctc -lcudart -ldl -lz diff --git a/mpi-dpd/dumper.cu b/mpi-dpd/dumper.cu index 86b1b9e54..fcacf6f3d 100644 --- a/mpi-dpd/dumper.cu +++ b/mpi-dpd/dumper.cu @@ -31,6 +31,8 @@ Dumper::Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm) : iocom } MPI_CHECK(MPI_Comm_rank(iocomm, &rank)); + + avgVels.resize(XSIZE_SUBDOMAIN*YSIZE_SUBDOMAIN*ZSIZE_SUBDOMAIN); } void Dumper::qoi(Particle* rbcs, Particle * ctcs, int nrbcparts, int nctcparts, const float tm) @@ -182,6 +184,7 @@ void Dumper::do_dump() Acceleration* a = &accelerations[0]; MPI_CHECK( MPI_Recv(p, n, Particle::datatype(), rank, 0, intercomm, &status) ); MPI_CHECK( MPI_Recv(a, n, Acceleration::datatype(), rank, 0, intercomm, &status) ); + MPI_CHECK( MPI_Recv(&avgVels[0], avgVels.size()*3, MPI_FLOAT, rank, 0, intercomm, &status) ); double t0 = MPI_Wtime(); @@ -243,6 +246,23 @@ void Dumper::do_dump() if (hdf5field_dumps) { dump_field.dump(iocomm, p, nparticles, iddatadump * steps_per_dump); + + // Dump avg vels as well + char filepath[512]; + sprintf(filepath, "h5/avgvels-%04d.h5", iddatadump); + + vector vx(avgVels.size()), vy(avgVels.size()), vz(avgVels.size()); + for (int i=0; i particles; vector accelerations; + vector avgVels; int nrbcverts, nctcverts, qoiid, rank; diff --git a/mpi-dpd/io.h b/mpi-dpd/io.h index fc77ae4e2..bc3d58ddb 100644 --- a/mpi-dpd/io.h +++ b/mpi-dpd/io.h @@ -56,10 +56,6 @@ class H5FieldDump int last_idtimestep, globalsize[3]; MPI_Comm cartcomm; - - void _write_fields(const char * const path2h5, - const float * const channeldata[], const char * const * const channelnames, const int nchannels, - MPI_Comm comm, const float time); void _xdmf_header(FILE * xmf); void _xdmf_grid(FILE * xmf, float time, const char * const h5path, const char * const * channelnames, int nchannels); @@ -67,6 +63,10 @@ class H5FieldDump public: + void _write_fields(const char * const path2h5, + const float * const channeldata[], const char * const * const channelnames, const int nchannels, + MPI_Comm comm, const float time); + H5FieldDump(MPI_Comm cartcomm); void dump(MPI_Comm comm, const Particle * const p, const int n, int idtimestep); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index d4605e44a..a88e703b1 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -25,25 +25,25 @@ __global__ void make_texture( float4 * __restrict xyzouvwo, ushort4 * __restrict const float2 * base = ( float2* )( xyzuvw + i * 6 ); #pragma unroll 3 for( uint j = lane; j < 96; j += 32 ) { - float2 u = base[j]; - // NVCC bug: no operator = between volatile float2 and float2 - asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); + float2 u = base[j]; + // NVCC bug: no operator = between volatile float2 and float2 + asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); } // SMEM: XYZUVW XYZUVW ... uint pid = lane / 2; const uint x_or_v = ( lane % 2 ) * 3; xyzouvwo[ i * 2 + lane ] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); pid += 16; xyzouvwo[ i * 2 + lane + 32] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); xyzo_half[i + lane] = make_ushort4( __float2half_rn( smem[ warpid * 192 + lane * 6 + 0 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); -// } + __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), + __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); + // } } void Simulation::_update_helper_arrays() @@ -56,7 +56,7 @@ void Simulation::_update_helper_arrays() xyzo_half.resize(np); if (np) - make_texture <<< (np + 1023) / 1024, 1024, 1024 * 6 * sizeof( float )>>>(xyzouvwo.data, xyzo_half.data, (float *)particles->xyzuvw.data, np ); + make_texture <<< (np + 1023) / 1024, 1024, 1024 * 6 * sizeof( float )>>>(xyzouvwo.data, xyzo_half.data, (float *)particles->xyzuvw.data, np ); CUDA_CHECK(cudaPeekAtLastError()); } @@ -70,19 +70,19 @@ std::vector Simulation::_ic() const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int iz = 0; iz < L[2]; iz++) - for(int iy = 0; iy < L[1]; iy++) - for(int ix = 0; ix < L[0]; ix++) - for(int l = 0; l < numberdensity; ++l) - { - const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); - - ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); - ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); - ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); - ic[p].u[0] = 0; - ic[p].u[1] = 0; - ic[p].u[2] = 0; - } + for(int iy = 0; iy < L[1]; iy++) + for(int ix = 0; ix < L[0]; ix++) + for(int l = 0; l < numberdensity; ++l) + { + const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); + + ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); + ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); + ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); + ic[p].u[0] = 0; + ic[p].u[1] = 0; + ic[p].u[2] = 0; + } /* use this to check robustness for(int i = 0; i < ic.size(); ++i) @@ -91,7 +91,7 @@ std::vector Simulation::_ic() ic[i].x[c] = -L[c] * 0.5 + drand48() * L[c]; ic[i].u[c] = 0; } - */ + */ return ic; } @@ -105,18 +105,18 @@ void Simulation::_redistribute() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.send(); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.bulk(particles->size, cells.start, cells.count, mainstream); @@ -126,17 +126,17 @@ void Simulation::_redistribute() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -149,10 +149,10 @@ void Simulation::_redistribute() swap(particles, newparticles); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -166,61 +166,61 @@ void Simulation::_report(const bool verbose, const int idtimestep) report_host_memory_usage(activecomm, stdout); { - static double t0 = MPI_Wtime(), t1; + static double t0 = MPI_Wtime(), t1; - t1 = MPI_Wtime(); + t1 = MPI_Wtime(); - float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; + float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; - host_busy_time *= 1e3 / steps_per_report; + host_busy_time *= 1e3 / steps_per_report; - float sumval, maxval, minval; - MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); + float sumval, maxval, minval; + MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); - int commsize; - MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); + int commsize; + MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); - const double imbalance = 100 * (maxval / sumval * commsize - 1); + const double imbalance = 100 * (maxval / sumval * commsize - 1); - if (verbose && imbalance >= 0) - printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", - imbalance , minval, sumval / commsize, maxval); + if (verbose && imbalance >= 0) + printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", + imbalance , minval, sumval / commsize, maxval); - localcomm.print_particles(particles->size); + localcomm.print_particles(particles->size); - host_idle_time = 0; - t0 = t1; + host_idle_time = 0; + t0 = t1; } { - static double t0 = MPI_Wtime(), t1; - - t1 = MPI_Wtime(); - - if (verbose) - { - printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); - printf("in more details, per time step:\n"); - double tt = 0; - for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) - { - printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); - tt += it->second; - it->second = 0; - } - printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); - } - - t0 = t1; + static double t0 = MPI_Wtime(), t1; + + t1 = MPI_Wtime(); + + if (verbose) + { + printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); + printf("in more details, per time step:\n"); + double tt = 0; + for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) + { + printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); + tt += it->second; + it->second = 0; + } + printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); + } + + t0 = t1; } } void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) { if (!coll || !coll->count()) - return; + return; SimpleDeviceBuffer marks(coll->pcount()); @@ -235,13 +235,13 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) std::vector tokill; for(int i = 0; i < nbodies; ++i) { - bool valid = true; + bool valid = true; - for(int j = 0; j < nvertices && valid; ++j) - valid &= 0 == tmp[j + nvertices * i]; + for(int j = 0; j < nvertices && valid; ++j) + valid &= 0 == tmp[j + nvertices * i]; - if (!valid) - tokill.push_back(i); + if (!valid) + tokill.push_back(i); } coll->remove(&tokill.front(), tokill.size()); @@ -253,7 +253,7 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) void Simulation::_create_walls(const bool verbose, bool & termination_request) { if (verbose) - printf("creation of the walls...\n"); + printf("creation of the walls...\n"); int nsurvived = 0; ExpectedMessageSizes new_sizes; @@ -261,30 +261,30 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) //adjust the message sizes if we're pushing the flow in x { - const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; - const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; - const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; - - for(int code = 0; code < 27; ++code) - { - const int d[3] = { - (code % 3) - 1, - ((code / 3) % 3) - 1, - ((code / 9) % 3) - 1 - }; - - const double IudotnI = - fabs(d[0] * xvelavg) + - fabs(d[1] * yvelavg) + - fabs(d[2] * zvelavg) ; - - const float factor = 1 + IudotnI * dt * 10 * numberdensity; - - //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", - //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); - - new_sizes.msgsizes[code] *= factor; - } + const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; + const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; + const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; + + for(int code = 0; code < 27; ++code) + { + const int d[3] = { + (code % 3) - 1, + ((code / 3) % 3) - 1, + ((code / 9) % 3) - 1 + }; + + const double IudotnI = + fabs(d[0] * xvelavg) + + fabs(d[1] * yvelavg) + + fabs(d[2] * zvelavg) ; + + const float factor = 1 + IudotnI * dt * 10 * numberdensity; + + //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", + //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); + + new_sizes.msgsizes[code] *= factor; + } } //MPI_CHECK(MPI_Barrier(activecomm)); @@ -316,7 +316,7 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) return; } } - */ + */ particles->resize(nsurvived); particles->clear_velocity(); @@ -331,14 +331,14 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) _remove_bodies_from_wall(ctcscoll); { - H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); - Particle * p = new Particle[particles->size]; + H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); + Particle * p = new Particle[particles->size]; - CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); + CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); - sd.dump(p, particles->size); + sd.dump(p, particles->size); - delete [] p; + delete [] p; } } @@ -351,10 +351,10 @@ void Simulation::_forces(bool firsttime) std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -363,10 +363,10 @@ void Simulation::_forces(bool firsttime) particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); @@ -375,7 +375,7 @@ void Simulation::_forces(bool firsttime) CUDA_CHECK(cudaPeekAtLastError()); dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + cells.start, cells.count, mainstream); dpd.post(particles->xyzuvw.data, particles->size, mainstream, downloadstream); @@ -384,14 +384,14 @@ void Simulation::_forces(bool firsttime) CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -400,7 +400,7 @@ void Simulation::_forces(bool firsttime) solutex.recv_p(uploadstream, mainstream); if (contactforces) - contact.attach_bulk(wsolutes); + contact.attach_bulk(wsolutes); solutex.halo(uploadstream, mainstream, downloadstream); @@ -412,11 +412,11 @@ void Simulation::_forces(bool firsttime) if (nsubsteps == 0) { - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); } CUDA_CHECK(cudaPeekAtLastError()); @@ -481,24 +481,26 @@ void Simulation::_datadump(const int idtimestep) if (stress) { - for(int c = 0; c < 6; ++c) - stresses_datadump[c].resize(n); + for(int c = 0; c < 6; ++c) + stresses_datadump[c].resize(n); - for(int c = 0; c < 6; ++c) - CUDA_CHECK(cudaMemcpyAsync(stresses_datadump[c].data, stresses[c].data, sizeof(float) * n, cudaMemcpyDeviceToHost,0)); + for(int c = 0; c < 6; ++c) + CUDA_CHECK(cudaMemcpyAsync(stresses_datadump[c].data, stresses[c].data, sizeof(float) * n, cudaMemcpyDeviceToHost,0)); } if (rbcscoll) - n += rbcscoll->pcount(); + n += rbcscoll->pcount(); if (ctcscoll) - n += ctcscoll->pcount(); + n += ctcscoll->pcount(); particles_datadump.resize(n); accelerations_datadump.resize(n); CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost,0)); CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data, particles->axayaz.data, sizeof(Acceleration) * particles->size, cudaMemcpyDeviceToHost,0)); + vector& avgVels = velsampler->getAvgVel(0); + if (nsubsteps > 0) { CUDA_CHECK( cudaStreamSynchronize(0) ); @@ -511,18 +513,18 @@ void Simulation::_datadump(const int idtimestep) if (rbcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += rbcscoll->pcount(); + start += rbcscoll->pcount(); } if (ctcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += ctcscoll->pcount(); + start += ctcscoll->pcount(); } assert(start == n); @@ -536,10 +538,12 @@ void Simulation::_datadump(const int idtimestep) MPI_CHECK( MPI_Send(&datadump_nrbcs, 1, MPI_INT, rank, 0, intercomm) ); MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); - CUDA_CHECK(cudaEventSynchronize(evdownloaded)); + CUDA_CHECK(cudaEventSynchronize(evdownloaded)); + CUDA_CHECK( cudaStreamSynchronize(0) ); MPI_CHECK( MPI_Send(particles_datadump.data, n, Particle::datatype(), rank, 0, intercomm) ); MPI_CHECK( MPI_Send(accelerations_datadump.data, n, Acceleration::datatype(), rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&avgVels[0], avgVels.size()*3, MPI_FLOAT, rank, 0, intercomm) ); timings["data-dump"] += MPI_Wtime() - tstart; } @@ -547,18 +551,20 @@ void Simulation::_datadump(const int idtimestep) void Simulation::_update_and_bounce() { double tstart = MPI_Wtime(); + + velcontrol->push(cells.start, particles->xyzuvw.data, particles->axayaz.data, mainstream); particles->update_stage2_and_1(driving_acceleration, mainstream); CUDA_CHECK(cudaPeekAtLastError()); if (nsubsteps == 0) { - if (rbcscoll) + if (rbcscoll) rbcscoll->update_stage2_and_1(0.0f, mainstream); - CUDA_CHECK(cudaPeekAtLastError()); + CUDA_CHECK(cudaPeekAtLastError()); - if (ctcscoll) + if (ctcscoll) ctcscoll->update_stage2_and_1(0.0f, mainstream); } @@ -566,33 +572,33 @@ void Simulation::_update_and_bounce() if (wall) { - tstart = MPI_Wtime(); - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + tstart = MPI_Wtime(); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); if (nsubsteps == 0) { - if (rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); } - timings["bounce-walls"] += MPI_Wtime() - tstart; + timings["bounce-walls"] += MPI_Wtime() - tstart; } CUDA_CHECK(cudaPeekAtLastError()); } Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) : - cartcomm(cartcomm), activecomm(activecomm), intercomm(intercomm), - /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), - rbcscoll(NULL), ctcscoll(NULL), wall(NULL), - redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), - dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), - check_termination(check_termination), - driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), - datadump_pending(false), simulation_is_done(false) + cartcomm(cartcomm), activecomm(activecomm), intercomm(intercomm), + /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), + rbcscoll(NULL), ctcscoll(NULL), wall(NULL), + redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), + dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), + check_termination(check_termination), + driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), + datadump_pending(false), simulation_is_done(false) { MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); @@ -600,35 +606,35 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercom solutex.attach_halocomputation(fsi); if (contactforces) - solutex.attach_halocomputation(contact); + solutex.attach_halocomputation(contact); int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); { - particles = &particles_pingpong[0]; - newparticles = &particles_pingpong[1]; + particles = &particles_pingpong[0]; + newparticles = &particles_pingpong[1]; - vector ic = _ic(); + vector ic = _ic(); - for(int c = 0; c < 2; ++c) - { - particles_pingpong[c].resize(ic.size()); + for(int c = 0; c < 2; ++c) + { + particles_pingpong[c].resize(ic.size()); - particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, - (0.5 + coords[1]) * YSIZE_SUBDOMAIN, - (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); + particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, + (0.5 + coords[1]) * YSIZE_SUBDOMAIN, + (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); - particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, - dims[1] * YSIZE_SUBDOMAIN, - dims[2] * ZSIZE_SUBDOMAIN); - } + particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, + dims[1] * YSIZE_SUBDOMAIN, + dims[2] * ZSIZE_SUBDOMAIN); + } - CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); + CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); - cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); + cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); - _update_helper_arrays(); + _update_helper_arrays(); } CUDA_CHECK(cudaStreamCreate(&mainstream)); @@ -637,20 +643,25 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercom if (rbcs) { - rbcscoll = new CollectionRBC(cartcomm); - rbcscoll->setup("rbcs-ic.txt"); + rbcscoll = new CollectionRBC(cartcomm); + rbcscoll->setup("rbcs-ic.txt"); } if (ctcs) { - ctcscoll = new CollectionCTC(cartcomm); - ctcscoll->setup("ctcs-ic.txt"); + ctcscoll = new CollectionCTC(cartcomm); + ctcscoll->setup("ctcs-ic.txt"); } - CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); - particles_datadump.resize(particles->size * 1.5); - accelerations_datadump.resize(particles->size * 1.5); - } + CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); + particles_datadump.resize(particles->size * 1.5); + accelerations_datadump.resize(particles->size * 1.5); + + int xl[3] = {0, 0, 0}; + int xh[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + velcontrol = new VelController(xl, xh, coords, make_float3(2, 0, 0), activecomm); + velsampler = new VelSampler(); +} void Simulation::_lockstep() { @@ -661,10 +672,10 @@ void Simulation::_lockstep() std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -673,17 +684,17 @@ void Simulation::_lockstep() particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); solutex.pack_p(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + cells.start, cells.count, mainstream); solutex.post_p(mainstream, downloadstream); @@ -692,8 +703,8 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -702,7 +713,7 @@ void Simulation::_lockstep() solutex.recv_p(uploadstream, mainstream); if (contactforces) - contact.attach_bulk(wsolutes); + contact.attach_bulk(wsolutes); solutex.halo(uploadstream, mainstream, downloadstream); @@ -714,20 +725,21 @@ void Simulation::_lockstep() if (nsubsteps == 0) { - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); } CUDA_CHECK(cudaPeekAtLastError()); solutex.post_a(); + velcontrol->push(cells.start, particles->xyzuvw.data, particles->axayaz.data, mainstream); particles->update_stage2_and_1(driving_acceleration, mainstream); if (wall) - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -740,10 +752,10 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -751,17 +763,17 @@ void Simulation::_lockstep() if (nsubsteps == 0) { - if (rbcscoll) + if (rbcscoll) rbcscoll->update_stage2_and_1(0.0f, mainstream); - if (ctcscoll) + if (ctcscoll) ctcscoll->update_stage2_and_1(0.0f, mainstream); - if (wall && rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (wall && rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (wall && ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (wall && ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); } else { // TSS @@ -805,16 +817,16 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -828,25 +840,25 @@ void Simulation::_lockstep() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -857,7 +869,7 @@ void Simulation::_lockstep() void Simulation::run() { if (rank == 0 && !walls) - printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); + printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); double time_simulation_start = MPI_Wtime(); @@ -865,124 +877,139 @@ void Simulation::run() _forces(nsubsteps > 0); if (!walls && pushtheflow) - driving_acceleration = hydrostatic_a; + driving_acceleration = hydrostatic_a; particles->update_stage1(driving_acceleration, mainstream); if (nsubsteps == 0) { - if (rbcscoll) + if (rbcscoll) rbcscoll->update_stage1(0.0f, mainstream); - if (ctcscoll) + if (ctcscoll) ctcscoll->update_stage1(0.0f, mainstream); } int it; + FILE* fforce = fopen("frcdump.txt", "w"); + for(it = 0; it < nsteps; ++it) { - const bool verbose = it > 0 && rank == 0; + const bool verbose = it > 0 && rank == 0; #ifdef _USE_NVTX_ - if (it == nvtxstart) - { - NvtxTracer::currently_profiling = true; - CUDA_CHECK(cudaProfilerStart()); - } - else if (it == nvtxstop) - { - CUDA_CHECK(cudaProfilerStop()); - NvtxTracer::currently_profiling = false; - CUDA_CHECK(cudaDeviceSynchronize()); - - if (rank == 0) - printf("profiling session ended. terminating the simulation now...\n"); - - break; - } + if (it == nvtxstart) + { + NvtxTracer::currently_profiling = true; + CUDA_CHECK(cudaProfilerStart()); + } + else if (it == nvtxstop) + { + CUDA_CHECK(cudaProfilerStop()); + NvtxTracer::currently_profiling = false; + CUDA_CHECK(cudaDeviceSynchronize()); + + if (rank == 0) + printf("profiling session ended. terminating the simulation now...\n"); + + break; + } #endif - if (it % steps_per_report == 0) - { - CUDA_CHECK(cudaStreamSynchronize(mainstream)); + if (it % steps_per_report == 0) + { + CUDA_CHECK(cudaStreamSynchronize(mainstream)); - if (simulation_is_done = check_termination()) - break; + if (simulation_is_done = check_termination()) + break; - _report(verbose, it); - } + _report(verbose, it); + } - _redistribute(); + _redistribute(); #if 1 - lockstep_check: + lockstep_check: + + const bool lockstep_OK = + !(walls && it >= wall_creation_stepid && wall == NULL) && + !(it % steps_per_dump == 0) && + !(it + 1 == nvtxstart) && + !(it + 1 == nvtxstop) && + !((it + 1) % steps_per_report == 0) && + !(it + 1 == nsteps); + + if (lockstep_OK) + { + _lockstep(); + + if (it % 10 == 0) + velcontrol->sample(cells.start, particles->xyzuvw.data, mainstream); - const bool lockstep_OK = - !(walls && it >= wall_creation_stepid && wall == NULL) && - !(it % steps_per_dump == 0) && - !(it + 1 == nvtxstart) && - !(it + 1 == nvtxstop) && - !((it + 1) % steps_per_report == 0) && - !(it + 1 == nsteps); + if (it % 5 == 0) + velsampler->sample(cells.start, particles->xyzuvw.data, mainstream); - if (lockstep_OK) - { - _lockstep(); + if (wall && it % 1000 == 0) + { + float3 f = velcontrol->adjustF(mainstream); + if (rank == 0) fprintf(fforce, "Force applied: %f\n", f.x); + fflush(fforce); + } - ++it; + ++it; - goto lockstep_check; - } + goto lockstep_check; + } #endif - if (walls && it >= wall_creation_stepid && wall == NULL) - { - CUDA_CHECK(cudaDeviceSynchronize()); + if (walls && it >= wall_creation_stepid && wall == NULL) + { + CUDA_CHECK(cudaDeviceSynchronize()); + + bool termination_request = false; - bool termination_request = false; + _create_walls(verbose, termination_request); - _create_walls(verbose, termination_request); + _redistribute(); - _redistribute(); + if (termination_request) + break; - if (termination_request) - break; + time_simulation_start = MPI_Wtime(); - time_simulation_start = MPI_Wtime(); + if (pushtheflow) + driving_acceleration = 0; - if (pushtheflow) - driving_acceleration = hydrostatic_a; + if (rank == 0) + printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); + } - if (rank == 0) - printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); - } + if(stress && it % steps_per_dump == 0) + { + for(int c = 0; c < 6; ++c) + stresses[c].resize(particles->size); - if(stress && it % steps_per_dump == 0) - { - for(int c = 0; c < 6; ++c) - stresses[c].resize(particles->size); + dpd.set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); - dpd.set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + if (wall) + wall->set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + } - if (wall) - wall->set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); - } + _forces(); - _forces(); + if (stress && it % steps_per_dump == 0 ) + { + dpd.clr_stress_buffers(); - if (stress && it % steps_per_dump == 0 ) - { - dpd.clr_stress_buffers(); + if (wall) + wall->clr_stress_buffers(); + } - if (wall) - wall->clr_stress_buffers(); - } + if (it % steps_per_dump == 0) + _datadump(it); - if (it % steps_per_dump == 0) - _datadump(it); - - _update_and_bounce(); + _update_and_bounce(); } const double time_simulation_stop = MPI_Wtime(); @@ -996,12 +1023,12 @@ void Simulation::run() MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); if (rank == 0) - if (it == nsteps) - printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", - telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); - else - if (it != wall_creation_stepid) - printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); + if (it == nsteps) + printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", + telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); + else + if (it != wall_creation_stepid) + printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); fflush(stdout); } @@ -1013,11 +1040,11 @@ Simulation::~Simulation() CUDA_CHECK(cudaStreamDestroy(downloadstream)); if (wall) - delete wall; + delete wall; if (rbcscoll) - delete rbcscoll; + delete rbcscoll; if (ctcscoll) - delete ctcscoll; + delete ctcscoll; } diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 8161f442a..49925fac2 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -34,6 +34,8 @@ #include "redistribute-rbcs.h" #include "ctc.h" #include "io.h" +#include "velcontroller.h" +#include "velsampler.h" class Simulation { @@ -100,6 +102,9 @@ class Simulation void _datadump_async(); + VelController* velcontrol; + VelSampler* velsampler; + public: Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) ; diff --git a/mpi-dpd/velcontroller.cu b/mpi-dpd/velcontroller.cu new file mode 100644 index 000000000..a08cf6fe8 --- /dev/null +++ b/mpi-dpd/velcontroller.cu @@ -0,0 +1,185 @@ +/* + * velcontroller.cu + * ctc falcon + * + * Created by Dmitry Alexeev on Sep 24, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + + +#include "velcontroller.h" +#include "helper_math.h" + +//==================================================================================== +// Kernels +//==================================================================================== + +namespace VelContKernels +{ + __global__ void sample(const int * const __restrict__ cellsstart, const float2* const __restrict__ p, float3* res, VelController::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.n[0] && ccoos.y < info.n[1] && ccoos.z < info.n[2]) + { + const uint cid = (ccoos.x + info.xl[0]) + (ccoos.y + info.xl[1]) * info.cellsx + (ccoos.z + info.xl[2]) * info.cellsx * info.cellsy; + const uint resid = ccoos.x + ccoos.y * info.n[0] + ccoos.z * info.n[0] * info.n[1]; + + float3 myres = make_float3(0.0f, 0.0f, 0.0f); + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + float num = cellsstart[cid+1] - cellsstart[cid]; + float2 tmp1 = p[3*pid + 1]; + float2 tmp2 = p[3*pid + 2]; + myres.x += tmp1.y / num; + myres.y += tmp2.x / num; + myres.z += tmp2.y / num; + } + res[resid] += make_float3(myres.x, myres.y, myres.z); + } + } + + __global__ void push(const int * const __restrict__ cellsstart, Acceleration* acc, float3 f, VelController::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.n[0] && ccoos.y < info.n[1] && ccoos.z < info.n[2]) + { + const uint cid = (ccoos.x + info.xl[0]) + (ccoos.y + info.xl[1]) * info.cellsx + (ccoos.z + info.xl[2]) * info.cellsx * info.cellsy; + + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + acc[pid].a[0] += f.x; + //acc[pid].a[1] += f.y; + //acc[pid].a[2] += f.z; + } + } + } + + __inline__ __device__ float3 warpReduceSum(float3 val) + { + for (int offset = warpSize/2; offset > 0; offset /= 2) + { + val.x += __shfl_down(val.x, offset); + val.y += __shfl_down(val.y, offset); + val.z += __shfl_down(val.z, offset); + } + return val; + } + + __global__ void reduceByWarp(float3 *res, const float3 * const __restrict__ vel, const uint total) + { + assert(blockDim.x == 32); + const uint id = threadIdx.x + blockIdx.x*blockDim.x; + const uint ch = blockIdx.x; + if (id >= total) return; + + const float3 val = vel[id]; + const float3 rval = warpReduceSum(val); + + if ((threadIdx.x % warpSize) == 0) + res[ch]=rval; + } +} + +//==================================================================================== +// Methods +//==================================================================================== + +VelController::VelController(int xl[3], int xh[3], int mpicoos[3], float3 desired, MPI_Comm comm) : + desired(desired), Kp(2), Ki(1), Kd(8), factor(0.01), sampleid(0) +{ + MPI_CHECK( MPI_Comm_dup(comm, &this->comm) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); + MPI_CHECK( MPI_Comm_rank(comm, &rank) ); + const int L[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + + int myxl[3], myxh[3], n[3]; + for (int d=0; d<3; d++) + { + myxl[d] = max(xl[d], L[d]*mpicoos[d] ); + myxh[d] = min(xh[d], L[d]*(mpicoos[d]+1)); + + info.n[d] = n[d] = myxh[d] - myxl[d]; + info.xl[d] = myxl[d] % (L[d]+1); + } + + if (n[0] > 0 && n[1] > 0 && n[2] > 0) + total = n[0] * n[1] * n[2]; + else + total = 0; + + MPI_CHECK( MPI_Allreduce(&total, &globtot, 1, MPI_INT, MPI_SUM, comm) ); + + vel.resize(total); + if (total) + CUDA_CHECK( cudaMemset(vel.data, 0, n[0] * n[1] * n[2] * sizeof(float3)) ); + + info.cellsx = L[0]; + info.cellsy = L[1]; + info.cellsz = L[2]; + + s = f = make_float3(0, 0, 0); + old = desired; +} + +void VelController::sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream) +{ + dim3 block(8, 8, 1); + dim3 grid( (info.n[0] + block.x - 1) / block.x, + (info.n[1] + block.y - 1) / block.y, + (info.n[2] + block.z - 1) / block.z ); + + sampleid++; + if (total) + VelContKernels::sample <<>> (cellsstart, (float2*)p, vel.data, info); +} + +void VelController::push(const int * const cellsstart, const Particle* const p, Acceleration* acc, cudaStream_t stream) +{ + dim3 block(8, 8, 1); + dim3 grid( (info.n[0] + block.x - 1) / block.x, + (info.n[1] + block.y - 1) / block.y, + (info.n[2] + block.z - 1) / block.z ); + + if (total) + VelContKernels::push <<>> (cellsstart, acc, f, info); +} + +float3 VelController::adjustF(cudaStream_t stream) +{ + const int chunks = (total+31) / 32; + if (avgvel.size < chunks) avgvel.resize(chunks); + + if (total) + { + VelContKernels::reduceByWarp <<< (total + 31) / 32, 32, 0, stream >>> (avgvel.devptr, vel.data, total); + CUDA_CHECK( cudaStreamSynchronize(stream) ); + } + + float3 cur = make_float3(0, 0, 0); + for (int i=0; i vel; + PinnedHostBuffer avgvel; + int total, globtot; + + float3 desired; + float Kp, Ki, Kd, factor; + float3 s, old; + float3 f; + + int sampleid; + +public: + VelController(int xl[3], int xh[3], int mpicoos[3], float3 desired, MPI_Comm comm); + + void sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream); + void push (const int * const cellsstart, const Particle* const p, Acceleration* acc, cudaStream_t stream); + float3 adjustF(cudaStream_t stream); +}; + + + + diff --git a/mpi-dpd/velsampler.cu b/mpi-dpd/velsampler.cu new file mode 100644 index 000000000..c7d34d95a --- /dev/null +++ b/mpi-dpd/velsampler.cu @@ -0,0 +1,97 @@ +/* + * velsampler.cu + * ctc PANDA + * + * Created by Dmitry Alexeev on Nov 27, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +#include "velsampler.h" +#include "helper_math.h" + + +//==================================================================================== +// Kernels +//==================================================================================== + +namespace VelSmpKernels +{ + __global__ void sample(const int * const __restrict__ cellsstart, const float2* const __restrict__ p, float3* res, VelSampler::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.cellsx && ccoos.y < info.cellsy && ccoos.z < info.cellsz) + { + const uint cid = ccoos.x + (ccoos.y + ccoos.z * info.cellsy) * info.cellsx; + + float3 myres = make_float3(0.0f, 0.0f, 0.0f); + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + float num = cellsstart[cid+1] - cellsstart[cid]; + float2 tmp1 = p[3*pid + 1]; + float2 tmp2 = p[3*pid + 2]; + myres.x += tmp1.y / num; + myres.y += tmp2.x / num; + myres.z += tmp2.y / num; + } + res[cid] += make_float3(myres.x, myres.y, myres.z); + } + } + + __global__ void scale(int n, float a, float* res) + { + const uint id = threadIdx.x + blockIdx.x*blockDim.x; + + if (id < n) + { + res[id] *= a; + } + } +} + +//==================================================================================== +// Methods +//==================================================================================== + + +VelSampler::VelSampler() +{ + const int L[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + + vels.resize(L[0]*L[1]*L[2]); + hostVels.resize(vels.size); + CUDA_CHECK( cudaMemset(vels.data, 0, vels.size * sizeof(float3)) ); + + info.cellsx = L[0]; + info.cellsy = L[1]; + info.cellsz = L[2]; +} + +void VelSampler::sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream) +{ + dim3 block(4, 4, 4); + dim3 grid( (info.cellsx + block.x - 1) / block.x, + (info.cellsy + block.y - 1) / block.y, + (info.cellsz + block.z - 1) / block.z ); + + sampleid++; + VelSmpKernels::sample <<>> (cellsstart, (float2*)p, vels.data, info); + CUDA_CHECK(cudaPeekAtLastError()); +} + +vector& VelSampler::getAvgVel(cudaStream_t stream) +{ + if (sampleid > 0) + VelSmpKernels::scale <<<(3*vels.size + 127) / 128, 128, 0, stream>>> (3*vels.size, 1.0f / sampleid, (float*)vels.data); + CUDA_CHECK(cudaPeekAtLastError()); + CUDA_CHECK(cudaMemcpyAsync(&hostVels[0], vels.data, sizeof(float3) * vels.size, cudaMemcpyDeviceToHost, stream)); + CUDA_CHECK(cudaMemsetAsync(vels.data, 0, sizeof(float3) * vels.size, stream)); + + sampleid = 0; + return hostVels; +} + + diff --git a/mpi-dpd/velsampler.h b/mpi-dpd/velsampler.h new file mode 100644 index 000000000..febe13999 --- /dev/null +++ b/mpi-dpd/velsampler.h @@ -0,0 +1,49 @@ +/* + * velsampler.h + * ctc PANDA + * + * Created by Dmitry Alexeev on Nov 27, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + + +#pragma once + +#include +#include "common.h" + +using namespace std; + +class VelSampler +{ +public: + struct CellInfo + { + uint cellsx, cellsy, cellsz; + }; + +private: + CellInfo info; + int size, rank; + + SimpleDeviceBuffer vels; + vector hostVels; + int total, globtot; + + float3 desired; + float Kp, Ki, Kd, factor; + float3 s, old; + float3 f; + + int sampleid; + +public: + VelSampler(); + + void sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream); + vector& getAvgVel(cudaStream_t stream); +}; + + +