diff --git a/README.md b/README.md index 09588d229..54d26c1fc 100644 --- a/README.md +++ b/README.md @@ -1,13 +1,14 @@ uDeviceX comes with the GPLv2 LICENSE. === The uDeviceX folders is organized into: -balaprep: preprocessing tool for assigning ranks to compute nodes. -cell-placement: preprocessing tool for generating the initial RBC/CTC displacement -cuda-ctc: code for the CTC model -cuda-dpd: code for the DPD interactions -cuda-rbc: code for the RBC model -device-gen: preprocessing tool to generate new device geometries -halo-bench: OSU-like benchmark to measure latency and bandwidth across the MPI ranks. -mpi-dpd: the simulation code -proof-of-concept: tests and hacks. +* balaprep: preprocessing tool for assigning ranks to compute nodes. +* cell-placement: preprocessing tool for generating the initial RBC/CTC displacement +* cuda-ctc: code for the CTC model +* cuda-dpd: code for the DPD interactions +* cuda-rbc: code for the RBC model +* device-gen: preprocessing tool to generate new device geometries +* mpi-dpd: the simulation code +* postprocessing: auxiliary tools to extract quantities from simulations the output +* proof-of-concept: tests and hacks +* tests: accuracy and performance tests diff --git a/cuda-ctc/ctc-cuda.cu b/cuda-ctc/ctc-cuda.cu index a89a7f66e..284f2f321 100644 --- a/cuda-ctc/ctc-cuda.cu +++ b/cuda-ctc/ctc-cuda.cu @@ -27,27 +27,26 @@ using namespace std; namespace CudaCTC { + int nparticles; + int ntriang; + int nbonds; + int ndihedrals; -int nparticles; -int ntriang; -int nbonds; -int ndihedrals; - -int *triangles; -int *dihedrals; + int *triangles; + int *dihedrals; int *triangles_host; int *triplets; -// Helper pointers + // Helper pointers int maxCells; __constant__ real *totA_V; real *host_av; -// Original configuration -real* orig_xyzuvw; + // Original configuration + real* orig_xyzuvw; -map bufmap; -__constant__ float A[4][4]; + map bufmap; + __constant__ float A[4][4]; Extent* dummy; @@ -59,143 +58,152 @@ __constant__ float A[4][4]; float totArea0, float totVolume0, float lunit, float tunit, int ndens, bool prn); void setup(int& nvertices, Extent& host_extent) -{ + { const float scale=1; - const bool report = false; + const bool report = false; // 0.0945, 0.00141, 1.642599, // 1, 1.8, a, v, a/m.ntriang, 945, 0, 472.5, // 90, 30, sin(phi), cos(phi), 6.048 const char* fname = "../cuda-ctc/sphere20.dat"; - ifstream in(fname); - string line; - - if (report) - if (in.good()) - { - cout << "Reading file " << fname << endl; - } - else - { - cout << fname << ": no such file" << endl; - exit(1); - } - - in >> nparticles >> nbonds >> ntriang >> ndihedrals; - - if (report) - if (in.good()) - { - cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang << " triangles and " << ndihedrals << " dihedrals" << endl; - } - else - { - cout << "Couldn't parse the file" << endl; - exit(1); - } - - // Atoms section - real *xyzuvw_host = new real[6*nparticles]; - - int cur = 0; - int tmp1, tmp2, aid; - while (in.good() && cur < nparticles) - { - in >> tmp1 >> tmp2 >> aid >> xyzuvw_host[6*cur+0] >> xyzuvw_host[6*cur+1] >> xyzuvw_host[6*cur+2]; - xyzuvw_host[6*cur+3] = xyzuvw_host[6*cur+4] = xyzuvw_host[6*cur+5] = 0; + ifstream in(fname); + string line; + + if (report) + if (in.good()) + { + cout << "Reading file " << fname << endl; + } + else + { + cout << fname << ": no such file" << endl; + exit(1); + } + + in >> nparticles >> nbonds >> ntriang >> ndihedrals; + + + if (in.good()) + { + if (report || nparticles <= 0 || ntriang <= 0) + cout << "File contains " << nparticles << " atoms, " << nbonds << " bonds, " << ntriang << " triangles and " << ndihedrals << " dihedrals" << endl; + } + else + { + cout << "Couldn't parse the file" << endl; + exit(1); + } + + if (nparticles <= 0 || ntriang <= 0) abort(); + + // Atoms section + real *xyzuvw_host = new real[6*nparticles]; + + int cur = 0; + int tmp1, tmp2, aid; + while (in.good() && cur < nparticles) + { + in >> tmp1 >> tmp2 >> aid >> xyzuvw_host[6*cur+0] >> xyzuvw_host[6*cur+1] >> xyzuvw_host[6*cur+2]; + xyzuvw_host[6*cur+3] = xyzuvw_host[6*cur+4] = xyzuvw_host[6*cur+5] = 0; // Scale in dpd units xyzuvw_host[6*cur+0] *= scale; xyzuvw_host[6*cur+1] *= scale; xyzuvw_host[6*cur+2] *= scale; - if (aid != 1) break; - cur++; - } + if (aid != 1) break; + cur++; + } - // Shift the origin of "zeroth" rbc to 0,0,0 - float xmin[3] = { 1e10, 1e10, 1e10}; - float xmax[3] = {-1e10, -1e10, -1e10}; + // Shift the origin of "zeroth" rbc to 0,0,0 + float xmin[3] = { 1e10, 1e10, 1e10}; + float xmax[3] = {-1e10, -1e10, -1e10}; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1; - id0--; id1--; - bonds_host[2*i + 0] = id0; - bonds_host[2*i + 1] = id1; - } + int *bonds_host = new int[nbonds * 2]; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1; + id0--; id1--; + bonds_host[2*i + 0] = id0; + bonds_host[2*i + 1] = id1; + } - // Angles section --> triangles + // Angles section --> triangles triangles_host = new int[4*ntriang]; triplets = new int[3*ntriang]; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2; - id0--; id1--; id2--; + id0--; id1--; id2--; triangles_host[4*i + 0] = triplets[3*i + 0] = id0; triangles_host[4*i + 1] = triplets[3*i + 1] = id1; triangles_host[4*i + 2] = triplets[3*i + 2] = id2; - } + } - // Dihedrals section + // Dihedrals section - int *dihedrals_host = new int[4*ndihedrals]; - for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2 >> id3; - id0--; id1--; id2--; id3--; + int *dihedrals_host = new int[4*ndihedrals]; + for (int i=0; i> tmp1 >> tmp2 >> id0 >> id1 >> id2 >> id3; + id0--; id1--; id2--; id3--; + + dihedrals_host[4*i + 0] = id0; + dihedrals_host[4*i + 1] = id1; + dihedrals_host[4*i + 2] = id2; + dihedrals_host[4*i + 3] = id3; + } - dihedrals_host[4*i + 0] = id0; - dihedrals_host[4*i + 1] = id1; - dihedrals_host[4*i + 2] = id2; - dihedrals_host[4*i + 3] = id3; - } + in.close(); - in.close(); + nvertices = nparticles; + int *dummyiii; + if ( cudaMalloc(&dummyiii, sizeof(int)) == cudaErrorDevicesUnavailable ) + return; + else + gpuErrchk(cudaFree(dummyiii)); - gpuErrchk( cudaMalloc(&orig_xyzuvw, nparticles * 6 * sizeof(float)) ); + gpuErrchk( cudaMalloc(&orig_xyzuvw, nparticles * 6 * sizeof(float)) ); gpuErrchk( cudaMalloc(&triangles, ntriang * 4 * sizeof(int)) ); - gpuErrchk( cudaMalloc(&dihedrals, ndihedrals * 4 * sizeof(int)) ); + gpuErrchk( cudaMalloc(&dihedrals, ndihedrals * 4 * sizeof(int)) ); - gpuErrchk( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nparticles * 6 * sizeof(float), cudaMemcpyHostToDevice) ); + gpuErrchk( cudaMemcpy(orig_xyzuvw, xyzuvw_host, nparticles * 6 * sizeof(float), cudaMemcpyHostToDevice) ); gpuErrchk( cudaMemcpy(triangles, triangles_host, ntriang * 4 * sizeof(int), cudaMemcpyHostToDevice) ); - gpuErrchk( cudaMemcpy(dihedrals, dihedrals_host, ndihedrals * 4 * sizeof(int), cudaMemcpyHostToDevice) ); + gpuErrchk( cudaMemcpy(dihedrals, dihedrals_host, ndihedrals * 4 * sizeof(int), cudaMemcpyHostToDevice) ); - delete[] xyzuvw_host; - delete[] dihedrals_host; + delete[] xyzuvw_host; + delete[] dihedrals_host; - nvertices = nparticles; - host_extent.xmin = xmin[0] - origin[0]; - host_extent.ymin = xmin[1] - origin[1]; - host_extent.zmin = xmin[2] - origin[2]; + host_extent.xmin = xmin[0] - origin[0]; + host_extent.ymin = xmin[1] - origin[1]; + host_extent.zmin = xmin[2] - origin[2]; - host_extent.xmax = xmax[0] - origin[0]; - host_extent.ymax = xmax[1] - origin[1]; - host_extent.zmax = xmax[2] - origin[2]; + host_extent.xmax = xmax[0] - origin[0]; + host_extent.ymax = xmax[1] - origin[1]; + host_extent.zmax = xmax[2] - origin[2]; maxCells = 5; gpuErrchk( cudaMalloc(&host_av, maxCells * 2 * sizeof(float)) ); @@ -225,7 +233,7 @@ __constant__ float A[4][4]; dummy = new Extent[maxCells]; - unitsSetup(1.64, 0.00141, 19.0476, 120, 12000, 12000, 0, 1256, 4189, 1e-6/ scale, 2.4295e-6, 4, false); + unitsSetup(1.64, 0.00141, 19.0476, 120, 40000, 40000, 0, 660, 1596, 1e-6/ scale, 2.4295e-6, 4, false); } void unitsSetup(float lmax, float p, float cq, float kb, float ka, float kv, float gammaC, @@ -243,10 +251,10 @@ __constant__ float A[4][4]; params.kbT = 580 * 250 * pow(ll, -2.0) * pow(tt, 2.0); params.p = p / ll; params.lmax = lmax / ll; - params.q = 1; + params.q = 1; params.Cq = cq * params.kbT * pow(ll, -2.0); params.totArea0 = totArea0 * pow(ll, -2.0); - params.area0 = params.totArea0 / (float)ntriang; + params.area0 = params.totArea0 / (float)ntriang; params.totVolume0 = totVolume0 * pow(ll, -3.0); params.ka = params.kbT * ka / (l0*l0); params.kd = params.kbT * 0.0 / (l0*l0); @@ -254,15 +262,15 @@ __constant__ float A[4][4]; params.gammaC = gammaC * 580 * pow(tt, 1.0); params.gammaT = 3.0 * params.gammaC; - params.rc = 0.5; - params.aij = 100; + params.rc = 0.5; + params.aij = 100; params.gamma = 15; - params.sigma = sqrt(2 * params.gamma * params.kbT); + params.sigma = sqrt(2 * params.gamma * params.kbT); // params.dt = dt; float phi = 2.7 / 180.0*M_PI; //float phi = 3.1 / 180.0*M_PI; - params.sinTheta0 = sin(phi); - params.cosTheta0 = cos(phi); + params.sinTheta0 = sin(phi); + params.cosTheta0 = cos(phi); params.kb = kb * params.kbT; params.mass = 1.1 / 0.995 * params.totVolume0 * ndens / nparticles; @@ -316,12 +324,12 @@ __constant__ float A[4][4]; printf("\t area %12.5f (%12.5f)\n", totArea0, params.totArea0); printf("\t volume %12.5f (%12.5f)\n", totVolume0, params.totVolume0); printf("************* **************** *************\n\n"); -} + } } int get_nvertices() { - return nparticles; + return nparticles; } Params& get_params() @@ -329,90 +337,90 @@ __constant__ float A[4][4]; return params; } -__global__ void transformKernel(float* xyzuvw, int n) -{ - int i = blockIdx.x * blockDim.x + threadIdx.x; - if (i >= n) return; + __global__ void transformKernel(float* xyzuvw, int n) + { + int i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= n) return; - float x = xyzuvw[6*i + 0]; - float y = xyzuvw[6*i + 1]; - float z = xyzuvw[6*i + 2]; + float x = xyzuvw[6*i + 0]; + float y = xyzuvw[6*i + 1]; + float z = xyzuvw[6*i + 2]; - xyzuvw[6*i + 0] = A[0][0]*x + A[0][1]*y + A[0][2]*z + A[0][3]; - xyzuvw[6*i + 1] = A[1][0]*x + A[1][1]*y + A[1][2]*z + A[1][3]; - xyzuvw[6*i + 2] = A[2][0]*x + A[2][1]*y + A[2][2]*z + A[2][3]; -} + xyzuvw[6*i + 0] = A[0][0]*x + A[0][1]*y + A[0][2]*z + A[0][3]; + xyzuvw[6*i + 1] = A[1][0]*x + A[1][1]*y + A[1][2]*z + A[1][3]; + xyzuvw[6*i + 2] = A[2][0]*x + A[2][1]*y + A[2][2]*z + A[2][3]; + } -void initialize(float *device_xyzuvw, const float (*transform)[4]) -{ - const int threads = 128; - const int blocks = (nparticles + threads - 1) / threads; + void initialize(float *device_xyzuvw, const float (*transform)[4]) + { + const int threads = 128; + const int blocks = (nparticles + threads - 1) / threads; - gpuErrchk( cudaMemcpyToSymbol(A, transform, 16 * sizeof(float)) ); - gpuErrchk( cudaMemcpy(device_xyzuvw, orig_xyzuvw, 6*nparticles * sizeof(float), cudaMemcpyDeviceToDevice) ); - transformKernel<<>>(device_xyzuvw, nparticles); -} + gpuErrchk( cudaMemcpyToSymbol(A, transform, 16 * sizeof(float)) ); + gpuErrchk( cudaMemcpy(device_xyzuvw, orig_xyzuvw, 6*nparticles * sizeof(float), cudaMemcpyDeviceToDevice) ); + transformKernel<<>>(device_xyzuvw, nparticles); + } -__inline__ __host__ __device__ float3 fminf(float3 a, float3 b) -{ - return make_float3(min(a.x,b.x), min(a.y,b.y), min(a.z,b.z)); -} + __inline__ __host__ __device__ float3 fminf(float3 a, float3 b) + { + return make_float3(min(a.x,b.x), min(a.y,b.y), min(a.z,b.z)); + } -__inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) -{ - return make_float3(max(a.x,b.x), max(a.y,b.y), max(a.z,b.z)); -} + __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) + { + return make_float3(max(a.x,b.x), max(a.y,b.y), max(a.z,b.z)); + } __device__ __inline__ float atomicMin(float *addr, float value) -{ - float old = *addr, assumed; - if(old <= value) return old; + { + float old = *addr, assumed; + if(old <= value) return old; - do - { - assumed = old; - old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(min(value, assumed))) ); - }while(old!=assumed); + do + { + assumed = old; + old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(min(value, assumed))) ); + }while(old!=assumed); - return old; -} + return old; + } __device__ __inline__ float atomicMax(float *addr, float value) -{ - float old = *addr, assumed; - if(old >= value) return old; + { + float old = *addr, assumed; + if(old >= value) return old; - do - { - assumed = old; - old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(max(value, assumed))) ); - }while(old!=assumed); + do + { + assumed = old; + old = __int_as_float( atomicCAS((unsigned int*)addr, __float_as_int(assumed), __float_as_int(max(value, assumed))) ); + }while(old!=assumed); - return old; -} + return old; + } __global__ void extentKernel(const float* const __restrict__ xyzuvw, Extent* extent, int npart) -{ - float3 loBound = make_float3( 1e10f, 1e10f, 1e10f); - float3 hiBound = make_float3(-1e10f, -1e10f, -1e10f); + { + float3 loBound = make_float3( 1e10f, 1e10f, 1e10f); + float3 hiBound = make_float3(-1e10f, -1e10f, -1e10f); const int cid = blockIdx.y; - for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < npart; i += blockDim.x * gridDim.x) - { + for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < npart; i += blockDim.x * gridDim.x) + { const float* addr = xyzuvw + 6 * (devParams.nparticles*cid + i); float3 v = make_float3(addr[0], addr[1], addr[2]); - loBound = fminf(loBound, v); - hiBound = fmaxf(hiBound, v); - } + loBound = fminf(loBound, v); + hiBound = fmaxf(hiBound, v); + } - loBound = warpReduceMin(loBound); - __syncthreads(); - hiBound = warpReduceMax(hiBound); + loBound = warpReduceMin(loBound); + __syncthreads(); + hiBound = warpReduceMax(hiBound); - if ((threadIdx.x & (warpSize - 1)) == 0) - { + if ((threadIdx.x & (warpSize - 1)) == 0) + { atomicMin(&extent[cid].xmin, loBound.x); atomicMin(&extent[cid].ymin, loBound.y); atomicMin(&extent[cid].zmin, loBound.z); @@ -420,11 +428,11 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) atomicMax(&extent[cid].xmax, hiBound.x); atomicMax(&extent[cid].ymax, hiBound.y); atomicMax(&extent[cid].zmax, hiBound.z); - } -} + } + } void extent_nohost(cudaStream_t stream, int ncells, const float * const xyzuvw, Extent * device_extent, int n) -{ + { if (ncells == 0) return; dim3 threads(32*3, 1); @@ -449,10 +457,10 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) gpuErrchk( cudaMemcpy(device_extent, dummy, ncells * sizeof(Extent), cudaMemcpyHostToDevice) ); - if (n == -1) n = nparticles; - extentKernel<<>>(xyzuvw, device_extent, n); + if (n == -1) n = nparticles; + extentKernel<<>>(xyzuvw, device_extent, n); gpuErrchk( cudaPeekAtLastError() ); -} + } __device__ __inline__ vec3 tex2vec(int id) { @@ -462,29 +470,29 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) } __global__ void areaAndVolumeKernel() -{ - float2 a_v = make_float2(0.0f, 0.0f); + { + float2 a_v = make_float2(0.0f, 0.0f); const int cid = blockIdx.y; for(int i = blockIdx.x * blockDim.x + threadIdx.x; i < devParams.ntriang; i += blockDim.x * gridDim.x) - { + { int4 ids = tex1Dfetch(texTriangles4, i); vec3 v0( tex2vec(6*(ids.x+cid*devParams.nparticles)) ); vec3 v1( tex2vec(6*(ids.y+cid*devParams.nparticles)) ); vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); - a_v.x += 0.5f * norm(cross(v1 - v0, v2 - v0)); - a_v.y += 0.1666666667f * (- v0.z*v1.y*v2.x + v0.z*v1.x*v2.y + v0.y*v1.z*v2.x - - v0.x*v1.z*v2.y - v0.y*v1.x*v2.z + v0.x*v1.y*v2.z); - } + a_v.x += 0.5f * norm(cross(v1 - v0, v2 - v0)); + a_v.y += 0.1666666667f * (- v0.z*v1.y*v2.x + v0.z*v1.x*v2.y + v0.y*v1.z*v2.x + - v0.x*v1.z*v2.y - v0.y*v1.x*v2.z + v0.x*v1.y*v2.z); + } - a_v = warpReduceSum(a_v); - if ((threadIdx.x & (warpSize - 1)) == 0) - { + a_v = warpReduceSum(a_v); + if ((threadIdx.x & (warpSize - 1)) == 0) + { atomicAdd(&totA_V[2*cid+0], a_v.x); atomicAdd(&totA_V[2*cid+1], a_v.y); - } -} + } + } __global__ void perTriangle(float* fxfyfz) { @@ -500,44 +508,44 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) vec3 v1( tex2vec(6*(ids.y+cid*devParams.nparticles)) ); vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); - vec3 ksi = cross(v1 - v0, v2 - v0); - float area = 0.5f * norm(ksi); + vec3 ksi = cross(v1 - v0, v2 - v0); + float area = 0.5f * norm(ksi); - // in-plane + // in-plane float alpha = 0.25f * devParams.q*devParams.Cq / powf(area, devParams.q+2.0f); - // area conservation + // area conservation float beta_a = -0.25f * ( devParams.ka*(totArea - devParams.totArea0) / (devParams.totArea0*area) + devParams.kd * (area - devParams.area0) / (devParams.area0 * area) ); - alpha += beta_a; - vec3 f0, f1, f2; + alpha += beta_a; + vec3 f0, f1, f2; - f0 = cross(ksi, v2-v1)*alpha; - f1 = cross(ksi, v0-v2)*alpha; - f2 = cross(ksi, v1-v0)*alpha; + f0 = cross(ksi, v2-v1)*alpha; + f1 = cross(ksi, v0-v2)*alpha; + f2 = cross(ksi, v1-v0)*alpha; - // volume conservation + // volume conservation // "-" here is because the normals look inside - vec3 ksi_3 = ksi*0.333333333f; - vec3 t_c = (v0 + v1 + v2) * 0.333333333f; + vec3 ksi_3 = ksi*0.333333333f; + vec3 t_c = (v0 + v1 + v2) * 0.333333333f; float beta_v = -0.1666666667f * devParams.kv * (totVolume - devParams.totVolume0) / (devParams.totVolume0); - f0 += (ksi_3 + cross(t_c, v2-v1)) * beta_v; - f1 += (ksi_3 + cross(t_c, v0-v2)) * beta_v; - f2 += (ksi_3 + cross(t_c, v1-v0)) * beta_v; + f0 += (ksi_3 + cross(t_c, v2-v1)) * beta_v; + f1 += (ksi_3 + cross(t_c, v0-v2)) * beta_v; + f2 += (ksi_3 + cross(t_c, v1-v0)) * beta_v; float* addr = fxfyfz + 3*cid*devParams.nparticles; #pragma unroll for (int d = 0; d<3; d++) - { + { atomicAdd(addr + 3*ids.x + d, f0[d]); atomicAdd(addr + 3*ids.y + d, f1[d]); atomicAdd(addr + 3*ids.z + d, f2[d]); - } -} + } + } __global__ void perDihedral(float* fxfyfz) -{ + { const int i = blockIdx.x * blockDim.x + threadIdx.x; const int cid = blockIdx.y; if (i >= devParams.ndihedrals) return; @@ -548,67 +556,67 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) vec3 v2( tex2vec(6*(ids.z+cid*devParams.nparticles)) ); vec3 v3( tex2vec(6*(ids.w+cid*devParams.nparticles)) ); - vec3 f0, f1, f2, f3; + vec3 f0, f1, f2, f3; - vec3 d21 = v2 - v1; - float r = norm(d21); - if (r < 0.0001) r = 0.0001; + vec3 d21 = v2 - v1; + float r = norm(d21); + if (r < 0.0001) r = 0.0001; float xx = r/devParams.lmax; float IbforceI = devParams.kbT / devParams.p * ( 0.25f/((1.0f-xx)*(1.0f-xx)) - 0.25f + xx ) / r; // TODO: minus?? - vec3 bforce = d21*IbforceI; - f1 += bforce; - f2 -= bforce; + vec3 bforce = d21*IbforceI; + f1 += bforce; + f2 -= bforce; - // Friction force + // Friction force vec3 u1( tex2vec(6*(ids.y+cid*devParams.nparticles) + 3) ); vec3 u2( tex2vec(6*(ids.z+cid*devParams.nparticles) + 3) ); - vec3 du21 = u2 - u1; + vec3 du21 = u2 - u1; vec3 dforce = du21*devParams.gammaT + d21 * devParams.gammaC * dot(du21, d21) / (r*r); - f1 += dforce; - f2 -= dforce; - //printf("%f %f %f\n", dforce.x, dforce.y, dforce.z); + f1 += dforce; + f2 -= dforce; + //printf("%f %f %f\n", dforce.x, dforce.y, dforce.z); - vec3 ksi = cross(v0 - v1, v0 - v2); - vec3 dzeta = cross(v2 - v3, v1 - v3); - vec3 t_c0 = (v0 + v1 + v2) * 0.3333333333f; - vec3 t_c1 = (v1 + v2 + v3) * 0.3333333333f; + vec3 ksi = cross(v0 - v1, v0 - v2); + vec3 dzeta = cross(v2 - v3, v1 - v3); + vec3 t_c0 = (v0 + v1 + v2) * 0.3333333333f; + vec3 t_c1 = (v1 + v2 + v3) * 0.3333333333f; - float IksiI = norm(ksi); - float IdzetaI = norm(dzeta); - float cosTheta = dot(ksi, dzeta) / (IksiI * IdzetaI); + float IksiI = norm(ksi); + float IdzetaI = norm(dzeta); + float cosTheta = dot(ksi, dzeta) / (IksiI * IdzetaI); - float IsinThetaI = sqrt(fabs(1.0f - cosTheta*cosTheta)); // TODO use copysign - if (fabs(IsinThetaI) < 0.001f) IsinThetaI = 0.001f; + float IsinThetaI = sqrt(fabs(1.0f - cosTheta*cosTheta)); // TODO use copysign + if (fabs(IsinThetaI) < 0.001f) IsinThetaI = 0.001f; - float sinTheta = IsinThetaI; - if (dot(ksi - dzeta, t_c0 - t_c1) > 0.0f) sinTheta = -sinTheta; // ">" because the normals look inside + float sinTheta = IsinThetaI; + if (dot(ksi - dzeta, t_c0 - t_c1) > 0.0f) sinTheta = -sinTheta; // ">" because the normals look inside float beta_b = devParams.kb * (sinTheta * devParams.cosTheta0 - cosTheta * devParams.sinTheta0) / sinTheta; - float b11 = -beta_b * cosTheta / (IksiI*IksiI); - float b12 = beta_b / (IksiI*IdzetaI); - float b22 = -beta_b * cosTheta / (IdzetaI*IdzetaI); + float b11 = -beta_b * cosTheta / (IksiI*IksiI); + float b12 = beta_b / (IksiI*IdzetaI); + float b22 = -beta_b * cosTheta / (IdzetaI*IdzetaI); - f0 += cross(ksi, v2 - v1)*b11 + cross(dzeta, v2 - v1)*b12; - f1 += cross(ksi, v0 - v2)*b11 + ( cross(ksi, v2 - v3) + cross(dzeta, v0 - v2) )*b12 + cross(dzeta, v2 - v3)*b22; - f2 += cross(ksi, v1 - v0)*b11 + ( cross(ksi, v3 - v1) + cross(dzeta, v1 - v0) )*b12 + cross(dzeta, v3 - v1)*b22; - f3 += cross(ksi, v1 - v2)*b12 + cross(dzeta, v1 - v2)*b22; + f0 += cross(ksi, v2 - v1)*b11 + cross(dzeta, v2 - v1)*b12; + f1 += cross(ksi, v0 - v2)*b11 + ( cross(ksi, v2 - v3) + cross(dzeta, v0 - v2) )*b12 + cross(dzeta, v2 - v3)*b22; + f2 += cross(ksi, v1 - v0)*b11 + ( cross(ksi, v3 - v1) + cross(dzeta, v1 - v0) )*b12 + cross(dzeta, v3 - v1)*b22; + f3 += cross(ksi, v1 - v2)*b12 + cross(dzeta, v1 - v2)*b22; float* addr = fxfyfz + 3*cid*devParams.nparticles; #pragma unroll for (int d = 0; d<3; d++) - { + { atomicAdd(addr + 3*ids.x + d, f0[d]); atomicAdd(addr + 3*ids.y + d, f1[d]); atomicAdd(addr + 3*ids.z + d, f2[d]); atomicAdd(addr + 3*ids.w + d, f3[d]); - } -} + } + } void forces_nohost(cudaStream_t stream, int ncells, const float * const device_xyzuvw, float * const device_axayaz) -{ + { if (ncells == 0) return; if (ncells > maxCells) @@ -620,7 +628,7 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) delete[] dummy; dummy = new Extent[maxCells]; - } + } size_t textureoffset; gpuErrchk( cudaBindTexture(&textureoffset, &texParticles, device_xyzuvw, &texParticles.channelDesc, ncells * nparticles * 6 * sizeof(float)) ); @@ -646,17 +654,17 @@ __inline__ __host__ __device__ float3 fmaxf(float3 a, float3 b) perTriangle<<>>(device_axayaz); gpuErrchk( cudaPeekAtLastError() ); gpuErrchk( cudaUnbindTexture(texParticles) ); - } + } -void get_triangle_indexing(int (*&host_triplets_ptr)[3], int& ntriangles) -{ + void get_triangle_indexing(int (*&host_triplets_ptr)[3], int& ntriangles) + { host_triplets_ptr = (int(*)[3])triplets; - ntriangles = ntriang; -} + ntriangles = ntriang; + } float* get_orig_xyzuvw() - { + { return orig_xyzuvw; - } + } } diff --git a/cuda-dpd/dpd/Makefile b/cuda-dpd/dpd/Makefile index 81c92431f..3a8d0cc06 100644 --- a/cuda-dpd/dpd/Makefile +++ b/cuda-dpd/dpd/Makefile @@ -35,8 +35,8 @@ NVCCFLAGS += -DVISCOSITY_S_LEVEL=$(slevel) -lineinfo -Xptxas -v test-dpd: main.cpp libcuda-dpd.so $(CXX) $(CXXFLAGS) $^ -lcudart -lcurand -o test-dpd -libcuda-dpd.a: cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o celllists - ar rcs $@ cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o +libcuda-dpd.a: cuda-dpd.o cuda-dpd-bipartite.o stress.o ../profiler-dpd.o celllists + ar rcs $@ cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o stress.o libcuda-dpd.so: cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o celllists $(CXX) $(CXXFLAGS) -shared cuda-dpd.o cuda-dpd-bipartite.o ../profiler-dpd.o ../cell-lists.o ../cell-lists-faster.o -o libcuda-dpd.so -lcudart -lcurand @@ -48,6 +48,9 @@ cuda-dpd.o: $(CUDADPD) cuda-dpd.h cuda-dpd-bipartite.o: $(CUDADPDBIP) cuda-dpd.h $(NVCC) $(NVCCFLAGS) -c $(CUDADPDBIP) -o $@ +stress.o: stress.cu cuda-dpd.h + $(NVCC) $(NVCCFLAGS) -c $< -o $@ + ../%.o: make -C ../ $(@:../%=%) CXX="$(CXX)" NVCC="$(NVCC)" diff --git a/cuda-dpd/dpd/cuda-dpd.h b/cuda-dpd/dpd/cuda-dpd.h index dfeca681e..174870b52 100644 --- a/cuda-dpd/dpd/cuda-dpd.h +++ b/cuda-dpd/dpd/cuda-dpd.h @@ -29,7 +29,7 @@ template<> inline __device__ float viscosity_function<0>(float x){ return x; } void forces_dpd_cuda_nohost(const float * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, float * const _axayaz, const int np, - const int * const cellsstart, const int * const cellscount, + const int * const cellsstart, const int * const cellscount, const float rc, const float XL, const float YL, const float ZL, const float aij, @@ -37,7 +37,7 @@ void forces_dpd_cuda_nohost(const float * const xyzuvw, const float4 * const xyz const float sigma, const float invsqrtdt, const float seed1, - cudaStream_t stream); + cudaStream_t stream); void forces_dpd_cuda(const float * const xp, const float * const yp, const float * const zp, const float * const xv, const float * const yv, const float * const zv, @@ -62,3 +62,13 @@ void forces_dpd_cuda_bipartite_nohost(cudaStream_t stream, const float2 * const const int3 halo_ncells, const float aij, const float gamma, const float sigmaf, const float seed, const int mask, float * const axayaz); + +void compute_stress(const float * const xyzuvw, + const int np, + const int * const cellsstart, const int * const cellscount, + const int XL, const int YL, const int ZL, + const float aij, const float gamma, const float sigmaf, const float seed, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, + float * const sigma_yy, float * const sigma_yz, float * const sigma_zz, + float * const axayaz, + cudaStream_t stream); diff --git a/cuda-dpd/dpd/stress.cu b/cuda-dpd/dpd/stress.cu new file mode 100644 index 000000000..f4160e071 --- /dev/null +++ b/cuda-dpd/dpd/stress.cu @@ -0,0 +1,264 @@ +/* + * stress.cu + * Part of uDeviceX/cuda-dpd-sem/dpd/ + * + * Created and authored by Diego Rossinelli on 2015-09-29. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include + +#include "cuda-dpd.h" +#include "../dpd-rng.h" +#include "../hacks.h" + +namespace StressKernels +{ + struct InfoStress + { + int3 ncells; + float aij, gamma, sigmaf; + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz, *axayaz; + float seed; + }; + + __constant__ InfoStress info; + + texture texParticles; + texture texStart, texCount; + +#define _XCPB_ 2 +#define _YCPB_ 2 +#define _ZCPB_ 1 +#define CPB (_XCPB_ * _YCPB_ * _ZCPB_) + + __device__ float3 _dpd_interaction(const int dpid, const float3 xdest, const float3 udest, const int spid, const float2 stmp0, const float2 stmp1) + { + const int sentry = 3 * spid; + const float2 stmp2 = tex1Dfetch(texParticles, sentry + 2); + + const float _xr = xdest.x - stmp0.x; + const float _yr = xdest.y - stmp0.y; + const float _zr = xdest.z - stmp1.x; + const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; + assert(rij2 < 1); + + const float invrij = rsqrtf(rij2); + const float rij = rij2 * invrij; + const float argwr = 1 - rij; + const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(argwr); + + const float xr = _xr * invrij; + const float yr = _yr * invrij; + const float zr = _zr * invrij; + + const float rdotv = + xr * (udest.x - stmp1.y) + + yr * (udest.y - stmp2.x) + + zr * (udest.z - stmp2.y); + + const float myrandnr = Logistic::mean0var1(info.seed, min(spid, dpid), max(spid, dpid)); + + const float strength = info.aij * argwr - (info.gamma * wr * rdotv + info.sigmaf * myrandnr) * wr; + + return make_float3(strength * xr, strength * yr, strength * zr); + } + +#define __IMOD(x,y) ((x)-((x)/(y))*(y)) + + template + __global__ void stress_kernel() + { + int mycount = 0, myscan = 0; + + __shared__ int volatile starts[CPB][16], scan[CPB][16]; + + if (threadIdx.x < 14) + { + const int cbase = blockIdx.x * blockDim.y + threadIdx.y; + + int dx, dy, dz; + dx = dy = dz = threadIdx.x / 3; + dx = threadIdx.x - dx * 3 - 1; + dy = __IMOD(dy, 3) - 1; + dz = __IMOD(dz / 3, 3) - 1; + + int cid = cbase + dz * info.ncells.x * info.ncells.y + dy * info.ncells.x + dx; + + const bool valid_cid = (cid >= 0) && (cid < info.ncells.x * info.ncells.y * info.ncells.z); + + starts[threadIdx.y][threadIdx.x] = (valid_cid) ? tex1Dfetch(texStart, cid) : 0; + + myscan = mycount = (valid_cid) ? tex1Dfetch(texCount, cid) : 0; + } + +#pragma unroll + for(int L = 1; L < 16; L <<= 1) + myscan += (threadIdx.x >= L) * __shfl_up(myscan, L); + + if (threadIdx.x < 15) + scan[threadIdx.y][threadIdx.x] = myscan - mycount; + + const int subtid = threadIdx.x % COLS; + const int slot = threadIdx.x / COLS; + + const int dststart = starts[threadIdx.y][13]; + const int lastdst = dststart + scan[threadIdx.y][14] - scan[threadIdx.y][13]; + + const int nsrc = scan[threadIdx.y][14]; + const int nsrcext = scan[threadIdx.y][13]; + + for(int pid = subtid; pid < nsrc; pid += COLS) + { + const int key9 = 9 * (pid >= scan[threadIdx.y][9]); + + int key3 = 3 * (pid >= scan[threadIdx.y][key9 + 3]); + key3 += (key9 < 9) ? 3 * (pid >= scan[threadIdx.y][key9 + 6]) : 0; + + int spid = pid - scan[threadIdx.y][key3 + key9] + starts[threadIdx.y][key3 + key9]; + + const int sentry = 3 * spid; + const float2 stmp0 = tex1Dfetch(texParticles, sentry); + const float2 stmp1 = tex1Dfetch(texParticles, sentry + 1); + + for(int dpid = dststart + slot; dpid < lastdst; dpid += ROWS) + { + float3 xdest, udest; + + float2 dtmp0 = tex1Dfetch(texParticles, 3 * dpid); + xdest.x = dtmp0.x; + xdest.y = dtmp0.y; + + dtmp0 = tex1Dfetch(texParticles, 3 * dpid + 1); + xdest.z = dtmp0.x; + udest.x = dtmp0.y; + + dtmp0 = tex1Dfetch(texParticles, 3 * dpid + 2); + udest.y = dtmp0.x; + udest.z = dtmp0.y; + + const float rx = xdest.x - stmp0.x; + const float ry = xdest.y - stmp0.y; + const float rz = xdest.z - stmp1.x; + + const float d2 = rx * rx + ry * ry + rz * rz; + + if ((dpid != spid) && (d2 < 1.0f)) + { + const float3 f = _dpd_interaction(dpid, xdest, udest, spid, stmp0, stmp1); + + atomicAdd(info.sigma_xx + dpid, f.x * rx); + atomicAdd(info.sigma_xy + dpid, f.x * ry); + atomicAdd(info.sigma_xz + dpid, f.x * rz); + atomicAdd(info.sigma_yy + dpid, f.y * ry); + atomicAdd(info.sigma_yz + dpid, f.y * rz); + atomicAdd(info.sigma_zz + dpid, f.z * rz); + + if (info.axayaz) + { + atomicAdd(info.axayaz + 3 * dpid , f.x); + atomicAdd(info.axayaz + 3 * dpid + 1, f.y); + atomicAdd(info.axayaz + 3 * dpid + 2, f.z); + } + + if (pid < nsrcext) + { + atomicAdd(info.sigma_xx + spid, f.x * rx); + atomicAdd(info.sigma_xy + spid, f.x * ry); + atomicAdd(info.sigma_xz + spid, f.x * rz); + atomicAdd(info.sigma_yy + spid, f.y * ry); + atomicAdd(info.sigma_yz + spid, f.y * rz); + atomicAdd(info.sigma_zz + spid, f.z * rz); + + if (info.axayaz) + { + atomicAdd(info.axayaz + 3*spid , -f.x); + atomicAdd(info.axayaz + 3*spid + 1, -f.y); + atomicAdd(info.axayaz + 3*spid + 2, -f.z); + } + } + } + } + } + } + + bool computestress_init = false; +} + +using namespace StressKernels; + +void compute_stress(const float * const xyzuvw, + const int np, + const int * const cellsstart, const int * const cellscount, + const int XL, const int YL, const int ZL, + const float aij, const float gamma, const float sigmaf, const float seed, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, + float * const sigma_yy, float * const sigma_yz, float * const sigma_zz, + float * const axayaz, + cudaStream_t stream) +{ + if (np == 0) + { + printf("WARNING: stress_nohost called with np = %d\n", np); + return; + } + + if (!computestress_init) + { + texStart.channelDesc = cudaCreateChannelDesc(); + texStart.filterMode = cudaFilterModePoint; + texStart.mipmapFilterMode = cudaFilterModePoint; + texStart.normalized = 0; + + texCount.channelDesc = cudaCreateChannelDesc(); + texCount.filterMode = cudaFilterModePoint; + texCount.mipmapFilterMode = cudaFilterModePoint; + texCount.normalized = 0; + + texParticles.channelDesc = cudaCreateChannelDesc(); + texParticles.filterMode = cudaFilterModePoint; + texParticles.mipmapFilterMode = cudaFilterModePoint; + texParticles.normalized = 0; + + CUDA_CHECK(cudaFuncSetCacheConfig(stress_kernel<32, 1>, cudaFuncCachePreferL1)); + + computestress_init = true; + } + + size_t textureoffset; + CUDA_CHECK(cudaBindTexture(&textureoffset, &texParticles, xyzuvw, &texParticles.channelDesc, sizeof(float) * 6 * np)); + assert(textureoffset == 0); + + const int ncells = XL * YL * ZL; + + CUDA_CHECK(cudaBindTexture(&textureoffset, &texStart, cellsstart, &texStart.channelDesc, sizeof(int) * ncells)); + assert(textureoffset == 0); + CUDA_CHECK(cudaBindTexture(&textureoffset, &texCount, cellscount, &texCount.channelDesc, sizeof(int) * ncells)); + assert(textureoffset == 0); + + { + static InfoStress c = { make_int3(XL, YL, ZL), aij, gamma, sigmaf, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, + axayaz, seed }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(info, &c, sizeof(c), 0, cudaMemcpyHostToDevice, stream)); + } + + if (axayaz) + CUDA_CHECK(cudaMemsetAsync(axayaz, 0, sizeof(float) * 3 * np, stream)); + + float * const ptrs[] = { sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + for(int c = 0; c < 6; ++c) + CUDA_CHECK(cudaMemsetAsync(ptrs[c], 0, sizeof(float) * np, stream)); + + stress_kernel<32, 1><<<(ncells + CPB - 1) / CPB, dim3(32, CPB), 0, stream>>>(); + + CUDA_CHECK(cudaPeekAtLastError()); +} + diff --git a/cuda-rbc/rbc-cuda.cu b/cuda-rbc/rbc-cuda.cu index 340547181..bca181b94 100644 --- a/cuda-rbc/rbc-cuda.cu +++ b/cuda-rbc/rbc-cuda.cu @@ -309,7 +309,7 @@ namespace CudaRBC maxCells = 0; CUDA_CHECK( cudaMalloc(&host_av, 1 * 2 * sizeof(float)) ); - unitsSetup(1.64, 0.001412, 19.0476, 35, 2500, 3500, 50, 135, 91, 1e-6, 2.4295e-6, 4, report); + unitsSetup(1.233, 0.00198, 30, 13*8.8, 5000, 10000, 30, 135, 95, 1.0e-6, 8.7e-3, 4, report); CUDA_CHECK( cudaFuncSetCacheConfig(fall_kernel<498>, cudaFuncCachePreferL1) ); } @@ -317,14 +317,15 @@ namespace CudaRBC void unitsSetup(float lmax, float p, float cq, float kb, float ka, float kv, float gammaC, float totArea0, float totVolume0, float lunit, float tunit, int ndens, bool prn) { - const float lrbc = 1.000000e-06; - const float trbc = 3.009441e-03; + const float lrbc = 1.0e-6; + const float trbc = 8.7e-3;//3.009441e-03; //const float mrbc = 3.811958e-13; float ll = lunit / lrbc; float tt = tunit / trbc; + float EE = pow(ll, -2.0) * pow(tt, 2.0); - params.kbT = 580 * 250 * pow(ll, -2.0) * pow(tt, 2.0); + params.kbT = 0.05 * EE; params.p = p / ll; params.lmax = lmax / ll; params.q = 1; @@ -332,12 +333,12 @@ namespace CudaRBC params.totArea0 = totArea0 * pow(ll, -2.0); params.totVolume0 = totVolume0 * pow(ll, -3.0); params.l0 = sqrt(params.totArea0 / (2.0*params.nvertices - 4.) * 4.0/sqrt(3.0)); - params.ka = ka * params.kbT / (params.totArea0 * params.l0 * params.l0); - params.kv = kv * params.kbT / (6 * params.totVolume0 * powf(params.l0, 3)); - params.gammaC = gammaC * 580 * pow(tt, 1.0); + params.ka = ka * EE / (params.totArea0 * params.l0 * params.l0); + params.kv = kv * EE / (6 * params.totVolume0 * powf(params.l0, 3)); + params.gammaC = gammaC * pow(tt, 1.0); params.gammaT = 3.0 * params.gammaC; - float phi = 6.97 / 180.0*M_PI; + float phi = 0*6.97 / 180.0*M_PI; params.sinTheta0 = sin(phi); params.cosTheta0 = cos(phi); params.kb = kb * params.kbT; diff --git a/device-gen/2Dto3D/Makefile b/device-gen/2Dto3D/Makefile deleted file mode 100644 index b6d431f68..000000000 --- a/device-gen/2Dto3D/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -2Dto3D: main.cpp - g++ main.cpp -O0 -g3 -fopenmp -o 2Dto3D - -clean: - rm 2Dto3D - -.PHONY = clean diff --git a/device-gen/2Dto3D/main.cpp b/device-gen/2Dto3D/main.cpp deleted file mode 100644 index 6d8b48908..000000000 --- a/device-gen/2Dto3D/main.cpp +++ /dev/null @@ -1,90 +0,0 @@ -/* - * main.cpp - * Part of uDeviceX/device-gen/2Dto3D/ - * - * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include -#include -#include -#include -#include -#include - -using namespace std; - -int main(int argc, char ** argv) -{ - if (argc != 6) - { - printf("usage: ./2to3 \n"); - return 1; - } - - const float zextent = atof(argv[2]); - const float zmargin = atof(argv[3]); - const int NZ = atoi(argv[4]); - - int NX, NY; - float xextent, yextent; - - vector slice; - - { - printf("Reading file %s...\n", argv[1]); - FILE * f = fopen(argv[1], "r"); - assert(f != 0); - float zextentOld; - int NZOld; - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextentOld); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZOld); - printf("Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextentOld, NX, NY,NZOld); - slice.resize(NX * NY, 0.0f); - fread(&slice[0], sizeof(float), slice.size(), f); - fclose(f); - } - - printf("Generating data with extent [%f, %f, %f], dimensions [%d, %d, %d], zmargin %f\n", - xextent, yextent, zextent + 2 * zmargin, NX, NY, NZ, zmargin); - vector volume(NX * NY * NZ, 0.0f); - - const float z0 = -zextent * 0.5 - zmargin; - const float dz = (zextent + 2 * zmargin) / (NZ - 1); - -//#pragma omp parallel for - for(int iz = 0; iz < NZ; ++iz) - { - const float z = z0 + iz * dz; - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const float xysdf = slice[ix + NX * (NY - 1 - iy)]; // NY -1 to change Y-axis direction - const float zsdf = fabs(z) - zextent * 0.5; - float val; - if (xysdf < 0) - val = max(zsdf, xysdf); - else - val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); - - assert(iy + NY * (ix + NX * iz) < volume.size()); - assert(volume[iy + NY * (ix + NX * iz)] == 0.0f); - volume[iy + NY * (ix + NX * iz)] = val; - } - } - - { - FILE * f = fopen(argv[5], "w"); - assert(f != 0); - fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); //exchange X and Y - fprintf(f, "%d %d %d\n", NY, NX, NZ); - fwrite(&volume[0], sizeof(float), volume.size(), f); - fclose(f); - } -} - diff --git a/device-gen/README.md b/device-gen/README.md new file mode 100644 index 000000000..105e34647 --- /dev/null +++ b/device-gen/README.md @@ -0,0 +1,30 @@ +# Generate microfluidic geometry + +Set of scripts to generate device geometries as Signed Distance Function in \*.dat format. +The dat format consists of header and the binary float data: +``` + + + +``` +Where size is the geometry length units (typically microns), grid size defines how many grid points are there. + +## Parabolic funnels +Geometry mimicing the microfluidic device by McFaul et al [Cell separation based on size and deformability using microfluidic funnel ratchets](http://www.ncbi.nlm.nih.gov/pubmed/22517056) +To generate a this geometry with 10 rows and 20 columns and with walls in z direction of width 4: +``` +cd funnels +make +./funnel -nColumns=20 -nRows=10 -zMargin=4 -out=geom.dat +``` + +## Later displacement device +Geometry reproducing CTC-iChip1 module by Karabacak et al [Microfluidic, marker-free isolation of circulating tumor cells from blood samples](http://www.nature.com/nprot/journal/v9/n3/full/nprot.2014.044.html) + +To build geometry constisting of 13 columns and 59 rows repeated twice, with wall widht 2 and such grid resolution that 0.5 grid points correspond to 1 unit of length: +``` +cd ctc-ichip +make +./ctc-ichip -nColumns=13 -nRows=59 -nRepeat=2 -zMargin=2.0 -out=13x59x2-05.dat -zResolution=0.5 +``` + diff --git a/device-gen/common/2Dto3D.cpp b/device-gen/common/2Dto3D.cpp new file mode 100644 index 000000000..c1bab4dcc --- /dev/null +++ b/device-gen/common/2Dto3D.cpp @@ -0,0 +1,83 @@ +/* + * 2Dto3D.cpp + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#include "2Dto3D.h" +#include +#include +#include +#include +#include +#include +#include "common.h" + +using namespace std; + +void conver2Dto3D(const int NX, const int NY, const float xextent, const float yextent, const std::vector& slice, + const int NZ, const float zextent, const float zmargin, const std::string& fileName) +{ + printf("Generating data with extent [%f, %f, %f], dimensions [%d, %d, %d], zmargin %f\n", + xextent, yextent, zextent + 2 * zmargin, NX, NY, NZ, zmargin); + + vector outputslice(NX * NY, 0.0f); + + const float z0 = -zextent * 0.5 - zmargin; + const float dz = (zextent + 2 * zmargin) / (NZ - 1); + + FILE * f = fopen(fileName.c_str(), "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", yextent, xextent, zextent + 2.0f * zmargin); + fprintf(f, "%d %d %d\n", NY, NX, NZ); + + for(int iz = 0; iz < NZ; ++iz) + { + const float z = z0 + iz * dz; + +#pragma omp parallel for + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const float xysdf = slice[ix + NX * (NY - 1 - iy)]; // NY -1 to change Y-axis direction + float val = xysdf; + if (zmargin != 0.0f) { + const float zsdf = fabs(z) - zextent * 0.5; + if (xysdf < 0) + val = max(zsdf, xysdf); + else + val = (zsdf < 0) ? xysdf : sqrt(zsdf * zsdf + xysdf * xysdf); + } + assert(iy + NY * (ix) < outputslice.size()); + + assert(fabs(val) < 1e3); // to check that the value has reasonable range + outputslice[iy + NY * ix] = val; + } + + if (iz == 0) + { + unsigned char * ptr = (unsigned char *)&outputslice[0]; + if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) + { + ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; + printf("INFO: some symbols were changed while writing\n"); + } + } + + int result = fwrite(&outputslice.front(), sizeof(float), NX * NY, f); + + if (result != NX * NY) { + printf("ERROR: written less than expected"); + exit(3); + } + + } + + fclose(f); +} + diff --git a/device-gen/common/2Dto3D.h b/device-gen/common/2Dto3D.h new file mode 100644 index 000000000..482f7a36b --- /dev/null +++ b/device-gen/common/2Dto3D.h @@ -0,0 +1,17 @@ +/* + * 2Dto3D.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#pragma once +#include +#include + +void conver2Dto3D(const int NX, const int NY, const float xextent, const float yextent, const std::vector& slice, + const int NZ, const float zextent, const float zmargin, const std::string& fileName); diff --git a/device-gen/common/collage.cpp b/device-gen/common/collage.cpp new file mode 100644 index 000000000..6d2730d07 --- /dev/null +++ b/device-gen/common/collage.cpp @@ -0,0 +1,130 @@ +/* + * collage.cpp + * Part of CTC/device-gen/sdf-collage/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#include "collage.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "common.h" +using namespace std; + +void collageSDF(int NX, int NY, const vector< vector >& sampleSDF, vector& outputSDF) +{ + outputSDF.resize(sampleSDF.size() * sampleSDF[0].size()); + printf("SIZE: %d\n", outputSDF.size()); + const int stride = NX; + for(int iy = 0; iy < sampleSDF.size() * NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const int dst = ix + stride * iy; + const int iobst = iy / NY; + assert(ix + NX * (iy - iobst*NY) < sampleSDF[iobst].size()); + outputSDF[dst] = sampleSDF[iobst][ix + NX * (iy - iobst*NY)]; + } +} + +void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, + const int xtimes, const int ytimes, std::vector& outputSDF) +{ + printf("Populate %d * %d times\n", xtimes, ytimes); + const int stride = xtimes * NX; + outputSDF.resize(xtimes * ytimes * NX * NY); + + for(int ty = 0; ty < ytimes; ++ty) + for(int tx = 0; tx < xtimes; ++tx) { + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + { + const int gx = ix + NX * tx; + const int gy = iy + NY * ty; + const int dst = gx + stride * gy; + + assert(dst < outputSDF.size()); + assert(ix + NX * iy < sampleSDF.size()); + outputSDF[dst] = sampleSDF[ix + NX * iy]; + } + } +} + +void collageSDFWithWall(const int NX, const int NY, const float xextent, const float yextent, + const vector< vector >& sampleSDF, const int ytimes, + const float paddingAdd, vector& outputSDF) +{ + collageSDF(NX, NY, sampleSDF, outputSDF); + const int xtimes = 1; + int outputSDFNX = xtimes * NX; + int outputSDFNY = ytimes * NY; + const float x0 = -xtimes * xextent * 0.5; + const float dx = xtimes * xextent / (outputSDFNX - 1); + + const float y0 = -ytimes * yextent * 0.5; + const float dy = ytimes * yextent / (outputSDFNY - 1); + + const float angle = (1.8/180.)*M_PI; + const float normal[] = {-cos(angle), sin(angle)}; + const float wallWidth = -2*y0*tan(angle); + + float ypick = 25.0f; //15 + float widthOfBufferZone = paddingAdd - wallWidth; + float xpick = (wallWidth) * (-2.0f*y0 - ypick) / (-2.0f*y0); + const float angle2 = atan(xpick/ypick); + std::cout << "YY = " << xpick << ", " << y0 + ypick << " ANGLE = " << angle2/M_PI*180 << std::endl; + const float normal2[] = {-cos(angle2), -sin(angle2)}; + + const float linePoint[] = {-x0 - wallWidth, y0}; + const float linePoint2[] = {-x0 - xpick -wallWidth, -y0 - ypick}; + + for (int iy = 0; iy < outputSDFNY; ++iy) + for (int ix = 0; ix < outputSDFNX; ++ix) + { + const float signX = sign(dx*ix + x0); + float p[] = {dx*ix + x0, dy*iy + y0}; + float padding = signbit(-p[0])*widthOfBufferZone; + float xsdf = -1e6; + + if ((signX == -1 && p[1] > (y0 + ypick)) || (signX == 1 && p[1] < (-y0 - ypick))) { + xsdf = -(normal[0]*(fabs(p[0]) - linePoint[0] + padding) - signX*normal[1]*(p[1] - linePoint[1])); + } else { + xsdf = -(normal2[0]*(fabs(p[0]) - linePoint2[0] - signbit(p[0])*wallWidth + padding) - normal2[1]*(fabs(p[1]) - linePoint2[1])); + } + + outputSDF[ix + outputSDFNX*iy] = std::max(outputSDF[ix + outputSDFNX*iy], xsdf); + } +} + +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& inputGrid, + const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid) +{ + float h = xextent / (NX - 1); + int ixshift = xshift / h; + int ipadding = xpadding / h + 1; + newNX = NX + ipadding; + newXextent = xextent + xpadding; + + float minVal = *std::min(inputGrid.begin(), inputGrid.end()); + outGrid.resize(newNX * NY, -1e6); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX ; ++ix) + { + int newIx = (ix + ixshift) % newNX; + assert(fabs(inputGrid[ix + NX * iy]) < 1e3); + outGrid[newIx + newNX * iy] = inputGrid[ix + NX * iy]; + } +} + diff --git a/device-gen/common/collage.h b/device-gen/common/collage.h new file mode 100644 index 000000000..ffe304c19 --- /dev/null +++ b/device-gen/common/collage.h @@ -0,0 +1,26 @@ +/* + * collage.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#pragma once + +#include + +void collageSDF(int NX, int NY, const std::vector< std::vector >& sampleSDF, std::vector& outputSDF); + +void collageSDFWithWall(const int NX, const int NY, const float xextent, const float yextent, + const std::vector< std::vector >& sampleSDF, const int ytimes, + const float paddingAdd, std::vector& outputSDF); + +void populateSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& sampleSDF, + const int xtimes, const int ytimes, std::vector& outputSDF); + +void shiftSDF(const int NX, const int NY, const float xextent, const float yextent, const std::vector& inputGrid, + const float xshift, const float xpadding, int& newNX, float& newXextent, std::vector& outGrid); diff --git a/device-gen/common/common.h b/device-gen/common/common.h new file mode 100644 index 000000000..1bc90cfc3 --- /dev/null +++ b/device-gen/common/common.h @@ -0,0 +1,67 @@ +/* + * common.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#pragma once +#include +#include +#include +#include +#include + +inline float sign(float x) { + return 1.0f - 2.0f*std::signbit(x); +} + +inline void readDAT(const std::string& fileName, std::vector& data, + int& NX, int& NY, int& NZ, float& xextent, float& yextent, float& zextent) +{ + FILE * f = fopen(fileName.c_str(), "r"); + assert(f != 0); + int result = fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); + assert(result == 3); + result = fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); + assert(result == 3); + printf("Read file %s. Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", + fileName.c_str(), xextent, yextent, zextent, NX, NY, NZ); + data.resize(NX * NY * NZ, 0.0f); + result = fread(&data[0], sizeof(float), NX * NY * NZ, f); + if (result != data.size()) { + printf("ERROR: read less than expected"); + exit(3); + } + fclose(f); +} + +inline void writeDAT(const std::string& fileName, std::vector& data, + const int NX, const int NY, const int NZ, float xextent, float yextent, float zextent) +{ + FILE * f = fopen(fileName.c_str(), "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", xextent, yextent, zextent); + fprintf(f, "%d %d %d\n", NX, NY, NZ); + + unsigned char * ptr = (unsigned char *)&data[0]; + if ((ptr[0] >= 9 && ptr[0] <= 13) || ptr[0] == 32 ) + { + ptr[0] = (ptr[0] == 32) ? 33 : (ptr[0] < 11) ? 8 : 14; + printf("INFO: some symbols were changed while writing\n"); + } + + int result = fwrite(&data[0], sizeof(float), (int)data.size(), f); + if (result != data.size()) { + printf("ERROR: written less than expected"); + exit(3); + } + + fclose(f); +} + diff --git a/device-gen/common/device-builder.h b/device-gen/common/device-builder.h new file mode 100644 index 000000000..736c9e01e --- /dev/null +++ b/device-gen/common/device-builder.h @@ -0,0 +1,36 @@ + +/* + * device-builder.h + * Part of CTC/device-gen/ctc-ichip/ + * + * Created and authored by Kirill Lykov on 2015-09-7. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#pragma once + +class DeviceBuilder +{ +protected: + typedef vector SDF; + int m_ncolumns, m_nrows; + float m_resolution, m_zmargin; + float m_unitSizeX, m_unitSizeY, m_unitSizeZ; // size of the egg with the empty space aroung it + std::string m_outFileName2D, m_outFileName3D; + int m_niterRedistance; + int m_unitNX, m_unitNY, m_unitNZ; +public: + DeviceBuilder(float unitSizeX, float unitSizeY, float unitSizeZ) + : m_ncolumns(0), m_nrows(0), m_resolution(0), m_zmargin(0), + m_unitSizeX(unitSizeX), m_unitSizeY(unitSizeY), m_unitSizeZ(unitSizeZ), + m_niterRedistance(1e3), m_unitNX(0), m_unitNY(0), m_unitNZ(0) + {} + + virtual void build() = 0; + + virtual ~DeviceBuilder() {} +}; diff --git a/device-gen/common/redistance.cpp b/device-gen/common/redistance.cpp new file mode 100644 index 000000000..ef32a05af --- /dev/null +++ b/device-gen/common/redistance.cpp @@ -0,0 +1,134 @@ +/* + * resistance.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ +#include "redistance.h" +#include +#include + + float Redistance::sussman_scheme(int ix, int iy, float sgn0) + { + const float phicenter = _ACCESS(m_phi, ix, iy); + + const float dphidxm = phicenter - _ACCESS(m_phi, ix - 1, iy); + const float dphidxp = _ACCESS(m_phi, ix + 1, iy) - phicenter; + const float dphidym = phicenter - _ACCESS(m_phi, ix, iy - 1); + const float dphidyp = _ACCESS(m_phi, ix, iy + 1) - phicenter; + + if (sgn0 == 1) + { + const float xgrad0 = std::max( max(0.0f, dphidxm), -std::min(0.0f, dphidxp)) * m_invdx; + const float ygrad0 = std::max( max(0.0f, dphidym), -min(0.0f, dphidyp)) * m_invdy; + + const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1.0f; + + return phicenter - m_dt * sgn0 * G0; + } + else + { + const float xgrad1 = std::max( -min(0.0f, dphidxm), std::max(0.0f, dphidxp)) * m_invdx; + const float ygrad1 = std::max( -min(0.0f, dphidym), std::max(0.0f, dphidyp)) * m_invdy; + + const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1.0f; + + return phicenter - m_dt * sgn0 * G1; + } + } + +Redistance::Redistance(const float dt, const float dx, const float dy, + const int xsize, const int ysize) +: m_xsize(xsize), m_ysize(ysize), m_dt(dt), m_dx(dx), m_dy(dy), m_invdx(1.0f/dx), m_invdy(1.0f/dy), m_phi0(nullptr), m_phi(nullptr) +{} + + void Redistance::run(const int iterations, float * field) + { + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const float deltax = m_dx * ((code % 3) - 1); + const float deltay = m_dy * ((code % 9) / 3 - 1); + + const float dl = sqrtf(deltax * deltax + deltay * deltay); + + m_dls[code] = dl; + } + + m_phi0 = new float[m_xsize * m_ysize]; + memcpy(m_phi0, field, sizeof(float) * m_xsize * m_ysize); + m_phi = field; + + float * tmp = new float[m_xsize * m_ysize]; + for(int t = 0; t < iterations; ++t) + { + if (t % 100 == 0) + printf("t: %d, size: %d %d\n", t, m_xsize, m_ysize); + +#pragma omp parallel for + for(int iy = 0; iy < m_ysize; ++iy) + for(int ix = 0; ix < m_xsize; ++ix) + { + const float myval0 = _ACCESS(m_phi0, ix, iy); + const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); + + const bool boundary = ( + ix == 0 || ix == m_xsize - 1 || + iy == 0 || iy == m_ysize - 1); + if (boundary) + tmp[ix + m_xsize * iy] = simple_scheme(ix, iy, sgn0, myval0); + else + { + if (anycrossing(ix, iy, sgn0)) + tmp[ix + m_xsize * iy] = myval0; + else + tmp[ix + m_xsize * iy] = sussman_scheme(ix, iy, sgn0); + } + assert(fabs(tmp[ix + m_xsize * iy]) < 1e7); + } + + memcpy(field, tmp, sizeof(float) * m_xsize * m_ysize); + } + + delete [] tmp; + delete [] m_phi0; + m_phi0 = nullptr; + } + + float Redistance::simple_scheme(int ix, int iy, float sgn0, float myphi0) + { + float mindistance = 1e6f; + for(int code = 0; code < 3 * 3; ++code) + { + if (code == 1 + 3) continue; + + const int xneighbor = ix + (code % 3) - 1; + const int yneighbor = iy + (code % 9) / 3 - 1; + + if (xneighbor < 0 || xneighbor >= m_xsize) continue; + if (yneighbor < 0 || yneighbor >= m_ysize) continue; + + const float phi0_neighbor = _ACCESS(m_phi0, xneighbor, yneighbor); + const float phi_neighbor = _ACCESS(m_phi, xneighbor, yneighbor); + + const float dl = m_dls[code]; + + float distance = 0; + + if (sgn0 * phi0_neighbor < 0) + distance = - myphi0 * dl / (phi0_neighbor - myphi0); + else + distance = dl + abs(phi_neighbor); + + mindistance = std::min(mindistance, distance); + } + + return sgn0 * mindistance; + } + diff --git a/device-gen/common/redistance.h b/device-gen/common/redistance.h new file mode 100644 index 000000000..6e815e960 --- /dev/null +++ b/device-gen/common/redistance.h @@ -0,0 +1,52 @@ +/* + * resistance.h + * Part of CTC/device-gen/common/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include + +using namespace std; + +#define _ACCESS(f, x, y) f[(x) + m_xsize * (y)] + +class Redistance +{ + int m_xsize, m_ysize; + float * m_phi0, * m_phi; + float m_dt, m_dx, m_dy, m_invdx, m_invdy; + float m_dls[9]; + + template + inline bool anycrossing_dir(int ix, int iy, const float sgn0) + { + const int dx = d == 0, dy = d == 1, dz = d == 2; + + const float fm1 = _ACCESS(m_phi0, ix - dx, iy - dy); + const float fp1 = _ACCESS(m_phi0, ix + dx, iy + dy); + + return (fm1 * sgn0 < 0 || fp1 * sgn0 < 0); + } + + inline bool anycrossing(int ix, int iy, const float sgn0) + { + return + anycrossing_dir<0>(ix, iy, sgn0) || + anycrossing_dir<1>(ix, iy, sgn0); + } + + float simple_scheme(int ix, int iy, float sgn0, float myphi0); + + float sussman_scheme(int ix, int iy, float sgn0); +public: + Redistance(const float dt, const float dx, const float dy, + const int xsize, const int ysize); + void run(const int iterations, float * field); +}; diff --git a/device-gen/ctc-ichip/Makefile b/device-gen/ctc-ichip/Makefile new file mode 100644 index 000000000..93dccdc3e --- /dev/null +++ b/device-gen/ctc-ichip/Makefile @@ -0,0 +1,19 @@ +CXX = g++-4.9 +CXXFLAGS += -O0 -g3 -std=c++11 -fopenmp + +ctc-ichip: *.cpp collage.o redistance.o 2Dto3D.o + $(CXX) $(CXXFLAGS) -I../../mpi-dpd/ collage.o redistance.o 2Dto3D.o main.cpp -o ctc-ichip + +collage.o: ../common/collage.h ../common/collage.cpp + $(CXX) $(CXXFLAGS) -c $^ + +redistance.o: ../common/redistance.h ../common/redistance.cpp + $(CXX) $(CXXFLAGS) -c $^ + +2Dto3D.o: ../common/2Dto3D.h ../common/2Dto3D.cpp + $(CXX) $(CXXFLAGS) -c $^ + +clean: + rm -f ctc-ichip *.o *.d *.h.gch + + diff --git a/device-gen/ctc-ichip/main.cpp b/device-gen/ctc-ichip/main.cpp new file mode 100644 index 000000000..2fa4f12ae --- /dev/null +++ b/device-gen/ctc-ichip/main.cpp @@ -0,0 +1,297 @@ +/* + * main.cpp + * Part of CTC/device-gen/ctc-ichip/ + * + * Created and authored by Kirill Lykov on 2015-09-7. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include "../common/device-builder.h" +#include "../common/common.h" +#include "../common/collage.h" +#include "../common/redistance.h" +#include "../common/2Dto3D.h" + +using namespace std; + +struct Egg +{ + float r1, r2, alpha; + + Egg() + : r1(12.0f), r2(8.5f), alpha(0.03f) + { + } + + float x2y(float x) const { + return sqrt(r2*r2 * exp(-alpha * x) * (1.0f - x*x/r1/r1)); + } + + void run(vector& vx, vector& vy) { + int N = 500; + float dx = 2.0f * r1 / (N - 1); + for (int i = 0; i < N; ++i) { + float x = i * dx - r1; + float y = x2y(x); + vx.push_back(x); + vy.push_back(y); + } + + auto vxRev = vx; + vx.insert(vx.end(), vxRev.rbegin(), vxRev.rend()); + + auto vyRev = vy; + for_each(vyRev.begin(), vyRev.end(), [](float& i) { i *= -1.0f; }); + vy.insert(vy.end(), vyRev.rbegin(), vyRev.rend()); + } +}; + +class CTCiChip1Builder : public DeviceBuilder +{ + int m_nrepeat; + const float m_angle; + float m_desiredSubdomainSzX; +public: + CTCiChip1Builder() + : DeviceBuilder(56.0f, 32.0f, 128.0f), + m_nrepeat(0), m_angle(1.7f * M_PI / 180.0f) + {} + + CTCiChip1Builder& setNColumns(int ncolumns) + { + m_ncolumns = ncolumns; + return *this; + } + + CTCiChip1Builder& setNRows(int nrows) + { + m_nrows = nrows; + return *this; + } + + CTCiChip1Builder& setRepeat(float nrepeat) + { + m_nrepeat = nrepeat; + return *this; + } + + CTCiChip1Builder& setResolution(float resolution) + { + m_resolution = resolution; + return *this; + } + + CTCiChip1Builder& setZWallWidth(float zmargin) + { + m_zmargin = zmargin; + return *this; + } + + CTCiChip1Builder& setDiseredSubdomainX(float x) + { + m_desiredSubdomainSzX = x; + return *this; + } + + CTCiChip1Builder& setFileNameFor2D(const std::string& outFileName2D) + { + m_outFileName2D = outFileName2D; + return *this; + } + + CTCiChip1Builder& setFileNameFor3D(const std::string& outFileName3D) + { + m_outFileName3D = outFileName3D; + return *this; + } + + void build(); + +private: + void generateUnitSDF(vector& sdf) const; + + void shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, + float& padding, float& addPadding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const; +}; + +void CTCiChip1Builder::build() +{ + if (m_ncolumns * m_nrows * m_nrepeat * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) + throw std::runtime_error("Invalid parameters"); + + // 1 Create 1 obstacle + m_unitNX = static_cast(m_unitSizeX * m_resolution); + m_unitNY = static_cast(m_unitSizeY * m_resolution); + m_unitNZ = static_cast(m_unitSizeZ * m_resolution); + + SDF eggSdf; + generateUnitSDF(eggSdf); + + // 2 Create 1 row of obstacles + int rowNX = m_ncolumns*m_unitNX; + int rowNY = m_unitNY; + int rowSizeX = m_ncolumns * m_unitSizeX; + int rowSizeY = m_unitSizeY; + SDF rowObstacles; + populateSDF(m_unitNX, m_unitNY, m_unitSizeX, m_unitSizeY, eggSdf, m_ncolumns, 1, rowObstacles); + + // 3 Shift rows + float padding = 0.0f; + float addPadding = 0.0f; + int shiftedRowNX = 0; // they are all the same length + float shiftedRowSizeX = 0.0f; + + std::vector shiftedRows; + shiftRows(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, padding, addPadding, shiftedRowNX, shiftedRowSizeX, shiftedRows); + + // 4 Collage rows + SDF finalSDF; + collageSDFWithWall(shiftedRowNX, rowNY, shiftedRowSizeX, rowSizeY, shiftedRows, m_nrows, addPadding, finalSDF); + + // 5 Apply redistancing for the result + float finalExtent[] = {shiftedRowSizeX, static_cast(m_nrows * rowSizeY)}; + int finalN[] = {shiftedRowNX, m_nrows*rowNY}; + const float dx = finalExtent[0] / (finalN[0] - 1); + const float dy = finalExtent[1] / (finalN[1] - 1); + Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); + redistancer.run(m_niterRedistance, &finalSDF[0]); + + // 6 Repeat this pattern + SDF finalSDF2; + populateSDF(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, 1, m_nrepeat, finalSDF2); + std::swap(finalSDF, finalSDF2); + + // 6 Write result to the file + if (m_outFileName2D.length() != 0) + writeDAT(m_outFileName2D, finalSDF, finalN[0], m_nrepeat * finalN[1], 1, finalExtent[0], m_nrepeat * finalExtent[1], 1.0f); + + conver2Dto3D(finalN[0], m_nrepeat * finalN[1], finalExtent[0], m_nrepeat*finalExtent[1], finalSDF, + m_unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); +} + +void CTCiChip1Builder::generateUnitSDF(vector& sdf) const +{ + vector xs, ys; + Egg egg; + egg.run(xs, ys); + + const float xlb = -m_unitSizeX/2.0f; + const float ylb = -m_unitSizeY/2.0f; + + sdf.resize(m_unitNX * m_unitNY, 0.0f); + const float dx = m_unitSizeX / (m_unitNX - 1); + const float dy = m_unitSizeY / (m_unitNY - 1); + const int nsamples = xs.size(); + + for(int iy = 0; iy < m_unitNY; ++iy) + for(int ix = 0; ix < m_unitNX; ++ix) + { + const float x = xlb + ix * dx; + const float y = ylb + iy * dy; + + float distance2 = 1e6; + int iclosest = 0; + for(int i = 0; i < nsamples ; ++i) + { + const float xd = xs[i] - x; + const float yd = ys[i] - y; + const float candidate = xd * xd + yd * yd; + + if (candidate < distance2) + { + iclosest = i; + distance2 = candidate; + } + } + + float s = -1; + + { + const float ycurve = egg.x2y(x); + if (x >= -egg.r1 && x <= egg.r1 && fabs(y) <= ycurve) + s = +1; + } + + + sdf[ix + m_unitNX * iy] = s * sqrt(distance2); + } +} + +void CTCiChip1Builder::shiftRows(int rowNX, int rowNY, float rowSizeX, float rowSizeY, const SDF& rowObstacles, + float& padding, float& addPadding, int& shiftedRowNX, float& shiftedRowSizeX,std::vector& shiftedRows) const +{ + const int nRowsPerShift = static_cast(ceil(m_unitSizeX / (m_unitSizeY * tan(m_angle)))); + if (fabs(m_unitSizeX / (m_unitSizeY * tan(m_angle)) - nRowsPerShift) > 1e-1) { + throw std::runtime_error("Suggest changing the angle"); + } + + padding = float(ceil(m_nrows * m_unitSizeY * tan(m_angle))); + // TODO Do I need this nUniqueRows? + int nUniqueRows = m_nrows; + if (m_nrows > nRowsPerShift) { + nUniqueRows = nRowsPerShift; + padding = float(round(nRowsPerShift * m_unitSizeY * tan(m_angle))); + } + + // TODO fix this stupid workaround + if (padding < 32.0f) + padding = 0.0f; + if (padding == 57.0f) + padding = m_unitSizeX; + + // additional hack to have domain size in X direction to be devisible by desiredSubdomainSzX + { + float origSzX = m_ncolumns*m_unitSizeX + padding; + addPadding = (int(origSzX/m_desiredSubdomainSzX) + 1)*m_desiredSubdomainSzX - origSzX; + padding = padding + addPadding; // adjust padding to have desired size + } + + std::cout << "Launching rows generation. New size = "<< m_ncolumns*m_unitSizeX + padding << std::endl; + shiftedRows.resize(nUniqueRows); + for (int i = 0; i < nUniqueRows; ++i) { + float xshift = (nUniqueRows - i - 1) * 32.0f * tan(m_angle); + shiftSDF(rowNX, rowNY, rowSizeX, rowSizeY, rowObstacles, xshift, padding, shiftedRowNX, shiftedRowSizeX, shiftedRows[i]); + } +} + + +int main(int argc, char ** argv) +{ + ArgumentParser argp(vector(argv, argv + argc)); + + int nColumns = argp("-nColumns").asInt(1); + int nRows = argp("-nRows").asInt(1); + int nRepeat = argp("-nRepeat").asInt(1); + float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); + float resolution = static_cast(argp("-zResolution").asDouble(1.0)); + std::string outFileName = argp("-out").asString("3d"); + + CTCiChip1Builder builder; + try { + builder.setNColumns(nColumns) + .setNRows(nRows) + .setRepeat(nRepeat) + .setResolution(resolution) + .setZWallWidth(zMargin) + .setFileNameFor2D("2d") + .setFileNameFor3D(outFileName) + .setDiseredSubdomainX(64.0f) + .build(); + } catch(const std::exception& ex) { + std::cout << "ERROR: " << ex.what() << std::endl; + } + return 0; +} + diff --git a/device-gen/funnels/Makefile b/device-gen/funnels/Makefile new file mode 100644 index 000000000..5c7799168 --- /dev/null +++ b/device-gen/funnels/Makefile @@ -0,0 +1,19 @@ +CXX = g++-4.9 +CXXFLAGS += -O0 -g3 -std=c++11 -fopenmp + +funnel: *.cpp collage.o redistance.o 2Dto3D.o + $(CXX) $(CXXFLAGS) -I../../mpi-dpd/ collage.o redistance.o 2Dto3D.o main.cpp -o funnel + +collage.o: ../common/collage.h ../common/collage.cpp + $(CXX) $(CXXFLAGS) -c $^ + +redistance.o: ../common/redistance.h ../common/redistance.cpp + $(CXX) $(CXXFLAGS) -c $^ + +2Dto3D.o: ../common/2Dto3D.h ../common/2Dto3D.cpp + $(CXX) $(CXXFLAGS) -c $^ + +clean: + rm -f test *.o *.d *.h.gch + + diff --git a/device-gen/funnels/main.cpp b/device-gen/funnels/main.cpp new file mode 100644 index 000000000..cf56b3aed --- /dev/null +++ b/device-gen/funnels/main.cpp @@ -0,0 +1,234 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include +#include "../common/device-builder.h" +#include "../common/common.h" +#include "../common/collage.h" +#include "../common/redistance.h" +#include "../common/2Dto3D.h" + +using namespace std; + +struct Parabola +{ + float x0; + float ymax; // length of obstacle, for cutting the pick + float y0; + + Parabola(float gap, float xextent) : ymax(48.0) + { + x0 = xextent/2.0f - gap/2.0f; + if (gap > 9.0) + y0 = ymax; + else + y0 = ymax + 6.5; + } + + void line1(vector& vx, vector& vy) { + int N = 500; + float dx = 2.0f * fabs(x0) / (N - 1); + for (int i = 0; i < N; ++i) { + float x = i * dx - x0; + float y = 0.0f; + vx.push_back(x); + vy.push_back(y); + } + } + + void line2(vector& vx, vector& vy) { + int N = 1500; + float dx = 2.0f * fabs(x0) / (N - 1); + float alpha = -y0 / (x0 * x0); + for (int i = 0; i < N; ++i) { + float x = i * dx - x0; + float y = min(ymax, alpha * x*x + y0); + vx.push_back(x); + vy.push_back(y); + } + } +}; + +class FunnelsBuilder : public DeviceBuilder +{ + float m_gapSpace; // unit gap between obstacles +public: + FunnelsBuilder() + : DeviceBuilder(24.0f, 96.0f, 58.0f), m_gapSpace(1.0f) + {} + + FunnelsBuilder& setNColumns(int ncolumns) + { + m_ncolumns = ncolumns; + return *this; + } + + FunnelsBuilder& setNRows(int nrows) + { + m_nrows = nrows; + return *this; + } + + FunnelsBuilder& setResolution(float resolution) + { + m_resolution = resolution; + return *this; + } + + FunnelsBuilder& setZWallWidth(float zmargin) + { + m_zmargin = zmargin; + return *this; + } + + FunnelsBuilder& setFileNameFor2D(const std::string& outFileName2D) + { + m_outFileName2D = outFileName2D; + return *this; + } + + FunnelsBuilder& setFileNameFor3D(const std::string& outFileName3D) + { + m_outFileName3D = outFileName3D; + return *this; + } + + void build(); + +private: + void generateUnitSDF(float gap, vector& sdf) const; +}; + +void FunnelsBuilder::generateUnitSDF(float gap, vector& sdf) const +{ + assert(m_unitNX * m_unitNY * m_unitNZ != 0); + vector xs, ys; + Parabola par(gap, m_unitSizeX); + par.line1(xs, ys); + par.line2(xs, ys); + + const float xlb = -m_unitSizeX/2.0f; + const float ylb = -(m_unitSizeY - par.ymax)/2.0f; + + sdf.resize(m_unitNX * m_unitNY, 0.0f); + const float dx = m_unitSizeX / (m_unitNX - 1); //TODO NX-1 + const float dy = m_unitSizeY / (m_unitNY - 1); + const int nsamples = xs.size(); + + for(int iy = 0; iy < m_unitNY; ++iy) + for(int ix = 0; ix < m_unitNX; ++ix) + { + const float x = xlb + ix * dx; + const float y = ylb + iy * dy; + + float distance2 = 1e6; + int iclosest = 0; + for(int i = 0; i < nsamples ; ++i) + { + const float xd = xs[i] - x; + const float yd = ys[i] - y; + const float candidate = xd * xd + yd * yd; + + if (candidate < distance2) + { + iclosest = i; + distance2 = candidate; + } + } + + float s = -1; + + { + const float alpha = -par.y0 / (par.x0 * par.x0); + const float ycurve = min(par.ymax, alpha * x*x + par.y0); + + if (x >= -par.x0 && x <= par.x0 && y >= 0 && y <= ycurve) + s = +1; + } + + + sdf[ix + m_unitNX * iy] = s * sqrt(distance2); + } +} + +void FunnelsBuilder::build() +{ + if (m_ncolumns * m_nrows * m_resolution * m_zmargin == 0.0f || m_outFileName3D.length() == 0) + throw std::runtime_error("Invalid parameters"); + + // 1 Create obstacles with different gaps + m_unitNX = static_cast(m_unitSizeX * m_resolution); + m_unitNY = static_cast(m_unitSizeY * m_resolution); + m_unitNZ = static_cast(m_unitSizeZ * m_resolution); + + std::vector unitSDF(m_nrows); + for (int i = 0; i < m_nrows; ++i) { + float gap = m_gapSpace * (i + 3); + generateUnitSDF(gap, unitSDF[i]); + } + + // 2 Create rows of obstacles + std::vector rows(m_nrows); + for (int i = 0; i < m_nrows; ++i) { + populateSDF(m_unitNX, m_unitNY, m_unitSizeX, m_unitSizeY, unitSDF[i], m_ncolumns, 1, rows[i]); + } + + // 3 Collage rows + SDF finalSDF; + collageSDF(m_unitNX * m_ncolumns, m_unitNY, rows, finalSDF); + //collageSDF(m_unitNX * m_ncolumns, m_unitNY, m_unitSizeX * m_ncolumns, m_unitSizeY, rows, m_nrows, false, finalSDF); + + // 4 Apply redistancing for the result + float finalExtent[] = {m_unitSizeX * m_ncolumns, m_unitSizeY * m_nrows}; + int finalN[] = {m_unitNX * m_ncolumns, m_unitNY * m_nrows}; + const float dx = finalExtent[0] / (finalN[0] - 1); + const float dy = finalExtent[1] / (finalN[1] - 1); + Redistance redistancer(0.25f * min(dx, dy), dx, dy, finalN[0], finalN[1]); + redistancer.run(m_niterRedistance, &finalSDF[0]); + + if (m_outFileName2D.length() != 0) + writeDAT(m_outFileName2D.c_str(), finalSDF, finalExtent[0], finalExtent[1], 1.0f, finalN[0], finalN[1], 1); + + conver2Dto3D(finalN[0], finalN[1], finalExtent[0], finalExtent[1], finalSDF, m_unitNZ, m_unitSizeZ - 2.0f*m_zmargin, m_zmargin, m_outFileName3D); +} + +int main(int argc, char ** argv) +{ + ArgumentParser argp(vector(argv, argv + argc)); + + int nColumns = argp("-nColumns").asInt(1); + int nRows = argp("-nRows").asInt(1); + float zMargin = static_cast(argp("-zMargin").asDouble(5.0)); + float resolution = static_cast(argp("-zResolution").asDouble(1.0)); + + std::string outFileName = argp("-out").asString("3d"); + + FunnelsBuilder builder; + try { + builder.setNColumns(nColumns) + .setNRows(nRows) + .setResolution(1.0f) + .setZWallWidth(zMargin) + .setFileNameFor3D(outFileName) + .setResolution(resolution) + .build(); + } catch(const std::exception& ex) { + std::cout << "ERROR: " << ex.what() << std::endl; + return 1; + } + return 0; +} diff --git a/device-gen/pipe/Makefile b/device-gen/pipe/Makefile new file mode 100644 index 000000000..2d07290c1 --- /dev/null +++ b/device-gen/pipe/Makefile @@ -0,0 +1,7 @@ +sdf-unit: main.cpp + g++-4.9 main.cpp -O0 -g3 -o sdf-unit + +clean: + rm sdf-unit + +.PHONY = clean diff --git a/device-gen/pipe/main.cpp b/device-gen/pipe/main.cpp new file mode 100644 index 000000000..1299aca7f --- /dev/null +++ b/device-gen/pipe/main.cpp @@ -0,0 +1,61 @@ +/* + * main.cpp + * Part of CTC/device-gen/sdf-unit-par/ + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include "../common/common.h" +using namespace std; + +#define REAL float + +REAL distToSide(REAL x, REAL y, REAL radius) { + return sqrt(x*x + y*y) - radius; +} + +int main(int argc, char ** argv) +{ + if (argc != 4) + { + printf("usage: ./sdf-cylinder \n"); + return 1; + } + + const int N = atoi(argv[1]); + const REAL radius = atof(argv[2]);; + const REAL extent = 2.0f*radius + 4.0f; + + std:cout << "Will generate SDF with extent " << extent << " " << extent + << ". Grid size " << N << " x " << N << std::endl; + + const REAL xlb = -extent/2.0f; + const REAL ylb = -extent/2.0f; + + vector sdf(N * N, 0.0f); + const REAL dx = extent / (N-1); + const REAL dy = extent / (N-1); + + for(int iy = 0; iy < N; ++iy) + for(int ix = 0; ix < N; ++ix) + { + const REAL x = xlb + ix * dx; + const REAL y = ylb + iy * dy; + + sdf[ix + N * iy] = distToSide(x, y, radius); + } + + writeDAT(argv[3], sdf, extent, extent, REAL(1.0), N, N, 1); + + return 0; +} diff --git a/device-gen/plates/Makefile b/device-gen/plates/Makefile new file mode 100644 index 000000000..4e67bbfb1 --- /dev/null +++ b/device-gen/plates/Makefile @@ -0,0 +1,9 @@ +CXXFLAGS += -O3 -DNDEBUG + +plates: plates.cpp + $(CXX) $(CXXFLAGS) $< -o plates + +clean: + rm -f plates + +.PHONY = clean diff --git a/device-gen/plates/plates.cpp b/device-gen/plates/plates.cpp new file mode 100644 index 000000000..6e3a4546f --- /dev/null +++ b/device-gen/plates/plates.cpp @@ -0,0 +1,49 @@ +#include +#include +#include +#include + +int main(const int argc, const char * argv[]) +{ + if (argc != 5) + { + printf("usage: ./plates \n"); + return 1; + } + + const int NX = atoi(argv[1]); + const int NY = atoi(argv[2]); + const int NZ = atoi(argv[3]); + const int zmargin = atoi(argv[4]); + + float * data = new float[NX * NY * NZ]; + + for(int iz = 0; iz < NZ; ++iz) + { + const float zval = fabs(iz + 0.5 - NZ * 0.5) - (NZ / 2 - zmargin); + + for(int iy = 0; iy < NY; ++iy) + for(int ix = 0; ix < NX; ++ix) + data[ix + NX * (iy + NY * iz)] = zval; + } + + FILE * f = fopen("sdf.dat", "w"); + assert(f != 0); + fprintf(f, "%f %f %f\n", (float)NX, (float)NY, (float)NZ); + fprintf(f, "%d %d %d\n", NY, NX, NZ); + fwrite(data, sizeof(float), NX * NY * NZ, f); + fclose(f); + +#ifndef NDEBUG + { + FILE * f = fopen("sdf.raw", "w"); + assert(f != 0); + fwrite(data, sizeof(float), NX * NY * NZ, f); + fclose(f); + } +#endif + + delete [] data; + + return 0; +} diff --git a/device-gen/post-process/h52ply.py b/device-gen/post-process/h52ply.py new file mode 100755 index 000000000..c2a2dd31e --- /dev/null +++ b/device-gen/post-process/h52ply.py @@ -0,0 +1,37 @@ +#!/usr/bin/env /Applications/paraview.app/Contents/bin/pvpython + +''' + * Part of CTC/device-gen/post-processing/h52ply.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +import argparse +import os +from paraview.simple import * + +print("h52ply started") +parser = argparse.ArgumentParser(description='Transforms h5 which has xmf to ply using Paraview python lib.', + usage= './h52ply.py -i -o ') +parser.add_argument('-i','--inputFile', help='XMF', required=True) +parser.add_argument('-o','--outputFile', help='PLY', required=True) +args = vars(parser.parse_args()) + +# paraview wants to have absolute path +fullPath = os.path.dirname(os.path.abspath(args['inputFile'])) + '/' +print fullPath + +a13x59xmf = XDMFReader(FileNames=[fullPath + args['inputFile']]) +#a13x59xmf.GridStatus = ['Grid_26'] + +contour1 = Contour(Input=a13x59xmf) +contour1.Isosurfaces = [0.0] + +# save data +SaveData(fullPath + args['outputFile'], proxy=contour1) + diff --git a/device-gen/post-process/plyScale.py b/device-gen/post-process/plyScale.py new file mode 100755 index 000000000..82a06ea04 --- /dev/null +++ b/device-gen/post-process/plyScale.py @@ -0,0 +1,118 @@ +#!/usr/bin/env python + +''' + * Part of CTC/device-gen/post-processing/plyScale.py + * + * Created and authored by Kirill Lykov on 2015-08-28. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. +''' + +from plyfile import PlyData, PlyElement +import argparse +import copy +import numpy + +def computeExtent(vertices): + extentMax = [-10e6] * 3 + extentMin = [10e6] * 3 + for i in range(0, len(vertices)): + vi = vertices[i] + for dim in range(0, 3): + extentMax[dim] = max(extentMax[dim], vi[dim]) + extentMin[dim] = min(extentMin[dim], vi[dim]) + + origOrigin = [(extentMax[i] + extentMin[i])/2.0 for i in range(0, 3)] + origExtent = [extentMax[i] - extentMin[i] for i in range(0, 3)] + return (origOrigin, origExtent) + +parser = argparse.ArgumentParser(description='Modifies ply file to be used for rendering.\n Example: ./plyScale.py -f input.ply -o out.ply -r 210 -cutX 10.0') +parser.add_argument('-f','--inputFile', help='Input file name', required=True) +parser.add_argument('-o','--outputFile', help='Output file name', required=True) +parser.add_argument('--lx', help='Desired size of bounding box (X axis)', required=False, default="0") +parser.add_argument('--ly', help='Desired size of bounding box (Y axis)', required=False, default="0") +parser.add_argument('--lz', help='Desired size of bounding box (Z axis)', required=False, default="0") +parser.add_argument('-r','--order', help='Reorder axis. By default 012, to swap x and z use 210', required=False, default="012") +helpStringForCut = 'Remove all the faces which are above specified value for %s. The origin is in the center of mass. If axis reordering was applied, axis are in new coordinates.' +parser.add_argument('--cutX', help=helpStringForCut%('X'), required=False, default="none") +parser.add_argument('--cutY', help=helpStringForCut%('Y'), required=False, default="none") +parser.add_argument('--cutZ', help=helpStringForCut%('Z'), required=False, default="none") +args = vars(parser.parse_args()) + +desiredBox = [float(args['lx']), float(args['ly']), float(args['lz'])] + +plydata = PlyData.read(args['inputFile']) +vertices = plydata['vertex'].data + +# swap coords +order = args['order'] +if (order != "012"): + idx = [int(order[i]) for i in range(0, len(order))] + assert(len(idx) == 3) + print "Swapping axis!" + for i in range(0, len(vertices)): + v = copy.deepcopy(vertices[i]) + for dim in range(0, 3): + vertices[i][dim] = v[ idx[dim] ] + + +# Current box +(origOrigin, origExtent) = computeExtent(vertices) +for dim in range(0, 3): + if desiredBox[dim] == 0: + desiredBox[dim] = origExtent[dim] +print ("Extent is (%f, %f, %f). Center is (%f, %f, %f)."%(origExtent[0], origExtent[1], origExtent[2], + origOrigin[0], origOrigin[1], origOrigin[2])) +for i in range(0, len(vertices)): + for dim in range(0, 3): + vertices[i][dim] -= origOrigin[dim] + vertices[i][dim] *= desiredBox[dim]/origExtent[dim] + +if (args['cutX'] != "none" or args['cutY'] != "none" or args['cutZ'] != "none"): + print "Cut it!" + cut = [1e6]*3 + if args['cutX'] != "none": + cut[0] = float(args['cutX']) + if args['cutY'] != "none": + cut[1] = float(args['cutY']) + if args['cutZ'] != "none": + cut[2] = float(args['cutZ']) + + toDelete = list() + newInx = [None]*len(vertices) + j = 0 + for i in range(0, len(vertices)): + if (vertices[i][0] > cut[0] or vertices[i][1] > cut[1] or vertices[i][2] > cut[2]): + toDelete.append(i) + else: + newInx[i] = j + j += 1 + + plydata['vertex'].data = numpy.delete(plydata['vertex'].data, toDelete, axis=0) + # remove polygons containing these vertices + setVertToDel = set(toDelete) + faces = plydata['face'].data + facesToDelete = list() + for i in range(0, len(faces)): + curr = set(faces[i][0]) + common = curr & setVertToDel + if (common): + facesToDelete.append(i) + plydata['face'].data = numpy.delete(plydata['face'].data, facesToDelete, axis=0) + + #update vertices in polygons + faces = plydata['face'].data + for i in range(0, len(faces)): + f = faces[i][0] + for i in range(0, len(f)): + f[i] = newInx[ f[i] ] + +(finalOrigin, finalExtent) = computeExtent(vertices) +print ("Extent is (%f, %f, %f). Center is (%f, %f, %f)."%(finalExtent[0], finalExtent[1], finalExtent[2], + finalOrigin[0], finalOrigin[1], finalOrigin[2])) + +plydata.write(args['outputFile']) + diff --git a/device-gen/scripts/README b/device-gen/scripts/README deleted file mode 100644 index 6d30e57db..000000000 --- a/device-gen/scripts/README +++ /dev/null @@ -1,5 +0,0 @@ -Generates parabolic funnels. -./makeall.sh -./run.sh - -The result is in the file sdf.dat diff --git a/device-gen/scripts/cleanall.sh b/device-gen/scripts/cleanall.sh deleted file mode 100755 index b9e3a1b26..000000000 --- a/device-gen/scripts/cleanall.sh +++ /dev/null @@ -1,7 +0,0 @@ -#! /bin/bash -cd ../ -for d in */ ; do - pushd $d - make clean - popd -done diff --git a/device-gen/scripts/files.txt b/device-gen/scripts/files.txt deleted file mode 100644 index 217f12814..000000000 --- a/device-gen/scripts/files.txt +++ /dev/null @@ -1,12 +0,0 @@ -r4.dat -r5.dat -r6.dat -r7.dat -r8.dat -r9.dat -r10.dat -r11.dat -r12.dat -r13.dat -r14.dat -r15.dat \ No newline at end of file diff --git a/device-gen/scripts/makeall.sh b/device-gen/scripts/makeall.sh deleted file mode 100755 index 8c8723c2a..000000000 --- a/device-gen/scripts/makeall.sh +++ /dev/null @@ -1,7 +0,0 @@ -#! /bin/bash -cd ../ -for d in */ ; do - pushd $d - make - popd -done diff --git a/device-gen/scripts/run.sh b/device-gen/scripts/run.sh deleted file mode 100755 index ea81e33be..000000000 --- a/device-gen/scripts/run.sh +++ /dev/null @@ -1,19 +0,0 @@ -#! /usr/local/bin/bash - -unitXRes=32 -unitYRes=128 -unitZRes=64 - -nColumns=2 -# nRows is defined by files.txt - -for i in `seq 3 15`; do - ../sdf-unit-par/sdf-unit $unitXRes $unitYRes 24 96 $i gap$i.dat -done - -for i in `seq 3 15`; do - ../sdf-collage/sdf-collage gap$i.dat $nColumns 1 r$i.dat -done - -../sdf-collage/sdf-collage files.txt 1 1 collage.dat -../2Dto3D/2Dto3D collage.dat 40.0 4.0 $unitZRes sdf.dat diff --git a/device-gen/sdf-collage/Makefile b/device-gen/sdf-collage/Makefile deleted file mode 100644 index 306311353..000000000 --- a/device-gen/sdf-collage/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -sdf-collage: main.cpp - g++ main.cpp -O0 -g3 -fopenmp -o sdf-collage - -clean: - rm sdf-collage - -.PHONY = clean diff --git a/device-gen/sdf-collage/main.cpp b/device-gen/sdf-collage/main.cpp deleted file mode 100644 index 2d2ac7dbc..000000000 --- a/device-gen/sdf-collage/main.cpp +++ /dev/null @@ -1,227 +0,0 @@ -/* - * main.cpp - * Part of uDeviceX/device-gen/sdf-collage/ - * - * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include -#include -#include -#include -#include -#include -#include -#include - -using namespace std; - -#define _ACCESS(f, x, y) f[(x) + xsize * (y)] - -namespace Redistancing -{ - int xsize; - float * phi0, * phi; - float dt, invdx, invdy; - - template - inline bool anycrossing_dir(int ix, int iy, const float sgn0) - { - const int dx = d == 0, dy = d == 1, dz = d == 2; - - const float fm1 = _ACCESS(phi0, ix - dx, iy - dy); - const float fp1 = _ACCESS(phi0, ix + dx, iy + dy); - - return (fm1 * sgn0 < 0 || fp1 * sgn0 < 0); - } - - inline bool anycrossing(int ix, int iy, const float sgn0) - { - return - anycrossing_dir<0>(ix, iy, sgn0) || - anycrossing_dir<1>(ix, iy, sgn0) ; - } - - float sussman_scheme(int ix, int iy, float sgn0) - { - const float phicenter = _ACCESS(phi, ix, iy); - - const float dphidxm = phicenter - _ACCESS(phi, ix - 1, iy); - const float dphidxp = _ACCESS(phi, ix + 1, iy) - phicenter; - const float dphidym = phicenter - _ACCESS(phi, ix, iy - 1); - const float dphidyp = _ACCESS(phi, ix, iy + 1) - phicenter; - - if (sgn0 == 1) - { - const float xgrad0 = max( max((float)0, dphidxm), -min((float)0, dphidxp)) * invdx; - const float ygrad0 = max( max((float)0, dphidym), -min((float)0, dphidyp)) * invdy; - - const float G0 = sqrtf(xgrad0 * xgrad0 + ygrad0 * ygrad0) - 1; - - return phicenter - dt * sgn0 * G0; - } - else - { - const float xgrad1 = max( -min((float)0, dphidxm), max((float)0, dphidxp)) * invdx; - const float ygrad1 = max( -min((float)0, dphidym), max((float)0, dphidyp)) * invdy; - - const float G1 = sqrtf(xgrad1 * xgrad1 + ygrad1 * ygrad1) - 1; - - return phicenter - dt * sgn0 * G1; - } - } - - void redistancing(const int iterations, const float dt, const float dx, const float dy, - const int xsize, const int ysize, - float * field) - { - Redistancing::xsize = xsize; - Redistancing::dt = dt; - Redistancing::invdx = 1. / dx; - Redistancing::invdy = 1. / dy; - - Redistancing::phi0 = new float[xsize * ysize]; - memcpy(phi0, field, sizeof(float) * xsize * ysize); - Redistancing::phi = field; - - float * tmp = new float[xsize * ysize]; - for(int t = 0; t < iterations; ++t) - { - if (t % 30 == 0) - printf("t: %d\n", t); - -#pragma omp parallel for - for(int iy = 0; iy < ysize; ++iy) - for(int ix = 0; ix < xsize; ++ix) - { - const float myval0 = _ACCESS(phi0, ix, iy); - const float sgn0 = myval0 > 0 ? 1 : (myval0 < 0 ? -1 : 0); - - if (anycrossing(ix, iy, sgn0) || ix == 0 || ix == xsize - 1 || iy == 0 || iy == ysize - 1) - tmp[ix + xsize * iy] = myval0; - else - tmp[ix + xsize * iy] = sussman_scheme(ix, iy, sgn0); - } - - memcpy(field, tmp, sizeof(float) * xsize * ysize); - } - - delete [] tmp; - delete [] phi0; - phi0 = NULL; - } -} - -void mergeSDF(int NX, int NY, vector< vector >& cookie, vector& cake) -{ - cake.resize(cookie.size() * cookie[0].size()); - printf("SIZE: %d\n", cake.size()); - const int stride = NX; - for(int iy = 0; iy < cookie.size() * NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const int dst = ix + stride * iy; - const int iobst = iy / NY; - cake[dst] = cookie[iobst][ix + NX * (iy - iobst*NY)]; - } -} - -int main(int argc, char ** argv) -{ - if (argc != 5) - { - printf("usage: ./sdf-collage \n"); - return -1; - } - - const int xtimes = atoi(argv[2]); - int ytimes = atoi(argv[3]); - - float xextent, yextent, zextent; - int NX, NY,NZ; - vector< vector > cookie; - vector cake; - if (string(argv[1]) != "files.txt") - { - cookie.resize(1); - // for one file - FILE * f = fopen(argv[1], "r"); - assert(f != 0); - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); - printf("Extent: [%f, %f, %f]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextent, NX, NY,NZ); - assert(NZ == 1); - cookie[0].resize(NX * NY * NZ, 0.0f); - fread(&cookie[0][0], sizeof(float), NX * NY * NZ, f); - fclose(f); - - printf("Populate %d * %d times\n", xtimes, ytimes); - const int stride = xtimes * NX; - cake.resize(xtimes * ytimes * NX * NY); - - for(int ty = 0; ty < ytimes; ++ty) - for(int tx = 0; tx < xtimes; ++tx) { - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const int gx = ix + NX * tx; - const int gy = iy + NY * ty; - const int dst = gx + stride * gy; - - assert(dst < cake.size()); - assert(ix + NX * iy < cookie[0].size()); - cake[dst] = cookie[0][ix + NX * iy]; - } - } - } else { - vector files; - - FILE* fs = fopen(argv[1], "r"); - assert(fs != 0); - string buf(127, ' '); - while(fscanf(fs, "%s\n", &buf[0]) == 1) { - files.push_back(buf); - } - fclose(fs); - - ytimes = files.size(); - cookie.resize(ytimes); - assert(xtimes == 1); - - for (int i = files.size() - 1; i >= 0; --i) - { - printf("Reading file %s ...\n", files[i].c_str()); - FILE * f = fopen(files[i].c_str(), "r"); - assert(f != 0); - fscanf(f, "%f %f %f\n", &xextent, &yextent, &zextent); - fscanf(f, "%d %d %d\n", &NX, &NY, &NZ); - printf("Extent: [%g, %g, %g]. Grid size: [%d, %d, %d]\n", xextent, yextent, zextent, NX, NY,NZ); - assert(NZ == 1); - cookie[i].resize(NX * NY * NZ, 0.0f); - fread(&cookie[i][0], sizeof(float), NX * NY * NZ, f); - fclose(f); - } - - mergeSDF(NX, NY, cookie, cake); - } - - const float dx = xextent / NX; - const float dy = yextent / NY; - Redistancing::redistancing(240, 0.25 * min(dx, dy), dx, dy, xtimes * NX, ytimes * NY, &cake[0]); - - { - FILE * f = fopen(argv[4], "w"); - assert(f != 0); - fprintf(f, "%f %f %f\n", xtimes * xextent, ytimes * yextent, 1.0f); - fprintf(f, "%d %d %d\n", xtimes * NX, ytimes * NY, 1); - fwrite(&cake[0], sizeof(float), cake.size(), f); - fclose(f); - } - - return 0; -} diff --git a/device-gen/sdf-unit-par/Makefile b/device-gen/sdf-unit-par/Makefile deleted file mode 100644 index 8d0bef89c..000000000 --- a/device-gen/sdf-unit-par/Makefile +++ /dev/null @@ -1,7 +0,0 @@ -sdf-unit: main.cpp - g++ main.cpp -o sdf-unit - -clean: - rm sdf-unit - -.PHONY = clean \ No newline at end of file diff --git a/device-gen/sdf-unit-par/main.cpp b/device-gen/sdf-unit-par/main.cpp deleted file mode 100644 index 8c5ceb639..000000000 --- a/device-gen/sdf-unit-par/main.cpp +++ /dev/null @@ -1,132 +0,0 @@ -/* - * main.cpp - * Part of uDeviceX/device-gen/sdf-unit-par/ - * - * Created and authored by Diego Rossinelli and Kirill Lykov on 2015-03-20. - * Copyright 2015. All rights reserved. - * - * Users are NOT authorized - * to employ the present software for their own publications - * before getting a written permission from the author of this file. - */ - -#include -#include -#include -#include -#include -using namespace std; - -struct Parabola -{ - float x0; - float ymax; // length of obstacle, for cutting the pick - float y0; - - Parabola(float gap, float xextent) : ymax(48.0) - { - x0 = xextent/2.0f - gap/2.0f; - if (gap > 9.0) - y0 = ymax; - else - y0 = ymax + 6.5; - } - - void line1(vector& vx, vector& vy) { - int N = 500; - float dx = 2.0f * fabs(x0) / (N - 1); - for (int i = 0; i < N; ++i) { - float x = i * dx - x0; - float y = 0.0f; - vx.push_back(x); - vy.push_back(y); - } - } - - void line2(vector& vx, vector& vy) { - int N = 1500; - float dx = 2.0f * fabs(x0) / (N - 1); - float alpha = -y0 / (x0 * x0); - for (int i = 0; i < N; ++i) { - float x = i * dx - x0; - float y = min(ymax, alpha * x*x + y0); - vx.push_back(x); - vy.push_back(y); - } - } -}; - -int main(int argc, char ** argv) -{ - if (argc != 7) - { - printf("usage: ./sdf-unit \n"); - return 1; - } - - const int NX = atoi(argv[1]); - const int NY = atoi(argv[2]); - const float xextent = atof(argv[3]); - const float yextent = atof(argv[4]); - const float gap = atof(argv[5]); - - vector xs, ys; - Parabola par(gap, xextent); - par.line1(xs, ys); - par.line2(xs, ys); - - const float xlb = -xextent/2.0f; - const float ylb = -(yextent - par.ymax)/2.0f; - printf("starting brute force sdf with %d x %d starting from %f %f to %f %f\n", - NX, NY, xlb, ylb, xlb + xextent, ylb + yextent); - - float * sdf = new float[NX * NY]; - const float dx = xextent / NX; - const float dy = yextent / NY; - const int nsamples = xs.size(); - - for(int iy = 0; iy < NY; ++iy) - for(int ix = 0; ix < NX; ++ix) - { - const float x = xlb + ix * dx; - const float y = ylb + iy * dy; - - float distance2 = 1e6; - int iclosest = 0; - for(int i = 0; i < nsamples ; ++i) - { - const float xd = xs[i] - x; - const float yd = ys[i] - y; - const float candidate = xd * xd + yd * yd; - - if (candidate < distance2) - { - iclosest = i; - distance2 = candidate; - } - } - - float s = -1; - - { - const float alpha = -par.y0 / (par.x0 * par.x0); - const float ycurve = min(par.ymax, alpha * x*x + par.y0); - - if (x >= -par.x0 && x <= par.x0 && y >= 0 && y <= ycurve) - s = +1; - } - - - sdf[ix + NX * iy] = s * sqrt(distance2); - } - - FILE * f = fopen(argv[6], "w"); - fprintf(f, "%f %f %f\n", xextent, yextent, 1.0f); - fprintf(f, "%d %d %d\n", NX, NY, 1); - fwrite(sdf, sizeof(float), NX * NY, f); - fclose(f); - - delete [] sdf; - - return 0; -} diff --git a/mpi-dpd/Makefile b/mpi-dpd/Makefile index 178d49f5d..02cabaa9b 100644 --- a/mpi-dpd/Makefile +++ b/mpi-dpd/Makefile @@ -16,7 +16,8 @@ OBJS = dpd.o wall.o fsi.o contact.o \ solvent-exchange.o solute-exchange.o \ common.o containers.o io.o \ scan.o minmax.o redistancing.o \ - simulation.o main.o + simulation.o main.o dumper.o \ + velcontroller.o velsampler.o LIBS = -lcuda-dpd -lcuda-rbc -lcuda-ctc -lcudart -ldl -lz @@ -50,16 +51,6 @@ slevel ?= 0 flops ?= 0 datadump ?= 1 -ifeq "$(datadump)" "0" -NVCCFLAGS += -D_NO_DUMPS_ -endif -ifeq "$(datadump)" "1" -NVCCFLAGS += -D_SYNC_DUMPS_ -endif -ifeq "$(datadump)" "2" -NVCCFLAGS += -D_ASYNC_DUMPS_ -endif - NVCCFLAGS += -DVISCOSITY_S_LEVEL=$(slevel) inquire: $(bash [ `cat slevel.txt` == "$(slevel)" ] || { echo "cleanall" ; } ) diff --git a/mpi-dpd/README.md b/mpi-dpd/README.md new file mode 100644 index 000000000..146275372 --- /dev/null +++ b/mpi-dpd/README.md @@ -0,0 +1,51 @@ +The main folder: mpi-dpd +======= + +This is the "main" folder containing the object files orchestrating the various kernels. +The generated object files are the following ones: + +* common.o: global simulation parameters, common datastructures like device arrays, cell lists etc. +* contact.o: computation of the contact/lubrication force across "touching" solute particles +* containers.o: particle arrays, collections encapsulating the data of RBCs and CTCs +* dpd.o: code coordinating the computation of cuda-dpd +* fsi.o: computation of the "flow-structure interaction" force between solvent and solute particles +* io.o: data dumps in XYZ, PLY (for cells), H5Part, HDF5 structured grids +* main.o: home sweet home +* minmax.o: computation of the extent of an array of RBCs or CTCs +* redistancing.o: computation of the distance transform for the implicit description of the wall geometry +* redistribute-particles.o: redistribution of the solvent across the MPI ranks once particles have moved +* redistribute-rbcs.o: redistribution of RBCs across MPI ranks once they have moved +* scan.o: computation of the prefix sum for the solvent cell lists count +* simulation.o: simulation "driver" coordinating the other object files, except for main.cu +* solute-exchange.o: exchange the "halo" solute particles across the MPI ranks close by to compute FSI and contact forces. +* solvent-exchange.o: exchange the "halo" solvent particles across the MPI ranks to compute the DPD interactions +* wall.o: computation of the particles interacting with the no-slip boundary conditions of the wall + +Compiling uDeviceX +------------- +The makefile will check for a .cache.Makefile, that can be optionally put in this folder. +For example my .cache.Makefile on Piz Daint is + +`h5part = 0 +NVCC = nvcc -I$(CRAY_MPICH2_DIR)/include -L$(CRAY_MPICH2_DIR)/lib -I/scratch/daint/diegor/h5part/include -I$(HDF5_DIR)/include -I/users/diegor/vtk/install/include/vtk-6.2 -I/users/diegor/h5part/include/ +CXX = CC $(CRAY_CUDATOOLKIT_POST_LINK_OPTS) $(CRAY_CUDATOOLKIT_INCLUDE_OPTS) -L/users/diegor/h5part/lib -L$(HDF5_DIR)/lib -L/users/diegor/vtk/install/lib` + +To clean uDeviceX entirely (mpi-dpd, cuda-dpd, cuda-rbc, cuda-ctc): +`make cleanall` + +To just cleanup the mpi-dpd folder: +`make clean` + +To compile: +`make -j` + +Running uDeviceX +----------- +Running uDeviceX consists of these steps: + +1. Generation of the geometry file (optional), the file should always be named `sdf.dat` (as Signed Distance Function). +See the folder `device-gen`. +2. Generation of the initial positioning of the RBCs and CTCs (optional). The IC files should be called +`rbcs-ic.txt and ctcs-ic.txt` See the folder `cell-placement`. +3. Execution of uDeviceX, for example `mpirun ./test 4 4 2 -walls -couette=1 -tend=5e4 -steps_per_dump=1000 -rbcs -contactforces` +4. Post processing of the simulation data (optional), for example `ls ./stress/* -rt1 | tail -n 50 | mpirun -n 32 -N 1 ../postprocessing/stress/stress -origin=0,0,5 -extent=192,192,85 -project=1,1,0 > stress-profile.txt` diff --git a/mpi-dpd/common-kernels.h b/mpi-dpd/common-kernels.h index a2b8a1572..59c165675 100644 --- a/mpi-dpd/common-kernels.h +++ b/mpi-dpd/common-kernels.h @@ -170,8 +170,7 @@ void write_AOS3f(float * const data, const int nparticles, float& s0, float& s1, data[laneid + 64] = s2; } -template -__global__ void subindex_local(const int nparticles, const float2 * particles, int * const partials, +__global__ static void subindex_local(const int nparticles, const float2 * particles, int * const partials, uchar4 * const subindices) { assert(blockDim.x == 128 && blockDim.x * gridDim.x >= nparticles); @@ -192,29 +191,18 @@ __global__ void subindex_local(const int nparticles, const float2 * particles, read_AOS6f(particles + 3 * base, nsrc, data0, data1, data2); - const bool inside = project || + const bool inside = (data0.x >= -XSIZE_SUBDOMAIN / 2 && data0.x < XSIZE_SUBDOMAIN / 2 && data0.y >= -YSIZE_SUBDOMAIN / 2 && data0.y < YSIZE_SUBDOMAIN / 2 && data1.x >= -ZSIZE_SUBDOMAIN / 2 && data1.x < ZSIZE_SUBDOMAIN / 2 ); if (lane < nsrc && inside) { - if (project) - { - const int xcid = min(XSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2))); - const int ycid = min(YSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2))); - const int zcid = min(ZSIZE_SUBDOMAIN - 1, max(0, (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2))); + const int xcid = (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2); + const int ycid = (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2); + const int zcid = (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2); - cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); - } - else - { - const int xcid = (int)floor((double)data0.x + XSIZE_SUBDOMAIN / 2); - const int ycid = (int)floor((double)data0.y + YSIZE_SUBDOMAIN / 2); - const int zcid = (int)floor((double)data1.x + ZSIZE_SUBDOMAIN / 2); - - cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); - } + cid = xcid + XSIZE_SUBDOMAIN * (ycid + YSIZE_SUBDOMAIN * zcid); } } diff --git a/mpi-dpd/common.h b/mpi-dpd/common.h index 481592ee6..ef95f74c0 100644 --- a/mpi-dpd/common.h +++ b/mpi-dpd/common.h @@ -20,26 +20,26 @@ enum { - XSIZE_SUBDOMAIN = 48, - YSIZE_SUBDOMAIN = 48, - ZSIZE_SUBDOMAIN = 48, - XMARGIN_WALL = 6, - YMARGIN_WALL = 6, - ZMARGIN_WALL = 6, + XSIZE_SUBDOMAIN = 64, + YSIZE_SUBDOMAIN = 64, + ZSIZE_SUBDOMAIN = 64, + XMARGIN_WALL = 10, + YMARGIN_WALL = 10, + ZMARGIN_WALL = 10, }; const int numberdensity = 4; -const float dt = 0.001; -const float kBT = 0.0945; -const float gammadpd = 45; +const float dt = 0.0025; +const float kBT = 1.0; +const float gammadpd = 20; const float sigma = sqrt(2 * gammadpd * kBT); const float sigmaf = sigma / sqrt(dt); -const float aij = 25; -const float hydrostatic_a = 0.05; +const float aij = 50; +const float hydrostatic_a = 0.022; -extern float tend; -extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces; -extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; +extern float tend, couette; +extern bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, contactforces, stress; +extern int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop, nsubsteps; #include #include diff --git a/mpi-dpd/contact.cu b/mpi-dpd/contact.cu index 42059759f..d91d80e60 100644 --- a/mpi-dpd/contact.cu +++ b/mpi-dpd/contact.cu @@ -42,11 +42,6 @@ namespace KernelsContact texture texCellsStart, texCellEntries; - __global__ void bulk_3tpp(const float2 * const particles, const int np, const int ncellentries, const int nsolutes, - float * const acc, const float seed, const int mysoluteid); - - __global__ void halo(const int nparticles_padded, const int ncellentries, const int nsolutes, const float seed); - void setup() { texCellsStart.channelDesc = cudaCreateChannelDesc(); @@ -58,10 +53,9 @@ namespace KernelsContact texCellEntries.filterMode = cudaFilterModePoint; texCellEntries.mipmapFilterMode = cudaFilterModePoint; texCellEntries.normalized = 0; - - CUDA_CHECK(cudaFuncSetCacheConfig(bulk_3tpp, cudaFuncCachePreferL1)); - CUDA_CHECK(cudaFuncSetCacheConfig(halo, cudaFuncCachePreferL1)); } + + __global__ void bulk_3tpp(const int nsolutes, const float seed); } ComputeContact::ComputeContact(MPI_Comm comm): @@ -77,11 +71,13 @@ cellsstart(KernelsContact::NCELLS + 16), cellscount(KernelsContact::NCELLS + 16) CUDA_CHECK(cudaMemcpyToSymbol(KernelsContact::params, ¶ms, sizeof(params))); CUDA_CHECK(cudaPeekAtLastError()); + + CUDA_CHECK(cudaFuncSetCacheConfig(KernelsContact::bulk_3tpp , cudaFuncCachePreferL1)); } namespace KernelsContact { - __global__ void populate(const uchar4 * const subindices, const int * const cellstart, + __global__ void populate(const uchar4 * const subindices, const int * const cellstart, const int nparticles, const int soluteid, const int ntotalparticles, CellEntry * const entrycells) { @@ -141,7 +137,7 @@ namespace KernelsContact const int ncells = XSIZE_SUBDOMAIN * YSIZE_SUBDOMAIN * ZSIZE_SUBDOMAIN; - CUDA_CHECK(cudaBindTexture(&textureoffset, &texCellsStart, cellsstart, &texCellsStart.channelDesc, sizeof(int) * ncells)); + CUDA_CHECK(cudaBindTexture(&textureoffset, &texCellsStart, cellsstart, &texCellsStart.channelDesc, sizeof(int) * (1 + ncells))); assert(textureoffset == 0); const int n = wsolutes.size(); @@ -161,84 +157,43 @@ namespace KernelsContact CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutes, ps, sizeof(float2 *) * n, 0, cudaMemcpyHostToDevice, stream)); CUDA_CHECK(cudaMemcpyToSymbolAsync(csolutesacc, as, sizeof(float *) * n, 0, cudaMemcpyHostToDevice, stream)); } -} - -void ComputeContact::build_cells(std::vector wsolutes, cudaStream_t stream) -{ - this->nsolutes = wsolutes.size(); - - int ntotal = 0; - - for(int i = 0; i < wsolutes.size(); ++i) - ntotal += wsolutes[i].n; - - subindices.resize(ntotal); - cellsentries.resize(ntotal); - - CUDA_CHECK(cudaMemsetAsync(cellscount.data, 0, sizeof(int) * cellscount.size, stream)); - -#ifndef NDEBUG - CUDA_CHECK(cudaMemsetAsync(cellsentries.data, 0xff, sizeof(int) * cellsentries.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(subindices.data, 0xff, sizeof(int) * subindices.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(compressed_cellscount.data, 0xff, sizeof(unsigned char) * compressed_cellscount.capacity, stream)); - CUDA_CHECK(cudaMemsetAsync(cellsstart.data, 0xff, sizeof(int) * cellsstart.capacity, stream)); -#endif - - CUDA_CHECK(cudaPeekAtLastError()); - int ctr = 0; - for(int i = 0; i < wsolutes.size(); ++i) + __global__ __launch_bounds__(128, 10) void bulk_3tpp(const int nsolutes, const float seed) { - const ParticlesWrap it = wsolutes[i]; + const int np = tex1Dfetch(texCellsStart, XCELLS * YCELLS * ZCELLS); - if (it.n) - subindex_local<<< (it.n + 127) / 128, 128, 0, stream >>> - (it.n, (float2 *)it.p, cellscount.data, subindices.data + ctr); - - ctr += it.n; - } - - compress_counts<<< (compressed_cellscount.size + 127) / 128, 128, 0, stream >>> - (compressed_cellscount.size, (int4 *)cellscount.data, (uchar4 *)compressed_cellscount.data); + assert(blockDim.x * gridDim.x >= np * 3); - scan(compressed_cellscount.data, compressed_cellscount.size, stream, (uint *)cellsstart.data); + const int gid = threadIdx.x + blockDim.x * blockIdx.x; + const int myslot = gid / 3; + const int zplane = gid % 3; - ctr = 0; - for(int i = 0; i < wsolutes.size(); ++i) - { - const ParticlesWrap it = wsolutes[i]; + if (myslot >= np) + return; - if (it.n) - KernelsContact::populate<<< (it.n + 127) / 128, 128, 0, stream >>> - (subindices.data + ctr, cellsstart.data, it.n, i, ntotal, (KernelsContact::CellEntry *)cellsentries.data); + float2 dst0, dst1, dst2; + int mysoluteid, actualpid; - ctr += it.n; - } + { + CellEntry ce; + ce.pid = tex1Dfetch(texCellEntries, myslot); - CUDA_CHECK(cudaPeekAtLastError()); + mysoluteid = ce.code.w; - KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); -} - -namespace KernelsContact -{ - __global__ __launch_bounds__(128, 10) - void bulk_3tpp(const float2 * const particles, - const int np, const int ncellentries, const int nsolutes, - float * const acc, const float seed, const int mysoluteid) - { - assert(blockDim.x * gridDim.x >= np * 3); + ce.code.w = 0; + actualpid = ce.pid; - const int gid = threadIdx.x + blockDim.x * blockIdx.x; - const int pid = gid / 3; - const int zplane = gid % 3; + assert(mysoluteid < nsolutes); + assert(actualpid >= 0 && actualpid < cnsolutes[mysoluteid]); - if (pid >= np) - return; + dst0 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 0); + dst1 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 1); + dst2 = _ACCESS(csolutes[mysoluteid] + 3 * actualpid + 2); - const float2 dst0 = _ACCESS(particles + 3 * pid + 0); - const float2 dst1 = _ACCESS(particles + 3 * pid + 1); - const float2 dst2 = _ACCESS(particles + 3 * pid + 2); + assert(dst0.x >= -XOFFSET && dst0.x < XOFFSET); + assert(dst0.y >= -YOFFSET && dst0.y < YOFFSET); + assert(dst1.x >= -ZOFFSET && dst1.x < ZOFFSET); + } int scan1, scan2, ncandidates, spidbase; int deltaspid1, deltaspid2; @@ -295,13 +250,15 @@ namespace KernelsContact float xforce = 0, yforce = 0, zforce = 0; -#pragma unroll 3 for(int i = 0; i < ncandidates; ++i) { const int m1 = (int)(i >= scan1); const int m2 = (int)(i >= scan2); const int slot = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - assert(slot >= 0 && slot < ncellentries); + assert(slot >= 0 && slot < np); + + if (slot >= myslot) + continue; CellEntry ce; ce.pid = tex1Dfetch(texCellEntries, slot); @@ -313,9 +270,6 @@ namespace KernelsContact const int spid = ce.pid; assert(spid >= 0 && spid < cnsolutes[soluteid]); - if (mysoluteid < soluteid || mysoluteid == soluteid && pid <= spid) - continue; - const int sentry = 3 * spid; const float2 stmp0 = _ACCESS(csolutes[soluteid] + sentry ); const float2 stmp1 = _ACCESS(csolutes[soluteid] + sentry + 1); @@ -339,7 +293,7 @@ namespace KernelsContact const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + const float lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); @@ -352,7 +306,7 @@ namespace KernelsContact yr * (dst2.x - stmp2.x) + zr * (dst2.y - stmp2.y); - const float myrandnr = Logistic::mean0var1(seed, pid, spid); + const float myrandnr = Logistic::mean0var1(seed, myslot, slot); const float strength = lj + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; @@ -368,93 +322,57 @@ namespace KernelsContact assert(!isnan(yinteraction)); assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); + assert(fabs(xinteraction) < 1e5); + assert(fabs(yinteraction) < 1e5); + assert(fabs(zinteraction) < 1e5); atomicAdd(csolutesacc[soluteid] + sentry , -xinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 1, -yinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 2, -zinteraction); } - atomicAdd(acc + 3 * pid + 0, xforce); - atomicAdd(acc + 3 * pid + 1, yforce); - atomicAdd(acc + 3 * pid + 2, zforce); + const float xacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 0, xforce); + const float yacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 1, yforce); + const float zacc = atomicAdd(csolutesacc[mysoluteid] + 3 * actualpid + 2, zforce); - for(int c = 0; c < 3; ++c) - assert(!isnan(acc[3 * pid + c])); + assert(!isnan(xacc)); + assert(!isnan(yacc)); + assert(!isnan(zacc)); } -} - -void ComputeContact::bulk(std::vector wsolutes, cudaStream_t stream) -{ - NVTX_RANGE("Contact/bulk", NVTX_C6); - if (wsolutes.size() == 0) - return; - - for(int i = 0; i < wsolutes.size(); ++i) + __global__ void halo(const float2 * halo, const int nhalo, const int nsolutes, const float seed, float * const acc) { - ParticlesWrap it = wsolutes[i]; - - if (it.n) - KernelsContact::bulk_3tpp<<< (3 * it.n + 127) / 128, 128, 0, stream >>> - ((float2 *)it.p, it.n, cellsentries.size, wsolutes.size(), (float *)it.a, local_trunk.get_float(), i); - - CUDA_CHECK(cudaPeekAtLastError()); - } -} + const int nbulk = tex1Dfetch(texCellsStart, XCELLS * YCELLS * ZCELLS); -namespace KernelsContact -{ - __constant__ int packstarts_padded[27], packcount[26]; - __constant__ Particle * packstates[26]; - __constant__ Acceleration * packresults[26]; - - __global__ void halo(const int nparticles_padded, const int ncellentries, const int nsolutes, const float seed) - { - assert(blockDim.x * gridDim.x >= nparticles_padded); + assert(blockDim.x * gridDim.x >= nhalo); const int laneid = threadIdx.x & 0x1f; const int warpid = threadIdx.x >> 5; - const int localbase = 32 * (warpid + 4 * blockIdx.x); - const int pid = localbase + laneid; + const int unpackbase = 32 * (warpid + 4 * blockIdx.x); + const int nunpack = min(32, nhalo - unpackbase); - if (localbase >= nparticles_padded) - return; - - int nunpack; float2 dst0, dst1, dst2; - float * dst = NULL; - - { - const uint key9 = 9 * (localbase >= packstarts_padded[9]) + 9 * (localbase >= packstarts_padded[18]); - const uint key3 = 3 * (localbase >= packstarts_padded[key9 + 3]) + 3 * (localbase >= packstarts_padded[key9 + 6]); - const uint key1 = (localbase >= packstarts_padded[key9 + key3 + 1]) + (localbase >= packstarts_padded[key9 + key3 + 2]); - const int code = key9 + key3 + key1; - assert(code >= 0 && code < 26); - assert(localbase >= packstarts_padded[code] && localbase < packstarts_padded[code + 1]); - - const int unpackbase = localbase - packstarts_padded[code]; - assert (unpackbase >= 0); - assert(unpackbase < packcount[code]); + read_AOS6f((float2 *)(halo + 3 * unpackbase), nunpack, dst0, dst1, dst2); - nunpack = min(32, packcount[code] - unpackbase); + float xforce, yforce, zforce; + read_AOS3f(acc + 3 * unpackbase, nunpack, xforce, yforce, zforce); - if (nunpack == 0) - return; + const bool outside_plus = + dst0.x >= XOFFSET || + dst0.x >= -XOFFSET && dst0.y >= YOFFSET || + dst0.x >= -XOFFSET && dst0.y >= -YOFFSET && dst1.x >= ZOFFSET; - read_AOS6f((float2 *)(packstates[code] + unpackbase), nunpack, dst0, dst1, dst2); + const bool inside_outerhalo = + dst0.x < XOFFSET + 1 && + dst0.y < YOFFSET + 1 && + dst1.x < ZOFFSET + 1 ; - dst = (float*)(packresults[code] + unpackbase); - } + const bool valid = laneid < nunpack && outside_plus && inside_outerhalo; - float xforce, yforce, zforce; - read_AOS3f(dst, nunpack, xforce, yforce, zforce); - - const int nzplanes = laneid < nunpack ? 3 : 0; + if (!valid) + return; - for(int zplane = 0; zplane < nzplanes; ++zplane) + for(int zplane = 0; zplane < 3; ++zplane) { int scan1, scan2, ncandidates, spidbase; int deltaspid1, deltaspid2; @@ -470,8 +388,8 @@ namespace KernelsContact assert(xcount >= 0); const int ycenter = YOFFSET + (int)floorf(dst0.y); - const int zcenter = ZOFFSET + (int)floorf(dst1.x); + const int zmy = zcenter - 1 + zplane; const bool zvalid = zmy >= 0 && zmy < ZCELLS; @@ -515,7 +433,7 @@ namespace KernelsContact const int m2 = (int)(i >= scan2); const int slot = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - assert(slot >= 0 && slot < ncellentries); + assert(slot >= 0 && slot < nbulk); CellEntry ce; ce.pid = tex1Dfetch(texCellEntries, slot); const int soluteid = ce.code.w; @@ -548,7 +466,7 @@ namespace KernelsContact const float t2 = ljsigma2 * invr2; const float t4 = t2 * t2; const float t6 = t4 * t2; - const float lj = min(1e3f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + const float lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(1.f - rij); @@ -561,7 +479,7 @@ namespace KernelsContact yr * (dst2.x - stmp2.x) + zr * (dst2.y - stmp2.y); - const float myrandnr = Logistic::mean0var1(seed, pid, spid); + const float myrandnr = Logistic::mean0var1(seed, unpackbase + laneid, spid); const float strength = lj + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; @@ -577,9 +495,9 @@ namespace KernelsContact assert(!isnan(yinteraction)); assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); + assert(fabs(xinteraction) < 1e5); + assert(fabs(yinteraction) < 1e5); + assert(fabs(zinteraction) < 1e5); atomicAdd(csolutesacc[soluteid] + sentry , -xinteraction); atomicAdd(csolutesacc[soluteid] + sentry + 1, -yinteraction); @@ -587,54 +505,86 @@ namespace KernelsContact } } - write_AOS3f(dst, nunpack, xforce, yforce, zforce); + acc[3 * (unpackbase + laneid) + 0] = xforce; + acc[3 * (unpackbase + laneid) + 1] = yforce; + acc[3 * (unpackbase + laneid) + 2] = zforce; } } -void ComputeContact::halo(ParticlesWrap halos[26], cudaStream_t stream) +void ComputeContact::halo(ParticlesWrap halowrap, cudaStream_t stream) { NVTX_RANGE("Contact/halo", NVTX_C7); - int nremote_padded = 0; + CUDA_CHECK(cudaPeekAtLastError()); - { - int recvpackcount[26], recvpackstarts_padded[27]; + wsolutes.push_back(halowrap); + + int ntotal = 0; + + for(int i = 0; i < wsolutes.size(); ++i) + ntotal += wsolutes[i].n; - for(int i = 0; i < 26; ++i) - recvpackcount[i] = halos[i].n; + subindices.resize(ntotal); + cellsentries.resize(ntotal); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packcount, recvpackcount, - sizeof(recvpackcount), 0, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemsetAsync(cellscount.data, 0, sizeof(int) * cellscount.size, stream)); + +#ifndef NDEBUG + CUDA_CHECK(cudaMemsetAsync(cellsentries.data, 0xff, sizeof(int) * cellsentries.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(subindices.data, 0xff, sizeof(int) * subindices.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(compressed_cellscount.data, 0xff, sizeof(unsigned char) * compressed_cellscount.capacity, stream)); + CUDA_CHECK(cudaMemsetAsync(cellsstart.data, 0xff, sizeof(int) * cellsstart.capacity, stream)); +#endif - recvpackstarts_padded[0] = 0; - for(int i = 0, s = 0; i < 26; ++i) - recvpackstarts_padded[i + 1] = (s += 32 * ((halos[i].n + 31) / 32)); + CUDA_CHECK(cudaPeekAtLastError()); - nremote_padded = recvpackstarts_padded[26]; + int ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packstarts_padded, recvpackstarts_padded, - sizeof(recvpackstarts_padded), 0, cudaMemcpyHostToDevice, stream)); + if (it.n) + subindex_local<<< (it.n + 127) / 128, 128, 0, stream >>> + (it.n, (float2 *)it.p, cellscount.data, subindices.data + ctr); - const Particle * recvpackstates[26]; + ctr += it.n; + } - for(int i = 0; i < 26; ++i) - recvpackstates[i] = halos[i].p; + compress_counts<<< (compressed_cellscount.size + 127) / 128, 128, 0, stream >>> + (compressed_cellscount.size, (int4 *)cellscount.data, (uchar4 *)compressed_cellscount.data); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packstates, recvpackstates, - sizeof(recvpackstates), 0, cudaMemcpyHostToDevice, stream)); + scan(compressed_cellscount.data, compressed_cellscount.size, stream, (uint *)cellsstart.data); - Acceleration * packresults[26]; + ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; - for(int i = 0; i < 26; ++i) - packresults[i] = halos[i].a; + if (it.n) + KernelsContact::populate<<< (it.n + 127) / 128, 128, 0, stream >>> + (subindices.data + ctr, cellsstart.data, it.n, i, ntotal, (KernelsContact::CellEntry *)cellsentries.data); - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsContact::packresults, packresults, - sizeof(packresults), 0, cudaMemcpyHostToDevice, stream)); + ctr += it.n; } - if(nremote_padded) - KernelsContact::halo<<< (nremote_padded + 127) / 128, 128, 0, stream>>> - (nremote_padded, cellsentries.size, nsolutes, local_trunk.get_float()); + CUDA_CHECK(cudaPeekAtLastError()); + + KernelsContact::bind(cellsstart.data, cellsentries.data, ntotal, wsolutes, stream, cellscount.data); + + KernelsContact::bulk_3tpp<<< (3 * cellsentries.size + 127) / 128, 128, 0, stream >>> + (wsolutes.size(), local_trunk.get_float()); + + ctr = 0; + for(int i = 0; i < wsolutes.size(); ++i) + { + const ParticlesWrap it = wsolutes[i]; + + if (it.n) + KernelsContact::halo<<< (it.n + 127) / 128, 128, 0, stream>>> + ((float2 *)it.p, it.n, wsolutes.size(), local_trunk.get_float(), (float *)it.a); + + ctr += it.n; + } CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/contact.h b/mpi-dpd/contact.h index d02aa692e..3242a72c9 100644 --- a/mpi-dpd/contact.h +++ b/mpi-dpd/contact.h @@ -21,24 +21,20 @@ class ComputeContact : public SoluteExchange::Visitor { - //cudaEvent_t evuploaded; - - int nsolutes; - + std::vector wsolutes; + SimpleDeviceBuffer subindices; SimpleDeviceBuffer compressed_cellscount; SimpleDeviceBuffer cellsentries, cellsstart, cellscount; Logistic::KISS local_trunk; - + public: ComputeContact(MPI_Comm comm); - void build_cells(std::vector wsolutes, cudaStream_t stream); - - void bulk(std::vector wsolutes, cudaStream_t stream); + void attach_bulk(std::vector wsolutes) { this->wsolutes = wsolutes; } /*override of SoluteExchange::Visitor::halo*/ - void halo(ParticlesWrap solutes[26], cudaStream_t stream); + void halo(ParticlesWrap allhalos, cudaStream_t stream); }; diff --git a/mpi-dpd/containers.cu b/mpi-dpd/containers.cu index dfd96c564..75466846e 100644 --- a/mpi-dpd/containers.cu +++ b/mpi-dpd/containers.cu @@ -227,18 +227,18 @@ namespace ParticleKernels } } -void ParticleArray::update_stage1(const float driving_acceleration, cudaStream_t stream) +void ParticleArray::update_stage1(const float driving_acceleration, cudaStream_t stream, const float timestep) { if (size) ParticleKernels::update_stage1<<<(xyzuvw.size + 127) / 128, 128, 0, stream>>>( - xyzuvw.data, axayaz.data, xyzuvw.size, dt, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille, false); + xyzuvw.data, axayaz.data, xyzuvw.size, timestep, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille, false); } -void ParticleArray::update_stage2_and_1(const float driving_acceleration, cudaStream_t stream) +void ParticleArray::update_stage2_and_1(const float driving_acceleration, cudaStream_t stream, const float timestep) { if (size) ParticleKernels::update_stage2_and_1<<<(xyzuvw.size + 127) / 128, 128, 0, stream>>> - ((float2 *)xyzuvw.data, (float *)axayaz.data, xyzuvw.size, dt, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille); + ((float2 *)xyzuvw.data, (float *)axayaz.data, xyzuvw.size, timestep, driving_acceleration, globalextent.y * 0.5 - origin.y, doublepoiseuille); } void ParticleArray::resize(int n) @@ -281,6 +281,7 @@ void CollectionRBC::resize(const int count) ncells = count; ParticleArray::resize(count * get_nvertices()); + fsi_axayaz.resize(count * get_nvertices()); } void CollectionRBC::preserve_resize(const int count) @@ -288,6 +289,7 @@ void CollectionRBC::preserve_resize(const int count) ncells = count; ParticleArray::preserve_resize(count * get_nvertices()); + fsi_axayaz.preserve_resize(count * get_nvertices()); } struct TransformedExtent diff --git a/mpi-dpd/containers.h b/mpi-dpd/containers.h index 0bd850d44..592afcc0a 100644 --- a/mpi-dpd/containers.h +++ b/mpi-dpd/containers.h @@ -28,8 +28,8 @@ struct ParticleArray void resize(int n); void preserve_resize(int n); - void update_stage1(const float driving_acceleration, cudaStream_t stream); - void update_stage2_and_1(const float driving_acceleration, cudaStream_t stream); + void update_stage1(const float driving_acceleration, cudaStream_t stream, const float timestep = dt); + void update_stage2_and_1(const float driving_acceleration, cudaStream_t stream, const float timestep = dt); void clear_velocity(); void clear_acc(cudaStream_t stream) @@ -43,12 +43,13 @@ class CollectionRBC : public ParticleArray static int (*indices)[3]; static int ntriangles; static int nvertices; + SimpleDeviceBuffer fsi_axayaz; protected: MPI_Comm cartcomm; - int ncells, myrank, dims[3], periods[3], coords[3]; + int ncells, myrank, dims[3], periods[3], coords[3]; virtual int _ntriangles() const { return ntriangles; } @@ -70,6 +71,7 @@ class CollectionRBC : public ParticleArray Particle * data() { return xyzuvw.data; } Acceleration * acc() { return axayaz.data; } + Acceleration * fsiacc() { return fsi_axayaz.data; } void remove(const int * const entries, const int nentries); void resize(const int rbcs_count); void preserve_resize(int n); diff --git a/mpi-dpd/ctc.h b/mpi-dpd/ctc.h index b9f2d4293..a8ac40ea2 100644 --- a/mpi-dpd/ctc.h +++ b/mpi-dpd/ctc.h @@ -60,7 +60,7 @@ class CollectionCTC : public CollectionRBC CollectionCTC(MPI_Comm cartcomm) : CollectionRBC(cartcomm) { - if (ctcs) + //if (ctcs) { CudaCTC::Extent extent; CudaCTC::setup(nvertices, extent); diff --git a/mpi-dpd/dpd.cu b/mpi-dpd/dpd.cu index c573642b8..3eafd9145 100644 --- a/mpi-dpd/dpd.cu +++ b/mpi-dpd/dpd.cu @@ -20,11 +20,15 @@ using namespace std; -ComputeDPD::ComputeDPD(MPI_Comm cartcomm): SolventExchange(cartcomm, 0), local_trunk(0, 0, 0, 0) +ComputeDPD::ComputeDPD(MPI_Comm cartcomm): +SolventExchange(cartcomm, 0), local_trunk(0, 0, 0, 0), +sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), sigma_zz(NULL) { int myrank; MPI_CHECK(MPI_Comm_rank(cartcomm, &myrank)); + local_trunk = Logistic::KISS(9078 - 2 * myrank, 321 - myrank, 552, 456); + for(int i = 0; i < 26; ++i) { int d[3] = { (i + 2) % 3 - 1, (i / 3 + 2) % 3 - 1, (i / 9 + 2) % 3 - 1 }; @@ -79,10 +83,17 @@ void ComputeDPD::local_interactions(const Particle * const xyzuvw, const float4 NVTX_RANGE("DPD/local", NVTX_C5); if (n > 0) + { forces_dpd_cuda_nohost((float*)xyzuvw, xyzouvwo, xyzo_half, (float *)a, n, cellsstart, cellscount, 1, XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN, aij, gammadpd, - sigma, 1. / sqrt(dt), local_trunk.get_float(), stream); + sigma, 1. / sqrt(dt), current_lseed = local_trunk.get_float(), stream); + + if (sigma_xx) + compute_stress((float *)xyzuvw, n, cellsstart, cellscount, XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN, + aij, gammadpd, sigmaf, current_lseed, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, (float *)a, stream); + } } namespace BipsBatch @@ -102,9 +113,17 @@ namespace BipsBatch __constant__ BatchInfo batchinfos[26]; - __global__ void - interaction_kernel(const float aij, const float gamma, const float sigmaf, - const int ndstall, float * const adst, const int sizeadst) + struct StressInfo + { + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz; + }; + + __constant__ StressInfo stressinfo; + + + template < bool computestresses > __global__ + void interaction_kernel(const float aij, const float gamma, const float sigmaf, + const int ndstall, float * const adst, const int sizeadst) { #if !defined(__CUDA_ARCH__) #warning __CUDA_ARCH__ not defined! assuming 350 @@ -151,7 +170,8 @@ namespace BipsBatch const float vp = info.xdst[4 + dpid * 6]; const float wp = info.xdst[5 + dpid * 6]; - const int dstbase = 3 * info.scattered_entries[dpid]; + const int dstentry = info.scattered_entries[dpid]; + const int dstbase = 3 * dstentry; assert(dstbase < sizeadst * 3); uint scan1, scan2, ncandidates, spidbase; @@ -281,6 +301,16 @@ namespace BipsBatch xforce += strength * xr; yforce += strength * yr; zforce += strength * zr; + + if (computestresses) + { + atomicAdd(stressinfo.sigma_xx + dstentry, strength * xr * _xr); + atomicAdd(stressinfo.sigma_xy + dstentry, strength * xr * _yr); + atomicAdd(stressinfo.sigma_xz + dstentry, strength * xr * _zr); + atomicAdd(stressinfo.sigma_yy + dstentry, strength * yr * _yr); + atomicAdd(stressinfo.sigma_yz + dstentry, strength * yr * _zr); + atomicAdd(stressinfo.sigma_zz + dstentry, strength * zr * _zr); + } } atomicAdd(adst + dstbase + 0, xforce); @@ -295,12 +325,15 @@ namespace BipsBatch cudaEvent_t evhalodone; void interactions(const float aij, const float gamma, const float sigma, const float invsqrtdt, - const BatchInfo infos[20], cudaStream_t computestream, cudaStream_t uploadstream, float * const acc, const int n) + const BatchInfo infos[20], cudaStream_t computestream, cudaStream_t uploadstream, float * const acc, + float * const sigma_xx, float * const sigma_xy, float * const sigma_xz, float * const sigma_yy, + float * const sigma_yz, float * const sigma_zz, const int n) { if (firstcall) { CUDA_CHECK(cudaEventCreate(&evhalodone, cudaEventDisableTiming)); - CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interaction_kernel, cudaFuncCachePreferL1)); firstcall = false; } @@ -316,12 +349,24 @@ namespace BipsBatch const int nthreads = 2 * hstart_padded[26]; + if (sigma_xx) + { + StressInfo strinfo = {sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, uploadstream)); + } + CUDA_CHECK(cudaEventRecord(evhalodone, uploadstream)); CUDA_CHECK(cudaStreamWaitEvent(computestream, evhalodone, 0)); if (nthreads) - interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + { + if (sigma_xx) + interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + else + interaction_kernel<<< (nthreads + 127) / 128, 128, 0, computestream>>>(aij, gamma, sigma * invsqrtdt, nthreads, acc, n); + } CUDA_CHECK(cudaPeekAtLastError()); } @@ -346,7 +391,7 @@ void ComputeDPD::remote_interactions(const Particle * const p, const int n, Acce const int m2 = 0 == dz; BipsBatch::BatchInfo entry = { - (float *)sendhalos[i].dbuf.data, (float2 *)recvhalos[i].dbuf.data, interrank_trunks[i].get_float(), + (float *)sendhalos[i].dbuf.data, (float2 *)recvhalos[i].dbuf.data, current_rseeds[i] = interrank_trunks[i].get_float(), sendhalos[i].dbuf.size, recvhalos[i].dbuf.size, interrank_masks[i], recvhalos[i].dcellstarts.data, sendhalos[i].scattered_entries.data, dx, dy, dz, @@ -357,7 +402,8 @@ void ComputeDPD::remote_interactions(const Particle * const p, const int n, Acce infos[i] = entry; } - BipsBatch::interactions(aij, gammadpd, sigma, 1. / sqrt(dt), infos, stream, uploadstream, (float *)a, n); + BipsBatch::interactions(aij, gammadpd, sigma, 1. / sqrt(dt), infos, stream, uploadstream, (float *)a, + sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz, n); CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/dpd.h b/mpi-dpd/dpd.h index 290fc7c49..059ac0999 100644 --- a/mpi-dpd/dpd.h +++ b/mpi-dpd/dpd.h @@ -25,18 +25,43 @@ //see the vanilla version of this code for details about how this class operates class ComputeDPD : public SolventExchange -{ +{ Logistic::KISS local_trunk; Logistic::KISS interrank_trunks[26]; + float current_lseed, current_rseeds[26], + * sigma_xx, * sigma_xy, * sigma_xz, * sigma_yy, + * sigma_yz, * sigma_zz; + bool interrank_masks[26]; - + public: - + ComputeDPD(MPI_Comm cartcomm); + void set_stress_buffers(float * const stress_xx, float * const stress_xy, float * const stress_xz, float * const stress_yy, + float * const stress_yz, float * const stress_zz) + { + sigma_xx = stress_xx; + sigma_xy = stress_xy; + sigma_xz = stress_xz; + sigma_yy = stress_yy; + sigma_yz = stress_yz; + sigma_zz = stress_zz; + } + + void clr_stress_buffers() + { + sigma_xx = NULL; + sigma_xy = NULL; + sigma_xz = NULL; + sigma_yy = NULL; + sigma_yz = NULL; + sigma_zz = NULL; + } + void remote_interactions(const Particle * const p, const int n, Acceleration * const a, cudaStream_t stream, cudaStream_t uploadstream); - void local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, Acceleration * const a, - const int * const cellsstart, const int * const cellscount, cudaStream_t stream); + void local_interactions(const Particle * const xyzuvw, const float4 * const xyzouvwo, const ushort4 * const xyzo_half, const int n, + Acceleration * const a, const int * const cellsstart, const int * const cellscount, cudaStream_t stream); }; diff --git a/mpi-dpd/dumper.cu b/mpi-dpd/dumper.cu new file mode 100644 index 000000000..fcacf6f3d --- /dev/null +++ b/mpi-dpd/dumper.cu @@ -0,0 +1,286 @@ +/* + * dumper.cu + * ctc daint + * + * Created by Dmitry Alexeev on Sep 24, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +#include +#include + +#include "dumper.h" +#include "containers.h" +#include "ctc.h" +#include "io.h" + +using namespace std; + +Dumper::Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm) : iocomm(iocomm), iocartcomm(iocartcomm), intercomm(intercomm) +{ + CollectionRBC *rdummy = new CollectionRBC(iocartcomm); + CollectionCTC *cdummy = new CollectionCTC(iocartcomm); + nrbcverts = rdummy->get_nvertices(); + nctcverts = cdummy->get_nvertices(); + + if (nrbcverts < 0 || nctcverts < 0) + { + printf("RBC vertices: %d, CTC vertices: %d\n", nrbcverts, nctcverts); + abort(); + } + + MPI_CHECK(MPI_Comm_rank(iocomm, &rank)); + + avgVels.resize(XSIZE_SUBDOMAIN*YSIZE_SUBDOMAIN*ZSIZE_SUBDOMAIN); +} + +void Dumper::qoi(Particle* rbcs, Particle * ctcs, int nrbcparts, int nctcparts, const float tm) +{ + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(iocartcomm, 3, dims, periods, coords) ); + + const int subdomain[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + const int nbins = 15; // number of rows + + vector locRBChisto(nbins, 0); + vector locCTChisto(nbins, 0); + float totcom[3] = {0, 0, 0}; + + const float rwidth = 56; + const float offset = -40; + const float tan_a = tan(1.7 / 180.0 * M_PI); + + const int nrbcs = nrbcparts / nrbcverts; + const int nctcs = nctcparts / nctcverts; + + if (nrbcs) + { + for (int p = 0; p < nrbcs; p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = rbcs + p * nrbcverts; + + for (int i=0; i < nrbcverts; i++) + for (int d = 0; d<3; d++) + { + totcom[d] += cur[i].x[d] + (coords[d] + 0.5) * subdomain[d]; + com[d] += cur[i].x[d]; + } + + for (int d = 0; d<3; d++) + com[d] = com[d] / nrbcverts + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locRBChisto[irow]++; + } + } + + if (nctcs) + for (int p = 0; p < nctcs; p++) + { + float com[3] = {0, 0, 0}; + Particle * cur = ctcs + p * nctcverts; + + for (int i=0; i < nctcverts; i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / nctcverts + (coords[d] + 0.5) * subdomain[d]; + + int irow = floor( (com[1] - offset - com[0] * tan_a) / rwidth ); + if (irow >= nbins) irow = nbins - 1; + if (irow < 0) irow = 0; + + locCTChisto[irow]++; + } + + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locRBChisto[0], &locRBChisto[0], nbins, MPI_INT, MPI_SUM, 0, iocomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &locCTChisto[0], &locCTChisto[0], nbins, MPI_INT, MPI_SUM, 0, iocomm) ); + MPI_CHECK( MPI_Reduce(rank == 0 ? MPI_IN_PLACE : totcom, totcom, 3, MPI_FLOAT, MPI_SUM, 0, iocomm) ); + + + if (nctcs) + { + float com[3] = {0, 0, 0}; + + Particle * cur = ctcs; + for (int i=0; i < nctcverts; i++) + for (int d = 0; d<3; d++) + com[d] += cur[i].x[d]; + + for (int d = 0; d<3; d++) + com[d] = com[d] / nctcverts + (coords[d] + 0.5) * subdomain[d]; + + FILE* f = fopen("ctccom.txt", qoiid == 0 ? "w" : "a"); + fprintf(f, "%f %e %e %e\n", tm, com[0], com[1], com[2]); + fclose(f); + } + + if (rank == 0) + { + if (nrbcs) + { + FILE* fout = fopen("rbchisto.dat", qoiid == 0 ? "w" : "a"); + fprintf(fout, "\n %f\n", tm); + for (int i=0; i particles.size()) particles.resize(1.1*n); + if (n > accelerations.size()) accelerations.resize(1.1*n); + + Particle* p = &particles[0]; + Acceleration* a = &accelerations[0]; + MPI_CHECK( MPI_Recv(p, n, Particle::datatype(), rank, 0, intercomm, &status) ); + MPI_CHECK( MPI_Recv(a, n, Acceleration::datatype(), rank, 0, intercomm, &status) ); + MPI_CHECK( MPI_Recv(&avgVels[0], avgVels.size()*3, MPI_FLOAT, rank, 0, intercomm, &status) ); + + double t0 = MPI_Wtime(); + + H5PartDump dump_part("allparticles->h5part", iocomm, iocartcomm), *dump_part_solvent = NULL; + H5FieldDump dump_field(iocartcomm); + + + if (rank == 0) + mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + + MPI_CHECK(MPI_Barrier(iocomm)); + + { + NVTX_RANGE("diagnostics", NVTX_C1); + diagnostics(iocomm, iocartcomm, p, n, dt, iddatadump, a); + } + + if (xyz_dumps) + { + NVTX_RANGE("xyz dump", NVTX_C2); + + if (walls && iddatadump >= wall_creation_stepid && !wallcreated) + { + if (rank == 0) + { + if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); + + if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); + + if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) + rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); + } + + MPI_CHECK(MPI_Barrier(iocomm)); + + wallcreated = true; + } + + xyz_dump(iocomm, iocartcomm, "xyz/particles->xyz", "all-particles", p, n, iddatadump > 0); + } + + if (hdf5part_dumps) + { + if (!dump_part_solvent && walls && iddatadump >= wall_creation_stepid) + { + dump_part.close(); + + dump_part_solvent = new H5PartDump("solvent-particles->h5part", iocomm, iocartcomm); + } + + if (dump_part_solvent) + dump_part_solvent->dump(p, n); + else + dump_part.dump(p, n); + } + + if (hdf5field_dumps) + { + dump_field.dump(iocomm, p, nparticles, iddatadump * steps_per_dump); + + // Dump avg vels as well + char filepath[512]; + sprintf(filepath, "h5/avgvels-%04d.h5", iddatadump); + + vector vx(avgVels.size()), vy(avgVels.size()), vz(avgVels.size()); + for (int i=0; i +#include + +#include "common.h" + +using namespace std; + +class Dumper +{ + MPI_Comm iocomm, intercomm, iocartcomm; + vector particles; + vector accelerations; + vector avgVels; + + int nrbcverts, nctcverts, qoiid, rank; + + void qoi(Particle* rbcs, Particle * ctcs, int nrbcparts, int nctcparts, const float tm); + +public: + Dumper(MPI_Comm iocomm, MPI_Comm iocartcomm, MPI_Comm intercomm); + void do_dump(); +}; + diff --git a/mpi-dpd/fsi.cu b/mpi-dpd/fsi.cu index 4c8cdbec8..727db7f7b 100644 --- a/mpi-dpd/fsi.cu +++ b/mpi-dpd/fsi.cu @@ -31,7 +31,7 @@ ComputeFSI::ComputeFSI(MPI_Comm comm) //TODO: use CUDA_CHECK(cudaEventCreateWithFlags(&evuploaded, cudaEventDisableTiming)); - KernelsFSI::Params params = {12.5 , gammadpd, sigmaf}; + KernelsFSI::Params params = {0.0f, gammadpd, sigmaf}; CUDA_CHECK(cudaMemcpyToSymbol(KernelsFSI::params, ¶ms, sizeof(params))); @@ -157,6 +157,7 @@ namespace KernelsFSI const float _zr = dst1.x - stmp1.x; const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; + assert(rij2 > 0); const float invrij = rsqrtf(rij2); @@ -271,188 +272,7 @@ void ComputeFSI::bulk(std::vector wsolutes, cudaStream_t stream) CUDA_CHECK(cudaPeekAtLastError()); } -namespace KernelsFSI -{ - __constant__ int packstarts_padded[27], packcount[26]; - __constant__ Particle * packstates[26]; - __constant__ Acceleration * packresults[26]; - - __global__ void interactions_halo(const int nparticles_padded, const int nsolvent, float * const accsolvent, const float seed) - { - assert(blockDim.x * gridDim.x >= nparticles_padded); - - const int laneid = threadIdx.x & 0x1f; - const int warpid = threadIdx.x >> 5; - const int localbase = 32 * (warpid + 4 * blockIdx.x); - const int pid = localbase + laneid; - - if (localbase >= nparticles_padded) - return; - - int nunpack; - float2 dst0, dst1, dst2; - float * dst = NULL; - - { - const uint key9 = 9 * (localbase >= packstarts_padded[9]) + 9 * (localbase >= packstarts_padded[18]); - const uint key3 = 3 * (localbase >= packstarts_padded[key9 + 3]) + 3 * (localbase >= packstarts_padded[key9 + 6]); - const uint key1 = (localbase >= packstarts_padded[key9 + key3 + 1]) + (localbase >= packstarts_padded[key9 + key3 + 2]); - const int code = key9 + key3 + key1; - assert(code >= 0 && code < 26); - assert(localbase >= packstarts_padded[code] && localbase < packstarts_padded[code + 1]); - - const int unpackbase = localbase - packstarts_padded[code]; - assert (unpackbase >= 0); - assert(unpackbase < packcount[code]); - - nunpack = min(32, packcount[code] - unpackbase); - - if (nunpack == 0) - return; - - read_AOS6f((float2 *)(packstates[code] + unpackbase), nunpack, dst0, dst1, dst2); - - dst = (float*)(packresults[code] + unpackbase); - } - - float xforce = 0, yforce = 0, zforce = 0; - - const int nzplanes = laneid < nunpack ? 3 : 0; - - for(int zplane = 0; zplane < nzplanes; ++zplane) - { - int scan1, scan2, ncandidates, spidbase; - int deltaspid1, deltaspid2; - - { - enum - { - XCELLS = XSIZE_SUBDOMAIN, - YCELLS = YSIZE_SUBDOMAIN, - ZCELLS = ZSIZE_SUBDOMAIN, - XOFFSET = XCELLS / 2, - YOFFSET = YCELLS / 2, - ZOFFSET = ZCELLS / 2 - }; - - const int NCELLS = XSIZE_SUBDOMAIN * YSIZE_SUBDOMAIN * ZSIZE_SUBDOMAIN; - const int xcenter = XOFFSET + (int)floorf(dst0.x); - const int xstart = max(0, xcenter - 1); - const int xcount = min(XCELLS, xcenter + 2) - xstart; - - if (xcenter - 1 >= XCELLS || xcenter + 2 <= 0) - continue; - - assert(xcount >= 0); - - const int ycenter = YOFFSET + (int)floorf(dst0.y); - - const int zcenter = ZOFFSET + (int)floorf(dst1.x); - const int zmy = zcenter - 1 + zplane; - const bool zvalid = zmy >= 0 && zmy < ZCELLS; - - int count0 = 0, count1 = 0, count2 = 0; - - if (zvalid && ycenter - 1 >= 0 && ycenter - 1 < YCELLS) - { - const int cid0 = xstart + XCELLS * (ycenter - 1 + YCELLS * zmy); - assert(cid0 >= 0 && cid0 + xcount <= NCELLS); - spidbase = tex1Dfetch(texCellsStart, cid0); - count0 = ((cid0 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid0 + xcount)) - spidbase; - } - - if (zvalid && ycenter >= 0 && ycenter < YCELLS) - { - const int cid1 = xstart + XCELLS * (ycenter + YCELLS * zmy); - assert(cid1 >= 0 && cid1 + xcount <= NCELLS); - deltaspid1 = tex1Dfetch(texCellsStart, cid1); - count1 = ((cid1 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid1 + xcount)) - deltaspid1; - } - - if (zvalid && ycenter + 1 >= 0 && ycenter + 1 < YCELLS) - { - const int cid2 = xstart + XCELLS * (ycenter + 1 + YCELLS * zmy); - deltaspid2 = tex1Dfetch(texCellsStart, cid2); - assert(cid2 >= 0 && cid2 + xcount <= NCELLS); - count2 = ((cid2 + xcount == NCELLS) ? nsolvent : tex1Dfetch(texCellsStart, cid2 + xcount)) - deltaspid2; - } - - scan1 = count0; - scan2 = count0 + count1; - ncandidates = scan2 + count2; - - deltaspid1 -= scan1; - deltaspid2 -= scan2; - } - - for(int i = 0; i < ncandidates; ++i) - { - const int m1 = (int)(i >= scan1); - const int m2 = (int)(i >= scan2); - const int spid = i + (m2 ? deltaspid2 : m1 ? deltaspid1 : spidbase); - - assert(spid >= 0 && spid < nsolvent); - - const int sentry = 3 * spid; - const float2 stmp0 = tex1Dfetch(texSolventParticles, sentry ); - const float2 stmp1 = tex1Dfetch(texSolventParticles, sentry + 1); - const float2 stmp2 = tex1Dfetch(texSolventParticles, sentry + 2); - - const float _xr = dst0.x - stmp0.x; - const float _yr = dst0.y - stmp0.y; - const float _zr = dst1.x - stmp1.x; - - const float rij2 = _xr * _xr + _yr * _yr + _zr * _zr; - - const float invrij = rsqrtf(rij2); - - const float rij = rij2 * invrij; - - if (rij2 >= 1) - continue; - - const float argwr = 1.f - rij; - const float wr = viscosity_function<-VISCOSITY_S_LEVEL>(argwr); - - const float xr = _xr * invrij; - const float yr = _yr * invrij; - const float zr = _zr * invrij; - - const float rdotv = - xr * (dst1.y - stmp1.y) + - yr * (dst2.x - stmp2.x) + - zr * (dst2.y - stmp2.y); - - const float myrandnr = Logistic::mean0var1(seed, pid, spid); - - const float strength = params.aij * argwr + (- params.gamma * wr * rdotv + params.sigmaf * myrandnr) * wr; - - const float xinteraction = strength * xr; - const float yinteraction = strength * yr; - const float zinteraction = strength * zr; - - xforce += xinteraction; - yforce += yinteraction; - zforce += zinteraction; - - assert(!isnan(xinteraction)); - assert(!isnan(yinteraction)); - assert(!isnan(zinteraction)); - assert(fabs(xinteraction) < 1e4); - assert(fabs(yinteraction) < 1e4); - assert(fabs(zinteraction) < 1e4); - - atomicAdd(accsolvent + sentry , -xinteraction); - atomicAdd(accsolvent + sentry + 1, -yinteraction); - atomicAdd(accsolvent + sentry + 2, -zinteraction); - } - } - - write_AOS3f(dst, nunpack, xforce, yforce, zforce); - } -} - -void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) +void ComputeFSI::halo(ParticlesWrap halowrap, cudaStream_t stream) { NVTX_RANGE("FSI/halo", NVTX_C7); @@ -460,50 +280,9 @@ void ComputeFSI::halo(ParticlesWrap halos[26], cudaStream_t stream) CUDA_CHECK(cudaPeekAtLastError()); - int nremote_padded = 0; - - { - int recvpackcount[26], recvpackstarts_padded[27]; - - for(int i = 0; i < 26; ++i) - recvpackcount[i] = halos[i].n; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packcount, recvpackcount, - sizeof(recvpackcount), 0, cudaMemcpyHostToDevice, stream)); - - recvpackstarts_padded[0] = 0; - for(int i = 0, s = 0; i < 26; ++i) - recvpackstarts_padded[i + 1] = (s += 32 * ((halos[i].n + 31) / 32)); - - nremote_padded = recvpackstarts_padded[26]; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packstarts_padded, recvpackstarts_padded, - sizeof(recvpackstarts_padded), 0, cudaMemcpyHostToDevice, stream)); - } - - { - const Particle * recvpackstates[26]; - - for(int i = 0; i < 26; ++i) - recvpackstates[i] = halos[i].p; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packstates, recvpackstates, - sizeof(recvpackstates), 0, cudaMemcpyHostToDevice, stream)); - } - - { - Acceleration * packresults[26]; - - for(int i = 0; i < 26; ++i) - packresults[i] = halos[i].a; - - CUDA_CHECK(cudaMemcpyToSymbolAsync(KernelsFSI::packresults, packresults, - sizeof(packresults), 0, cudaMemcpyHostToDevice, stream)); - } - - if(nremote_padded) - KernelsFSI::interactions_halo<<< (nremote_padded + 127) / 128, 128, 0, stream>>> - (nremote_padded, wsolvent.n, (float *)wsolvent.a, local_trunk.get_float()); + if (halowrap.n) + KernelsFSI::interactions_3tpp<<< (3 * halowrap.n + 127) / 128, 128, 0, stream >>> + ((float2 *)halowrap.p, halowrap.n, wsolvent.n, (float *)halowrap.a, (float *)wsolvent.a, local_trunk.get_float()); CUDA_CHECK(cudaPeekAtLastError()); } diff --git a/mpi-dpd/fsi.h b/mpi-dpd/fsi.h index 130e753da..735f6031b 100644 --- a/mpi-dpd/fsi.h +++ b/mpi-dpd/fsi.h @@ -36,5 +36,5 @@ class ComputeFSI : public SoluteExchange::Visitor void bulk(std::vector wsolutes, cudaStream_t stream); /*override of SoluteExchange::Visitor::halo*/ - void halo(ParticlesWrap solutes[26], cudaStream_t stream); + void halo(ParticlesWrap halowrap, cudaStream_t stream); }; diff --git a/mpi-dpd/io.cu b/mpi-dpd/io.cu index 11b963c29..d5c87c979 100644 --- a/mpi-dpd/io.cu +++ b/mpi-dpd/io.cu @@ -27,6 +27,9 @@ #include #include +//#define USE_POSIX_IO +#include + #include "io.h" using namespace std; @@ -44,7 +47,7 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha bool filenotthere; if (rank == 0) - filenotthere = access(filename, F_OK ) == -1; + filenotthere = access(filename, F_OK ) == -1; MPI_CHECK( MPI_Bcast(&filenotthere, 1, MPI_INT, 0, comm) ); @@ -54,7 +57,7 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), MPI_INFO_NULL, &f) ); if (!append) - MPI_CHECK( MPI_File_set_size (f, 0)); + MPI_CHECK( MPI_File_set_size (f, 0)); MPI_Offset base; MPI_CHECK( MPI_File_get_position(f, &base)); @@ -63,17 +66,17 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha if (rank == 0) { - ss << n << "\n"; - ss << particlename << "\n"; + ss << n << "\n"; + ss << particlename << "\n"; - printf("xyz dump <%s>: total number of particles: %d\n", filename, n); + printf("xyz dump <%s>: total number of particles: %d\n", filename, n); } for(int i = 0; i < nlocal; ++i) - ss << rank << " " - << (particles[i].x[0] + XSIZE_SUBDOMAIN / 2 + coords[0] * XSIZE_SUBDOMAIN) << " " - << (particles[i].x[1] + YSIZE_SUBDOMAIN / 2 + coords[1] * YSIZE_SUBDOMAIN) << " " - << (particles[i].x[2] + ZSIZE_SUBDOMAIN / 2 + coords[2] * ZSIZE_SUBDOMAIN) << "\n"; + ss << rank << " " + << (particles[i].x[0] + XSIZE_SUBDOMAIN / 2 + coords[0] * XSIZE_SUBDOMAIN) << " " + << (particles[i].x[1] + YSIZE_SUBDOMAIN / 2 + coords[1] * YSIZE_SUBDOMAIN) << " " + << (particles[i].x[2] + ZSIZE_SUBDOMAIN / 2 + coords[2] * ZSIZE_SUBDOMAIN) << "\n"; string content = ss.str(); @@ -83,33 +86,188 @@ void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const cha MPI_Status status; - MPI_CHECK( MPI_File_write_at_all(f, base + offset, const_cast(content.c_str()), len, MPI_CHAR, &status)); + MPI_CHECK( MPI_File_write_at(f, base + offset, const_cast(content.c_str()), len, MPI_CHAR, &status)); MPI_CHECK( MPI_File_close(&f)); } -void _write_bytes(const void * const ptr, const int nbytes32, MPI_File f, MPI_Comm comm) +void _write_bytes_mpi(const void * const ptr, const MPI_Offset nbytes, MPI_File f, const MPI_Offset base, MPI_Offset offset) { - MPI_Offset base; - MPI_CHECK( MPI_File_get_position(f, &base)); + MPI_Status status; + MPI_CHECK( MPI_File_write_at_all(f, base + offset, ptr, nbytes, MPI_CHAR, &status)); +} - MPI_Offset offset = 0, nbytes = nbytes32; - MPI_CHECK( MPI_Exscan(&nbytes, &offset, 1, MPI_OFFSET, MPI_SUM, comm)); +void _write_bytes_posix(const void * const ptr, const int nbytes32, int f, MPI_Offset base0, MPI_Offset offset0, MPI_Comm comm) +{ + MPI_Offset base = base0; + MPI_Offset offset = offset0; + MPI_Offset nbytes = nbytes32; - MPI_Status status; + MPI_Offset rc; + rc = lseek64(f, base + offset, SEEK_SET); + if (rc == -1) + { + printf("lseek64 failed!\n"); + MPI_Abort(MPI_COMM_WORLD, rc); + } - MPI_CHECK( MPI_File_write_at_all(f, base + offset, ptr, nbytes, MPI_CHAR, &status)); + char * ptr0 = (char *)ptr; + MPI_Offset remaining = nbytes; + while (remaining > 0) + { + rc = write(f, ptr0, nbytes); + if (rc == -1) + { + printf("write failed!\n"); + MPI_Abort(MPI_COMM_WORLD, rc); + } + if (rc < remaining) + { + } + remaining -= rc; + ptr0 += rc; + } +} + +void ply_dump_mpi(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) +{ + double t0 = MPI_Wtime(); + std::vector particles(_particles, _particles + ninstances * nvertices_per_instance); + + int rank, size; + MPI_CHECK( MPI_Comm_rank(comm, &rank) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); + + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + + /* Part II: particles */ + int64_t NPOINTS = 0; + const int64_t n = particles.size(); + MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_LONG_LONG, MPI_SUM, comm) ); + + /* Part III: triangles */ + const int64_t ntriangles = ntriangles_per_instance * ninstances; + int64_t NTRIANGLES = 0; + MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_LONG_LONG, MPI_SUM, comm) ); + + /* Part I: header */ + std::stringstream ss; + if (rank == 0) + { + ss << "ply\n"; + ss << "format binary_little_endian 1.0\n"; + ss << "element vertex " << NPOINTS << "\n"; + ss << "property float x\nproperty float y\nproperty float z\n"; + ss << "property float u\nproperty float v\nproperty float w\n"; + //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; + ss << "element face " << NTRIANGLES << "\n"; + ss << "property list int int vertex_index\n"; + ss << "end_header\n"; + } + string content = ss.str(); + + const int headersize = content.size(); + int HEADERSIZE = 0; + MPI_CHECK( MPI_Allreduce(&headersize, &HEADERSIZE, 1, MPI_INT, MPI_SUM, comm) ); + + int64_t size0 = HEADERSIZE; + int64_t size1 = NPOINTS*sizeof(Particle); + // unsigned long size2 = NTRIANGLES*4*sizeof(int); - MPI_Offset ntotal = 0; - MPI_CHECK( MPI_Allreduce(&nbytes, &ntotal, 1, MPI_OFFSET, MPI_SUM, comm) ); + MPI_Offset base0 = 0; + MPI_Offset base1 = base0 + size0; + MPI_Offset base2 = base1 + size1; - MPI_CHECK( MPI_File_seek(f, ntotal, MPI_SEEK_CUR)); + int64_t ioffset0 = 0; + int64_t ioffset1 = 0; + int64_t ioffset2 = 0; + MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_LONG_LONG, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_LONG_LONG, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_LONG_LONG, MPI_SUM, comm)); + + MPI_Offset poffset0 = ioffset0*sizeof(char); + MPI_Offset poffset1 = ioffset1*sizeof(Particle); + MPI_Offset poffset2 = ioffset2*4*sizeof(int); + + MPI_File f; + double t1 = MPI_Wtime(); + + int cb = 1; + while (cb*2 <= size) cb *= 2; + + cb = min(cb, 128); + char cbstr[100]; + sprintf(cbstr, "%d", cb); + + MPI_Info info; + MPI_Info_create(&info); + MPI_Info_set(info, "cb_nodes", cbstr); + MPI_Info_set(info, "romio_cb_write", "enable"); + MPI_Info_set(info, "romio_ds_write", "disable"); + MPI_Info_set(info, "striping_factor", cbstr); + + MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), info, &f) ); + double t2 = MPI_Wtime(); + +#if 0 + if (!append) { + // MPI_CHECK( MPI_File_set_size (f, 0)); + MPI_CHECK (MPI_File_set_size (f, size0 + size1 + size2)); + } +#endif + + _write_bytes_mpi(content.c_str(), content.size(), f, base0, poffset0); + const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; + + for(int i = 0; i < n; ++i) + for(int c = 0; c < 3; ++c) + particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; + + _write_bytes_mpi(&particles.front(), sizeof(Particle) * n, f, base1, poffset1); + + int poffset = ioffset1; + std::vector buf; + + for(int j = 0; j < ninstances; ++j) + for(int i = 0; i < ntriangles_per_instance; ++i) + { + int primitive[4] = { 3, + poffset + nvertices_per_instance * j + mesh_indices[i][0], + poffset + nvertices_per_instance * j + mesh_indices[i][1], + poffset + nvertices_per_instance * j + mesh_indices[i][2] }; + + buf.insert(buf.end(), primitive, primitive + 4); + } + + _write_bytes_mpi(&buf.front(), sizeof(int) * buf.size(), f, base2, poffset2); + + MPI_Barrier(comm); + +// double t3 = MPI_Wtime(); + MPI_CHECK( MPI_File_close(&f)); +// double t4 = MPI_Wtime(); +// +// double d0 = 1e3*(t1 - t0); +// double d1 = 1e3*(t2 - t1); +// double d2 = 1e3*(t3 - t2); +// double d3 = 1e3*(t4 - t3); +// +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d0, &d0, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d1, &d1, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d2, &d2, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// MPI_Reduce(rank == 0 ? MPI_IN_PLACE : &d3, &d3, 1, MPI_DOUBLE, MPI_MAX, 0, comm); +// +// if (!rank) printf("ply_dump_mpi:\t %.2f ms; \t prep: %.2f, open %.2f, write %.2f, close %.2f\n", d0+d1+d2+d3, d0, d1, d2, d3); } -void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, - int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, - Particle * _particles, int nvertices_per_instance, bool append) +void ply_dump_posix(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) { + double t0 = MPI_Wtime(); std::vector particles(_particles, _particles + ninstances * nvertices_per_instance); int rank; @@ -118,69 +276,195 @@ void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + /* Part II: particles */ int NPOINTS = 0; const int n = particles.size(); - MPI_CHECK( MPI_Reduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, 0, comm) ); + MPI_CHECK( MPI_Allreduce(&n, &NPOINTS, 1, MPI_INT, MPI_SUM, comm) ); + /* Part III: triangles */ const int ntriangles = ntriangles_per_instance * ninstances; int NTRIANGLES = 0; - MPI_CHECK( MPI_Reduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, 0, comm) ); + MPI_CHECK( MPI_Allreduce(&ntriangles, &NTRIANGLES, 1, MPI_INT, MPI_SUM, comm) ); - MPI_File f; - MPI_CHECK( MPI_File_open(comm, filename , MPI_MODE_WRONLY | (append ? MPI_MODE_APPEND : MPI_MODE_CREATE), MPI_INFO_NULL, &f) ); + /* Part I: header */ + std::stringstream ss; + if (rank == 0) + { + ss << "ply\n"; + ss << "format binary_little_endian 1.0\n"; + ss << "element vertex " << NPOINTS << "\n"; + ss << "property float x\nproperty float y\nproperty float z\n"; + ss << "property float u\nproperty float v\nproperty float w\n"; + //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; + ss << "element face " << NTRIANGLES << "\n"; + ss << "property list int int vertex_index\n"; + ss << "end_header\n"; + } + string content = ss.str(); - if (!append) - MPI_CHECK( MPI_File_set_size (f, 0)); + const int headersize = content.size(); + int HEADERSIZE = 0; + MPI_CHECK( MPI_Allreduce(&headersize, &HEADERSIZE, 1, MPI_INT, MPI_SUM, comm) ); - std::stringstream ss; + unsigned long size0 = HEADERSIZE; + unsigned long size1 = NPOINTS*sizeof(Particle); + // unsigned long size2 = NTRIANGLES*4*sizeof(int); + + MPI_Offset base0 = 0; + MPI_Offset base1 = base0 + size0; + MPI_Offset base2 = base1 + size1; + int ioffset0 = 0; + int ioffset1 = 0; + int ioffset2 = 0; + MPI_CHECK( MPI_Exscan(&headersize, &ioffset0, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&n, &ioffset1, 1, MPI_INTEGER, MPI_SUM, comm)); + MPI_CHECK( MPI_Exscan(&ntriangles, &ioffset2, 1, MPI_INTEGER, MPI_SUM, comm)); + + MPI_Offset poffset0 = ioffset0*sizeof(char); + MPI_Offset poffset1 = ioffset1*sizeof(Particle); + MPI_Offset poffset2 = ioffset2*4*sizeof(int); + + int f; if (rank == 0) { - ss << "ply\n"; - ss << "format binary_little_endian 1.0\n"; - ss << "element vertex " << NPOINTS << "\n"; - ss << "property float x\nproperty float y\nproperty float z\n"; - ss << "property float u\nproperty float v\nproperty float w\n"; - //ss << "property float xnormal\nproperty float ynormal\nproperty float znormal\n"; - ss << "element face " << NTRIANGLES << "\n"; - ss << "property list int int vertex_index\n"; - ss << "end_header\n"; + f = open((char *)filename, O_CREAT | O_WRONLY | O_DIRECT, 0664); + if (f <= 0) + { + printf("File creation failed!\n"); + MPI_Abort(MPI_COMM_WORLD, f); + } + //ftruncate(f, size0+size1+size2); + close(f); } - string content = ss.str(); - - _write_bytes(content.c_str(), content.size(), f, comm); +#if 1 + MPI_CHECK(MPI_Barrier(comm)); +#else + int flag = 0; + MPI_Status status; + MPI_Request req; + MPI_Ibarrier(comm, &req); + while (1) + { + MPI_Test(&req, &flag, &status); + if (flag == 1) + break; + else + usleep(100); + } +#endif + f = open((char *)filename, O_WRONLY, 0664); + if (f <= 0) + { + printf("File opening failed!\n"); + MPI_Abort(MPI_COMM_WORLD, f); + } + _write_bytes_posix(content.c_str(), content.size(), f, base0, poffset0, comm); const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int i = 0; i < n; ++i) - for(int c = 0; c < 3; ++c) - particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; - - _write_bytes(&particles.front(), sizeof(Particle) * n, f, comm); + for(int c = 0; c < 3; ++c) + particles[i].x[c] += L[c] / 2 + coords[c] * L[c]; - int poffset = 0; - - MPI_CHECK( MPI_Exscan(&n, &poffset, 1, MPI_INTEGER, MPI_SUM, comm)); + _write_bytes_posix(&particles.front(), sizeof(Particle) * n, f, base1, poffset1, comm); + int poffset = ioffset1; std::vector buf; for(int j = 0; j < ninstances; ++j) - for(int i = 0; i < ntriangles_per_instance; ++i) - { - int primitive[4] = { 3, - poffset + nvertices_per_instance * j + mesh_indices[i][0], - poffset + nvertices_per_instance * j + mesh_indices[i][1], - poffset + nvertices_per_instance * j + mesh_indices[i][2] }; + for(int i = 0; i < ntriangles_per_instance; ++i) + { + int primitive[4] = { 3, + poffset + nvertices_per_instance * j + mesh_indices[i][0], + poffset + nvertices_per_instance * j + mesh_indices[i][1], + poffset + nvertices_per_instance * j + mesh_indices[i][2] }; + + buf.insert(buf.end(), primitive, primitive + 4); + } + + _write_bytes_posix(&buf.front(), sizeof(int) * buf.size(), f, base2, poffset2, comm); + + close(f); + + double t1 = MPI_Wtime(); + if (!rank && (t1-t0 > 0.05)) printf("ply_dump_posix:\t %f ms\n", 1e3*(t1-t0)); +} + +void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, + int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, + Particle * _particles, int nvertices_per_instance, bool append) +{ +#ifdef USE_POSIX_IO + ply_dump_posix(comm, cartcomm, filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); +#else + ply_dump_mpi(comm, cartcomm, filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); + +#if 0 // only for debugging + char new_filename[256]; + strcpy(new_filename, filename); + strcat(new_filename, ".posix"); + ply_dump_posix(comm, cartcomm, new_filename, mesh_indices, ninstances, ntriangles_per_instance, + _particles, nvertices_per_instance, append); +#endif +#endif + +} + +void stress_dump(MPI_Comm cartcomm, const char * filename, const int nparticles, + const Particle * const particles, + const float * const stress_xx, const float * const stress_xy, const float * const stress_xz, + const float * const stress_yy, const float * const stress_yz, const float * const stress_zz) +{ + std::vector buf(nparticles * 12); - buf.insert(buf.end(), primitive, primitive + 4); - } + int rank; + MPI_CHECK( MPI_Comm_rank(cartcomm, &rank) ); + + int dims[3], periods[3], coords[3]; + MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); + + int NALL = 0; + const int n = nparticles; + MPI_CHECK( MPI_Allreduce(&n, &NALL, 1, MPI_INT, MPI_SUM, cartcomm) ); + + MPI_File f; + MPI_CHECK( MPI_File_open(cartcomm, filename , MPI_MODE_WRONLY | MPI_MODE_CREATE, MPI_INFO_NULL, &f) ); + + MPI_CHECK( MPI_File_set_size (f, sizeof(float) * 12 * NALL )); - _write_bytes(&buf.front(), sizeof(int) * buf.size(), f, comm); + const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; + + for(int i = 0; i < n; ++i) + { + const int base = 12 * i; + + for(int c = 0; c < 3; ++c) + buf[base + c] = particles[i].x[c] + L[c] / 2 + coords[c] * L[c]; + + for(int c = 0; c < 3; ++c) + buf[base + 3 + c] = particles[i].u[c]; + + buf[base + 6] = stress_xx[i] / 2; + buf[base + 7] = stress_xy[i] / 2; + buf[base + 8] = stress_xz[i] / 2; + buf[base + 9] = stress_yy[i] / 2; + buf[base + 10] = stress_yz[i] / 2; + buf[base + 11] = stress_zz[i] / 2; + } + + MPI_Offset offset = 0, nbytes = (MPI_Offset)sizeof(float) * 12 * n; + MPI_CHECK( MPI_Exscan(&nbytes, &offset, 1, MPI_OFFSET, MPI_SUM, cartcomm)); + + _write_bytes_mpi(buf.data(), nbytes, f, 0, offset); MPI_CHECK( MPI_File_close(&f)); } + H5PartDump::H5PartDump(const string fname, MPI_Comm comm, MPI_Comm cartcomm): tstamp(0), disposed(false) { _initialize(fname, comm, cartcomm); @@ -195,7 +479,7 @@ void H5PartDump::_initialize(const std::string filename, MPI_Comm comm, MPI_Comm const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int c = 0; c < 3; ++c) - origin[c] = L[c] / 2 + coords[c] * L[c]; + origin[c] = L[c] / 2 + coords[c] * L[c]; mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); @@ -215,7 +499,7 @@ void H5PartDump::dump(Particle * host_particles, int n) { #ifndef NO_H5PART if (disposed) - return; + return; H5PartFile * f = (H5PartFile *)handler; @@ -227,12 +511,12 @@ void H5PartDump::dump(Particle * host_particles, int n) for(int c = 0; c < 3; ++c) { - vector data(n); + vector data(n); - for(int i = 0; i < n; ++i) - data[i] = host_particles[i].x[c] + origin[c]; + for(int i = 0; i < n; ++i) + data[i] = host_particles[i].x[c] + origin[c]; - H5PartWriteDataFloat32(f, labels[c].c_str(), &data.front()); + H5PartWriteDataFloat32(f, labels[c].c_str(), &data.front()); } tstamp++; @@ -244,20 +528,20 @@ void H5PartDump::_dispose() #ifndef NO_H5PART if (!disposed) { - H5PartFile * f = (H5PartFile *)handler; + H5PartFile * f = (H5PartFile *)handler; - H5PartCloseFile(f); + H5PartCloseFile(f); - disposed = true; + disposed = true; - handler = NULL; + handler = NULL; } #endif } H5PartDump::~H5PartDump() { - _dispose(); + _dispose(); } void H5FieldDump::_xdmf_header(FILE * xmf) @@ -269,12 +553,12 @@ void H5FieldDump::_xdmf_header(FILE * xmf) } void H5FieldDump::_xdmf_grid(FILE * xmf, float time, - const char * const h5path, const char * const * channelnames, int nchannels) + const char * const h5path, const char * const * channelnames, int nchannels) { fprintf(xmf, " \n"); fprintf(xmf, " \n"); - } +} void H5FieldDump::_xdmf_epilogue(FILE * xmf) { @@ -309,14 +593,30 @@ void H5FieldDump::_xdmf_epilogue(FILE * xmf) } void H5FieldDump::_write_fields(const char * const path2h5, - const float * const channeldata[], const char * const * const channelnames, const int nchannels, - MPI_Comm comm, const float time) + const float * const channeldata[], const char * const * const channelnames, const int nchannels, + MPI_Comm comm, const float time) { #ifndef NO_H5 - int nranks[3], periods[3], myrank[3]; + int nranks[3], periods[3], myrank[3], size; MPI_CHECK( MPI_Cart_get(cartcomm, 3, nranks, periods, myrank) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); id_t plist_id_access = H5Pcreate(H5P_FILE_ACCESS); + + int cb = 1; + while (cb*2 <= size) cb *= 2; + + cb = min(cb, 128); + char cbstr[100]; + sprintf(cbstr, "%d", cb); + + MPI_Info info; + MPI_Info_create(&info); + MPI_Info_set(info, "cb_nodes", cbstr); + MPI_Info_set(info, "romio_cb_write", "enable"); + MPI_Info_set(info, "romio_ds_write", "disable"); + MPI_Info_set(info, "striping_factor", cbstr); + H5Pset_fapl_mpio(plist_id_access, comm, MPI_INFO_NULL); hid_t file_id = H5Fcreate(path2h5, H5F_ACC_TRUNC, H5P_DEFAULT, plist_id_access); @@ -328,23 +628,23 @@ void H5FieldDump::_write_fields(const char * const path2h5, for(int ichannel = 0; ichannel < nchannels; ++ichannel) { - hid_t dset_id = H5Dcreate(file_id, channelnames[ichannel], H5T_NATIVE_FLOAT, filespace_simple, H5P_DEFAULT, H5P_DEFAULT, H5P_DEFAULT); - id_t plist_id = H5Pcreate(H5P_DATASET_XFER); + hid_t dset_id = H5Dcreate(file_id, channelnames[ichannel], H5T_NATIVE_FLOAT, filespace_simple, H5P_DEFAULT, H5P_DEFAULT, H5P_DEFAULT); + id_t plist_id = H5Pcreate(H5P_DATASET_XFER); - H5Pset_dxpl_mpio(plist_id, H5FD_MPIO_COLLECTIVE); + H5Pset_dxpl_mpio(plist_id, H5FD_MPIO_COLLECTIVE); - hsize_t start[4] = { myrank[2] * L[2], myrank[1] * L[1], myrank[0] * L[0], 0}; - hsize_t extent[4] = { L[2], L[1], L[0], 1}; - hid_t filespace = H5Dget_space(dset_id); - H5Sselect_hyperslab(filespace, H5S_SELECT_SET, start, NULL, extent, NULL); + hsize_t start[4] = { myrank[2] * L[2], myrank[1] * L[1], myrank[0] * L[0], 0}; + hsize_t extent[4] = { L[2], L[1], L[0], 1}; + hid_t filespace = H5Dget_space(dset_id); + H5Sselect_hyperslab(filespace, H5S_SELECT_SET, start, NULL, extent, NULL); - hid_t memspace = H5Screate_simple(4, extent, NULL); - herr_t status = H5Dwrite(dset_id, H5T_NATIVE_FLOAT, memspace, filespace, plist_id, channeldata[ichannel]); + hid_t memspace = H5Screate_simple(4, extent, NULL); + herr_t status = H5Dwrite(dset_id, H5T_NATIVE_FLOAT, memspace, filespace, plist_id, channeldata[ichannel]); - H5Sclose(memspace); - H5Sclose(filespace); - H5Pclose(plist_id); - H5Dclose(dset_id); + H5Sclose(memspace); + H5Sclose(filespace); + H5Pclose(plist_id); + H5Dclose(dset_id); } H5Sclose(filespace_simple); @@ -355,17 +655,17 @@ void H5FieldDump::_write_fields(const char * const path2h5, if (!rankscalar) { - char wrapper[256]; - sprintf(wrapper, "%s.xmf", string(path2h5).substr(0, string(path2h5).find_last_of(".h5") - 2).data()); + char wrapper[256]; + sprintf(wrapper, "%s.xmf", string(path2h5).substr(0, string(path2h5).find_last_of(".h5") - 2).data()); - FILE * xmf = fopen(wrapper, "w"); - assert(xmf); + FILE * xmf = fopen(wrapper, "w"); + assert(xmf); - _xdmf_header(xmf); - _xdmf_grid(xmf, time, string(path2h5).substr(string(path2h5).find_last_of("/") + 1).c_str(), channelnames, nchannels); - _xdmf_epilogue(xmf); + _xdmf_header(xmf); + _xdmf_grid(xmf, time, string(path2h5).substr(string(path2h5).find_last_of("/") + 1).c_str(), channelnames, nchannels); + _xdmf_epilogue(xmf); - fclose(xmf); + fclose(xmf); } #endif // NO_H5 } @@ -378,7 +678,7 @@ H5FieldDump::H5FieldDump(MPI_Comm cartcomm): cartcomm(cartcomm), last_idtimestep const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int c = 0; c < 3; ++c) - globalsize[c] = L[c] * dims[c]; + globalsize[c] = L[c] * dims[c]; } void H5FieldDump::dump_scalarfield(MPI_Comm comm, const float * const data, const char * channelname) @@ -399,41 +699,41 @@ void H5FieldDump::dump(MPI_Comm comm, const Particle * const p, const int n, int vector rho(ncells), u[3]; for(int c = 0; c < 3; ++c) - u[c].resize(ncells); + u[c].resize(ncells); for(int i = 0; i < n; ++i) { - const int cellindex[3] = { - max(0, min(XSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[0])) + XSIZE_SUBDOMAIN / 2)), - max(0, min(YSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[1])) + YSIZE_SUBDOMAIN / 2)), - max(0, min(ZSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[2])) + ZSIZE_SUBDOMAIN / 2)) + const int cellindex[3] = { + max(0, min(XSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[0])) + XSIZE_SUBDOMAIN / 2)), + max(0, min(YSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[1])) + YSIZE_SUBDOMAIN / 2)), + max(0, min(ZSIZE_SUBDOMAIN - 1, (int)(floor(p[i].x[2])) + ZSIZE_SUBDOMAIN / 2)) }; - const int entry = cellindex[0] + XSIZE_SUBDOMAIN * (cellindex[1] + YSIZE_SUBDOMAIN * cellindex[2]); + const int entry = cellindex[0] + XSIZE_SUBDOMAIN * (cellindex[1] + YSIZE_SUBDOMAIN * cellindex[2]); - rho[entry] += 1; + rho[entry] += 1; - for(int c = 0; c < 3; ++c) - u[c][entry] += p[i].u[c]; + for(int c = 0; c < 3; ++c) + u[c][entry] += p[i].u[c]; } for(int c = 0; c < 3; ++c) - for(int i = 0; i < ncells; ++i) - u[c][i] = rho[i] ? u[c][i] / rho[i] : 0; + for(int i = 0; i < ncells; ++i) + u[c][i] = rho[i] ? u[c][i] / rho[i] : 0; const char * names[] = { "density", "u", "v", "w" }; if (!directory_exists) { - int rank; - MPI_CHECK(MPI_Comm_rank(comm, &rank)); - - if (rank == 0) - mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); + int rank; + MPI_CHECK(MPI_Comm_rank(comm, &rank)); + + if (rank == 0) + mkdir("h5", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); - directory_exists = true; + directory_exists = true; - MPI_CHECK(MPI_Barrier(comm)); + MPI_CHECK(MPI_Barrier(comm)); } char filepath[512]; @@ -450,7 +750,7 @@ H5FieldDump::~H5FieldDump() { #ifndef NO_H5 if (last_idtimestep == 0) - return; + return; FILE * xmf = fopen("h5/flowfields-sequence.xmf", "w"); @@ -463,10 +763,10 @@ H5FieldDump::~H5FieldDump() const char * channelnames[] = { "density", "u", "v", "w" }; for(int it = 0; it <= last_idtimestep; it += steps_per_dump) { - char filepath[512]; - sprintf(filepath, "h5/flowfields-%04d.h5", it / steps_per_dump); + char filepath[512]; + sprintf(filepath, "h5/flowfields-%04d.h5", it / steps_per_dump); - _xdmf_grid(xmf, it * dt, string(filepath).substr(string(filepath).find_last_of("/") + 1).c_str(), channelnames, 4); + _xdmf_grid(xmf, it * dt, string(filepath).substr(string(filepath).find_last_of("/") + 1).c_str(), channelnames, 4); } fprintf(xmf, " \n"); @@ -479,3 +779,4 @@ H5FieldDump::~H5FieldDump() } bool H5FieldDump::directory_exists = false; + diff --git a/mpi-dpd/io.h b/mpi-dpd/io.h index a970f5983..bc3d58ddb 100644 --- a/mpi-dpd/io.h +++ b/mpi-dpd/io.h @@ -16,13 +16,17 @@ #include "common.h" - void xyz_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, const char * particlename, Particle * particles, int n, bool append); void ply_dump(MPI_Comm comm, MPI_Comm cartcomm, const char * filename, int (*mesh_indices)[3], const int ninstances, const int ntriangles_per_instance, Particle * _particles, int nvertices_per_instance, bool append); +void stress_dump(MPI_Comm comm, const char * filename, const int nparticles, + const Particle * const particles, + const float * const stress_xx, const float * const stress_xy, const float * const stress_xz, + const float * const stress_yy, const float * const stress_yz, const float * const stress_zz); + class H5PartDump { float origin[3]; @@ -52,10 +56,6 @@ class H5FieldDump int last_idtimestep, globalsize[3]; MPI_Comm cartcomm; - - void _write_fields(const char * const path2h5, - const float * const channeldata[], const char * const * const channelnames, const int nchannels, - MPI_Comm comm, const float time); void _xdmf_header(FILE * xmf); void _xdmf_grid(FILE * xmf, float time, const char * const h5path, const char * const * channelnames, int nchannels); @@ -63,6 +63,10 @@ class H5FieldDump public: + void _write_fields(const char * const path2h5, + const float * const channeldata[], const char * const * const channelnames, const int nchannels, + MPI_Comm comm, const float time); + H5FieldDump(MPI_Comm cartcomm); void dump(MPI_Comm comm, const Particle * const p, const int n, int idtimestep); diff --git a/mpi-dpd/main.cu b/mpi-dpd/main.cu index 3e5a7e5f4..078aa9c68 100644 --- a/mpi-dpd/main.cu +++ b/mpi-dpd/main.cu @@ -21,11 +21,13 @@ #include "argument-parser.h" #include "simulation.h" +#include "dumper.h" bool currently_profiling = false; -float tend; -bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces; -int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; +float tend, couette; +bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, + hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces, stress; +int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop, nsubsteps; LocalComm localcomm; @@ -84,12 +86,9 @@ int main(int argc, char ** argv) nvtxstop = argp("-nvtxstop").asInt(10500); adjust_message_sizes = argp("-adjust_message_sizes").asBool(false); contactforces = argp("-contactforces").asBool(false); - -#ifndef _NO_DUMPS_ - const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(true); -#else - const bool mpi_thread_safe = argp("-mpi_thread_safe").asBool(false); -#endif + stress = argp("-stress").asBool(false); + couette = argp("-couette").asDouble(0); + nsubsteps = argp("-nsubsteps").asInt(0); SignalHandling::setup(); @@ -117,40 +116,27 @@ int main(int argc, char ** argv) int nranks, rank; - if (mpi_thread_safe) - { - //needed for the asynchronous data dumps - setenv("MPICH_MAX_THREAD_SAFETY", "multiple", 0); - - int provided_safety_level; - MPI_CHECK( MPI_Init_thread(&argc, &argv, MPI_THREAD_MULTIPLE, &provided_safety_level)); + MPI_CHECK(MPI_Init(&argc, &argv)); MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); - if (provided_safety_level != MPI_THREAD_MULTIPLE) - { - if (rank == 0) - printf("ooooooooops MPI thread safety level is just %d. Aborting now.\n", provided_safety_level); - abort(); - } + MPI_Comm iocomm, activecomm, intercomm, splitcomm; + + assert(nranks & 0x1 == 0); + int computeTask = (rank+1) % 2; + MPI_CHECK( MPI_Comm_split(MPI_COMM_WORLD, computeTask, rank, &splitcomm) ); + if (computeTask) + MPI_CHECK( MPI_Comm_dup(splitcomm, &activecomm) ); else - if (rank == 0) - printf("I have set MPICH_MAX_THREAD_SAFETY=multiple\n"); - } - else - { - MPI_CHECK(MPI_Init(&argc, &argv)); - MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); - MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); + MPI_CHECK( MPI_Comm_dup(splitcomm, &iocomm) ); - const char * env_thread_safety = getenv("MPICH_MAX_THREAD_SAFETY"); + if (computeTask) + MPI_CHECK( MPI_Intercomm_create(activecomm, 0, MPI_COMM_WORLD, 1, 0, &intercomm) ); + else + MPI_CHECK( MPI_Intercomm_create(iocomm, 0, MPI_COMM_WORLD, 0, 0, &intercomm) ); - if (rank == 0 && env_thread_safety) - printf("I read MPICH_MAX_THREAD_SAFETY=%s", env_thread_safety); - } - MPI_Comm activecomm = MPI_COMM_WORLD; #if defined(CUSTOM_REORDERING) activecomm = setup_reorder_comm(MPI_COMM_WORLD, rank, nranks); #endif @@ -160,7 +146,7 @@ int main(int argc, char ** argv) const char * env_reorder = getenv("MPICH_RANK_REORDER_METHOD"); //reordering of the ranks according to the computational domain and environment variables - if (atoi(env_reorder ? env_reorder : "-1") == atoi("3")) + if (computeTask && atoi(env_reorder ? env_reorder : "-1") == atoi("3")) { reordering = false; @@ -182,18 +168,22 @@ int main(int argc, char ** argv) return 0; } - MPI_CHECK(MPI_Barrier(activecomm)); + MPI_CHECK(MPI_Barrier(MPI_COMM_WORLD)); } - MPI_Comm cartcomm; + MPI_Comm cartcomm, iocartcomm; int periods[] = {1, 1, 1}; + if (computeTask) MPI_CHECK( MPI_Cart_create(activecomm, 3, ranks, periods, (int)reordering, &cartcomm) ); + else + MPI_CHECK( MPI_Cart_create(iocomm, 3, ranks, periods, (int)reordering, &iocartcomm) ); activecomm = cartcomm; //print the rank-to-node mapping + if (computeTask) { char name[1024]; int len; @@ -219,6 +209,8 @@ int main(int argc, char ** argv) //RAII { + if (computeTask) + { MPI_CHECK(MPI_Barrier(activecomm)); if (rank == 0) @@ -231,21 +223,38 @@ int main(int argc, char ** argv) MPI_CHECK(MPI_Barrier(activecomm)); - Simulation simulation(cartcomm, activecomm, SignalHandling::check_termination_request); - + Simulation simulation(cartcomm, activecomm, intercomm, SignalHandling::check_termination_request); simulation.run(); } + else + { + Dumper dumper(iocomm, iocartcomm, intercomm); + dumper.do_dump(); + } + } + if (computeTask) + { if (activecomm != cartcomm) MPI_CHECK(MPI_Comm_free(&activecomm)); MPI_CHECK(MPI_Comm_free(&cartcomm)); + MPI_CHECK(MPI_Comm_free(&intercomm)); + } + else + { + MPI_CHECK(MPI_Comm_free(&iocomm)); + MPI_CHECK(MPI_Comm_free(&intercomm)); + } MPI_CHECK(MPI_Finalize()); + if (computeTask) + { CUDA_CHECK(cudaDeviceSynchronize()); CUDA_CHECK(cudaDeviceReset()); + } return 0; } diff --git a/mpi-dpd/redistribute-particles.cu b/mpi-dpd/redistribute-particles.cu index e31ec5d3b..e4a88480c 100644 --- a/mpi-dpd/redistribute-particles.cu +++ b/mpi-dpd/redistribute-particles.cu @@ -744,7 +744,7 @@ void RedistributeParticles::bulk(const int nparticles, int * const cellstarts, i subindices.resize(nparticles); if (nparticles) - subindex_local<<< (nparticles + 127) / 128, 128, 0, mystream>>> + subindex_local<<< (nparticles + 127) / 128, 128, 0, mystream>>> (nparticles, RedistributeParticlesKernels::texparticledata, cellcounts, subindices.data); /* #ifndef NDEBUG diff --git a/mpi-dpd/redistribute-particles.h b/mpi-dpd/redistribute-particles.h index f7e281b41..48f5b01ce 100644 --- a/mpi-dpd/redistribute-particles.h +++ b/mpi-dpd/redistribute-particles.h @@ -79,7 +79,7 @@ class RedistributeParticles const double tstart = MPI_Wtime(); MPI_Status statuses[n]; - MPI_CHECK( MPI_Waitall(n, reqs, statuses) ); + MPI_CHECK( MPI_Waitall(n, reqs, statuses) ); return MPI_Wtime() - tstart; } diff --git a/mpi-dpd/redistribute-rbcs.cu b/mpi-dpd/redistribute-rbcs.cu index cb3998844..2129112d5 100644 --- a/mpi-dpd/redistribute-rbcs.cu +++ b/mpi-dpd/redistribute-rbcs.cu @@ -109,8 +109,8 @@ namespace ReorderingRBC ddestinations[idrbc][offset] = val; } - SimpleDeviceBuffer _ddestinations; - SimpleDeviceBuffer _dsources; + SimpleDeviceBuffer *_ddestinations = new SimpleDeviceBuffer; + SimpleDeviceBuffer *_dsources = new SimpleDeviceBuffer; void pack_all(cudaStream_t stream, const int nrbcs, const int nvertices, const float ** const sources, float ** const destinations) { @@ -128,13 +128,13 @@ namespace ReorderingRBC } else { - _ddestinations.resize(nrbcs); - _dsources.resize(nrbcs); + _ddestinations->resize(nrbcs); + _dsources->resize(nrbcs); - CUDA_CHECK(cudaMemcpyAsync(_ddestinations.data, destinations, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); - CUDA_CHECK(cudaMemcpyAsync(_dsources.data, sources, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemcpyAsync(_ddestinations->data, destinations, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); + CUDA_CHECK(cudaMemcpyAsync(_dsources->data, sources, sizeof(float *) * nrbcs, cudaMemcpyHostToDevice, stream)); - pack_all_kernel<<<(nthreads + 127) / 128, 128, 0, stream>>>(nrbcs, nvertices, _dsources.data, _ddestinations.data); + pack_all_kernel<<<(nthreads + 127) / 128, 128, 0, stream>>>(nrbcs, nvertices, _dsources->data, _ddestinations->data); } CUDA_CHECK(cudaPeekAtLastError()); diff --git a/mpi-dpd/simulation.cu b/mpi-dpd/simulation.cu index 4ecf0f4a5..a88e703b1 100644 --- a/mpi-dpd/simulation.cu +++ b/mpi-dpd/simulation.cu @@ -25,25 +25,25 @@ __global__ void make_texture( float4 * __restrict xyzouvwo, ushort4 * __restrict const float2 * base = ( float2* )( xyzuvw + i * 6 ); #pragma unroll 3 for( uint j = lane; j < 96; j += 32 ) { - float2 u = base[j]; - // NVCC bug: no operator = between volatile float2 and float2 - asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); + float2 u = base[j]; + // NVCC bug: no operator = between volatile float2 and float2 + asm volatile( "st.volatile.shared.v2.f32 [%0], {%1, %2};" : : "r"( ( warpid * 96 + j )*8 ), "f"( u.x ), "f"( u.y ) : "memory" ); } // SMEM: XYZUVW XYZUVW ... uint pid = lane / 2; const uint x_or_v = ( lane % 2 ) * 3; xyzouvwo[ i * 2 + lane ] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); pid += 16; xyzouvwo[ i * 2 + lane + 32] = make_float4( smem[ warpid * 192 + pid * 6 + x_or_v + 0 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], - smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); + smem[ warpid * 192 + pid * 6 + x_or_v + 1 ], + smem[ warpid * 192 + pid * 6 + x_or_v + 2 ], 0 ); xyzo_half[i + lane] = make_ushort4( __float2half_rn( smem[ warpid * 192 + lane * 6 + 0 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), - __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); -// } + __float2half_rn( smem[ warpid * 192 + lane * 6 + 1 ] ), + __float2half_rn( smem[ warpid * 192 + lane * 6 + 2 ] ), 0 ); + // } } void Simulation::_update_helper_arrays() @@ -56,7 +56,7 @@ void Simulation::_update_helper_arrays() xyzo_half.resize(np); if (np) - make_texture <<< (np + 1023) / 1024, 1024, 1024 * 6 * sizeof( float )>>>(xyzouvwo.data, xyzo_half.data, (float *)particles->xyzuvw.data, np ); + make_texture <<< (np + 1023) / 1024, 1024, 1024 * 6 * sizeof( float )>>>(xyzouvwo.data, xyzo_half.data, (float *)particles->xyzuvw.data, np ); CUDA_CHECK(cudaPeekAtLastError()); } @@ -70,19 +70,19 @@ std::vector Simulation::_ic() const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; for(int iz = 0; iz < L[2]; iz++) - for(int iy = 0; iy < L[1]; iy++) - for(int ix = 0; ix < L[0]; ix++) - for(int l = 0; l < numberdensity; ++l) - { - const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); - - ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); - ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); - ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); - ic[p].u[0] = 0; - ic[p].u[1] = 0; - ic[p].u[2] = 0; - } + for(int iy = 0; iy < L[1]; iy++) + for(int ix = 0; ix < L[0]; ix++) + for(int l = 0; l < numberdensity; ++l) + { + const int p = l + numberdensity * (ix + L[0] * (iy + L[1] * iz)); + + ic[p].x[0] = -L[0]/2 + ix + 0.99 * drand48(); + ic[p].x[1] = -L[1]/2 + iy + 0.99 * drand48(); + ic[p].x[2] = -L[2]/2 + iz + 0.99 * drand48(); + ic[p].u[0] = 0; + ic[p].u[1] = 0; + ic[p].u[2] = 0; + } /* use this to check robustness for(int i = 0; i < ic.size(); ++i) @@ -91,7 +91,7 @@ std::vector Simulation::_ic() ic[i].x[c] = -L[c] * 0.5 + drand48() * L[c]; ic[i].u[c] = 0; } - */ + */ return ic; } @@ -105,18 +105,18 @@ void Simulation::_redistribute() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.send(); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); redistribute.bulk(particles->size, cells.start, cells.count, mainstream); @@ -126,17 +126,17 @@ void Simulation::_redistribute() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -149,10 +149,10 @@ void Simulation::_redistribute() swap(particles, newparticles); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -166,61 +166,61 @@ void Simulation::_report(const bool verbose, const int idtimestep) report_host_memory_usage(activecomm, stdout); { - static double t0 = MPI_Wtime(), t1; + static double t0 = MPI_Wtime(), t1; - t1 = MPI_Wtime(); + t1 = MPI_Wtime(); - float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; + float host_busy_time = (MPI_Wtime() - t0) - host_idle_time; - host_busy_time *= 1e3 / steps_per_report; + host_busy_time *= 1e3 / steps_per_report; - float sumval, maxval, minval; - MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); - MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); + float sumval, maxval, minval; + MPI_CHECK(MPI_Reduce(&host_busy_time, &sumval, 1, MPI_FLOAT, MPI_SUM, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &maxval, 1, MPI_FLOAT, MPI_MAX, 0, activecomm)); + MPI_CHECK(MPI_Reduce(&host_busy_time, &minval, 1, MPI_FLOAT, MPI_MIN, 0, activecomm)); - int commsize; - MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); + int commsize; + MPI_CHECK(MPI_Comm_size(activecomm, &commsize)); - const double imbalance = 100 * (maxval / sumval * commsize - 1); + const double imbalance = 100 * (maxval / sumval * commsize - 1); - if (verbose && imbalance >= 0) - printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", - imbalance , minval, sumval / commsize, maxval); + if (verbose && imbalance >= 0) + printf("\x1b[93moverall imbalance: %.f%%, host workload min/avg/max: %.2f/%.2f/%.2f ms\x1b[0m\n", + imbalance , minval, sumval / commsize, maxval); - localcomm.print_particles(particles->size); + localcomm.print_particles(particles->size); - host_idle_time = 0; - t0 = t1; + host_idle_time = 0; + t0 = t1; } { - static double t0 = MPI_Wtime(), t1; - - t1 = MPI_Wtime(); - - if (verbose) - { - printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); - printf("in more details, per time step:\n"); - double tt = 0; - for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) - { - printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); - tt += it->second; - it->second = 0; - } - printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); - } - - t0 = t1; + static double t0 = MPI_Wtime(), t1; + + t1 = MPI_Wtime(); + + if (verbose) + { + printf("\x1b[92mbeginning of time step %d (%.3f ms)\x1b[0m\n", idtimestep, (t1 - t0) * 1e3 / steps_per_report); + printf("in more details, per time step:\n"); + double tt = 0; + for(std::map::iterator it = timings.begin(); it != timings.end(); ++it) + { + printf("%s: %.3f ms\n", it->first.c_str(), it->second * 1e3 / steps_per_report); + tt += it->second; + it->second = 0; + } + printf("discrepancy: %.3f ms\n", ((t1 - t0) - tt) * 1e3 / steps_per_report); + } + + t0 = t1; } } void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) { if (!coll || !coll->count()) - return; + return; SimpleDeviceBuffer marks(coll->pcount()); @@ -235,13 +235,13 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) std::vector tokill; for(int i = 0; i < nbodies; ++i) { - bool valid = true; + bool valid = true; - for(int j = 0; j < nvertices && valid; ++j) - valid &= 0 == tmp[j + nvertices * i]; + for(int j = 0; j < nvertices && valid; ++j) + valid &= 0 == tmp[j + nvertices * i]; - if (!valid) - tokill.push_back(i); + if (!valid) + tokill.push_back(i); } coll->remove(&tokill.front(), tokill.size()); @@ -253,38 +253,38 @@ void Simulation::_remove_bodies_from_wall(CollectionRBC * coll) void Simulation::_create_walls(const bool verbose, bool & termination_request) { if (verbose) - printf("creation of the walls...\n"); + printf("creation of the walls...\n"); int nsurvived = 0; ExpectedMessageSizes new_sizes; - wall = new ComputeWall(cartcomm, particles->xyzuvw.data, particles->size, nsurvived, new_sizes, verbose); + wall = new ComputeWall(cartcomm, particles->xyzuvw.data, particles->size, nsurvived, new_sizes, couette); //adjust the message sizes if we're pushing the flow in x { - const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; - const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; - const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; - - for(int code = 0; code < 27; ++code) - { - const int d[3] = { - (code % 3) - 1, - ((code / 3) % 3) - 1, - ((code / 9) % 3) - 1 - }; - - const double IudotnI = - fabs(d[0] * xvelavg) + - fabs(d[1] * yvelavg) + - fabs(d[2] * zvelavg) ; - - const float factor = 1 + IudotnI * dt * 10 * numberdensity; - - //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", - //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); - - new_sizes.msgsizes[code] *= factor; - } + const double xvelavg = getenv("XVELAVG") ? atof(getenv("XVELAVG")) : pushtheflow; + const double yvelavg = getenv("YVELAVG") ? atof(getenv("YVELAVG")) : 0; + const double zvelavg = getenv("ZVELAVG") ? atof(getenv("ZVELAVG")) : 0; + + for(int code = 0; code < 27; ++code) + { + const int d[3] = { + (code % 3) - 1, + ((code / 3) % 3) - 1, + ((code / 9) % 3) - 1 + }; + + const double IudotnI = + fabs(d[0] * xvelavg) + + fabs(d[1] * yvelavg) + + fabs(d[2] * zvelavg) ; + + const float factor = 1 + IudotnI * dt * 10 * numberdensity; + + //printf("RANK %d: direction %d %d %d -> IudotnI is %f and final factor is %f\n", + //rank, d[0], d[1], d[2], IudotnI, 1 + IudotnI * dt * numberdensity); + + new_sizes.msgsizes[code] *= factor; + } } //MPI_CHECK(MPI_Barrier(activecomm)); @@ -316,7 +316,7 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) return; } } - */ + */ particles->resize(nsurvived); particles->clear_velocity(); @@ -331,18 +331,18 @@ void Simulation::_create_walls(const bool verbose, bool & termination_request) _remove_bodies_from_wall(ctcscoll); { - H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); - Particle * p = new Particle[particles->size]; + H5PartDump sd("survived-particles->h5part", activecomm, cartcomm); + Particle * p = new Particle[particles->size]; - CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); + CUDA_CHECK(cudaMemcpy(p, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost)); - sd.dump(p, particles->size); + sd.dump(p, particles->size); - delete [] p; + delete [] p; } } -void Simulation::_forces() +void Simulation::_forces(bool firsttime) { double tstart = MPI_Wtime(); @@ -351,10 +351,10 @@ void Simulation::_forces() std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -363,10 +363,10 @@ void Simulation::_forces() particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); @@ -374,11 +374,8 @@ void Simulation::_forces() CUDA_CHECK(cudaPeekAtLastError()); - if (contactforces) - contact.build_cells(wsolutes, mainstream); - dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + cells.start, cells.count, mainstream); dpd.post(particles->xyzuvw.data, particles->size, mainstream, downloadstream); @@ -387,37 +384,40 @@ void Simulation::_forces() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); dpd.recv(mainstream, uploadstream); - solutex.recv_p(uploadstream); + solutex.recv_p(uploadstream, mainstream); + + if (contactforces) + contact.attach_bulk(wsolutes); - solutex.halo(uploadstream, mainstream); + solutex.halo(uploadstream, mainstream, downloadstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); fsi.bulk(wsolutes, mainstream); - if (contactforces) - contact.bulk(wsolutes, mainstream); - CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (nsubsteps == 0) + { + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + } CUDA_CHECK(cudaPeekAtLastError()); @@ -425,6 +425,49 @@ void Simulation::_forces() solutex.recv_a(mainstream); + if (nsubsteps) + { // TSS + if (rbcscoll) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->fsiacc(), rbcscoll->acc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + if (ctcscoll) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->fsiacc(), ctcscoll->acc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + for (int sstep = 0; sstep < nsubsteps; sstep++) + { + // Start with acc induced by solvent + if (rbcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->acc(), rbcscoll->fsiacc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + + if (firsttime) + rbcscoll->update_stage1(0.0, mainstream, (dt / nsubsteps)); + else + rbcscoll->update_stage2_and_1(0.0, mainstream, (dt / nsubsteps)); + + if (wall) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + + if (ctcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->acc(), ctcscoll->fsiacc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + + if (firsttime) + ctcscoll->update_stage2_and_1(0.0, mainstream, dt / (nsubsteps)); + else + ctcscoll->update_stage1(0.0, mainstream, dt / (nsubsteps)); + + if (wall) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + } + } + timings["interactions"] += MPI_Wtime() - tstart; CUDA_CHECK(cudaPeekAtLastError()); @@ -434,243 +477,128 @@ void Simulation::_datadump(const int idtimestep) { double tstart = MPI_Wtime(); - pthread_mutex_lock(&mutex_datadump); + int n = particles->size; - while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); + if (stress) + { + for(int c = 0; c < 6; ++c) + stresses_datadump[c].resize(n); - int n = particles->size; + for(int c = 0; c < 6; ++c) + CUDA_CHECK(cudaMemcpyAsync(stresses_datadump[c].data, stresses[c].data, sizeof(float) * n, cudaMemcpyDeviceToHost,0)); + } if (rbcscoll) - n += rbcscoll->pcount(); + n += rbcscoll->pcount(); if (ctcscoll) - n += ctcscoll->pcount(); + n += ctcscoll->pcount(); particles_datadump.resize(n); accelerations_datadump.resize(n); CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data, particles->xyzuvw.data, sizeof(Particle) * particles->size, cudaMemcpyDeviceToHost,0)); CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data, particles->axayaz.data, sizeof(Acceleration) * particles->size, cudaMemcpyDeviceToHost,0)); + vector& avgVels = velsampler->getAvgVel(0); + + if (nsubsteps > 0) + { + CUDA_CHECK( cudaStreamSynchronize(0) ); + for (int i=0; isize; i++) + for (int c=0; c<3; c++) + particles_datadump.data[i].u[c] += dt * accelerations_datadump.data[i].a[c]; + } int start = particles->size; if (rbcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, rbcscoll->xyzuvw.data, sizeof(Particle) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, rbcscoll->axayaz.data, sizeof(Acceleration) * rbcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += rbcscoll->pcount(); + start += rbcscoll->pcount(); } if (ctcscoll) { - CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(particles_datadump.data + start, ctcscoll->xyzuvw.data, sizeof(Particle) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); + CUDA_CHECK(cudaMemcpyAsync(accelerations_datadump.data + start, ctcscoll->axayaz.data, sizeof(Acceleration) * ctcscoll->pcount(), cudaMemcpyDeviceToHost, 0)); - start += ctcscoll->pcount(); + start += ctcscoll->pcount(); } assert(start == n); - CUDA_CHECK(cudaEventRecord(evdownloaded, 0)); - datadump_idtimestep = idtimestep; datadump_nsolvent = particles->size; datadump_nrbcs = rbcscoll ? rbcscoll->pcount() : 0; datadump_nctcs = ctcscoll ? ctcscoll->pcount() : 0; - datadump_pending = true; - - pthread_cond_signal(&request_datadump); -#if defined(_SYNC_DUMPS_) - while (datadump_pending) - pthread_cond_wait(&done_datadump, &mutex_datadump); -#endif - - pthread_mutex_unlock(&mutex_datadump); - - timings["data-dump"] += MPI_Wtime() - tstart; -} - -void Simulation::_datadump_async() -{ -#ifdef _USE_NVTX_ - nvtxNameOsThread(pthread_self(), "DATADUMP_THREAD"); -#endif - - int iddatadump = 0, rank; - int curr_idtimestep = -1; - bool wallcreated = false; - - MPI_Comm myactivecomm, mycartcomm; - - MPI_CHECK(MPI_Comm_dup(activecomm, &myactivecomm) ); - MPI_CHECK(MPI_Comm_dup(cartcomm, &mycartcomm) ); - - H5PartDump dump_part("allparticles->h5part", activecomm, cartcomm), *dump_part_solvent = NULL; - H5FieldDump dump_field(cartcomm); - - MPI_CHECK(MPI_Comm_rank(myactivecomm, &rank)); - - if (rank == 0) - mkdir("xyz", S_IRWXU | S_IRWXG | S_IROTH | S_IXOTH); - - MPI_CHECK(MPI_Barrier(myactivecomm)); - - while (true) - { - pthread_mutex_lock(&mutex_datadump); - async_thread_initialized = 1; - - while (!datadump_pending) - pthread_cond_wait(&request_datadump, &mutex_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - if (curr_idtimestep == datadump_idtimestep) - if (simulation_is_done) - break; - - CUDA_CHECK(cudaEventSynchronize(evdownloaded)); - - const int n = particles_datadump.size; - Particle * p = particles_datadump.data; - Acceleration * a = accelerations_datadump.data; - - { - NVTX_RANGE("diagnostics", NVTX_C1); - diagnostics(myactivecomm, mycartcomm, p, n, dt, datadump_idtimestep, a); - } - - if (xyz_dumps) - { - NVTX_RANGE("xyz dump", NVTX_C2); - - if (walls && datadump_idtimestep >= wall_creation_stepid && !wallcreated) - { - if (rank == 0) - { - if( access("xyz/particles-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/particles.xyz", "xyz/particles-equilibration.xyz"); - - if( access( "xyz/rbcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/rbcs.xyz", "xyz/rbcs-equilibration.xyz"); - - if( access( "xyz/ctcs-equilibration.xyz", F_OK ) == -1 ) - rename ("xyz/ctcs.xyz", "xyz/ctcs-equilibration.xyz"); - } - - MPI_CHECK(MPI_Barrier(myactivecomm)); - - wallcreated = true; - } - - xyz_dump(myactivecomm, mycartcomm, "xyz/particles->xyz", "all-particles", p, n, datadump_idtimestep > 0); - } - - if (hdf5part_dumps) - { - NVTX_RANGE("h5part dump", NVTX_C3); - - if (!dump_part_solvent && walls && datadump_idtimestep >= wall_creation_stepid) - { - dump_part.close(); - - dump_part_solvent = new H5PartDump("solvent-particles->h5part", activecomm, cartcomm); - } - - if (dump_part_solvent) - dump_part_solvent->dump(p, n); - else - dump_part.dump(p, n); - } - - if (hdf5field_dumps) - { - NVTX_RANGE("hdf5 field dump", NVTX_C4); - dump_field.dump(activecomm, p, datadump_nsolvent, datadump_idtimestep); - } + MPI_CHECK( MPI_Send(&datadump_nsolvent, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nrbcs, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); - { - NVTX_RANGE("ply dump", NVTX_C5); + CUDA_CHECK(cudaEventSynchronize(evdownloaded)); + CUDA_CHECK( cudaStreamSynchronize(0) ); - if (rbcscoll) - CollectionRBC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent, a + datadump_nsolvent, datadump_nrbcs, iddatadump); + MPI_CHECK( MPI_Send(particles_datadump.data, n, Particle::datatype(), rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(accelerations_datadump.data, n, Acceleration::datatype(), rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&avgVels[0], avgVels.size()*3, MPI_FLOAT, rank, 0, intercomm) ); - if (ctcscoll) - CollectionCTC::dump(myactivecomm, mycartcomm, p + datadump_nsolvent + datadump_nrbcs, - a + datadump_nsolvent + datadump_nrbcs, datadump_nctcs, iddatadump); - } - - curr_idtimestep = datadump_idtimestep; - - pthread_mutex_lock(&mutex_datadump); - - if (simulation_is_done) - { - pthread_mutex_unlock(&mutex_datadump); - break; - } - - datadump_pending = false; - - pthread_cond_signal(&done_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - ++iddatadump; - } - - if (dump_part_solvent) - delete dump_part_solvent; - - CUDA_CHECK(cudaEventDestroy(evdownloaded)); + timings["data-dump"] += MPI_Wtime() - tstart; } void Simulation::_update_and_bounce() { double tstart = MPI_Wtime(); + + velcontrol->push(cells.start, particles->xyzuvw.data, particles->axayaz.data, mainstream); particles->update_stage2_and_1(driving_acceleration, mainstream); CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - rbcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage2_and_1(0.0f, mainstream); - CUDA_CHECK(cudaPeekAtLastError()); + CUDA_CHECK(cudaPeekAtLastError()); - if (ctcscoll) - ctcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage2_and_1(0.0f, mainstream); + } timings["update"] += MPI_Wtime() - tstart; if (wall) { - tstart = MPI_Wtime(); - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + tstart = MPI_Wtime(); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); - if (rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + } - timings["bounce-walls"] += MPI_Wtime() - tstart; + timings["bounce-walls"] += MPI_Wtime() - tstart; } CUDA_CHECK(cudaPeekAtLastError()); } -Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_termination)()) : - cartcomm(cartcomm), activecomm(activecomm), - /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), - rbcscoll(NULL), ctcscoll(NULL), wall(NULL), - redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), - dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), - check_termination(check_termination), - driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), - datadump_pending(false), simulation_is_done(false) +Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) : + cartcomm(cartcomm), activecomm(activecomm), intercomm(intercomm), + /*particles(_ic()),*/ cells(XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN), + rbcscoll(NULL), ctcscoll(NULL), wall(NULL), + redistribute(cartcomm), redistribute_rbcs(cartcomm), redistribute_ctcs(cartcomm), + dpd(cartcomm), fsi(cartcomm), contact(cartcomm), solutex(cartcomm), + check_termination(check_termination), + driving_acceleration(0), host_idle_time(0), nsteps((int)(tend / dt)), + datadump_pending(false), simulation_is_done(false) { MPI_CHECK( MPI_Comm_size(activecomm, &nranks) ); MPI_CHECK( MPI_Comm_rank(activecomm, &rank) ); @@ -678,36 +606,35 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term solutex.attach_halocomputation(fsi); if (contactforces) - solutex.attach_halocomputation(contact); - //localcomm.initialize(activecomm); + solutex.attach_halocomputation(contact); int dims[3], periods[3], coords[3]; MPI_CHECK( MPI_Cart_get(cartcomm, 3, dims, periods, coords) ); { - particles = &particles_pingpong[0]; - newparticles = &particles_pingpong[1]; + particles = &particles_pingpong[0]; + newparticles = &particles_pingpong[1]; - vector ic = _ic(); + vector ic = _ic(); - for(int c = 0; c < 2; ++c) - { - particles_pingpong[c].resize(ic.size()); + for(int c = 0; c < 2; ++c) + { + particles_pingpong[c].resize(ic.size()); - particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, - (0.5 + coords[1]) * YSIZE_SUBDOMAIN, - (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); + particles_pingpong[c].origin = make_float3((0.5 + coords[0]) * XSIZE_SUBDOMAIN, + (0.5 + coords[1]) * YSIZE_SUBDOMAIN, + (0.5 + coords[2]) * ZSIZE_SUBDOMAIN); - particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, - dims[1] * YSIZE_SUBDOMAIN, - dims[2] * ZSIZE_SUBDOMAIN); - } + particles_pingpong[c].globalextent = make_float3(dims[0] * XSIZE_SUBDOMAIN, + dims[1] * YSIZE_SUBDOMAIN, + dims[2] * ZSIZE_SUBDOMAIN); + } - CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); + CUDA_CHECK(cudaMemcpy(particles->xyzuvw.data, &ic.front(), sizeof(Particle) * ic.size(), cudaMemcpyHostToDevice)); - cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); + cells.build(particles->xyzuvw.data, particles->size, 0, NULL, NULL); - _update_helper_arrays(); + _update_helper_arrays(); } CUDA_CHECK(cudaStreamCreate(&mainstream)); @@ -716,47 +643,24 @@ Simulation::Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_term if (rbcs) { - rbcscoll = new CollectionRBC(cartcomm); - rbcscoll->setup("rbcs-ic.txt"); + rbcscoll = new CollectionRBC(cartcomm); + rbcscoll->setup("rbcs-ic.txt"); } if (ctcs) { - ctcscoll = new CollectionCTC(cartcomm); - ctcscoll->setup("ctcs-ic.txt"); + ctcscoll = new CollectionCTC(cartcomm); + ctcscoll->setup("ctcs-ic.txt"); } -#ifndef _NO_DUMPS_ - //setting up the asynchronous data dumps - { - CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); - - particles_datadump.resize(particles->size * 1.5); - accelerations_datadump.resize(particles->size * 1.5); - - int rc = pthread_mutex_init(&mutex_datadump, NULL); - rc |= pthread_cond_init(&done_datadump, NULL); - rc |= pthread_cond_init(&request_datadump, NULL); - async_thread_initialized = 0; - rc |= pthread_create(&thread_datadump, NULL, datadump_trampoline, this); - - while (1) - { - pthread_mutex_lock(&mutex_datadump); - int done = async_thread_initialized; - pthread_mutex_unlock(&mutex_datadump); - - if (done) - break; - } - - if (rc) - { - printf("ERROR; return code from pthread_create() is %d\n", rc); - exit(-1); - } - } -#endif + CUDA_CHECK(cudaEventCreate(&evdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); + particles_datadump.resize(particles->size * 1.5); + accelerations_datadump.resize(particles->size * 1.5); + + int xl[3] = {0, 0, 0}; + int xh[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + velcontrol = new VelController(xl, xh, coords, make_float3(2, 0, 0), activecomm); + velsampler = new VelSampler(); } void Simulation::_lockstep() @@ -768,10 +672,10 @@ void Simulation::_lockstep() std::vector wsolutes; if (rbcscoll) - wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); + wsolutes.push_back(ParticlesWrap(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc())); if (ctcscoll) - wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); + wsolutes.push_back(ParticlesWrap(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc())); fsi.bind_solvent(wsolvent); @@ -780,20 +684,17 @@ void Simulation::_lockstep() particles->clear_acc(mainstream); if (rbcscoll) - rbcscoll->clear_acc(mainstream); + rbcscoll->clear_acc(mainstream); if (ctcscoll) - ctcscoll->clear_acc(mainstream); + ctcscoll->clear_acc(mainstream); solutex.pack_p(mainstream); dpd.pack(particles->xyzuvw.data, particles->size, cells.start, cells.count, mainstream); dpd.local_interactions(particles->xyzuvw.data, xyzouvwo.data, xyzo_half.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); - - if (contactforces) - contact.build_cells(wsolutes, mainstream); + cells.start, cells.count, mainstream); solutex.post_p(mainstream, downloadstream); @@ -802,40 +703,43 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (wall) - wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, - cells.start, cells.count, mainstream); + wall->interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, + cells.start, cells.count, mainstream); CUDA_CHECK(cudaPeekAtLastError()); dpd.recv(mainstream, uploadstream); - solutex.recv_p(uploadstream); + solutex.recv_p(uploadstream, mainstream); + + if (contactforces) + contact.attach_bulk(wsolutes); - solutex.halo(uploadstream, mainstream); + solutex.halo(uploadstream, mainstream, downloadstream); dpd.remote_interactions(particles->xyzuvw.data, particles->size, particles->axayaz.data, mainstream, uploadstream); fsi.bulk(wsolutes, mainstream); - if (contactforces) - contact.bulk(wsolutes, mainstream); - CUDA_CHECK(cudaPeekAtLastError()); - if (rbcscoll) - CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); - - if (ctcscoll) - CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + if (nsubsteps == 0) + { + if (rbcscoll) + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + if (ctcscoll) + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + } CUDA_CHECK(cudaPeekAtLastError()); solutex.post_a(); + velcontrol->push(cells.start, particles->xyzuvw.data, particles->axayaz.data, mainstream); particles->update_stage2_and_1(driving_acceleration, mainstream); if (wall) - wall->bounce(particles->xyzuvw.data, particles->size, mainstream); + wall->bounce(particles->xyzuvw.data, particles->size, mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -848,42 +752,81 @@ void Simulation::_lockstep() CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll && wall) - wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(rbcscoll->data(), rbcscoll->pcount(), rbcscoll->acc(), NULL, NULL, mainstream); if (ctcscoll && wall) - wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); + wall->interactions(ctcscoll->data(), ctcscoll->pcount(), ctcscoll->acc(), NULL, NULL, mainstream); CUDA_CHECK(cudaPeekAtLastError()); solutex.recv_a(mainstream); - if (rbcscoll) - rbcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage2_and_1(0.0f, mainstream); - if (ctcscoll) - ctcscoll->update_stage2_and_1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage2_and_1(0.0f, mainstream); - if (wall && rbcscoll) - wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); + if (wall && rbcscoll) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream); - if (wall && ctcscoll) - wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + if (wall && ctcscoll) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream); + } + else + { // TSS + if (rbcscoll) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->fsiacc(), rbcscoll->acc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + if (ctcscoll) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->fsiacc(), ctcscoll->acc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + + for (int sstep = 0; sstep < nsubsteps; sstep++) + { + // Start with acc induced by solvent + if (rbcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(rbcscoll->acc(), rbcscoll->fsiacc(), 3*rbcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaRBC::forces_nohost(mainstream, rbcscoll->count(), (float *)rbcscoll->data(), (float *)rbcscoll->acc()); + + rbcscoll->update_stage2_and_1(0.0, mainstream, (dt / nsubsteps)); + + if (wall) + wall->bounce(rbcscoll->data(), rbcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + + if (ctcscoll) + { + if (sstep > 0) + CUDA_CHECK( cudaMemcpyAsync(ctcscoll->acc(), ctcscoll->fsiacc(), 3*ctcscoll->pcount()*sizeof(float), cudaMemcpyDeviceToDevice, mainstream) ); + CudaCTC::forces_nohost(mainstream, ctcscoll->count(), (float *)ctcscoll->data(), (float *)ctcscoll->acc()); + + ctcscoll->update_stage1(0.0, mainstream, dt / (nsubsteps)); + + if (wall) + wall->bounce(ctcscoll->data(), ctcscoll->pcount(), mainstream, dt / (nsubsteps)); + } + } + } const int newnp = redistribute.recv_count(mainstream, host_idle_time); CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.extent(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.extent(ctcscoll->data(), ctcscoll->count(), mainstream); if (rbcscoll) - redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.pack_sendcount(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.pack_sendcount(ctcscoll->data(), ctcscoll->count(), mainstream); newparticles->resize(newnp); xyzouvwo.resize(newnp * 2); @@ -897,25 +840,25 @@ void Simulation::_lockstep() int nrbcs; if (rbcscoll) - nrbcs = redistribute_rbcs.post(); + nrbcs = redistribute_rbcs.post(); int nctcs; if (ctcscoll) - nctcs = redistribute_ctcs.post(); + nctcs = redistribute_ctcs.post(); if (rbcscoll) - rbcscoll->resize(nrbcs); + rbcscoll->resize(nrbcs); if (ctcscoll) - ctcscoll->resize(nctcs); + ctcscoll->resize(nctcs); CUDA_CHECK(cudaPeekAtLastError()); if (rbcscoll) - redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); + redistribute_rbcs.unpack(rbcscoll->data(), rbcscoll->count(), mainstream); if (ctcscoll) - redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); + redistribute_ctcs.unpack(ctcscoll->data(), ctcscoll->count(), mainstream); CUDA_CHECK(cudaPeekAtLastError()); @@ -926,112 +869,147 @@ void Simulation::_lockstep() void Simulation::run() { if (rank == 0 && !walls) - printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); + printf("the simulation begins now and it consists of %.3e steps\n", (double)nsteps); double time_simulation_start = MPI_Wtime(); _redistribute(); - _forces(); + _forces(nsubsteps > 0); if (!walls && pushtheflow) - driving_acceleration = hydrostatic_a; + driving_acceleration = hydrostatic_a; particles->update_stage1(driving_acceleration, mainstream); - if (rbcscoll) - rbcscoll->update_stage1(driving_acceleration, mainstream); + if (nsubsteps == 0) + { + if (rbcscoll) + rbcscoll->update_stage1(0.0f, mainstream); - if (ctcscoll) - ctcscoll->update_stage1(driving_acceleration, mainstream); + if (ctcscoll) + ctcscoll->update_stage1(0.0f, mainstream); + } int it; + FILE* fforce = fopen("frcdump.txt", "w"); for(it = 0; it < nsteps; ++it) { - const bool verbose = it > 0 && rank == 0; + const bool verbose = it > 0 && rank == 0; #ifdef _USE_NVTX_ - if (it == nvtxstart) - { - NvtxTracer::currently_profiling = true; - CUDA_CHECK(cudaProfilerStart()); - } - else if (it == nvtxstop) - { - CUDA_CHECK(cudaProfilerStop()); - NvtxTracer::currently_profiling = false; - CUDA_CHECK(cudaDeviceSynchronize()); - - if (rank == 0) - printf("profiling session ended. terminating the simulation now...\n"); - - break; - } + if (it == nvtxstart) + { + NvtxTracer::currently_profiling = true; + CUDA_CHECK(cudaProfilerStart()); + } + else if (it == nvtxstop) + { + CUDA_CHECK(cudaProfilerStop()); + NvtxTracer::currently_profiling = false; + CUDA_CHECK(cudaDeviceSynchronize()); + + if (rank == 0) + printf("profiling session ended. terminating the simulation now...\n"); + + break; + } #endif - if (it % steps_per_report == 0) - { - CUDA_CHECK(cudaStreamSynchronize(mainstream)); + if (it % steps_per_report == 0) + { + CUDA_CHECK(cudaStreamSynchronize(mainstream)); - if (simulation_is_done = check_termination()) - break; + if (simulation_is_done = check_termination()) + break; - _report(verbose, it); - } + _report(verbose, it); + } - _redistribute(); + _redistribute(); #if 1 - lockstep_check: + lockstep_check: + + const bool lockstep_OK = + !(walls && it >= wall_creation_stepid && wall == NULL) && + !(it % steps_per_dump == 0) && + !(it + 1 == nvtxstart) && + !(it + 1 == nvtxstop) && + !((it + 1) % steps_per_report == 0) && + !(it + 1 == nsteps); + + if (lockstep_OK) + { + _lockstep(); - const bool lockstep_OK = - !(walls && it >= wall_creation_stepid && wall == NULL) && - !(it % steps_per_dump == 0) && - !(it + 1 == nvtxstart) && - !(it + 1 == nvtxstop) && - !((it + 1) % steps_per_report == 0) && - !(it + 1 == nsteps); + if (it % 10 == 0) + velcontrol->sample(cells.start, particles->xyzuvw.data, mainstream); - if (lockstep_OK) - { - _lockstep(); + if (it % 5 == 0) + velsampler->sample(cells.start, particles->xyzuvw.data, mainstream); - ++it; + if (wall && it % 1000 == 0) + { + float3 f = velcontrol->adjustF(mainstream); + if (rank == 0) fprintf(fforce, "Force applied: %f\n", f.x); + fflush(fforce); + } - goto lockstep_check; - } + ++it; + + goto lockstep_check; + } #endif - if (walls && it >= wall_creation_stepid && wall == NULL) - { - CUDA_CHECK(cudaDeviceSynchronize()); + if (walls && it >= wall_creation_stepid && wall == NULL) + { + CUDA_CHECK(cudaDeviceSynchronize()); - bool termination_request = false; + bool termination_request = false; - _create_walls(verbose, termination_request); + _create_walls(verbose, termination_request); - _redistribute(); + _redistribute(); - if (termination_request) - break; + if (termination_request) + break; - time_simulation_start = MPI_Wtime(); + time_simulation_start = MPI_Wtime(); - if (pushtheflow) - driving_acceleration = hydrostatic_a; + if (pushtheflow) + driving_acceleration = 0; - if (rank == 0) - printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); - } + if (rank == 0) + printf("the simulation begins now and it consists of %.3e steps\n", (double)(nsteps - it)); + } - _forces(); + if(stress && it % steps_per_dump == 0) + { + for(int c = 0; c < 6; ++c) + stresses[c].resize(particles->size); -#ifndef _NO_DUMPS_ - if (it % steps_per_dump == 0) - _datadump(it); -#endif - _update_and_bounce(); + dpd.set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + + if (wall) + wall->set_stress_buffers(stresses[0].data, stresses[1].data, stresses[2].data, stresses[3].data, stresses[4].data, stresses[5].data); + } + + _forces(); + + if (stress && it % steps_per_dump == 0 ) + { + dpd.clr_stress_buffers(); + + if (wall) + wall->clr_stress_buffers(); + } + + if (it % steps_per_dump == 0) + _datadump(it); + + _update_and_bounce(); } const double time_simulation_stop = MPI_Wtime(); @@ -1039,40 +1017,34 @@ void Simulation::run() simulation_is_done = true; + datadump_nsolvent = datadump_nrbcs = datadump_nctcs = -1; + MPI_CHECK( MPI_Send(&datadump_nsolvent, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nrbcs, 1, MPI_INT, rank, 0, intercomm) ); + MPI_CHECK( MPI_Send(&datadump_nctcs, 1, MPI_INT, rank, 0, intercomm) ); + if (rank == 0) - if (it == nsteps) - printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", - telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); - else - if (it != wall_creation_stepid) - printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); + if (it == nsteps) + printf("simulation is done after %.2lf s (%dm%ds). Ciao.\n", + telapsed, (int)(telapsed / 60), (int)(telapsed) % 60); + else + if (it != wall_creation_stepid) + printf("external termination request (signal) after %.3e s. Bye.\n", telapsed); fflush(stdout); } Simulation::~Simulation() { -#ifndef _NO_DUMPS_ - pthread_mutex_lock(&mutex_datadump); - - datadump_pending = true; - pthread_cond_signal(&request_datadump); - - pthread_mutex_unlock(&mutex_datadump); - - pthread_join(thread_datadump, NULL); -#endif - CUDA_CHECK(cudaStreamDestroy(mainstream)); CUDA_CHECK(cudaStreamDestroy(uploadstream)); CUDA_CHECK(cudaStreamDestroy(downloadstream)); if (wall) - delete wall; + delete wall; if (rbcscoll) - delete rbcscoll; + delete rbcscoll; if (ctcscoll) - delete ctcscoll; + delete ctcscoll; } diff --git a/mpi-dpd/simulation.h b/mpi-dpd/simulation.h index 5a097d522..49925fac2 100644 --- a/mpi-dpd/simulation.h +++ b/mpi-dpd/simulation.h @@ -34,6 +34,8 @@ #include "redistribute-rbcs.h" #include "ctc.h" #include "io.h" +#include "velcontroller.h" +#include "velsampler.h" class Simulation { @@ -41,6 +43,7 @@ class Simulation ParticleArray * particles, * newparticles; SimpleDeviceBuffer xyzouvwo; SimpleDeviceBuffer xyzo_half; + SimpleDeviceBuffer stresses[6]; CellLists cells; CollectionRBC * rbcscoll; @@ -60,7 +63,7 @@ class Simulation bool (*check_termination)(); bool simulation_is_done; - MPI_Comm activecomm, cartcomm; + MPI_Comm activecomm, cartcomm, intercomm; //LocalComm localcomm; cudaStream_t mainstream, uploadstream, downloadstream; @@ -70,7 +73,7 @@ class Simulation const size_t nsteps; float driving_acceleration; float host_idle_time; - int nranks, rank; + int nranks, rank; std::vector _ic(); void _update_helper_arrays(); @@ -79,7 +82,7 @@ class Simulation void _report(const bool verbose, const int idtimestep); void _create_walls(const bool verbose, bool & termination_request); void _remove_bodies_from_wall(CollectionRBC * coll); - void _forces(); + void _forces(bool firsttime = false); void _datadump(const int idtimestep); void _update_and_bounce(); void _lockstep(); @@ -93,18 +96,20 @@ class Simulation PinnedHostBuffer particles_datadump; PinnedHostBuffer accelerations_datadump; + PinnedHostBuffer stresses_datadump[6]; cudaEvent_t evdownloaded; void _datadump_async(); + VelController* velcontrol; + VelSampler* velsampler; + public: - Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, bool (*check_termination)()) ; + Simulation(MPI_Comm cartcomm, MPI_Comm activecomm, MPI_Comm intercomm, bool (*check_termination)()) ; void run(); ~Simulation(); - - static void * datadump_trampoline(void * x) { ((Simulation *)x)->_datadump_async(); return NULL; } }; diff --git a/mpi-dpd/solute-exchange.cu b/mpi-dpd/solute-exchange.cu index e2d63c03d..627fdeb84 100644 --- a/mpi-dpd/solute-exchange.cu +++ b/mpi-dpd/solute-exchange.cu @@ -59,7 +59,8 @@ iterationcount(-1), packstotalstart(27), host_packstotalstart(27), host_packstot _adjust_packbuffers(); CUDA_CHECK(cudaEventCreateWithFlags(&evPpacked, cudaEventDisableTiming | cudaEventBlockingSync)); - CUDA_CHECK(cudaEventCreateWithFlags(&evAcomputed, cudaEventDisableTiming | cudaEventBlockingSync)); + CUDA_CHECK(cudaEventCreateWithFlags(&evAcomputed, cudaEventDisableTiming)); + CUDA_CHECK(cudaEventCreateWithFlags(&evAdownloaded, cudaEventDisableTiming | cudaEventBlockingSync)); CUDA_CHECK(cudaPeekAtLastError()); } @@ -306,7 +307,7 @@ void SoluteExchange::_pack_attempt(cudaStream_t stream) { const ParticlesWrap it = wsolutes[i]; - if (it.n) + if (it.n) { CUDA_CHECK(cudaMemcpyToSymbolAsync(SolutePUP::coffsets, packsoffset.data + 26 * i, sizeof(int) * 26, 0, cudaMemcpyDeviceToDevice, stream)); @@ -351,7 +352,7 @@ void SoluteExchange::pack_p(cudaStream_t stream) if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/pack", NVTX_C4); + NVTX_RANGE("SOLUTEX/pack", NVTX_C4); ++iterationcount; @@ -371,7 +372,7 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) //consolidate the packing { - NVTX_RANGE("FSI/consolidate", NVTX_C5); + NVTX_RANGE("SOLUTEX/consolidate", NVTX_C5); CUDA_CHECK(cudaEventSynchronize(evPpacked)); @@ -446,7 +447,7 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) //post the sending of the packs { - NVTX_RANGE("FSI/send", NVTX_C6); + NVTX_RANGE("SOLUTEX/send", NVTX_C6); reqsendC.resize(26); @@ -485,13 +486,13 @@ void SoluteExchange::post_p(cudaStream_t stream, cudaStream_t downloadstream) } } -void SoluteExchange::recv_p(cudaStream_t uploadstream) +void SoluteExchange::recv_p(cudaStream_t uploadstream, cudaStream_t computestream) { if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/recv-p", NVTX_C7); - + NVTX_RANGE("SOLUTEX/recv-p", NVTX_C7); + _wait(reqrecvC); _wait(reqrecvP); @@ -504,7 +505,6 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) remote[i].preserve_resize(count); #ifndef NDEBUG - CUDA_CHECK(cudaMemsetAsync(remote[i].dstate.data, 0xff, sizeof(Particle) * remote[i].dstate.capacity, uploadstream)); CUDA_CHECK(cudaMemsetAsync(remote[i].result.data, 0xff, sizeof(Acceleration) * remote[i].result.capacity, uploadstream)); #endif @@ -524,39 +524,71 @@ void SoluteExchange::recv_p(cudaStream_t uploadstream) } _postrecvC(); - - for(int i = 0; i < 26; ++i) - CUDA_CHECK(cudaMemcpyAsync(remote[i].dstate.data, remote[i].hstate.data, sizeof(Particle) * remote[i].hstate.size, - cudaMemcpyHostToDevice, uploadstream)); + + //collate halos + { + int c = 0; + for(int i = 0; i < 26; ++i) + c += remote[i].hstate.size; + +#ifndef NDEBUG + CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0xff, sizeof(Acceleration) * allremotehalosacc.capacity, computestream)); + CUDA_CHECK(cudaMemsetAsync(allremotehalos.data, 0xff, sizeof(Particle) * allremotehalos.capacity, uploadstream)); +#endif + + allremotehalos.resize(c); + allremotehalosacc.resize(c); + + CUDA_CHECK(cudaMemsetAsync(allremotehalosacc.data, 0, sizeof(Acceleration) * allremotehalosacc.size, computestream)); + + c = 0; + for(int i = 0; i < 26; ++i) + { + CUDA_CHECK(cudaMemcpyAsync(allremotehalos.data + c, remote[i].hstate.data, sizeof(Particle) * remote[i].hstate.size, cudaMemcpyHostToDevice, uploadstream)); + + c += remote[i].hstate.size; + } + } } -void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t stream) +void SoluteExchange::halo(cudaStream_t uploadstream, cudaStream_t computestream, cudaStream_t downloadstream) { - NVTX_RANGE("FSI/halo", NVTX_C7); + NVTX_RANGE("SOLUTEX/halo", NVTX_C7); if (wsolutes.size() == 0) return; - + if (iterationcount) _wait(reqsendA); - - ParticlesWrap halos[26]; - - for(int i = 0; i < 26; ++i) - halos[i] = ParticlesWrap(remote[i].dstate.data, remote[i].dstate.size, remote[i].result.devptr); - + + ParticlesWrap halowrap(allremotehalos.data, allremotehalos.size, allremotehalosacc.data); + CUDA_CHECK(cudaStreamSynchronize(uploadstream)); - + for(int i = 0; i < visitors.size(); ++i) - visitors[i]->halo(halos, stream); - + visitors[i]->halo(halowrap, computestream); + CUDA_CHECK(cudaPeekAtLastError()); - - CUDA_CHECK(cudaEventRecord(evAcomputed, stream)); - + + CUDA_CHECK(cudaEventRecord(evAcomputed, computestream)); + + CUDA_CHECK(cudaStreamWaitEvent(downloadstream, evAcomputed, 0)); + + //split back halos + { + int c = 0; + for(int i = 0; i < 26; ++i) + { + CUDA_CHECK(cudaMemcpyAsync(remote[i].result.data, allremotehalosacc.data + c, sizeof(Acceleration) * remote[i].hstate.size, cudaMemcpyDeviceToHost, downloadstream)); + c += remote[i].hstate.size; + } + } + + CUDA_CHECK(cudaEventRecord(evAdownloaded, downloadstream)); + for(int i = 0; i < 26; ++i) local[i].update(); - + #ifndef _DUMBCRAY_ _postrecvP(); #endif @@ -567,9 +599,9 @@ void SoluteExchange::post_a() if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/send-a", NVTX_C1); + NVTX_RANGE("SOLUTEX/send-a", NVTX_C1); - CUDA_CHECK(cudaEventSynchronize(evAcomputed)); + CUDA_CHECK(cudaEventSynchronize(evAdownloaded)); reqsendA.resize(26); for(int i = 0; i < 26; ++i) @@ -632,7 +664,7 @@ void SoluteExchange::recv_a(cudaStream_t stream) if (wsolutes.size() == 0) return; - NVTX_RANGE("FSI/merge", NVTX_C2); + NVTX_RANGE("SOLUTEX/merge", NVTX_C2); { float * recvbags[26]; @@ -667,4 +699,5 @@ SoluteExchange::~SoluteExchange() CUDA_CHECK(cudaEventDestroy(evPpacked)); CUDA_CHECK(cudaEventDestroy(evAcomputed)); + CUDA_CHECK(cudaEventDestroy(evAdownloaded)); } diff --git a/mpi-dpd/solute-exchange.h b/mpi-dpd/solute-exchange.h index bf315c628..a522172a7 100644 --- a/mpi-dpd/solute-exchange.h +++ b/mpi-dpd/solute-exchange.h @@ -19,11 +19,11 @@ class SoluteExchange { enum { TAGBASE_C = 113, TAGBASE_P = 365, TAGBASE_A = 668, TAGBASE_P2 = 1055, TAGBASE_A2 = 1501 }; - + public: - - struct Visitor { virtual void halo(ParticlesWrap solutehalos[26], cudaStream_t stream) = 0; }; - + + struct Visitor { virtual void halo(ParticlesWrap allhalos, cudaStream_t stream) = 0; }; + protected: MPI_Comm cartcomm; @@ -34,20 +34,20 @@ class SoluteExchange dims[3], periods[3], coords[3], myrank, recv_tags[26], recv_counts[26], send_counts[26]; - cudaEvent_t evPpacked, evAcomputed; + cudaEvent_t evPpacked, evAcomputed, evAdownloaded; SimpleDeviceBuffer packscount, packsstart, packsoffset, packstotalstart; PinnedHostBuffer host_packstotalstart, host_packstotalcount; - + SimpleDeviceBuffer packbuf; PinnedHostBuffer host_packbuf; - + std::vector wsolutes; - + std::vector reqsendC, reqrecvC, reqsendP, reqrecvP, reqsendA, reqrecvA; std::vector visitors; - + class TimeSeriesWindow { static const int N = 200; @@ -77,14 +77,12 @@ class SoluteExchange public: - SimpleDeviceBuffer dstate; PinnedHostBuffer hstate; PinnedHostBuffer result; std::vector pmessage; void preserve_resize(int n) { - dstate.resize(n); hstate.preserve_resize(n); result.resize(n); history.update(n); @@ -92,10 +90,13 @@ class SoluteExchange int expected() const { return (int)ceil(history.max() * 1.1); } - int capacity() const { assert(hstate.capacity == dstate.capacity); return dstate.capacity; } + int capacity() const { return hstate.capacity; } } remote[26]; + SimpleDeviceBuffer allremotehalos; + SimpleDeviceBuffer allremotehalosacc; + class LocalHalo { TimeSeriesWindow history; @@ -202,7 +203,7 @@ class SoluteExchange void _pack_attempt(cudaStream_t stream); public: - + SoluteExchange(MPI_Comm cartcomm); void bind_solutes(std::vector wsolutes) { this->wsolutes = wsolutes; } @@ -213,9 +214,9 @@ class SoluteExchange void post_p(cudaStream_t stream, cudaStream_t downloadstream); - void recv_p(cudaStream_t uploadstream); - - void halo(cudaStream_t uploadstream, cudaStream_t stream); + void recv_p(cudaStream_t uploadstream, cudaStream_t computestream); + + void halo(cudaStream_t uploadstream, cudaStream_t computestream, cudaStream_t downloadstream); void post_a(); diff --git a/mpi-dpd/velcontroller.cu b/mpi-dpd/velcontroller.cu new file mode 100644 index 000000000..a08cf6fe8 --- /dev/null +++ b/mpi-dpd/velcontroller.cu @@ -0,0 +1,185 @@ +/* + * velcontroller.cu + * ctc falcon + * + * Created by Dmitry Alexeev on Sep 24, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + + +#include "velcontroller.h" +#include "helper_math.h" + +//==================================================================================== +// Kernels +//==================================================================================== + +namespace VelContKernels +{ + __global__ void sample(const int * const __restrict__ cellsstart, const float2* const __restrict__ p, float3* res, VelController::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.n[0] && ccoos.y < info.n[1] && ccoos.z < info.n[2]) + { + const uint cid = (ccoos.x + info.xl[0]) + (ccoos.y + info.xl[1]) * info.cellsx + (ccoos.z + info.xl[2]) * info.cellsx * info.cellsy; + const uint resid = ccoos.x + ccoos.y * info.n[0] + ccoos.z * info.n[0] * info.n[1]; + + float3 myres = make_float3(0.0f, 0.0f, 0.0f); + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + float num = cellsstart[cid+1] - cellsstart[cid]; + float2 tmp1 = p[3*pid + 1]; + float2 tmp2 = p[3*pid + 2]; + myres.x += tmp1.y / num; + myres.y += tmp2.x / num; + myres.z += tmp2.y / num; + } + res[resid] += make_float3(myres.x, myres.y, myres.z); + } + } + + __global__ void push(const int * const __restrict__ cellsstart, Acceleration* acc, float3 f, VelController::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.n[0] && ccoos.y < info.n[1] && ccoos.z < info.n[2]) + { + const uint cid = (ccoos.x + info.xl[0]) + (ccoos.y + info.xl[1]) * info.cellsx + (ccoos.z + info.xl[2]) * info.cellsx * info.cellsy; + + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + acc[pid].a[0] += f.x; + //acc[pid].a[1] += f.y; + //acc[pid].a[2] += f.z; + } + } + } + + __inline__ __device__ float3 warpReduceSum(float3 val) + { + for (int offset = warpSize/2; offset > 0; offset /= 2) + { + val.x += __shfl_down(val.x, offset); + val.y += __shfl_down(val.y, offset); + val.z += __shfl_down(val.z, offset); + } + return val; + } + + __global__ void reduceByWarp(float3 *res, const float3 * const __restrict__ vel, const uint total) + { + assert(blockDim.x == 32); + const uint id = threadIdx.x + blockIdx.x*blockDim.x; + const uint ch = blockIdx.x; + if (id >= total) return; + + const float3 val = vel[id]; + const float3 rval = warpReduceSum(val); + + if ((threadIdx.x % warpSize) == 0) + res[ch]=rval; + } +} + +//==================================================================================== +// Methods +//==================================================================================== + +VelController::VelController(int xl[3], int xh[3], int mpicoos[3], float3 desired, MPI_Comm comm) : + desired(desired), Kp(2), Ki(1), Kd(8), factor(0.01), sampleid(0) +{ + MPI_CHECK( MPI_Comm_dup(comm, &this->comm) ); + MPI_CHECK( MPI_Comm_size(comm, &size) ); + MPI_CHECK( MPI_Comm_rank(comm, &rank) ); + const int L[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + + int myxl[3], myxh[3], n[3]; + for (int d=0; d<3; d++) + { + myxl[d] = max(xl[d], L[d]*mpicoos[d] ); + myxh[d] = min(xh[d], L[d]*(mpicoos[d]+1)); + + info.n[d] = n[d] = myxh[d] - myxl[d]; + info.xl[d] = myxl[d] % (L[d]+1); + } + + if (n[0] > 0 && n[1] > 0 && n[2] > 0) + total = n[0] * n[1] * n[2]; + else + total = 0; + + MPI_CHECK( MPI_Allreduce(&total, &globtot, 1, MPI_INT, MPI_SUM, comm) ); + + vel.resize(total); + if (total) + CUDA_CHECK( cudaMemset(vel.data, 0, n[0] * n[1] * n[2] * sizeof(float3)) ); + + info.cellsx = L[0]; + info.cellsy = L[1]; + info.cellsz = L[2]; + + s = f = make_float3(0, 0, 0); + old = desired; +} + +void VelController::sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream) +{ + dim3 block(8, 8, 1); + dim3 grid( (info.n[0] + block.x - 1) / block.x, + (info.n[1] + block.y - 1) / block.y, + (info.n[2] + block.z - 1) / block.z ); + + sampleid++; + if (total) + VelContKernels::sample <<>> (cellsstart, (float2*)p, vel.data, info); +} + +void VelController::push(const int * const cellsstart, const Particle* const p, Acceleration* acc, cudaStream_t stream) +{ + dim3 block(8, 8, 1); + dim3 grid( (info.n[0] + block.x - 1) / block.x, + (info.n[1] + block.y - 1) / block.y, + (info.n[2] + block.z - 1) / block.z ); + + if (total) + VelContKernels::push <<>> (cellsstart, acc, f, info); +} + +float3 VelController::adjustF(cudaStream_t stream) +{ + const int chunks = (total+31) / 32; + if (avgvel.size < chunks) avgvel.resize(chunks); + + if (total) + { + VelContKernels::reduceByWarp <<< (total + 31) / 32, 32, 0, stream >>> (avgvel.devptr, vel.data, total); + CUDA_CHECK( cudaStreamSynchronize(stream) ); + } + + float3 cur = make_float3(0, 0, 0); + for (int i=0; i vel; + PinnedHostBuffer avgvel; + int total, globtot; + + float3 desired; + float Kp, Ki, Kd, factor; + float3 s, old; + float3 f; + + int sampleid; + +public: + VelController(int xl[3], int xh[3], int mpicoos[3], float3 desired, MPI_Comm comm); + + void sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream); + void push (const int * const cellsstart, const Particle* const p, Acceleration* acc, cudaStream_t stream); + float3 adjustF(cudaStream_t stream); +}; + + + + diff --git a/mpi-dpd/velsampler.cu b/mpi-dpd/velsampler.cu new file mode 100644 index 000000000..c7d34d95a --- /dev/null +++ b/mpi-dpd/velsampler.cu @@ -0,0 +1,97 @@ +/* + * velsampler.cu + * ctc PANDA + * + * Created by Dmitry Alexeev on Nov 27, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +#include "velsampler.h" +#include "helper_math.h" + + +//==================================================================================== +// Kernels +//==================================================================================== + +namespace VelSmpKernels +{ + __global__ void sample(const int * const __restrict__ cellsstart, const float2* const __restrict__ p, float3* res, VelSampler::CellInfo info) + { + const uint3 ccoos = {threadIdx.x + blockIdx.x*blockDim.x, + threadIdx.y + blockIdx.y*blockDim.y, + threadIdx.z + blockIdx.z*blockDim.z}; + + if (ccoos.x < info.cellsx && ccoos.y < info.cellsy && ccoos.z < info.cellsz) + { + const uint cid = ccoos.x + (ccoos.y + ccoos.z * info.cellsy) * info.cellsx; + + float3 myres = make_float3(0.0f, 0.0f, 0.0f); + for (uint pid = cellsstart[cid]; pid < cellsstart[cid+1]; pid++) + { + float num = cellsstart[cid+1] - cellsstart[cid]; + float2 tmp1 = p[3*pid + 1]; + float2 tmp2 = p[3*pid + 2]; + myres.x += tmp1.y / num; + myres.y += tmp2.x / num; + myres.z += tmp2.y / num; + } + res[cid] += make_float3(myres.x, myres.y, myres.z); + } + } + + __global__ void scale(int n, float a, float* res) + { + const uint id = threadIdx.x + blockIdx.x*blockDim.x; + + if (id < n) + { + res[id] *= a; + } + } +} + +//==================================================================================== +// Methods +//==================================================================================== + + +VelSampler::VelSampler() +{ + const int L[3] = {XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN}; + + vels.resize(L[0]*L[1]*L[2]); + hostVels.resize(vels.size); + CUDA_CHECK( cudaMemset(vels.data, 0, vels.size * sizeof(float3)) ); + + info.cellsx = L[0]; + info.cellsy = L[1]; + info.cellsz = L[2]; +} + +void VelSampler::sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream) +{ + dim3 block(4, 4, 4); + dim3 grid( (info.cellsx + block.x - 1) / block.x, + (info.cellsy + block.y - 1) / block.y, + (info.cellsz + block.z - 1) / block.z ); + + sampleid++; + VelSmpKernels::sample <<>> (cellsstart, (float2*)p, vels.data, info); + CUDA_CHECK(cudaPeekAtLastError()); +} + +vector& VelSampler::getAvgVel(cudaStream_t stream) +{ + if (sampleid > 0) + VelSmpKernels::scale <<<(3*vels.size + 127) / 128, 128, 0, stream>>> (3*vels.size, 1.0f / sampleid, (float*)vels.data); + CUDA_CHECK(cudaPeekAtLastError()); + CUDA_CHECK(cudaMemcpyAsync(&hostVels[0], vels.data, sizeof(float3) * vels.size, cudaMemcpyDeviceToHost, stream)); + CUDA_CHECK(cudaMemsetAsync(vels.data, 0, sizeof(float3) * vels.size, stream)); + + sampleid = 0; + return hostVels; +} + + diff --git a/mpi-dpd/velsampler.h b/mpi-dpd/velsampler.h new file mode 100644 index 000000000..febe13999 --- /dev/null +++ b/mpi-dpd/velsampler.h @@ -0,0 +1,49 @@ +/* + * velsampler.h + * ctc PANDA + * + * Created by Dmitry Alexeev on Nov 27, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + + +#pragma once + +#include +#include "common.h" + +using namespace std; + +class VelSampler +{ +public: + struct CellInfo + { + uint cellsx, cellsy, cellsz; + }; + +private: + CellInfo info; + int size, rank; + + SimpleDeviceBuffer vels; + vector hostVels; + int total, globtot; + + float3 desired; + float Kp, Ki, Kd, factor; + float3 s, old; + float3 f; + + int sampleid; + +public: + VelSampler(); + + void sample(const int * const cellsstart, const Particle* const p, cudaStream_t stream); + vector& getAvgVel(cudaStream_t stream); +}; + + + diff --git a/mpi-dpd/wall.cu b/mpi-dpd/wall.cu index aea985207..9080ef101 100644 --- a/mpi-dpd/wall.cu +++ b/mpi-dpd/wall.cu @@ -57,8 +57,10 @@ namespace SolidWallsKernel texture texWallParticles; texture texWallCellStart, texWallCellCount; + template __global__ void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, - float * const acc, const float seed, const float sigmaf); + float * const acc, const float seed, const float sigmaf, const float xvel, const float y0); + void setup() { texSDF.normalized = 0; @@ -83,7 +85,8 @@ namespace SolidWallsKernel texWallCellCount.mipmapFilterMode = cudaFilterModePoint; texWallCellCount.normalized = 0; - CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); + CUDA_CHECK(cudaFuncSetCacheConfig(interactions_3tpp, cudaFuncCachePreferL1)); } __device__ float sdf(float x, float y, float z) @@ -201,8 +204,6 @@ namespace SolidWallsKernel return make_float3(xmygrad, ymygrad, zmygrad); } - - __global__ void fill_keys(const Particle * const particles, const int n, int * const key) { assert(blockDim.x * gridDim.x >= n); @@ -377,8 +378,18 @@ namespace SolidWallsKernel } } + struct StressInfo + { + float *sigma_xx, *sigma_xy, *sigma_xz, *sigma_yy, *sigma_yz, *sigma_zz; + }; + + __constant__ StressInfo stressinfo; + + + template __global__ __launch_bounds__(128, 16) void interactions_3tpp(const float2 * const particles, const int np, const int nsolid, - float * const acc, const float seed, const float sigmaf) + float * const acc, const float seed, const float sigmaf, + const float xvelocity_wall, const float z0) { assert(blockDim.x * gridDim.x >= np * 3); @@ -487,9 +498,11 @@ namespace SolidWallsKernel const float xr = _xr * invrij; const float yr = _yr * invrij; const float zr = _zr * invrij; - + + const float xvel = zq > z0 ? xvelocity_wall : 0; + const float rdotv = - xr * (dst1.y - 0) + + xr * (dst1.y - xvel) + yr * (dst2.x - 0) + zr * (dst2.y - 0); @@ -500,6 +513,16 @@ namespace SolidWallsKernel xforce += strength * xr; yforce += strength * yr; zforce += strength * zr; + + if (computestresses) + { + atomicAdd(stressinfo.sigma_xx + pid, strength * xr * _xr); + atomicAdd(stressinfo.sigma_xy + pid, strength * xr * _yr); + atomicAdd(stressinfo.sigma_xz + pid, strength * xr * _zr); + atomicAdd(stressinfo.sigma_yy + pid, strength * yr * _yr); + atomicAdd(stressinfo.sigma_yz + pid, strength * yr * _zr); + atomicAdd(stressinfo.sigma_zz + pid, strength * zr * _zr); + } } atomicAdd(acc + 3 * pid + 0, xforce); @@ -756,8 +779,10 @@ struct FieldSampler }; ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, - ExpectedMessageSizes& new_sizes, const bool verbose): - cartcomm(cartcomm), arrSDF(NULL), solid4(NULL), solid_size(0), + ExpectedMessageSizes& new_sizes, const float xvelocity): + sigma_xx(NULL), sigma_xy(NULL), sigma_xz(NULL), sigma_yy(NULL), sigma_yz(NULL), sigma_zz(NULL), + cartcomm(cartcomm), arrSDF(NULL), solid4(NULL), solid_size(0), xvelocity(xvelocity), + cells(XSIZE_SUBDOMAIN + 2 * XMARGIN_WALL, YSIZE_SUBDOMAIN + 2 * YMARGIN_WALL, ZSIZE_SUBDOMAIN + 2 * ZMARGIN_WALL) { MPI_CHECK( MPI_Comm_rank(cartcomm, &myrank)); @@ -766,6 +791,7 @@ ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& float * field = new float[ XTEXTURESIZE * YTEXTURESIZE * ZTEXTURESIZE]; + static const bool verbose = false; FieldSampler sampler("sdf.dat", cartcomm, verbose); const int L[3] = { XSIZE_SUBDOMAIN, YSIZE_SUBDOMAIN, ZSIZE_SUBDOMAIN }; @@ -1090,12 +1116,12 @@ ComputeWall::ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& CUDA_CHECK(cudaPeekAtLastError()); } -void ComputeWall::bounce(Particle * const p, const int n, cudaStream_t stream) +void ComputeWall::bounce(Particle * const p, const int n, cudaStream_t stream, const float deltat) { NVTX_RANGE("WALL/bounce", NVTX_C3) if (n > 0) - SolidWallsKernel::bounce<<< (n + 127) / 128, 128, 0, stream>>>((float2 *)p, n, myrank, dt); + SolidWallsKernel::bounce<<< (n + 127) / 128, 128, 0, stream>>>((float2 *)p, n, myrank, deltat); CUDA_CHECK(cudaPeekAtLastError()); } @@ -1104,7 +1130,6 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati const int * const cellsstart, const int * const cellscount, cudaStream_t stream) { NVTX_RANGE("WALL/interactions", NVTX_C3); - //cellsstart and cellscount IGNORED for now if (n > 0 && solid_size > 0) { @@ -1121,8 +1146,21 @@ void ComputeWall::interactions(const Particle * const p, const int n, Accelerati &SolidWallsKernel::texWallCellCount.channelDesc, sizeof(int) * cells.ncells)); assert(textureoffset == 0); - SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> - ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf); + const float z0 = (dims[2] - 1 - 2 * coords[2]) * ZSIZE_SUBDOMAIN / 2; + + if (sigma_xx) + { + SolidWallsKernel::StressInfo strinfo = { sigma_xx, sigma_xy, sigma_xz, sigma_yy, sigma_yz, sigma_zz }; + + CUDA_CHECK(cudaMemcpyToSymbolAsync(SolidWallsKernel::stressinfo, &strinfo, sizeof(strinfo), 0, cudaMemcpyHostToDevice, stream)); + + SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, z0); + } + else + SolidWallsKernel::interactions_3tpp<<< (3 * n + 127) / 128, 128, 0, stream>>> + ((float2 *)p, n, solid_size, (float *)acc, trunk.get_float(), sigmaf, xvelocity, z0); + CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallParticles)); CUDA_CHECK(cudaUnbindTexture(SolidWallsKernel::texWallCellStart)); diff --git a/mpi-dpd/wall.h b/mpi-dpd/wall.h index 466c54977..717c68523 100644 --- a/mpi-dpd/wall.h +++ b/mpi-dpd/wall.h @@ -31,6 +31,8 @@ class ComputeWall int solid_size; float4 * solid4; + float * sigma_xx, * sigma_xy, * sigma_xz, * sigma_yy, + * sigma_yz, * sigma_zz, xvelocity; cudaArray * arrSDF; @@ -38,11 +40,32 @@ class ComputeWall public: - ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, ExpectedMessageSizes& new_sizes, const bool verbose); + ComputeWall(MPI_Comm cartcomm, Particle* const p, const int n, int& nsurvived, ExpectedMessageSizes& new_sizes, const float xvelocity); ~ComputeWall(); - void bounce(Particle * const p, const int n, cudaStream_t stream); + void bounce(Particle * const p, const int n, cudaStream_t stream, const float deltat = dt); + + void set_stress_buffers(float * const stress_xx, float * const stress_xy, float * const stress_xz, float * const stress_yy, + float * const stress_yz, float * const stress_zz) + { + sigma_xx = stress_xx; + sigma_xy = stress_xy; + sigma_xz = stress_xz; + sigma_yy = stress_yy; + sigma_yz = stress_yz; + sigma_zz = stress_zz; + } + + void clr_stress_buffers() + { + sigma_xx = NULL; + sigma_xy = NULL; + sigma_xz = NULL; + sigma_yy = NULL; + sigma_yz = NULL; + sigma_zz = NULL; + } void interactions(const Particle * const p, const int n, Acceleration * const acc, const int * const cellsstart, const int * const cellscount, cudaStream_t stream); diff --git a/postprocessing/argument-parser.h b/postprocessing/argument-parser.h new file mode 100644 index 000000000..d01b9f87d --- /dev/null +++ b/postprocessing/argument-parser.h @@ -0,0 +1,243 @@ +/* + * ArgumentParser.h + * Cubism + * + *This argument parser assumes that all arguments are optional ie, each of the argument names is preceded by a '-' + *all arguments are however NOT optional to avoid a mess with default values and returned values when not found! + * + *More converter could be required: + *add as needed + *TypeName as{TypeName}() in Value + * + * Created by Christian Conti on 6/7/10. That is a long time ago. + * Modified by Diego Rossinelli several times after his dreadlocks hair cut. + * Copyright 2010 ETH Zurich. All rights reserved. + * + */ + +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace std; + +class Value +{ +private: + string content; + +public: + +Value() : content("") {} + +Value(string content_) : content(content_) { /*printf("%s\n",content.c_str());*/ } + + double asDouble(double def=0) const + { + if (content == "") return def; + return (double) atof(content.c_str()); + } + + int asInt(int def=0) const + { + if (content == "") return def; + return atoi(content.c_str()); + } + + bool asBool(bool def=false) const + { + if (content == "") return def; + if (content == "0") return false; + if (content == "false") return false; + + return true; + } + + string asString(string def="") const + { + if (content == "") return def; + + return content; + } + + vector asVecFloat(const int musthave_size = -1) const + { + //printf("mycontent is %s\n", content.c_str()); + std::stringstream ss(content); + //assert(ss.good()); + vector retval; + double e; + + while (ss >> e) + { + retval.push_back(e); + // printf("reading %f\n", e); + if (ss.peek() == ',') + ss.ignore(); + } + + if (musthave_size > 0) + assert(musthave_size == (int)retval.size()); + + return retval; + } +}; + +class ArgumentParser +{ +private: + + map mapArguments; + + const int iArgC; + const char** vArgV; + bool bStrictMode, bVerbose; + + const char delimiter; +public: + + Value operator()(const string arg) + { + map::const_iterator it = mapArguments.find(arg); + + if (bStrictMode) + { + if (it == mapArguments.end()) + { + printf("Runtime option NOT SPECIFIED! ABORTING! name: %s\n",arg.data()); + abort(); + } + } + + if (bVerbose) + printf("%s is %s\n", arg.data(), mapArguments[arg].asString().data()); + + if (it != mapArguments.end()) + return mapArguments[arg]; + else + return Value(); + } + + bool check(const string arg) const + { + return mapArguments.find(arg) != mapArguments.end(); + } + +ArgumentParser(const int argc, const char ** argv, bool bVerbose = false, const char delimiter = '=') : + mapArguments(), iArgC(argc), vArgV(argv), bStrictMode(false), bVerbose(bVerbose), delimiter(delimiter) + { + for (int i = 1; i args, bool bVerbose = false, const char delimiter = '='): + mapArguments(), iArgC(args.size()), vArgV(NULL), bStrictMode(false), bVerbose(bVerbose), delimiter(delimiter) + { + for(vector::iterator it = args.begin(); it != args.end(); ++it) + { + const char * arg = it->c_str(); + + int sep = 0; + while(arg[sep] != '\0' && arg[sep] != delimiter) + ++sep; + + string value; + + if (arg[sep] != '\0') + value = string(arg + sep + 1); + else + value = "1"; + + mapArguments[string(arg, sep)] = Value(value); + } + + mute(); + } + + int getargc() const { return iArgC; } + + const char** getargv() const { return vArgV; } + + void set_strict_mode() + { + bStrictMode = true; + } + + void unset_strict_mode() + { + bStrictMode = false; + } + + void mute() + { + bVerbose = false; + } + + void loud() + { + bVerbose = true; + } + + void print_arguments(FILE * f = stdout) + { + printf("PRINTOUT OF THE RUNTIME OPTIONS\n"); + + for(map::const_iterator it=mapArguments.begin(); it!=mapArguments.end(); it++) + fprintf(f, "%s: <%s>\n", it->first.c_str(), it->second.asString().c_str()); + + printf("END OF THE PRINTOUT.\n"); + } + + void print_arguments(string path2log) + { + FILE * f = fopen(path2log.c_str(), "w"); + + if (f == NULL) + { + printf("could not save the log to <%s>. Exiting now\n", path2log.c_str()); + exit(-1); + } + + print_arguments(f); + + fclose(f); + } + + vector find(string name) + { + map::iterator itb = mapArguments.lower_bound(name); + map::iterator ite = mapArguments.end(); + + vector retval; + for(map::iterator it = itb; it != ite; ++it) + { + if (it->first.find(name) == string::npos) + break; + + retval.push_back(it->first); + } + return retval; + } +}; diff --git a/postprocessing/mpi-check.h b/postprocessing/mpi-check.h new file mode 100644 index 000000000..31900eaab --- /dev/null +++ b/postprocessing/mpi-check.h @@ -0,0 +1,19 @@ +#include + +#include + +#define MPI_CHECK(ans) do { mpiAssert((ans), __FILE__, __LINE__); } while(0) + +inline void mpiAssert(int code, const char *file, int line, bool abort=true) +{ + if (code != MPI_SUCCESS) + { + char error_string[2048]; + int length_of_error_string = sizeof(error_string); + MPI_Error_string(code, error_string, &length_of_error_string); + + printf("mpiAssert: %s %d %s\n", file, line, error_string); + + MPI_Abort(MPI_COMM_WORLD, code); + } +} diff --git a/postprocessing/ply2vtkpts/Makefile b/postprocessing/ply2vtkpts/Makefile new file mode 100644 index 000000000..ec76ea0e7 --- /dev/null +++ b/postprocessing/ply2vtkpts/Makefile @@ -0,0 +1,17 @@ +CXX ?= CC + +ply2vtk: main.cpp + $(CXX) -Ofast -std=c++11 -fopenmp main.cpp \ + -I/apps/daint/VTK/6.2/gnu_491/include/vtk-6.2 -L/apps/daint/VTK/6.2/gnu_491/lib \ + -lvtkIOImage-6.2 -lvtkCommonDataModel-6.2 -lvtkpng-6.2 -lvtktiff-6.2 \ + -lvtkmetaio-6.2 -lvtkDICOMParser-6.2 -lvtkzlib-6.2 -lvtksys-6.2 \ + -lvtkIOXMLParser-6.2 -lvtkCommonExecutionModel-6.2 -lvtkCommonTransforms-6.2 \ + -lvtkCommonCore-6.2 -lvtkIOXML-6.2 -lvtkexpat-6.2 -lvtkjpeg-6.2 -lvtkIOCore-6.2 \ + -lvtkCommonSystem-6.2 -lvtkCommonTransforms-6.2 -lvtkCommonMath-6.2 \ + -lvtkCommonMisc-6.2 \ + -o ply2vtk + +clean: + rm -f ply2vtk + +.PHONY = clean diff --git a/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh b/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh new file mode 100644 index 000000000..b5c9c0ade --- /dev/null +++ b/postprocessing/ply2vtkpts/daint-ply2vtkpts.sh @@ -0,0 +1,45 @@ +module swap PrgEnv-cray PrgEnv-gnu +module load cray-hdf5-parallel +module unload cray-mpich/7.0.4 +module load cray-mpich/7.1.1 +module unload gcc +module load gcc/4.9.1 +module load vtk + +convert_some() +{ + MYFOLDER=$1 + SRCPATH=$2 + SRCPATTERN=$3 + NVERTPERCELLS=$4 + + mkdir -p $MYFOLDER + + echo `date` "convert_some: $*" >> ${MYFOLDER}/log.txt + + find "$SRCPATH" -name "$SRCPATTERN" > /tmp/asd.txt + + for F in $(cat /tmp/asd.txt) + do + SRC=`basename $F` + + DST=${MYFOLDER}/${SRC%.ply}.vtp + + aprun ./ply2vtk $NVERTPERCELLS $F $DST + done +} + +if (( $# != 3)) +then + echo "usage ./convert-all.sh " + + exit 1 +fi + + +MYFOLDER=$1 #for example "ichip31" +NVERTRBC=$2 #for example 498 +NVERTCTC=$3 #for example 5220 + +convert_some "$MYFOLDER" /scratch/daint/alexeedm/ctc/"$MYFOLDER"/ply/ "rbcs-*.ply" $NVERTRBC +convert_some "$MYFOLDER" /scratch/daint/alexeedm/ctc/"$MYFOLDER"/ply/ "ctcs-*.ply" $NVERTCTC \ No newline at end of file diff --git a/postprocessing/ply2vtkpts/main.cpp b/postprocessing/ply2vtkpts/main.cpp new file mode 100644 index 000000000..c1465f927 --- /dev/null +++ b/postprocessing/ply2vtkpts/main.cpp @@ -0,0 +1,223 @@ +#include +#include + +#include +#include +#include +#include + + +#define MPI_CHECK(ans) do { mpiAssert((ans), __FILE__, __LINE__); } while(0) + +inline void mpiAssert(int code, const char *file, int line, bool abort=true) +{ + if (code != MPI_SUCCESS) + { + char error_string[2048]; + int length_of_error_string = sizeof(error_string); + MPI_Error_string(code, error_string, &length_of_error_string); + + printf("mpiAssert: %s %d %s\n", file, line, error_string); + + MPI_Abort(MPI_COMM_WORLD, code); + } +} + +using namespace std; + +#include +#include +#include +#include +#include +#include + +int dump_vtk_points(const char * dstpath, const int nrbcs, const float * xs, const float * ys, const float * zs ) +{ + vtkSmartPointer points = + vtkSmartPointer::New(); + + for ( unsigned int i = 0; i < nrbcs; ++i ) + points->InsertNextPoint ( xs[i], ys[i], zs[i] ); + + // Create a polydata object and add the points to it. + vtkSmartPointer polydata = + vtkSmartPointer::New(); + polydata->SetPoints(points); + + // Write the file + vtkSmartPointer writer = + vtkSmartPointer::New(); + writer->SetFileName(dstpath); +#if VTK_MAJOR_VERSION <= 5 + writer->SetInput(polydata); +#else + writer->SetInputData(polydata); +#endif + + writer->Write(); + + return EXIT_SUCCESS; +} + +int main(int argc, char ** argv) +{ + MPI_CHECK(MPI_Init(&argc, &argv)); + + int nranks, rank; + MPI_CHECK(MPI_Comm_size(MPI_COMM_WORLD, &nranks)); + MPI_CHECK(MPI_Comm_rank(MPI_COMM_WORLD, &rank)); + + const bool verbose = false; + + if (argc != 4) + { + if (rank == 0) + printf("usage: test \n"); + + exit(EXIT_FAILURE); + } + + const int nvpc = atoi(argv[1]); + const char * path = argv[2]; + const char * dstpath = argv[3]; + + if (rank == 0) + printf("reading at location <%s>\n", path); + + int nallvertices, nallrbcs, headersize; + + const double tstart = omp_get_wtime(); + + if (rank == 0) + { + FILE * f = fopen(path, "r"); + assert(f); + char line[2048]; + + auto eat_line = [&] () + { + fgets(line, 2048, f); + + if (verbose) + printf("reading <%s>\n", line); + }; + + for(int i = 0; i < 3; ++i) + eat_line(); + + int retval = sscanf(line, "element vertex %d\n", &nallvertices); + assert(retval == 1); + + nallrbcs = nallvertices / nvpc; + + if (verbose) + printf("*** nvertices: %d\n", nallvertices); + + for(int i = 0; i < 7; ++i) + eat_line(); + + int nfaces = -1; + retval = sscanf(line, "element face %d\n", &nfaces); + assert(retval == 1); + + if (verbose) + printf("*** nfaces: %d\n", nfaces); + + for(int i = 0; i < 2; ++i) + eat_line(); + + headersize = ftell(f); + + fclose(f); + } + + MPI_CHECK(MPI_Bcast(&nallvertices, 1, MPI_INT, 0, MPI_COMM_WORLD)); + MPI_CHECK(MPI_Bcast(&nallrbcs, 1, MPI_INT, 0, MPI_COMM_WORLD)); + MPI_CHECK(MPI_Bcast(&headersize, 1, MPI_INT, 0, MPI_COMM_WORLD)); + + const double theader = omp_get_wtime(); + + const int myrbcs_size = nallrbcs / nranks + (int)(rank < (nallrbcs % nranks)); + const int myrbcs_start = nallrbcs / nranks * rank + min(rank, nallrbcs % nranks); + + float * data = new float[6 * myrbcs_size * nvpc]; + + { + MPI_File filehandle; + MPI_CHECK( MPI_File_open(MPI_COMM_WORLD, path, MPI_MODE_RDONLY, MPI_INFO_NULL, &filehandle) ); + + MPI_Status status; + MPI_CHECK( MPI_File_read_at(filehandle, headersize + myrbcs_start * nvpc * 6 * sizeof(float), + data, myrbcs_size * nvpc * 6, MPI_FLOAT, &status)); + + MPI_CHECK( MPI_File_close(&filehandle)); + } + + vector coords[3]; + + for(int i = 0; i < 3; ++i) + coords[i].resize(myrbcs_size); + +#pragma omp parallel for + for(int r = 0; r < myrbcs_size; ++r) + { + float com[3] = {0, 0, 0}; + + for(int v = 0; v < nvpc; ++v) + for(int c = 0; c < 3; ++c) + com[c] += data[c + 6 * (v + nvpc * r)]; + + for(int i = 0; i < 3; ++i) + coords[i][r] = com[i] /nvpc; + } + + delete [] data; + + vector allcoords[3]; + + if (rank == 0) + for(int i = 0; i < 3; ++i) + allcoords[i].resize(nranks * ((nallrbcs + nranks - 1) / nranks)); + + for(int i = 0; i < 3; ++i) + { + MPI_CHECK( MPI_Gather(&coords[i].front(), nallrbcs / nranks, MPI_FLOAT, + &allcoords[i].front(), nallrbcs / nranks, MPI_FLOAT, + 0, MPI_COMM_WORLD) ); + + MPI_CHECK( MPI_Gather(&coords[i].back(), 1, MPI_FLOAT, + (&allcoords[i].front()) + nranks * (nallrbcs / nranks), 1, MPI_FLOAT, + 0, MPI_COMM_WORLD) ); + } + + const double tthroughput = omp_get_wtime(); + + if (rank == 0) + dump_vtk_points(dstpath, nallrbcs, &allcoords[0].front(), &allcoords[1].front(), &allcoords[2].front()); + + const double tvtk = omp_get_wtime(); + + if (rank == 0) + { + const double ttotal = tvtk - tstart; + + printf("TOTAL TIME: %.2f\n", ttotal); + + printf("TDISTRIBUTION: HEADER:%.1f%%\tI/O+REDUCE:%.1f%%\tVTK:%.1f%%\t\n", + 100 / ttotal * (theader - tstart), + 100 / ttotal * (tthroughput - theader), + 100 / ttotal * (tvtk - tthroughput)); + + const double memfp_ply = 6. * nallvertices * sizeof(float) / pow(1024., 3); + const double memfp_vtk = 3. * nallrbcs * sizeof(float) / pow(1024., 3); + + printf("THROUGHPUT: %.1f GB/s\n", (memfp_ply + memfp_vtk) / (tthroughput - theader)); + printf("VTK DUMP: %.1f GB/s\n", memfp_vtk / (tvtk - tthroughput)); + } + + MPI_CHECK(MPI_Finalize()); + + return 0; +} + diff --git a/postprocessing/stress/Makefile b/postprocessing/stress/Makefile new file mode 100644 index 000000000..1707fd490 --- /dev/null +++ b/postprocessing/stress/Makefile @@ -0,0 +1,9 @@ +CXX = mpicxx + +stress: main.cpp + $(CXX) main.cpp -I../ -g -O3 -Wno-deprecated-declarations -Wno-unused-result -o stress + +clean: + rm -f stress + +.PHONY = clean diff --git a/postprocessing/stress/example.run b/postprocessing/stress/example.run new file mode 100644 index 000000000..8123d29ce --- /dev/null +++ b/postprocessing/stress/example.run @@ -0,0 +1,28 @@ +make + +echo EXAMPLE1: REDUCE ALL +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=1,1,1 + +echo EXAMPLE2: 1D PROFILE + GNUPLOT +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,48 -project=1,0,1 > profile.txt +gnuplot -persist <<- END_GNUPLOT + plot "profile.txt" u 1:2 w lp title "sigma_xx", "profile.txt" u 1:3 w lp title "sigma_xy", "profile.txt" u 1:4 w lp title "sigma_xz", "profile.txt" u 1:5 w lp title "sigma_yy", "profile.txt" u 1:6 w lp title "sigma_yz", "profile.txt" u 1:7 w lp title "sigma_zz" +END_GNUPLOT + +echo EXAMPLE3: HEIGHT FIELD + MATPLOTLIB +find ../../mpi-dpd/stress/* | ./stress -origin=0,0,0 -extent=48,48,32 -project=0,1,0 | csplit --suppress-matched - '/^$/' {*} -f channel. +python <<- END_PYTHON +import numpy as np +import matplotlib.pyplot as plt +import matplotlib.cm as cm + +for path in ["channel.00", "channel.01", "channel.02", "channel.03", "channel.04", "channel.05"]: + ncols, nrows = 48, 32 + temp = np.loadtxt(path).T + grid = temp.reshape((nrows, ncols)) + plt.figure(path) + plt.imshow(grid, interpolation='nearest', cmap=cm.gist_rainbow) + +plt.show() + +END_PYTHON \ No newline at end of file diff --git a/postprocessing/stress/main.cpp b/postprocessing/stress/main.cpp new file mode 100644 index 000000000..c90aeb02f --- /dev/null +++ b/postprocessing/stress/main.cpp @@ -0,0 +1,295 @@ +#include +#include +#include + +#include +#include +#include +#include + +#include +#include + +using namespace std; + +int main(int argc, const char ** argv) +{ + MPI_CHECK( MPI_Init(&argc, (char ***)&argv) ); + + int nranks, rank; + MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank)); + MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks)); + + ArgumentParser argp(argc, argv); + + const bool verbose = argp("-verbose").asBool(false); + const bool avg = argp("-average").asBool(true); + vector origin = argp("-origin").asVecFloat(3); + vector extent = argp("-extent").asVecFloat(3); + vector projectf = argp("-project").asVecFloat(3); + string contributions = argp("-contributions").asString("uf"); + + const double ufactor = contributions.find("u") != string::npos; + const double ffactor = contributions.find("f") != string::npos; + + bool project[3]; + for(int c = 0; c < 3; ++c) + project[c] = projectf[c] != 0; + + int nprojections = 0; + for(int c = 0; c < 3; ++c) + nprojections += project[c]; + + const int noutputchannels = 9; + const size_t chunksize = (1 << 29) / 12 / sizeof(float); + + float * const pbuf = new float[12 * chunksize]; + + float binsize[3]; + for(int c = 0; c < 3; ++c) + binsize[c] = project[c] ? extent[c] : 1; + + int nbins[3]; + for(int c = 0; c < 3; ++c) + nbins[c] = extent[c] / binsize[c]; + + const int ntotbins = nbins[0] * nbins[1] * nbins[2]; + + int * const bincount = new int[ntotbins]; + memset(bincount, 0, sizeof(int) * ntotbins); + + const int noutput = noutputchannels * ntotbins; + + double * const bindata = new double[noutput]; + memset(bindata, 0, sizeof(double) * noutput); + + vector paths; + + { + string myinput; + + if (rank == 0) + for (string line; getline(cin, line);) + myinput += line + "\n"; + + int inputsize = myinput.size(); + MPI_CHECK( MPI_Bcast(&inputsize, 1, MPI_INTEGER, 0, MPI_COMM_WORLD)); + + myinput.resize(inputsize); + + MPI_CHECK( MPI_Bcast(&myinput[0], inputsize, MPI_CHAR, 0, MPI_COMM_WORLD)); + + int c = 0; + istringstream iss(myinput); + + for (string line; getline(iss, line); ++c) + if (c % nranks == rank) + paths.push_back(line); + } + + int numfiles = paths.size(); + + size_t totalfootprint = 0; + double timeIO = 0; + + for(int ipath = 0; ipath < (int)paths.size(); ++ipath) + { + const char * const path = paths[ipath].c_str(); + + if (verbose) + fprintf(stderr, "working on <%s>\n", path); + + int fdin = open(path, O_RDONLY); + + if (!fdin) + { + fprintf(stderr, "can't access <%s> , exiting now.\n", path); + exit(-1); + } + + if (verbose) + perror("reading...\n"); + + const size_t filesize = lseek(fdin, 0, SEEK_END); + + totalfootprint += filesize; + + lseek(fdin, 0, SEEK_SET); + + const size_t nparticles = filesize / 12 / sizeof(float); + assert(filesize % (12 * sizeof(float)) == 0); + + if (verbose) + { + fprintf(stderr, "i have found %d particles\n", (int)nparticles); + fprintf(stderr, "particle chunk %d\n", (int)chunksize); + } + + for(size_t base = 0; base < nparticles; base += chunksize) + { + const int nhotparticles = min(nparticles - base, chunksize); + const size_t nhotbytes = nhotparticles * sizeof(float) * 12; + + size_t nreadbytes = 0; + int start = 0; + + while(start < nhotparticles) + { + const double tstart = MPI_Wtime(); + nreadbytes += read(fdin, pbuf, nhotbytes - nreadbytes); + timeIO += MPI_Wtime() - tstart; + + const int stop = nreadbytes / sizeof(float) / 12; + +#ifndef NDEBUG + if (verbose) + { + float avgs[12]; + for(int i = 0; i < 12; ++i) + avgs[i] = 0; + + for(int i = 0; i < nhotparticles; ++i) + for(int c = 0; c < 12; ++c) + avgs[c] += pbuf[12 * i + c]; + + for(int i = 0; i < 12; ++i) + printf("AVG %d: %.3e\n", i, avgs[i] / nhotparticles); + } +#endif + + for(int i = start; i < stop; ++i) + { + const int srcbase = 12 * i; + + int index[3]; + for(int c = 0; c < 3; ++c) + index[c] = (int)((pbuf[srcbase + c] - origin[c]) / binsize[c]); + + bool valid = true; + for(int c = 0; c < 3; ++c) + valid &= index[c] >= 0 && index[c] < nbins[c]; + + if (!valid) + continue; + + const int binid = index[0] + nbins[0] * (index[1] + nbins[1] * index[2]); + ++bincount[binid]; + + const int dstbase = noutputchannels * binid; + + const int v1[6] = {0, 0, 0, 1, 1, 2}; + const int v2[6] = {0, 1, 2, 1, 2, 2}; + + for(int c = 0; c < 6; ++c) + bindata[dstbase + c] += + ffactor * pbuf[srcbase + 6 + c] + + ufactor * pbuf[srcbase + 3 + v1[c]] * pbuf[srcbase + 3 + v2[c]]; + + for(int c = 0; c < 3; ++c) + bindata[dstbase + 6 + c] += pbuf[srcbase + 3 + c]; + } + + start = stop; + } + } + + close(fdin); + } + + if (rank == 0 && !numfiles) + { + perror("ooops zero files were read. Exiting now.\n"); + exit(-1); + } + + MPI_CHECK( MPI_Reduce(rank ? bincount : MPI_IN_PLACE, bincount, ntotbins, MPI_INT, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? bindata : MPI_IN_PLACE, bindata, noutput, MPI_DOUBLE, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? &timeIO : MPI_IN_PLACE, &timeIO, 1, MPI_DOUBLE, MPI_SUM, 0, MPI_COMM_WORLD) ); + MPI_CHECK( MPI_Reduce(rank ? &totalfootprint : MPI_IN_PLACE, &totalfootprint, 1, MPI_OFFSET, MPI_SUM, 0, MPI_COMM_WORLD) ); + + if (rank) + goto finalize; + + if (avg) + for(int i = 0; i < ntotbins; ++i) + for(int c = 0; c < noutputchannels; ++c) + bindata[noutputchannels * i + c] /= bincount[i]; + + if (nprojections == 3) + { + assert(noutput == noutputchannels); + + for(int c = 0; c < noutputchannels; ++c) + printf("%+.3e\t", bindata[c]); + + printf("\n"); + } + else if (nprojections == 2) + { + int ctr = 0; + for(int iz = 0; iz < nbins[2]; ++iz) + for(int iy = 0; iy < nbins[1]; ++iy) + for(int ix = 0; ix < nbins[0]; ++ix) + { + printf("%03d ", ctr); + + for(int c = 0; c < noutputchannels; ++c) + printf("%+.4e ", bindata[noutputchannels * ctr + c]); + + printf("\n"); + + ++ctr; + } + } + else if (nprojections == 1) + { + int nx = 0; + for(int c = 0; c < 3; ++c) + if (nbins[c] > 1) + { + nx = nbins[c]; + break; + } + + for(int c = 0; c < noutputchannels; ++c) + { + int ctr = 0; + + for(int iz = 0; iz < nbins[2]; ++iz) + for(int iy = 0; iy < nbins[1]; ++iy) + for(int ix = 0; ix < nbins[0]; ++ix) + { + printf("%+.5e ", bindata[noutputchannels * ctr + c]); + + ++ctr; + + if (ctr % nx == 0) + printf("\n"); + } + + if (c < noutputchannels - 1) + printf("\n"); + } + } + else + { + perror("woops invalid number of projections. Exiting now...\n"); + exit(-1); + } + + if (verbose) + perror("all is done. ciao.\n"); + + fprintf(stderr, "total footprint: %.3f MB, I/O time: %.3f ms\n", totalfootprint * 1. / 1024 / 1024, timeIO * 1e3); + fprintf(stderr, "read throughput: %.3f GB/s\n", totalfootprint /( 1024 * 1024) / timeIO / 1024); + +finalize: + + delete [] pbuf; + delete [] bincount; + delete [] bindata; + + MPI_CHECK( MPI_Finalize() ); + + return 0; +} diff --git a/tests/contact/Makefile b/tests/contact/Makefile new file mode 100644 index 000000000..e14c7da00 --- /dev/null +++ b/tests/contact/Makefile @@ -0,0 +1,26 @@ +-include ../../mpi-dpd/.cache.Makefile + +NVCC ?= nvcc -ccbin $(CXX) +ARCH_VAL ?= compute_35 +CODE_VAL ?= sm_35 + + +NVCCFLAGS += -I$(HDF5_DIR)/include -Xcudafe "--diag_suppress=unrecognized_gcc_pragma" +NVCCFLAGS += -arch $(ARCH_VAL) -code $(CODE_VAL) -O3 -use_fast_math -g -DNDEBUG -Xcompiler "-fopenmp" +CXXFLAGS += -L../../cuda-dpd/dpd -L../../cuda-rbc/ -L../../cuda-ctc/ -O3 -g -std=c++11 -DNDEBUG -fopenmp +NVCCFLAGS += -I../../cuda-dpd/dpd -I../../cuda-dpd/ -I../../cuda-rbc/ -I../../cuda-ctc -I../../mpi-dpd +LIBS = -lcuda-dpd -lcuda-rbc -lcuda-ctc -lcudart -ldl -lz -fopenmp + +testcontact: ../../mpi-dpd/test testcontact.o + make -C ../../mpi-dpd + rm -f ../../mpi-dpd/main.o + $(CXX) $(CXXFLAGS) testcontact.o ../../mpi-dpd/*.o $(LIBS) -o testcontact + +../test: + make -C ../ + cp ../*.o . + +testcontact.o: testcontact.cu ../../mpi-dpd/argument-parser.h ../../mpi-dpd/common.h ../../mpi-dpd/containers.h ../../mpi-dpd/contact.h ../../cuda-dpd/dpd-rng.h + $(NVCC) $(NVCCFLAGS) testcontact.cu -c -o testcontact.o + +.PHONY: ../test diff --git a/tests/contact/testcontact.cu b/tests/contact/testcontact.cu new file mode 100644 index 000000000..d221b35de --- /dev/null +++ b/tests/contact/testcontact.cu @@ -0,0 +1,300 @@ +/* + * main.cu + * ctc PANDA + * + * Created by Dmitry Alexeev on Oct 20, 2015 + * Copyright 2015 ETH Zurich. All rights reserved. + * + */ + +/* + * main.cu + * Part of uDeviceX/mpi-dpd/ + * + * Created and authored by Diego Rossinelli on 2014-11-14. + * Copyright 2015. All rights reserved. + * + * Users are NOT authorized + * to employ the present software for their own publications + * before getting a written permission from the author of this file. + */ + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + enum + { + XCELLS = XSIZE_SUBDOMAIN, + YCELLS = YSIZE_SUBDOMAIN, + ZCELLS = ZSIZE_SUBDOMAIN, + XOFFSET = XCELLS / 2, + YOFFSET = YCELLS / 2, + ZOFFSET = ZCELLS / 2 + }; +using namespace std; + +float tend, couette; +bool walls, pushtheflow, doublepoiseuille, rbcs, ctcs, xyz_dumps, hdf5field_dumps, hdf5part_dumps, is_mps_enabled, adjust_message_sizes, contactforces, stress; +int steps_per_report, steps_per_dump, wall_creation_stepid, nvtxstart, nvtxstop; + +LocalComm localcomm; + +static const float ljsigma = 0.5; +static const float ljsigma2 = ljsigma * ljsigma; + +template +inline float _viscosity_function(float x) +{ + return sqrtf(viscosity_function(x)); +} + +template<> inline float _viscosity_function<1>(float x) { return sqrtf(x); } +template<> inline float _viscosity_function<0>(float x){ return x; } + +int main(int argc, char ** argv) +{ + CUDA_CHECK(cudaSetDevice(0)); + CUDA_CHECK(cudaDeviceReset()); + + { + is_mps_enabled = false; + + const char * mps_variables[] = { + "CRAY_CUDA_MPS", + "CUDA_MPS", + "CRAY_CUDA_PROXY", + "CUDA_PROXY" + }; + + for(int i = 0; i < 4; ++i) + is_mps_enabled |= getenv(mps_variables[i])!= NULL && atoi(getenv(mps_variables[i])) != 0; + } + + int nranks, rank; + MPI_CHECK(MPI_Init(&argc, &argv)); + MPI_CHECK( MPI_Comm_size(MPI_COMM_WORLD, &nranks) ); + MPI_CHECK( MPI_Comm_rank(MPI_COMM_WORLD, &rank) ); + MPI_Comm activecomm = MPI_COMM_WORLD; + + bool reordering = true; + const char * env_reorder = getenv("MPICH_RANK_REORDER_METHOD"); + + MPI_Comm cartcomm; + int periods[] = {1, 1, 1}; + int ranks[] = {1, 1, 1}; + + + MPI_CHECK( MPI_Cart_create(activecomm, 3, ranks, periods, (int)reordering, &cartcomm) ); + activecomm = cartcomm; + + { + MPI_CHECK(MPI_Barrier(activecomm)); + localcomm.initialize(activecomm); + + MPI_CHECK(MPI_Barrier(activecomm)); + + // test here + const size_t myseed = 0x563d00cf;//time(NULL); + srand48(myseed); + printf("myseed: 0x%x\n", myseed); + + int n = 25e3; + vector ic(n); + vector acc(n); + for (int i=0; i gpuacc(n); + + Logistic::KISS local_trunk = Logistic::KISS(7119 - rank, 187 + rank, 18278, 15674); + + const double center[3] = { -XSIZE_SUBDOMAIN/2, -YSIZE_SUBDOMAIN/2, -ZSIZE_SUBDOMAIN/2} ;//YSIZE_SUBDOMAIN/2 -}; + const double halfwidth[3] = {XSIZE_SUBDOMAIN/10., YSIZE_SUBDOMAIN/10., ZSIZE_SUBDOMAIN / 10.}; + + for(int i = 0; i < n; ++i) + { + ic[i].x[0] = center[0] + halfwidth[0] * 2 * (drand48() - 0.5); + ic[i].x[1] = center[1] + halfwidth[1] * 2 * (drand48() - 0.5); + ic[i].x[2] = center[2] + halfwidth[2] * 2 * (drand48() - 0.5); + ic[i].u[0] = 0.5 - drand48(); + ic[i].u[1] = 0.5 - drand48(); + ic[i].u[2] = 0.5 - drand48(); + } + + if (false)//if (true) + { + ic.resize(2); + acc.resize(2); + gpuacc.resize(2); + n = 2; + //ic[0].x[0] = 24.413; ic[0].x[1] = +14.924; ic[0].x[2] = +7.326; + //ic[1].x[0] = +23.895; ic[1].x[1] = +14.887; ic[1].x[2] = +7.455 ; + + ic[0].x[0] = +23.670 ; ic[0].x[1] =+23.494; ic[0].x[2] =-18.980; + ic[1].x[0] = -23.851 ; ic[1].x[1] =+23.696; ic[1].x[2] =-18.790; + } + + float seed = local_trunk.get_float(); + +#pragma omp parallel for + for (int i=0; i= 1) + continue; + + const double invr2 = invrij * invrij; + const double t2 = ljsigma2 * invr2; + const double t4 = t2 * t2; + const double t6 = t4 * t2; + const double lj = min(1e4f, max(0.f, 24.f * invrij * t6 * (2.f * t6 - 1.f))); + + const double wr = _viscosity_function<0>(1.f - rij); + + const double xr = _xr * invrij; + const double yr = _yr * invrij; + const double zr = _zr * invrij; + + const double strength = lj; + + const double xinteraction = strength * xr; + const double yinteraction = strength * yr; + const double zinteraction = strength * zr; + + acc[i].a[0] += xinteraction; + acc[i].a[1] += yinteraction; + acc[i].a[2] += zinteraction; + } + + ParticleArray p; + p.resize(n); + + CUDA_CHECK( cudaMemcpy(p.xyzuvw.data, &ic[0], n * sizeof(Particle), cudaMemcpyHostToDevice) ); + CUDA_CHECK( cudaMemset(p.axayaz.data, 0, n * sizeof(Acceleration)) ); + std::vector wsolutes; + wsolutes.push_back(ParticlesWrap(p.xyzuvw.data, n, p.axayaz.data)); + + ComputeContact contact(cartcomm); + SoluteExchange solutex(cartcomm); + + solutex.attach_halocomputation(contact); + contact.attach_bulk(wsolutes); + + solutex.bind_solutes(wsolutes); + solutex.pack_p(0); + solutex.post_p(0, 0); + solutex.recv_p(0); + solutex.halo(0, 0); + solutex.post_a(); + solutex.recv_a(0); + + CUDA_CHECK( cudaMemcpy(&gpuacc[0], p.axayaz.data, n * sizeof(Acceleration), cudaMemcpyDeviceToHost) ); + + { + double fx = 0, fy = 0, fz = 0; + double hfx = 0, hfy = 0, hfz = 0; + + for (int i=0; i= tol && fabs(err) >= tol; + + if (failed) + printf("p %d c %d: %e ref: %e -> %e %e\n", i / 3, i % 3, res[i], ref[i], err, relerr); + + if (i % 3 == 2 && failed) + { + const int pid = i/3; + + const bool inside = + ic[i].x[0] >= -XOFFSET && ic[pid].x[0] < XOFFSET && + ic[i].x[1] >= -YOFFSET && ic[pid].x[1] < YOFFSET && + ic[i].x[2] >= -ZOFFSET && ic[pid].x[2] < ZOFFSET ; + + printf("%d: CPU [%+.3f %+.3f %+.3f] GPU [%+.3f %+.3f %+.3f] -> p %+.3f %+.3f %+.3f -> inside: %d\n", + i, acc[pid].a[0], acc[pid].a[1], acc[pid].a[2], + gpuacc[pid].a[0], gpuacc[pid].a[1], gpuacc[pid].a[2], + ic[pid].x[0], ic[pid].x[1], ic[pid].x[2], inside); + + failed = false; + } + + assert(fabs(relerr) < tol || fabs(err) < tol); + + l1 += fabs(err); + l1_rel += fabs(relerr); + + linf = std::max(linf, fabs(err)); + linf_rel = std::max(linf_rel, fabs(relerr)); + } + + printf("l-infinity errors: %.03e (absolute) %.03e (relative)\n", linf, linf_rel); + printf(" l-1 errors: %.03e (absolute) %.03e (relative)\n", l1, l1_rel); + } + } + + if (activecomm != cartcomm) + MPI_CHECK(MPI_Comm_free(&activecomm)); + + MPI_CHECK(MPI_Comm_free(&cartcomm)); + + MPI_CHECK(MPI_Finalize()); + + CUDA_CHECK(cudaDeviceSynchronize()); + + CUDA_CHECK(cudaDeviceReset()); + + return 0; +} \ No newline at end of file diff --git a/halo-bench/Makefile b/tests/halo-bench/Makefile similarity index 100% rename from halo-bench/Makefile rename to tests/halo-bench/Makefile diff --git a/halo-bench/byte_latency.cpp b/tests/halo-bench/byte_latency.cpp similarity index 100% rename from halo-bench/byte_latency.cpp rename to tests/halo-bench/byte_latency.cpp diff --git a/halo-bench/halo-exchanger.cu b/tests/halo-bench/halo-exchanger.cu similarity index 100% rename from halo-bench/halo-exchanger.cu rename to tests/halo-bench/halo-exchanger.cu diff --git a/halo-bench/halo_bench.cpp b/tests/halo-bench/halo_bench.cpp similarity index 100% rename from halo-bench/halo_bench.cpp rename to tests/halo-bench/halo_bench.cpp diff --git a/halo-bench/hpm.cpp b/tests/halo-bench/hpm.cpp similarity index 100% rename from halo-bench/hpm.cpp rename to tests/halo-bench/hpm.cpp diff --git a/halo-bench/mesh_distances.cpp b/tests/halo-bench/mesh_distances.cpp similarity index 100% rename from halo-bench/mesh_distances.cpp rename to tests/halo-bench/mesh_distances.cpp diff --git a/halo-bench/mesh_topo.cpp b/tests/halo-bench/mesh_topo.cpp similarity index 100% rename from halo-bench/mesh_topo.cpp rename to tests/halo-bench/mesh_topo.cpp diff --git a/halo-bench/osu_latency.c b/tests/halo-bench/osu_latency.c similarity index 100% rename from halo-bench/osu_latency.c rename to tests/halo-bench/osu_latency.c diff --git a/halo-bench/osu_latency_rdp.cpp b/tests/halo-bench/osu_latency_rdp.cpp similarity index 100% rename from halo-bench/osu_latency_rdp.cpp rename to tests/halo-bench/osu_latency_rdp.cpp diff --git a/halo-bench/scripts/env.sh b/tests/halo-bench/scripts/env.sh similarity index 100% rename from halo-bench/scripts/env.sh rename to tests/halo-bench/scripts/env.sh diff --git a/halo-bench/scripts/exp_12x12x2.sh b/tests/halo-bench/scripts/exp_12x12x2.sh similarity index 100% rename from halo-bench/scripts/exp_12x12x2.sh rename to tests/halo-bench/scripts/exp_12x12x2.sh diff --git a/halo-bench/scripts/exp_14x14x4.sh b/tests/halo-bench/scripts/exp_14x14x4.sh similarity index 100% rename from halo-bench/scripts/exp_14x14x4.sh rename to tests/halo-bench/scripts/exp_14x14x4.sh diff --git a/halo-bench/scripts/exp_3x3x3.sh b/tests/halo-bench/scripts/exp_3x3x3.sh similarity index 100% rename from halo-bench/scripts/exp_3x3x3.sh rename to tests/halo-bench/scripts/exp_3x3x3.sh diff --git a/halo-bench/scripts/exp_3x3x3_new.sh b/tests/halo-bench/scripts/exp_3x3x3_new.sh similarity index 100% rename from halo-bench/scripts/exp_3x3x3_new.sh rename to tests/halo-bench/scripts/exp_3x3x3_new.sh diff --git a/halo-bench/scripts/modules.sh b/tests/halo-bench/scripts/modules.sh similarity index 100% rename from halo-bench/scripts/modules.sh rename to tests/halo-bench/scripts/modules.sh diff --git a/halo-bench/scripts/unsetenv.sh b/tests/halo-bench/scripts/unsetenv.sh similarity index 100% rename from halo-bench/scripts/unsetenv.sh rename to tests/halo-bench/scripts/unsetenv.sh