Split simulationWork.useGpuBufferOps into separate x and f flags

[alexxy/gromacs.git] / src / gromacs / mdrun / runner.cpp
diff --git a/src/gromacs/mdrun/runner.cpp b/src/gromacs/mdrun/runner.cpp

index 8c3fbdf517855a22e7f5e9689a515619c193e7c5..6e245a2df99f4c3d0b2cc0786a6c6626f9af2637 100644 (file)
--- a/src/gromacs/mdrun/runner.cpp
+++ b/src/gromacs/mdrun/runner.cpp
@@ -203,8 +203,8 @@ static DevelopmentFeatureFlags manageDevelopmentFeatures(const gmx::MDLogger& md
  {
      DevelopmentFeatureFlags devFlags;
  
-    devFlags.enableGpuBufferOps =
-            GMX_GPU_CUDA && useGpuForNonbonded && (getenv("GMX_USE_GPU_BUFFER_OPS") != nullptr);
+    devFlags.enableGpuBufferOps = (GMX_GPU_CUDA || GMX_GPU_SYCL) && useGpuForNonbonded
+                                  && (getenv("GMX_USE_GPU_BUFFER_OPS") != nullptr);
      devFlags.enableGpuHaloExchange = GMX_MPI && GMX_GPU_CUDA && getenv("GMX_GPU_DD_COMMS") != nullptr;
      devFlags.forceGpuUpdateDefault = (getenv("GMX_FORCE_UPDATE_DEFAULT_GPU") != nullptr) || GMX_FAHCORE;
      devFlags.enableGpuPmePPComm = GMX_MPI && GMX_GPU_CUDA && getenv("GMX_GPU_PME_PP_COMMS") != nullptr;
@@ -907,6 +907,7 @@ int Mdrunner::mdrunner()
                      hw_opt.nthreads_tmpi);
              useGpuForPme = decideWhetherToUseGpusForPmeWithThreadMpi(useGpuForNonbonded,
                                                                       pmeTarget,
+                                                                     pmeFftTarget,
                                                                       numAvailableDevices,
                                                                       userGpuTaskAssignment,
                                                                       *hwinfo_,
@@ -937,7 +938,7 @@ int Mdrunner::mdrunner()
          // master and spawned threads joins at the end of this block.
      }
  
-    GMX_RELEASE_ASSERT(ms || simulationCommunicator != MPI_COMM_NULL,
+    GMX_RELEASE_ASSERT(!GMX_MPI || ms || simulationCommunicator != MPI_COMM_NULL,
                         "Must have valid communicator unless running a multi-simulation");
      CommrecHandle crHandle = init_commrec(simulationCommunicator);
      t_commrec*    cr       = crHandle.get();
@@ -969,13 +970,6 @@ int Mdrunner::mdrunner()
      GMX_RELEASE_ASSERT(inputrec != nullptr, "All ranks should have a valid inputrec now");
      partialDeserializedTpr.reset(nullptr);
  
-    GMX_RELEASE_ASSERT(
-            !inputrec->useConstantAcceleration,
-            "Linear acceleration has been removed in GROMACS 2022, and was broken for many years "
-            "before that. Use GROMACS 4.5 or earlier if you need this feature.");
-
-    // Now we decide whether to use the domain decomposition machinery.
-    // Note that this does not necessarily imply actually using multiple domains.
      // Now the number of ranks is known to all ranks, and each knows
      // the inputrec read by the master rank. The ranks can now all run
      // the task-deciding functions and will agree on the result
@@ -1012,6 +1006,7 @@ int Mdrunner::mdrunner()
                  gpusWereDetected);
          useGpuForPme    = decideWhetherToUseGpusForPme(useGpuForNonbonded,
                                                      pmeTarget,
+                                                    pmeFftTarget,
                                                      userGpuTaskAssignment,
                                                      *hwinfo_,
                                                      *inputrec,
@@ -1408,22 +1403,6 @@ int Mdrunner::mdrunner()
      int                deviceId   = -1;
      DeviceInformation* deviceInfo = gpuTaskAssignments.initDevice(&deviceId);
  
-    // timing enabling - TODO put this in gpu_utils (even though generally this is just option handling?)
-    bool useTiming = true;
-
-    if (GMX_GPU_CUDA)
-    {
-        /* WARNING: CUDA timings are incorrect with multiple streams.
-         *          This is the main reason why they are disabled by default.
-         */
-        // TODO: Consider turning on by default when we can detect nr of streams.
-        useTiming = (getenv("GMX_ENABLE_GPU_TIMING") != nullptr);
-    }
-    else if (GMX_GPU_OPENCL)
-    {
-        useTiming = (getenv("GMX_DISABLE_GPU_TIMING") == nullptr);
-    }
-
      // TODO Currently this is always built, yet DD partition code
      // checks if it is built before using it. Probably it should
      // become an MDModule that is made only when another module
@@ -1511,8 +1490,9 @@ int Mdrunner::mdrunner()
          {
              dd_setup_dlb_resource_sharing(cr, deviceId);
          }
-        deviceStreamManager = std::make_unique<DeviceStreamManager>(
-                *deviceInfo, havePPDomainDecomposition(cr), runScheduleWork.simulationWork, useTiming);
+        const bool useGpuTiming = decideGpuTimingsUsage();
+        deviceStreamManager     = std::make_unique<DeviceStreamManager>(
+                *deviceInfo, havePPDomainDecomposition(cr), runScheduleWork.simulationWork, useGpuTiming);
      }
  
      // If the user chose a task assignment, give them some hints
@@ -2016,7 +1996,7 @@ int Mdrunner::mdrunner()
              makeBondedLinks(cr->dd, mtop, fr->atomInfoForEachMoleculeBlock);
          }
  
-        if (runScheduleWork.simulationWork.useGpuBufferOps)
+        if (runScheduleWork.simulationWork.useGpuFBufferOps)
          {
              fr->gpuForceReduction[gmx::AtomLocality::Local] = std::make_unique<gmx::GpuForceReduction>(
                      deviceStreamManager->context(),
@@ -2031,7 +2011,7 @@ int Mdrunner::mdrunner()
          std::unique_ptr<gmx::StatePropagatorDataGpu> stateGpu;
          if (gpusWereDetected
              && ((runScheduleWork.simulationWork.useGpuPme && thisRankHasDuty(cr, DUTY_PME))
-                || runScheduleWork.simulationWork.useGpuBufferOps))
+                || runScheduleWork.simulationWork.useGpuXBufferOps))
          {
              GpuApiCallBehavior transferKind =
                      (inputrec->eI == IntegrationAlgorithm::MD && !doRerun && !useModularSimulator)