Logo ROOT  
Reference Guide
 
Loading...
Searching...
No Matches
Evaluator.cxx
Go to the documentation of this file.
1/*
2 * Project: RooFit
3 * Authors:
4 * Jonas Rembser, CERN 2021
5 * Emmanouil Michalainas, CERN 2021
6 *
7 * Copyright (c) 2021, CERN
8 *
9 * Redistribution and use in source and binary forms,
10 * with or without modification, are permitted according to the terms
11 * listed in LICENSE (http://roofit.sourceforge.net/license.txt)
12 */
13
14/**
15\file Evaluator.cxx
16\class RooFit::Evaluator
17\ingroup Roofitcore
18
19Evaluates a RooAbsReal object in other ways than recursive graph
20traversal. Currently, it is being used for evaluating a RooAbsReal object and
21supplying the value to the minimizer, during a fit. The class scans the
22dependencies and schedules the computations in a secure and efficient way. The
23computations take place in the RooBatchCompute library and can be carried off
24by either the CPU or a CUDA-supporting GPU. The Evaluator class takes care
25of data transfers. An instance of this class is created every time
26RooAbsPdf::fitTo() is called and gets destroyed when the fitting ends.
27**/
28
29#include <RooFit/Evaluator.h>
30
31#include <RooAbsCategory.h>
32#include <RooAbsData.h>
33#include <RooAbsReal.h>
34#include <RooRealVar.h>
35#include <RooBatchCompute.h>
36#include <RooMsgService.h>
37#include <RooNameReg.h>
38#include <RooSimultaneous.h>
39
40#include <RooBatchCompute.h>
41
43#include "RooFitImplHelpers.h"
44
45#include <atomic>
46#include <iomanip>
47#include <mutex>
48#include <numeric>
49#include <set>
50#include <unordered_set>
51
52namespace RooFit {
53
54namespace {
55
56// To avoid deleted move assignment.
57template <class T>
58void assignSpan(std::span<T> &to, std::span<T> const &from)
59{
60 to = from;
61}
62
64{
65 // We have to exit early if the message stream is not active. Otherwise it's
66 // possible that this function skips logging because it thinks it has
67 // already logged, but actually it didn't.
68 if (!RooMsgService::instance().isActive(nullptr, RooFit::Fitting, RooFit::INFO)) {
69 return;
70 }
71
72 // Don't repeat logging architecture info if the useGPU option didn't change
73 {
74 // Second element of pair tracks whether this function has already been called
75 static std::pair<bool, bool> lastUseGPU;
76 if (lastUseGPU.second && lastUseGPU.first == useGPU)
77 return;
78 lastUseGPU = {useGPU, true};
79 }
80
81 auto log = [](std::string_view message) {
82 oocxcoutI(static_cast<RooAbsArg *>(nullptr), Fitting) << message << std::endl;
83 };
84
86 log("using generic CPU library compiled with no vectorizations");
87 } else {
88 log(std::string("using CPU computation library compiled with -m") + RooBatchCompute::cpuArchitectureName());
89 }
90 if (useGPU) {
91 log("using CUDA computation library");
92 }
93}
94
95/// Advise the user to implement CUDA support for a class that had to be
96/// evaluated on the CPU even though the CUDA backend was requested. Just like
97/// for the analogous message about the missing batch evaluation interface in
98/// RooAbsReal::doEval(), the message is only printed once per class, because
99/// the computation graph is evaluated many times, e.g. in every minimizer
100/// iteration.
101void logMissingCudaSupport(RooAbsArg const &arg)
102{
104 return;
105 }
106 static std::set<std::string> warnedClasses;
107 static std::mutex warnedClassesMutex;
108 std::scoped_lock guard{warnedClassesMutex};
109 if (warnedClasses.insert(arg.ClassName()).second) {
110 oocoutI(&arg, FastEvaluations) << "The class " << arg.ClassName()
111 << " could not be evaluated on the GPU because it doesn't support it."
112 << " Consider requesting or implementing it to benefit from a speed up."
113 << std::endl;
114 }
115}
116
117} // namespace
118
119/// A struct used by the Evaluator to store information on the RooAbsArgs in
120/// the computation graph.
121struct NodeInfo {
122
123 bool isScalar() const { return outputSize == 1; }
124
125 RooAbsArg *absArg = nullptr;
127
128 std::shared_ptr<RooBatchCompute::AbsBuffer> buffer;
129 std::size_t iNode = 0;
130 int remClients = 0;
132 bool fromArrayInput = false;
133 bool isVariable = false;
134 bool isDirty = true;
135 bool isCategory = false;
136 bool hasLogged = false;
137 bool computeInGPU = false;
138 bool isValueServer = false; // if this node is a value server to the top node
139 std::size_t outputSize = 1;
140 std::size_t lastSetValCount = std::numeric_limits<std::size_t>::max();
141 int lastCatVal = std::numeric_limits<int>::max();
142 double scalarBuffer = 0.0;
143 std::vector<NodeInfo *> serverInfos;
144 std::vector<NodeInfo *> clientInfos;
145
146 /// Check the servers of a node that has been computed and release its
147 /// resources if they are no longer needed. Buffers of nodes whose results
148 /// are copied between host and device (copyAfterEvaluation) must not be
149 /// released eagerly: their pinned host memory can still be the source of
150 /// an asynchronous copy that was enqueued on the CUDA stream, and a new
151 /// owner would overwrite it from the CPU without any stream ordering.
152 /// Those buffers are released at the beginning of the next evaluation
153 /// instead, after the stream was synchronized at the end of this one.
155 {
156 if (--remClients == 0 && !fromArrayInput && !copyAfterEvaluation) {
157 buffer.reset();
158 }
159 }
160};
161
162/// Construct a new Evaluator. The constructor analyzes and saves metadata about the graph,
163/// useful for the evaluation of it that will be done later. In case the CUDA mode is selected,
164/// there's also some CUDA-related initialization.
165///
166/// \param[in] absReal The RooAbsReal object that sits on top of the
167/// computation graph that we want to evaluate.
168/// \param[in] useGPU Whether the evaluation should be preferably done on the GPU.
170 : _topNode{const_cast<RooAbsReal &>(absReal)}, _useGPU{useGPU}
171{
173 if (useGPU && RooBatchCompute::initCUDA() != 0) {
174 throw std::runtime_error("Can't create Evaluator in CUDA mode because RooBatchCompute CUDA could not be loaded!");
175 }
176 // Some checks and logging of used architectures
178
181
184
186 if (useGPU) {
188 }
189
190 std::map<RooFit::Detail::DataKey, NodeInfo *> nodeInfos;
191
192 // Fill the ordered nodes list and initialize the node info structs.
193 _nodes.reserve(serverSet.size());
194 std::size_t iNode = 0;
195 for (RooAbsArg *arg : serverSet) {
196
197 _nodes.emplace_back();
198 auto &nodeInfo = _nodes.back();
199 _nodesMap[arg->namePtr()] = &nodeInfo;
200
201 nodeInfo.absArg = arg;
202 nodeInfo.originalOperMode = arg->operMode();
203 nodeInfo.iNode = iNode;
204 nodeInfos[arg] = &nodeInfo;
205
206 if (dynamic_cast<RooRealVar const *>(arg)) {
207 nodeInfo.isVariable = true;
208 } else {
209 arg->setDataToken(iNode);
210 }
211 if (dynamic_cast<RooAbsCategory const *>(arg)) {
212 nodeInfo.isCategory = true;
213 }
214
215 ++iNode;
216 }
217
218 for (NodeInfo &info : _nodes) {
219 info.serverInfos.reserve(info.absArg->servers().size());
220 for (RooAbsArg *server : info.absArg->servers()) {
221 if (server->isValueServer(*info.absArg)) {
222 auto *serverInfo = nodeInfos.at(server);
223 info.serverInfos.emplace_back(serverInfo);
224 serverInfo->clientInfos.emplace_back(&info);
225 }
226 }
227 }
228
229 // Figure out which nodes are value servers to the top node
230 _nodes.back().isValueServer = true; // the top node itself
231 for (auto iter = _nodes.rbegin(); iter != _nodes.rend(); ++iter) {
232 if (!iter->isValueServer)
233 continue;
234 for (auto &serverInfo : iter->serverInfos) {
235 serverInfo->isValueServer = true;
236 }
237 }
238
240
241 if (_useGPU) {
242 // Create the single CUDA stream on which all GPU computations and data
243 // transfers of this Evaluator are enqueued. The graph is evaluated in
244 // topological order, so ordering the operations by the stream is enough
245 // to guarantee correct results.
249 for (auto &info : _nodes) {
250 _evalContextCUDA.setConfig(info.absArg, cfg);
251 }
252 }
253}
254
255/// If there are servers with the same name that got de-duplicated in the
256/// `_nodes` list, we need to set their data tokens too. We find such nodes by
257/// visiting the servers of every known node.
259{
260 for (NodeInfo &info : _nodes) {
261 std::size_t iValueServer = 0;
262 for (RooAbsArg *server : info.absArg->servers()) {
263 if (server->isValueServer(*info.absArg)) {
264 auto *knownServer = info.serverInfos[iValueServer]->absArg;
265 if (knownServer->hasDataToken()) {
266 server->setDataToken(knownServer->dataToken());
267 }
268 ++iValueServer;
269 }
270 }
271 }
272}
273
274void Evaluator::setInput(std::string const &name, std::span<const double> inputArray, bool isOnDevice)
275{
276 if (isOnDevice && !_useGPU) {
277 throw std::runtime_error("Evaluator can only take device array as input in CUDA mode!");
278 }
279
280 // Check if "name" is used in the computation graph. If yes, add the span to
281 // the data map and set the node info accordingly.
282
283 auto found = _nodesMap.find(RooNameReg::ptr(name.c_str()));
284
285 if (found == _nodesMap.end())
286 return;
287
289
290 // Invalidate the caches that reducer nodes key on the input data, like the
291 // cached sum of event weights in RooNLLVarNew. The counter is global so
292 // that generation values can never alias between different Evaluators.
293 {
294 static std::atomic<std::size_t> nextInputGeneration{1};
295 const std::size_t gen = ++nextInputGeneration;
298 }
299
300 NodeInfo &info = *found->second;
301
302 info.fromArrayInput = true;
303 info.absArg->setDataToken(info.iNode);
304 info.outputSize = inputArray.size();
305
306 if (!_useGPU) {
308 return;
309 }
310
311 if (info.outputSize <= 1) {
312 // Empty or scalar observables from the data don't need to be
313 // copied to the GPU.
316 return;
317 }
318
319 // For simplicity, we put the data on both host and device for
320 // now. This could be optimized by inspecting the clients of the
321 // variable.
322 if (isOnDevice) {
324 auto gpuSpan = _evalContextCUDA.at(info.absArg);
325 info.buffer = _bufferManager->makeCpuBuffer(gpuSpan.size());
326 info.buffer->assignFromDevice(gpuSpan);
327 _evalContextCPU.set(info.absArg, {info.buffer->hostReadPtr(), gpuSpan.size()});
328 } else {
330 auto cpuSpan = _evalContextCPU.at(info.absArg);
331 info.buffer = _bufferManager->makeGpuBuffer(cpuSpan.size());
332 info.buffer->assignFromHost(cpuSpan);
333 _evalContextCUDA.set(info.absArg, {info.buffer->deviceReadPtr(), cpuSpan.size()});
334 }
335}
336
338{
339 std::map<RooFit::Detail::DataKey, std::size_t> sizeMap;
340 for (auto &info : _nodes) {
341 if (info.fromArrayInput) {
342 sizeMap[info.absArg] = info.outputSize;
343 } else {
344 // any buffer for temporary results is invalidated by resetting the output sizes
345 info.buffer.reset();
346 }
347 }
348
349 auto outputSizeMap =
350 RooFit::BatchModeDataHelpers::determineOutputSizes(_topNode, [&](RooFit::Detail::DataKey key) -> int {
351 auto found = sizeMap.find(key);
352 return found != sizeMap.end() ? found->second : -1;
353 });
354
355 for (auto &info : _nodes) {
356 info.outputSize = outputSizeMap.at(info.absArg);
357 info.isDirty = true;
358 }
359
360 if (_useGPU) {
361 markGPUNodes();
362 }
363
365}
366
368{
369 for (auto &info : _nodes) {
370 if (!info.isVariable) {
371 info.absArg->resetDataToken();
372 }
373 }
374 if (_cudaStream) {
376 }
377}
378
380{
381 using namespace Detail;
382
383 const std::size_t nOut = info.outputSize;
384
385 double *buffer = nullptr;
386 if (nOut == 1) {
387 buffer = &info.scalarBuffer;
388 if (_useGPU) {
389 _evalContextCUDA.set(node, {buffer, nOut});
390 }
391 } else {
392 // Advising to implement the CUDA evaluation makes only sense if the batch was not a scalar.
393 // Otherwise, there would be no speedup benefit.
394 if (!info.hasLogged && _useGPU) {
396 info.hasLogged = true;
397 }
398 if (!info.buffer) {
399 info.buffer = info.copyAfterEvaluation ? _bufferManager->makePinnedBuffer(nOut, _cudaStream)
400 : _bufferManager->makeCpuBuffer(nOut);
401 }
402 buffer = info.buffer->hostWritePtr();
403 }
405 _evalContextCPU.set(node, {buffer, nOut});
406 if (nOut > 1) {
408 }
409 if (info.isCategory) {
410 auto nodeAbsCategory = static_cast<RooAbsCategory const *>(node);
411 if (nOut == 1) {
412 buffer[0] = nodeAbsCategory->getCurrentIndex();
413 } else {
414 throw std::runtime_error("RooFit::Evaluator - non-scalar category values are not supported!");
415 }
416 } else {
417 auto nodeAbsReal = static_cast<RooAbsReal const *>(node);
419 }
422 if (info.copyAfterEvaluation) {
423 // The deviceReadPtr() call triggers the copy of the result to the GPU.
424 // The copy is ordered by the CUDA stream, so GPU clients enqueued later
425 // will see the result without any further synchronization.
426 _evalContextCUDA.set(node, {info.buffer->deviceReadPtr(), nOut});
427 }
428}
429
430/// Process a variable in the computation graph. This is a separate non-inlined
431/// function such that we can see in performance profiles how long this takes.
433{
434 RooAbsArg *node = nodeInfo.absArg;
435 auto *var = static_cast<RooRealVar const *>(node);
436 if (nodeInfo.lastSetValCount != var->valueResetCounter()) {
437 nodeInfo.lastSetValCount = var->valueResetCounter();
438 for (NodeInfo *clientInfo : nodeInfo.clientInfos) {
439 clientInfo->isDirty = true;
440 }
442 nodeInfo.isDirty = false;
443 }
444}
445
446/// Process a category in the computation graph. This is a separate non-inlined
447/// function such that we can see in performance profiles how long this takes.
449{
450 RooAbsArg *node = nodeInfo.absArg;
451 auto *cat = static_cast<RooAbsCategory const *>(node);
452 if (nodeInfo.lastCatVal != cat->getCurrentIndex()) {
453 nodeInfo.lastCatVal = cat->getCurrentIndex();
454 for (NodeInfo *clientInfo : nodeInfo.clientInfos) {
455 clientInfo->isDirty = true;
456 }
458 nodeInfo.isDirty = false;
459 }
460}
461
462/// Flags all the clients of a given node dirty. This is a separate non-inlined
463/// function such that we can see in performance profiles how long this takes.
465{
466 for (NodeInfo *clientInfo : nodeInfo.clientInfos) {
467 clientInfo->isDirty = true;
468 }
469}
470
471/// Returns the value of the top node in the computation graph
472std::span<const double> Evaluator::run()
473{
476
478
479 // Discard leftover deferred actions in case a previous evaluation was
480 // aborted by an exception.
483
484 if (_useGPU) {
485 return getValHeterogeneous();
486 }
487
488 for (auto &nodeInfo : _nodes) {
489 if (!nodeInfo.fromArrayInput) {
490 if (nodeInfo.isVariable) {
492 } else if (nodeInfo.isCategory) {
494 } else {
495 if (nodeInfo.isDirty) {
498 nodeInfo.isDirty = false;
499 }
500 }
501 }
502 }
503
505 action();
506 }
508
509 // return the final output
510 return _evalContextCPU.at(&_topNode);
511}
512
513/// Returns the value of the top node in the computation graph
514std::span<const double> Evaluator::getValHeterogeneous()
515{
516 for (auto &info : _nodes) {
517 info.remClients = info.clientInfos.size();
518 if (info.buffer && !info.fromArrayInput) {
519 info.buffer.reset();
520 }
521 }
522
523 // Iterate over the nodes in topological order. Nodes that are computed on
524 // the GPU only enqueue their computation on the single CUDA stream and
525 // return immediately, so independent CPU nodes that come later in the
526 // ordering naturally overlap with the GPU computations. Ordering by the
527 // stream guarantees that GPU nodes see the results of their GPU servers,
528 // and host-side reads of GPU results synchronize on the stream in the
529 // buffer implementation.
530 try {
531 for (auto &info : _nodes) {
532 if (!info.fromArrayInput) {
533 if (info.computeInGPU) {
535 } else {
536 computeCPUNode(info.absArg, info);
537 }
538 }
539
540 // Release the buffers of server nodes that are no longer needed. For
541 // device-only buffers this is safe to do right away even if GPU work
542 // is still in flight, because any reuse of a released device buffer
543 // happens through operations that are enqueued later on the same
544 // stream. Pinned buffers are exempted from the eager release, see
545 // the comment in NodeInfo::decrementRemainingClients().
546 for (auto *serverInfo : info.serverInfos) {
547 serverInfo->decrementRemainingClients();
548 }
549 }
550 } catch (...) {
551 // The evaluation was aborted, but readbacks that compute() calls
552 // deferred may still be armed. Deliver them now, while the destination
553 // memory in the nodes of the computation graph is guaranteed to be
554 // alive, so that no armed readback survives into a later evaluation.
555 try {
557 } catch (...) {
558 // The stream is in an unrecoverable error state. The deferred
559 // readbacks are dropped together with the scratch memory when the
560 // stream gets deleted.
561 }
564 throw;
565 }
566
567 // Ensure that all enqueued GPU work has completed when run() returns. For
568 // the usual likelihood evaluations this is mostly a no-op, because the
569 // final reduction has synchronized the stream already. It also guarantees
570 // that recycling the buffers at the beginning of the next evaluation is
571 // safe, and it delivers the deferred readbacks like the evaluation error
572 // counters.
574
575 // Run the deferred actions now that all results have arrived on the host,
576 // e.g. the logging of evaluation errors that were counted on the GPU.
577 // Nodes evaluated on the CPU register their actions in the CPU context,
578 // so both contexts are drained.
579 for (auto *ctx : {&_evalContextCUDA, &_evalContextCPU}) {
580 for (auto &action : ctx->_deferredActions) {
581 action();
582 }
583 ctx->_deferredActions.clear();
584 }
585
586 // return the final value
588}
589
590/// Enqueue the computation of a node on the GPU.
592{
593 using namespace Detail;
594
595 auto node = static_cast<RooAbsReal const *>(info.absArg);
596
597 const std::size_t nOut = info.outputSize;
598
599 double *buffer = nullptr;
600 if (nOut == 1) {
601 buffer = &info.scalarBuffer;
602 _evalContextCPU.set(node, {buffer, nOut});
603 } else {
604 info.buffer = info.copyAfterEvaluation ? _bufferManager->makePinnedBuffer(nOut, _cudaStream)
605 : _bufferManager->makeGpuBuffer(nOut);
606 buffer = info.buffer->deviceWritePtr();
607 }
609 _evalContextCUDA.set(node, {buffer, nOut});
610 node->doEval(_evalContextCUDA);
611 if (info.copyAfterEvaluation) {
612 // The hostReadPtr() call triggers the copy of the result to the host,
613 // which waits for the enqueued computation via the CUDA stream.
614 _evalContextCPU.set(node, {info.buffer->hostReadPtr(), nOut});
615 }
616}
617
618/// Decides which nodes are assigned to the GPU in a CUDA fit.
620{
621 // Decide which nodes get evaluated on the GPU: we select nodes that support
622 // CUDA evaluation and have at least one input of size greater than one.
623 for (auto &info : _nodes) {
624 info.computeInGPU = false;
625 if (!info.absArg->canComputeBatchWithCuda()) {
626 continue;
627 }
628 for (NodeInfo const *serverInfo : info.serverInfos) {
629 if (serverInfo->outputSize > 1) {
630 info.computeInGPU = true;
631 break;
632 }
633 }
634 }
635
636 // In a second pass, figure out which nodes need to copy over their results.
637 for (auto &info : _nodes) {
638 info.copyAfterEvaluation = false;
639 // scalar nodes don't need copying
640 if (!info.isScalar()) {
641 for (auto *clientInfo : info.clientInfos) {
642 if (info.computeInGPU != clientInfo->computeInGPU) {
643 info.copyAfterEvaluation = true;
644 break;
645 }
646 }
647 }
648 }
649}
650
651/// \brief Sets the number of threads to use for the evaluation of a single node.
652///
653/// With a value greater than one, the computation functions and reductions of
654/// the CPU backend process large batches multi-threaded, using up to the
655/// given number of threads. Nodes evaluated on the CPU with fewer events than
656/// an internal threshold are still evaluated single-threaded, so requesting
657/// multiple threads never introduces scheduling overhead for small fits.
658void Evaluator::setNThreads(int nThreads)
659{
660 for (auto &info : _nodes) {
661 if (info.isVariable) {
662 continue;
663 }
665 cfg.setNThreads(nThreads);
666 _evalContextCPU.setConfig(info.absArg, cfg);
667 }
668}
669
670/// Temporarily change the operation mode of a RooAbsArg until the
671/// Evaluator gets deleted.
673{
674 if (!_operModeChanges)
675 _operModeChanges = std::make_unique<ChangeOperModeRAII>();
676 _operModeChanges->change(arg, opMode);
677}
678
679// Change the operation modes of all RooAbsArgs in the computation graph.
680// The changes are reset when the returned RAII object goes out of scope.
681//
682// We also walk transitively through value clients of the nodes to cover any
683// node that RooAbsReal::doEval (the fallback scalar implementation) might
684// inadvertently propagate the ADirty mode to via its recursive restore: that
685// helper sets servers temporarily to AClean and then calls
686// setOperMode(oldOperMode) to restore, which recurses to value clients when
687// oldOperMode is ADirty. If we did not protect those clients here, any node
688// outside the computation graph that shares a fundamental (e.g. a parameter
689// like a RooRealVar) would be left permanently in ADirty after the first
690// minimization, dramatically slowing down later scalar evaluations (for
691// example on pdfs held by the legacy test statistics' internal cache).
692std::unique_ptr<ChangeOperModeRAII> Evaluator::setOperModes(RooAbsArg::OperMode opMode)
693{
694 auto out = std::make_unique<ChangeOperModeRAII>();
695 std::unordered_set<RooAbsArg *> visited;
696
697 std::vector<RooAbsArg *> queue;
698 queue.reserve(_nodes.size());
699 for (auto &info : _nodes) {
700 queue.push_back(info.absArg);
701 }
702
703 while (!queue.empty()) {
704 RooAbsArg *node = queue.back();
705 queue.pop_back();
706 if (!visited.insert(node).second)
707 continue;
708
709 out->change(node, opMode);
710
711 // Only follow value-client links: that is exactly the propagation path
712 // used by RooAbsArg::setOperMode with mode==ADirty.
713 if (opMode == RooAbsArg::ADirty) {
714 for (auto *client : node->valueClients()) {
715 queue.push_back(client);
716 }
717 }
718 }
719 return out;
720}
721
722void Evaluator::print(std::ostream &os)
723{
724 std::cout << "--- RooFit BatchMode evaluation ---\n";
725
726 std::vector<int> widths{9, 37, 20, 9, 10, 20};
727
728 auto printElement = [&](int iCol, auto const &t) {
729 const char separator = ' ';
730 os << separator << std::left << std::setw(widths[iCol]) << std::setfill(separator) << t;
731 os << "|";
732 };
733
734 auto printHorizontalRow = [&]() {
735 int n = 0;
736 for (int w : widths) {
737 n += w + 2;
738 }
739 for (int i = 0; i < n; i++) {
740 os << '-';
741 }
742 os << "|\n";
743 };
744
746
747 os << "|";
748 printElement(0, "Index");
749 printElement(1, "Name");
750 printElement(2, "Class");
751 printElement(3, "Size");
752 printElement(4, "From Data");
753 printElement(5, "1st value");
754 std::cout << "\n";
755
757
758 for (std::size_t iNode = 0; iNode < _nodes.size(); ++iNode) {
759 auto &nodeInfo = _nodes[iNode];
760 RooAbsArg *node = nodeInfo.absArg;
761
762 auto span = _evalContextCPU.at(node);
763
764 os << "|";
765 printElement(0, iNode);
766 printElement(1, node->GetName());
767 printElement(2, node->ClassName());
768 printElement(3, nodeInfo.outputSize);
769 printElement(4, nodeInfo.fromArrayInput);
770 printElement(5, span[0]);
771
772 std::cout << "\n";
773 }
774
776}
777
778/// Gets all the parameters of the RooAbsReal. This is in principle not
779/// necessary, because we can always ask the RooAbsReal itself, but the
780/// Evaluator has the cached information to get the answer quicker.
781/// Therefore, this is not meant to be used in general, just where it matters.
782/// \warning If we find another solution to get the parameters efficiently,
783/// this function might be removed without notice.
785{
786 RooArgSet parameters;
787 for (auto &nodeInfo : _nodes) {
788 if (nodeInfo.isValueServer && nodeInfo.absArg->isFundamental()) {
789 parameters.add(*nodeInfo.absArg);
790 }
791 }
792 // Just like in RooAbsArg::getParameters(), we sort the parameters alphabetically.
793 parameters.sort();
794 return parameters;
795}
796
797/// \brief Sets the offset mode for evaluation.
798///
799/// This function sets the offset mode for evaluation to the specified mode.
800/// It updates the offset mode for both CPU and CUDA evaluation contexts.
801///
802/// \param mode The offset mode to be set.
803///
804/// \note This function marks reducer nodes as dirty if the offset mode is
805/// changed, because only reducer nodes can use offsetting.
807{
809 return;
810
813
814 for (auto &nodeInfo : _nodes) {
815 if (nodeInfo.absArg->isReducerNode()) {
816 nodeInfo.isDirty = true;
817 }
818 }
819}
820
821} // namespace RooFit
#define oocoutI(o, a)
#define oocxcoutI(o, a)
ROOT::Detail::TRangeCast< T, true > TRangeDynCast
TRangeDynCast is an adapter class that allows the typed iteration through a TCollection.
Option_t Option_t TPoint TPoint const char mode
char name[80]
Definition TGX11.cxx:142
const_iterator end() const
Common abstract base class for objects that represent a value and a "shape" in RooFit.
Definition RooAbsArg.h:76
const TNamed * namePtr() const
De-duplicated pointer to this object's name.
Definition RooAbsArg.h:482
void setDataToken(std::size_t index)
Sets the token for retrieving results in the BatchMode. For internal use only.
const RefCountList_t & valueClients() const
List of all value clients of this object. Value clients receive value updates.
Definition RooAbsArg.h:139
OperMode operMode() const
Query the operation mode of this node.
Definition RooAbsArg.h:398
A space to attach TBranches.
virtual bool add(const RooAbsArg &var, bool silent=false)
Add the specified argument to list.
void sort(bool reverse=false)
Sort collection using std::sort and name comparison.
Abstract base class for objects that represent a real value and implements functionality common to al...
Definition RooAbsReal.h:63
RooArgSet is a container object that can hold multiple RooAbsArg objects.
Definition RooArgSet.h:24
Minimal configuration struct to steer the evaluation of a single node with the RooBatchCompute librar...
void setCudaStream(CudaInterface::CudaStream *cudaStream)
void setNThreads(int nThreads)
Number of threads to use for CPU batch computations and reductions.
virtual void synchronizeCudaStream(CudaInterface::CudaStream *) const =0
Wait until all work that was enqueued on the stream has completed.
virtual std::unique_ptr< AbsBufferManager > createBufferManager() const =0
virtual CudaInterface::CudaStream * newCudaStream() const =0
virtual void deleteCudaStream(CudaInterface::CudaStream *) const =0
std::size_t _inputGeneration
std::vector< std::function< void()> > _deferredActions
void set(RooAbsArg const *arg, std::span< const double > const &span)
Definition EvalContext.h:92
std::span< const double > at(RooAbsArg const *arg, RooAbsArg const *caller=nullptr)
void enableVectorBuffers(bool enable)
OffsetMode _offsetMode
RooBatchCompute::Config config(RooAbsArg const *arg) const
void setConfig(RooAbsArg const *arg, RooBatchCompute::Config const &config)
std::span< double > _currentOutput
void resize(std::size_t n)
void print(std::ostream &os)
void setClientsDirty(NodeInfo &nodeInfo)
Flags all the clients of a given node dirty.
std::unique_ptr< ChangeOperModeRAII > setOperModes(RooAbsArg::OperMode opMode)
RooArgSet getParameters() const
Gets all the parameters of the RooAbsReal.
void setOffsetMode(RooFit::EvalContext::OffsetMode)
Sets the offset mode for evaluation.
void syncDataTokens()
If there are servers with the same name that got de-duplicated in the _nodes list,...
const bool _useGPU
Definition Evaluator.h:68
std::unordered_map< TNamed const *, NodeInfo * > _nodesMap
Definition Evaluator.h:74
std::unique_ptr< ChangeOperModeRAII > _operModeChanges
Definition Evaluator.h:75
std::vector< NodeInfo > _nodes
Definition Evaluator.h:73
bool _needToUpdateOutputSizes
Definition Evaluator.h:70
std::span< const double > getValHeterogeneous()
Returns the value of the top node in the computation graph.
std::span< const double > run()
Returns the value of the top node in the computation graph.
Evaluator(const RooAbsReal &absReal, bool useGPU=false)
Construct a new Evaluator.
void setNThreads(int nThreads)
Sets the number of threads to use for the evaluation of a single node.
void processVariable(NodeInfo &nodeInfo)
Process a variable in the computation graph.
void processCategory(NodeInfo &nodeInfo)
Process a category in the computation graph.
RooBatchCompute::CudaInterface::CudaStream * _cudaStream
Definition Evaluator.h:77
std::unique_ptr< RooBatchCompute::AbsBufferManager > _bufferManager
Definition Evaluator.h:66
void markGPUNodes()
Decides which nodes are assigned to the GPU in a CUDA fit.
void assignToGPU(NodeInfo &info)
Enqueue the computation of a node on the GPU.
void setInput(std::string const &name, std::span< const double > inputArray, bool isOnDevice)
RooFit::EvalContext _evalContextCUDA
Definition Evaluator.h:72
RooFit::EvalContext _evalContextCPU
Definition Evaluator.h:71
void computeCPUNode(const RooAbsArg *node, NodeInfo &info)
void setOperMode(RooAbsArg *arg, RooAbsArg::OperMode opMode)
Temporarily change the operation mode of a RooAbsArg until the Evaluator gets deleted.
RooAbsReal & _topNode
Definition Evaluator.h:67
static RooMsgService & instance()
Return reference to singleton instance.
static const TNamed * ptr(const char *stringPtr)
Return a unique TNamed pointer for given C++ string.
Variable that can be changed from the outside.
Definition RooRealVar.h:37
const char * GetName() const override
Returns name of object.
Definition TNamed.h:49
virtual const char * ClassName() const
Returns name of class to which the object belongs.
Definition TObject.cxx:226
RVec< PromoteType< T > > log(const RVec< T > &v)
Definition RVec.hxx:1821
const Int_t n
Definition legend1.C:16
R__EXTERN RooBatchComputeInterface * dispatchCUDA
std::string cpuArchitectureName()
R__EXTERN RooBatchComputeInterface * dispatchCPU
This dispatch pointer points to an implementation of the compute library, provided one has been loade...
Architecture cpuArchitecture()
int initCPU()
Inspect hardware capabilities, and load the optimal library for RooFit computations.
The namespace RooFit contains mostly switches that change the behaviour of functions of PDFs (or othe...
Definition CodegenImpl.h:73
@ FastEvaluations
void getSortedComputationGraph(RooAbsArg const &func, RooArgSet &out)
A struct used by the Evaluator to store information on the RooAbsArgs in the computation graph.
RooAbsArg * absArg
bool isScalar() const
std::size_t iNode
std::size_t lastSetValCount
std::vector< NodeInfo * > serverInfos
RooAbsArg::OperMode originalOperMode
std::size_t outputSize
std::vector< NodeInfo * > clientInfos
std::shared_ptr< RooBatchCompute::AbsBuffer > buffer
void decrementRemainingClients()
Check the servers of a node that has been computed and release its resources if they are no longer ne...