Logo ROOT  
Reference Guide
 
Loading...
Searching...
No Matches
RClusterPool.cxx
Go to the documentation of this file.
1/// \file RClusterPool.cxx
2/// \author Jakob Blomer <jblomer@cern.ch>
3/// \date 2020-03-11
4/// \warning This is part of the ROOT 7 prototype! It will change without notice. It might trigger earthquakes. Feedback
5/// is welcome!
6
7/*************************************************************************
8 * Copyright (C) 1995-2020, Rene Brun and Fons Rademakers. *
9 * All rights reserved. *
10 * *
11 * For the licensing terms see $ROOTSYS/LICENSE. *
12 * For the list of contributors see $ROOTSYS/README/CREDITS. *
13 *************************************************************************/
14
15#include <ROOT/RClusterPool.hxx>
17#include <ROOT/RPageStorage.hxx>
18
19#include <TError.h>
20
21#include <algorithm>
22#include <chrono>
23#include <future>
24#include <iostream>
25#include <iterator>
26#include <map>
27#include <memory>
28#include <mutex>
29#include <set>
30#include <utility>
31
33{
34 if (fClusterKey.fClusterId == other.fClusterKey.fClusterId) {
35 if (fClusterKey.fPhysicalColumnSet.size() == other.fClusterKey.fPhysicalColumnSet.size()) {
36 for (auto itr1 = fClusterKey.fPhysicalColumnSet.begin(), itr2 = other.fClusterKey.fPhysicalColumnSet.begin();
38 if (*itr1 == *itr2)
39 continue;
40 return *itr1 < *itr2;
41 }
42 // *this == other
43 return false;
44 }
45 return fClusterKey.fPhysicalColumnSet.size() < other.fClusterKey.fPhysicalColumnSet.size();
46 }
47 return fClusterKey.fClusterId < other.fClusterKey.fClusterId;
48}
49
52{
54
56 fCounters = std::make_unique<RCounters>(
57 RCounters{*fMetrics.MakeCounter<RNTupleAtomicCounter *>("nCluster", "", "number of currently cached clusters")});
58}
59
61{
62 StopBackgroundThread();
63}
64
66{
67 if (fThreadIo.joinable())
68 return;
69
70 fThreadIo = std::thread(&RClusterPool::ExecReadClusters, this);
71}
72
74{
75 if (!fThreadIo.joinable())
76 return;
77
78 {
79 // Controlled shutdown of the I/O thread
80 std::unique_lock<std::mutex> lock(fLockWorkQueue);
81 fReadQueue.emplace_back(RReadItem());
82 fCvHasReadWork.notify_one();
83 }
84 fThreadIo.join();
85}
86
88{
89 std::deque<RReadItem> readItems;
90 while (true) {
91 {
92 std::unique_lock<std::mutex> lock(fLockWorkQueue);
93 fCvHasReadWork.wait(lock, [&]{ return !fReadQueue.empty(); });
94 std::swap(readItems, fReadQueue);
95 }
96
97 while (!readItems.empty()) {
98 std::vector<RCluster::RKey> clusterKeys;
99 std::int64_t bunchId = -1;
100 for (unsigned i = 0; i < readItems.size(); ++i) {
101 const auto &item = readItems[i];
102 // `kInvalidDescriptorId` is used as a marker for thread cancellation. Such item causes the
103 // thread to terminate; thus, it must appear last in the queue.
104 if (R__unlikely(item.fClusterKey.fClusterId == ROOT::kInvalidDescriptorId)) {
105 R__ASSERT(i == (readItems.size() - 1));
106 return;
107 }
108 if ((bunchId >= 0) && (item.fBunchId != bunchId))
109 break;
110 bunchId = item.fBunchId;
111 clusterKeys.emplace_back(item.fClusterKey);
112 }
113
114 auto clusters = fPageSource.LoadClusters(clusterKeys);
115 for (std::size_t i = 0; i < clusters.size(); ++i) {
116 readItems[i].fPromise.set_value(std::move(clusters[i]));
117 }
118 readItems.erase(readItems.begin(), readItems.begin() + clusters.size());
119 }
120 } // while (true)
121}
122
123namespace {
124
125/// Helper class for the (cluster, column list) pairs that should be loaded in the background
126class RProvides {
128 using ColumnSet_t = ROOT::Internal::RCluster::ColumnSet_t;
129
130public:
131 struct RInfo {
132 std::int64_t fBunchId = -1;
133 std::int64_t fFlags = 0;
134 ColumnSet_t fPhysicalColumnSet;
135 };
136
137 static constexpr std::int64_t kFlagRequired = 0x01;
138 static constexpr std::int64_t kFlagLast = 0x02;
139
140private:
141 std::map<DescriptorId_t, RInfo> fMap;
142
143public:
144 void Insert(DescriptorId_t clusterId, const RInfo &info)
145 {
146 fMap.emplace(clusterId, info);
147 }
148
149 bool Contains(DescriptorId_t clusterId) {
150 return fMap.count(clusterId) > 0;
151 }
152
153 std::size_t GetSize() const { return fMap.size(); }
154
155 void Erase(DescriptorId_t clusterId, const ColumnSet_t &physicalColumns)
156 {
157 auto itr = fMap.find(clusterId);
158 if (itr == fMap.end())
159 return;
160 ColumnSet_t d;
161 std::copy_if(itr->second.fPhysicalColumnSet.begin(), itr->second.fPhysicalColumnSet.end(),
162 std::inserter(d, d.end()),
163 [&physicalColumns](DescriptorId_t needle) { return physicalColumns.count(needle) == 0; });
164 if (d.empty()) {
165 fMap.erase(itr);
166 } else {
167 itr->second.fPhysicalColumnSet = d;
168 }
169 }
170
171 decltype(fMap)::iterator begin() { return fMap.begin(); }
172 decltype(fMap)::iterator end() { return fMap.end(); }
173};
174
175} // anonymous namespace
176
179{
180 StartBackgroundThread(); // ensure that the thread is started (no-op if it is already running)
181
182 std::unordered_set<ROOT::DescriptorId_t> keep{fPageSource.GetPinnedClusters()};
183 for (auto cid : fPageSource.GetPinnedClusters()) {
184 for (ROOT::DescriptorId_t i = 1, next = cid; i < 2 * fClusterBunchSize; ++i) {
185 const auto currentId = next;
186 auto descriptorGuard = fPageSource.FindNextClusterId(currentId, next);
187 if (next == ROOT::kInvalidDescriptorId ||
188 !fPageSource.GetEntryRange().IntersectsWith(descriptorGuard->GetClusterDescriptor(next))) {
189 break;
190 }
191
192 keep.insert(next);
193 }
194 }
195
196 RProvides provide;
197 // Determine following cluster ids and the column ids that we want to make available
198 RProvides::RInfo provideInfo;
199 provideInfo.fPhysicalColumnSet = physicalColumns;
200 provideInfo.fBunchId = fBunchId;
201 provideInfo.fFlags = RProvides::kFlagRequired;
202 for (ROOT::DescriptorId_t i = 0, next = clusterId; i < 2 * fClusterBunchSize; ++i) {
203 if (i == fClusterBunchSize)
204 provideInfo.fBunchId = ++fBunchId;
205
206 auto cid = next;
207 auto descriptorGuard = fPageSource.FindNextClusterId(cid, next);
208 if (next != ROOT::kInvalidNTupleIndex) {
209 if (!fPageSource.GetEntryRange().IntersectsWith(descriptorGuard->GetClusterDescriptor(next)))
211 }
212 if (next == ROOT::kInvalidDescriptorId)
213 provideInfo.fFlags |= RProvides::kFlagLast;
214
215 provide.Insert(cid, provideInfo);
216
217 if (next == ROOT::kInvalidDescriptorId)
218 break;
219 provideInfo.fFlags = 0;
220 }
221
222 // Clear the cache from clusters not the in the look-ahead window or the set of pinned clusters
223 for (auto itr = fPool.begin(); itr != fPool.end();) {
224 if (provide.Contains(itr->first)) {
225 ++itr;
226 continue;
227 }
228 if (keep.count(itr->first) > 0) {
229 ++itr;
230 continue;
231 }
232 itr = fPool.erase(itr);
233 fCounters->fNCluster.Dec();
234 }
235
236 // Move clusters that meanwhile arrived into cache pool
237 {
238 // This lock is held during iteration over several data structures: the collection of in-flight clusters,
239 // the current pool of cached clusters, and the set of cluster ids to be preloaded.
240 // All three collections are expected to be small (certainly < 100, more likely < 10). All operations
241 // are non-blocking and moving around small items (pointers, ids, etc). Thus the overall locking time should
242 // still be reasonably small and the lock is rarely taken (usually once per cluster).
243 std::lock_guard<std::mutex> lockGuard(fLockWorkQueue);
244
245 for (auto itr = fInFlightClusters.begin(); itr != fInFlightClusters.end(); ) {
246 R__ASSERT(itr->fFuture.valid());
247 if (itr->fFuture.wait_for(std::chrono::seconds(0)) != std::future_status::ready) {
248 // Remove the set of columns that are already scheduled for being loaded
249 provide.Erase(itr->fClusterKey.fClusterId, itr->fClusterKey.fPhysicalColumnSet);
250 ++itr;
251 continue;
252 }
253
254 auto cptr = itr->fFuture.get();
256
257 const bool isExpired =
258 !provide.Contains(itr->fClusterKey.fClusterId) && (keep.count(itr->fClusterKey.fClusterId) == 0);
259 if (isExpired) {
260 cptr.reset();
261 itr = fInFlightClusters.erase(itr);
262 continue;
263 }
264
265 // Noop unless the page source has a task scheduler
266 fPageSource.UnzipCluster(cptr.get());
267
268 // We either put a fresh cluster into a free slot or we merge the cluster with an existing one
269 auto existingCluster = fPool.find(cptr->GetId());
270 if (existingCluster != fPool.end()) {
271 existingCluster->second->Adopt(std::move(*cptr));
272 } else {
273 const auto cid = cptr->GetId();
274 fPool.emplace(cid, std::move(cptr));
275 fCounters->fNCluster.Inc();
276 }
277 itr = fInFlightClusters.erase(itr);
278 }
279
280 // Determine clusters which get triggered for background loading
281 for (const auto &[_, cptr] : fPool) {
282 provide.Erase(cptr->GetId(), cptr->GetAvailPhysicalColumns());
283 }
284
285 // Figure out if enough work accumulated to justify I/O calls
286 bool skipPrefetch = false;
287 if (provide.GetSize() < fClusterBunchSize) {
288 skipPrefetch = true;
289 for (const auto &kv : provide) {
290 if ((kv.second.fFlags & (RProvides::kFlagRequired | RProvides::kFlagLast)) == 0)
291 continue;
292 skipPrefetch = false;
293 break;
294 }
295 }
296
297 // Update the work queue and the in-flight cluster list with new requests. We already hold the work queue
298 // mutex
299 // TODO(jblomer): we should ensure that clusterId is given first to the I/O thread. That is usually the
300 // case but it's not ensured by the code
301 if (!skipPrefetch) {
302 for (const auto &kv : provide) {
303 R__ASSERT(!kv.second.fPhysicalColumnSet.empty());
304
306 readItem.fClusterKey.fClusterId = kv.first;
307 readItem.fBunchId = kv.second.fBunchId;
308 readItem.fClusterKey.fPhysicalColumnSet = kv.second.fPhysicalColumnSet;
309
311 inFlightCluster.fClusterKey.fClusterId = kv.first;
312 inFlightCluster.fClusterKey.fPhysicalColumnSet = kv.second.fPhysicalColumnSet;
313 inFlightCluster.fFuture = readItem.fPromise.get_future();
314 fInFlightClusters.emplace_back(std::move(inFlightCluster));
315
316 fReadQueue.emplace_back(std::move(readItem));
317 }
318 if (!fReadQueue.empty())
319 fCvHasReadWork.notify_one();
320 }
321 } // work queue lock guard
322
323 return WaitFor(clusterId, physicalColumns);
324}
325
328{
329 while (true) {
330 // Fast exit: the cluster happens to be already present in the cache pool
331 auto result = fPool.find(clusterId);
332 if (result != fPool.end()) {
333 bool hasMissingColumn = false;
334 for (auto cid : physicalColumns) {
335 if (result->second->ContainsColumn(cid))
336 continue;
337
338 hasMissingColumn = true;
339 break;
340 }
341 if (!hasMissingColumn)
342 return result->second.get();
343 }
344
345 // Otherwise the missing data must have been triggered for loading by now, so block and wait
346 decltype(fInFlightClusters)::iterator itr;
347 {
348 std::lock_guard<std::mutex> lockGuardInFlightClusters(fLockWorkQueue);
349 itr = fInFlightClusters.begin();
350 for (; itr != fInFlightClusters.end(); ++itr) {
351 if (itr->fClusterKey.fClusterId == clusterId)
352 break;
353 }
354 R__ASSERT(itr != fInFlightClusters.end());
355 // Note that the fInFlightClusters is accessed concurrently only by the I/O thread. The I/O thread
356 // never changes the structure of the in-flight clusters array (it does not add, remove, or swap elements).
357 // Therefore, it is safe to access the element pointed to by itr here even after fLockWorkQueue
358 // is released. We need to release the lock before potentially blocking on the cluster future.
359 }
360
361 auto cptr = itr->fFuture.get();
362 // We were blocked waiting for the cluster, so assume that nobody discarded it.
363 R__ASSERT(cptr != nullptr);
364
365 // Noop unless the page source has a task scheduler
366 fPageSource.UnzipCluster(cptr.get());
367
368 if (result != fPool.end()) {
369 result->second->Adopt(std::move(*cptr));
370 } else {
371 const auto cid = cptr->GetId();
372 fPool.emplace(cid, std::move(cptr));
373 fCounters->fNCluster.Inc();
374 }
375
376 std::lock_guard<std::mutex> lockGuardInFlightClusters(fLockWorkQueue);
377 fInFlightClusters.erase(itr);
378 }
379}
380
382{
383 while (true) {
384 decltype(fInFlightClusters)::iterator itr;
385 {
386 std::lock_guard<std::mutex> lockGuardInFlightClusters(fLockWorkQueue);
387 itr = fInFlightClusters.begin();
388 while (itr != fInFlightClusters.end() &&
389 itr->fFuture.wait_for(std::chrono::seconds(0)) == std::future_status::ready) {
390 ++itr;
391 }
392 if (itr == fInFlightClusters.end())
393 break;
394 }
395
396 itr->fFuture.wait();
397 }
398}
uint32_t fFlags
#define R__unlikely(expr)
Definition RConfig.hxx:568
#define d(i)
Definition RSha256.hxx:102
ROOT::Detail::TRangeCast< T, true > TRangeDynCast
TRangeDynCast is an adapter class that allows the typed iteration through a TCollection.
#define R__ASSERT(e)
Checks condition e and reports a fatal error if it's false.
Definition TError.h:130
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t Float_t Float_t Float_t Int_t Int_t UInt_t UInt_t Rectangle_t result
#define _(A, B)
Definition cfortran.h:108
A thread-safe integral performance counter.
CounterPtrT MakeCounter(const std::string &name, Args &&... args)
RCluster * WaitFor(ROOT::DescriptorId_t clusterId, const RCluster::ColumnSet_t &physicalColumns)
Returns the given cluster from the pool, which needs to contain at least the columns physicalColumns.
ROOT::Experimental::Detail::RNTupleMetrics fMetrics
The cluster pool counters are observed by the page source.
unsigned int fClusterBunchSize
The number of clusters that are being read in a single vector read.
void WaitForInFlightClusters()
Used by the unit tests to drain the queue of clusters to be preloaded.
std::unique_ptr< RCounters > fCounters
void StopBackgroundThread()
Stop the I/O background thread. No-op if already stopped. Called by the destructor.
void ExecReadClusters()
The I/O thread routine, there is exactly one I/O thread in-flight for every cluster pool.
RCluster * GetCluster(ROOT::DescriptorId_t clusterId, const RCluster::ColumnSet_t &physicalColumns)
Returns the requested cluster either from the pool or, in case of a cache miss, lets the I/O thread l...
RClusterPool(ROOT::Internal::RPageSource &pageSource, unsigned int clusterBunchSize)
void StartBackgroundThread()
Spawn the I/O background thread. No-op if already started.
ROOT::Internal::RPageSource & fPageSource
Every cluster pool is responsible for exactly one page source that triggers loading of the clusters (...
An in-memory subset of the packed and compressed pages of a cluster.
Definition RCluster.hxx:147
std::unordered_set< ROOT::DescriptorId_t > ColumnSet_t
Definition RCluster.hxx:149
Abstract interface to read data from an ntuple.
const_iterator begin() const
const_iterator end() const
void Erase(const T &that, std::vector< T > &v)
Erase that element from vector v
Definition Utils.hxx:204
std::uint64_t DescriptorId_t
Distriniguishes elements of the same type within a descriptor, e.g. different fields.
constexpr NTupleSize_t kInvalidNTupleIndex
constexpr DescriptorId_t kInvalidDescriptorId
Performance counters that get registered in fMetrics.
Clusters that are currently being processed by the pipeline.
bool operator<(const RInFlightCluster &other) const
First order by cluster id, then by number of columns, than by the column ids in fColumns.
Request to load a subset of the columns of a particular cluster.
ROOT::DescriptorId_t fClusterId
Definition RCluster.hxx:152