Logo ROOT  
Reference Guide
 
Loading...
Searching...
No Matches
CpuTensor.h
Go to the documentation of this file.
1// @(#)root/tmva/tmva/dnn:$Id$
2// Authors: Sitong An, Lorenzo Moneta 10/2019
3
4/*************************************************************************
5 * Copyright (C) 2019, ROOT *
6 * All rights reserved. *
7 * *
8 * For the licensing terms see $ROOTSYS/LICENSE. *
9 * For the list of contributors see $ROOTSYS/README/CREDITS. *
10 *************************************************************************/
11
12//////////////////////////////////////////////////////////
13// Definition of the CpuTensor class used to represent //
14// tensor data in deep neural nets (CNN, RNN, etc..) //
15//////////////////////////////////////////////////////////
16
17#ifndef TMVA_DNN_ARCHITECTURES_CPU_CPUTENSOR
18#define TMVA_DNN_ARCHITECTURES_CPU_CPUTENSOR
19
20#include <cstddef>
21#include <cstdint>
22#include <iostream>
23#include <memory>
24#include <sstream>
25#include <stdexcept>
26#include <vector>
27
28#include "TMatrix.h"
29#include "TMVA/Config.h"
30#include "CpuBuffer.h"
31#include "CpuMatrix.h"
32
33namespace TMVA {
34namespace DNN {
35
36/// Memory layout type (row- or column-major storage of the tensor elements)
37enum class MemoryLayout : uint8_t {
38 RowMajor = 0x01,
39 ColumnMajor = 0x02
40};
41
42// CPU Tensor Class
43// It is a tensor container with contiguous storage whose memory is
44// owned by a CPU Buffer.
45// We need to keep a pointer for CPUBuffer for fast conversion
46// without copying to TCpuMatrix
47// also provides compatibility with old interface
48
49template <typename AFloat>
51
52public:
53 friend class TCpuMatrix<AFloat>;
54
55 using Shape_t = std::vector<std::size_t>;
58 using Scalar_t = AFloat;
59
60private:
61 Shape_t fShape; ///< Shape of the tensor
62 Shape_t fStrides; ///< Strides of the tensor
63 std::size_t fSize; ///< Total number of elements
64 MemoryLayout fLayout; ///< Memory layout of the tensor
65 AFloat *fData = nullptr; ///< Pointer to the first element
66 std::shared_ptr<TCpuBuffer<AFloat>> fContainer; ///< Buffer owning the data
67
68 /// Compute the total number of elements from a shape vector.
69 /// An empty shape has size 0.
70 static std::size_t GetSizeFromShape(const Shape_t &shape)
71 {
72 if (shape.size() == 0)
73 return 0;
74 std::size_t size = 1;
75 for (auto &s : shape)
76 size *= s;
77 return size;
78 }
79
80 /// Compute strides from a shape vector.
81 /// This information is needed for the multi-dimensional indexing:
82 /// for row-major layout the last dimension varies fastest, while for
83 /// column-major layout the first dimension varies fastest.
85 {
86 const auto size = shape.size();
87 Shape_t strides(size);
89 for (std::size_t i = 0; i < size; i++) {
90 if (i == 0) {
91 strides[size - 1 - i] = 1;
92 } else {
93 strides[size - 1 - i] = strides[size - 1 - i + 1] * shape[size - 1 - i + 1];
94 }
95 }
96 } else {
97 for (std::size_t i = 0; i < size; i++) {
98 if (i == 0) {
99 strides[i] = 1;
100 } else {
101 strides[i] = strides[i - 1] * shape[i - 1];
102 }
103 }
104 }
105 return strides;
106 }
107
108public:
109 /// Construct a tensor sharing the given buffer.
110 /// The container holds the memory and its destructor releases it when
111 /// all tensors sharing it are destroyed.
114 {
115 fSize = GetSizeFromShape(shape);
117 fData = fContainer->data();
118 }
119
120 // default constructor
122
123 /** constructors from n m */
127
128 /** constructors from batch size, depth, height*width */
135
136 /** constructors from batch size, depth, height, width */
145
146 /** constructors from a shape.*/
150
151 /* constructors from a AFloat pointer and a shape. This is a copy */
152
154 : TCpuTensor(std::make_shared<TCpuBuffer<AFloat>>(GetSizeFromShape(shape)), shape, memlayout)
155 {
156 auto& container = *(this->GetContainer());
157 for (size_t i = 0; i < this->GetSize(); ++i) container[i] = data[i];
158 }
159
160
161
162 /** constructors from a TCpuBuffer and a shape */
163 //unsafe method for backwards compatibility, const not promised. A view.
165 : TCpuTensor(std::make_shared<TCpuBuffer<AFloat>>(buffer), shape, memlayout)
166 {
167 R__ASSERT(this->GetSize() <= this->GetContainer()->GetSize());
168 }
169
170 /** constructors from a TCpuMatrix. Memory layout is forced to be same as matrix (i.e. columnlayout) */
171 //unsafe method for backwards compatibility, const not promised. A view of underlying data.
173 : TCpuTensor(std::make_shared<TCpuBuffer<AFloat>>(matrix.GetBuffer()), {matrix.GetNrows(), matrix.GetNcols()},
174 memlayout)
175 {
176
177 if (dim > 2) {
178 Shape_t shape = this->GetShape();
179
180 if (this->GetLayout() == MemoryLayout::ColumnMajor) {
181 shape.insert(shape.end(),dim-2, 1);
182 } else {
183 shape.insert(shape.begin(), dim - 2, 1);
184 }
185 this->ReshapeInplace(shape);
186 }
187 }
188
189
190 /** Convert to a TMatrixT<AFloat_t> object. Performs a deep copy of the matrix
191 * elements. */
192
193 operator TMatrixT<AFloat>() const {
194 // this should work only for size 2 or 4 tensors
195 if (this->GetShape().size() == 2 || (this->GetShape().size() == 3 && GetFirstSize() == 1)) {
197 return temp;
198 }
199 // convert as a flat vector
200 return TMatrixT<AFloat>(1, this->GetSize(), this->GetData());
201 }
202
203 // Access properties
204 std::size_t GetSize() const { return fSize; }
205 const Shape_t &GetShape() const { return fShape; }
206 const Shape_t &GetStrides() const { return fStrides; }
207 AFloat *GetData() { return fData; }
208 const AFloat *GetData() const { return fData; }
209 std::shared_ptr<TCpuBuffer<AFloat>> GetContainer() { return fContainer; }
210 const std::shared_ptr<TCpuBuffer<AFloat>> GetContainer() const { return fContainer; }
212
213 /// Reshape tensor in place.
214 /// The new shape must have the same overall size as the old one.
215 void ReshapeInplace(const Shape_t &shape)
216 {
217 const auto size = GetSizeFromShape(shape);
218 if (size != fSize) {
219 std::stringstream ss;
220 ss << "Cannot reshape tensor with size " << fSize << " into shape { ";
221 for (std::size_t i = 0; i < shape.size(); i++) {
222 if (i != shape.size() - 1) {
223 ss << shape[i] << ", ";
224 } else {
225 ss << shape[i] << " }.";
226 }
227 }
228 throw std::runtime_error(ss.str());
229 }
230
231 // Compute new strides from shape
233 fShape = shape;
234 }
235
236 /** Return raw pointer to the elements stored contiguously in column-major
237 * order. */
238 AFloat *GetRawDataPointer() { return this->GetContainer()->data(); }
239 const AFloat *GetRawDataPointer() const { return this->GetContainer()->data(); }
240
241 // for same API as CudaTensor (device buffer is the CpuBuffer)
242 const TCpuBuffer<AFloat> & GetDeviceBuffer() const {return *(this->GetContainer());}
244
245
246 size_t GetNoElements() const { return this->GetSize(); }
247
248 // return the size of the first dimension (if in row order) or last dimension if in column order
249 // Tensor is F x H x W x...for row order layout FHWC
250 // or H x W x ... x F for column order layout CHWF
251 // logic copied from TCudaTensor
252 size_t GetFirstSize() const
253 {
254 auto& shape = this->GetShape();
255 return (this->GetMemoryLayout() == MemoryLayout::ColumnMajor) ? shape.back() : shape.front();
256 }
257
258 size_t GetCSize() const
259 {
260 auto& shape = this->GetShape();
261 if (shape.size() == 2) return 1;
262 return (this->GetMemoryLayout() == MemoryLayout::ColumnMajor) ? shape.front() : shape[1]; // assume NHWC
263 }
264 //
265 size_t GetHSize() const
266 {
267 auto& shape = this->GetShape();
268 if (shape.size() == 2) return shape[0];
269 if (shape.size() == 3) return (this->GetMemoryLayout() == MemoryLayout::ColumnMajor) ? shape[0] : shape[1] ;
270 if (shape.size() >= 4) return shape[2] ;
271 return 0;
272
273 }
274 size_t GetWSize() const
275 {
276 auto& shape = this->GetShape();
277 if (shape.size() == 2) return shape[1];
278 if (shape.size() == 3) return (this->GetMemoryLayout() == MemoryLayout::ColumnMajor) ? shape[1] : shape[2] ;
279 if (shape.size() >= 4) return shape[3] ;
280 return 0;
281
282 }
283
284 // for backward compatibility (assume column-major
285 // for backward compatibility : for CM tensor (n1,n2,n3,n4) -> ( n1*n2*n3, n4)
286 // for RM tensor (n1,n2,n3,n4) -> ( n2*n3*n4, n1 ) ???
287 size_t GetNrows() const { return (GetLayout() == MemoryLayout::ColumnMajor ) ? this->GetStrides().back() : this->GetShape().front();}
288 size_t GetNcols() const { return (GetLayout() == MemoryLayout::ColumnMajor ) ? this->GetShape().back() : this->GetStrides().front(); }
289
290
291 MemoryLayout GetLayout() const { return this->GetMemoryLayout(); }
292
293 //this will be an unsafe view. Method exists for backwards compatibility only
295 {
296 [[maybe_unused]] size_t ndims = 0;
297 auto& shape = this->GetShape();
298 //check if squeezable but do not actually squeeze
299 for (auto& shape_i : shape){
300 if (shape_i != 1) {
301 ndims++;
302 }
303 }
304 assert(ndims <= 2 && shape.size() > 1); // to support shape cases {n,1}
305 return TCpuMatrix<AFloat>(*(this->GetContainer()), GetHSize(), GetWSize());
306 }
307
308 // Create copy, replace and return
310 {
311 TCpuTensor<AFloat> x(*this);
312 x.ReshapeInplace(shape);
313 return x;
314 }
315
316 // return a view of slices in the first dimension (if row wise) or last dimension if column wise
317 // so single event slices
319 {
320 auto &shape = this->GetShape();
321 auto layout = this->GetMemoryLayout();
322 Shape_t sliced_shape = (layout == MemoryLayout::RowMajor) ? Shape_t(shape.begin() + 1, shape.end())
323 : Shape_t(shape.begin(), shape.end() - 1);
324
325 size_t buffsize = (layout == MemoryLayout::RowMajor) ? this->GetStrides().front() : this->GetStrides().back();
326 size_t offset = i * buffsize;
327
328 return TCpuTensor<AFloat>(this->GetContainer()->GetSubBuffer(offset, buffsize), sliced_shape, layout);
329 }
330
331 TCpuTensor<AFloat> At(size_t i) const { return (const_cast<TCpuTensor<AFloat> &>(*this)).At(i); }
332
333 // for compatibility with old tensor (std::vector<matrix>)
336 return At(i).GetMatrix();
337 }
338
339 // set all the tensor contents to zero
340 void Zero()
341 {
342 AFloat *data = this->GetContainer()->data();
343 for (size_t i = 0; i < this->GetSize(); ++i)
344 data[i] = 0;
345 }
346
347 // access single element - assume tensor dim is 2
348 AFloat &operator()(size_t i, size_t j)
349 {
350 auto &shape = this->GetShape();
351 assert(shape.size() == 2);
352 return (this->GetMemoryLayout() == MemoryLayout::RowMajor) ? (*(this->GetContainer()))[i * shape[1] + j]
353 : (*(this->GetContainer()))[j * shape[0] + i];
354 }
355
356 // access single element - assume tensor dim is 3. First index i is always the major independent of row-major or
357 // column major row- major I - J - K . Column- major is J - K - I
358 AFloat &operator()(size_t i, size_t j, size_t k)
359 {
360 auto &shape = this->GetShape();
361 assert(shape.size() == 3);
362
363 return (this->GetMemoryLayout() == MemoryLayout::RowMajor)
364 ? (*(this->GetContainer()))[i * shape[1] * shape[2] + j * shape[2] + k]
365 : (*(this->GetContainer()))[i * shape[0] * shape[1] + k * shape[0] + j]; // note that is J-K-I
366 }
367
368 // access single element - assume tensor dim is 2
369 AFloat operator()(size_t i, size_t j) const
370 {
371 auto &shape = this->GetShape();
372 assert(shape.size() == 2);
373 return (this->GetMemoryLayout() == MemoryLayout::RowMajor) ? (this->GetData())[i * shape[1] + j]
374 : (this->GetData())[j * shape[0] + i];
375 }
376
377 AFloat operator()(size_t i, size_t j, size_t k) const
378 {
379 auto &shape = this->GetShape();
380 assert(shape.size() == 3);
381
382 return (this->GetMemoryLayout() == MemoryLayout::RowMajor)
383 ? (this->GetData())[i * shape[1] * shape[2] + j * shape[2] + k]
384 : (this->GetData())[i * shape[0] * shape[1] + k * shape[0] + j]; // note that is J-K-I
385 }
386
387 /** Map the given function over the matrix elements. Executed in parallel
388 * using TThreadExecutor. */
389 template <typename Function_t>
390 void Map(Function_t & f);
391
392 /** Same as maps but takes the input values from the tensor \p A and writes
393 * the results in this tensor. */
394 template <typename Function_t>
395 void MapFrom(Function_t & f, const TCpuTensor<AFloat> &A);
396
397 size_t GetBufferUseCount() const { return this->GetContainer()->GetUseCount(); }
398
399 void Print(const char *name = "Tensor") const
400 {
402
403 for (size_t i = 0; i < this->GetSize(); i++)
404 std::cout << (this->GetData())[i] << " ";
405 std::cout << std::endl;
406 }
407 void PrintShape(const char *name = "Tensor") const
408 {
409 std::string memlayout = (GetLayout() == MemoryLayout::RowMajor) ? "RowMajor" : "ColMajor";
410 std::cout << name << " shape : { ";
411 auto &shape = this->GetShape();
412 for (size_t i = 0; i < shape.size() - 1; ++i)
413 std::cout << shape[i] << " , ";
414 std::cout << shape.back() << " } "
415 << " Layout : " << memlayout << std::endl;
416 }
417};
418
419//______________________________________________________________________________
420template <typename AFloat>
421template <typename Function_t>
423{
424 AFloat *data = GetRawDataPointer();
425 size_t nelements = GetNoElements();
427
428 auto ff = [data, &nsteps, &nelements, &f](UInt_t workerID) {
429 size_t jMax = std::min(workerID + nsteps, nelements);
430 for (size_t j = workerID; j < jMax; ++j) {
431 data[j] = f(data[j]);
432 }
433 return 0;
434 };
435
436 if (nsteps < nelements) {
437 TMVA::Config::Instance().GetThreadExecutor().Foreach(ff, ROOT::TSeqI(0, nelements, nsteps));
438
439 // for (size_t i = 0; i < nelements; i+=nsteps)
440 // ff(i);
441
442 } else {
444 ff(0);
445 }
446}
447
448//______________________________________________________________________________
449template <typename AFloat>
450template <typename Function_t>
452{
453 AFloat *dataB = GetRawDataPointer();
454 const AFloat *dataA = A.GetRawDataPointer();
455
456 size_t nelements = GetNoElements();
457 R__ASSERT(nelements == A.GetNoElements());
459
460 auto ff = [&dataB, &dataA, &nsteps, &nelements, &f](UInt_t workerID) {
461 size_t jMax = std::min(workerID + nsteps, nelements);
462 for (size_t j = workerID; j < jMax; ++j) {
463 dataB[j] = f(dataA[j]);
464 }
465 return 0;
466 };
467 if (nsteps < nelements) {
468 TMVA::Config::Instance().GetThreadExecutor().Foreach(ff, ROOT::TSeqI(0, nelements, nsteps));
469 // for (size_t i = 0; i < nelements; i+=nsteps)
470 // ff(i);
471
472 } else {
474 ff(0);
475 }
476}
477
478
479} // namespace DNN
480} // namespace TMVA
481
482#endif
#define f(i)
Definition RSha256.hxx:104
size_t size(const MatrixT &matrix)
retrieve the size of a square matrix
ROOT::Detail::TRangeCast< T, true > TRangeDynCast
TRangeDynCast is an adapter class that allows the typed iteration through a TCollection.
#define R__ASSERT(e)
Checks condition e and reports a fatal error if it's false.
Definition TError.h:130
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void data
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t Float_t Float_t Float_t Int_t Int_t UInt_t UInt_t Rectangle_t Int_t Int_t Window_t TString Int_t GCValues_t GetPrimarySelectionOwner GetDisplay GetScreen GetColormap GetNativeEvent const char const char dpyName wid window const char font_name cursor keysym reg const char only_if_exist regb h Point_t winding char text const char depth char const char Int_t count const char ColorStruct_t color const char Pixmap_t Pixmap_t PictureAttributes_t attr const char char ret_data h unsigned char height h offset
Option_t Option_t width
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t height
char name[80]
Definition TGX11.cxx:142
static Config & Instance()
static function: returns TMVA instance
Definition Config.cxx:97
The TCpuMatrix class.
Definition CpuMatrix.h:86
static size_t GetNWorkItems(size_t nelements)
Definition CpuMatrix.h:191
size_t GetBufferUseCount() const
Definition CpuTensor.h:397
const Shape_t & GetStrides() const
Definition CpuTensor.h:206
AFloat operator()(size_t i, size_t j, size_t k) const
Definition CpuTensor.h:377
TCpuTensor(size_t n, size_t m, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from n m
Definition CpuTensor.h:124
TCpuTensor(size_t bsize, size_t depth, size_t height, size_t width, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from batch size, depth, height, width
Definition CpuTensor.h:137
AFloat * GetRawDataPointer()
Return raw pointer to the elements stored contiguously in column-major order.
Definition CpuTensor.h:238
size_t GetNoElements() const
Definition CpuTensor.h:246
size_t GetWSize() const
Definition CpuTensor.h:274
void Map(Function_t &f)
Map the given function over the matrix elements.
Definition CpuTensor.h:422
std::size_t GetSize() const
Definition CpuTensor.h:204
const TCpuBuffer< AFloat > & GetDeviceBuffer() const
Definition CpuTensor.h:242
TCpuTensor(std::shared_ptr< TCpuBuffer< AFloat > > container, Shape_t shape, MemoryLayout layout)
Construct a tensor sharing the given buffer.
Definition CpuTensor.h:112
MemoryLayout fLayout
Memory layout of the tensor.
Definition CpuTensor.h:64
TCpuTensor(size_t bsize, size_t depth, size_t hw, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from batch size, depth, height*width
Definition CpuTensor.h:129
TCpuMatrix< AFloat > operator[](size_t i) const
Definition CpuTensor.h:334
const AFloat * GetRawDataPointer() const
Definition CpuTensor.h:239
std::shared_ptr< TCpuBuffer< AFloat > > GetContainer()
Definition CpuTensor.h:209
TCpuBuffer< AFloat > & GetDeviceBuffer()
Definition CpuTensor.h:243
size_t GetCSize() const
Definition CpuTensor.h:258
AFloat & operator()(size_t i, size_t j, size_t k)
Definition CpuTensor.h:358
const AFloat * GetData() const
Definition CpuTensor.h:208
std::shared_ptr< TCpuBuffer< AFloat > > fContainer
Buffer owning the data.
Definition CpuTensor.h:66
const Shape_t & GetShape() const
Definition CpuTensor.h:205
std::size_t fSize
Total number of elements.
Definition CpuTensor.h:63
TCpuTensor(const TCpuBuffer< AFloat > &buffer, Shape_t shape, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from a TCpuBuffer and a shape
Definition CpuTensor.h:164
AFloat * fData
Pointer to the first element.
Definition CpuTensor.h:65
void MapFrom(Function_t &f, const TCpuTensor< AFloat > &A)
Same as maps but takes the input values from the tensor A and writes the results in this tensor.
Definition CpuTensor.h:451
AFloat operator()(size_t i, size_t j) const
Definition CpuTensor.h:369
Shape_t fStrides
Strides of the tensor.
Definition CpuTensor.h:62
TCpuTensor(const TCpuMatrix< AFloat > &matrix, size_t dim=3, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from a TCpuMatrix.
Definition CpuTensor.h:172
TCpuMatrix< AFloat > GetMatrix() const
Definition CpuTensor.h:294
TCpuTensor< AFloat > At(size_t i) const
Definition CpuTensor.h:331
size_t GetNcols() const
Definition CpuTensor.h:288
TCpuTensor(Shape_t shape, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
constructors from a shape.
Definition CpuTensor.h:147
size_t GetFirstSize() const
Definition CpuTensor.h:252
size_t GetNrows() const
Definition CpuTensor.h:287
void PrintShape(const char *name="Tensor") const
Definition CpuTensor.h:407
std::vector< std::size_t > Shape_t
Definition CpuTensor.h:55
static Shape_t ComputeStridesFromShape(const Shape_t &shape, MemoryLayout layout)
Compute strides from a shape vector.
Definition CpuTensor.h:84
AFloat & operator()(size_t i, size_t j)
Definition CpuTensor.h:348
void ReshapeInplace(const Shape_t &shape)
Reshape tensor in place.
Definition CpuTensor.h:215
Shape_t fShape
Shape of the tensor.
Definition CpuTensor.h:61
const std::shared_ptr< TCpuBuffer< AFloat > > GetContainer() const
Definition CpuTensor.h:210
TCpuTensor< AFloat > At(size_t i)
Definition CpuTensor.h:318
friend class TCpuMatrix< AFloat >
Definition CpuTensor.h:53
MemoryLayout GetLayout() const
Definition CpuTensor.h:291
void Print(const char *name="Tensor") const
Definition CpuTensor.h:399
static std::size_t GetSizeFromShape(const Shape_t &shape)
Compute the total number of elements from a shape vector.
Definition CpuTensor.h:70
TCpuTensor< AFloat > Reshape(Shape_t shape) const
Definition CpuTensor.h:309
MemoryLayout GetMemoryLayout() const
Definition CpuTensor.h:211
TCpuTensor(AFloat *data, const Shape_t &shape, MemoryLayout memlayout=MemoryLayout::ColumnMajor)
Definition CpuTensor.h:153
size_t GetHSize() const
Definition CpuTensor.h:265
Double_t x[n]
Definition legend1.C:17
const Int_t n
Definition legend1.C:16
MemoryLayout
Memory layout type (row- or column-major storage of the tensor elements)
Definition CpuTensor.h:37
create variable transformations
TMarker m
Definition textangle.C:8