Logo ROOT  
Reference Guide
 
Loading...
Searching...
No Matches
Reference.h
Go to the documentation of this file.
1// @(#)root/tmva/tmva/dnn:$Id$
2// Author: Simon Pfreundschuh 20/06/16
3
4/*************************************************************************
5 * Copyright (C) 2016, Simon Pfreundschuh *
6 * All rights reserved. *
7 * *
8 * For the licensing terms see $ROOTSYS/LICENSE. *
9 * For the list of contributors see $ROOTSYS/README/CREDITS. *
10 *************************************************************************/
11
12///////////////////////////////////////////////////////////////////////
13// Declaration of the TReference architecture, which provides a //
14// reference implementation of the low-level interface for the DNN //
15// implementation based on ROOT's TMatrixT matrix type. //
16///////////////////////////////////////////////////////////////////////
17
18#ifndef TMVA_DNN_ARCHITECTURES_REFERENCE
19#define TMVA_DNN_ARCHITECTURES_REFERENCE
20
21#include "TMatrix.h"
22#include "TMVA/DNN/Functions.h"
26#include <vector>
27
28
29class TRandom;
30
31namespace TMVA
32{
33namespace DNN
34{
35// struct TDescriptors {
36// };
37// struct TWorkspace {
38// };
39
40/*! The reference architecture class.
41*
42* Class template that contains the reference implementation of the low-level
43* interface for the DNN implementation. The reference implementation uses the
44* TMatrixT class template to represent matrices.
45*
46* \tparam AReal The floating point type used to represent scalars.
47*/
48
49
50template<typename AReal>
52{
53private:
55public:
56 using Scalar_t = AReal;
59
60 //____________________________________________________________________________
61 //
62 // Propagation
63 //____________________________________________________________________________
64
65 /** @name Forward Propagation
66 * Low-level functions required for the forward propagation of activations
67 * through the network.
68 */
69 ///@{
70 /** Matrix-multiply \p input with the transpose of \p weights and
71 * write the results into \p output. */
72
73 static void MultiplyTranspose(TMatrixT<Scalar_t> &output,
75 const TMatrixT<Scalar_t> &weights);
76
77
78 /** Add the vectors biases row-wise to the matrix output */
79 static void AddRowWise(TMatrixT<Scalar_t> &output,
80 const TMatrixT<Scalar_t> &biases);
81 ///@}
82
83 /** @name Backward Propagation
84 * Low-level functions required for the forward propagation of activations
85 * through the network.
86 */
87 ///@{
88 /** Perform the complete backward propagation step. If the provided
89 * \p activationGradientsBackward matrix is not empty, compute the
90 * gradients of the objective function with respect to the activations
91 * of the previous layer (backward direction).
92 * Also compute the weight and the bias gradients. Modifies the values
93 * in \p df and thus produces only a valid result, if it is applied the
94 * first time after the corresponding forward propagation has been per-
95 * formed. */
96 static void Backward(TMatrixT<Scalar_t> & activationGradientsBackward,
97 TMatrixT<Scalar_t> & weightGradients,
98 TMatrixT<Scalar_t> & biasGradients,
100 const TMatrixT<Scalar_t> & activationGradients,
101 const TMatrixT<Scalar_t> & weights,
102 const TMatrixT<Scalar_t> & activationBackward);
103 /** Backpropagation step for a Recurrent Neural Network */
104 static Matrix_t & RecurrentLayerBackward(TMatrixT<Scalar_t> & state_gradients_backward, // BxH
105 TMatrixT<Scalar_t> & input_weight_gradients,
106 TMatrixT<Scalar_t> & state_weight_gradients,
107 TMatrixT<Scalar_t> & bias_gradients,
108 TMatrixT<Scalar_t> & df, //DxH
109 const TMatrixT<Scalar_t> & state, // BxH
110 const TMatrixT<Scalar_t> & weights_input, // HxD
111 const TMatrixT<Scalar_t> & weights_state, // HxH
112 const TMatrixT<Scalar_t> & input, // BxD
113 TMatrixT<Scalar_t> & input_gradient);
114
115
116
117 /** Backward pass for LSTM Network */
118 static Matrix_t & LSTMLayerBackward(TMatrixT<Scalar_t> & state_gradients_backward,
119 TMatrixT<Scalar_t> & cell_gradients_backward,
120 TMatrixT<Scalar_t> & input_weight_gradients,
121 TMatrixT<Scalar_t> & forget_weight_gradients,
122 TMatrixT<Scalar_t> & candidate_weight_gradients,
123 TMatrixT<Scalar_t> & output_weight_gradients,
124 TMatrixT<Scalar_t> & input_state_weight_gradients,
125 TMatrixT<Scalar_t> & forget_state_weight_gradients,
126 TMatrixT<Scalar_t> & candidate_state_weight_gradients,
127 TMatrixT<Scalar_t> & output_state_weight_gradients,
128 TMatrixT<Scalar_t> & input_bias_gradients,
129 TMatrixT<Scalar_t> & forget_bias_gradients,
130 TMatrixT<Scalar_t> & candidate_bias_gradients,
131 TMatrixT<Scalar_t> & output_bias_gradients,
135 TMatrixT<Scalar_t> & dout,
136 const TMatrixT<Scalar_t> & precStateActivations,
137 const TMatrixT<Scalar_t> & precCellActivations,
138 const TMatrixT<Scalar_t> & fInput,
139 const TMatrixT<Scalar_t> & fForget,
140 const TMatrixT<Scalar_t> & fCandidate,
141 const TMatrixT<Scalar_t> & fOutput,
142 const TMatrixT<Scalar_t> & weights_input,
143 const TMatrixT<Scalar_t> & weights_forget,
144 const TMatrixT<Scalar_t> & weights_candidate,
145 const TMatrixT<Scalar_t> & weights_output,
146 const TMatrixT<Scalar_t> & weights_input_state,
147 const TMatrixT<Scalar_t> & weights_forget_state,
148 const TMatrixT<Scalar_t> & weights_candidate_state,
149 const TMatrixT<Scalar_t> & weights_output_state,
151 TMatrixT<Scalar_t> & input_gradient,
152 TMatrixT<Scalar_t> & cell_gradient,
153 TMatrixT<Scalar_t> & cell_tanh);
154
155
156 /** Backward pass for GRU Network */
157 static Matrix_t & GRULayerBackward(TMatrixT<Scalar_t> & state_gradients_backward,
158 TMatrixT<Scalar_t> & reset_weight_gradients,
159 TMatrixT<Scalar_t> & update_weight_gradients,
160 TMatrixT<Scalar_t> & candidate_weight_gradients,
161 TMatrixT<Scalar_t> & reset_state_weight_gradients,
162 TMatrixT<Scalar_t> & update_state_weight_gradients,
163 TMatrixT<Scalar_t> & candidate_state_weight_gradients,
164 TMatrixT<Scalar_t> & reset_bias_gradients,
165 TMatrixT<Scalar_t> & update_bias_gradients,
166 TMatrixT<Scalar_t> & candidate_bias_gradients,
170 const TMatrixT<Scalar_t> & precStateActivations,
171 const TMatrixT<Scalar_t> & fReset,
172 const TMatrixT<Scalar_t> & fUpdate,
173 const TMatrixT<Scalar_t> & fCandidate,
174 const TMatrixT<Scalar_t> & weights_reset,
175 const TMatrixT<Scalar_t> & weights_update,
176 const TMatrixT<Scalar_t> & weights_candidate,
177 const TMatrixT<Scalar_t> & weights_reset_state,
178 const TMatrixT<Scalar_t> & weights_update_state,
179 const TMatrixT<Scalar_t> & weights_candidate_state,
181 TMatrixT<Scalar_t> & input_gradient);
182
183 /** Adds a the elements in matrix B scaled by c to the elements in
184 * the matrix A. This is required for the weight update in the gradient
185 * descent step.*/
186 static void ScaleAdd(TMatrixT<Scalar_t> & A,
187 const TMatrixT<Scalar_t> & B,
188 Scalar_t beta = 1.0);
189
190 static void Copy(TMatrixT<Scalar_t> & A,
191 const TMatrixT<Scalar_t> & B);
192
193 // copy from another type of matrix
194 template<typename AMatrix_t>
195 static void CopyDiffArch(TMatrixT<Scalar_t> & A, const AMatrix_t & B);
196
197
198 /** Above functions extended to vectors */
199 static void ScaleAdd(std::vector<TMatrixT<Scalar_t>> & A,
200 const std::vector<TMatrixT<Scalar_t>> & B,
201 Scalar_t beta = 1.0);
202
203 static void Copy(std::vector<TMatrixT<Scalar_t>> & A, const std::vector<TMatrixT<Scalar_t>> & B);
204
205 // copy from another architecture
206 template<typename AMatrix_t>
207 static void CopyDiffArch(std::vector<TMatrixT<Scalar_t> > & A, const std::vector<AMatrix_t> & B);
208
209
210 ///@}
211
212 //____________________________________________________________________________
213 //
214 // Activation Functions
215 //____________________________________________________________________________
216
217 /** @name Activation Functions
218 * For each activation function, the low-level interface contains two routines.
219 * One that applies the activation function to a matrix and one that evaluate
220 * the derivatives of the activation function at the elements of a given matrix
221 * and writes the results into the result matrix.
222 */
223 ///@{
224 static void Identity(TMatrixT<AReal> & B);
225 static void IdentityDerivative(TMatrixT<AReal> & B,
226 const TMatrixT<AReal> & A);
227
228 static void Relu(TMatrixT<AReal> & B);
229 static void ReluDerivative(TMatrixT<AReal> & B,
230 const TMatrixT<AReal> & A);
231
232 static void Sigmoid(TMatrixT<AReal> & B);
233 static void SigmoidDerivative(TMatrixT<AReal> & B,
234 const TMatrixT<AReal> & A);
235
236 static void Tanh(TMatrixT<AReal> & B);
237 static void TanhDerivative(TMatrixT<AReal> & B,
238 const TMatrixT<AReal> & A);
239
240 static void FastTanh(Tensor_t &B) { return Tanh(B); }
241 static void FastTanhDerivative(Tensor_t &B, const Tensor_t &A) { return TanhDerivative(B, A); }
242
243 static void SymmetricRelu(TMatrixT<AReal> & B);
245 const TMatrixT<AReal> & A);
246
247 static void SoftSign(TMatrixT<AReal> & B);
248 static void SoftSignDerivative(TMatrixT<AReal> & B,
249 const TMatrixT<AReal> & A);
250
251 static void Gauss(TMatrixT<AReal> & B);
252 static void GaussDerivative(TMatrixT<AReal> & B,
253 const TMatrixT<AReal> & A);
254
255
256 ///@}
257
258 //____________________________________________________________________________
259 //
260 // Loss Functions
261 //____________________________________________________________________________
262
263 /** @name Loss Functions
264 * Loss functions compute a scalar value given the \p output of the network
265 * for a given training input and the expected network prediction \p Y that
266 * quantifies the quality of the prediction. For each function also a routing
267 * that computes the gradients (suffixed by Gradients) must be provided for
268 * the starting of the backpropagation algorithm.
269 */
270 ///@{
271
272 static AReal MeanSquaredError(const TMatrixT<AReal> &Y, const TMatrixT<AReal> &output,
273 const TMatrixT<AReal> &weights);
274 static void MeanSquaredErrorGradients(TMatrixT<AReal> &dY, const TMatrixT<AReal> &Y, const TMatrixT<AReal> &output,
275 const TMatrixT<AReal> &weights);
276
277 /** Sigmoid transformation is implicitly applied, thus \p output should
278 * hold the linear activations of the last layer in the net. */
279 static AReal CrossEntropy(const TMatrixT<AReal> &Y, const TMatrixT<AReal> &output, const TMatrixT<AReal> &weights);
280
281 static void CrossEntropyGradients(TMatrixT<AReal> &dY, const TMatrixT<AReal> &Y, const TMatrixT<AReal> &output,
282 const TMatrixT<AReal> &weights);
283
284 /** Softmax transformation is implicitly applied, thus \p output should
285 * hold the linear activations of the last layer in the net. */
286 static AReal SoftmaxCrossEntropy(const TMatrixT<AReal> &Y, const TMatrixT<AReal> &output,
287 const TMatrixT<AReal> &weights);
289 const TMatrixT<AReal> &output, const TMatrixT<AReal> &weights);
290 ///@}
291
292 //____________________________________________________________________________
293 //
294 // Output Functions
295 //____________________________________________________________________________
296
297 /** @name Output Functions
298 * Output functions transform the activations \p output of the
299 * output layer in the network to a valid prediction \p YHat for
300 * the desired usage of the network, e.g. the identity function
301 * for regression or the sigmoid transformation for two-class
302 * classification.
303 */
304 ///@{
305 static void Sigmoid(TMatrixT<AReal> &YHat,
306 const TMatrixT<AReal> & );
307 static void Softmax(TMatrixT<AReal> &YHat,
308 const TMatrixT<AReal> & );
309 ///@}
310
311 //____________________________________________________________________________
312 //
313 // Regularization
314 //____________________________________________________________________________
315
316 /** @name Regularization
317 * For each regularization type, two functions are required, one named
318 * `<Type>Regularization` that evaluates the corresponding
319 * regularization functional for a given weight matrix and the
320 * `Add<Type>RegularizationGradients`, that adds the regularization
321 * component in the gradients to the provided matrix.
322 */
323 ///@{
324
325 static AReal L1Regularization(const TMatrixT<AReal> & W);
327 const TMatrixT<AReal> & W,
329
330 static AReal L2Regularization(const TMatrixT<AReal> & W);
332 const TMatrixT<AReal> & W,
334 ///@}
335
336 //____________________________________________________________________________
337 //
338 // Initialization
339 //____________________________________________________________________________
340
341 /** @name Initialization
342 * For each initialization method, one function in the low-level interface
343 * is provided. The naming scheme is `Initialize<Type>` for a given
344 * initialization method Type.
345 */
346 ///@{
347
348 static void InitializeGauss(TMatrixT<AReal> & A);
349
350 static void InitializeUniform(TMatrixT<AReal> & A);
351
352 static void InitializeIdentity(TMatrixT<AReal> & A);
353
354 static void InitializeZero(TMatrixT<AReal> & A);
355
357
359
360 // return static instance of random generator used for initialization
361 // if generator does not exist it is created the first time with a random seed (e.g. seed = 0)
362 static TRandom & GetRandomGenerator();
363 // set random seed for the static generator
364 // if the static generator does not exists it is created
365 static void SetRandomSeed(size_t seed);
366
367
368 ///@}
369
370 //____________________________________________________________________________
371 //
372 // Dropout
373 //____________________________________________________________________________
374
375 /** @name Dropout
376 */
377 ///@{
378
379 /** Apply dropout with activation probability \p p to the given
380 * matrix \p A and scale the result by reciprocal of \p p. */
381 //static void Dropout(TMatrixT<AReal> & A, AReal dropoutProbability);
382 static void DropoutForward(Tensor_t &A, TDescriptors *descriptors, TWorkspace *workspace, Scalar_t p);
384 {
385 Tensor_t & tA = A; // Tensor and matrix are same types
386 DropoutForward(tA, static_cast<TDescriptors *>(nullptr), static_cast<TWorkspace *>(nullptr), p);
387 }
388
389 ///@}
390
391
392 //____________________________________________________________________________
393 //
394 // Convolutional Layer Propagation
395 //____________________________________________________________________________
396
397 /** @name Forward Propagation in Convolutional Layer
398 */
399 ///@{
400
401 /** Transform the matrix \p B in local view format, suitable for
402 * convolution, and store it in matrix \p A. */
403 static void Im2col(TMatrixT<AReal> &A,
404 const TMatrixT<AReal> &B,
405 size_t imgHeight,
406 size_t imgWidth,
407 size_t fltHeight,
408 size_t fltWidth,
409 size_t strideRows,
410 size_t strideCols,
411 size_t zeroPaddingHeight,
412 size_t zeroPaddingWidth);
413
414 static void Im2colIndices(std::vector<int> &, const TMatrixT<AReal> &, size_t, size_t, size_t, size_t ,
415 size_t , size_t , size_t , size_t ,size_t ) {
416 Fatal("Im2ColIndices","This function is not implemented for ref architectures");
417 }
418 static void Im2colFast(TMatrixT<AReal> &, const TMatrixT<AReal> &, const std::vector<int> & ) {
419 Fatal("Im2ColFast","This function is not implemented for ref architectures");
420 }
421
422 /** Rotates the matrix \p B, which is representing a weights,
423 * and stores them in the matrix \p A. */
424 static void RotateWeights(TMatrixT<AReal> &A, const TMatrixT<AReal> &B, size_t filterDepth, size_t filterHeight,
425 size_t filterWidth, size_t numFilters);
426
427 /** Add the biases in the Convolutional Layer. */
428 static void AddConvBiases(TMatrixT<AReal> &output, const TMatrixT<AReal> &biases);
429 ///@}
430
431 /** Dummy placeholder - preparation is currently only required for the CUDA architecture. */
432 static void PrepareInternals(std::vector<TMatrixT<AReal>> &) {}
433
434 /** Forward propagation in the Convolutional layer */
435 static void ConvLayerForward(std::vector<TMatrixT<AReal>> & /*output*/,
436 std::vector<TMatrixT<AReal>> & /*derivatives*/,
437 const std::vector<TMatrixT<AReal>> & /*input*/,
438 const TMatrixT<AReal> & /*weights*/, const TMatrixT<AReal> & /*biases*/,
439 const DNN::CNN::TConvParams & /*params*/, EActivationFunction /*activFunc*/,
440 std::vector<TMatrixT<AReal>> & /*inputPrime*/) {
441 Fatal("ConvLayerForward","This function is not implemented for ref architectures");
442 }
443
444
445 /** @name Backward Propagation in Convolutional Layer
446 */
447 ///@{
448
449 /** Perform the complete backward propagation step in a Convolutional Layer.
450 * If the provided \p activationGradientsBackward matrix is not empty, compute the
451 * gradients of the objective function with respect to the activations
452 * of the previous layer (backward direction).
453 * Also compute the weight and the bias gradients. Modifies the values
454 * in \p df and thus produces only a valid result, if it is applied the
455 * first time after the corresponding forward propagation has been per-
456 * formed. */
457 static void ConvLayerBackward(std::vector<TMatrixT<AReal>> &,
459 std::vector<TMatrixT<AReal>> &,
460 const std::vector<TMatrixT<AReal>> &,
461 const TMatrixT<AReal> &, const std::vector<TMatrixT<AReal>> &,
462 size_t , size_t , size_t , size_t , size_t,
463 size_t , size_t , size_t , size_t , size_t) {
464 Fatal("ConvLayerBackward","This function is not implemented for ref architectures");
465
466 }
467
468#ifdef HAVE_CNN_REFERENCE
469 /** Utility function for calculating the activation gradients of the layer
470 * before the convolutional layer. */
471 static void CalculateConvActivationGradients(std::vector<TMatrixT<AReal>> &activationGradientsBackward,
472 const std::vector<TMatrixT<AReal>> &df, const TMatrixT<AReal> &weights,
473 size_t batchSize, size_t inputHeight, size_t inputWidth, size_t depth,
474 size_t height, size_t width, size_t filterDepth, size_t filterHeight,
475 size_t filterWidth);
476
477 /** Utility function for calculating the weight gradients of the convolutional
478 * layer. */
479 static void CalculateConvWeightGradients(TMatrixT<AReal> &weightGradients, const std::vector<TMatrixT<AReal>> &df,
480 const std::vector<TMatrixT<AReal>> &activationBackward, size_t batchSize,
481 size_t inputHeight, size_t inputWidth, size_t depth, size_t height,
482 size_t width, size_t filterDepth, size_t filterHeight, size_t filterWidth,
483 size_t nLocalViews);
484
485 /** Utility function for calculating the bias gradients of the convolutional
486 * layer. */
487 static void CalculateConvBiasGradients(TMatrixT<AReal> &biasGradients, const std::vector<TMatrixT<AReal>> &df,
488 size_t batchSize, size_t depth, size_t nLocalViews);
489 ///@}
490
491#endif
492
493 //____________________________________________________________________________
494 //
495 // Max Pooling Layer Propagation
496 //____________________________________________________________________________
497 /** @name Forward Propagation in Max Pooling Layer
498 */
499 ///@{
500
501 /** Downsample the matrix \p C to the matrix \p A, using max
502 * operation, such that the winning indices are stored in matrix
503 * \p B. */
504 static void Downsample(TMatrixT<AReal> &A, TMatrixT<AReal> &B, const TMatrixT<AReal> &C, size_t imgHeight,
505 size_t imgWidth, size_t fltHeight, size_t fltWidth, size_t strideRows, size_t strideCols);
506
507 ///@}
508
509 /** @name Backward Propagation in Max Pooling Layer
510 */
511 ///@{
512
513 /** Perform the complete backward propagation step in a Max Pooling Layer. Based on the
514 * winning indices stored in the index matrix, it just forwards the activation
515 * gradients to the previous layer. */
516 static void MaxPoolLayerBackward(TMatrixT<AReal> &activationGradientsBackward,
517 const TMatrixT<AReal> &activationGradients,
518 const TMatrixT<AReal> &indexMatrix,
519 size_t imgHeight,
520 size_t imgWidth,
521 size_t fltHeight,
522 size_t fltWidth,
523 size_t strideRows,
524 size_t strideCol,
525 size_t nLocalViews);
526 ///@}
527 //____________________________________________________________________________
528 //
529 // Reshape Layer Propagation
530 //____________________________________________________________________________
531 /** @name Forward and Backward Propagation in Reshape Layer
532 */
533 ///@{
534
535 /** Transform the matrix \p B to a matrix with different dimensions \p A */
536 static void Reshape(TMatrixT<AReal> &A, const TMatrixT<AReal> &B);
537
538 /** Flattens the tensor \p B, such that each matrix, is stretched in one row, resulting with a matrix \p A. */
539 static void Flatten(TMatrixT<AReal> &A, const std::vector<TMatrixT<AReal>> &B, size_t size, size_t nRows,
540 size_t nCols);
541
542 /** Transforms each row of \p B to a matrix and stores it in the tensor \p B. */
543 static void Deflatten(std::vector<TMatrixT<AReal>> &A, const TMatrixT<Scalar_t> &B, size_t index, size_t nRows,
544 size_t nCols);
545 /** Rearrage data according to time fill B x T x D out with T x B x D matrix in*/
546 static void Rearrange(std::vector<TMatrixT<AReal>> &out, const std::vector<TMatrixT<AReal>> &in);
547
548 ///@}
549
550 //____________________________________________________________________________
551 //
552 // Additional Arithmetic Functions
553 //____________________________________________________________________________
554
555 /** Sum columns of (m x n) matrix \p A and write the results into the first
556 * m elements in \p A.
557 */
558 static void SumColumns(TMatrixT<AReal> &B, const TMatrixT<AReal> &A);
559
560 /** In-place Hadamard (element-wise) product of matrices \p A and \p B
561 * with the result being written into \p A.
562 */
563 static void Hadamard(TMatrixT<AReal> &A, const TMatrixT<AReal> &B);
564
565 /** Add the constant \p beta to all the elements of matrix \p A and write the
566 * result into \p A.
567 */
568 static void ConstAdd(TMatrixT<AReal> &A, AReal beta);
569
570 /** Multiply the constant \p beta to all the elements of matrix \p A and write the
571 * result into \p A.
572 */
573 static void ConstMult(TMatrixT<AReal> &A, AReal beta);
574
575 /** Reciprocal each element of the matrix \p A and write the result into
576 * \p A
577 */
579
580 /** Square each element of the matrix \p A and write the result into
581 * \p A
582 */
583 static void SquareElementWise(TMatrixT<AReal> &A);
584
585 /** Square root each element of the matrix \p A and write the result into
586 * \p A
587 */
588 static void SqrtElementWise(TMatrixT<AReal> &A);
589
590 // optimizer update functions
591
592 /// Update functions for ADAM optimizer
593 static void AdamUpdate(TMatrixT<AReal> & A, const TMatrixT<AReal> & M, const TMatrixT<AReal> & V, AReal alpha, AReal eps);
594 static void AdamUpdateFirstMom(TMatrixT<AReal> & A, const TMatrixT<AReal> & B, AReal beta);
595 static void AdamUpdateSecondMom(TMatrixT<AReal> & A, const TMatrixT<AReal> & B, AReal beta);
596
597
598
599 //____________________________________________________________________________
600 //
601 // AutoEncoder Propagation
602 //____________________________________________________________________________
603
604 // Add Biases to the output
605 static void AddBiases(TMatrixT<AReal> &A,
606 const TMatrixT<AReal> &biases);
607
608 // Updating parameters after every backward pass. Weights and biases are
609 // updated.
610 static void
612 TMatrixT<AReal> &z, TMatrixT<AReal> &fVBiases,
613 TMatrixT<AReal> &fHBiases, TMatrixT<AReal> &fWeights,
614 TMatrixT<AReal> &VBiasError, TMatrixT<AReal> &HBiasError,
615 AReal learningRate, size_t fBatchSize);
616
617 // Softmax functions redefined
618 static void SoftmaxAE(TMatrixT<AReal> & A);
619
620
621 // Corrupt the input values randomly on corruption Level.
622 //Basically inputs are masked currently.
623 static void CorruptInput(TMatrixT<AReal> & input,
624 TMatrixT<AReal> & corruptedInput,
625 AReal corruptionLevel);
626
627 //Encodes the input Values in the compressed form.
628 static void EncodeInput(TMatrixT<AReal> &input,
629 TMatrixT<AReal> &compressedInput,
630 TMatrixT<AReal> &Weights);
631
632 // reconstructs the input. The reconstructed Input has same dimensions as that
633 // of the input.
634 static void ReconstructInput(TMatrixT<AReal> & compressedInput,
635 TMatrixT<AReal> & reconstructedInput,
636 TMatrixT<AReal> &fWeights);
637
638
641 TMatrixT<AReal> &fWeights);
642
644 TMatrixT<AReal> &output,
645 TMatrixT<AReal> &difference,
647 TMatrixT<AReal> &fWeights,
648 TMatrixT<AReal> &fBiases,
649 AReal learningRate,
650 size_t fBatchSize);
651
652};
653
654
655// implement the templated member functions
656template <typename AReal>
657template <typename AMatrix_t>
659{
660 TMatrixT<AReal> tmp = B;
661 A = tmp;
662}
663
664template <typename AReal>
665template <typename AMatrix_t>
666void TReference<AReal>::CopyDiffArch(std::vector<TMatrixT<AReal>> &A, const std::vector<AMatrix_t> &B)
667{
668 for (size_t i = 0; i < A.size(); ++i) {
669 CopyDiffArch(A[i], B[i]);
670 }
671}
672
673
674
675} // namespace DNN
676} // namespace TMVA
677
678#endif
size_t size(const MatrixT &matrix)
retrieve the size of a square matrix
void Fatal(const char *location, const char *msgfmt,...)
Use this function in case of a fatal error. It will abort the program.
Definition TError.cxx:267
winID h TVirtualViewer3D TVirtualGLPainter p
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void input
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t index
Option_t Option_t width
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t height
The reference architecture class.
Definition Reference.h:52
static void AdamUpdate(TMatrixT< AReal > &A, const TMatrixT< AReal > &M, const TMatrixT< AReal > &V, AReal alpha, AReal eps)
Update functions for ADAM optimizer.
static void AdamUpdateSecondMom(TMatrixT< AReal > &A, const TMatrixT< AReal > &B, AReal beta)
static void DropoutForward(Matrix_t &A, Scalar_t p)
Definition Reference.h:383
static void SymmetricRelu(TMatrixT< AReal > &B)
static void InitializeIdentity(TMatrixT< AReal > &A)
static void MultiplyTranspose(TMatrixT< Scalar_t > &output, const TMatrixT< Scalar_t > &input, const TMatrixT< Scalar_t > &weights)
Matrix-multiply input with the transpose of weights and write the results into output.
static void InitializeGlorotNormal(TMatrixT< AReal > &A)
Truncated normal initialization (Glorot, called also Xavier normal) The values are sample with a norm...
static void AdamUpdateFirstMom(TMatrixT< AReal > &A, const TMatrixT< AReal > &B, AReal beta)
static void Flatten(TMatrixT< AReal > &A, const std::vector< TMatrixT< AReal > > &B, size_t size, size_t nRows, size_t nCols)
Flattens the tensor B, such that each matrix, is stretched in one row, resulting with a matrix A.
static void Relu(TMatrixT< AReal > &B)
static void MaxPoolLayerBackward(TMatrixT< AReal > &activationGradientsBackward, const TMatrixT< AReal > &activationGradients, const TMatrixT< AReal > &indexMatrix, size_t imgHeight, size_t imgWidth, size_t fltHeight, size_t fltWidth, size_t strideRows, size_t strideCol, size_t nLocalViews)
Perform the complete backward propagation step in a Max Pooling Layer.
static void GaussDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static void SoftmaxAE(TMatrixT< AReal > &A)
static void AddL1RegularizationGradients(TMatrixT< AReal > &A, const TMatrixT< AReal > &W, AReal weightDecay)
static void AddRowWise(TMatrixT< Scalar_t > &output, const TMatrixT< Scalar_t > &biases)
Add the vectors biases row-wise to the matrix output.
static void CrossEntropyGradients(TMatrixT< AReal > &dY, const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
static void Im2colFast(TMatrixT< AReal > &, const TMatrixT< AReal > &, const std::vector< int > &)
Definition Reference.h:418
static void Downsample(TMatrixT< AReal > &A, TMatrixT< AReal > &B, const TMatrixT< AReal > &C, size_t imgHeight, size_t imgWidth, size_t fltHeight, size_t fltWidth, size_t strideRows, size_t strideCols)
Downsample the matrix C to the matrix A, using max operation, such that the winning indices are store...
static void EncodeInput(TMatrixT< AReal > &input, TMatrixT< AReal > &compressedInput, TMatrixT< AReal > &Weights)
static void TanhDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static void ReconstructInput(TMatrixT< AReal > &compressedInput, TMatrixT< AReal > &reconstructedInput, TMatrixT< AReal > &fWeights)
static AReal L2Regularization(const TMatrixT< AReal > &W)
static void Im2colIndices(std::vector< int > &, const TMatrixT< AReal > &, size_t, size_t, size_t, size_t, size_t, size_t, size_t, size_t, size_t)
Definition Reference.h:414
static void IdentityDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static AReal SoftmaxCrossEntropy(const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
Softmax transformation is implicitly applied, thus output should hold the linear activations of the l...
static Matrix_t & GRULayerBackward(TMatrixT< Scalar_t > &state_gradients_backward, TMatrixT< Scalar_t > &reset_weight_gradients, TMatrixT< Scalar_t > &update_weight_gradients, TMatrixT< Scalar_t > &candidate_weight_gradients, TMatrixT< Scalar_t > &reset_state_weight_gradients, TMatrixT< Scalar_t > &update_state_weight_gradients, TMatrixT< Scalar_t > &candidate_state_weight_gradients, TMatrixT< Scalar_t > &reset_bias_gradients, TMatrixT< Scalar_t > &update_bias_gradients, TMatrixT< Scalar_t > &candidate_bias_gradients, TMatrixT< Scalar_t > &dr, TMatrixT< Scalar_t > &du, TMatrixT< Scalar_t > &dc, const TMatrixT< Scalar_t > &precStateActivations, const TMatrixT< Scalar_t > &fReset, const TMatrixT< Scalar_t > &fUpdate, const TMatrixT< Scalar_t > &fCandidate, const TMatrixT< Scalar_t > &weights_reset, const TMatrixT< Scalar_t > &weights_update, const TMatrixT< Scalar_t > &weights_candidate, const TMatrixT< Scalar_t > &weights_reset_state, const TMatrixT< Scalar_t > &weights_update_state, const TMatrixT< Scalar_t > &weights_candidate_state, const TMatrixT< Scalar_t > &input, TMatrixT< Scalar_t > &input_gradient)
Backward pass for GRU Network.
static void ConstAdd(TMatrixT< AReal > &A, AReal beta)
Add the constant beta to all the elements of matrix A and write the result into A.
static void FastTanhDerivative(Tensor_t &B, const Tensor_t &A)
Definition Reference.h:241
static void SetRandomSeed(size_t seed)
static void SigmoidDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static void SoftSignDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static AReal CrossEntropy(const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
Sigmoid transformation is implicitly applied, thus output should hold the linear activations of the l...
static void InitializeZero(TMatrixT< AReal > &A)
static void Softmax(TMatrixT< AReal > &YHat, const TMatrixT< AReal > &)
static void ReciprocalElementWise(TMatrixT< AReal > &A)
Reciprocal each element of the matrix A and write the result into A.
static void Backward(TMatrixT< Scalar_t > &activationGradientsBackward, TMatrixT< Scalar_t > &weightGradients, TMatrixT< Scalar_t > &biasGradients, TMatrixT< Scalar_t > &df, const TMatrixT< Scalar_t > &activationGradients, const TMatrixT< Scalar_t > &weights, const TMatrixT< Scalar_t > &activationBackward)
Perform the complete backward propagation step.
static void SquareElementWise(TMatrixT< AReal > &A)
Square each element of the matrix A and write the result into A.
static void MeanSquaredErrorGradients(TMatrixT< AReal > &dY, const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
static TRandom * fgRandomGen
Definition Reference.h:54
static void RotateWeights(TMatrixT< AReal > &A, const TMatrixT< AReal > &B, size_t filterDepth, size_t filterHeight, size_t filterWidth, size_t numFilters)
Rotates the matrix B, which is representing a weights, and stores them in the matrix A.
static void Rearrange(std::vector< TMatrixT< AReal > > &out, const std::vector< TMatrixT< AReal > > &in)
Rearrage data according to time fill B x T x D out with T x B x D matrix in.
static Matrix_t & LSTMLayerBackward(TMatrixT< Scalar_t > &state_gradients_backward, TMatrixT< Scalar_t > &cell_gradients_backward, TMatrixT< Scalar_t > &input_weight_gradients, TMatrixT< Scalar_t > &forget_weight_gradients, TMatrixT< Scalar_t > &candidate_weight_gradients, TMatrixT< Scalar_t > &output_weight_gradients, TMatrixT< Scalar_t > &input_state_weight_gradients, TMatrixT< Scalar_t > &forget_state_weight_gradients, TMatrixT< Scalar_t > &candidate_state_weight_gradients, TMatrixT< Scalar_t > &output_state_weight_gradients, TMatrixT< Scalar_t > &input_bias_gradients, TMatrixT< Scalar_t > &forget_bias_gradients, TMatrixT< Scalar_t > &candidate_bias_gradients, TMatrixT< Scalar_t > &output_bias_gradients, TMatrixT< Scalar_t > &di, TMatrixT< Scalar_t > &df, TMatrixT< Scalar_t > &dc, TMatrixT< Scalar_t > &dout, const TMatrixT< Scalar_t > &precStateActivations, const TMatrixT< Scalar_t > &precCellActivations, const TMatrixT< Scalar_t > &fInput, const TMatrixT< Scalar_t > &fForget, const TMatrixT< Scalar_t > &fCandidate, const TMatrixT< Scalar_t > &fOutput, const TMatrixT< Scalar_t > &weights_input, const TMatrixT< Scalar_t > &weights_forget, const TMatrixT< Scalar_t > &weights_candidate, const TMatrixT< Scalar_t > &weights_output, const TMatrixT< Scalar_t > &weights_input_state, const TMatrixT< Scalar_t > &weights_forget_state, const TMatrixT< Scalar_t > &weights_candidate_state, const TMatrixT< Scalar_t > &weights_output_state, const TMatrixT< Scalar_t > &input, TMatrixT< Scalar_t > &input_gradient, TMatrixT< Scalar_t > &cell_gradient, TMatrixT< Scalar_t > &cell_tanh)
Backward pass for LSTM Network.
static void Deflatten(std::vector< TMatrixT< AReal > > &A, const TMatrixT< Scalar_t > &B, size_t index, size_t nRows, size_t nCols)
Transforms each row of B to a matrix and stores it in the tensor B.
static void Im2col(TMatrixT< AReal > &A, const TMatrixT< AReal > &B, size_t imgHeight, size_t imgWidth, size_t fltHeight, size_t fltWidth, size_t strideRows, size_t strideCols, size_t zeroPaddingHeight, size_t zeroPaddingWidth)
Transform the matrix B in local view format, suitable for convolution, and store it in matrix A.
static void Hadamard(TMatrixT< AReal > &A, const TMatrixT< AReal > &B)
In-place Hadamard (element-wise) product of matrices A and B with the result being written into A.
static void UpdateParams(TMatrixT< AReal > &x, TMatrixT< AReal > &tildeX, TMatrixT< AReal > &y, TMatrixT< AReal > &z, TMatrixT< AReal > &fVBiases, TMatrixT< AReal > &fHBiases, TMatrixT< AReal > &fWeights, TMatrixT< AReal > &VBiasError, TMatrixT< AReal > &HBiasError, AReal learningRate, size_t fBatchSize)
static void ScaleAdd(TMatrixT< Scalar_t > &A, const TMatrixT< Scalar_t > &B, Scalar_t beta=1.0)
Adds a the elements in matrix B scaled by c to the elements in the matrix A.
static void Sigmoid(TMatrixT< AReal > &B)
static void CopyDiffArch(TMatrixT< Scalar_t > &A, const AMatrix_t &B)
Definition Reference.h:658
static void SymmetricReluDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static void Identity(TMatrixT< AReal > &B)
static void UpdateParamsLogReg(TMatrixT< AReal > &input, TMatrixT< AReal > &output, TMatrixT< AReal > &difference, TMatrixT< AReal > &p, TMatrixT< AReal > &fWeights, TMatrixT< AReal > &fBiases, AReal learningRate, size_t fBatchSize)
static void ConvLayerBackward(std::vector< TMatrixT< AReal > > &, TMatrixT< AReal > &, TMatrixT< AReal > &, std::vector< TMatrixT< AReal > > &, const std::vector< TMatrixT< AReal > > &, const TMatrixT< AReal > &, const std::vector< TMatrixT< AReal > > &, size_t, size_t, size_t, size_t, size_t, size_t, size_t, size_t, size_t, size_t)
Perform the complete backward propagation step in a Convolutional Layer.
Definition Reference.h:457
static void SoftmaxCrossEntropyGradients(TMatrixT< AReal > &dY, const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
static AReal L1Regularization(const TMatrixT< AReal > &W)
static void AddConvBiases(TMatrixT< AReal > &output, const TMatrixT< AReal > &biases)
Add the biases in the Convolutional Layer.
static void DropoutForward(Tensor_t &A, TDescriptors *descriptors, TWorkspace *workspace, Scalar_t p)
Apply dropout with activation probability p to the given matrix A and scale the result by reciprocal ...
static void ConvLayerForward(std::vector< TMatrixT< AReal > > &, std::vector< TMatrixT< AReal > > &, const std::vector< TMatrixT< AReal > > &, const TMatrixT< AReal > &, const TMatrixT< AReal > &, const DNN::CNN::TConvParams &, EActivationFunction, std::vector< TMatrixT< AReal > > &)
Forward propagation in the Convolutional layer.
Definition Reference.h:435
static void AddL2RegularizationGradients(TMatrixT< AReal > &A, const TMatrixT< AReal > &W, AReal weightDecay)
static void Copy(TMatrixT< Scalar_t > &A, const TMatrixT< Scalar_t > &B)
static void CorruptInput(TMatrixT< AReal > &input, TMatrixT< AReal > &corruptedInput, AReal corruptionLevel)
static void InitializeGauss(TMatrixT< AReal > &A)
static AReal MeanSquaredError(const TMatrixT< AReal > &Y, const TMatrixT< AReal > &output, const TMatrixT< AReal > &weights)
static void SqrtElementWise(TMatrixT< AReal > &A)
Square root each element of the matrix A and write the result into A.
static void SumColumns(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
Sum columns of (m x n) matrix A and write the results into the first m elements in A.
static void ReluDerivative(TMatrixT< AReal > &B, const TMatrixT< AReal > &A)
static void ForwardLogReg(TMatrixT< AReal > &input, TMatrixT< AReal > &p, TMatrixT< AReal > &fWeights)
static void ConstMult(TMatrixT< AReal > &A, AReal beta)
Multiply the constant beta to all the elements of matrix A and write the result into A.
static void InitializeGlorotUniform(TMatrixT< AReal > &A)
Sample from a uniform distribution in range [ -lim,+lim] where lim = sqrt(6/N_in+N_out).
static void FastTanh(Tensor_t &B)
Definition Reference.h:240
static void Reshape(TMatrixT< AReal > &A, const TMatrixT< AReal > &B)
Transform the matrix B to a matrix with different dimensions A.
static void AddBiases(TMatrixT< AReal > &A, const TMatrixT< AReal > &biases)
static TRandom & GetRandomGenerator()
static void PrepareInternals(std::vector< TMatrixT< AReal > > &)
Dummy placeholder - preparation is currently only required for the CUDA architecture.
Definition Reference.h:432
static Matrix_t & RecurrentLayerBackward(TMatrixT< Scalar_t > &state_gradients_backward, TMatrixT< Scalar_t > &input_weight_gradients, TMatrixT< Scalar_t > &state_weight_gradients, TMatrixT< Scalar_t > &bias_gradients, TMatrixT< Scalar_t > &df, const TMatrixT< Scalar_t > &state, const TMatrixT< Scalar_t > &weights_input, const TMatrixT< Scalar_t > &weights_state, const TMatrixT< Scalar_t > &input, TMatrixT< Scalar_t > &input_gradient)
Backpropagation step for a Recurrent Neural Network.
static void InitializeUniform(TMatrixT< AReal > &A)
This is the base class for the ROOT Random number generators.
Definition TRandom.h:28
Double_t y[n]
Definition legend1.C:17
Double_t x[n]
Definition legend1.C:17
std::shared_ptr< std::function< double(double)> > Tanh
Definition NeuralNet.cxx:29
double weightDecay(double error, ItWeight itWeight, ItWeight itWeightEnd, double factorWeightDecay, EnumRegularization eRegularization)
compute the weight decay for regularization (L1 or L2)
EActivationFunction
Enum that represents layer activation functions.
Definition Functions.h:32
std::shared_ptr< std::function< double(double)> > Gauss
Definition NeuralNet.cxx:12
std::shared_ptr< std::function< double(double)> > Sigmoid
Definition NeuralNet.cxx:26
std::shared_ptr< std::function< double(double)> > SoftSign
Definition NeuralNet.cxx:32
create variable transformations