76#ifdef MethodMLP_UseMinuit__
94 fUseRegulator(false), fCalculateErrors(false),
95 fPrior(0.0), fPriorDev(0), fUpdateLimit(0),
96 fTrainingMethod(kBFGS), fTrainMethodS(
"BFGS"),
97 fSamplingFraction(1.0), fSamplingEpoch(0.0), fSamplingWeight(0.0),
98 fSamplingTraining(false), fSamplingTesting(false),
99 fLastAlpha(0.0), fTau(0.),
100 fResetStep(0), fLearnRate(0.0), fDecayRate(0.0),
101 fBPMode(kSequential), fBpModeS(
"None"),
102 fBatchSize(0), fTestRate(0), fEpochMon(false),
103 fGA_nsteps(0), fGA_preCalc(0), fGA_SC_steps(0),
104 fGA_SC_rate(0), fGA_SC_factor(0.0),
105 fDeviationsFromTargets(0),
117 fUseRegulator(false), fCalculateErrors(false),
118 fPrior(0.0), fPriorDev(0), fUpdateLimit(0),
119 fTrainingMethod(kBFGS), fTrainMethodS(
"BFGS"),
120 fSamplingFraction(1.0), fSamplingEpoch(0.0), fSamplingWeight(0.0),
121 fSamplingTraining(false), fSamplingTesting(false),
122 fLastAlpha(0.0), fTau(0.),
123 fResetStep(0), fLearnRate(0.0), fDecayRate(0.0),
124 fBPMode(kSequential), fBpModeS(
"None"),
125 fBatchSize(0), fTestRate(0), fEpochMon(false),
126 fGA_nsteps(0), fGA_preCalc(0), fGA_SC_steps(0),
127 fGA_SC_rate(0), fGA_SC_factor(0.0),
128 fDeviationsFromTargets(0),
166 SetSignalReferenceCut( 0.5 );
167#ifdef MethodMLP_UseMinuit__
196 DeclareOptionRef(fTrainMethodS=
"BP",
"TrainingMethod",
197 "Train with Back-Propagation (BP), BFGS Algorithm (BFGS), or Genetic Algorithm (GA - slower and worse)");
202 DeclareOptionRef(fLearnRate=0.02,
"LearningRate",
"ANN learning rate parameter");
203 DeclareOptionRef(fDecayRate=0.01,
"DecayRate",
"Decay rate for learning parameter");
204 DeclareOptionRef(fTestRate =10,
"TestRate",
"Test for overtraining performed at each #th epochs");
205 DeclareOptionRef(fEpochMon =
kFALSE,
"EpochMonitoring",
"Provide epoch-wise monitoring plots according to TestRate (caution: causes big ROOT output file!)" );
207 DeclareOptionRef(fSamplingFraction=1.0,
"Sampling",
"Only 'Sampling' (randomly selected) events are trained each epoch");
208 DeclareOptionRef(fSamplingEpoch=1.0,
"SamplingEpoch",
"Sampling is used for the first 'SamplingEpoch' epochs, afterwards, all events are taken for training");
209 DeclareOptionRef(fSamplingWeight=1.0,
"SamplingImportance",
" The sampling weights of events in epochs which successful (worse estimator than before) are multiplied with SamplingImportance, else they are divided.");
211 DeclareOptionRef(fSamplingTraining=
kTRUE,
"SamplingTraining",
"The training sample is sampled");
212 DeclareOptionRef(fSamplingTesting=
kFALSE,
"SamplingTesting" ,
"The testing sample is sampled");
214 DeclareOptionRef(fResetStep=50,
"ResetStep",
"How often BFGS should reset history");
215 DeclareOptionRef(fTau =3.0,
"Tau",
"LineSearch \"size step\"");
217 DeclareOptionRef(fBpModeS=
"sequential",
"BPMode",
218 "Back-propagation learning mode: sequential or batch");
219 AddPreDefVal(
TString(
"sequential"));
220 AddPreDefVal(
TString(
"batch"));
222 DeclareOptionRef(fBatchSize=-1,
"BatchSize",
223 "Batch size: number of events/batch, only set if in Batch Mode, -1 for BatchSize=number_of_events");
225 DeclareOptionRef(fImprovement=1
e-30,
"ConvergenceImprove",
226 "Minimum improvement which counts as improvement (<0 means automatic convergence check is turned off)");
228 DeclareOptionRef(fSteps=-1,
"ConvergenceTests",
229 "Number of steps (without improvement) required for convergence (<0 means automatic convergence check is turned off)");
231 DeclareOptionRef(fUseRegulator=
kFALSE,
"UseRegulator",
232 "Use regulator to avoid over-training");
233 DeclareOptionRef(fUpdateLimit=10000,
"UpdateLimit",
234 "Maximum times of regulator update");
235 DeclareOptionRef(fCalculateErrors=
kFALSE,
"CalculateErrors",
236 "Calculates inverse Hessian matrix at the end of the training to be able to calculate the uncertainties of an MVA value");
238 DeclareOptionRef(fWeightRange=1.0,
"WeightRange",
239 "Take the events for the estimator calculations from small deviations from the desired value to large deviations only over the weight range");
251 if (IgnoreEventsWithNegWeightsInTraining()) {
253 <<
"Will ignore negative events in training!"
258 if (fTrainMethodS ==
"BP" ) fTrainingMethod = kBP;
259 else if (fTrainMethodS ==
"BFGS") fTrainingMethod = kBFGS;
260 else if (fTrainMethodS ==
"GA" ) fTrainingMethod = kGA;
262 if (fBpModeS ==
"sequential") fBPMode = kSequential;
263 else if (fBpModeS ==
"batch") fBPMode = kBatch;
267 if (fBPMode == kBatch) {
269 Int_t numEvents = Data()->GetNEvents();
270 if (fBatchSize < 1 || fBatchSize > numEvents) fBatchSize = numEvents;
279 Log() << kDEBUG <<
"Initialize learning rates" <<
Endl;
281 Int_t numSynapses = fSynapses->GetEntriesFast();
282 for (
Int_t i = 0; i < numSynapses; i++) {
283 synapse = (
TSynapse*)fSynapses->At(i);
284 synapse->SetLearningRate(fLearnRate);
295 Log() << kFATAL <<
"<CalculateEstimator> fatal error: wrong tree type: " << treeType <<
Endl;
299 Data()->SetCurrentType(treeType);
310 if (fEpochMon && iEpoch >= 0 && !DoRegression()) {
311 histS =
new TH1F( nameS, nameS, nbin, -limit, limit );
312 histB =
new TH1F( nameB, nameB, nbin, -limit, limit );
318 Int_t nEvents = GetNEvents();
319 UInt_t nClasses = DataInfo().GetNClasses();
320 UInt_t nTgts = DataInfo().GetNTargets();
324 if( fWeightRange < 1.f ){
325 fDeviationsFromTargets =
new std::vector<std::pair<Float_t,Float_t> >(nEvents);
328 for (
Int_t i = 0; i < nEvents; i++) {
330 const Event* ev = GetEvent(i);
332 if ((ev->GetWeight() < 0) && IgnoreEventsWithNegWeightsInTraining()
339 ForceNetworkInputs( ev );
340 ForceNetworkCalculations();
343 if (DoRegression()) {
344 for (
UInt_t itgt = 0; itgt < nTgts; itgt++) {
345 v = GetOutputNeuron( itgt )->GetActivationValue();
346 Double_t targetValue = ev->GetTarget( itgt );
351 }
else if (DoMulticlass() ) {
352 UInt_t cls = ev->GetClass();
353 if (fEstimator==kCE){
355 for (
UInt_t icls = 0; icls < nClasses; icls++) {
356 Float_t activationValue = GetOutputNeuron( icls )->GetActivationValue();
357 norm += exp( activationValue );
359 d = exp( activationValue );
364 for (
UInt_t icls = 0; icls < nClasses; icls++) {
365 Double_t desired = (icls==cls) ? 1.0 : 0.0;
366 v = GetOutputNeuron( icls )->GetActivationValue();
367 d = (desired-
v)*(desired-
v);
372 Double_t desired = DataInfo().IsSignal(ev)?1.:0.;
373 v = GetOutputNeuron()->GetActivationValue();
374 if (fEstimator==kMSE)
d = (desired-
v)*(desired-
v);
379 if( fDeviationsFromTargets )
380 fDeviationsFromTargets->push_back(std::pair<Float_t,Float_t>(
d,
w));
386 if (DataInfo().IsSignal(ev) && histS != 0) histS->Fill(
float(
v), float(
w) );
387 else if (histB != 0) histB->Fill(
float(
v), float(
w) );
391 if( fDeviationsFromTargets ) {
392 std::sort(fDeviationsFromTargets->begin(),fDeviationsFromTargets->end());
394 Float_t sumOfWeightsInRange = fWeightRange*sumOfWeights;
397 Float_t weightRangeCut = fWeightRange*sumOfWeights;
399 for(std::vector<std::pair<Float_t,Float_t> >::iterator itDev = fDeviationsFromTargets->begin(), itDevEnd = fDeviationsFromTargets->end(); itDev != itDevEnd; ++itDev ){
400 float deviation = (*itDev).first;
401 float devWeight = (*itDev).second;
402 weightSum += devWeight;
403 if( weightSum <= weightRangeCut ) {
404 estimator += devWeight*deviation;
408 sumOfWeights = sumOfWeightsInRange;
409 delete fDeviationsFromTargets;
412 if (histS != 0) fEpochMonHistS.push_back( histS );
413 if (histB != 0) fEpochMonHistB.push_back( histB );
418 estimator = estimator/
Float_t(sumOfWeights);
423 Data()->SetCurrentType( saveType );
426 if (fEpochMon && iEpoch >= 0 && !DoRegression() && treeType ==
Types::kTraining) {
427 CreateWeightMonitoringHists(
TString::Format(
"epochmonitoring___epoch_%04i_weights_hist", iEpoch), &fEpochMonHistW );
439 Log() << kFATAL <<
"ANN Network is not initialized, doing it now!"<<
Endl;
440 SetAnalysisType(GetAnalysisType());
442 Log() << kDEBUG <<
"reinitialize learning rates" <<
Endl;
443 InitializeLearningRates();
445 PrintMessage(
"Training Network");
447 Int_t nEvents=GetNEvents();
448 Int_t nSynapses=fSynapses->GetEntriesFast();
449 if (nSynapses>nEvents)
450 Log()<<kWARNING<<
"ANN too complicated: #events="<<nEvents<<
"\t#synapses="<<nSynapses<<
Endl;
453#ifdef MethodMLP_UseMinuit__
454 if (useMinuit) MinuitMinimize();
456 if (fTrainingMethod == kGA) GeneticMinimize();
457 else if (fTrainingMethod == kBFGS) BFGSMinimize(nEpochs);
458 else BackPropagationMinimize(nEpochs);
464 Log()<<kINFO<<
"Finalizing handling of Regulator terms, trainE="<<trainE<<
" testE="<<testE<<
Endl;
466 Log()<<kINFO<<
"Done with handling of Regulator terms"<<
Endl;
469 if( fCalculateErrors || fUseRegulator )
471 Int_t numSynapses=fSynapses->GetEntriesFast();
472 fInvHessian.ResizeTo(numSynapses,numSynapses);
473 GetApproxInvHessian( fInvHessian ,
false);
482 Timer timer( (fSteps>0?100:nEpochs), GetName() );
488 fEstimatorHistTrain =
new TH1F(
"estimatorHistTrain",
"training estimator",
489 nbinTest,
Int_t(fTestRate/2), nbinTest*fTestRate+
Int_t(fTestRate/2) );
490 fEstimatorHistTest =
new TH1F(
"estimatorHistTest",
"test estimator",
491 nbinTest,
Int_t(fTestRate/2), nbinTest*fTestRate+
Int_t(fTestRate/2) );
494 Int_t nSynapses = fSynapses->GetEntriesFast();
495 Int_t nWeights = nSynapses;
497 for (
Int_t i=0;i<nSynapses;i++) {
502 std::vector<Double_t> buffer( nWeights );
503 for (
Int_t i=0;i<nWeights;i++) buffer[i] = 0.;
506 TMatrixD Hessian ( nWeights, nWeights );
510 Int_t RegUpdateTimes=0;
518 if(fSamplingTraining || fSamplingTesting)
519 Data()->InitSampling(1.0,1.0,fRandomSeed);
521 if (fSteps > 0) Log() << kINFO <<
"Inaccurate progress timing for MLP... " <<
Endl;
522 timer.DrawProgressBar( 0 );
525 for (
Int_t i = 0; i < nEpochs; i++) {
527 if (
Float_t(i)/nEpochs < fSamplingEpoch) {
528 if ((i+1)%fTestRate == 0 || (i == 0)) {
529 if (fSamplingTraining) {
531 Data()->InitSampling(fSamplingFraction,fSamplingWeight);
532 Data()->CreateSampling();
534 if (fSamplingTesting) {
536 Data()->InitSampling(fSamplingFraction,fSamplingWeight);
537 Data()->CreateSampling();
543 Data()->InitSampling(1.0,1.0);
545 Data()->InitSampling(1.0,1.0);
556 SetGammaDelta( Gamma, Delta, buffer );
558 if (i % fResetStep == 0 && i<0.5*nEpochs) {
564 if (GetHessian( Hessian, Gamma, Delta )) {
569 else SetDir( Hessian, Dir );
573 if (DerivDir( Dir ) > 0) {
578 if (LineSearch( Dir, buffer, &dError )) {
582 if (LineSearch(Dir, buffer, &dError)) {
584 Log() << kFATAL <<
"Line search failed! Huge troubles somewhere..." <<
Endl;
589 if (dError<0) Log()<<kWARNING<<
"\nnegative dError=" <<dError<<
Endl;
592 if ( fUseRegulator && RegUpdateTimes<fUpdateLimit && RegUpdateCD>=5 && fabs(dError)<0.1*AccuError) {
593 Log()<<kDEBUG<<
"\n\nUpdate regulators "<<RegUpdateTimes<<
" on epoch "<<i<<
"\tdError="<<dError<<
Endl;
603 if ((i+1)%fTestRate == 0) {
610 fEstimatorHistTrain->Fill( i+1, trainE );
611 fEstimatorHistTest ->Fill( i+1, testE );
614 if ((testE < GetCurrentValue()) || (GetCurrentValue()<1
e-100)) {
617 Data()->EventResult( success );
619 SetCurrentValue( testE );
620 if (HasConverged()) {
621 if (
Float_t(i)/nEpochs < fSamplingEpoch) {
622 Int_t newEpoch =
Int_t(fSamplingEpoch*nEpochs);
624 ResetConvergenceCounter();
634 if (
Float_t(i)/nEpochs < fSamplingEpoch)
636 progress = Progress()*fSamplingFraction*100*fSamplingEpoch;
640 progress = 100.0*(fSamplingFraction*fSamplingEpoch+(1.0-fSamplingEpoch)*Progress());
642 Float_t progress2= 100.0*RegUpdateTimes/fUpdateLimit;
643 if (progress2>progress) progress=progress2;
644 timer.DrawProgressBar(
Int_t(progress), convText );
648 if (progress<i) progress=i;
649 timer.DrawProgressBar( progress, convText );
664 Int_t nWeights = fSynapses->GetEntriesFast();
667 Int_t nSynapses = fSynapses->GetEntriesFast();
668 for (
Int_t i=0;i<nSynapses;i++) {
670 Gamma[IDX++][0] = -synapse->GetDEDw();
673 for (
Int_t i=0;i<nWeights;i++) Delta[i][0] = buffer[i];
678 for (
Int_t i=0;i<nSynapses;i++)
681 Gamma[IDX++][0] += synapse->GetDEDw();
689 Int_t nSynapses = fSynapses->GetEntriesFast();
690 for (
Int_t i=0;i<nSynapses;i++) {
695 Int_t nEvents = GetNEvents();
696 Int_t nPosEvents = nEvents;
697 for (
Int_t i=0;i<nEvents;i++) {
699 const Event* ev = GetEvent(i);
700 if ((ev->GetWeight() < 0) && IgnoreEventsWithNegWeightsInTraining()
708 for (
Int_t j=0;j<nSynapses;j++) {
710 synapse->
SetDEDw( synapse->GetDEDw() + synapse->GetDelta() );
714 for (
Int_t i=0;i<nSynapses;i++) {
717 if (fUseRegulator) DEDw+=fPriorDev[i];
718 synapse->SetDEDw( DEDw / nPosEvents );
726 Double_t eventWeight = ev->GetWeight();
728 ForceNetworkInputs( ev );
729 ForceNetworkCalculations();
731 if (DoRegression()) {
732 UInt_t ntgt = DataInfo().GetNTargets();
733 for (
UInt_t itgt = 0; itgt < ntgt; itgt++) {
734 Double_t desired = ev->GetTarget(itgt);
735 Double_t error = ( GetOutputNeuron( itgt )->GetActivationValue() - desired )*eventWeight;
736 GetOutputNeuron( itgt )->SetError(error);
738 }
else if (DoMulticlass()) {
739 UInt_t nClasses = DataInfo().GetNClasses();
740 UInt_t cls = ev->GetClass();
741 for (
UInt_t icls = 0; icls < nClasses; icls++) {
742 Double_t desired = ( cls==icls ? 1.0 : 0.0 );
743 Double_t error = ( GetOutputNeuron( icls )->GetActivationValue() - desired )*eventWeight;
744 GetOutputNeuron( icls )->SetError(error);
747 Double_t desired = GetDesiredOutput( ev );
749 if (fEstimator==kMSE) error = ( GetOutputNeuron()->GetActivationValue() - desired )*eventWeight;
750 else if (fEstimator==kCE) error = -eventWeight/(GetOutputNeuron()->GetActivationValue() -1 + desired);
751 GetOutputNeuron()->SetError(error);
754 CalculateNeuronDeltas();
755 for (
Int_t j=0;j<fSynapses->GetEntriesFast();j++) {
758 synapse->CalculateDelta();
767 Int_t nSynapses = fSynapses->GetEntriesFast();
769 for (
Int_t i=0;i<nSynapses;i++) {
771 Dir[IDX++][0] = -synapse->
GetDEDw();
801 Int_t nSynapses = fSynapses->GetEntriesFast();
804 for (
Int_t i=0;i<nSynapses;i++) {
806 DEDw[IDX++][0] = synapse->
GetDEDw();
809 dir = Hessian * DEDw;
810 for (
Int_t i=0;i<IDX;i++) dir[i][0] = -dir[i][0];
818 Int_t nSynapses = fSynapses->GetEntriesFast();
821 for (
Int_t i=0;i<nSynapses;i++) {
823 Result += Dir[IDX++][0] * synapse->GetDEDw();
833 Int_t nSynapses = fSynapses->GetEntriesFast();
834 Int_t nWeights = nSynapses;
836 std::vector<Double_t> Origin(nWeights);
837 for (
Int_t i=0;i<nSynapses;i++) {
848 if (alpha2 < 0.01) alpha2 = 0.01;
849 else if (alpha2 > 2.0) alpha2 = 2.0;
853 SetDirWeights( Origin, Dir, alpha2 );
861 for (
Int_t i=0;i<100;i++) {
863 SetDirWeights(Origin, Dir, alpha3);
875 SetDirWeights(Origin, Dir, 0.);
880 for (
Int_t i=0;i<100;i++) {
883 Log() << kWARNING <<
"linesearch, starting to investigate direction opposite of steepestDIR" <<
Endl;
884 alpha2 = -alpha_original;
886 SetDirWeights(Origin, Dir, alpha2);
896 SetDirWeights(Origin, Dir, 0.);
897 Log() << kWARNING <<
"linesearch, failed even in opposite direction of steepestDIR" <<
Endl;
903 if (alpha1>0 && alpha2>0 && alpha3 > 0) {
904 fLastAlpha = 0.5 * (alpha1 + alpha3 -
905 (err3 - err1) / ((err3 - err2) / ( alpha3 - alpha2 )
906 - ( err2 - err1 ) / (alpha2 - alpha1 )));
912 fLastAlpha = fLastAlpha < 10000 ? fLastAlpha : 10000;
914 SetDirWeights(Origin, Dir, fLastAlpha);
921 if (finalError > err1) {
922 Log() << kWARNING <<
"Line search increased error! Something is wrong."
923 <<
"fLastAlpha=" << fLastAlpha <<
"al123=" << alpha1 <<
" "
924 << alpha2 <<
" " << alpha3 <<
" err1="<< err1 <<
" errfinal=" << finalError <<
Endl;
927 for (
Int_t i=0;i<nSynapses;i++) {
929 buffer[IDX] = synapse->GetWeight() - Origin[IDX];
933 if (dError) (*dError)=(errOrigin-finalError)/finalError;
943 Int_t nSynapses = fSynapses->GetEntriesFast();
945 for (
Int_t i=0;i<nSynapses;i++) {
947 synapse->
SetWeight( Origin[IDX] + Dir[IDX][0] * alpha );
950 if (fUseRegulator) UpdatePriors();
958 Int_t nEvents = GetNEvents();
959 UInt_t ntgts = GetNTargets();
962 for (
Int_t i=0;i<nEvents;i++) {
963 const Event* ev = GetEvent(i);
965 if ((ev->GetWeight() < 0) && IgnoreEventsWithNegWeightsInTraining()
972 if (DoRegression()) {
973 for (
UInt_t itgt = 0; itgt < ntgts; itgt++) {
974 error += GetMSEErr( ev, itgt );
976 }
else if ( DoMulticlass() ){
977 for(
UInt_t icls = 0, iclsEnd = DataInfo().GetNClasses(); icls < iclsEnd; icls++ ){
978 error += GetMSEErr( ev, icls );
981 if (fEstimator==kMSE) error = GetMSEErr( ev );
982 else if (fEstimator==kCE) error= GetCEErr( ev );
984 Result += error * ev->GetWeight();
986 if (fUseRegulator) Result+=fPrior;
987 if (Result<0) Log()<<kWARNING<<
"\nNegative Error!!! :"<<Result-fPrior<<
"+"<<fPrior<<
Endl;
996 Double_t output = GetOutputNeuron(
index )->GetActivationValue();
998 if (DoRegression())
target = ev->GetTarget(
index );
999 else if (DoMulticlass())
target = (ev->GetClass() ==
index ? 1.0 : 0.0 );
1000 else target = GetDesiredOutput( ev );
1013 Double_t output = GetOutputNeuron(
index )->GetActivationValue();
1015 if (DoRegression())
target = ev->GetTarget(
index );
1016 else if (DoMulticlass())
target = (ev->GetClass() ==
index ? 1.0 : 0.0 );
1017 else target = GetDesiredOutput( ev );
1030 Timer timer( (fSteps>0?100:nEpochs), GetName() );
1037 fEstimatorHistTrain =
new TH1F(
"estimatorHistTrain",
"training estimator",
1038 nbinTest,
Int_t(fTestRate/2), nbinTest*fTestRate+
Int_t(fTestRate/2) );
1039 fEstimatorHistTest =
new TH1F(
"estimatorHistTest",
"test estimator",
1040 nbinTest,
Int_t(fTestRate/2), nbinTest*fTestRate+
Int_t(fTestRate/2) );
1042 if(fSamplingTraining || fSamplingTesting)
1043 Data()->InitSampling(1.0,1.0,fRandomSeed);
1045 if (fSteps > 0) Log() << kINFO <<
"Inaccurate progress timing for MLP... " <<
Endl;
1046 timer.DrawProgressBar(0);
1053 for (
Int_t i = 0; i < nEpochs; i++) {
1055 if (
Float_t(i)/nEpochs < fSamplingEpoch) {
1056 if ((i+1)%fTestRate == 0 || (i == 0)) {
1057 if (fSamplingTraining) {
1059 Data()->InitSampling(fSamplingFraction,fSamplingWeight);
1060 Data()->CreateSampling();
1062 if (fSamplingTesting) {
1064 Data()->InitSampling(fSamplingFraction,fSamplingWeight);
1065 Data()->CreateSampling();
1071 Data()->InitSampling(1.0,1.0);
1073 Data()->InitSampling(1.0,1.0);
1078 DecaySynapseWeights(i >= lateEpoch);
1081 if ((i+1)%fTestRate == 0) {
1086 fEstimatorHistTrain->Fill( i+1, trainE );
1087 fEstimatorHistTest ->Fill( i+1, testE );
1090 if ((testE < GetCurrentValue()) || (GetCurrentValue()<1
e-100)) {
1093 Data()->EventResult( success );
1095 SetCurrentValue( testE );
1096 if (HasConverged()) {
1097 if (
Float_t(i)/nEpochs < fSamplingEpoch) {
1098 Int_t newEpoch =
Int_t(fSamplingEpoch*nEpochs);
1100 ResetConvergenceCounter();
1103 if (lateEpoch > i) lateEpoch = i;
1113 if (
Float_t(i)/nEpochs < fSamplingEpoch)
1114 progress = Progress()*fSamplingEpoch*fSamplingFraction*100;
1116 progress = 100*(fSamplingEpoch*fSamplingFraction+(1.0-fSamplingFraction*fSamplingEpoch)*Progress());
1118 timer.DrawProgressBar(
Int_t(progress), convText );
1121 timer.DrawProgressBar( i, convText );
1131 Int_t nEvents = Data()->GetNEvents();
1135 for (
Int_t i = 0; i < nEvents; i++)
index[i] = i;
1136 Shuffle(
index, nEvents);
1139 for (
Int_t i = 0; i < nEvents; i++) {
1142 if ((ev->GetWeight() < 0) && IgnoreEventsWithNegWeightsInTraining()
1147 TrainOneEvent(
index[i]);
1150 if (fBPMode == kBatch && (i+1)%fBatchSize == 0) {
1151 AdjustSynapseWeights();
1152 if (fgPRINT_BATCH) {
1181 for (
Int_t i = 0; i <
n; i++) {
1182 j = (
Int_t) (frgen->Rndm() *
a);
1198 Int_t numSynapses = fSynapses->GetEntriesFast();
1199 for (
Int_t i = 0; i < numSynapses; i++) {
1200 synapse = (
TSynapse*)fSynapses->At(i);
1201 if (lateEpoch) synapse->DecayLearningRate(
TMath::Sqrt(fDecayRate));
1202 else synapse->DecayLearningRate(fDecayRate);
1223 if (
type == 0) desired = fOutput->GetMin();
1224 else desired = fOutput->GetMax();
1230 for (
UInt_t j = 0; j < GetNvar(); j++) {
1233 neuron = GetInputNeuron(j);
1237 ForceNetworkCalculations();
1238 UpdateNetwork(desired, eventWeight);
1252 const Event * ev = GetEvent(ievt);
1253 Double_t eventWeight = ev->GetWeight();
1254 ForceNetworkInputs( ev );
1255 ForceNetworkCalculations();
1256 if (DoRegression()) UpdateNetwork( ev->GetTargets(), eventWeight );
1257 if (DoMulticlass()) UpdateNetwork( *DataInfo().GetTargetsForMulticlass( ev ), eventWeight );
1258 else UpdateNetwork( GetDesiredOutput( ev ), eventWeight );
1266 return DataInfo().IsSignal(ev)?fOutput->GetMax():fOutput->GetMin();
1275 Double_t error = GetOutputNeuron()->GetActivationValue() - desired;
1276 if (fEstimator==kMSE) error = GetOutputNeuron()->GetActivationValue() - desired ;
1277 else if (fEstimator==kCE) error = -1./(GetOutputNeuron()->GetActivationValue() -1 + desired);
1278 else Log() << kFATAL <<
"Estimator type unspecified!!" <<
Endl;
1279 error *= eventWeight;
1280 GetOutputNeuron()->SetError(error);
1281 CalculateNeuronDeltas();
1293 for (
UInt_t i = 0, iEnd = desired.size(); i < iEnd; ++i) {
1294 Double_t act = GetOutputNeuron(i)->GetActivationValue();
1299 for (
UInt_t i = 0, iEnd = desired.size(); i < iEnd; ++i) {
1300 Double_t act = GetOutputNeuron(i)->GetActivationValue();
1302 Double_t error = output - desired.at(i);
1303 error *= eventWeight;
1304 GetOutputNeuron(i)->SetError(error);
1308 CalculateNeuronDeltas();
1319 Int_t numLayers = fNetwork->GetEntriesFast();
1324 for (
Int_t i = numLayers-1; i >= 0; i--) {
1326 numNeurons = curLayer->GetEntriesFast();
1328 for (
Int_t j = 0; j < numNeurons; j++) {
1329 neuron = (
TNeuron*) curLayer->At(j);
1345 PrintMessage(
"Minimizing Estimator with GA");
1351 fGA_SC_factor = 0.95;
1355 std::vector<Interval*> ranges;
1357 Int_t numWeights = fSynapses->GetEntriesFast();
1358 for (
Int_t ivar=0; ivar< numWeights; ivar++) {
1359 ranges.push_back(
new Interval( 0, GetXmax(ivar) - GetXmin(ivar) ));
1365 Double_t estimator = CalculateEstimator();
1366 Log() << kINFO <<
"GA: estimator after optimization: " << estimator <<
Endl;
1374 return ComputeEstimator( parameters );
1383 Int_t numSynapses = fSynapses->GetEntriesFast();
1385 for (
Int_t i = 0; i < numSynapses; i++) {
1386 synapse = (
TSynapse*)fSynapses->At(i);
1387 synapse->SetWeight(parameters.at(i));
1389 if (fUseRegulator) UpdatePriors();
1391 Double_t estimator = CalculateEstimator();
1404 Int_t numLayers = fNetwork->GetEntriesFast();
1406 for (
Int_t i = 0; i < numLayers; i++) {
1408 numNeurons = curLayer->GetEntriesFast();
1410 for (
Int_t j = 0; j < numNeurons; j++) {
1411 neuron = (
TNeuron*) curLayer->At(j);
1426 Int_t numLayers = fNetwork->GetEntriesFast();
1428 for (
Int_t i = numLayers-1; i >= 0; i--) {
1430 numNeurons = curLayer->GetEntriesFast();
1432 for (
Int_t j = 0; j < numNeurons; j++) {
1433 neuron = (
TNeuron*) curLayer->At(j);
1445 Int_t nSynapses = fSynapses->GetEntriesFast();
1446 for (
Int_t i=0;i<nSynapses;i++) {
1448 fPrior+=0.5*fRegulators[fRegulatorIdx[i]]*(synapse->GetWeight())*(synapse->GetWeight());
1449 fPriorDev.push_back(fRegulators[fRegulatorIdx[i]]*(synapse->GetWeight()));
1458 GetApproxInvHessian(InvH);
1459 Int_t numSynapses=fSynapses->GetEntriesFast();
1460 Int_t numRegulators=fRegulators.size();
1463 std::vector<Int_t> nWDP(numRegulators);
1464 std::vector<Double_t> trace(numRegulators),weightSum(numRegulators);
1465 for (
int i=0;i<numSynapses;i++) {
1467 Int_t idx=fRegulatorIdx[i];
1469 trace[idx]+=InvH[i][i];
1470 gamma+=1-fRegulators[idx]*InvH[i][i];
1471 weightSum[idx]+=(synapses->GetWeight())*(synapses->GetWeight());
1473 if (fEstimator==kMSE) {
1474 if (GetNEvents()>gamma) variance=CalculateEstimator(
Types::kTraining, 0 )/(1-(gamma/GetNEvents()));
1479 for (
int i=0;i<numRegulators;i++)
1482 fRegulators[i]=variance*nWDP[i]/(weightSum[i]+variance*trace[i]);
1483 if (fRegulators[i]<0) fRegulators[i]=0;
1484 Log()<<kDEBUG<<
"R"<<i<<
":"<<fRegulators[i]<<
"\t";
1489 Log()<<kDEBUG<<
"\n"<<
"trainE:"<<trainE<<
"\ttestE:"<<testE<<
"\tvariance:"<<variance<<
"\tgamma:"<<gamma<<
Endl;
1497 Int_t numSynapses=fSynapses->GetEntriesFast();
1498 InvHessian.
ResizeTo( numSynapses, numSynapses );
1502 Int_t nEvents = GetNEvents();
1503 for (
Int_t i=0;i<nEvents;i++) {
1505 double outputValue=GetMvaValue();
1506 GetOutputNeuron()->SetError(1./fOutput->EvalDerivative(GetOutputNeuron()->GetValue()));
1507 CalculateNeuronDeltas();
1508 for (
Int_t j = 0; j < numSynapses; j++){
1511 synapses->CalculateDelta();
1512 sens[j][0]=sensT[0][j]=synapses->GetDelta();
1514 if (fEstimator==kMSE ) InvHessian+=sens*sensT;
1515 else if (fEstimator==kCE) InvHessian+=(outputValue*(1-outputValue))*sens*sensT;
1520 for (
Int_t i = 0; i < numSynapses; i++){
1521 InvHessian[i][i]+=fRegulators[fRegulatorIdx[i]];
1525 for (
Int_t i = 0; i < numSynapses; i++){
1526 InvHessian[i][i]+=1
e-6;
1541 if (!fCalculateErrors || errLower==0 || errUpper==0)
1544 Double_t MvaUpper,MvaLower,median,variance;
1545 Int_t numSynapses=fSynapses->GetEntriesFast();
1546 if (fInvHessian.GetNcols()!=numSynapses) {
1547 Log() << kWARNING <<
"inconsistent dimension " << fInvHessian.GetNcols() <<
" vs " << numSynapses <<
Endl;
1551 GetOutputNeuron()->SetError(1./fOutput->EvalDerivative(GetOutputNeuron()->GetValue()));
1553 CalculateNeuronDeltas();
1554 for (
Int_t i = 0; i < numSynapses; i++){
1557 synapses->CalculateDelta();
1558 sensT[0][i]=synapses->GetDelta();
1560 sens.Transpose(sensT);
1561 TMatrixD sig=sensT*fInvHessian*sens;
1563 median=GetOutputNeuron()->GetValue();
1566 Log()<<kWARNING<<
"Negative variance!!! median=" << median <<
"\tvariance(sigma^2)=" << variance <<
Endl;
1569 variance=sqrt(variance);
1572 MvaUpper=fOutput->Eval(median+variance);
1574 *errUpper=MvaUpper-MvaValue;
1577 MvaLower=fOutput->Eval(median-variance);
1579 *errLower=MvaValue-MvaLower;
1585#ifdef MethodMLP_UseMinuit__
1590void TMVA::MethodMLP::MinuitMinimize()
1592 fNumberOfWeights = fSynapses->GetEntriesFast();
1601 tfitter->ExecuteCommand(
"SET PRINTOUT", args, 1 );
1602 tfitter->ExecuteCommand(
"SET NOWARNINGS", args, 0 );
1607 for (
Int_t ipar=0; ipar < fNumberOfWeights; ipar++) {
1609 tfitter->SetParameter( ipar,
1610 parName,
w[ipar], 0.1, 0, 0 );
1614 tfitter->SetFCN( &IFCN );
1618 tfitter->ExecuteCommand(
"SET STRATEGY", args, 1 );
1622 tfitter->ExecuteCommand(
"MIGRAD", args, 1 );
1628 tfitter->ExecuteCommand(
"IMPROVE", args, 1 );
1632 tfitter->ExecuteCommand(
"MINOS", args, 1 );
1653 ((MethodMLP*)GetThisPtr())->FCN( npars, grad,
f, fitPars, iflag );
1656TTHREAD_TLS(
Int_t) nc = 0;
1657TTHREAD_TLS(
double) minf = 1000000;
1662 for (
Int_t ipar=0; ipar<fNumberOfWeights; ipar++) {
1668 f = CalculateEstimator();
1671 if (
f < minf) minf =
f;
1672 for (
Int_t ipar=0; ipar<fNumberOfWeights; ipar++)
Log() <<
kDEBUG << fitPars[ipar] <<
" ";
1674 Log() <<
kDEBUG <<
"***** New estimator: " <<
f <<
" min: " << minf <<
" --> ncalls: " << nc <<
Endl;
1708 Log() << col <<
"--- Short description:" << colres <<
Endl;
1710 Log() <<
"The MLP artificial neural network (ANN) is a traditional feed-" <<
Endl;
1711 Log() <<
"forward multilayer perceptron implementation. The MLP has a user-" <<
Endl;
1712 Log() <<
"defined hidden layer architecture, while the number of input (output)" <<
Endl;
1713 Log() <<
"nodes is determined by the input variables (output classes, i.e., " <<
Endl;
1714 Log() <<
"signal and one background). " <<
Endl;
1716 Log() << col <<
"--- Performance optimisation:" << colres <<
Endl;
1718 Log() <<
"Neural networks are stable and performing for a large variety of " <<
Endl;
1719 Log() <<
"linear and non-linear classification problems. However, in contrast" <<
Endl;
1720 Log() <<
"to (e.g.) boosted decision trees, the user is advised to reduce the " <<
Endl;
1721 Log() <<
"number of input variables that have only little discrimination power. " <<
Endl;
1722 Log() <<
"" <<
Endl;
1723 Log() <<
"In the tests we have carried out so far, the MLP and ROOT networks" <<
Endl;
1724 Log() <<
"(TMlpANN, interfaced via TMVA) performed equally well, with however" <<
Endl;
1725 Log() <<
"a clear speed advantage for the MLP. The Clermont-Ferrand neural " <<
Endl;
1726 Log() <<
"net (CFMlpANN) exhibited worse classification performance in these" <<
Endl;
1727 Log() <<
"tests, which is partly due to the slow convergence of its training" <<
Endl;
1728 Log() <<
"(at least 10k training cycles are required to achieve approximately" <<
Endl;
1729 Log() <<
"competitive results)." <<
Endl;
1731 Log() << col <<
"Overtraining: " << colres
1732 <<
"only the TMlpANN performs an explicit separation of the" <<
Endl;
1733 Log() <<
"full training sample into independent training and validation samples." <<
Endl;
1734 Log() <<
"We have found that in most high-energy physics applications the " <<
Endl;
1735 Log() <<
"available degrees of freedom (training events) are sufficient to " <<
Endl;
1736 Log() <<
"constrain the weights of the relatively simple architectures required" <<
Endl;
1737 Log() <<
"to achieve good performance. Hence no overtraining should occur, and " <<
Endl;
1738 Log() <<
"the use of validation samples would only reduce the available training" <<
Endl;
1739 Log() <<
"information. However, if the performance on the training sample is " <<
Endl;
1740 Log() <<
"found to be significantly better than the one found with the inde-" <<
Endl;
1741 Log() <<
"pendent test sample, caution is needed. The results for these samples " <<
Endl;
1742 Log() <<
"are printed to standard output at the end of each training job." <<
Endl;
1744 Log() << col <<
"--- Performance tuning via configuration options:" << colres <<
Endl;
1746 Log() <<
"The hidden layer architecture for all ANNs is defined by the option" <<
Endl;
1747 Log() <<
"\"HiddenLayers=N+1,N,...\", where here the first hidden layer has N+1" <<
Endl;
1748 Log() <<
"neurons and the second N neurons (and so on), and where N is the number " <<
Endl;
1749 Log() <<
"of input variables. Excessive numbers of hidden layers should be avoided," <<
Endl;
1750 Log() <<
"in favour of more neurons in the first hidden layer." <<
Endl;
1751 Log() <<
"" <<
Endl;
1752 Log() <<
"The number of cycles should be above 500. As said, if the number of" <<
Endl;
1753 Log() <<
"adjustable weights is small compared to the training sample size," <<
Endl;
1754 Log() <<
"using a large number of training samples should not lead to overtraining." <<
Endl;
#define REGISTER_METHOD(CLASS)
for example
bool Bool_t
Boolean (0=false, 1=true) (bool)
int Int_t
Signed integer 4 bytes (int)
float Float_t
Float 4 bytes (float)
double Double_t
Double 8 bytes.
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t Float_t Float_t Float_t Int_t Int_t UInt_t UInt_t Rectangle_t Int_t Int_t Window_t TString Int_t GCValues_t GetPrimarySelectionOwner GetDisplay GetScreen GetColormap GetNativeEvent const char const char dpyName wid window const char font_name cursor keysym reg const char only_if_exist regb h Point_t winding char text const char depth char const char Int_t count const char ColorStruct_t color const char Pixmap_t Pixmap_t PictureAttributes_t attr const char char ret_data h unsigned char height h Atom_t Int_t ULong_t ULong_t unsigned char prop_list Atom_t Atom_t target
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t index
Option_t Option_t TPoint TPoint const char GetTextMagnitude GetFillStyle GetLineColor GetLineWidth GetMarkerStyle GetTextAlign GetTextColor GetTextSize void char Point_t Rectangle_t WindowAttributes_t Float_t Float_t Float_t Int_t Int_t UInt_t UInt_t Rectangle_t Int_t Int_t Window_t TString Int_t GCValues_t GetPrimarySelectionOwner GetDisplay GetScreen GetColormap GetNativeEvent const char const char dpyName wid window const char font_name cursor keysym reg const char only_if_exist regb h Point_t winding char text const char depth char const char Int_t count const char ColorStruct_t color const char Pixmap_t Pixmap_t PictureAttributes_t attr const char char ret_data h unsigned char height h Atom_t Int_t ULong_t ULong_t unsigned char prop_list Atom_t Atom_t Atom_t Time_t type
TMatrixT< Double_t > TMatrixD
<div class="legacybox"><h2>Legacy Code</h2> TFitter is a legacy interface: there will be no bug fixes...
1-D histogram with a float per channel (see TH1 documentation)
TH1 is the base class of all histogram classes in ROOT.
Bool_t WriteOptionsReference() const
Class that contains all the data information.
Base class for TMVA fitters.
Fitter using a Genetic Algorithm.
The TMVA::Interval Class.
Base class for all TMVA methods using artificial neural networks.
void ProcessOptions() override
do nothing specific at this moment
void MakeClassSpecific(std::ostream &, const TString &) const override
write specific classifier response
Double_t GetMvaValue(Double_t *err=nullptr, Double_t *errUpper=nullptr) override
get the mva value generated by the NN
Multilayer Perceptron class built off of MethodANNBase.
void GetHelpMessage() const override
get help message text
void BackPropagationMinimize(Int_t nEpochs)
minimize estimator / train network with back propagation algorithm
Double_t GetMSEErr(const Event *ev, UInt_t index=0)
zjh
void MakeClassSpecific(std::ostream &, const TString &) const override
write specific classifier response
void DeclareOptions() override
define the options (their key words) that can be set in the option string
void AdjustSynapseWeights()
just adjust the synapse weights (should be called in batch mode)
void SteepestDir(TMatrixD &Dir)
void TrainOneEpoch()
train network over a single epoch/cycle of events
Bool_t GetHessian(TMatrixD &Hessian, TMatrixD &Gamma, TMatrixD &Delta)
Double_t ComputeEstimator(std::vector< Double_t > ¶meters)
this function is called by GeneticANN for GA optimization
void InitializeLearningRates()
initialize learning rates of synapses, used only by back propagation
void CalculateNeuronDeltas()
have each neuron calculate its delta by back propagation
Double_t EstimatorFunction(std::vector< Double_t > ¶meters) override
interface to the estimate
Double_t DerivDir(TMatrixD &Dir)
Double_t GetCEErr(const Event *ev, UInt_t index=0)
zjh
virtual ~MethodMLP()
destructor nothing to be done
void SetDir(TMatrixD &Hessian, TMatrixD &Dir)
void Shuffle(Int_t *index, Int_t n)
Input:
void SimulateEvent(const Event *ev)
void SetDirWeights(std::vector< Double_t > &Origin, TMatrixD &Dir, Double_t alpha)
void SetGammaDelta(TMatrixD &Gamma, TMatrixD &Delta, std::vector< Double_t > &Buffer)
void GetApproxInvHessian(TMatrixD &InvHessian, bool regulate=true)
rank-1 approximation, neglect 2nd derivatives. //zjh
void BFGSMinimize(Int_t nEpochs)
train network with BFGS algorithm
void UpdateSynapses()
update synapse error fields and adjust the weights (if in sequential mode)
void TrainOneEvent(Int_t ievt)
train network over a single event this uses the new event model
Double_t GetDesiredOutput(const Event *ev)
get the desired output of this event
void GeneticMinimize()
create genetics class similar to GeneticCut give it vector of parameter ranges (parameters = weights)...
void Init() override
default initializations
Bool_t HasAnalysisType(Types::EAnalysisType type, UInt_t numberClasses, UInt_t numberTargets) override
MLP can handle classification with 2 classes and regression with one regression-target.
void DecaySynapseWeights(Bool_t lateEpoch)
decay synapse weights in last 10 epochs, lower learning rate even more to find a good minimum
void TrainOneEventFast(Int_t ievt, Float_t *&branchVar, Int_t &type)
fast per-event training
void UpdateNetwork(Double_t desired, Double_t eventWeight=1.0)
update the network based on how closely the output matched the desired output
MethodMLP(const TString &jobName, const TString &methodTitle, DataSetInfo &theData, const TString &theOption)
standard constructor
void UpdateRegulators()
zjh
Bool_t LineSearch(TMatrixD &Dir, std::vector< Double_t > &Buffer, Double_t *dError=nullptr)
zjh
Double_t GetMvaValue(Double_t *err=nullptr, Double_t *errUpper=nullptr) override
get the mva value generated by the NN
Double_t CalculateEstimator(Types::ETreeType treeType=Types::kTraining, Int_t iEpoch=-1)
calculate the estimator that training is attempting to minimize
void ProcessOptions() override
process user options
Neuron class used by TMVA artificial neural network methods.
void AdjustSynapseWeights()
adjust the pre-synapses' weights for each neuron (input neuron has no pre-synapse) this method should...
void ForceValue(Double_t value)
force the value, typically for input and bias neurons
void UpdateSynapsesSequential()
update the pre-synapses for each neuron (input neuron has no pre-synapse) this method should only be ...
void UpdateSynapsesBatch()
update and adjust the pre-synapses for each neuron (input neuron has no pre-synapse) this method shou...
void CalculateDelta()
calculate error field
Synapse class used by TMVA artificial neural network methods.
void SetWeight(Double_t weight)
set synapse weight
void SetDEDw(Double_t DEDw)
Timing information for training and evaluation of MVA methods.
Singleton class for Global types used by TMVA.
virtual TMatrixTBase< Element > & UnitMatrix()
Make a unit matrix (matrix need not be a square one).
TMatrixTBase< Element > & ResizeTo(Int_t nrows, Int_t ncols, Int_t=-1) override
Set size of the matrix to nrows x ncols New dynamic elements are created, the overlapping part of the...
TMatrixT< Element > & Invert(Double_t *det=nullptr)
Invert the matrix and calculate its determinant.
TObject * At(Int_t idx) const override
static TString Format(const char *fmt,...)
Static method which formats a string using a printf style format descriptor and return a TString.
This is a simple weighted bidirectional connection between two neurons.
void SetWeight(Double_t w)
Sets the weight of the synapse.
MsgLogger & Endl(MsgLogger &ml)
Double_t Exp(Double_t x)
Returns the base-e exponential function of x, which is e raised to the power x.
Double_t Log(Double_t x)
Returns the natural logarithm of x.
Double_t Sqrt(Double_t x)
Returns the square root of x.