From 30c4d38fcd2487e5f85c0ef96b06aaf6ec51c206 Mon Sep 17 00:00:00 2001 From: Jon Smit Date: Wed, 6 Jul 2016 01:12:39 +0200 Subject: [PATCH] Added another version of the Rsquared test. --- src/Numerics/GoodnessOfFit.cs | 66 +++++++++++++++++++++++++++++++++-- 1 file changed, 64 insertions(+), 2 deletions(-) diff --git a/src/Numerics/GoodnessOfFit.cs b/src/Numerics/GoodnessOfFit.cs index 5972384f..29ad6cca 100644 --- a/src/Numerics/GoodnessOfFit.cs +++ b/src/Numerics/GoodnessOfFit.cs @@ -3,6 +3,8 @@ // http://numerics.mathdotnet.com // http://github.com/mathnet/mathnet-numerics // +// Copyright (c) 2009-2018 Math.NET +// // Permission is hereby granted, free of charge, to any person // obtaining a copy of this software and associated documentation // files (the "Software"), to deal in the Software without @@ -61,7 +63,7 @@ namespace MathNet.Numerics /// /// Calculates the Standard Error of the regression, given a sequence of - /// modeled/predicted values, and a sequence of actual/observed values + /// modeled/predicted values, and a sequence of actual/observed values /// /// The modelled/predicted values /// The observed/actual values @@ -77,7 +79,7 @@ namespace MathNet.Numerics /// /// The modelled/predicted values /// The observed/actual values - /// The degrees of freedom by which the + /// The degrees of freedom by which the /// number of samples is reduced for performing the Standard Error calculation /// The Standard Error of the regression public static double StandardError(IEnumerable modelledValues, IEnumerable observedValues, int degreesOfFreedom) @@ -107,5 +109,65 @@ namespace MathNet.Numerics return Math.Sqrt(accumulator / (n - degreesOfFreedom)); } } + + /// + /// Calculates the R-Squared value, also known as coefficient of determination, + /// given some modelled and observed values. + /// + /// The values expected from the model. + /// The actual values obtained. + /// Coefficient of determination. + public static double CoefficientOfDetermination(IEnumerable modelledValues, IEnumerable observedValues) + { + var y = observedValues; + var f = modelledValues; + int n = 0; + + double meanY = 0; + double ssTot = 0; + double ssRes = 0; + + using (IEnumerator ieY = y.GetEnumerator()) + using (IEnumerator ieF = f.GetEnumerator()) + { + while (ieY.MoveNext()) + { + if (!ieF.MoveNext()) + { + throw new ArgumentOutOfRangeException("modelledValues", Resources.ArgumentArraysSameLength); + } + + double currentY = ieY.Current; + double currentF = ieF.Current; + + // If a large constant C is added to every y value, + // then each new y have an error of about C*eps, + // thus each new deltaY will change by about C*eps (compared to the old deltaY), + // and thus ssTot will change by only C*eps*deltaY on each step + // thus C*eps*deltaY*n in total. + // (This error cannot be eliminated by a Kahan algorithm, + // because it is introduced when C is added to the old Y value). + // + // This is better than summing the square of y values + // and then substracting the correct multiple of the square of the sum of y values, + // in this latter case ssTot will change by eps*n*(C^2+2*C*meanY) in total. + double deltaY = currentY - meanY; + double scaleDeltaY = deltaY / ++n; + + meanY += scaleDeltaY; + ssTot += scaleDeltaY* deltaY* (n - 1); + + // This calculation is as safe as ssTot + // in the case when a constant is added to both y and f. + ssRes += (currentY - currentF)* (currentY-currentF); + } + + if (ieF.MoveNext()) + { + throw new ArgumentOutOfRangeException("observedValues", Resources.ArgumentArraysSameLength); + } + } + return 1 - ssRes/ssTot; + } } }