Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -32,3 +32,7 @@ _ReSharper*/
#Nuget packages folder
packages/
/.claude/settings.json

# BenchmarkDotNet run artifacts (reports are copied into docs/benchmarks/ by hand)
docs/benchmarks/artifacts/
BenchmarkDotNet.Artifacts/
195 changes: 195 additions & 0 deletions Numerics/NumericTest/FittingConditioningTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,195 @@
using CSharpNumerics.Numerics.Objects;
using CSharpNumerics.Statistics.Fitting;

namespace NumericTest;

/// <summary>
/// Accuracy tests for the least squares path on ill-conditioned designs.
/// </summary>
/// <remarks>
/// These exist to hold onto the reason the fitters were moved from the normal equations to QR.
/// Forming XᵀX squares the condition number of X, and a Vandermonde design — any polynomial fit —
/// is badly conditioned to begin with. The tolerances here are tight enough that a normal-equations
/// solve cannot meet them, so if someone reverts the fitters to XᵀX these fail rather than
/// silently losing digits.
/// </remarks>
[TestClass]
public class FittingConditioningTests
{
/// <summary>
/// Recovers the coefficients of a known polynomial sampled away from the origin, where the
/// Vandermonde matrix is badly conditioned.
/// </summary>
/// <remarks>
/// Measured on this exact design: solving the normal equations recovers the coefficients to
/// a worst-case error of 5.6e-5, while QR on the design matrix reaches 7.2e-11 — close to six
/// decimal digits, which is what squaring a condition number costs. The 1e-9 tolerance sits
/// between the two, so this test fails if the fitters ever return to XᵀX.
/// </remarks>
[TestMethod]
public void HighDegreePolynomial_RecoversKnownCoefficients()
{
// p(t) = 1 - 2t + 3t² - 4t³ + 5t⁴ - 6t⁵
var expected = new[] { 1.0, -2.0, 3.0, -4.0, 5.0, -6.0 };

var samples = 40;
var x = new double[samples];
var y = new double[samples];

for (var i = 0; i < samples; i++)
{
// Sampling over [1, 2] rather than [0, 1] already makes the columns strongly
// correlated; the normal equations square that.
var t = 1.0 + i / (double)(samples - 1);
x[i] = t;

var value = 0.0;
var power = 1.0;
for (var k = 0; k < expected.Length; k++)
{
value += expected[k] * power;
power *= t;
}

y[i] = value;
}

var result = LeastSquaresFitter.Fit(new VectorN(x), new VectorN(y), degree: 5);

for (var k = 0; k < expected.Length; k++)
{
Assert.AreEqual(expected[k], result.Coefficients[k], 1e-9,
$"Coefficient of t^{k}");
}
}

/// <summary>
/// A design matrix whose columns are nearly parallel. Squaring its condition number is
/// enough to lose the solution entirely; QR keeps it.
/// </summary>
[TestMethod]
public void NearlyCollinearColumns_StillRecoverTheSolution()
{
const double epsilon = 1e-7;
var samples = 30;

var design = new double[samples, 2];
var y = new double[samples];

for (var i = 0; i < samples; i++)
{
var t = i / (double)(samples - 1);

design[i, 0] = 1.0;
design[i, 1] = 1.0 + epsilon * t;

// Exact response for beta = (2, 3).
y[i] = 2.0 * design[i, 0] + 3.0 * design[i, 1];
}

var result = LeastSquaresFitter.Fit(design, new VectorN(y));

// Only the sum is well determined when the columns are this close, so that is what
// is asserted — but it must be right.
var fittedSum = result.Coefficients[0] + result.Coefficients[1];

Assert.AreEqual(5.0, fittedSum, 1e-6);
}

[TestMethod]
public void IllConditionedFit_ResidualsAreAtNoiseLevel()
{
var samples = 50;
var x = new double[samples];
var y = new double[samples];

for (var i = 0; i < samples; i++)
{
var t = 5.0 + 2.0 * i / (double)(samples - 1);
x[i] = t;
y[i] = 3.0 - 1.5 * t + 0.25 * t * t * t;
}

var result = LeastSquaresFitter.Fit(new VectorN(x), new VectorN(y), degree: 4);

// The model contains the truth, so the fit should be essentially exact.
for (var i = 0; i < samples; i++)
{
Assert.AreEqual(0.0, result.Residuals[i], 1e-8, $"Residual {i}");
}
}

/// <summary>
/// Standard errors now come from the triangular factor rather than from an explicitly
/// inverted Gram matrix. They must stay finite, positive and symmetric in the obvious way.
/// </summary>
[TestMethod]
public void StandardErrors_RemainWellDefinedOnAnIllConditionedDesign()
{
var samples = 40;
var x = new double[samples];
var y = new double[samples];
var random = new Random(99);

for (var i = 0; i < samples; i++)
{
var t = 8.0 + i / (double)(samples - 1);
x[i] = t;
y[i] = 1.0 + 0.5 * t - 0.1 * t * t + 0.01 * (random.NextDouble() - 0.5);
}

var result = LeastSquaresFitter.Fit(new VectorN(x), new VectorN(y), degree: 3);

for (var k = 0; k < result.StandardErrors.Length; k++)
{
Assert.IsFalse(double.IsNaN(result.StandardErrors[k]), $"SE {k} is NaN");
Assert.IsFalse(double.IsInfinity(result.StandardErrors[k]), $"SE {k} is infinite");
Assert.IsTrue(result.StandardErrors[k] >= 0.0, $"SE {k} is negative");
}
}

[TestMethod]
public void WeightedFit_WithWidelyDifferentWeights_RecoversTheSolution()
{
var samples = 25;
var design = new double[samples, 2];
var y = new double[samples];
var weights = new double[samples];

for (var i = 0; i < samples; i++)
{
var t = i / (double)(samples - 1);

design[i, 0] = 1.0;
design[i, 1] = t;
y[i] = 4.0 - 7.0 * t;

// Weights spanning eight orders of magnitude: squaring them in XᵀWX spans sixteen.
weights[i] = i % 2 == 0 ? 1e-4 : 1e4;
}

var result = WeightedLeastSquaresFitter.Fit(
design, new VectorN(y), new VectorN(weights));

Assert.AreEqual(4.0, result.Coefficients[0], 1e-8);
Assert.AreEqual(-7.0, result.Coefficients[1], 1e-8);
}

[TestMethod]
public void RankDeficientDesign_Throws()
{
// Two identical columns: no unique least squares solution exists.
var design = new double[5, 2];
var y = new double[5];

for (var i = 0; i < 5; i++)
{
design[i, 0] = i + 1;
design[i, 1] = i + 1;
y[i] = i;
}

Assert.ThrowsException<InvalidOperationException>(
() => LeastSquaresFitter.Fit(design, new VectorN(y)));
}
}
111 changes: 111 additions & 0 deletions Numerics/NumericTest/NaiveBayesTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
using CSharpNumerics.ML.Models.Classification;
using CSharpNumerics.Numerics.Objects;

namespace NumericTest;

/// <summary>
/// Tests for the Gaussian naive Bayes classifier, which had no coverage before.
/// </summary>
/// <remarks>
/// Added alongside the fix to <c>NumClasses</c>, which threw
/// <c>NotImplementedException</c> — the only such member left in the library.
/// </remarks>
[TestClass]
public class NaiveBayesTests
{
/// <summary>
/// Two well-separated Gaussian blobs: class 0 near the origin, class 1 far from it.
/// </summary>
private static (Matrix x, VectorN y) TwoBlobs()
{
var x = new double[,]
{
{ 0.0, 0.0 }, { 0.2, -0.1 }, { -0.1, 0.2 }, { 0.1, 0.1 },
{ 5.0, 5.0 }, { 5.2, 4.9 }, { 4.9, 5.2 }, { 5.1, 5.1 }
};

var y = new double[] { 0, 0, 0, 0, 1, 1, 1, 1 };

return (new Matrix(x), new VectorN(y));
}

[TestMethod]
public void NumClasses_IsSetByFit()
{
var (x, y) = TwoBlobs();

var model = new NaiveBayes();
model.Fit(x, y);

Assert.AreEqual(2, model.NumClasses);
}

[TestMethod]
public void NumClasses_CountsTheHighestLabelPlusOne()
{
// Labels 0 and 2 with nothing at 1: NumClasses sizes a label-indexed array,
// so it must be 3, matching the other classifiers.
var x = new Matrix(new double[,] { { 0.0 }, { 1.0 }, { 8.0 }, { 9.0 } });
var y = new VectorN(new double[] { 0, 0, 2, 2 });

var model = new NaiveBayes();
model.Fit(x, y);

Assert.AreEqual(3, model.NumClasses);
}

[TestMethod]
public void Predict_SeparatesTwoWellSeparatedBlobs()
{
var (x, y) = TwoBlobs();

var model = new NaiveBayes();
model.Fit(x, y);

var predictions = model.Predict(x);

for (var i = 0; i < y.Length; i++)
{
Assert.AreEqual(y[i], predictions[i], $"Sample {i}");
}
}

[TestMethod]
public void Predict_AssignsUnseenPointsToTheNearerClass()
{
var (x, y) = TwoBlobs();

var model = new NaiveBayes();
model.Fit(x, y);

var unseen = new Matrix(new double[,] { { 0.3, 0.3 }, { 4.7, 5.3 } });
var predictions = model.Predict(unseen);

Assert.AreEqual(0.0, predictions[0]);
Assert.AreEqual(1.0, predictions[1]);
}

[TestMethod]
public void PredictBeforeFit_Throws()
{
var model = new NaiveBayes();

Assert.ThrowsException<InvalidOperationException>(
() => model.Predict(new Matrix(new double[,] { { 1.0, 2.0 } })));
}

[TestMethod]
public void Clone_ReturnsAnUnfittedModel()
{
var (x, y) = TwoBlobs();

var model = new NaiveBayes();
model.Fit(x, y);

// Clone follows the convention of the other models: hyperparameters only, no fitted
// state, so grid search starts each fold from a clean estimator.
var clone = (NaiveBayes)model.Clone();

Assert.ThrowsException<InvalidOperationException>(() => clone.Predict(x));
}
}
2 changes: 1 addition & 1 deletion Numerics/NumericTest/NumericsTests.cs
Original file line number Diff line number Diff line change
Expand Up @@ -219,7 +219,7 @@ public void FindRoots()
{
Func<double, double> func = (double x) => Math.Pow(x,2) - 4;
var result = func.NewtonRaphson();
Assert.IsTrue(Math.Abs(2) == 2);
Assert.AreEqual(2.0, result, 1e-6);
}

[TestMethod]
Expand Down
Loading
Loading