From 33451e8dfa94456b90b6e942778f8e8ede064974 Mon Sep 17 00:00:00 2001 From: Skorinn <42702903+Skorinn@users.noreply.github.com> Date: Wed, 9 Sep 2026 22:04:02 -0500 Subject: [PATCH 1/4] Say what shift the sessions could have shown, and warn when it is too coarse A verdict of nothing found means nothing at all unless a shift worth finding could have been seen. Two sessions of a few seconds will report no significant shift whatever the generator did, and that reading was being left to stand on its own as though it said something about the generator rather than about the length of the recording. SmallestDetectableDifference is the half width of the confidence interval for the pair, which is the test read the other way round: rather than asking whether the shift that happened beats the noise, it asks how large a shift would have to be before it could. A test pins the two together by shifting readings by exactly that much and checking the probability lands on the significance level. The verdict states it either way. Below SHIFT_OF_INTEREST, one part in ten thousand and the order of the effect reported in the published work, it says so in the warning colour and asks for a longer recording. The three outcomes are three weights, which is what VerdictWeights carries. A shift found is notable however short the sessions, because it cleared the limit by being found at all. Nothing found from sessions that could have found something is the ordinary outcome of the experiment. Nothing found from sessions that could not is the one that misleads, and is the only one that warns. Worked out from each session's own spread rather than from a count of readings, because a device reading averages 262,144 bits and a simulated one 16,384: the simulated spread is four times wider and needs sixteen times the readings to pin its mean down as finely. A fixed count would have meant different things for the two. The verdict band had to grow with the wording. It was two lines and the warning needs three, so it was clipped mid-sentence when first run - which only showed up in the built application, as the tests read the text rather than look at it. Sizing the label to its content was worse: an AutoSize label grows sideways rather than wrapping, so it left the card entirely. It is a taller fixed band, checked against the narrowest the window is allowed to be, where the wording needs 45px of the 78 it now has. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_0153VVkWg7DQmdaNLcvtY37w --- CLAUDE.md | 20 +- GeneratorForm.Designer.cs | 4 +- GeneratorForm.cs | 84 +++++- README.md | 13 + SignificanceTest.cs | 112 +++++++ .../SignificanceTest.Test.cs | 283 ++++++++++++++++++ tests/manual/README.md | 1 + tests/manual/check-sensitivity.ps1 | 124 ++++++++ tests/manual/run-all.ps1 | 1 + 9 files changed, 627 insertions(+), 15 deletions(-) create mode 100644 tests/manual/check-sensitivity.ps1 diff --git a/CLAUDE.md b/CLAUDE.md index 265c721..f0c3638 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -190,9 +190,27 @@ spreads, because a baseline is usually recorded for far longer than the run comp `CompareWithExpected` tests one session against the 0.5 an unbiased generator gives. Both are two-tailed: a one-tailed test would find a shift toward a target more easily, but only holds when the direction was predicted before the readings were taken, which the application cannot know. Every path that cannot produce -a number — fewer than two readings, or readings with no spread — returns `Valid = false` rather than a +a number - fewer than two readings, or readings with no spread - returns `Valid = false` rather than a probability, and `Significant` is false whenever the test did not run. +A verdict of nothing found means nothing at all unless a shift worth finding could have been seen, so the +analysis states what it could have seen. `SmallestDetectableDifference` is the half width of the confidence +interval for the pair, which is the test read the other way round: instead of asking whether the shift that +happened beats the noise, it asks how large a shift would have to be before it could. Below +`SHIFT_OF_INTEREST` - one part in ten thousand, the order of the effect reported in the published work - the +sessions cannot speak to the question, and the verdict says so in the warning colour rather than reporting a +null result that reads as evidence of absence. + +The three verdicts are different weights, which is what `VerdictWeights` carries. A shift found is notable +whatever the sensitivity, because it cleared the limit by being found at all. Nothing found from sessions +that could have found something is the ordinary outcome. Nothing found from sessions that could not is the +one that misleads, and is the only one that warns. + +Note that the device and the simulator have different noise floors: a device reading averages 262,144 bits +and a simulated one 16,384, so the simulated spread is about four times wider and needs about sixteen times +the readings to pin its mean down as finely. That is why the limit is worked out from each session's own +spread rather than from a count of readings. + The verdict is stated in words under the table rather than left to be read off the numbers, and is emphasised only when there is a shift to notice. `HistogramChart` plots each session as a percentage of its own readings, not as counts: on counts the longer session stands taller in every bin and hides the shift the diff --git a/GeneratorForm.Designer.cs b/GeneratorForm.Designer.cs index 0e1834b..acd104a 100644 --- a/GeneratorForm.Designer.cs +++ b/GeneratorForm.Designer.cs @@ -696,7 +696,7 @@ private void InitializeComponent() this.m_AnalyzeLayout.Name = "m_AnalyzeLayout"; this.m_AnalyzeLayout.RowCount = 3; this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 126F)); - this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 234F)); + this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 290F)); this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Percent, 100F)); this.m_AnalyzeLayout.Size = new System.Drawing.Size(741, 652); this.m_AnalyzeLayout.TabIndex = 0; @@ -867,7 +867,7 @@ private void InitializeComponent() this.m_VerdictLabel.Location = new System.Drawing.Point(14, 188); this.m_VerdictLabel.Name = "m_VerdictLabel"; this.m_VerdictLabel.Padding = new System.Windows.Forms.Padding(0, 10, 0, 0); - this.m_VerdictLabel.Size = new System.Drawing.Size(713, 42); + this.m_VerdictLabel.Size = new System.Drawing.Size(713, 88); this.m_VerdictLabel.TabIndex = 1; this.m_VerdictLabel.TextAlign = System.Drawing.ContentAlignment.MiddleLeft; // diff --git a/GeneratorForm.cs b/GeneratorForm.cs index 319ec84..9450d8a 100644 --- a/GeneratorForm.cs +++ b/GeneratorForm.cs @@ -81,6 +81,18 @@ public enum RngGuiStates RNG_GUI_STATES_SIZE // Keep at end }; + /// + /// How much attention a verdict is worth. Nothing found is the ordinary outcome of the experiment + /// and is not worth shouting about; a shift is worth noticing; and a session that could not have + /// shown the shift it was looking for is worth warning about, because its silence means nothing. + /// + public enum VerdictWeights + { + Ordinary = 0, + Notable = 1, + Warning = 2, + }; + #endregion #region Constructors @@ -2450,32 +2462,60 @@ private void UpdateComparisonVerdict(List baselineReadings, List bool bBothLoaded = ((null != baselineReadings) && (null != resultReadings)); if (false == bBothLoaded) { - SetVerdict(m_sVERDICT_NEEDS_BOTH, false); + SetVerdict(m_sVERDICT_NEEDS_BOTH, VerdictWeights.Ordinary); return; } SignificanceResult test = SignificanceTest.CompareMeans(baselineReadings, resultReadings); if (false == test.Valid) { - SetVerdict(m_sVERDICT_NOT_ENOUGH_DATA, false); + SetVerdict(m_sVERDICT_NOT_ENOUGH_DATA, VerdictWeights.Warning); return; } double fShift = (Mean(resultReadings) - Mean(baselineReadings)); string sDirection = (0 <= fShift) ? m_sDIRECTION_HIGHER : m_sDIRECTION_LOWER; string sProbability = FormatProbability(test); + string sFreedom = test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT); + string sShift = Math.Abs(fShift).ToString(m_sVALUE_FORMAT); + + // How finely the two sessions between them pin a difference down. A verdict of nothing found + // means nothing at all unless a shift worth finding could have been seen, so the limit is + // stated either way rather than left for the reader to work out from the reading counts. + double fDetectable = SignificanceTest.SmallestDetectableDifference(baselineReadings, resultReadings); + bool bSensitive = SignificanceTest.IsSensitiveEnough(fDetectable); + string sDetectable = fDetectable.ToString(m_sVALUE_FORMAT); + string sOfInterest = SignificanceTest.SHIFT_OF_INTEREST.ToString(m_sVALUE_FORMAT); if (true == test.Significant) { - SetVerdict($"Significant shift: the result sits {sDirection} than the baseline by " + - $"{Math.Abs(fShift).ToString(m_sVALUE_FORMAT)}. A shift this large would arise by " + - $"chance {sProbability} of the time (Welch's t, two-tailed, {test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT)} df).", true); + // The shift cleared the limit by having been found at all, so this reports the limit rather + // than warning about it, however short the sessions were + SetVerdict($"Significant shift: the result sits {sDirection} than the baseline by {sShift}. " + + $"A shift this large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions can show a difference of " + + $"{sDetectable} or larger.", VerdictWeights.Notable); + } + else if (true == bSensitive) + { + SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by {sShift}, " + + $"but a shift that large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions can show a difference of " + + $"{sDetectable} or larger, so a shift of {sOfInterest} would have been found.", + VerdictWeights.Ordinary); } else { - SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by " + - $"{Math.Abs(fShift).ToString(m_sVALUE_FORMAT)}, but a shift that large would arise by " + - $"chance {sProbability} of the time (Welch's t, two-tailed, {test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT)} df).", false); + // Nothing was found and nothing could have been. This is the reading that misleads if it is + // left to stand on its own, so it is the one that carries the warning. The measurement is + // still reported in full first: the warning is about what the numbers can be taken to mean, + // not a reason to stop showing them. + SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by {sShift}, " + + $"but a shift that large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions are too short to conclude " + + $"anything from that: they can only show a difference of {sDetectable} or larger, so " + + $"a shift of {sOfInterest}, the size this looks for, could be real here and still " + + $"never reach significance. Record for longer.", VerdictWeights.Warning); } } @@ -2484,12 +2524,32 @@ private void UpdateComparisonVerdict(List baselineReadings, List /// ordinary outcome and is not worth shouting about. /// /// IN - The wording to show - /// IN - Whether the verdict reports a shift worth noticing - private void SetVerdict(string sVerdict, bool bSignificant) + /// IN - How much attention the verdict is worth + private void SetVerdict(string sVerdict, VerdictWeights weight) { m_VerdictLabel.Text = sVerdict; - m_VerdictLabel.Font = bSignificant ? m_VerdictFontBold : m_VerdictFontRegular; - m_VerdictLabel.ForeColor = bSignificant ? UiPalette.CardText : UiPalette.MutedText; + + // Anything other than the ordinary outcome is set in the heavier face, so a verdict that needs + // reading is not the same weight as the one that says nothing happened + bool bOrdinary = (VerdictWeights.Ordinary == weight); + m_VerdictLabel.Font = bOrdinary ? m_VerdictFontRegular : m_VerdictFontBold; + + // A warning takes the severity colour, which stands aside under a high contrast scheme and + // leaves the wording to carry it + switch (weight) + { + case VerdictWeights.Warning: + m_VerdictLabel.ForeColor = StatusPalette.WarningText; + break; + + case VerdictWeights.Notable: + m_VerdictLabel.ForeColor = UiPalette.CardText; + break; + + default: + m_VerdictLabel.ForeColor = UiPalette.MutedText; + break; + } } /// diff --git a/README.md b/README.md index c86af0b..158f165 100644 --- a/README.md +++ b/README.md @@ -35,6 +35,19 @@ directly: mean, standard deviation, skewness and kurtosis for each, the differen a histogram of the two distributions overlaid. Statistics are calculated with [Math.NET Numerics](https://numerics.mathdotnet.com/). +### Enough readings to answer + +A session that found nothing has only said something if it could have found something. The verdict therefore +states the smallest difference the two sessions could show as significant, and warns when that is coarser +than the shift being looked for — one part in ten thousand, a generator running at 0.5001 rather than 0.5, +which is the order of the effect reported in the published work. + +A minute or so of recording from the device is enough to speak to a shift that size. Below that, a real +shift can sit in the readings and never reach significance, and a verdict of *no significant shift* would +be reporting the length of the session rather than anything about the generator. The limit is worked out +from each session's own spread rather than from a count of readings, because a simulated reading averages +far fewer bits than a device one and is correspondingly noisier. + ### Targets A session can record a target value of `0` or `1` — the outcome the operator is attempting to influence — diff --git a/SignificanceTest.cs b/SignificanceTest.cs index 671d8d0..18365ce 100644 --- a/SignificanceTest.cs +++ b/SignificanceTest.cs @@ -156,6 +156,107 @@ public static SignificanceResult CompareWithExpected(IList readings, dou return BuildResult(fStatistic, (readings.Count - 1)); } + /// + /// The smallest shift away from the expected value that a set of readings could show as significant. + /// A session answers the question it was recorded to answer only if the shift being looked for is + /// larger than this: below it, a shift could be perfectly real and still not reach significance, + /// because the readings are not pinned down finely enough to tell it from noise. + /// NOTE: This is the half width of the confidence interval, the same arithmetic the test itself + /// uses, read the other way round. The test asks whether the shift that happened is larger than the + /// noise; this asks how large a shift would have to be before it could. + /// + /// IN - The readings to measure (cannot be null) + /// The smallest shift that could reach significance, or NaN when there is too little to say + /// Thrown when the readings are null + public static double SmallestDetectableShift(IList readings) + { + if (null == readings) + { + throw new ArgumentNullException(nameof(readings), "Readings cannot be null"); + } + + // A spread cannot be measured from fewer than two readings, so nothing can be said about how + // finely they pin the mean down + if (m_iMINIMUM_READINGS > readings.Count) + { + return double.NaN; + } + + double fMean = Mean(readings); + double fStandardError = (Variance(readings, fMean) / readings.Count); + if (m_fMINIMUM_ERROR >= fStandardError) + { + return double.NaN; + } + + // The critical value comes from the same distribution the test uses, so the answer agrees with + // the test rather than approximating it. It is well above two for a handful of readings, which + // is exactly the case this exists to report on. + double fFreedom = (readings.Count - 1); + double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); + + return (fCritical * Math.Sqrt(fStandardError)); + } + + /// + /// The smallest difference between two sets of readings that could be shown as significant. This is + /// the pair's answer to the same question, and it is the one the verdict reports, because the + /// verdict is about the comparison rather than about either session on its own. + /// + /// IN - The baseline readings (cannot be null) + /// IN - The result readings (cannot be null) + /// The smallest difference that could reach significance, or NaN when there is too little to say + /// Thrown when either set of readings is null + public static double SmallestDetectableDifference(IList baseline, IList result) + { + if (null == baseline) + { + throw new ArgumentNullException(nameof(baseline), "Baseline readings cannot be null"); + } + + if (null == result) + { + throw new ArgumentNullException(nameof(result), "Result readings cannot be null"); + } + + bool bEnoughData = ((m_iMINIMUM_READINGS <= baseline.Count) && (m_iMINIMUM_READINGS <= result.Count)); + if (false == bEnoughData) + { + return double.NaN; + } + + double fBaselineError = (Variance(baseline, Mean(baseline)) / baseline.Count); + double fResultError = (Variance(result, Mean(result)) / result.Count); + double fCombinedError = (fBaselineError + fResultError); + if (m_fMINIMUM_ERROR >= fCombinedError) + { + return double.NaN; + } + + // Welch's degrees of freedom, as in the test the two are compared with + double fDenominator = (((fBaselineError * fBaselineError) / (baseline.Count - 1)) + + ((fResultError * fResultError) / (result.Count - 1))); + double fFreedom = ((fCombinedError * fCombinedError) / fDenominator); + if ((double.IsNaN(fFreedom)) || (0.0 >= fFreedom)) + { + return double.NaN; + } + + double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); + + return (fCritical * Math.Sqrt(fCombinedError)); + } + + /// + /// Whether readings pin their mean down finely enough to speak to the shift being looked for + /// + /// IN - The smallest shift that could reach significance + /// true if a shift of the size being looked for could be shown; otherwise, false + public static bool IsSensitiveEnough(double fDetectableShift) + { + return ((false == double.IsNaN(fDetectableShift)) && (SHIFT_OF_INTEREST >= fDetectableShift)); + } + /// /// Turns a statistic and its degrees of freedom into an outcome, taking the two-tailed probability /// from the distribution the statistic follows @@ -222,6 +323,17 @@ private static double Variance(IList readings, double fMean) #endregion #region Constants + // A spread cannot be measured from fewer than two readings + /// + /// The size of shift a session is expected to be able to speak to. Readings that cannot pin their + /// mean down at least this finely are reported as too few, because a shift of this size could be + /// entirely real in them and still not reach significance. + /// NOTE: This is the order of the effect reported in the published work on influencing a generator: + /// around one part in ten thousand, a generator running at 0.5001 rather than 0.5. It is the figure + /// the wording in the interface quotes, so the two have to be changed together. + /// + public const double SHIFT_OF_INTEREST = 0.0001; + // A spread cannot be measured from fewer than two readings private const int m_iMINIMUM_READINGS = 2; diff --git a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs index 294145f..f93a679 100644 --- a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs +++ b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs @@ -376,6 +376,289 @@ private static List BuildReadings(Random source, int iCount, double fMea private const double m_fFREEDOM_TOLERANCE = 0.001; private const double m_fPROBABILITY_TOLERANCE = 0.000001; + /// + /// Tests the smallest detectable shift agrees with the test it is derived from. A shift of exactly + /// that size, applied to the readings, should sit right on the edge of significance: this is the + /// same arithmetic read the other way round, so the two have to meet. + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_ShiftOfThatSize_SitsOnTheEdgeOfSignificance() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fEXPECTED_MEAN = 0.5; + const double fEDGE_TOLERANCE = 0.002; + + // Readings with a spread, sitting on the expected value so the shift applied below is the whole + // of the difference the test sees + List readings = MakeSpreadReadings(200, fEXPECTED_MEAN, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fDetectable = SignificanceTest.SmallestDetectableShift(readings); + + // Move every reading by exactly that much and ask the test what it makes of it + List shifted = new List(); + foreach (double fReading in readings) + { + shifted.Add(fReading + fDetectable); + } + SignificanceResult onTheEdge = SignificanceTest.CompareWithExpected(shifted, fEXPECTED_MEAN); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify a shift of exactly the detectable size lands on the significance level rather than + // somewhere unrelated to it + Assert.IsTrue(onTheEdge.Valid); + Assert.AreEqual(SignificanceResult.SIGNIFICANCE_LEVEL, onTheEdge.Probability, fEDGE_TOLERANCE, + $"A shift of {fDetectable} gave a probability of {onTheEdge.Probability}"); + } + + /// + /// Tests a shift smaller than the detectable size cannot reach significance, which is the whole + /// point of warning about it + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_SmallerShift_CannotReachSignificance() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fEXPECTED_MEAN = 0.5; + const double fWELL_UNDER = 0.5; + + List readings = MakeSpreadReadings(200, fEXPECTED_MEAN, 0.001); + double fDetectable = SignificanceTest.SmallestDetectableShift(readings); + + //**************************************************************// + // Act + //**************************************************************// + + // Half the shift the readings could show + List shifted = new List(); + foreach (double fReading in readings) + { + shifted.Add(fReading + (fDetectable * fWELL_UNDER)); + } + SignificanceResult tooSmall = SignificanceTest.CompareWithExpected(shifted, fEXPECTED_MEAN); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify a real shift of that size goes unfound, which is what the warning exists to say + Assert.IsTrue(tooSmall.Valid); + Assert.IsFalse(tooSmall.Significant, + $"A shift of half the detectable size reached significance at {tooSmall.Probability}"); + } + + /// + /// Tests more readings pin the mean down more finely, at the rate the arithmetic says they should + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_MoreReadings_PinsTheMeanDownMoreFinely() + { + //**************************************************************// + // Arrange + //**************************************************************// + + // Four times the readings should halve the detectable shift, as it falls with the root of the + // count. The tolerance is loose because the critical value also moves with the count. + const double fEXPECTED_RATIO = 2.0; + const double fRATIO_TOLERANCE = 0.15; + + List few = MakeSpreadReadings(100, 0.5, 0.001); + List many = MakeSpreadReadings(400, 0.5, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fFew = SignificanceTest.SmallestDetectableShift(few); + double fMany = SignificanceTest.SmallestDetectableShift(many); + + //**************************************************************// + // Assert + //**************************************************************// + + Assert.IsTrue(fMany < fFew, "More readings should pin the mean down more finely"); + Assert.AreEqual(fEXPECTED_RATIO, (fFew / fMany), fRATIO_TOLERANCE, + $"Four times the readings gave a ratio of {(fFew / fMany)}"); + } + + /// + /// Tests readings too few to measure a spread report no detectable shift rather than a number that + /// looks like one + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_TooFewReadings_ReportsNoAnswer() + { + //**************************************************************// + // Arrange + //**************************************************************// + + List single = new List { 0.5 }; + List empty = new List(); + List identical = new List { 0.5, 0.5, 0.5, 0.5 }; + + //**************************************************************// + // Act & Assert + //**************************************************************// + + // Verify each reports no answer rather than zero, which would read as "any shift is detectable" + Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(single))); + Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(empty))); + Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(identical))); + } + + /// + /// Tests a session that cannot show the shift being looked for is reported as not sensitive enough, + /// and one that can is not + /// + [TestMethod] + [TestCategory("Component")] + public void IsSensitiveEnough_AgainstTheShiftOfInterest_AnswersEitherWay() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fJUST_INSIDE = 0.9; + const double fJUST_OUTSIDE = 1.1; + + //**************************************************************// + // Act & Assert + //**************************************************************// + + // Verify the answer turns on the shift being looked for rather than on a reading count + Assert.IsTrue(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST * fJUST_INSIDE)); + Assert.IsTrue(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST)); + Assert.IsFalse(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST * fJUST_OUTSIDE)); + + // Verify no answer is not mistaken for a good one + Assert.IsFalse(SignificanceTest.IsSensitiveEnough(double.NaN)); + } + + /// + /// Tests a short device session cannot show the shift being looked for while a longer one can, using + /// the spread a real device actually produces. This is the case the warning was added for. + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_RealDeviceSpread_TurnsOverAtTheExpectedLength() + { + //**************************************************************// + // Arrange + //**************************************************************// + + // Each device reading is the average of 32768 bytes, so its spread is about 0.5 over the root + // of that many bits. A minute of recording is roughly 550 readings. + const double fDEVICE_SPREAD = 0.00098; + const int iTEN_SECONDS = 90; + const int iTWO_MINUTES = 1100; + + List shortSession = MakeSpreadReadings(iTEN_SECONDS, 0.5, fDEVICE_SPREAD); + List longSession = MakeSpreadReadings(iTWO_MINUTES, 0.5, fDEVICE_SPREAD); + + //**************************************************************// + // Act + //**************************************************************// + + bool bShortEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.SmallestDetectableShift(shortSession)); + bool bLongEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.SmallestDetectableShift(longSession)); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify ten seconds of device readings cannot speak to the shift and two minutes can + Assert.IsFalse(bShortEnough, "Ten seconds of readings should not be enough"); + Assert.IsTrue(bLongEnough, "Two minutes of readings should be enough"); + } + + /// + /// Tests the difference two sessions can show between them is reported, and that it is coarser than + /// what either could show on its own - the noise of both stands between them + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableDifference_TwoSessions_IsCoarserThanEitherAlone() + { + //**************************************************************// + // Arrange + //**************************************************************// + + List baseline = MakeSpreadReadings(300, 0.5, 0.001); + List result = MakeSpreadReadings(300, 0.5, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fBaselineAlone = SignificanceTest.SmallestDetectableShift(baseline); + double fBetween = SignificanceTest.SmallestDetectableDifference(baseline, result); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify comparing two sessions is harder than measuring one against a fixed value, because both + // sides carry noise + Assert.IsFalse(double.IsNaN(fBetween)); + Assert.IsTrue(fBetween > fBaselineAlone, + $"Between them: {fBetween}, baseline alone: {fBaselineAlone}"); + } + + /// + /// Tests a null set of readings is refused rather than dereferenced + /// + [TestMethod] + [TestCategory("Component")] + public void SmallestDetectableShift_NullReadings_Exception() + { + //**************************************************************// + // Act & Assert + //**************************************************************// + + Assert.ThrowsException(() => SignificanceTest.SmallestDetectableShift(null)); + Assert.ThrowsException(() => + SignificanceTest.SmallestDetectableDifference(null, new List { 0.5, 0.6 })); + Assert.ThrowsException(() => + SignificanceTest.SmallestDetectableDifference(new List { 0.5, 0.6 }, null)); + } + + /// + /// Builds readings with a known mean and spread, alternating either side of the mean so the spread + /// is exactly what was asked for rather than whatever a random draw happened to give + /// + /// IN - How many readings to make + /// IN - The value to centre them on + /// IN - The standard deviation to give them + /// The readings + private static List MakeSpreadReadings(int iCount, double fMean, double fSpread) + { + List readings = new List(); + for (int iIndex = 0; iIndex < iCount; ++iIndex) + { + // Half above and half below, so the mean lands where it was asked to and the spread is the + // offset itself + double fOffset = ((0 == (iIndex % 2)) ? fSpread : -fSpread); + readings.Add(fMean + fOffset); + } + return readings; + } + #endregion } } diff --git a/tests/manual/README.md b/tests/manual/README.md index 7fe0fe4..84dc049 100644 --- a/tests/manual/README.md +++ b/tests/manual/README.md @@ -39,6 +39,7 @@ already does. | `test-analyze.ps1` | no | The Analyse tab: the comparison table, the significance verdict and its wording, the histogram, and recovering a session file left unterminated by killing the application mid-write. | | `check-chart.ps1` | no | That the chart and the statistics beside it agree on every path — recording, a second session into the same file, a different file being chosen, an existing file being reopened, and Clear. | | `inproc-length.ps1` | no | The session length, and the data file surviving a session ending. Drives the form directly, so it covers the state machine rather than the pixels. | +| `check-sensitivity.ps1` | no | That the verdict states what shift the two sessions could show, and warns when nothing was found because nothing could have been. | | `check-appended.ps1` | no | That a file grown by a second session reads back correctly on the Record tab and analyses correctly on the Analyse tab. | | `check-device.ps1` | **yes** | Device discovery, the native reads, a session recorded from the hardware, a second session appended to it, a timed session, and the recorded file analysing. Also that a port with nothing on it is refused. | | `check-reinit.ps1` | **yes** | Switching between the device and the simulator and back, which tears down the native interface and builds a new one each time. | diff --git a/tests/manual/check-sensitivity.ps1 b/tests/manual/check-sensitivity.ps1 new file mode 100644 index 0000000..a06f749 --- /dev/null +++ b/tests/manual/check-sensitivity.ps1 @@ -0,0 +1,124 @@ +######################################################################################################################### +# File Name: check-sensitivity.ps1 +# Description: Checks the analysis warns when sessions are too short to show the shift being looked for +# +# Copyright (c) 2026 Mike Pullen +# Licensed under the MIT License. See LICENSE in the repository root. +# +# Revision History: +#====================================================================================================================== +# 2026/09/09 - Mike Pullen - Original implementation. +######################################################################################################################### +# Drives the analysis tab through all three verdicts: a shift found, nothing found from sessions long +# enough to have found one, and nothing found from sessions too short to have found anything. The third is +# the one the warning exists for. +$ErrorActionPreference = "Continue" +Add-Type -AssemblyName System.Windows.Forms +try { + [System.Windows.Forms.Application]::EnableVisualStyles() + [System.Windows.Forms.Application]::SetCompatibleTextRenderingDefault($false) +} catch { } +# Which build to drive. Set RNG_BIN to test a different one. +if ([string]::IsNullOrEmpty($env:RNG_BIN)) { $env:RNG_BIN = (Resolve-Path (Join-Path $PSScriptRoot "..\..\bin\Release")).Path } +$binDir = $env:RNG_BIN +[System.Reflection.Assembly]::LoadFrom((Join-Path $binDir "Random Number Generator.exe")) | Out-Null +[Environment]::CurrentDirectory = $binDir + +$WorkRoot = Join-Path $PSScriptRoot "work" +if (-not (Test-Path $WorkRoot)) { $null = New-Item -ItemType Directory -Path $WorkRoot } + +$script:pass = 0; $script:fail = 0 +function Check($name, $condition, $detail) { + if ($condition) { $script:pass++; Write-Host (" PASS " + $name + " " + $detail) } + else { $script:fail++; Write-Host (" FAIL " + $name + " " + $detail) } +} + +# A session file with a known count, centre and spread. Readings alternate either side of the centre so the +# spread is exactly what was asked for, which makes the detectable shift predictable. +function WriteSession($path, $count, $centre, $spread) { + $sb = New-Object System.Text.StringBuilder + [void]$sb.AppendLine('') + [void]$sb.AppendLine('') + for ($i = 0; $i -lt $count; $i++) { + $offset = if (0 -eq ($i % 2)) { $spread } else { -$spread } + $v = ($centre + $offset).ToString("F12", [System.Globalization.CultureInfo]::InvariantCulture) + [void]$sb.AppendLine("`t$v") + } + [void]$sb.Append('') + [System.IO.File]::WriteAllText($path, $sb.ToString()) +} + +# The spread a real device reading has: the average of 32768 bytes of bits +$DEVICE_SPREAD = 0.00098 + +$ws = New-Object System.Xml.XmlWriterSettings +$writer = New-Object RandomNumberGenerator.RNGXMLWriter($ws) +$reader = New-Object RandomNumberGenerator.RNGXMLReader +$dataFile= New-Object RandomNumberGenerator.RNGSessionDataFile($writer, $reader) +$timer = New-Object RandomNumberGenerator.RNGSessionTimer +$data = New-Object RandomNumberGenerator.RNGSessionData($dataFile, $timer) +$device = New-Object RandomNumberGenerator.RNGDeviceTimer +$form = New-Object RandomNumberGenerator.GeneratorForm($data, $device) +$BF = [System.Reflection.BindingFlags]"NonPublic,Instance" +function Fld($name) { return $form.GetType().GetField($name, $BF).GetValue($form) } +function Call1($name, $arg) { return $form.GetType().GetMethod($name, $BF).Invoke($form, [object[]]@([string]$arg)) } +function Call2($name, $a, $b) { return $form.GetType().GetMethod($name, $BF).Invoke($form, [object[]]@([string]$a, [bool]$b)) } +function Verdict() { return (Fld "m_VerdictLabel").Text } +function VerdictColour() { return (Fld "m_VerdictLabel").ForeColor } + +$form.Show(); [System.Windows.Forms.Application]::DoEvents() + +function LoadPair($baseline, $result) { + $b = Call1 "LoadBaselineFile" $baseline + Call2 "OnBaselineLoadCompleted" $baseline $b | Out-Null + $r = Call1 "LoadResultFile" $result + Call2 "OnResultLoadCompleted" $result $r | Out-Null + [System.Windows.Forms.Application]::DoEvents() +} + +Write-Host "=== 1. TOO SHORT TO SAY ANYTHING - the case the warning exists for ===" +$shortA = Join-Path $WorkRoot "sens-short-a.rng" +$shortB = Join-Path $WorkRoot "sens-short-b.rng" +WriteSession $shortA 90 0.5 $DEVICE_SPREAD +WriteSession $shortB 90 0.5 $DEVICE_SPREAD +LoadPair $shortA $shortB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "warns that the sessions are too short" ($v -match "too short to conclude") "" +Check "states the limit" ($v -match "can only show a difference of") "" +Check "names the shift being looked for" ($v -match "0.000100") "" +Check "tells the user what to do" ($v -match "Record for longer") "" +Check "coloured as a warning" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.StatusPalette]::WarningText).ToArgb()) ("colour=" + (VerdictColour)) + +Write-Host "" +Write-Host "=== 2. LONG ENOUGH, NOTHING FOUND - the ordinary outcome ===" +$longA = Join-Path $WorkRoot "sens-long-a.rng" +$longB = Join-Path $WorkRoot "sens-long-b.rng" +WriteSession $longA 1500 0.5 $DEVICE_SPREAD +WriteSession $longB 1500 0.5 $DEVICE_SPREAD +LoadPair $longA $longB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "reports no significant shift" ($v -match "No significant shift\.") "" +Check "does not warn" (-not ($v -match "too short")) "" +Check "says the shift would have been found" ($v -match "would have been found") "" +Check "coloured as ordinary" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.UiPalette]::MutedText).ToArgb()) ("colour=" + (VerdictColour)) + +Write-Host "" +Write-Host "=== 3. A SHIFT FOUND - the limit is reported, not warned about ===" +$shiftA = Join-Path $WorkRoot "sens-shift-a.rng" +$shiftB = Join-Path $WorkRoot "sens-shift-b.rng" +WriteSession $shiftA 1500 0.5 $DEVICE_SPREAD +WriteSession $shiftB 1500 0.5005 $DEVICE_SPREAD +LoadPair $shiftA $shiftB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "reports a significant shift" ($v -match "Significant shift:") "" +Check "still states the limit" ($v -match "can show a difference of") "" +Check "does not warn" (-not ($v -match "too short")) "" +Check "coloured as notable" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.UiPalette]::CardText).ToArgb()) ("colour=" + (VerdictColour)) + +$form.Close() +[System.Windows.Forms.Application]::DoEvents() +Write-Host "" +Write-Host ("PASS " + $script:pass + " FAIL " + $script:fail) diff --git a/tests/manual/run-all.ps1 b/tests/manual/run-all.ps1 index cdd8363..86f49bb 100644 --- a/tests/manual/run-all.ps1 +++ b/tests/manual/run-all.ps1 @@ -37,6 +37,7 @@ $suites = @( "check-chart.ps1", "inproc-length.ps1", "check-appended.ps1", + "check-sensitivity.ps1", "check-device.ps1", "check-reinit.ps1", "check-nodevice.ps1" From 3fe3170192b96054cc6f93438013d990f20772fd Mon Sep 17 00:00:00 2001 From: Skorinn <42702903+Skorinn@users.noreply.github.com> Date: Thu, 10 Sep 2026 00:43:01 -0500 Subject: [PATCH 2/4] Take the stranded comment off the shift constant The comment about needing two readings to measure a spread belongs to the minimum readings constant below it, not to the shift being looked for. It was left stranded there by the edit that added the new constant: it anchored on the declaration alone, so the comment that was already above it stayed where it was and a fresh copy went in with the declaration that owned it. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_0153VVkWg7DQmdaNLcvtY37w --- SignificanceTest.cs | 1 - 1 file changed, 1 deletion(-) diff --git a/SignificanceTest.cs b/SignificanceTest.cs index 18365ce..9fba90a 100644 --- a/SignificanceTest.cs +++ b/SignificanceTest.cs @@ -323,7 +323,6 @@ private static double Variance(IList readings, double fMean) #endregion #region Constants - // A spread cannot be measured from fewer than two readings /// /// The size of shift a session is expected to be able to speak to. Readings that cannot pin their /// mean down at least this finely are reported as too few, because a shift of this size could be From f9818bf8f2974818b7d9e26e46ebe3ba4670add3 Mon Sep 17 00:00:00 2001 From: Skorinn <42702903+Skorinn@users.noreply.github.com> Date: Thu, 10 Sep 2026 01:42:13 -0500 Subject: [PATCH 3/4] Return what a test could have found along with what it found The verdict was asking for the detectable difference separately from the test it sits beside, which walked both sets of readings a second time to work out quantities the test had already worked out: the means, the variances, the combined error and Welch's degrees of freedom. Measured, the second walk costs under four milliseconds for an eight hour session, on a path that runs when a file is loaded rather than while anything is recording, so the cost was not the reason to change it. The reason is that the two were separate copies of the same arithmetic and had to agree. The test that shifts readings by exactly the detectable amount and expects the probability to land on the significance level only passed because they did; change Welch's degrees of freedom in one and the verdict would have quoted a limit that disagreed with the test printed beside it. The limit now comes back with the result, worked out where the probability is worked out and from the same critical value, so they cannot drift apart. The standalone calculations are gone and the tests go through the tests. An outcome with too little to test reports no answer for the limit rather than zero, which is why the invalid returns go through one place now. Zero would have read as "a difference of any size would have been found", and anything asking whether the readings were sensitive enough would have agreed with it. Also corrects the test helper's description: what it takes is how far each reading sits from the mean, not the standard deviation the readings end up with. Those are close but not equal, the sample deviation being the larger for dividing by one fewer than the count. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_0153VVkWg7DQmdaNLcvtY37w --- GeneratorForm.cs | 6 +- SignificanceTest.cs | 140 +++++------------- .../SignificanceTest.Test.cs | 66 +++++---- 3 files changed, 81 insertions(+), 131 deletions(-) diff --git a/GeneratorForm.cs b/GeneratorForm.cs index 9450d8a..33e143d 100644 --- a/GeneratorForm.cs +++ b/GeneratorForm.cs @@ -2481,8 +2481,10 @@ private void UpdateComparisonVerdict(List baselineReadings, List // How finely the two sessions between them pin a difference down. A verdict of nothing found // means nothing at all unless a shift worth finding could have been seen, so the limit is - // stated either way rather than left for the reader to work out from the reading counts. - double fDetectable = SignificanceTest.SmallestDetectableDifference(baselineReadings, resultReadings); + // stated either way rather than left for the reader to work out from the reading counts. It + // comes back with the test rather than being asked for separately, so the readings are walked + // once and the limit cannot disagree with the test it is quoted beside. + double fDetectable = test.DetectableDifference; bool bSensitive = SignificanceTest.IsSensitiveEnough(fDetectable); string sDetectable = fDetectable.ToString(m_sVALUE_FORMAT); string sOfInterest = SignificanceTest.SHIFT_OF_INTEREST.ToString(m_sVALUE_FORMAT); diff --git a/SignificanceTest.cs b/SignificanceTest.cs index 9fba90a..db79eae 100644 --- a/SignificanceTest.cs +++ b/SignificanceTest.cs @@ -47,6 +47,17 @@ public struct SignificanceResult /// public double Probability { get; set; } + /// + /// The smallest difference this test could have shown as significant, or NaN when there was too + /// little to test. A test that found nothing has said something only if it could have found + /// something, and this is what it could have found: below it a difference can be entirely real and + /// still not reach significance. + /// NOTE: This is the half width of the confidence interval, which is the same arithmetic the test + /// itself does, read the other way round. It comes back with the test rather than being worked out + /// separately so that the two cannot drift apart, and so the readings are walked once. + /// + public double DetectableDifference { get; set; } + /// /// Whether the difference is significant at the level the application reports against (read-only) /// @@ -99,7 +110,7 @@ public static SignificanceResult CompareMeans(IList baseline, IList baseline, IList= fCombinedError) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fStatistic = ((fResultMean - fBaselineMean) / Math.Sqrt(fCombinedError)); @@ -122,7 +133,7 @@ public static SignificanceResult CompareMeans(IList baseline, IList @@ -142,109 +153,18 @@ public static SignificanceResult CompareWithExpected(IList readings, dou if (m_iMINIMUM_READINGS > readings.Count) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fMean = Mean(readings); double fStandardError = (Variance(readings, fMean) / readings.Count); if (m_fMINIMUM_ERROR >= fStandardError) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fStatistic = ((fMean - fExpectedMean) / Math.Sqrt(fStandardError)); - return BuildResult(fStatistic, (readings.Count - 1)); - } - - /// - /// The smallest shift away from the expected value that a set of readings could show as significant. - /// A session answers the question it was recorded to answer only if the shift being looked for is - /// larger than this: below it, a shift could be perfectly real and still not reach significance, - /// because the readings are not pinned down finely enough to tell it from noise. - /// NOTE: This is the half width of the confidence interval, the same arithmetic the test itself - /// uses, read the other way round. The test asks whether the shift that happened is larger than the - /// noise; this asks how large a shift would have to be before it could. - /// - /// IN - The readings to measure (cannot be null) - /// The smallest shift that could reach significance, or NaN when there is too little to say - /// Thrown when the readings are null - public static double SmallestDetectableShift(IList readings) - { - if (null == readings) - { - throw new ArgumentNullException(nameof(readings), "Readings cannot be null"); - } - - // A spread cannot be measured from fewer than two readings, so nothing can be said about how - // finely they pin the mean down - if (m_iMINIMUM_READINGS > readings.Count) - { - return double.NaN; - } - - double fMean = Mean(readings); - double fStandardError = (Variance(readings, fMean) / readings.Count); - if (m_fMINIMUM_ERROR >= fStandardError) - { - return double.NaN; - } - - // The critical value comes from the same distribution the test uses, so the answer agrees with - // the test rather than approximating it. It is well above two for a handful of readings, which - // is exactly the case this exists to report on. - double fFreedom = (readings.Count - 1); - double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); - - return (fCritical * Math.Sqrt(fStandardError)); - } - - /// - /// The smallest difference between two sets of readings that could be shown as significant. This is - /// the pair's answer to the same question, and it is the one the verdict reports, because the - /// verdict is about the comparison rather than about either session on its own. - /// - /// IN - The baseline readings (cannot be null) - /// IN - The result readings (cannot be null) - /// The smallest difference that could reach significance, or NaN when there is too little to say - /// Thrown when either set of readings is null - public static double SmallestDetectableDifference(IList baseline, IList result) - { - if (null == baseline) - { - throw new ArgumentNullException(nameof(baseline), "Baseline readings cannot be null"); - } - - if (null == result) - { - throw new ArgumentNullException(nameof(result), "Result readings cannot be null"); - } - - bool bEnoughData = ((m_iMINIMUM_READINGS <= baseline.Count) && (m_iMINIMUM_READINGS <= result.Count)); - if (false == bEnoughData) - { - return double.NaN; - } - - double fBaselineError = (Variance(baseline, Mean(baseline)) / baseline.Count); - double fResultError = (Variance(result, Mean(result)) / result.Count); - double fCombinedError = (fBaselineError + fResultError); - if (m_fMINIMUM_ERROR >= fCombinedError) - { - return double.NaN; - } - - // Welch's degrees of freedom, as in the test the two are compared with - double fDenominator = (((fBaselineError * fBaselineError) / (baseline.Count - 1)) + - ((fResultError * fResultError) / (result.Count - 1))); - double fFreedom = ((fCombinedError * fCombinedError) / fDenominator); - if ((double.IsNaN(fFreedom)) || (0.0 >= fFreedom)) - { - return double.NaN; - } - - double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); - - return (fCritical * Math.Sqrt(fCombinedError)); + return BuildResult(fStatistic, (readings.Count - 1), Math.Sqrt(fStandardError)); } /// @@ -257,14 +177,28 @@ public static bool IsSensitiveEnough(double fDetectableShift) return ((false == double.IsNaN(fDetectableShift)) && (SHIFT_OF_INTEREST >= fDetectableShift)); } + /// + /// The outcome for readings there was too little of to test + /// + /// An outcome reporting itself invalid, with no answer for what it could have found + private static SignificanceResult NoResult() + { + // The detectable difference is set to no answer rather than left at zero. Zero would read as + // "a difference of any size would have been found", which is the opposite of what having too + // little to test means, and anything asking whether the readings are sensitive enough would + // agree with it. + return new SignificanceResult { Valid = false, DetectableDifference = double.NaN }; + } + /// /// Turns a statistic and its degrees of freedom into an outcome, taking the two-tailed probability /// from the distribution the statistic follows /// /// IN - The size of the difference in standard errors /// IN - The degrees of freedom of the test + /// IN - The standard error the statistic was measured against /// The completed outcome - private static SignificanceResult BuildResult(double fStatistic, double fFreedom) + private static SignificanceResult BuildResult(double fStatistic, double fFreedom, double fStandardError) { // A statistic that is not a number says the arithmetic ran out of meaning rather than that the // difference was enormous, so it is reported as no test rather than as a certainty @@ -272,19 +206,25 @@ private static SignificanceResult BuildResult(double fStatistic, double fFreedom (false == double.IsNaN(fFreedom)) && (0 < fFreedom)); if (false == bUsable) { - return new SignificanceResult { Valid = false }; + return NoResult(); } // Both tails, so a shift in either direction counts against the null double fUpperTail = (1.0 - StudentT.CDF(0.0, 1.0, fFreedom, Math.Abs(fStatistic))); double fProbability = Math.Min(1.0, (2.0 * fUpperTail)); + // How large a difference would have had to be to reach significance here. The critical value is + // taken from the same distribution the probability came from, so the two agree by construction + // rather than by two pieces of arithmetic happening to match. + double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); + return new SignificanceResult { Valid = true, Statistic = fStatistic, DegreesOfFreedom = fFreedom, - Probability = fProbability + Probability = fProbability, + DetectableDifference = (fCritical * fStandardError) }; } diff --git a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs index f93a679..31c6af3 100644 --- a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs +++ b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs @@ -383,7 +383,7 @@ private static List BuildReadings(Random source, int iCount, double fMea /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_ShiftOfThatSize_SitsOnTheEdgeOfSignificance() + public void DetectableDifference_ShiftOfThatSize_SitsOnTheEdgeOfSignificance() { //**************************************************************// // Arrange @@ -400,7 +400,7 @@ public void SmallestDetectableShift_ShiftOfThatSize_SitsOnTheEdgeOfSignificance( // Act //**************************************************************// - double fDetectable = SignificanceTest.SmallestDetectableShift(readings); + double fDetectable = SignificanceTest.CompareWithExpected(readings, fEXPECTED_MEAN).DetectableDifference; // Move every reading by exactly that much and ask the test what it makes of it List shifted = new List(); @@ -427,7 +427,7 @@ public void SmallestDetectableShift_ShiftOfThatSize_SitsOnTheEdgeOfSignificance( /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_SmallerShift_CannotReachSignificance() + public void DetectableDifference_SmallerShift_CannotReachSignificance() { //**************************************************************// // Arrange @@ -437,7 +437,7 @@ public void SmallestDetectableShift_SmallerShift_CannotReachSignificance() const double fWELL_UNDER = 0.5; List readings = MakeSpreadReadings(200, fEXPECTED_MEAN, 0.001); - double fDetectable = SignificanceTest.SmallestDetectableShift(readings); + double fDetectable = SignificanceTest.CompareWithExpected(readings, fEXPECTED_MEAN).DetectableDifference; //**************************************************************// // Act @@ -466,7 +466,7 @@ public void SmallestDetectableShift_SmallerShift_CannotReachSignificance() /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_MoreReadings_PinsTheMeanDownMoreFinely() + public void DetectableDifference_MoreReadings_PinsTheMeanDownMoreFinely() { //**************************************************************// // Arrange @@ -484,8 +484,8 @@ public void SmallestDetectableShift_MoreReadings_PinsTheMeanDownMoreFinely() // Act //**************************************************************// - double fFew = SignificanceTest.SmallestDetectableShift(few); - double fMany = SignificanceTest.SmallestDetectableShift(many); + double fFew = SignificanceTest.CompareWithExpected(few, 0.5).DetectableDifference; + double fMany = SignificanceTest.CompareWithExpected(many, 0.5).DetectableDifference; //**************************************************************// // Assert @@ -502,7 +502,7 @@ public void SmallestDetectableShift_MoreReadings_PinsTheMeanDownMoreFinely() /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_TooFewReadings_ReportsNoAnswer() + public void DetectableDifference_TooFewReadings_ReportsNoAnswer() { //**************************************************************// // Arrange @@ -517,9 +517,9 @@ public void SmallestDetectableShift_TooFewReadings_ReportsNoAnswer() //**************************************************************// // Verify each reports no answer rather than zero, which would read as "any shift is detectable" - Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(single))); - Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(empty))); - Assert.IsTrue(double.IsNaN(SignificanceTest.SmallestDetectableShift(identical))); + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(single, 0.5).DetectableDifference)); + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(empty, 0.5).DetectableDifference)); + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(identical, 0.5).DetectableDifference)); } /// @@ -556,7 +556,7 @@ public void IsSensitiveEnough_AgainstTheShiftOfInterest_AnswersEitherWay() /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_RealDeviceSpread_TurnsOverAtTheExpectedLength() + public void DetectableDifference_RealDeviceSpread_TurnsOverAtTheExpectedLength() { //**************************************************************// // Arrange @@ -575,8 +575,8 @@ public void SmallestDetectableShift_RealDeviceSpread_TurnsOverAtTheExpectedLengt // Act //**************************************************************// - bool bShortEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.SmallestDetectableShift(shortSession)); - bool bLongEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.SmallestDetectableShift(longSession)); + bool bShortEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.CompareWithExpected(shortSession, 0.5).DetectableDifference); + bool bLongEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.CompareWithExpected(longSession, 0.5).DetectableDifference); //**************************************************************// // Assert @@ -593,7 +593,7 @@ public void SmallestDetectableShift_RealDeviceSpread_TurnsOverAtTheExpectedLengt /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableDifference_TwoSessions_IsCoarserThanEitherAlone() + public void DetectableDifference_TwoSessions_IsCoarserThanEitherAlone() { //**************************************************************// // Arrange @@ -606,8 +606,8 @@ public void SmallestDetectableDifference_TwoSessions_IsCoarserThanEitherAlone() // Act //**************************************************************// - double fBaselineAlone = SignificanceTest.SmallestDetectableShift(baseline); - double fBetween = SignificanceTest.SmallestDetectableDifference(baseline, result); + double fBaselineAlone = SignificanceTest.CompareWithExpected(baseline, 0.5).DetectableDifference; + double fBetween = SignificanceTest.CompareMeans(baseline, result).DetectableDifference; //**************************************************************// // Assert @@ -625,36 +625,44 @@ public void SmallestDetectableDifference_TwoSessions_IsCoarserThanEitherAlone() /// [TestMethod] [TestCategory("Component")] - public void SmallestDetectableShift_NullReadings_Exception() + public void Tests_NullReadings_Exception() { + //**************************************************************// + // Arrange + //**************************************************************// + + List readings = new List { 0.5, 0.6 }; + //**************************************************************// // Act & Assert //**************************************************************// - Assert.ThrowsException(() => SignificanceTest.SmallestDetectableShift(null)); - Assert.ThrowsException(() => - SignificanceTest.SmallestDetectableDifference(null, new List { 0.5, 0.6 })); - Assert.ThrowsException(() => - SignificanceTest.SmallestDetectableDifference(new List { 0.5, 0.6 }, null)); + // The side each test is given as null, which the existing cover misses for all but the baseline + Assert.ThrowsException(() => SignificanceTest.CompareWithExpected(null, 0.5)); + Assert.ThrowsException(() => SignificanceTest.CompareMeans(null, readings)); + Assert.ThrowsException(() => SignificanceTest.CompareMeans(readings, null)); } /// - /// Builds readings with a known mean and spread, alternating either side of the mean so the spread - /// is exactly what was asked for rather than whatever a random draw happened to give + /// Builds readings sitting a fixed distance either side of a mean, alternating, so their spread is + /// settled rather than whatever a random draw happened to give. + /// NOTE: fOffset is how far each reading sits from the mean, not the sample standard deviation the + /// readings end up with. Those are close but not equal: the sample deviation divides by one fewer + /// than the count, so it comes out slightly the larger of the two. /// /// IN - How many readings to make /// IN - The value to centre them on - /// IN - The standard deviation to give them + /// IN - How far each reading sits either side of the mean /// The readings - private static List MakeSpreadReadings(int iCount, double fMean, double fSpread) + private static List MakeSpreadReadings(int iCount, double fMean, double fOffset) { List readings = new List(); for (int iIndex = 0; iIndex < iCount; ++iIndex) { // Half above and half below, so the mean lands where it was asked to and the spread is the // offset itself - double fOffset = ((0 == (iIndex % 2)) ? fSpread : -fSpread); - readings.Add(fMean + fOffset); + double fStep = ((0 == (iIndex % 2)) ? fOffset : -fOffset); + readings.Add(fMean + fStep); } return readings; } From 51bd0cc618aef9ab277720960c9859178ec297b9 Mon Sep 17 00:00:00 2001 From: Skorinn <42702903+Skorinn@users.noreply.github.com> Date: Thu, 10 Sep 2026 07:47:06 -0500 Subject: [PATCH 4/4] Put the new tests in the tests region and name them as the others are Three points from the review, all of them right. CLAUDE.md still named SmallestDetectableDifference, which the last change removed. It reads SignificanceResult.DetectableDifference now, and the paragraph around it says where the value is worked out. The tests added for the detectable difference were sitting inside the constants region rather than the tests region. The splice that added them anchored on the last endregion in the file, which closes the constants rather than the tests, so seven tests and a helper landed under a label that says none of them are there. They are now at the end of the tests region, after the other helper, which is where this file keeps them. The same splice put two tests in DeviceUpdateThread.Test.cs inside its helper types region. That file is not part of this change and the review could not see it, but it is the same mistake from the same edit and is a move of two methods, so it is corrected here rather than left to be found later. Tests_NullReadings_Exception was named after nothing: there is no method called Tests. It covered three cases, one of which already had a test of its own. It is two tests now, named for the method and the case each one covers, and the duplicate is gone. 252 tests pass on both configurations. No application code changed, so the verdict path is as it was; the sensitivity suite was run against the built application to confirm it. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_0153VVkWg7DQmdaNLcvtY37w --- CLAUDE.md | 8 ++- .../DeviceUpdateThread.Test.cs | 24 +++---- .../SignificanceTest.Test.cs | 69 +++++++++++-------- 3 files changed, 58 insertions(+), 43 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index f0c3638..e8b9737 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -194,9 +194,11 @@ a number - fewer than two readings, or readings with no spread - returns `Valid probability, and `Significant` is false whenever the test did not run. A verdict of nothing found means nothing at all unless a shift worth finding could have been seen, so the -analysis states what it could have seen. `SmallestDetectableDifference` is the half width of the confidence -interval for the pair, which is the test read the other way round: instead of asking whether the shift that -happened beats the noise, it asks how large a shift would have to be before it could. Below +analysis states what it could have seen. `SignificanceResult.DetectableDifference` is the half width of the +confidence interval for the pair, which is the test read the other way round: instead of asking whether the +shift that happened beats the noise, it asks how large a shift would have to be before it could. It comes +back with the test, worked out in `BuildResult` from the same critical value the probability comes from, so +the readings are walked once and the limit cannot disagree with the test it is quoted beside. Below `SHIFT_OF_INTEREST` - one part in ten thousand, the order of the effect reported in the published work - the sessions cannot speak to the question, and the verdict says so in the warning colour rather than reporting a null result that reads as evidence of absence. diff --git a/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs b/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs index 7b1782a..879d72d 100644 --- a/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs +++ b/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs @@ -644,17 +644,6 @@ public void ThreadProc_InfoBoxChangedDuringUpdate_NewerMessageKept() mockParent.Verify(mock => mock.SetStatusBoxState(m_sTEST_STATUS_TEXT, It.IsAny(), It.IsAny()), Times.Never); } - #endregion - #region Helper Types - - /// - /// Delegate for GetStatusBoxState callback to handle out parameters in Moq - /// - /// OUT - Status box text - /// OUT - Status box text color - /// OUT - Status box background color - private delegate void GetStatusBoxStateCallback(out string text, out Color textColor, out Color backColor); - /// /// Tests the device status colours are taken from the palette when they are used rather than held /// from when the class was first touched. The palette reports the scheme in force now, so anything @@ -734,6 +723,17 @@ public void StatusPalette_SeverityColours_AreReadEachTimeRatherThanFixed() } } + #endregion + #region Helper Types + + /// + /// Delegate for GetStatusBoxState callback to handle out parameters in Moq + /// + /// OUT - Status box text + /// OUT - Status box text color + /// OUT - Status box background color + private delegate void GetStatusBoxStateCallback(out string text, out Color textColor, out Color backColor); + #endregion } -} \ No newline at end of file +} diff --git a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs index 31c6af3..b05b2b7 100644 --- a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs +++ b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs @@ -352,30 +352,6 @@ private static List BuildReadings(Random source, int iCount, double fMea return readings; } - #endregion - #region Constants - - // A fixed seed, so a test that passes today passes tomorrow - private const int m_iFIXED_SEED = 20260907; - - // The value an unbiased generator's readings average to - private const double m_fEXPECTED_MEAN = 0.5; - - // How widely the readings are scattered, of the order the device produces - private const double m_fSCATTER_WIDTH = 0.02; - - // A shift large enough that it should be found rather than missed - private const double m_fCLEAR_SHIFT = 0.004; - - // Session sizes, of the order a real comparison is made from - private const int m_iTYPICAL_COUNT = 400; - private const int m_iLONG_COUNT = 2000; - private const int m_iSHORT_COUNT = 120; - - // Tolerances - private const double m_fFREEDOM_TOLERANCE = 0.001; - private const double m_fPROBABILITY_TOLERANCE = 0.000001; - /// /// Tests the smallest detectable shift agrees with the test it is derived from. A shift of exactly /// that size, applied to the readings, should sit right on the edge of significance: this is the @@ -625,7 +601,23 @@ public void DetectableDifference_TwoSessions_IsCoarserThanEitherAlone() /// [TestMethod] [TestCategory("Component")] - public void Tests_NullReadings_Exception() + public void CompareWithExpected_NullReadings_ThrowsArgumentNullException() + { + //**************************************************************// + // Act & Assert + //**************************************************************// + + Assert.ThrowsException( + () => SignificanceTest.CompareWithExpected(null, m_fEXPECTED_MEAN)); + } + + /// + /// Tests a null result is refused rather than dereferenced. The baseline side is covered above; this + /// is the other one. + /// + [TestMethod] + [TestCategory("Component")] + public void CompareMeans_NullResult_ThrowsArgumentNullException() { //**************************************************************// // Arrange @@ -637,9 +629,6 @@ public void Tests_NullReadings_Exception() // Act & Assert //**************************************************************// - // The side each test is given as null, which the existing cover misses for all but the baseline - Assert.ThrowsException(() => SignificanceTest.CompareWithExpected(null, 0.5)); - Assert.ThrowsException(() => SignificanceTest.CompareMeans(null, readings)); Assert.ThrowsException(() => SignificanceTest.CompareMeans(readings, null)); } @@ -667,6 +656,30 @@ private static List MakeSpreadReadings(int iCount, double fMean, double return readings; } + #endregion + #region Constants + + // A fixed seed, so a test that passes today passes tomorrow + private const int m_iFIXED_SEED = 20260907; + + // The value an unbiased generator's readings average to + private const double m_fEXPECTED_MEAN = 0.5; + + // How widely the readings are scattered, of the order the device produces + private const double m_fSCATTER_WIDTH = 0.02; + + // A shift large enough that it should be found rather than missed + private const double m_fCLEAR_SHIFT = 0.004; + + // Session sizes, of the order a real comparison is made from + private const int m_iTYPICAL_COUNT = 400; + private const int m_iLONG_COUNT = 2000; + private const int m_iSHORT_COUNT = 120; + + // Tolerances + private const double m_fFREEDOM_TOLERANCE = 0.001; + private const double m_fPROBABILITY_TOLERANCE = 0.000001; + #endregion } }