diff --git a/CLAUDE.md b/CLAUDE.md index 265c721..e8b9737 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -190,9 +190,29 @@ spreads, because a baseline is usually recorded for far longer than the run comp `CompareWithExpected` tests one session against the 0.5 an unbiased generator gives. Both are two-tailed: a one-tailed test would find a shift toward a target more easily, but only holds when the direction was predicted before the readings were taken, which the application cannot know. Every path that cannot produce -a number — fewer than two readings, or readings with no spread — returns `Valid = false` rather than a +a number - fewer than two readings, or readings with no spread - returns `Valid = false` rather than a probability, and `Significant` is false whenever the test did not run. +A verdict of nothing found means nothing at all unless a shift worth finding could have been seen, so the +analysis states what it could have seen. `SignificanceResult.DetectableDifference` is the half width of the +confidence interval for the pair, which is the test read the other way round: instead of asking whether the +shift that happened beats the noise, it asks how large a shift would have to be before it could. It comes +back with the test, worked out in `BuildResult` from the same critical value the probability comes from, so +the readings are walked once and the limit cannot disagree with the test it is quoted beside. Below +`SHIFT_OF_INTEREST` - one part in ten thousand, the order of the effect reported in the published work - the +sessions cannot speak to the question, and the verdict says so in the warning colour rather than reporting a +null result that reads as evidence of absence. + +The three verdicts are different weights, which is what `VerdictWeights` carries. A shift found is notable +whatever the sensitivity, because it cleared the limit by being found at all. Nothing found from sessions +that could have found something is the ordinary outcome. Nothing found from sessions that could not is the +one that misleads, and is the only one that warns. + +Note that the device and the simulator have different noise floors: a device reading averages 262,144 bits +and a simulated one 16,384, so the simulated spread is about four times wider and needs about sixteen times +the readings to pin its mean down as finely. That is why the limit is worked out from each session's own +spread rather than from a count of readings. + The verdict is stated in words under the table rather than left to be read off the numbers, and is emphasised only when there is a shift to notice. `HistogramChart` plots each session as a percentage of its own readings, not as counts: on counts the longer session stands taller in every bin and hides the shift the diff --git a/GeneratorForm.Designer.cs b/GeneratorForm.Designer.cs index 0e1834b..acd104a 100644 --- a/GeneratorForm.Designer.cs +++ b/GeneratorForm.Designer.cs @@ -696,7 +696,7 @@ private void InitializeComponent() this.m_AnalyzeLayout.Name = "m_AnalyzeLayout"; this.m_AnalyzeLayout.RowCount = 3; this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 126F)); - this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 234F)); + this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Absolute, 290F)); this.m_AnalyzeLayout.RowStyles.Add(new System.Windows.Forms.RowStyle(System.Windows.Forms.SizeType.Percent, 100F)); this.m_AnalyzeLayout.Size = new System.Drawing.Size(741, 652); this.m_AnalyzeLayout.TabIndex = 0; @@ -867,7 +867,7 @@ private void InitializeComponent() this.m_VerdictLabel.Location = new System.Drawing.Point(14, 188); this.m_VerdictLabel.Name = "m_VerdictLabel"; this.m_VerdictLabel.Padding = new System.Windows.Forms.Padding(0, 10, 0, 0); - this.m_VerdictLabel.Size = new System.Drawing.Size(713, 42); + this.m_VerdictLabel.Size = new System.Drawing.Size(713, 88); this.m_VerdictLabel.TabIndex = 1; this.m_VerdictLabel.TextAlign = System.Drawing.ContentAlignment.MiddleLeft; // diff --git a/GeneratorForm.cs b/GeneratorForm.cs index 319ec84..33e143d 100644 --- a/GeneratorForm.cs +++ b/GeneratorForm.cs @@ -81,6 +81,18 @@ public enum RngGuiStates RNG_GUI_STATES_SIZE // Keep at end }; + /// + /// How much attention a verdict is worth. Nothing found is the ordinary outcome of the experiment + /// and is not worth shouting about; a shift is worth noticing; and a session that could not have + /// shown the shift it was looking for is worth warning about, because its silence means nothing. + /// + public enum VerdictWeights + { + Ordinary = 0, + Notable = 1, + Warning = 2, + }; + #endregion #region Constructors @@ -2450,32 +2462,62 @@ private void UpdateComparisonVerdict(List baselineReadings, List bool bBothLoaded = ((null != baselineReadings) && (null != resultReadings)); if (false == bBothLoaded) { - SetVerdict(m_sVERDICT_NEEDS_BOTH, false); + SetVerdict(m_sVERDICT_NEEDS_BOTH, VerdictWeights.Ordinary); return; } SignificanceResult test = SignificanceTest.CompareMeans(baselineReadings, resultReadings); if (false == test.Valid) { - SetVerdict(m_sVERDICT_NOT_ENOUGH_DATA, false); + SetVerdict(m_sVERDICT_NOT_ENOUGH_DATA, VerdictWeights.Warning); return; } double fShift = (Mean(resultReadings) - Mean(baselineReadings)); string sDirection = (0 <= fShift) ? m_sDIRECTION_HIGHER : m_sDIRECTION_LOWER; string sProbability = FormatProbability(test); + string sFreedom = test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT); + string sShift = Math.Abs(fShift).ToString(m_sVALUE_FORMAT); + + // How finely the two sessions between them pin a difference down. A verdict of nothing found + // means nothing at all unless a shift worth finding could have been seen, so the limit is + // stated either way rather than left for the reader to work out from the reading counts. It + // comes back with the test rather than being asked for separately, so the readings are walked + // once and the limit cannot disagree with the test it is quoted beside. + double fDetectable = test.DetectableDifference; + bool bSensitive = SignificanceTest.IsSensitiveEnough(fDetectable); + string sDetectable = fDetectable.ToString(m_sVALUE_FORMAT); + string sOfInterest = SignificanceTest.SHIFT_OF_INTEREST.ToString(m_sVALUE_FORMAT); if (true == test.Significant) { - SetVerdict($"Significant shift: the result sits {sDirection} than the baseline by " + - $"{Math.Abs(fShift).ToString(m_sVALUE_FORMAT)}. A shift this large would arise by " + - $"chance {sProbability} of the time (Welch's t, two-tailed, {test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT)} df).", true); + // The shift cleared the limit by having been found at all, so this reports the limit rather + // than warning about it, however short the sessions were + SetVerdict($"Significant shift: the result sits {sDirection} than the baseline by {sShift}. " + + $"A shift this large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions can show a difference of " + + $"{sDetectable} or larger.", VerdictWeights.Notable); + } + else if (true == bSensitive) + { + SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by {sShift}, " + + $"but a shift that large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions can show a difference of " + + $"{sDetectable} or larger, so a shift of {sOfInterest} would have been found.", + VerdictWeights.Ordinary); } else { - SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by " + - $"{Math.Abs(fShift).ToString(m_sVALUE_FORMAT)}, but a shift that large would arise by " + - $"chance {sProbability} of the time (Welch's t, two-tailed, {test.DegreesOfFreedom.ToString(m_sMOMENT_FORMAT)} df).", false); + // Nothing was found and nothing could have been. This is the reading that misleads if it is + // left to stand on its own, so it is the one that carries the warning. The measurement is + // still reported in full first: the warning is about what the numbers can be taken to mean, + // not a reason to stop showing them. + SetVerdict($"No significant shift. The result sits {sDirection} than the baseline by {sShift}, " + + $"but a shift that large would arise by chance {sProbability} of the time " + + $"(Welch's t, two-tailed, {sFreedom} df). These sessions are too short to conclude " + + $"anything from that: they can only show a difference of {sDetectable} or larger, so " + + $"a shift of {sOfInterest}, the size this looks for, could be real here and still " + + $"never reach significance. Record for longer.", VerdictWeights.Warning); } } @@ -2484,12 +2526,32 @@ private void UpdateComparisonVerdict(List baselineReadings, List /// ordinary outcome and is not worth shouting about. /// /// IN - The wording to show - /// IN - Whether the verdict reports a shift worth noticing - private void SetVerdict(string sVerdict, bool bSignificant) + /// IN - How much attention the verdict is worth + private void SetVerdict(string sVerdict, VerdictWeights weight) { m_VerdictLabel.Text = sVerdict; - m_VerdictLabel.Font = bSignificant ? m_VerdictFontBold : m_VerdictFontRegular; - m_VerdictLabel.ForeColor = bSignificant ? UiPalette.CardText : UiPalette.MutedText; + + // Anything other than the ordinary outcome is set in the heavier face, so a verdict that needs + // reading is not the same weight as the one that says nothing happened + bool bOrdinary = (VerdictWeights.Ordinary == weight); + m_VerdictLabel.Font = bOrdinary ? m_VerdictFontRegular : m_VerdictFontBold; + + // A warning takes the severity colour, which stands aside under a high contrast scheme and + // leaves the wording to carry it + switch (weight) + { + case VerdictWeights.Warning: + m_VerdictLabel.ForeColor = StatusPalette.WarningText; + break; + + case VerdictWeights.Notable: + m_VerdictLabel.ForeColor = UiPalette.CardText; + break; + + default: + m_VerdictLabel.ForeColor = UiPalette.MutedText; + break; + } } /// diff --git a/README.md b/README.md index c86af0b..158f165 100644 --- a/README.md +++ b/README.md @@ -35,6 +35,19 @@ directly: mean, standard deviation, skewness and kurtosis for each, the differen a histogram of the two distributions overlaid. Statistics are calculated with [Math.NET Numerics](https://numerics.mathdotnet.com/). +### Enough readings to answer + +A session that found nothing has only said something if it could have found something. The verdict therefore +states the smallest difference the two sessions could show as significant, and warns when that is coarser +than the shift being looked for — one part in ten thousand, a generator running at 0.5001 rather than 0.5, +which is the order of the effect reported in the published work. + +A minute or so of recording from the device is enough to speak to a shift that size. Below that, a real +shift can sit in the readings and never reach significance, and a verdict of *no significant shift* would +be reporting the length of the session rather than anything about the generator. The limit is worked out +from each session's own spread rather than from a count of readings, because a simulated reading averages +far fewer bits than a device one and is correspondingly noisier. + ### Targets A session can record a target value of `0` or `1` — the outcome the operator is attempting to influence — diff --git a/SignificanceTest.cs b/SignificanceTest.cs index 671d8d0..db79eae 100644 --- a/SignificanceTest.cs +++ b/SignificanceTest.cs @@ -47,6 +47,17 @@ public struct SignificanceResult /// public double Probability { get; set; } + /// + /// The smallest difference this test could have shown as significant, or NaN when there was too + /// little to test. A test that found nothing has said something only if it could have found + /// something, and this is what it could have found: below it a difference can be entirely real and + /// still not reach significance. + /// NOTE: This is the half width of the confidence interval, which is the same arithmetic the test + /// itself does, read the other way round. It comes back with the test rather than being worked out + /// separately so that the two cannot drift apart, and so the readings are walked once. + /// + public double DetectableDifference { get; set; } + /// /// Whether the difference is significant at the level the application reports against (read-only) /// @@ -99,7 +110,7 @@ public static SignificanceResult CompareMeans(IList baseline, IList baseline, IList= fCombinedError) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fStatistic = ((fResultMean - fBaselineMean) / Math.Sqrt(fCombinedError)); @@ -122,7 +133,7 @@ public static SignificanceResult CompareMeans(IList baseline, IList @@ -142,18 +153,41 @@ public static SignificanceResult CompareWithExpected(IList readings, dou if (m_iMINIMUM_READINGS > readings.Count) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fMean = Mean(readings); double fStandardError = (Variance(readings, fMean) / readings.Count); if (m_fMINIMUM_ERROR >= fStandardError) { - return new SignificanceResult { Valid = false }; + return NoResult(); } double fStatistic = ((fMean - fExpectedMean) / Math.Sqrt(fStandardError)); - return BuildResult(fStatistic, (readings.Count - 1)); + return BuildResult(fStatistic, (readings.Count - 1), Math.Sqrt(fStandardError)); + } + + /// + /// Whether readings pin their mean down finely enough to speak to the shift being looked for + /// + /// IN - The smallest shift that could reach significance + /// true if a shift of the size being looked for could be shown; otherwise, false + public static bool IsSensitiveEnough(double fDetectableShift) + { + return ((false == double.IsNaN(fDetectableShift)) && (SHIFT_OF_INTEREST >= fDetectableShift)); + } + + /// + /// The outcome for readings there was too little of to test + /// + /// An outcome reporting itself invalid, with no answer for what it could have found + private static SignificanceResult NoResult() + { + // The detectable difference is set to no answer rather than left at zero. Zero would read as + // "a difference of any size would have been found", which is the opposite of what having too + // little to test means, and anything asking whether the readings are sensitive enough would + // agree with it. + return new SignificanceResult { Valid = false, DetectableDifference = double.NaN }; } /// @@ -162,8 +196,9 @@ public static SignificanceResult CompareWithExpected(IList readings, dou /// /// IN - The size of the difference in standard errors /// IN - The degrees of freedom of the test + /// IN - The standard error the statistic was measured against /// The completed outcome - private static SignificanceResult BuildResult(double fStatistic, double fFreedom) + private static SignificanceResult BuildResult(double fStatistic, double fFreedom, double fStandardError) { // A statistic that is not a number says the arithmetic ran out of meaning rather than that the // difference was enormous, so it is reported as no test rather than as a certainty @@ -171,19 +206,25 @@ private static SignificanceResult BuildResult(double fStatistic, double fFreedom (false == double.IsNaN(fFreedom)) && (0 < fFreedom)); if (false == bUsable) { - return new SignificanceResult { Valid = false }; + return NoResult(); } // Both tails, so a shift in either direction counts against the null double fUpperTail = (1.0 - StudentT.CDF(0.0, 1.0, fFreedom, Math.Abs(fStatistic))); double fProbability = Math.Min(1.0, (2.0 * fUpperTail)); + // How large a difference would have had to be to reach significance here. The critical value is + // taken from the same distribution the probability came from, so the two agree by construction + // rather than by two pieces of arithmetic happening to match. + double fCritical = StudentT.InvCDF(0.0, 1.0, fFreedom, (1.0 - (SignificanceResult.SIGNIFICANCE_LEVEL / 2.0))); + return new SignificanceResult { Valid = true, Statistic = fStatistic, DegreesOfFreedom = fFreedom, - Probability = fProbability + Probability = fProbability, + DetectableDifference = (fCritical * fStandardError) }; } @@ -222,6 +263,16 @@ private static double Variance(IList readings, double fMean) #endregion #region Constants + /// + /// The size of shift a session is expected to be able to speak to. Readings that cannot pin their + /// mean down at least this finely are reported as too few, because a shift of this size could be + /// entirely real in them and still not reach significance. + /// NOTE: This is the order of the effect reported in the published work on influencing a generator: + /// around one part in ten thousand, a generator running at 0.5001 rather than 0.5. It is the figure + /// the wording in the interface quotes, so the two have to be changed together. + /// + public const double SHIFT_OF_INTEREST = 0.0001; + // A spread cannot be measured from fewer than two readings private const int m_iMINIMUM_READINGS = 2; diff --git a/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs b/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs index 7b1782a..879d72d 100644 --- a/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs +++ b/tests/RandomNumberGenerator.Test/DeviceUpdateThread.Test.cs @@ -644,17 +644,6 @@ public void ThreadProc_InfoBoxChangedDuringUpdate_NewerMessageKept() mockParent.Verify(mock => mock.SetStatusBoxState(m_sTEST_STATUS_TEXT, It.IsAny(), It.IsAny()), Times.Never); } - #endregion - #region Helper Types - - /// - /// Delegate for GetStatusBoxState callback to handle out parameters in Moq - /// - /// OUT - Status box text - /// OUT - Status box text color - /// OUT - Status box background color - private delegate void GetStatusBoxStateCallback(out string text, out Color textColor, out Color backColor); - /// /// Tests the device status colours are taken from the palette when they are used rather than held /// from when the class was first touched. The palette reports the scheme in force now, so anything @@ -734,6 +723,17 @@ public void StatusPalette_SeverityColours_AreReadEachTimeRatherThanFixed() } } + #endregion + #region Helper Types + + /// + /// Delegate for GetStatusBoxState callback to handle out parameters in Moq + /// + /// OUT - Status box text + /// OUT - Status box text color + /// OUT - Status box background color + private delegate void GetStatusBoxStateCallback(out string text, out Color textColor, out Color backColor); + #endregion } -} \ No newline at end of file +} diff --git a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs index 294145f..b05b2b7 100644 --- a/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs +++ b/tests/RandomNumberGenerator.Test/SignificanceTest.Test.cs @@ -352,6 +352,310 @@ private static List BuildReadings(Random source, int iCount, double fMea return readings; } + /// + /// Tests the smallest detectable shift agrees with the test it is derived from. A shift of exactly + /// that size, applied to the readings, should sit right on the edge of significance: this is the + /// same arithmetic read the other way round, so the two have to meet. + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_ShiftOfThatSize_SitsOnTheEdgeOfSignificance() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fEXPECTED_MEAN = 0.5; + const double fEDGE_TOLERANCE = 0.002; + + // Readings with a spread, sitting on the expected value so the shift applied below is the whole + // of the difference the test sees + List readings = MakeSpreadReadings(200, fEXPECTED_MEAN, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fDetectable = SignificanceTest.CompareWithExpected(readings, fEXPECTED_MEAN).DetectableDifference; + + // Move every reading by exactly that much and ask the test what it makes of it + List shifted = new List(); + foreach (double fReading in readings) + { + shifted.Add(fReading + fDetectable); + } + SignificanceResult onTheEdge = SignificanceTest.CompareWithExpected(shifted, fEXPECTED_MEAN); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify a shift of exactly the detectable size lands on the significance level rather than + // somewhere unrelated to it + Assert.IsTrue(onTheEdge.Valid); + Assert.AreEqual(SignificanceResult.SIGNIFICANCE_LEVEL, onTheEdge.Probability, fEDGE_TOLERANCE, + $"A shift of {fDetectable} gave a probability of {onTheEdge.Probability}"); + } + + /// + /// Tests a shift smaller than the detectable size cannot reach significance, which is the whole + /// point of warning about it + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_SmallerShift_CannotReachSignificance() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fEXPECTED_MEAN = 0.5; + const double fWELL_UNDER = 0.5; + + List readings = MakeSpreadReadings(200, fEXPECTED_MEAN, 0.001); + double fDetectable = SignificanceTest.CompareWithExpected(readings, fEXPECTED_MEAN).DetectableDifference; + + //**************************************************************// + // Act + //**************************************************************// + + // Half the shift the readings could show + List shifted = new List(); + foreach (double fReading in readings) + { + shifted.Add(fReading + (fDetectable * fWELL_UNDER)); + } + SignificanceResult tooSmall = SignificanceTest.CompareWithExpected(shifted, fEXPECTED_MEAN); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify a real shift of that size goes unfound, which is what the warning exists to say + Assert.IsTrue(tooSmall.Valid); + Assert.IsFalse(tooSmall.Significant, + $"A shift of half the detectable size reached significance at {tooSmall.Probability}"); + } + + /// + /// Tests more readings pin the mean down more finely, at the rate the arithmetic says they should + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_MoreReadings_PinsTheMeanDownMoreFinely() + { + //**************************************************************// + // Arrange + //**************************************************************// + + // Four times the readings should halve the detectable shift, as it falls with the root of the + // count. The tolerance is loose because the critical value also moves with the count. + const double fEXPECTED_RATIO = 2.0; + const double fRATIO_TOLERANCE = 0.15; + + List few = MakeSpreadReadings(100, 0.5, 0.001); + List many = MakeSpreadReadings(400, 0.5, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fFew = SignificanceTest.CompareWithExpected(few, 0.5).DetectableDifference; + double fMany = SignificanceTest.CompareWithExpected(many, 0.5).DetectableDifference; + + //**************************************************************// + // Assert + //**************************************************************// + + Assert.IsTrue(fMany < fFew, "More readings should pin the mean down more finely"); + Assert.AreEqual(fEXPECTED_RATIO, (fFew / fMany), fRATIO_TOLERANCE, + $"Four times the readings gave a ratio of {(fFew / fMany)}"); + } + + /// + /// Tests readings too few to measure a spread report no detectable shift rather than a number that + /// looks like one + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_TooFewReadings_ReportsNoAnswer() + { + //**************************************************************// + // Arrange + //**************************************************************// + + List single = new List { 0.5 }; + List empty = new List(); + List identical = new List { 0.5, 0.5, 0.5, 0.5 }; + + //**************************************************************// + // Act & Assert + //**************************************************************// + + // Verify each reports no answer rather than zero, which would read as "any shift is detectable" + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(single, 0.5).DetectableDifference)); + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(empty, 0.5).DetectableDifference)); + Assert.IsTrue(double.IsNaN(SignificanceTest.CompareWithExpected(identical, 0.5).DetectableDifference)); + } + + /// + /// Tests a session that cannot show the shift being looked for is reported as not sensitive enough, + /// and one that can is not + /// + [TestMethod] + [TestCategory("Component")] + public void IsSensitiveEnough_AgainstTheShiftOfInterest_AnswersEitherWay() + { + //**************************************************************// + // Arrange + //**************************************************************// + + const double fJUST_INSIDE = 0.9; + const double fJUST_OUTSIDE = 1.1; + + //**************************************************************// + // Act & Assert + //**************************************************************// + + // Verify the answer turns on the shift being looked for rather than on a reading count + Assert.IsTrue(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST * fJUST_INSIDE)); + Assert.IsTrue(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST)); + Assert.IsFalse(SignificanceTest.IsSensitiveEnough(SignificanceTest.SHIFT_OF_INTEREST * fJUST_OUTSIDE)); + + // Verify no answer is not mistaken for a good one + Assert.IsFalse(SignificanceTest.IsSensitiveEnough(double.NaN)); + } + + /// + /// Tests a short device session cannot show the shift being looked for while a longer one can, using + /// the spread a real device actually produces. This is the case the warning was added for. + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_RealDeviceSpread_TurnsOverAtTheExpectedLength() + { + //**************************************************************// + // Arrange + //**************************************************************// + + // Each device reading is the average of 32768 bytes, so its spread is about 0.5 over the root + // of that many bits. A minute of recording is roughly 550 readings. + const double fDEVICE_SPREAD = 0.00098; + const int iTEN_SECONDS = 90; + const int iTWO_MINUTES = 1100; + + List shortSession = MakeSpreadReadings(iTEN_SECONDS, 0.5, fDEVICE_SPREAD); + List longSession = MakeSpreadReadings(iTWO_MINUTES, 0.5, fDEVICE_SPREAD); + + //**************************************************************// + // Act + //**************************************************************// + + bool bShortEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.CompareWithExpected(shortSession, 0.5).DetectableDifference); + bool bLongEnough = SignificanceTest.IsSensitiveEnough(SignificanceTest.CompareWithExpected(longSession, 0.5).DetectableDifference); + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify ten seconds of device readings cannot speak to the shift and two minutes can + Assert.IsFalse(bShortEnough, "Ten seconds of readings should not be enough"); + Assert.IsTrue(bLongEnough, "Two minutes of readings should be enough"); + } + + /// + /// Tests the difference two sessions can show between them is reported, and that it is coarser than + /// what either could show on its own - the noise of both stands between them + /// + [TestMethod] + [TestCategory("Component")] + public void DetectableDifference_TwoSessions_IsCoarserThanEitherAlone() + { + //**************************************************************// + // Arrange + //**************************************************************// + + List baseline = MakeSpreadReadings(300, 0.5, 0.001); + List result = MakeSpreadReadings(300, 0.5, 0.001); + + //**************************************************************// + // Act + //**************************************************************// + + double fBaselineAlone = SignificanceTest.CompareWithExpected(baseline, 0.5).DetectableDifference; + double fBetween = SignificanceTest.CompareMeans(baseline, result).DetectableDifference; + + //**************************************************************// + // Assert + //**************************************************************// + + // Verify comparing two sessions is harder than measuring one against a fixed value, because both + // sides carry noise + Assert.IsFalse(double.IsNaN(fBetween)); + Assert.IsTrue(fBetween > fBaselineAlone, + $"Between them: {fBetween}, baseline alone: {fBaselineAlone}"); + } + + /// + /// Tests a null set of readings is refused rather than dereferenced + /// + [TestMethod] + [TestCategory("Component")] + public void CompareWithExpected_NullReadings_ThrowsArgumentNullException() + { + //**************************************************************// + // Act & Assert + //**************************************************************// + + Assert.ThrowsException( + () => SignificanceTest.CompareWithExpected(null, m_fEXPECTED_MEAN)); + } + + /// + /// Tests a null result is refused rather than dereferenced. The baseline side is covered above; this + /// is the other one. + /// + [TestMethod] + [TestCategory("Component")] + public void CompareMeans_NullResult_ThrowsArgumentNullException() + { + //**************************************************************// + // Arrange + //**************************************************************// + + List readings = new List { 0.5, 0.6 }; + + //**************************************************************// + // Act & Assert + //**************************************************************// + + Assert.ThrowsException(() => SignificanceTest.CompareMeans(readings, null)); + } + + /// + /// Builds readings sitting a fixed distance either side of a mean, alternating, so their spread is + /// settled rather than whatever a random draw happened to give. + /// NOTE: fOffset is how far each reading sits from the mean, not the sample standard deviation the + /// readings end up with. Those are close but not equal: the sample deviation divides by one fewer + /// than the count, so it comes out slightly the larger of the two. + /// + /// IN - How many readings to make + /// IN - The value to centre them on + /// IN - How far each reading sits either side of the mean + /// The readings + private static List MakeSpreadReadings(int iCount, double fMean, double fOffset) + { + List readings = new List(); + for (int iIndex = 0; iIndex < iCount; ++iIndex) + { + // Half above and half below, so the mean lands where it was asked to and the spread is the + // offset itself + double fStep = ((0 == (iIndex % 2)) ? fOffset : -fOffset); + readings.Add(fMean + fStep); + } + return readings; + } + #endregion #region Constants diff --git a/tests/manual/README.md b/tests/manual/README.md index 7fe0fe4..84dc049 100644 --- a/tests/manual/README.md +++ b/tests/manual/README.md @@ -39,6 +39,7 @@ already does. | `test-analyze.ps1` | no | The Analyse tab: the comparison table, the significance verdict and its wording, the histogram, and recovering a session file left unterminated by killing the application mid-write. | | `check-chart.ps1` | no | That the chart and the statistics beside it agree on every path — recording, a second session into the same file, a different file being chosen, an existing file being reopened, and Clear. | | `inproc-length.ps1` | no | The session length, and the data file surviving a session ending. Drives the form directly, so it covers the state machine rather than the pixels. | +| `check-sensitivity.ps1` | no | That the verdict states what shift the two sessions could show, and warns when nothing was found because nothing could have been. | | `check-appended.ps1` | no | That a file grown by a second session reads back correctly on the Record tab and analyses correctly on the Analyse tab. | | `check-device.ps1` | **yes** | Device discovery, the native reads, a session recorded from the hardware, a second session appended to it, a timed session, and the recorded file analysing. Also that a port with nothing on it is refused. | | `check-reinit.ps1` | **yes** | Switching between the device and the simulator and back, which tears down the native interface and builds a new one each time. | diff --git a/tests/manual/check-sensitivity.ps1 b/tests/manual/check-sensitivity.ps1 new file mode 100644 index 0000000..a06f749 --- /dev/null +++ b/tests/manual/check-sensitivity.ps1 @@ -0,0 +1,124 @@ +######################################################################################################################### +# File Name: check-sensitivity.ps1 +# Description: Checks the analysis warns when sessions are too short to show the shift being looked for +# +# Copyright (c) 2026 Mike Pullen +# Licensed under the MIT License. See LICENSE in the repository root. +# +# Revision History: +#====================================================================================================================== +# 2026/09/09 - Mike Pullen - Original implementation. +######################################################################################################################### +# Drives the analysis tab through all three verdicts: a shift found, nothing found from sessions long +# enough to have found one, and nothing found from sessions too short to have found anything. The third is +# the one the warning exists for. +$ErrorActionPreference = "Continue" +Add-Type -AssemblyName System.Windows.Forms +try { + [System.Windows.Forms.Application]::EnableVisualStyles() + [System.Windows.Forms.Application]::SetCompatibleTextRenderingDefault($false) +} catch { } +# Which build to drive. Set RNG_BIN to test a different one. +if ([string]::IsNullOrEmpty($env:RNG_BIN)) { $env:RNG_BIN = (Resolve-Path (Join-Path $PSScriptRoot "..\..\bin\Release")).Path } +$binDir = $env:RNG_BIN +[System.Reflection.Assembly]::LoadFrom((Join-Path $binDir "Random Number Generator.exe")) | Out-Null +[Environment]::CurrentDirectory = $binDir + +$WorkRoot = Join-Path $PSScriptRoot "work" +if (-not (Test-Path $WorkRoot)) { $null = New-Item -ItemType Directory -Path $WorkRoot } + +$script:pass = 0; $script:fail = 0 +function Check($name, $condition, $detail) { + if ($condition) { $script:pass++; Write-Host (" PASS " + $name + " " + $detail) } + else { $script:fail++; Write-Host (" FAIL " + $name + " " + $detail) } +} + +# A session file with a known count, centre and spread. Readings alternate either side of the centre so the +# spread is exactly what was asked for, which makes the detectable shift predictable. +function WriteSession($path, $count, $centre, $spread) { + $sb = New-Object System.Text.StringBuilder + [void]$sb.AppendLine('') + [void]$sb.AppendLine('') + for ($i = 0; $i -lt $count; $i++) { + $offset = if (0 -eq ($i % 2)) { $spread } else { -$spread } + $v = ($centre + $offset).ToString("F12", [System.Globalization.CultureInfo]::InvariantCulture) + [void]$sb.AppendLine("`t$v") + } + [void]$sb.Append('') + [System.IO.File]::WriteAllText($path, $sb.ToString()) +} + +# The spread a real device reading has: the average of 32768 bytes of bits +$DEVICE_SPREAD = 0.00098 + +$ws = New-Object System.Xml.XmlWriterSettings +$writer = New-Object RandomNumberGenerator.RNGXMLWriter($ws) +$reader = New-Object RandomNumberGenerator.RNGXMLReader +$dataFile= New-Object RandomNumberGenerator.RNGSessionDataFile($writer, $reader) +$timer = New-Object RandomNumberGenerator.RNGSessionTimer +$data = New-Object RandomNumberGenerator.RNGSessionData($dataFile, $timer) +$device = New-Object RandomNumberGenerator.RNGDeviceTimer +$form = New-Object RandomNumberGenerator.GeneratorForm($data, $device) +$BF = [System.Reflection.BindingFlags]"NonPublic,Instance" +function Fld($name) { return $form.GetType().GetField($name, $BF).GetValue($form) } +function Call1($name, $arg) { return $form.GetType().GetMethod($name, $BF).Invoke($form, [object[]]@([string]$arg)) } +function Call2($name, $a, $b) { return $form.GetType().GetMethod($name, $BF).Invoke($form, [object[]]@([string]$a, [bool]$b)) } +function Verdict() { return (Fld "m_VerdictLabel").Text } +function VerdictColour() { return (Fld "m_VerdictLabel").ForeColor } + +$form.Show(); [System.Windows.Forms.Application]::DoEvents() + +function LoadPair($baseline, $result) { + $b = Call1 "LoadBaselineFile" $baseline + Call2 "OnBaselineLoadCompleted" $baseline $b | Out-Null + $r = Call1 "LoadResultFile" $result + Call2 "OnResultLoadCompleted" $result $r | Out-Null + [System.Windows.Forms.Application]::DoEvents() +} + +Write-Host "=== 1. TOO SHORT TO SAY ANYTHING - the case the warning exists for ===" +$shortA = Join-Path $WorkRoot "sens-short-a.rng" +$shortB = Join-Path $WorkRoot "sens-short-b.rng" +WriteSession $shortA 90 0.5 $DEVICE_SPREAD +WriteSession $shortB 90 0.5 $DEVICE_SPREAD +LoadPair $shortA $shortB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "warns that the sessions are too short" ($v -match "too short to conclude") "" +Check "states the limit" ($v -match "can only show a difference of") "" +Check "names the shift being looked for" ($v -match "0.000100") "" +Check "tells the user what to do" ($v -match "Record for longer") "" +Check "coloured as a warning" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.StatusPalette]::WarningText).ToArgb()) ("colour=" + (VerdictColour)) + +Write-Host "" +Write-Host "=== 2. LONG ENOUGH, NOTHING FOUND - the ordinary outcome ===" +$longA = Join-Path $WorkRoot "sens-long-a.rng" +$longB = Join-Path $WorkRoot "sens-long-b.rng" +WriteSession $longA 1500 0.5 $DEVICE_SPREAD +WriteSession $longB 1500 0.5 $DEVICE_SPREAD +LoadPair $longA $longB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "reports no significant shift" ($v -match "No significant shift\.") "" +Check "does not warn" (-not ($v -match "too short")) "" +Check "says the shift would have been found" ($v -match "would have been found") "" +Check "coloured as ordinary" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.UiPalette]::MutedText).ToArgb()) ("colour=" + (VerdictColour)) + +Write-Host "" +Write-Host "=== 3. A SHIFT FOUND - the limit is reported, not warned about ===" +$shiftA = Join-Path $WorkRoot "sens-shift-a.rng" +$shiftB = Join-Path $WorkRoot "sens-shift-b.rng" +WriteSession $shiftA 1500 0.5 $DEVICE_SPREAD +WriteSession $shiftB 1500 0.5005 $DEVICE_SPREAD +LoadPair $shiftA $shiftB +$v = Verdict +Write-Host (" verdict: " + $v) +Check "reports a significant shift" ($v -match "Significant shift:") "" +Check "still states the limit" ($v -match "can show a difference of") "" +Check "does not warn" (-not ($v -match "too short")) "" +Check "coloured as notable" ((VerdictColour).ToArgb() -eq ([RandomNumberGenerator.UiPalette]::CardText).ToArgb()) ("colour=" + (VerdictColour)) + +$form.Close() +[System.Windows.Forms.Application]::DoEvents() +Write-Host "" +Write-Host ("PASS " + $script:pass + " FAIL " + $script:fail) diff --git a/tests/manual/run-all.ps1 b/tests/manual/run-all.ps1 index cdd8363..86f49bb 100644 --- a/tests/manual/run-all.ps1 +++ b/tests/manual/run-all.ps1 @@ -37,6 +37,7 @@ $suites = @( "check-chart.ps1", "inproc-length.ps1", "check-appended.ps1", + "check-sensitivity.ps1", "check-device.ps1", "check-reinit.ps1", "check-nodevice.ps1"