diff --git a/Cargo.lock b/Cargo.lock index 9716d6b3..727a0ac8 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1117,9 +1117,9 @@ dependencies = [ [[package]] name = "aws-lc-rs" -version = "1.16.2" +version = "1.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a054912289d18629dc78375ba2c3726a3afe3ff71b4edba9dedfca0e3446d1fc" +checksum = "b281d307588d634de920874890732659e2e7672f72b5e10e81badc1a8a83621e" dependencies = [ "aws-lc-sys", "zeroize", @@ -1127,14 +1127,15 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.39.1" +version = "0.45.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83a25cf98105baa966497416dbd42565ce3a8cf8dbfd59803ec9ad46f3126399" +checksum = "9bff6c3b54fad79a2e60b8102caf565819711497c1f5f092f49508e2f5c31b27" dependencies = [ "cc", "cmake", "dunce", "fs_extra", + "pkg-config", ] [[package]] @@ -4925,11 +4926,12 @@ checksum = "2195bf6aa996a481483b29d62a7663eed3fe39600c460e323f8ff41e90bdd89b" [[package]] name = "mzannotate" version = "0.2.0" -source = "git+https://github.com/jspaezp/rustyms?rev=142c96ed1d31c56936c003af5f599d0c8f740005#142c96ed1d31c56936c003af5f599d0c8f740005" +source = "git+https://github.com/jspaezp/rustyms?rev=c033092561166e1b9960fa2a6678299af52a94e1#c033092561166e1b9960fa2a6678299af52a94e1" dependencies = [ "context_error", "indexmap", - "itertools 0.14.0", + "itertools 0.13.0", + "memchr", "mzcore", "mzcv", "mzdata", @@ -4944,12 +4946,12 @@ dependencies = [ [[package]] name = "mzcore" version = "0.2.0" -source = "git+https://github.com/jspaezp/rustyms?rev=142c96ed1d31c56936c003af5f599d0c8f740005#142c96ed1d31c56936c003af5f599d0c8f740005" +source = "git+https://github.com/jspaezp/rustyms?rev=c033092561166e1b9960fa2a6678299af52a94e1#c033092561166e1b9960fa2a6678299af52a94e1" dependencies = [ "bincode", "context_error", "flate2", - "itertools 0.14.0", + "itertools 0.13.0", "mzcv", "ordered-float 5.3.0", "roxmltree", @@ -4963,7 +4965,7 @@ dependencies = [ [[package]] name = "mzcv" version = "0.3.0" -source = "git+https://github.com/jspaezp/rustyms?rev=142c96ed1d31c56936c003af5f599d0c8f740005#142c96ed1d31c56936c003af5f599d0c8f740005" +source = "git+https://github.com/jspaezp/rustyms?rev=c033092561166e1b9960fa2a6678299af52a94e1#c033092561166e1b9960fa2a6678299af52a94e1" dependencies = [ "bincode", "chrono", @@ -7026,9 +7028,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.37" +version = "0.23.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "758025cb5fccfd3bc2fd74708fd4682be41d99e5dff73c377c0646c6012c73a4" +checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" dependencies = [ "aws-lc-rs", "once_cell", @@ -7072,9 +7074,9 @@ dependencies = [ [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "aws-lc-rs", "ring", @@ -8187,6 +8189,7 @@ dependencies = [ "array2d", "arrow 59.2.0", "bon", + "calibrt", "csv", "flate2", "half", diff --git a/Cargo.toml b/Cargo.toml index 6d622b81..fcde6247 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -77,18 +77,12 @@ insta = { version = "1.34.0" } bon = "3.8.1" tinyvec = { features = ["alloc", "serde"], version = "1.10.0" } smallvec = { version = "1.13", features = ["const_generics", "union"] } -# `mzannotate` carries the mzSpecLib reader. Two fixes are not in a release yet: -# named attribute sets were never applied, so every decoy in a target/decoy -# library read back as a target, and retention time was dropped for any spectrum -# whose writer grouped other attributes with it. `mzcore` and `mzcv` come from -# the same source so the graph holds one copy of each and the ontology types -# unify. -# -# Pinned by revision, not by branch: a branch can be force-pushed or deleted. -# All three become version numbers once the fixes ship. -mzannotate = { git = "https://github.com/jspaezp/rustyms", rev = "142c96ed1d31c56936c003af5f599d0c8f740005" } -mzcore = { git = "https://github.com/jspaezp/rustyms", rev = "142c96ed1d31c56936c003af5f599d0c8f740005" } -mzcv = { git = "https://github.com/jspaezp/rustyms", rev = "142c96ed1d31c56936c003af5f599d0c8f740005" } +# mzSpecLib record API preserves RT terms, grouped units and attribute-set +# provenance before projection (jspaezp/rustyms PR #6). Keep all ontology crates +# on the same reviewed revision so their types unify. +mzannotate = { git = "https://github.com/jspaezp/rustyms", rev = "c033092561166e1b9960fa2a6678299af52a94e1" } +mzcore = { git = "https://github.com/jspaezp/rustyms", rev = "c033092561166e1b9960fa2a6678299af52a94e1" } +mzcv = { git = "https://github.com/jspaezp/rustyms", rev = "c033092561166e1b9960fa2a6678299af52a94e1" } csv = "1.3" tempfile = "3.23.0" diff --git a/docs/development.md b/docs/development.md index f8441048..a8b9168f 100644 --- a/docs/development.md +++ b/docs/development.md @@ -31,6 +31,41 @@ q-value columns (`result_mode=raw` Parquet metadata). It bypasses calibration, rescoring and q-value filtering. Supplied decoys do not by themselves validate a decoy strategy for a new analyte class. +Library RT is declared once per library: seconds, normalized index, unspecified +coordinates, or absent. Measured minutes convert to seconds; normalized indices +retain their values, including zero and negative values. Declared normalized-scale +metadata is preserved from mzSpecLib headers and prediction provenance; records +select the actual axis. Mixed RT availability or +axes are rejected. An RT-free library searches unrestricted on RT, omits RT +prediction/residual features, and retains decoy competition, rescoring and q-values +when decoys are available. RT, m/z and mobility calibrate independently. Without +an RT fit, score-selected apexes still supply m/z and mobility measurements, +without ridge filtering. RT-free prescoring retains bounded candidates across +observed RT bands. Mobility calibration requires a searchable run axis; FAIMS and +absent mobility do not contribute sentinel measurements. An axis without measurements keeps its configured tolerance +in primary and secondary extraction. A failed RT fit never reinterprets an index +as seconds; initial extraction searches unrestricted RT. Secondary queries still +center on detected apex RT and observed mobility. + +Results format 5 includes `library_rt_axis` Parquet metadata. Unavailable library +RT, calibrated RT and RT residuals are NaN; observed apex RT remains seconds. +Calibration format v4 records the input library axis and effective tolerance +enums; an empty RT snapshot means no RT fit, while residuals can still contain +m/z and mobility calibration. Older calibration files +must be regenerated. Explicit `rt_seconds` JSON/Python inputs remain seconds. JSON `rt_axis` is +preserved whether optional precursor/fragment labels are supplied or filled in. +Rust source geometry exposes `LibraryRT` through `Target` and `OwnedTarget`. +`ExtractionQuery` borrows that geometry with a separately resolved acquisition +window. `RtSelection::Centered(ObservedRTSeconds(...))` or `FullRun` resolves +against the acquisition extent once. Collectors copy that window and reuse their +geometry/intensity buffers; they retain no source borrow. Index queries accept +`PeakTolerance` for m/z, mobility and quadrupole only. Isotope extraction shifts +collector buffers directly, without an owned target or library-axis clone. +The viewer searches full-run without an RT fit, including libraries measured in +seconds. Direct-query CLI restricted RT inputs explicitly mean acquisition +seconds; normalized library indices require calibration first. Python target +`rt_seconds` likewise specifies an acquisition coordinate. + `run_report.json` records the resolved `scoring_plan` once for the search library, plus `calibration_scoring_plan` when a separate calibration library is supplied. These are the same plans serialized in Parquet metadata: sequence-operation @@ -49,7 +84,7 @@ intensities, so those reader routes remain extraction-only. Standalone `calib_dash` reads saved `calibration.json`, not a spectral library. [Analyte module documentation](../rust/timsquery/src/chemistry/analyte.rs) documents chemistry storage, reader mappings, -library-wide sequence eligibility, and results format version 4. +library-wide sequence eligibility, and results format version 5. Precursor isotope envelopes retain the three-bin C/S approximation. Scoring finalization includes known modification C/S deltas or an explicitly based diff --git a/python/timsquery_pyo3/src/index.rs b/python/timsquery_pyo3/src/index.rs index c53c4a03..5485be6d 100644 --- a/python/timsquery_pyo3/src/index.rs +++ b/python/timsquery_pyo3/src/index.rs @@ -7,7 +7,6 @@ use timsquery::traits::queriable_data::QueriableData; use timsquery::{ ChromatogramCollector, MzMobilityStatsCollector, - OptionallyRestricted, SpectralCollector, Tolerance, }; @@ -21,28 +20,20 @@ use crate::spectrum::{ use crate::target::PyTarget; use crate::tolerance::PyTolerance; -/// Compute the RT range in milliseconds for a chromatogram query. -/// -/// If the tolerance is restricted, returns the tolerance-derived range. -/// If unrestricted, falls back to the full acquisition RT range from the -/// cycle mapping (giving a chromatogram spanning the entire run). -pub(crate) fn rt_range_ms_for_chromatogram( +/// Python target `rt_seconds` is an explicit acquisition coordinate. +pub(crate) fn resolve_query_rt( tol: &Tolerance, rt_seconds: f32, handle: &IndexedPeaksHandle, -) -> PyResult> { - match tol.rt_range_as_milis(rt_seconds) { - OptionallyRestricted::Restricted(range) => Ok(range), - OptionallyRestricted::Unrestricted => { - let (start, end) = handle.ms1_cycle_mapping().range_milis(); - timsquery::TupleRange::try_new(start, end).map_err(|e| { - PyErr::new::(format!( - "Empty RT range in index: {:?}", - e - )) - }) - } - } +) -> PyResult { + timsquery::ResolvedRt::from_mapping( + timsquery::RtSelection::Centered(timsquery::ObservedRTSeconds(rt_seconds)), + &tol.rt, + handle.ms1_cycle_mapping(), + ) + .map_err(|e| { + PyErr::new::(format!("Invalid query RT: {e:?}")) + }) } /// Resolved tolerances: either one shared or one per query. @@ -133,15 +124,18 @@ impl PyTimsIndex { target: &PyTarget, tolerance: &PyTolerance, ) -> PyResult { - let target = target.inner.clone(); + let target = &target.inner; let tol = &tolerance.inner; - let rt_range_ms = rt_range_ms_for_chromatogram(tol, target.rt_seconds(), &self.handle)?; + let resolved_rt = resolve_query_rt(tol, target.rt_seconds(), &self.handle)?; let ref_rt = self.handle.ms1_cycle_mapping(); - let mut collector = ChromatogramCollector::::new(&target, rt_range_ms, ref_rt) - .map_err(|e| PyErr::new::(format!("{:?}", e)))?; + let mut collector = ChromatogramCollector::::new( + &timsquery::ExtractionQuery::new(target, resolved_rt), + ref_rt, + ) + .map_err(|e| PyErr::new::(format!("{:?}", e)))?; - self.handle.add_query(&mut collector, tol); + self.handle.add_query(&mut collector, &tol.peak_tolerance()); Ok(PyChromatogramResult::new(collector, target.id())) } @@ -161,18 +155,22 @@ impl PyTimsIndex { target: &PyTarget, tolerance: &PyTolerance, ) -> PyResult<()> { - let target = target.inner.clone(); + let target = &target.inner; let tol = &tolerance.inner; - let rt_range_ms = rt_range_ms_for_chromatogram(tol, target.rt_seconds(), &self.handle)?; + let resolved_rt = resolve_query_rt(tol, target.rt_seconds(), &self.handle)?; let ref_rt = self.handle.ms1_cycle_mapping(); result .collector - .try_reset_with(&target, rt_range_ms, ref_rt) + .try_reset_with( + &timsquery::ExtractionQuery::new(target, resolved_rt), + ref_rt, + ) .map_err(|e| PyErr::new::(format!("{:?}", e)))?; result.source_id.set_from(target.id()); - self.handle.add_query(&mut result.collector, tol); + self.handle + .add_query(&mut result.collector, &tol.peak_tolerance()); Ok(()) } @@ -189,10 +187,13 @@ impl PyTimsIndex { target: &PyTarget, tolerance: &PyTolerance, ) -> PyResult { - let target = target.inner.clone(); + let target = &target.inner; let tol = &tolerance.inner; - let mut collector = SpectralCollector::::new(&target); - self.handle.add_query(&mut collector, tol); + let mut collector = SpectralCollector::::new(&timsquery::ExtractionQuery::new( + target, + resolve_query_rt(tol, target.rt_seconds(), &self.handle)?, + )); + self.handle.add_query(&mut collector, &tol.peak_tolerance()); Ok(PySpectralResult::new(collector, target.id())) } @@ -209,10 +210,15 @@ impl PyTimsIndex { target: &PyTarget, tolerance: &PyTolerance, ) -> PyResult { - let target = target.inner.clone(); + let target = &target.inner; let tol = &tolerance.inner; - let mut collector = SpectralCollector::::new(&target); - self.handle.add_query(&mut collector, tol); + let mut collector = SpectralCollector::::new( + &timsquery::ExtractionQuery::new( + target, + resolve_query_rt(tol, target.rt_seconds(), &self.handle)?, + ), + ); + self.handle.add_query(&mut collector, &tol.peak_tolerance()); Ok(PyMzMobilityResult::new(collector, target.id())) } @@ -239,13 +245,14 @@ impl PyTimsIndex { .iter() .enumerate() .map(|(i, target)| { - let inner = target.inner.clone(); + let inner = &target.inner; let tol = tolerances.get(i); - let rt_range_ms = - rt_range_ms_for_chromatogram(tol, inner.rt_seconds(), &self.handle)?; - ChromatogramCollector::::new(&inner, rt_range_ms, ref_rt).map_err(|e| { - PyErr::new::(format!("{:?}", e)) - }) + let resolved_rt = resolve_query_rt(tol, inner.rt_seconds(), &self.handle)?; + ChromatogramCollector::::new( + &timsquery::ExtractionQuery::new(inner, resolved_rt), + ref_rt, + ) + .map_err(|e| PyErr::new::(format!("{:?}", e))) }) .collect::>>()?; @@ -253,10 +260,19 @@ impl PyTimsIndex { let handle = &*self.handle; py.detach(|| match &tolerances { ResolvedTolerances::Single(tol) => { - handle.par_add_query_multi(&mut collectors[..], rayon::iter::repeat_n(tol, n)); + handle.par_add_query_multi( + &mut collectors[..], + rayon::iter::repeat_n(&tol.peak_tolerance(), n), + ); } ResolvedTolerances::PerQuery(tols) => { - handle.par_add_query_multi(&mut collectors[..], &tols[..]); + handle.par_add_query_multi( + &mut collectors[..], + &tols + .iter() + .map(Tolerance::peak_tolerance) + .collect::>()[..], + ); } }); diff --git a/python/timsquery_pyo3/src/iterator.rs b/python/timsquery_pyo3/src/iterator.rs index 2a305ed5..efb69007 100644 --- a/python/timsquery_pyo3/src/iterator.rs +++ b/python/timsquery_pyo3/src/iterator.rs @@ -9,7 +9,7 @@ use timsquery::{ Tolerance, }; -use crate::index::rt_range_ms_for_chromatogram; +use crate::index::resolve_query_rt; use crate::numpy_utils::array2d_to_numpy; use crate::target::PyTarget; use crate::tolerance::PyTolerance; @@ -34,7 +34,7 @@ pub struct PyChromatogramArrays { #[pyo3(get)] fragment_labels: Vec<(usize, f64)>, #[pyo3(get)] - rt_range_ms: (u32, u32), + resolved_rt: (u32, u32), #[pyo3(get)] num_cycles: usize, } @@ -92,7 +92,7 @@ fn extract_arrays( .iter() .map(|(k, mz)| (*k, *mz)) .collect(), - rt_range_ms: (rt.start(), rt.end()), + resolved_rt: (rt.start(), rt.end()), num_cycles: collector.num_cycles(), }) } @@ -104,7 +104,7 @@ pub struct PyChromatogramIterator { tol_source: ToleranceSource, target_source: Py, pool: Vec>, - chunk_tolerances: Vec, + chunk_tolerances: Vec, buffer: VecDeque, chunk_size: usize, exhausted: bool, @@ -140,7 +140,7 @@ impl PyChromatogramIterator { match next_result { Ok(obj) => { let target_ref: PyRef<'_, PyTarget> = obj.extract(py)?; - let target = target_ref.inner.clone(); + let target = &target_ref.inner; let tol = match &self.tol_source { ToleranceSource::Single(t) => t.clone(), @@ -159,27 +159,29 @@ impl PyChromatogramIterator { } }; - let rt_range_ms = - rt_range_ms_for_chromatogram(&tol, target.rt_seconds(), &self.handle)?; + let resolved_rt = resolve_query_rt(&tol, target.rt_seconds(), &self.handle)?; if i < self.pool.len() { self.pool[i] - .try_reset_with(&target, rt_range_ms, ref_rt) + .try_reset_with( + &timsquery::ExtractionQuery::new(target, resolved_rt), + ref_rt, + ) .map_err(|e| { PyErr::new::(format!("{e:?}")) })?; } else { - let collector = - ChromatogramCollector::::new(&target, rt_range_ms, ref_rt) - .map_err(|e| { - PyErr::new::(format!( - "{e:?}" - )) - })?; + let collector = ChromatogramCollector::::new( + &timsquery::ExtractionQuery::new(target, resolved_rt), + ref_rt, + ) + .map_err(|e| { + PyErr::new::(format!("{e:?}")) + })?; self.pool.push(collector); } source_ids.push(target.id().to_owned_id()); - self.chunk_tolerances.push(tol); + self.chunk_tolerances.push(tol.peak_tolerance()); n_this_chunk += 1; } Err(err) if err.is_instance_of::(py) => { diff --git a/python/timsquery_pyo3/src/target.rs b/python/timsquery_pyo3/src/target.rs index f785d9b4..327c3cff 100644 --- a/python/timsquery_pyo3/src/target.rs +++ b/python/timsquery_pyo3/src/target.rs @@ -1,5 +1,5 @@ use pyo3::prelude::*; -use timsquery::Target; +use timsquery::OwnedTarget; use timsquery::tinyvec::tiny_vec; /// A query target consists of one precursor and its fragments. @@ -11,7 +11,7 @@ use timsquery::tinyvec::tiny_vec; #[pyclass(skip_from_py_object)] #[derive(Debug, Clone)] pub struct PyTarget { - pub(crate) inner: Target, + pub(crate) inner: OwnedTarget, } #[pymethods] @@ -22,7 +22,7 @@ impl PyTarget { /// id: Numeric source ID for this target. /// precursor_mz: Monoisotopic precursor m/z. /// precursor_charge: Charge state. - /// rt_seconds: Expected retention time in seconds. + /// rt_seconds: Expected retention time on the queried acquisition, in seconds. /// mobility: Expected ion mobility (1/K0). /// fragment_mzs: List of fragment m/z values. /// fragment_labels: List of integer labels (one per fragment, same length as fragment_mzs). @@ -45,11 +45,11 @@ impl PyTarget { }; let fragment_labels_tv = fragment_labels.into_iter().collect(); - let target = Target::builder() + let target = OwnedTarget::builder() .id(id) .precursor(precursor_mz, precursor_charge) .mobility_ook0(mobility) - .rt_seconds(rt_seconds) + .rt_value(rt_seconds) .fragment_mzs(fragment_mzs) .fragment_labels(fragment_labels_tv) .precursor_labels(precursor_labels_tv) @@ -75,8 +75,26 @@ impl PyTarget { } #[getter] - fn rt_seconds(&self) -> f32 { - self.inner.rt_seconds() + fn rt_seconds(&self) -> Option { + self.inner + .rt() + .filter(|rt| *rt.axis == timsquery::RtAxis::Seconds) + .map(|rt| rt.value.0) + } + + #[getter] + fn library_rt(&self) -> Option { + self.inner.rt().map(|rt| rt.value.0) + } + + #[getter] + fn rt_axis(&self) -> &'static str { + match self.inner.rt().map(|rt| rt.axis) { + None => "absent", + Some(timsquery::RtAxis::Seconds) => "seconds", + Some(timsquery::RtAxis::NormalizedIndex { .. }) => "normalized_index", + _ => "unspecified", + } } #[getter] @@ -96,11 +114,12 @@ impl PyTarget { fn __repr__(&self) -> String { format!( - "Target(id={}, mz={:.4}, charge={}, rt={:.1}s, mob={:.3}, frags={}, precs={})", + "Target(id={}, mz={:.4}, charge={}, rt={:?} ({}), mob={:.3}, frags={}, precs={})", self.inner.id(), self.inner.precursor_mz(), self.inner.precursor_charge(), - self.inner.rt_seconds(), + self.library_rt(), + self.rt_axis(), self.inner.mobility_ook0(), self.inner.fragment_count(), self.inner.precursor_count(), diff --git a/rust/apex_sim/src/sim.rs b/rust/apex_sim/src/sim.rs index c9c07fdd..24a96f3a 100644 --- a/rust/apex_sim/src/sim.rs +++ b/rust/apex_sim/src/sim.rs @@ -341,13 +341,21 @@ pub fn build(params: &SimParams) -> SimData { let chromatograms = ChromatogramCollector:: { mobility_ook0: 1.0, - rt_seconds: (map((realized_apex as usize).min(n - 1)) as f32) / 1000.0, + rt: timsquery::ResolvedRt::resolve( + timsquery::RtSelection::Centered(timsquery::ObservedRTSeconds( + (map((realized_apex as usize).min(n - 1)) as f32) / 1000.0, + )), + &timsquery::models::tolerance::RtTolerance::Unrestricted, + rt_range_ms + .map_elems(|ms| timsquery::ObservedRTSeconds(ms as f32 / 1000.0)) + .unwrap(), + ) + .unwrap(), precursor_mono_mz: dummy_mz, precursor_charge: 2, precursor_mz_limits: (dummy_mz - 1.0, dummy_mz + 1.0), precursors, fragments, - rt_range_ms, n_precursor_peaks_added: n_prec_peaks, n_fragment_peaks_added: n_frag_peaks, n_quad_windows_matched: 1, diff --git a/rust/calib_dash/src/bin/calib_dash.rs b/rust/calib_dash/src/bin/calib_dash.rs index 0fedce0a..00e96860 100644 --- a/rust/calib_dash/src/bin/calib_dash.rs +++ b/rust/calib_dash/src/bin/calib_dash.rs @@ -28,6 +28,10 @@ fn main() { std::process::exit(1); } }; + if snapshot.points.is_empty() { + eprintln!("No RT fit in {}; no RT curve to replay.", path.display()); + return; + } if let Err(e) = validate_snapshot(&snapshot) { eprintln!("invalid calibration snapshot in {}: {e}", path.display()); std::process::exit(1); diff --git a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Curve.snap b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Curve.snap index 3b96dd1e..dbceb77b 100644 --- a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Curve.snap +++ b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Curve.snap @@ -2,7 +2,7 @@ source: rust/calib_dash/src/ui.rs expression: out --- - ┌ Fit -- observed RT (s) ↑ vs library RT (s) → ──────────────────────────────────────────── b0 ┐ + ┌ Fit -- observed RT (s) ↑ vs library RT coordinate → ───────────────────────────────────── b0 ┐ │ Showing: curve -- fitted calibration │ │ │ │ ▄▄▄▄▄▄│ diff --git a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_None.snap b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_None.snap index 38162e04..d9deb187 100644 --- a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_None.snap +++ b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_None.snap @@ -2,7 +2,7 @@ source: rust/calib_dash/src/ui.rs expression: out --- - ┌ Fit -- observed RT (s) ↑ vs library RT (s) → ──────────────────────────────────────────── b0 ┐ + ┌ Fit -- observed RT (s) ↑ vs library RT coordinate → ───────────────────────────────────── b0 ┐ │ Showing: none -- density only │ │ │ │ ▄▄▄▄▄▄│ diff --git a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Path.snap b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Path.snap index 045ae801..432e31e9 100644 --- a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Path.snap +++ b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_layer_Path.snap @@ -2,7 +2,7 @@ source: rust/calib_dash/src/ui.rs expression: out --- - ┌ Fit -- observed RT (s) ↑ vs library RT (s) → ──────────────────────────────────────────── b0 ┐ + ┌ Fit -- observed RT (s) ↑ vs library RT coordinate → ───────────────────────────────────── b0 ┐ │ Showing: path -- O chosen, X greedy tail │ │ │ │ ▄▄▄▄▄▄│ diff --git a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_tab_shows_a_banner_and_a_different_grid_when_scrubbing.snap b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_tab_shows_a_banner_and_a_different_grid_when_scrubbing.snap index 205bf24e..93aea185 100644 --- a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_tab_shows_a_banner_and_a_different_grid_when_scrubbing.snap +++ b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__fit_tab_shows_a_banner_and_a_different_grid_when_scrubbing.snap @@ -3,7 +3,7 @@ source: rust/calib_dash/src/ui.rs expression: "render_snapshot(&mut app, 100, 30)" --- SCRUBBED -- retained frame 3/5, batch 17 (not live; `>` returns to now) - ┌ Fit -- observed RT (s) ↑ vs library RT (s) → ──────────────────────────────────────────── b0 ┐ + ┌ Fit -- observed RT (s) ↑ vs library RT coordinate → ───────────────────────────────────── b0 ┐ │ Showing: none -- density only │ │ ████████████│ 15┤ ████████████│ diff --git a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__the_keys_overlay_lists_every_binding_over_the_tab_beneath_it.snap b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__the_keys_overlay_lists_every_binding_over_the_tab_beneath_it.snap index 48742401..fb117b1b 100644 --- a/rust/calib_dash/src/snapshots/calib_dash__ui__tests__the_keys_overlay_lists_every_binding_over_the_tab_beneath_it.snap +++ b/rust/calib_dash/src/snapshots/calib_dash__ui__tests__the_keys_overlay_lists_every_binding_over_the_tab_beneath_it.snap @@ -3,7 +3,7 @@ source: rust/calib_dash/src/ui.rs expression: "render(&mut app, 100, 30)" --- Fit │ Convergence │ Tolerances - ┌ Fit -- observed RT (s) ↑ vs library RT (s) → ──────────────────────────────────────────── b0 ┐ + ┌ Fit -- observed RT (s) ↑ vs library RT coordinate → ───────────────────────────────────── b0 ┐ │ Showing: none -- density only │ │ │ │ ▄▄▄▄▄▄│ diff --git a/rust/calib_dash/src/ui.rs b/rust/calib_dash/src/ui.rs index fc1abc14..797277c8 100644 --- a/rust/calib_dash/src/ui.rs +++ b/rust/calib_dash/src/ui.rs @@ -457,7 +457,7 @@ fn draw_heatmap(frame: &mut Frame, area: Rect, app: &App) { return; }; - let title_left = " Fit -- observed RT (s) \u{2191} vs library RT (s) \u{2192} "; + let title_left = " Fit -- observed RT (s) \u{2191} vs library RT coordinate \u{2192} "; // `batch` only -- the active layer gets its own subtitle row below. let title_right = format!(" b{} ", app.batch()); frame.render_widget( diff --git a/rust/calibrt/src/lib.rs b/rust/calibrt/src/lib.rs index 0c1faf98..0a323f7f 100644 --- a/rust/calibrt/src/lib.rs +++ b/rust/calibrt/src/lib.rs @@ -4,10 +4,16 @@ pub mod grid; mod pathfinding; +mod saved_calibration; pub mod types; +pub use saved_calibration::{ + CALIBRATION_FORMAT_VERSION, + SavedCalibration, +}; pub use types::{ LibraryRT, ObservedRTSeconds, + RtAxis, }; /// Minimum denominator for slope calculations to avoid division by zero. @@ -229,119 +235,6 @@ pub struct CalibrationSnapshot { pub lookback: usize, } -/// The only version [`SavedCalibration`] reads or writes. Named once so the -/// writer and the reader's gate cannot disagree. -pub const CALIBRATION_FORMAT_VERSION: &str = "v3"; - -/// JSON v3 calibration file format -- shared between CLI and viewer. -/// -/// `calibration` is the grid's own snapshot and the only record of the fit: the -/// curve and the ridge widths are recomputed by refitting it, so the file cannot -/// carry a curve that disagrees with the points that produced it. -/// -/// `R` is the residual block a writer measures beyond the curve. This crate does -/// not interpret it; a reader that has no use for it can leave it opaque. -#[derive(Debug, serde::Serialize, serde::Deserialize)] -pub struct SavedCalibration { - pub version: String, - pub rt_range_seconds: [f64; 2], - pub calibration: CalibrationSnapshot, - /// The uniform RT tolerance. Every writer has one -- it is what a query falls - /// back to where the grid measured no ridge. - pub rt_tolerance_minutes: f32, - /// What a search measured beyond the curve. `None` for a writer that measures - /// no residuals -- zeros there would read as "measured, and tight". - /// - /// The path is named because a bare `default` would demand `R: Default`. - #[serde(default = "Option::default")] - pub residuals: Option, - pub n_scored: usize, -} - -impl SavedCalibration { - /// Assemble a file record. The version is stamped here, not by callers. - pub fn new( - rt_range_seconds: [f64; 2], - calibration: CalibrationSnapshot, - rt_tolerance_minutes: f32, - residuals: Option, - n_scored: usize, - ) -> Self { - Self { - version: CALIBRATION_FORMAT_VERSION.to_string(), - rt_range_seconds, - calibration, - rt_tolerance_minutes, - residuals, - n_scored, - } - } - - /// Serialize to `path` in the layout [`Self::read`] expects. - pub fn write(&self, path: &std::path::Path) -> Result<(), String> - where - R: serde::Serialize, - { - let json = serde_json::to_string_pretty(self).map_err(|e| e.to_string())?; - std::fs::write(path, json).map_err(|e| e.to_string()) - } - - /// Parse a calibration file and check its provenance. The `Option` - /// is a reason to distrust the file, not an error: a calibration is only - /// valid for the run it was fit on, and `raw_rt_range` -- the RT span of the - /// run it is about to be used on -- is the one cheap way to catch the wrong - /// file. `None` there means nothing verifies it, which also warns. - pub fn read( - path: &std::path::Path, - raw_rt_range: Option<[f64; 2]>, - ) -> Result<(Self, Option), String> - where - R: serde::de::DeserializeOwned, - { - let json = std::fs::read_to_string(path).map_err(|e| e.to_string())?; - let saved: Self = serde_json::from_str(&json).map_err(|e| e.to_string())?; - if saved.version != CALIBRATION_FORMAT_VERSION { - return Err(format!( - "Unsupported calibration version: {} (expected {CALIBRATION_FORMAT_VERSION})", - saved.version - )); - } - let warning = saved.provenance_warning(raw_rt_range); - Ok((saved, warning)) - } - - /// Number of calibrant points the curve was fit on. - pub fn n_calibrants(&self) -> usize { - self.calibration.points.len() - } - - fn provenance_warning(&self, raw_rt_range: Option<[f64; 2]>) -> Option { - let Some(raw) = raw_rt_range else { - return Some( - "No raw RT range to check the calibration against -- nothing verifies it was \ - fit on this run" - .to_string(), - ); - }; - let overlap_lo = self.rt_range_seconds[0].max(raw[0]); - let overlap_hi = self.rt_range_seconds[1].min(raw[1]); - let overlap = (overlap_hi - overlap_lo).max(0.0); - let span = self.rt_range_seconds[1] - self.rt_range_seconds[0]; - if span <= 0.0 || overlap / span >= 0.5 { - return None; - } - Some(format!( - "Calibration RT range [{:.1}, {:.1}]s overlaps the raw file's [{:.1}, {:.1}]s by \ - {:.0}% -- it may have been fit on a different run", - self.rt_range_seconds[0], - self.rt_range_seconds[1], - raw[0], - raw[1], - (overlap / span) * 100.0, - )) - } -} - /// Grid geometry, so a consumer can lay out [`CalibrationState::grid_cells`] /// without inferring it from the node coordinates (which is impossible for an /// empty or single-occupied grid). diff --git a/rust/calibrt/src/saved_calibration.rs b/rust/calibrt/src/saved_calibration.rs new file mode 100644 index 00000000..45a46ad6 --- /dev/null +++ b/rust/calibrt/src/saved_calibration.rs @@ -0,0 +1,202 @@ +//! Saved calibration format, version validation, and file I/O. +use crate::{ + CalibrationSnapshot, + RtAxis, +}; + +/// The only version [`SavedCalibration`] reads or writes. Named once so the +/// writer and the reader's gate cannot disagree. +pub const CALIBRATION_FORMAT_VERSION: &str = "v4"; + +/// JSON v4 calibration file format -- shared between CLI and viewer. +/// +/// `calibration` is the grid's own snapshot and the only record of the fit: the +/// curve and the ridge widths are recomputed by refitting it, so the file cannot +/// carry a curve that disagrees with the points that produced it. +/// +/// `R` is the residual block a writer measures beyond the curve. This crate does +/// not interpret it; a reader that has no use for it can leave it opaque. +#[derive(Debug, serde::Serialize, serde::Deserialize)] +pub struct SavedCalibration { + #[serde(deserialize_with = "deserialize_version")] + pub version: String, + pub rt_range_seconds: [f64; 2], + /// Input-axis descriptor supplied by the library consumer. The fitter maps + /// numerical library coordinates to observed seconds without interpreting units. + pub library_rt_axis: RtAxis, + pub calibration: CalibrationSnapshot, + /// The uniform RT tolerance. Every writer has one -- it is what a query falls + /// back to where the grid measured no ridge. + pub rt_tolerance_minutes: f32, + /// What a search measured beyond the curve. `None` for a writer that measures + /// no residuals -- zeros there would read as "measured, and tight". + /// + /// The path is named because a bare `default` would demand `R: Default`. + #[serde(default = "Option::default")] + pub residuals: Option, + pub n_scored: usize, +} + +impl SavedCalibration { + /// Assemble a file record. The version is stamped here, not by callers. + pub fn new( + rt_range_seconds: [f64; 2], + calibration: CalibrationSnapshot, + rt_tolerance_minutes: f32, + residuals: Option, + n_scored: usize, + ) -> Self { + Self { + version: CALIBRATION_FORMAT_VERSION.to_string(), + library_rt_axis: RtAxis::Unspecified, + rt_range_seconds, + calibration, + rt_tolerance_minutes, + residuals, + n_scored, + } + } + + pub fn with_library_rt_axis(mut self, axis: RtAxis) -> Self { + self.library_rt_axis = axis; + self + } + + /// Serialize to `path` in the layout [`Self::read`] expects. + pub fn write(&self, path: &std::path::Path) -> Result<(), String> + where + R: serde::Serialize, + { + let json = serde_json::to_string_pretty(self).map_err(|e| e.to_string())?; + std::fs::write(path, json).map_err(|e| e.to_string()) + } + + /// Parse a calibration file and check its provenance. The `Option` + /// is a reason to distrust the file, not an error: a calibration is only + /// valid for the run it was fit on, and `raw_rt_range` -- the RT span of the + /// run it is about to be used on -- is the one cheap way to catch the wrong + /// file. `None` there means nothing verifies it, which also warns. + pub fn read( + path: &std::path::Path, + raw_rt_range: Option<[f64; 2]>, + ) -> Result<(Self, Option), String> + where + R: serde::de::DeserializeOwned, + { + let json = std::fs::read_to_string(path).map_err(|e| e.to_string())?; + let saved: Self = serde_json::from_str(&json).map_err(|e| e.to_string())?; + let warning = saved.provenance_warning(raw_rt_range); + Ok((saved, warning)) + } + + /// Number of calibrant points the curve was fit on. + pub fn n_calibrants(&self) -> usize { + self.calibration.points.len() + } + + fn provenance_warning(&self, raw_rt_range: Option<[f64; 2]>) -> Option { + let Some(raw) = raw_rt_range else { + return Some( + "No raw RT range to check the calibration against -- nothing verifies it was \ + fit on this run" + .to_string(), + ); + }; + let overlap_lo = self.rt_range_seconds[0].max(raw[0]); + let overlap_hi = self.rt_range_seconds[1].min(raw[1]); + let overlap = (overlap_hi - overlap_lo).max(0.0); + let span = self.rt_range_seconds[1] - self.rt_range_seconds[0]; + if span <= 0.0 || overlap / span >= 0.5 { + return None; + } + Some(format!( + "Calibration RT range [{:.1}, {:.1}]s overlaps the raw file's [{:.1}, {:.1}]s by \ + {:.0}% -- it may have been fit on a different run", + self.rt_range_seconds[0], + self.rt_range_seconds[1], + raw[0], + raw[1], + (overlap / span) * 100.0, + )) + } +} + +fn deserialize_version<'de, D: serde::Deserializer<'de>>( + deserializer: D, +) -> Result { + let version = ::deserialize(deserializer)?; + if version != CALIBRATION_FORMAT_VERSION { + return Err(serde::de::Error::custom(format!( + "Unsupported calibration version: {version} (expected {CALIBRATION_FORMAT_VERSION})" + ))); + } + Ok(version) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn fixture() -> serde_json::Value { + serde_json::json!({ + "version": "v4", + "rt_range_seconds": [0.0, 1200.0], + "library_rt_axis": {"kind": "normalized_index", "scale": "anchors"}, + "calibration": {"points": [], "grid_size": 16, "lookback": 4}, + "rt_tolerance_minutes": 1.25, + "n_scored": 0 + }) + } + + #[test] + fn typed_axis_preserves_the_v4_wire_format() { + let json = fixture(); + let saved: SavedCalibration = serde_json::from_value(json.clone()).unwrap(); + assert_eq!( + saved.library_rt_axis, + RtAxis::NormalizedIndex { + scale: Some("anchors".into()) + } + ); + assert_eq!( + serde_json::to_value(saved).unwrap()["library_rt_axis"], + json["library_rt_axis"] + ); + } + + #[test] + fn malformed_axis_is_rejected_at_deserialization() { + for axis in [ + serde_json::json!({"kind": "unknown"}), + serde_json::json!({"kind": "normalized_index", "scale": 42}), + ] { + let mut json = fixture(); + json["library_rt_axis"] = axis; + assert!(serde_json::from_value::(json).is_err()); + } + } + + #[test] + fn direct_deserialization_checks_version_even_without_v4_fields() { + for json in [ + r#"{"version":"v3","rt_range_seconds":[0,1200]}"#, + r#"{"rt_range_seconds":[0,1200],"version":"v3"}"#, + ] { + let error = serde_json::from_str::(json) + .unwrap_err() + .to_string(); + assert!( + error.contains("Unsupported calibration version: v3 (expected v4)"), + "{error}" + ); + } + let mut json = fixture(); + json["version"] = serde_json::json!("v2"); + assert!( + serde_json::from_value::(json) + .unwrap_err() + .to_string() + .contains("Unsupported calibration version") + ); + } +} diff --git a/rust/calibrt/src/types.rs b/rust/calibrt/src/types.rs index 2e108417..4e81886a 100644 --- a/rust/calibrt/src/types.rs +++ b/rust/calibrt/src/types.rs @@ -4,13 +4,39 @@ use serde::{ }; use std::fmt; +/// Coordinate domain of the library used to fit a calibration. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case")] +pub enum RtAxis { + #[default] + Absent, + Seconds, + NormalizedIndex { + scale: Option, + }, + Unspecified, +} + +impl std::fmt::Display for RtAxis { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Absent => f.write_str("no RT"), + Self::Seconds => f.write_str("s"), + Self::NormalizedIndex { scale: Some(scale) } => write!(f, "index ({scale})"), + Self::NormalizedIndex { scale: None } => f.write_str("index"), + Self::Unspecified => f.write_str("unspecified units"), + } + } +} + /// Library reference retention time. Unit-agnostic -- could be iRT, minutes, /// or arbitrary units depending on the spectral library. #[derive(Debug, Clone, Copy, PartialEq, PartialOrd, Serialize, Deserialize)] #[serde(transparent)] pub struct LibraryRT(pub T); -/// Observed retention time from raw instrument data, always in seconds. +/// Retention time on an acquisition axis, always in seconds. +/// May be a calibrated prediction, an explicit query center, or a measured apex. #[derive(Debug, Clone, Copy, PartialEq, PartialOrd, Serialize, Deserialize)] #[serde(transparent)] pub struct ObservedRTSeconds(pub T); diff --git a/rust/timsquery/Cargo.toml b/rust/timsquery/Cargo.toml index fb241177..0329808f 100644 --- a/rust/timsquery/Cargo.toml +++ b/rust/timsquery/Cargo.toml @@ -8,6 +8,7 @@ license.workspace = true # Workspace member deps array2d = { path = "../array2d" } +calibrt = { path = "../calibrt" } timscentroid = { path = "../timscentroid" } micromzpaf = { path = "../micromzpaf" } tims_stage = { path = "../tims_stage" } diff --git a/rust/timsquery/src/chemistry/analyte.rs b/rust/timsquery/src/chemistry/analyte.rs index 9a28697b..f581a4b7 100644 --- a/rust/timsquery/src/chemistry/analyte.rs +++ b/rust/timsquery/src/chemistry/analyte.rs @@ -55,7 +55,7 @@ //! modification spellings therefore collapse. Incomparable entries remain separate; //! formula equality and labels do not establish chemical identity or competition. //! -//! Results format version 4 resolves metadata from the library. `sequence` is nullable +//! Results format version 5 resolves metadata from the library. `sequence` is nullable //! and canonically formatted; `entry_name`, `molecular_formula`, and `formula_basis` //! are nullable columns. Unformattable properties produce null. The viewer's Analyte //! column displays sequence or formula. `Analyte` serialization preserves property @@ -871,7 +871,7 @@ mod tests { TargetCapabilities, TargetColumnsBuilder, }; - use crate::traits::QueryGeom; + use crate::traits::Target; #[test] fn geometry_and_formula_rows_need_no_peptide() { diff --git a/rust/timsquery/src/errors.rs b/rust/timsquery/src/errors.rs index a28a876f..65861e6d 100644 --- a/rust/timsquery/src/errors.rs +++ b/rust/timsquery/src/errors.rs @@ -86,6 +86,7 @@ impl Display for UnsupportedDataError { #[derive(Debug)] pub enum DataProcessingError { + InvalidRtQuery, ExpectedVectorLength { real: usize, expected: usize }, ExpectedNonEmptyData, InsufficientData { real: usize, expected: usize }, diff --git a/rust/timsquery/src/lib.rs b/rust/timsquery/src/lib.rs index 96023678..747ae7d5 100644 --- a/rust/timsquery/src/lib.rs +++ b/rust/timsquery/src/lib.rs @@ -17,7 +17,7 @@ pub use crate::models::base::{ MzMajorIntensityArray, RTMajorIntensityArray, }; -pub use crate::models::target::Target; +pub use crate::models::target::OwnedTarget; // Re-export traits pub use crate::models::PeakAddable; @@ -71,3 +71,20 @@ pub use crate::serde::index_serde::{ LoadIndexError, load_index, }; + +pub use crate::models::retention_time::{ + RtAxis, + RtCoordinate, +}; +pub use crate::traits::Target; + +pub use calibrt::{ + LibraryRT, + ObservedRTSeconds, +}; +pub use models::extraction_query::{ + ExtractionQuery, + ResolvedRt, + RtSelection, +}; +pub use models::tolerance::PeakTolerance; diff --git a/rust/timsquery/src/models/aggregators/chromatogram_agg.rs b/rust/timsquery/src/models/aggregators/chromatogram_agg.rs index 948bbc68..97c12841 100644 --- a/rust/timsquery/src/models/aggregators/chromatogram_agg.rs +++ b/rust/timsquery/src/models/aggregators/chromatogram_agg.rs @@ -1,3 +1,7 @@ +use crate::{ + ExtractionQuery, + ResolvedRt, +}; use serde::Serialize; use crate::errors::DataProcessingError; @@ -6,7 +10,7 @@ use crate::models::base::{ Chromatogram, MzMajorIntensityArray, }; -use crate::traits::QueryGeom; +use crate::traits::Target; use crate::traits::queriable_data::HasQueryData; use crate::{ KeyLike, @@ -27,7 +31,7 @@ use timscentroid::utils::TupleRange; pub struct ChromatogramCollector { // Query scalars carried from the eg at reset time. pub mobility_ook0: f32, - pub rt_seconds: f32, + pub rt: ResolvedRt, pub precursor_mono_mz: f64, pub precursor_charge: u8, /// Cached from `Target::precursor_mz_limits()` at reset @@ -38,7 +42,6 @@ pub struct ChromatogramCollector { // labels/mzs are NOT duplicated on the collector. pub precursors: MzMajorIntensityArray, pub fragments: MzMajorIntensityArray, - pub rt_range_ms: TupleRange, /// MS1 peaks written into any precursor chromatogram cell during the /// most recent `add_query`. Informational only -- downstream fast-path @@ -63,10 +66,11 @@ pub struct ChromatogramCollector { impl ChromatogramCollector { pub fn new( - eg: &impl QueryGeom