From dbd7adb288d35060230987af9f1701246cdd52cb Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 08:49:00 +0200 Subject: [PATCH 01/18] Derive an anomaly total instead of re-scanning the graph for it Every anomaly preflight is a SELECT ... LIMIT 100 paired with a *_count aggregate. Both scan the whole graph -- the LIMIT bounds what comes back, not what is examined -- so on the 17.1M-triple benchmark graph the pair cost 97.0 s and 94.6 s to report the same zero. Three such scans were 294.1 s of the 298.1 s that all fourteen preflights cost, against 17.1 s for the thirteen semantic queries they exist to protect. When the sample returns fewer rows than its limit it enumerated every match, so the exact total is the number of rows it returned. Deriving it is arithmetic, not approximation. At or above the limit the count is genuinely unknown and the aggregate still runs. The derived record carries derived/derivedFrom/derivedReason, and benchmark.csv gains a derived column, so a 0.0 s row is never mistaken for an impossibly fast scan when the timings are analysed. Recommendation 3 of implementation-improvements-from-benchmarking.md. Co-Authored-By: Claude Opus 5 --- src/validation/validation_runner.py | 130 +++++++++- test/test_validation_derived_counts_unit.py | 266 ++++++++++++++++++++ 2 files changed, 390 insertions(+), 6 deletions(-) create mode 100644 test/test_validation_derived_counts_unit.py diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index 71d9203..40bd61e 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -2942,6 +2942,103 @@ def normalize(query_id: str, path: Path) -> Any: return rows +def derive_anomaly_count_execution( + sample_execution: dict[str, Any], + sample_query_id: str, + raw_dir: Path, +) -> dict[str, Any] | None: + """Synthesize a ``*_count`` result from a sample that did not hit its limit. + + An anomaly preflight is ``SELECT ... LIMIT ANOMALY_SAMPLE_LIMIT``. When it + returns fewer rows than that limit it enumerated *every* match, so the exact + anomaly total is the number of rows returned and the companion aggregate can + only re-derive a number we already hold. Skipping it is not an approximation. + + That matters because the companion is not cheap. Both queries scan the whole + graph -- the LIMIT bounds what is returned, not what is examined -- so on a + 17.1M-triple graph ``preflight_empty_values`` cost 97.0 s and + ``preflight_empty_values_count`` a further 94.6 s to report the same zero. + Three such scans were 294.1 s of the 298.1 s that all fourteen preflights + cost, against 17.1 s for the thirteen semantic queries they precede. + + Returns ``None`` when the sample is truncated or unreadable, in which case + the real aggregate must run: the count is then genuinely unknown. + """ + if sample_execution.get("status") != "PASS": + return None + try: + rows = bindings(Path(sample_execution["rawResult"])) + except (OSError, ValueError, KeyError, json.JSONDecodeError): + return None + if len(rows) >= ANOMALY_SAMPLE_LIMIT: + return None + + # Write the derived answer in SPARQL Results JSON so every downstream + # consumer -- anomaly_count, benchmark.csv, the raw result tree -- reads it + # exactly as it reads an executed one. + raw_path = raw_dir / f"{sample_query_id}_count.json" + raw_path.parent.mkdir(parents=True, exist_ok=True) + write_json( + raw_path, + { + "head": {"vars": ["anomalyCount"]}, + "results": { + "bindings": [ + { + "anomalyCount": { + "type": "literal", + "datatype": "http://www.w3.org/2001/XMLSchema#integer", + "value": str(len(rows)), + } + } + ] + }, + }, + ) + return { + "status": "PASS", + "engine": sample_execution.get("engine"), + "exitCode": 0, + "wallSeconds": 0.0, + "query": None, + "rawResult": str(raw_path), + "stderr": None, + "resourceMetrics": None, + # The record says plainly that this was not executed, so a reader never + # mistakes a 0.0 s row in benchmark.csv for an impossibly fast scan. + "derived": True, + "derivedFrom": sample_query_id, + "derivedReason": ( + f"{sample_query_id} returned {len(rows)} of at most " + f"{ANOMALY_SAMPLE_LIMIT} rows, so it enumerated every match and the " + f"exact count is known without a second full-graph scan" + ), + } + + +def derived_count_for( + query_id: str, + executions: dict[str, dict[str, Any]], + raw_dir: Path, +) -> dict[str, Any] | None: + """The derived execution for ``query_id``, or None if it must really run. + + Answers one question for the query loop: is this a ``*_count`` companion + whose sample already enumerated every match? Anything else -- a semantic + query, a preflight that is not part of an anomaly pair, a sample that was + truncated or failed -- returns None and is executed normally. + """ + if not query_id.endswith("_count"): + return None + sample_query_id = query_id[: -len("_count")] + if sample_query_id not in ANOMALY_PREFLIGHT_QUERIES: + return None + sample = executions.get(sample_query_id) + if sample is None: + return None + return derive_anomaly_count_execution(sample, sample_query_id, raw_dir) + + def anomaly_count(executions: dict[str, dict[str, Any]], query_id: str) -> Any: """Exact anomaly total from the companion aggregate, or None if unavailable. @@ -3259,6 +3356,9 @@ def evaluate_validation( "oracle_wall_seconds", "engine_setup_seconds", "artifact_origin", + # 1 when the row was derived from a sibling query instead of executed. + # Exclude these before summing wall_seconds as measured query cost. + "derived", ] @@ -3309,6 +3409,9 @@ def build_benchmark( query_id: { "status": execution.get("status"), "wallSeconds": execution.get("wallSeconds"), + # True when the answer was derived from a sibling query rather + # than executed, so a 0.0 s row is never read as a measurement. + "derived": bool(execution.get("derived")), } for query_id, execution in executions.items() } @@ -3388,6 +3491,7 @@ def write_benchmark_csv(path: Path, benchmark: dict[str, Any]) -> Path: "oracle_wall_seconds": oracle_total, "engine_setup_seconds": engine["setupSeconds"], "artifact_origin": engine["artifactOrigin"] or "", + "derived": 1 if entry.get("derived") else 0, }) return path @@ -3641,13 +3745,27 @@ def run_validation(args: argparse.Namespace) -> int: f" ({engine_name})", flush=True, ) - executions[query_id] = engine.execute( - query_id, query_path(query_dir, query_id) - ) - progress.emit( - "progress", completed=completed, query_id=query_id, - detail=f"{engine_name}: completed {query_id}", + # An anomaly preflight's ``*_count`` companion re-scans + # the whole graph to total what the sample already + # enumerated, whenever the sample came back under its + # LIMIT. Derive it instead: same number, no second scan. + derived = derived_count_for( + query_id, executions, engine_raw_dir ) + if derived is not None: + executions[query_id] = derived + progress.emit( + "progress", completed=completed, query_id=query_id, + detail=f"{engine_name}: derived {query_id}", + ) + else: + executions[query_id] = engine.execute( + query_id, query_path(query_dir, query_id) + ) + progress.emit( + "progress", completed=completed, query_id=query_id, + detail=f"{engine_name}: completed {query_id}", + ) # A query that fails, times out, or returns the wrong # answer never stops the suite: its verdict is recorded diff --git a/test/test_validation_derived_counts_unit.py b/test/test_validation_derived_counts_unit.py new file mode 100644 index 0000000..216996d --- /dev/null +++ b/test/test_validation_derived_counts_unit.py @@ -0,0 +1,266 @@ +"""Deriving an anomaly total instead of re-scanning the graph for it. + +Every anomaly preflight is a ``SELECT ... LIMIT 100`` paired with a ``*_count`` +aggregate. Both scan the whole graph -- the LIMIT bounds what comes back, not +what is examined -- so on the 17.1M-triple benchmark graph the pair cost 97.0 s +and 94.6 s to report the same zero. Three such scans were 294.1 s of the 298.1 s +that all fourteen preflights cost, against 17.1 s for the thirteen semantic +queries they exist to protect. + +When the sample returns fewer rows than its limit it enumerated every match, so +the exact total is the number of rows it returned. Deriving it is not an +approximation, and these tests pin that distinction: derive below the limit, +execute at or above it. +""" + +import csv +import importlib.util +import json +import tempfile +import unittest +from pathlib import Path + +from test.helpers import VerboseTestCase + +RUNNER_PATH = Path(__file__).resolve().parents[1] / "src" / "validation" / "validation_runner.py" + + +def load_runner(): + spec = importlib.util.spec_from_file_location("validation_runner_derived", RUNNER_PATH) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +V = load_runner() + + +def write_sample(directory: Path, rows: int, query_id: str = "preflight_empty_values") -> dict: + """A PASSing sample execution whose raw result holds ``rows`` bindings.""" + raw = directory / f"{query_id}.json" + raw.write_text( + json.dumps( + { + "head": {"vars": ["s", "p", "o", "issue"]}, + "results": { + "bindings": [ + { + "s": {"type": "uri", "value": f"urn:s{index}"}, + "p": {"type": "uri", "value": "urn:p"}, + "o": {"type": "literal", "value": " "}, + "issue": {"type": "literal", "value": "EMPTY_LITERAL"}, + } + for index in range(rows) + ] + }, + } + ), + encoding="utf-8", + ) + return { + "status": "PASS", + "engine": "qlever", + "exitCode": 0, + "wallSeconds": 97.01, + "query": f"/queries/{query_id}.rq", + "rawResult": str(raw), + "stderr": None, + "resourceMetrics": None, + } + + +class DeriveAnomalyCountTests(VerboseTestCase): + def test_a_clean_sample_derives_a_zero_count_without_a_second_scan(self): + """Zero anomalies is the common case and must not cost a full scan.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + sample = write_sample(directory, rows=0) + derived = V.derive_anomaly_count_execution( + sample, "preflight_empty_values", directory + ) + self.assertIsNotNone(derived) + self.assertTrue(derived["derived"]) + self.assertEqual(derived["derivedFrom"], "preflight_empty_values") + self.assertEqual(derived["status"], "PASS") + self.assertEqual(derived["wallSeconds"], 0.0) + self.assertEqual( + V.anomaly_count({"preflight_empty_values_count": derived}, + "preflight_empty_values"), + 0, + ) + + def test_a_partial_sample_derives_exactly_the_rows_it_returned(self): + """Under the limit, the sample IS the population -- not an estimate.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + sample = write_sample(directory, rows=20) + derived = V.derive_anomaly_count_execution( + sample, "preflight_empty_values", directory + ) + self.assertIsNotNone(derived) + self.assertEqual( + V.anomaly_count({"preflight_empty_values_count": derived}, + "preflight_empty_values"), + 20, + ) + + def test_a_truncated_sample_refuses_to_derive_and_demands_the_aggregate(self): + """At the limit the true count is unknown; guessing it would be a lie.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + sample = write_sample(directory, rows=V.ANOMALY_SAMPLE_LIMIT) + self.assertIsNone( + V.derive_anomaly_count_execution( + sample, "preflight_empty_values", directory + ) + ) + + def test_a_failed_sample_refuses_to_derive(self): + """A sample that did not run says nothing about how many matches exist.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + sample = write_sample(directory, rows=0) + sample["status"] = "EXECUTION_FAILED" + self.assertIsNone( + V.derive_anomaly_count_execution( + sample, "preflight_empty_values", directory + ) + ) + + def test_an_unreadable_sample_refuses_to_derive(self): + """A corrupt result must not silently become a confident zero.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + sample = write_sample(directory, rows=0) + Path(sample["rawResult"]).write_text("not json", encoding="utf-8") + self.assertIsNone( + V.derive_anomaly_count_execution( + sample, "preflight_empty_values", directory + ) + ) + + def test_every_anomaly_preflight_has_a_count_companion_that_can_be_derived(self): + """The optimisation applies to the whole family, not just empty_values.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + for query_id in V.ANOMALY_PREFLIGHT_QUERIES: + self.assertIn(f"{query_id}_count", V.PREFLIGHT_COUNT_QUERIES) + sample = write_sample(directory, rows=0, query_id=query_id) + derived = V.derive_anomaly_count_execution(sample, query_id, directory) + self.assertIsNotNone(derived, query_id) + self.assertEqual( + V.anomaly_count({f"{query_id}_count": derived}, query_id), 0, query_id + ) + + def test_the_derived_record_names_why_it_was_not_executed(self): + """A 0.0 s row must carry its own explanation, not need one.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + derived = V.derive_anomaly_count_execution( + write_sample(directory, rows=3), "preflight_empty_values", directory + ) + self.assertIn("enumerated every match", derived["derivedReason"]) + self.assertIn("preflight_empty_values", derived["derivedReason"]) + + +class WhichQueriesTheLoopSkipsTests(VerboseTestCase): + """The decision the query loop makes, isolated from any engine.""" + + def test_a_semantic_query_is_always_executed(self): + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + self.assertIsNone( + V.derived_count_for("q05_sample_genotype_counts", {}, directory) + ) + + def test_a_sample_preflight_is_always_executed(self): + """The sample is the thing the count is derived FROM; it must run.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + self.assertIsNone( + V.derived_count_for("preflight_empty_values", {}, directory) + ) + + def test_a_count_with_no_sample_recorded_yet_is_executed(self): + """Ordering safety: never derive from a result that is not there.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + self.assertIsNone( + V.derived_count_for("preflight_empty_values_count", {}, directory) + ) + + def test_the_non_anomaly_distinct_triple_count_is_executed(self): + """It has a _count suffix but no sample pair, so it is not derivable.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + self.assertNotIn( + "preflight_distinct_triple_count", V.ANOMALY_PREFLIGHT_QUERIES + ) + self.assertIsNone( + V.derived_count_for("preflight_distinct_triple_count", {}, directory) + ) + + def test_a_count_whose_sample_came_back_clean_is_derived(self): + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + executions = {"preflight_empty_values": write_sample(directory, rows=0)} + derived = V.derived_count_for( + "preflight_empty_values_count", executions, directory + ) + self.assertIsNotNone(derived) + self.assertTrue(derived["derived"]) + + def test_the_sample_always_precedes_its_count_in_the_execution_order(self): + """Derivation depends on it; a reordering must fail here, not in a run.""" + order = V.PREFLIGHT_QUERIES + V.PREFLIGHT_COUNT_QUERIES + for query_id in V.ANOMALY_PREFLIGHT_QUERIES: + self.assertLess( + order.index(query_id), + order.index(f"{query_id}_count"), + query_id, + ) + + +class DerivedRowsAreMarkedInTheBenchmarkTests(VerboseTestCase): + def test_benchmark_csv_carries_a_derived_column(self): + """Analysis must be able to exclude derived rows before summing cost.""" + self.assertIn("derived", V.BENCHMARK_CSV_HEADER) + + def test_a_derived_row_is_flagged_and_an_executed_row_is_not(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "benchmark.csv" + V.write_benchmark_csv( + path, + { + "oracle": {"totalSeconds": 12.41, "perQuerySeconds": {}}, + "engines": { + "qlever": { + "setupSeconds": 22.5, + "artifactOrigin": "run artifact", + "queries": { + "preflight_empty_values": { + "status": "PASS", + "wallSeconds": 97.01, + "derived": False, + }, + "preflight_empty_values_count": { + "status": "PASS", + "wallSeconds": 0.0, + "derived": True, + }, + }, + } + }, + }, + ) + rows = { + row["query_id"]: row + for row in csv.DictReader(path.open(encoding="utf-8")) + } + self.assertEqual(rows["preflight_empty_values"]["derived"], "0") + self.assertEqual(rows["preflight_empty_values_count"]["derived"], "1") + + +if __name__ == "__main__": + unittest.main() From 6eafd73ab9091da61f86eb7cea08257c858ab002 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 08:53:40 +0200 Subject: [PATCH 02/18] Default validation to QLever, and say which choice decides query time The interface suggests the representation is the performance decision. The measurement says the opposite. Same thirteen questions, same machine: changing the ENGINE spans three orders of magnitude -- QLever 1.09 s, Comunica 23.30 s, HDT-backed 44.83 s, native pycottas 1402.37 s -- while changing the ARTIFACT under one engine moves nothing: on a 17.1M-triple graph QLever took 17.11 s over N-Triples, 17.16 s over HDT, 16.97 s over COTTAS, under 5% apart. So the default that mattered was the slow one. The benchmark campaign had to pass --validation-engine qlever by hand on every cell large enough for it to matter; this makes that the default and documents the separation in the CLI help and docs/validation.md, so the representation is chosen for size and build cost and the engine for speed. Recommendation 10 of implementation-improvements-from-benchmarking.md. Co-Authored-By: Claude Opus 5 --- docs/validation.md | 31 +++++++++++++++++---- test/test_validation_derived_counts_unit.py | 27 ++++++++++++++++++ vcf_rdfizer.py | 21 +++++++++++--- 3 files changed, 70 insertions(+), 9 deletions(-) diff --git a/docs/validation.md b/docs/validation.md index 6eae925..a1e6d37 100644 --- a/docs/validation.md +++ b/docs/validation.md @@ -59,11 +59,32 @@ bigger heap. | Graph size | Recommended | | --- | --- | -| Fixtures, small VCFs | `--validation-engine comunica` (default; no index build) | -| Anything above a few GiB of N-Triples | `--validation-engine qlever`, or validate the indexed artifact with `--validate-artifacts hdt` | +| Anything | `--validation-engine qlever` (default; builds an on-disk index) | +| Fixtures, where an index build is not worth its setup | `--validation-engine comunica` | Above 4 GiB the runner warns before it starts and names the alternatives. +### The engine is the performance decision; the artifact is not + +This is the most consequential thing to know before configuring a run, and it +is the opposite of what the interface suggests. Measured on one machine, same +thirteen questions, same graph: + +| What changes | Spread | +| --- | --- | +| The **engine**, on a 0.96M-triple graph | QLever 1.09 s, Comunica 23.30 s, HDT-backed 44.83 s, native pycottas 1402.37 s | +| The **artifact**, under QLever on a 17.1M-triple graph | N-Triples 17.11 s, HDT 17.16 s, COTTAS 16.97 s | + +Three orders of magnitude against under 5%. Each engine materializes what it +needs, so which compressed form the triples were stored in is nearly invisible +to query time. Choose the representation for size and build cost -- COTTAS is +0.37--0.55x the stored N-Triples where HDT is 1.36--1.82x, and HDT builds +1.5--2.8x faster -- and choose the engine for speed. + +`qlever` is the default for that reason. It was not, until the benchmark +campaign had to pass `--validation-engine qlever` by hand on every cell large +enough for the difference to matter. + ### If validation seems to hang First check `setupSeconds` in the engine report. Slow *setup* is a scale @@ -246,10 +267,10 @@ produced it. | Engine | Queries | Setup | Use it when | |---|---|---|---| -| `comunica` (default) | The N-Triples file directly | none | The graph fits comfortably in RAM | -| `qlever` | An on-disk [QLever](https://github.com/ad-freiburg/qlever) index, served on a container-local port | index build | The graph no longer fits in memory, or the aggregate queries are too slow | +| `qlever` (default) | An on-disk [QLever](https://github.com/ad-freiburg/qlever) index, served on a container-local port | index build | Almost always: fastest by a wide margin at every size measured | +| `comunica` | The N-Triples file directly | none | A fixture small enough that an index build is not worth its setup | | `hdt` | A `.hdt` artifact **in place**, through Comunica's HDT engine | reuses the run's HDT, or builds one | Checking that the compressed artifact is queryable, not just decodable | -| `cottas` | A `.cottas` artifact **in place**, through `pycottas`'s rdflib store | reuses the run's COTTAS, or builds one | Same, for COTTAS | +| `cottas` | A `.cottas` artifact **in place**, through `pycottas`'s rdflib store | reuses the run's COTTAS, or builds one | Same, for COTTAS. A conformance path, not a query path: 1286x slower than QLever on identical work, and it does not terminate at all on the condensed encoding (see [limitations](limitations.md)) | ### Validating a compressed artifact without decoding it diff --git a/test/test_validation_derived_counts_unit.py b/test/test_validation_derived_counts_unit.py index 216996d..5408914 100644 --- a/test/test_validation_derived_counts_unit.py +++ b/test/test_validation_derived_counts_unit.py @@ -264,3 +264,30 @@ def test_a_derived_row_is_flagged_and_an_executed_row_is_not(self): if __name__ == "__main__": unittest.main() + + +class EngineDefaultTests(VerboseTestCase): + """The default engine is the one performance decision a user rarely revisits.""" + + def test_the_default_validation_engine_is_qlever(self): + """Measured: 1.09 s vs Comunica's 23.30 s on the same thirteen queries.""" + import vcf_rdfizer + + self.assertEqual(vcf_rdfizer.DEFAULT_VALIDATION_ENGINE, "qlever") + self.assertEqual( + vcf_rdfizer.parse_validation_engines( + vcf_rdfizer.DEFAULT_VALIDATION_ENGINE + ), + ["qlever"], + ) + + def test_every_engine_remains_selectable(self): + """Changing the default must not narrow the choice.""" + import vcf_rdfizer + + self.assertEqual( + vcf_rdfizer.parse_validation_engines("all"), + list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), + ) + for engine in vcf_rdfizer.VALIDATION_ENGINE_CHOICES: + self.assertEqual(vcf_rdfizer.parse_validation_engines(engine), [engine]) diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 23d9e2f..5f6f71a 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -363,7 +363,16 @@ (".hdt", "hdt"), ) VALIDATION_ENGINE_CHOICES = ("comunica", "qlever", "hdt", "cottas") -DEFAULT_VALIDATION_ENGINE = "comunica" +#: The representation is a storage decision; the engine is the performance one. +#: Measured on one graph, one machine, the same thirteen questions: QLever 1.09 s, +#: Comunica 23.30 s, the HDT-backed engine 44.83 s, native pycottas 1402.37 s. +#: Against the SAME engine, the artifact queried barely matters -- on a +#: 17.1M-triple graph QLever took 17.11 s over N-Triples, 17.16 s over HDT and +#: 16.97 s over COTTAS, under 5% apart, because each engine materializes what it +#: needs. So the default that matters is this one, and it was the slow choice: +#: the benchmark campaign had to pass --validation-engine qlever by hand on +#: every cell large enough for it to matter. +DEFAULT_VALIDATION_ENGINE = "qlever" def parse_validation_engines(raw: str) -> list[str]: @@ -9033,9 +9042,13 @@ def main(): default=DEFAULT_VALIDATION_ENGINE, help=( "SPARQL engine(s) for validation, comma-separated, or 'all'. " - "comunica queries the graph in memory; qlever builds an on-disk " - "index and serves it; hdt and cottas query those compressed " - "artifacts natively without decoding them. Requesting several runs " + "This is the choice that decides query time: the artifact barely " + "matters (under 5%% between N-Triples, HDT and COTTAS for one " + "engine), while engines span three orders of magnitude on identical " + "work. qlever builds an on-disk index and serves it; comunica " + "queries the graph in memory; hdt and cottas query those compressed " + "artifacts natively without decoding them, which is a conformance " + "path rather than a fast one. Requesting several runs " "the whole query set on each, cross-checks their answers, and " "records comparable timings for benchmarking " f"(choices: {', '.join(VALIDATION_ENGINE_CHOICES)}; " From d76ac57d1eb5fb2669f1fe9976c0e321ad070409 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 08:59:21 +0200 Subject: [PATCH 03/18] Refuse a cohort-scale expanded run before it fills the volume The expanded representation emits per sample per record, so its cost is records x samples and is invisible in the input file size: 1000G_phase3_chr20.vcf.gz is 327 MB gzipped and, at 1,812,841 records x 2,504 samples, needs roughly 2 TB against a 189 GB volume. The benchmark harness guarded that in its own shell (bm_skip_if_cohort_scale); the tool did not, so the same file run directly filled the disk with no warning. The guard estimates peak workspace from two coefficients fitted on the sample ladder -- 25 triples per (record x sample), converging to 24.997 by 2504 samples, and 18.3 peak bytes per triple rounded to 20 -- compares it against actual free space, and refuses above 75% of it. The message names the condensed alternative and --allow-cohort-expansion rather than just failing. It only ever refuses expanded. Condensed is ~S + (V x F) and grew 0.9% in total across that same 2504-fold change in samples, so the cohort file still converts in the representation that can hold it. The guard sits ahead of the helper TSVs, which are themselves records x samples work. Recommendation 7 of implementation-improvements-from-benchmarking.md. Co-Authored-By: Claude Opus 5 --- test/test_cohort_guard_unit.py | 222 +++++++++++++++++++++++++++++++++ vcf_rdfizer.py | 126 +++++++++++++++++++ 2 files changed, 348 insertions(+) create mode 100644 test/test_cohort_guard_unit.py diff --git a/test/test_cohort_guard_unit.py b/test/test_cohort_guard_unit.py new file mode 100644 index 0000000..dfd574c --- /dev/null +++ b/test/test_cohort_guard_unit.py @@ -0,0 +1,222 @@ +"""Refusing an expanded run that cannot fit, before it starts filling the disk. + +The expanded representation emits per sample per record, so its cost is +records x samples and is invisible in the input file size: +``1000G_phase3_chr20.vcf.gz`` is 327 MB gzipped and, at 1,812,841 records x +2,504 samples, needs roughly 2 TB. The benchmark harness guarded that case in +its own shell; the tool did not, so a user running the same file directly got +no warning and filled the volume. + +The coefficients are fitted, not guessed -- see the constants -- and the guard +is deliberately one-sided: it only ever refuses ``expanded``, because condensed +grew 0.9% in total across the same 1-to-2504 sample ladder. +""" + +import csv +import shutil +import sys +import tempfile +import unittest +from contextlib import redirect_stderr, redirect_stdout +from io import StringIO +from pathlib import Path +from unittest import mock + +import vcf_rdfizer +from test.helpers import VerboseTestCase + + +def write_records_tsv(path: Path, *, sample_ids: list[str], records: int) -> Path: + """A records TSV with the column layout SampleRecordStream expects.""" + header = [ + "SOURCE_FILE", "ROW_ID", "CHROM", "POS", "ID", "REF", "ALT", + "QUAL", "FILTER", "INFO", "FORMAT", " ".join(sample_ids) or "SAMPLES", + ] + with path.open("w", newline="", encoding="utf-8") as handle: + writer = csv.writer(handle, delimiter="\t") + writer.writerow(header) + for index in range(records): + writer.writerow([ + "in.vcf", f"r{index}", "20", str(100 + index), ".", "A", "G", + "50", "PASS", "DP=30", "GT:DP", + " ".join("0/1:30" for _ in sample_ids), + ]) + return path + + +class EstimateTests(VerboseTestCase): + def test_the_estimate_is_the_product_of_records_samples_and_the_fits(self): + self.assertEqual( + vcf_rdfizer.estimate_expanded_workspace_bytes(10_000, 2_504), + 10_000 + * 2_504 + * vcf_rdfizer.EXPANDED_TRIPLES_PER_SAMPLE_CALL + * vcf_rdfizer.EXPANDED_PEAK_BYTES_PER_TRIPLE, + ) + + def test_the_fitted_coefficient_reproduces_the_measured_ladder_rung(self): + """s2504 x 10,000 records emitted 627,372,018 triples; the fit must be close.""" + measured_expanded_minus_condensed = 627_372_018 - 1_452_018 + predicted = ( + 10_000 * 2_504 * vcf_rdfizer.EXPANDED_TRIPLES_PER_SAMPLE_CALL + ) + relative_error = abs(predicted - measured_expanded_minus_condensed) / ( + measured_expanded_minus_condensed + ) + self.assertLess(relative_error, 0.01, f"predicted {predicted:,}") + + def test_a_degenerate_shape_estimates_zero_rather_than_raising(self): + self.assertEqual(vcf_rdfizer.estimate_expanded_workspace_bytes(0, 2_504), 0) + self.assertEqual(vcf_rdfizer.estimate_expanded_workspace_bytes(-5, 2_504), 0) + + +class RecordsTsvReadingTests(VerboseTestCase): + def test_sample_columns_are_read_from_the_tsv_header(self): + with tempfile.TemporaryDirectory() as tmp: + path = write_records_tsv( + Path(tmp) / "r.tsv", sample_ids=["NA1", "NA2", "NA3"], records=2 + ) + self.assertEqual(vcf_rdfizer.read_records_tsv_sample_count(path), 3) + + def test_a_sample_free_tsv_reports_zero_columns(self): + with tempfile.TemporaryDirectory() as tmp: + path = write_records_tsv(Path(tmp) / "r.tsv", sample_ids=[], records=2) + self.assertEqual(vcf_rdfizer.read_records_tsv_sample_count(path), 0) + + def test_rows_are_counted_without_parsing_fields(self): + with tempfile.TemporaryDirectory() as tmp: + path = write_records_tsv( + Path(tmp) / "r.tsv", sample_ids=["NA1"], records=37 + ) + self.assertEqual(vcf_rdfizer.count_records_tsv_rows(path), 37) + + def test_an_unreadable_tsv_reports_zero_samples_instead_of_raising(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "missing.tsv" + self.assertEqual(vcf_rdfizer.read_records_tsv_sample_count(path), 0) + + +class GuardDecisionTests(VerboseTestCase): + def _refusal(self, tmp, *, sample_ids, records, free_bytes, allow=False, + representation="expanded"): + records_tsv = write_records_tsv( + Path(tmp) / "r.tsv", sample_ids=sample_ids, records=records + ) + usage = shutil.disk_usage(tmp) + fake = type(usage)(usage.total, usage.total - free_bytes, free_bytes) + with mock.patch.object(vcf_rdfizer.shutil, "disk_usage", return_value=fake): + return vcf_rdfizer.cohort_scale_refusal( + records_tsv=records_tsv, + sample_representation=representation, + out_dir=Path(tmp), + allow=allow, + ) + + def test_a_cohort_expanded_run_that_cannot_fit_is_refused(self): + with tempfile.TemporaryDirectory() as tmp: + refusal = self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(200)], + records=50, + free_bytes=1_000_000, + ) + self.assertIsNotNone(refusal) + self.assertIn("expanded", refusal) + + def test_the_refusal_names_the_condensed_alternative_and_the_override(self): + """A refusal that does not say what to do instead is just a failure.""" + with tempfile.TemporaryDirectory() as tmp: + refusal = self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(200)], + records=50, + free_bytes=1_000_000, + ) + self.assertIn("--sample-representation condensed", refusal) + self.assertIn("--allow-cohort-expansion", refusal) + self.assertIn("records x samples", refusal) + + def test_condensed_is_never_refused(self): + """Condensed moved 0.9% over a 2504-fold change in samples.""" + with tempfile.TemporaryDirectory() as tmp: + self.assertIsNone( + self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(200)], + records=50, + free_bytes=1, + representation="condensed", + ) + ) + + def test_a_single_sample_file_is_never_refused(self): + """Single-sample expanded is the ordinary case and must stay unguarded.""" + with tempfile.TemporaryDirectory() as tmp: + self.assertIsNone( + self._refusal(tmp, sample_ids=["NA1"], records=100_000, free_bytes=1) + ) + + def test_a_run_that_fits_proceeds(self): + with tempfile.TemporaryDirectory() as tmp: + self.assertIsNone( + self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(4)], + records=10, + free_bytes=10 * 1024 ** 3, + ) + ) + + def test_the_override_lets_a_refused_run_proceed(self): + with tempfile.TemporaryDirectory() as tmp: + self.assertIsNone( + self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(200)], + records=50, + free_bytes=1_000_000, + allow=True, + ) + ) + + def test_the_guard_leaves_a_margin_rather_than_filling_the_volume(self): + """Needing exactly the free space is refused; the share is below 1.0.""" + self.assertLess(vcf_rdfizer.COHORT_GUARD_FREE_SPACE_SHARE, 1.0) + with tempfile.TemporaryDirectory() as tmp: + samples, records = 200, 50 + needed = vcf_rdfizer.estimate_expanded_workspace_bytes(records, samples) + self.assertIsNotNone( + self._refusal( + tmp, + sample_ids=[f"NA{i}" for i in range(samples)], + records=records, + free_bytes=needed, + ) + ) + + +class CliTests(VerboseTestCase): + def test_the_override_flag_is_accepted_by_the_cli(self): + """It must parse; without it the escape hatch is unreachable.""" + with redirect_stdout(StringIO()), redirect_stderr(StringIO()): + with mock.patch.object( + sys, "argv", ["vcf_rdfizer.py", "--allow-cohort-expansion", "--help"] + ), self.assertRaises(SystemExit) as exc: + vcf_rdfizer.main() + self.assertEqual(exc.exception.code, 0) + + def test_the_help_text_explains_why_the_guard_exists(self): + """A flag whose help says only what it toggles teaches nothing.""" + buffer = StringIO() + with redirect_stdout(buffer), redirect_stderr(StringIO()): + with mock.patch.object( + sys, "argv", ["vcf_rdfizer.py", "--help"] + ), self.assertRaises(SystemExit): + vcf_rdfizer.main() + help_text = buffer.getvalue() + self.assertIn("--allow-cohort-expansion", help_text) + self.assertIn("records x samples", help_text) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 5f6f71a..4b8d743 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -345,6 +345,103 @@ # line starting with one of these bytes and ending in " ." is a statement. _NTRIPLES_SUBJECT_STARTS = (b"<", b"_") SAMPLE_REPRESENTATION_CHOICES = {"expanded", "condensed"} + +#: How many triples the expanded representation adds per (record x sample). +#: Fitted from the sample ladder in 03_sample_representation, which re-emitted +#: the same 10,000 variant records against 1, 4, 16, 64, 256, 1024 and 2504 +#: sample columns: the expanded-minus-condensed difference divided by +#: records x samples converges to 24.997 by 2504 samples (25.0 at 1024, 24.97 +#: at 256). The condensed arm over that same ladder moved 0.9% in total, which +#: is why only the expanded side needs an estimate at all. +EXPANDED_TRIPLES_PER_SAMPLE_CALL = 25 +#: Peak workspace bytes per emitted triple, measured rather than assumed: the +#: s2504 expanded cell peaked at 11,481,042,944 bytes for 627,372,018 triples, +#: i.e. 18.3. Rounded up, because a guard that under-estimates is no guard. +EXPANDED_PEAK_BYTES_PER_TRIPLE = 20 +#: Refuse when the estimate needs more than this share of the free space that +#: remains. Leaving a quarter of the volume is not generosity: the estimate is +#: a fit, and filling a disk takes down everything else running on the host. +COHORT_GUARD_FREE_SPACE_SHARE = 0.75 + + +def estimate_expanded_workspace_bytes(record_count: int, sample_count: int) -> int: + """Peak bytes the expanded representation will need for this shape. + + Expanded emits per sample per record, so cost is records x samples and the + file size on disk says almost nothing about it: 1000G_phase3_chr20 is 327 MB + gzipped and, at 1,812,841 records x 2,504 samples, needs roughly 2 TB. + """ + return int( + max(0, record_count) + * max(0, sample_count) + * EXPANDED_TRIPLES_PER_SAMPLE_CALL + * EXPANDED_PEAK_BYTES_PER_TRIPLE + ) + + +def count_records_tsv_rows(records_tsv: Path) -> int: + """Data rows in a records TSV, read once in binary without parsing fields.""" + total = 0 + with records_tsv.open("rb") as handle: + while chunk := handle.read(8 * 1024 * 1024): + total += chunk.count(b"\n") + # The header line is not a record. A file whose final row has no trailing + # newline loses one here, which errs toward a smaller estimate by one row. + return max(0, total - 1) + + +def read_records_tsv_sample_count(records_tsv: Path) -> int: + """Sample columns declared by a records TSV, or 0 when it has none.""" + try: + with SampleRecordStream(records_tsv) as stream: + return len(stream.columns) + except (OSError, csv.Error, StopIteration): + return 0 + + +def cohort_scale_refusal( + *, + records_tsv: Path, + sample_representation: str, + out_dir: Path, + allow: bool, +) -> str | None: + """Refuse an expanded run that cannot fit, naming the condensed alternative. + + Only ``expanded`` is affected. Condensed is ~S + (V x F) and stays tractable + at cohort scale -- across a 1-to-2504 sample ladder it grew 0.9% in total -- + so the cohort file still converts, in the representation that can hold it. + + Returns the refusal message, or None when the run may proceed. + """ + if allow or sample_representation != "expanded": + return None + sample_count = read_records_tsv_sample_count(records_tsv) + if sample_count <= 1: + return None + try: + record_count = count_records_tsv_rows(records_tsv) + free_bytes = shutil.disk_usage(out_dir).free + except OSError: + # An unreadable TSV or volume is the pipeline's problem to report, not + # this guard's to guess at. Let the run proceed and fail where it means. + return None + needed = estimate_expanded_workspace_bytes(record_count, sample_count) + budget = int(free_bytes * COHORT_GUARD_FREE_SPACE_SHARE) + if needed <= budget: + return None + return ( + f"--sample-representation expanded needs an estimated " + f"{needed / 1e9:,.1f} GB of workspace for {record_count:,} records x " + f"{sample_count:,} samples ({record_count * sample_count:,} sample " + f"calls), and only {free_bytes / 1e9:,.1f} GB is free on " + f"{out_dir}. Expanded emits per sample per record, so its cost is " + f"records x samples and is not visible in the input file size.\n" + f" Use --sample-representation condensed, which encodes the same " + f"genotypes as S + (V x F) and stays flat in sample count, or pass " + f"--allow-cohort-expansion to proceed anyway and accept the risk of " + f"filling this volume." + ) # How the INFO column is represented. "raw" is the historical behaviour (an # opaque vcfc:infoRaw string). "structured" additionally emits one # vcfc:InfoFieldValue per record and key, plus the allele layer and the @@ -6949,6 +7046,7 @@ def run_full_mode( run_tracker: RunTracker | None = None, linking_manifests: list | None = None, linking_options: dict | None = None, + allow_cohort_expansion: bool = False, ): """Execute full pipeline: per-input TSV -> RDF -> compression -> validation.""" linking_options = dict(linking_options or {}) @@ -7108,6 +7206,21 @@ def fail_current(stage: str, message: str): triplet = triplets_by_prefix[expected_prefix] prefix = triplet["prefix"] + + # Refuse a cohort-scale expanded run before spending anything on it. + # This sits ahead of the helper TSVs deliberately: building those is + # already records x samples work, so a guard after them has let the + # failure mode start. + refusal = cohort_scale_refusal( + records_tsv=triplet["records"], + sample_representation=sample_workflow.representation, + out_dir=out_dir, + allow=allow_cohort_expansion, + ) + if refusal is not None: + fail_current("cohort-scale-guard", refusal) + continue + sample_calls_tsv = tsv_dir / f"{prefix}.sample_calls.tsv" sample_format_tsv = tsv_dir / f"{prefix}.sample_format_values.tsv" try: @@ -8819,6 +8932,18 @@ def main(): "one sample-ordered value vector per FORMAT key (default: expanded)" ), ) + parser.add_argument( + "--allow-cohort-expansion", + action="store_true", + help=( + "Proceed with --sample-representation expanded even when the " + "estimated workspace exceeds the free space on the output volume. " + "Expanded emits per sample per record, so cost is records x samples " + "and is invisible in the input file size: a 327 MB gzipped " + "2,504-sample cohort needs roughly 2 TB. Off by default, because " + "filling the volume takes down whatever else is running on the host" + ), + ) parser.add_argument( "--header-representation", choices=HEADER_REPRESENTATION_CHOICES, @@ -9759,6 +9884,7 @@ def execute_mode(): image_ref=image_ref, out_name=args.out_name, sample_workflow=sample_workflow, + allow_cohort_expansion=args.allow_cohort_expansion, info_representation=args.info_representation, header_representation=args.header_representation, vcf_version=args.vcf_version, From 7e11fdbe0396ceaa40774f88502d2785418663a4 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:02:46 +0200 Subject: [PATCH 04/18] Free the TSV intermediates when the last reader finishes, not at the end The wrapper always removed them. It removed them at the end of the per-input iteration, which is after the representation build -- 90% of end-to-end wall time. On the whole-file HG005 cell that meant a 2.63 GB TSV set whose last reader finished at 00:03:51 stayed on disk for the remaining 14.8 hours, inside a peak workspace of 16.23 GB that the run held within 90% of for 4.47 h. Every TSV consumer -- RMLStreamer, the sample emitter, the header emitter, the record-detail emitter -- has run by the point this now removes them, which is worth about 2.6 GB of that peak. The end-of-iteration sweep stays as an idempotent safety net over the same helper, and --keep-tsv still suppresses both. Recommendation 5 of implementation-improvements-from-benchmarking.md. Co-Authored-By: Claude Opus 5 --- test/test_tsv_cleanup_timing_unit.py | 146 +++++++++++++++++++++++++++ vcf_rdfizer.py | 96 +++++++++++++----- 2 files changed, 216 insertions(+), 26 deletions(-) create mode 100644 test/test_tsv_cleanup_timing_unit.py diff --git a/test/test_tsv_cleanup_timing_unit.py b/test/test_tsv_cleanup_timing_unit.py new file mode 100644 index 0000000..67ba9cb --- /dev/null +++ b/test/test_tsv_cleanup_timing_unit.py @@ -0,0 +1,146 @@ +"""When the TSV intermediates are freed, not merely that they are. + +The wrapper always removed them; it removed them at the end of the per-input +iteration, which is after the representation build. That build is 90% of +end-to-end wall time, so on the whole-file HG005 cell a 2.63 GB TSV set that +RMLStreamer finished reading at 00:03:51 stayed on disk for the remaining +14.8 hours, and peak workspace carried it: 16.23 GB, held within 90% of peak +for 4.47 h. + +Removing them at the point the last reader finishes is worth about 2.6 GB of +that peak. These tests pin the ordering, because "the files are gone when the +run ends" was already true and is not the property that matters. +""" + +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +import vcf_rdfizer +from test.helpers import VerboseTestCase + + +def make_triplet(directory: Path) -> dict: + triplet = {"prefix": "sample"} + for key, name in ( + ("records", "sample.records.tsv"), + ("headers", "sample.header_lines.tsv"), + ("metadata", "sample.file_metadata.tsv"), + ("sample_calls", "sample.sample_calls.tsv"), + ("sample_format_values", "sample.sample_format_values.tsv"), + ): + path = directory / name + path.write_text("x\n", encoding="utf-8") + triplet[key] = path + return triplet + + +class RemoveTsvTripletTests(VerboseTestCase): + def test_every_member_of_the_triplet_is_removed(self): + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + triplet = make_triplet(directory) + ok, failed = vcf_rdfizer.remove_tsv_triplet( + triplet, + tsv_dir=directory, + image_ref="image", + wrapper_log_path=directory / "log", + ) + self.assertTrue(ok) + self.assertIsNone(failed) + for key in ("records", "headers", "metadata", "sample_calls", + "sample_format_values"): + self.assertFalse(triplet[key].exists(), key) + + def test_calling_it_twice_is_a_no_op_not_a_failure(self): + """It runs once before the build and again as an end-of-loop sweep.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + triplet = make_triplet(directory) + kwargs = dict( + tsv_dir=directory, + image_ref="image", + wrapper_log_path=directory / "log", + ) + self.assertEqual( + vcf_rdfizer.remove_tsv_triplet(triplet, **kwargs), (True, None) + ) + self.assertEqual( + vcf_rdfizer.remove_tsv_triplet(triplet, **kwargs), (True, None) + ) + + def test_a_triplet_missing_optional_helper_paths_is_handled(self): + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + triplet = make_triplet(directory) + del triplet["sample_calls"] + triplet["sample_format_values"] = None + ok, failed = vcf_rdfizer.remove_tsv_triplet( + triplet, + tsv_dir=directory, + image_ref="image", + wrapper_log_path=directory / "log", + ) + self.assertTrue(ok) + self.assertIsNone(failed) + + def test_a_removal_failure_names_the_path_that_could_not_be_removed(self): + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + triplet = make_triplet(directory) + with mock.patch.object( + vcf_rdfizer, "remove_file_with_docker_fallback", return_value=False + ): + ok, failed = vcf_rdfizer.remove_tsv_triplet( + triplet, + tsv_dir=directory, + image_ref="image", + wrapper_log_path=directory / "log", + ) + self.assertFalse(ok) + self.assertEqual(failed, triplet["records"]) + + +class CleanupHappensBeforeTheRepresentationBuildTests(VerboseTestCase): + """Source-order check: the cheap, exact way to pin a sequencing property. + + Driving a full pipeline run to observe the ordering would need a container; + what actually has to hold is that the cleanup call precedes the compression + stage in the loop body, and that is directly checkable. + """ + + def _source(self) -> str: + return Path(vcf_rdfizer.__file__).read_text(encoding="utf-8") + + def test_the_tsvs_are_freed_before_the_representation_build_starts(self): + source = self._source() + cleanup_marker = "Every TSV reader has now run" + build_marker = "method_results_by_file: dict[str, dict[str, dict]] = {}" + self.assertIn(cleanup_marker, source) + self.assertIn(build_marker, source) + self.assertLess( + source.index(cleanup_marker), + source.index(build_marker), + "TSV cleanup must precede the representation build, which is 90% " + "of wall time and the whole reason the files lingered", + ) + + def test_the_last_tsv_reader_still_precedes_the_cleanup(self): + """Freeing them one line too early would break the record-detail emitter.""" + source = self._source() + self.assertLess( + source.rindex("header_lines_tsv=triplet[\"headers\"]"), + source.index("Every TSV reader has now run"), + "a TSV reader must not run after the cleanup", + ) + + def test_keep_tsv_still_suppresses_the_early_cleanup(self): + source = self._source() + marker = source.index("Every TSV reader has now run") + window = source[marker:marker + 700] + self.assertIn("if not keep_tsv:", window) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 4b8d743..af50bf2 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -7007,6 +7007,39 @@ def run_partitioned_representation_methods_for_rdf_files( ) +def remove_tsv_triplet( + triplet: dict, + *, + tsv_dir: Path, + image_ref: str, + wrapper_log_path: Path, +) -> tuple[bool, Path | None]: + """Remove one input's TSV intermediates. Returns (ok, first_failed_path). + + Idempotent: a path that is already gone is a success, so this can be called + at the point the TSVs stop being read AND again as an end-of-iteration + sweep without the second call reporting a failure. + """ + for tsv_path in ( + triplet.get("records"), + triplet.get("headers"), + triplet.get("metadata"), + triplet.get("sample_calls"), + triplet.get("sample_format_values"), + ): + if tsv_path is None or not tsv_path.exists(): + continue + if not remove_file_with_docker_fallback( + path=tsv_path, + mount_root=tsv_dir, + mount_point="/data/tsv", + image_ref=image_ref, + wrapper_log_path=wrapper_log_path, + ): + return False, tsv_path + return True, None + + def run_full_mode( *, input_mount_dir: Path, @@ -7567,6 +7600,28 @@ def fail_current(stage: str, message: str): for raw_rdf_path in raw_rdf_files: run_tracker.track_raw_rdf(raw_rdf_path) + # Every TSV reader has now run: RMLStreamer, the sample emitter, the + # header emitter and the record-detail emitter. Free them here rather + # than at the end of the iteration, because what comes next is the + # representation build -- 90% of end-to-end wall time, and the reason a + # 2.63 GB TSV set was sitting on disk for 14.8 hours after being read + # in the first three minutes. Measured on the whole-file HG005 cell, + # this alone takes peak workspace from 16.23 GB to about 13.6 GB. + if not keep_tsv: + tsv_removed, failed_path = remove_tsv_triplet( + triplet, + tsv_dir=tsv_dir, + image_ref=image_ref, + wrapper_log_path=wrapper_log_path, + ) + if not tsv_removed: + fail_current( + "tsv-cleanup", + f"failed to remove intermediate TSV '{failed_path.name}'. " + f"See log: {wrapper_log_path}", + ) + continue + method_results_by_file: dict[str, dict[str, dict]] = {} partitioned_representation_results: dict[str, dict] = {} if selected_methods: @@ -7952,32 +8007,21 @@ def fail_current(stage: str, message: str): print(f" - Final RDF size (no compression): {format_bytes(raw_total_size)}") if not keep_tsv: - # Cleanup only the triplet generated for this input iteration. - tsv_cleanup_failed = False - for tsv_path in ( - triplet["records"], - triplet["headers"], - triplet["metadata"], - triplet.get("sample_calls"), - triplet.get("sample_format_values"), - ): - if tsv_path is None: - continue - if tsv_path.exists(): - if not remove_file_with_docker_fallback( - path=tsv_path, - mount_root=tsv_dir, - mount_point="/data/tsv", - image_ref=image_ref, - wrapper_log_path=wrapper_log_path, - ): - fail_current( - "tsv-cleanup", - f"failed to remove intermediate TSV '{tsv_path.name}'. See log: {wrapper_log_path}", - ) - tsv_cleanup_failed = True - break - if tsv_cleanup_failed: + # Safety net. The triplet is normally freed before the + # representation build above; this catches a path that only + # appeared later, and is a no-op in the ordinary case. + tsv_removed, failed_path = remove_tsv_triplet( + triplet, + tsv_dir=tsv_dir, + image_ref=image_ref, + wrapper_log_path=wrapper_log_path, + ) + if not tsv_removed: + fail_current( + "tsv-cleanup", + f"failed to remove intermediate TSV '{failed_path.name}'. " + f"See log: {wrapper_log_path}", + ) continue if run_tracker is not None: From 55e9760576b179f750e7c8832a11a965e18e8947 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:10:33 +0200 Subject: [PATCH 05/18] Make the representation build's cost attributable The campaign could say the build was 90% of end-to-end wall time and no more. Three quantities were missing, and each blocked a decision: Per-chunk and per-merge timings existed -- StageRunner collected them all along -- and were dropped at the wrapper boundary, so of HDT's 19,480 s on the whole HG005 file the split between chunk conversion and ~8 rounds of pairwise merging was unknown. They are now persisted, bucketed by stage kind, with merge rounds counted per representation. The shared chunk pass was never timed. One decompress-and-chunk feeds every requested representation, and each chunk is unlinked once all have consumed it -- but with no timing, the archived records could not show that. The identical chunk_count and chunk_input_bytes under both methods reads equally as one pass or two. It is one; chunk_stream_seconds now says so, with the clock stopped across the yield so a consumer's build time is never charged to the stream. max_rss_kb was null for both representations, because GNU time is not in the image, leaving the stage taking 90% of the run reporting no memory while the mapping stage at under 0.3% reported 1.3-1.8 GB. A getrusage(RUSAGE_CHILDREN) high-water fallback fills it, tagged with its source. Container-volume peak workspace -- invisible to the host trace, and never to be added to it -- is derived from the free-space samples the runner already takes. Recommendation 4 of implementation-improvements-from-benchmarking.md, and a prerequisite for 1 and 2 of its section 2. Co-Authored-By: Claude Opus 5 --- src/partitioned_compression.py | 144 ++++++++++++++++++++- test/test_build_profile_unit.py | 213 ++++++++++++++++++++++++++++++++ vcf_rdfizer.py | 10 ++ 3 files changed, 364 insertions(+), 3 deletions(-) create mode 100644 test/test_build_profile_unit.py diff --git a/src/partitioned_compression.py b/src/partitioned_compression.py index d015f4e..010ae51 100644 --- a/src/partitioned_compression.py +++ b/src/partitioned_compression.py @@ -19,6 +19,11 @@ import subprocess import sys import time + +try: # POSIX only; the container is Linux, but the module imports on Windows too. + import resource +except ImportError: # pragma: no cover - Windows + resource = None from pathlib import Path @@ -162,6 +167,13 @@ def stream_chunks( "target_chunk_bytes": target_bytes, "min_chunk_bytes": min_bytes, "max_chunk_bytes": max_bytes, + # Wall time spent decompressing the aggregate and writing chunks, + # accumulated across resumptions of the generator. This pass is SHARED: + # one stream feeds every requested representation, and each chunk is + # unlinked once all of them have consumed it. Recording it separately is + # what lets a reader see that the 99 GB read happens once, which the + # per-method totals alone cannot show -- they exclude it entirely. + "chunk_stream_seconds": 0.0, "chunks": [], } @@ -175,9 +187,12 @@ def generate(): record_count = 0 chunk_index = 0 last_progress_offset = 0 + chunk_started = None + stream_seconds = 0.0 def open_chunk(): - nonlocal handle, chunk_path, chunk_size, chunk_start_offset, chunk_start_record, chunk_index + nonlocal handle, chunk_path, chunk_size, chunk_start_offset, chunk_start_record, chunk_index, chunk_started + chunk_started = time.perf_counter() chunk_path = chunk_dir / f"chunk-{chunk_index:05d}.nt" chunk_index += 1 handle = chunk_path.open("wb") @@ -186,7 +201,7 @@ def open_chunk(): chunk_start_record = record_count def close_chunk(): - nonlocal handle, chunk_path, chunk_size + nonlocal handle, chunk_path, chunk_size, chunk_started if handle is None or chunk_path is None: return None handle.close() @@ -199,6 +214,13 @@ def close_chunk(): "end_uncompressed_byte": logical_offset, "record_count": record_count - chunk_start_record, "payload_bytes": chunk_size, + # Time to read and write THIS chunk, so a slow one is + # attributable instead of hidden in a build total. + "write_seconds": ( + time.perf_counter() - chunk_started + if chunk_started is not None + else None + ), } plan["chunks"].append(metadata) plan["chunk_count"] = len(plan["chunks"]) @@ -221,6 +243,7 @@ def close_chunk(): return completed_path, metadata try: + resumed_at = time.perf_counter() for line in iter_rdf_lines(source): if not line.endswith(b"\n"): raise ValueError(f"RDF source contains a non-line-terminated record: {source}") @@ -233,7 +256,12 @@ def close_chunk(): ): completed_chunk = close_chunk() if completed_chunk is not None: + # Stop the clock across the yield: the consumer's build + # time is its own, not this stream's. + stream_seconds += time.perf_counter() - resumed_at + plan["chunk_stream_seconds"] = stream_seconds yield completed_chunk + resumed_at = time.perf_counter() open_chunk() handle.write(line) @@ -258,6 +286,8 @@ def close_chunk(): last_progress_offset = logical_offset completed_chunk = close_chunk() + stream_seconds += time.perf_counter() - resumed_at + plan["chunk_stream_seconds"] = stream_seconds if completed_chunk is not None: yield completed_chunk finally: @@ -425,6 +455,30 @@ def number(pattern: str, integer: bool = False): } +def children_peak_rss_kb() -> int | None: + """Peak RSS of every child process reaped so far, in KB. + + A fallback for the common case where GNU ``time -v`` is not in the image: + ``max_rss_kb`` was null for every hdt and cottas total in the benchmark + campaign, which left the stage taking 90% of wall time reporting no memory + at all -- while the mapping stage, taking under 0.3%, reported 1.3-1.8 GB. + A feasibility claim cannot rest on the wrong stage's number. + + This is a high-water mark across all children, not a per-stage delta, so it + is meaningful as a run maximum and is recorded that way. + """ + if resource is None: + return None + try: + peak = resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss + except (OSError, ValueError): + return None + if not peak: + return None + # Linux reports kilobytes; macOS and the BSDs report bytes. + return int(peak // 1024) if sys.platform == "darwin" else int(peak) + + class StageRunner: """Execute stages and accumulate both detailed and method-level metrics.""" @@ -434,6 +488,21 @@ def __init__(self, work_dir: Path, progress_path: Path | None = None): prepare_progress_path(progress_path) self.stages: list[dict] = [] + def peak_workspace_bytes(self) -> int | None: + """Highest container-volume usage seen across every stage so far. + + The host-side workspace trace cannot see this: the chunk scratch lives + in a Docker volume, and the two numbers must never be added. Recording + it here is what makes the volume half of peak disk measurable at all. + """ + peaks = [ + stage["workspace_total_bytes"] - stage[key] + for stage in self.stages + for key in ("workspace_free_bytes_before", "workspace_free_bytes_after") + if stage.get("workspace_total_bytes") is not None and stage.get(key) is not None + ] + return max(peaks) if peaks else None + def run( self, name: str, @@ -542,6 +611,15 @@ def run( if stderr_tail: result["stderr_tail"] = stderr_tail result.update(parse_time_log(time_path)) + if result.get("max_rss_kb") is None: + # GNU time is absent or did not report. Fall back to the kernel's + # own accounting rather than recording nothing. + fallback_rss = children_peak_rss_kb() + if fallback_rss is not None: + result["max_rss_kb"] = fallback_rss + result["max_rss_source"] = "rusage_children_highwater" + elif result.get("max_rss_source") is None: + result["max_rss_source"] = "gnu_time" self.stages.append({"name": name, **result}) time_path.unlink(missing_ok=True) return result @@ -646,6 +724,59 @@ def merge_pairwise( return (current[0] if current else None), rounds +def build_profile(runner: "StageRunner", plan: dict | None) -> dict: + """Where the representation build's time and disk actually went. + + The benchmark campaign could attribute 90% of end-to-end wall time to "the + representation build" and no further: per-chunk and per-merge timings were + collected by StageRunner and then dropped at the wrapper boundary, the + shared chunk pass was never timed at all, and max_rss_kb was null for both + representations. This turns that opaque block into a breakdown. + """ + buckets: dict[str, dict] = {} + for stage in runner.stages: + name = str(stage.get("name") or "") + if "-merge-r" in name: + kind = f"{name.split('-merge-r', 1)[0]}-merge" + elif name.startswith("hdt-build-"): + kind = "hdt-chunk-build" + elif name.startswith("cottas-build-"): + kind = "cottas-chunk-build" + else: + kind = name + bucket = buckets.setdefault( + kind, {"stage_count": 0, "wall_seconds": 0.0, "max_rss_kb": None} + ) + bucket["stage_count"] += 1 + bucket["wall_seconds"] += float(stage.get("wall_seconds") or 0.0) + rss = stage.get("max_rss_kb") + if rss is not None: + bucket["max_rss_kb"] = max(bucket["max_rss_kb"] or 0, int(rss)) + + merge_rounds: dict[str, int] = {} + for stage in runner.stages: + name = str(stage.get("name") or "") + if "-merge-r" in name: + prefix, _, rest = name.partition("-merge-r") + round_number = int(rest.split("-", 1)[0]) + merge_rounds[prefix] = max(merge_rounds.get(prefix, 0), round_number) + + return { + # Shared across every representation: one decompress, one chunk write. + "chunk_stream_seconds": (plan or {}).get("chunk_stream_seconds"), + "chunk_count": (plan or {}).get("chunk_count"), + "chunk_input_bytes": (plan or {}).get("chunk_input_bytes"), + "by_stage_kind": buckets, + "merge_rounds": merge_rounds, + "peak_volume_workspace_bytes": runner.peak_workspace_bytes(), + "max_rss_kb": max( + (int(stage["max_rss_kb"]) for stage in runner.stages + if stage.get("max_rss_kb") is not None), + default=None, + ), + } + + def main() -> int: args = parse_args() source = Path(args.source) @@ -1132,6 +1263,7 @@ def validate_artifact( "methods": results, "stages": runner.stages, "index_warnings": index_warnings, + "build_profile": build_profile(runner, plan), } ) return 0 @@ -1144,7 +1276,13 @@ def validate_artifact( "one raw chunk plus the in-progress HDT/COTTAS artifacts. Reduce " "--chunk-target-bytes and --chunk-max-bytes, or increase Docker's disk limit." ) - write_result({"exit_code": 1, "methods": results, "stages": runner.stages, "error": error}) + write_result({ + "exit_code": 1, + "methods": results, + "stages": runner.stages, + "build_profile": build_profile(runner, locals().get("plan")), + "error": error, + }) print(f"partitioned compression failed: {error}", file=sys.stderr) return 1 diff --git a/test/test_build_profile_unit.py b/test/test_build_profile_unit.py new file mode 100644 index 0000000..683d481 --- /dev/null +++ b/test/test_build_profile_unit.py @@ -0,0 +1,213 @@ +"""Making the representation build's cost attributable instead of opaque. + +The benchmark campaign could say that HDT and COTTAS construction was 90% of +end-to-end wall time and no more than that. Per-chunk and per-merge timings +were collected by StageRunner and then dropped at the wrapper boundary; the +shared chunk pass was never timed at all; and ``max_rss_kb`` was null for both +representations, so the stage taking 90% of the run reported no memory while +the mapping stage, at under 0.3%, reported 1.3-1.8 GB. + +That gap is not only inconvenient. Reading the archived records, the identical +``chunk_count`` and ``chunk_input_bytes`` under both methods is equally +consistent with one shared chunk pass and with two -- and it is in fact one. +These tests pin the breakdown that makes the difference visible. +""" + +import importlib.util +import tempfile +import unittest +from pathlib import Path + +from test.helpers import VerboseTestCase + +MODULE_PATH = ( + Path(__file__).resolve().parents[1] / "src" / "partitioned_compression.py" +) + + +def load_module(): + spec = importlib.util.spec_from_file_location("partitioned_build_profile", MODULE_PATH) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +P = load_module() + + +class FakeRunner: + """A StageRunner stand-in carrying only what build_profile reads.""" + + def __init__(self, stages): + self.stages = stages + + def peak_workspace_bytes(self): + return P.StageRunner.peak_workspace_bytes(self) + + +def stage(name, wall, *, rss=None, free_before=None, free_after=None, total=None): + record = {"name": name, "wall_seconds": wall} + if rss is not None: + record["max_rss_kb"] = rss + if free_before is not None: + record["workspace_free_bytes_before"] = free_before + if free_after is not None: + record["workspace_free_bytes_after"] = free_after + if total is not None: + record["workspace_total_bytes"] = total + return record + + +class ChunkStreamTimingTests(VerboseTestCase): + def test_the_shared_chunk_pass_is_timed(self): + """One decompress feeds every representation; its cost must be visible.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + source = directory / "graph.nt" + source.write_bytes(b"".join( + f" .\n".encode() for i in range(200) + )) + stream, plan = P.stream_chunks( + source, + directory / "chunks", + target_bytes=256, + min_bytes=64, + max_bytes=4096, + ) + for chunk_path, _metadata in stream: + chunk_path.unlink(missing_ok=True) + self.assertIn("chunk_stream_seconds", plan) + self.assertIsInstance(plan["chunk_stream_seconds"], float) + self.assertGreaterEqual(plan["chunk_stream_seconds"], 0.0) + self.assertGreater(plan["chunk_count"], 1) + + def test_every_chunk_carries_its_own_write_time(self): + """A slow chunk should be attributable, not averaged into a build total.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + source = directory / "graph.nt" + source.write_bytes(b"".join( + f" .\n".encode() for i in range(200) + )) + stream, plan = P.stream_chunks( + source, + directory / "chunks", + target_bytes=256, + min_bytes=64, + max_bytes=4096, + ) + for chunk_path, _metadata in stream: + chunk_path.unlink(missing_ok=True) + self.assertTrue(plan["chunks"]) + for chunk in plan["chunks"]: + self.assertIn("write_seconds", chunk) + self.assertIsNotNone(chunk["write_seconds"]) + + def test_consumer_time_is_not_charged_to_the_stream(self): + """The clock stops across the yield, or every build would look like I/O.""" + with tempfile.TemporaryDirectory() as tmp: + directory = Path(tmp) + source = directory / "graph.nt" + source.write_bytes(b"".join( + f" .\n".encode() for i in range(200) + )) + stream, plan = P.stream_chunks( + source, + directory / "chunks", + target_bytes=256, + min_bytes=64, + max_bytes=4096, + ) + import time as _time + + for chunk_path, _metadata in stream: + _time.sleep(0.02) # stand in for a chunk build + chunk_path.unlink(missing_ok=True) + # Several chunks x 20 ms of consumer time must not appear here. + self.assertLess(plan["chunk_stream_seconds"], 0.02 * plan["chunk_count"]) + + +class BuildProfileTests(VerboseTestCase): + def test_chunk_builds_and_merges_are_bucketed_separately(self): + runner = FakeRunner([ + stage("hdt-build-00000", 1.0), + stage("hdt-build-00001", 2.0), + stage("cottas-build-00000", 4.0), + stage("hdt-merge-r01-00000", 0.5), + stage("hdt-merge-r02-00000", 0.25), + ]) + profile = P.build_profile(runner, {"chunk_stream_seconds": 9.0, "chunk_count": 2}) + buckets = profile["by_stage_kind"] + self.assertEqual(buckets["hdt-chunk-build"]["stage_count"], 2) + self.assertEqual(buckets["hdt-chunk-build"]["wall_seconds"], 3.0) + self.assertEqual(buckets["cottas-chunk-build"]["wall_seconds"], 4.0) + self.assertEqual(buckets["hdt-merge"]["stage_count"], 2) + self.assertEqual(buckets["hdt-merge"]["wall_seconds"], 0.75) + + def test_merge_rounds_are_counted_per_representation(self): + """log2(chunks) rounds of pairwise merging was previously unrecorded.""" + runner = FakeRunner([ + stage("hdt-merge-r01-00000", 1.0), + stage("hdt-merge-r01-00001", 1.0), + stage("hdt-merge-r02-00000", 1.0), + stage("cottas-merge-r01-00000", 1.0), + ]) + profile = P.build_profile(runner, {}) + self.assertEqual(profile["merge_rounds"], {"hdt": 2, "cottas": 1}) + + def test_the_shared_chunk_cost_is_reported_beside_the_per_method_ones(self): + runner = FakeRunner([stage("hdt-build-00000", 1.0)]) + profile = P.build_profile( + runner, {"chunk_stream_seconds": 42.0, "chunk_count": 186, + "chunk_input_bytes": 99_325_167_164} + ) + self.assertEqual(profile["chunk_stream_seconds"], 42.0) + self.assertEqual(profile["chunk_count"], 186) + self.assertEqual(profile["chunk_input_bytes"], 99_325_167_164) + + def test_peak_memory_is_the_maximum_across_stages(self): + runner = FakeRunner([ + stage("hdt-build-00000", 1.0, rss=1_000), + stage("cottas-build-00000", 1.0, rss=8_000), + stage("hdt-merge-r01-00000", 1.0), + ]) + self.assertEqual(P.build_profile(runner, {})["max_rss_kb"], 8_000) + + def test_a_build_with_no_memory_readings_reports_none_not_zero(self): + """Absent is not zero; recording zero would be a false measurement.""" + runner = FakeRunner([stage("hdt-build-00000", 1.0)]) + self.assertIsNone(P.build_profile(runner, {})["max_rss_kb"]) + + def test_a_missing_plan_does_not_break_the_failure_path(self): + """build_profile also runs when the build raised before planning.""" + runner = FakeRunner([stage("hdt-build-00000", 1.0)]) + profile = P.build_profile(runner, None) + self.assertIsNone(profile["chunk_stream_seconds"]) + self.assertIn("by_stage_kind", profile) + + +class VolumeWorkspaceTests(VerboseTestCase): + def test_peak_volume_usage_is_derived_from_the_free_space_samples(self): + """The host trace cannot see the Docker volume; this is the only source.""" + runner = FakeRunner([ + stage("hdt-build-00000", 1.0, free_before=90, free_after=70, total=100), + stage("hdt-build-00001", 1.0, free_before=70, free_after=25, total=100), + ]) + self.assertEqual(P.build_profile(runner, {})["peak_volume_workspace_bytes"], 75) + + def test_stages_without_workspace_samples_report_none(self): + runner = FakeRunner([stage("hdt-build-00000", 1.0)]) + self.assertIsNone(P.build_profile(runner, {})["peak_volume_workspace_bytes"]) + + +class ChildRssFallbackTests(VerboseTestCase): + def test_the_fallback_returns_a_positive_reading_or_none(self): + """GNU time is often absent in the image; this must fill that gap.""" + value = P.children_peak_rss_kb() + if value is not None: + self.assertGreater(value, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index af50bf2..43b8a0f 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -5705,6 +5705,8 @@ def write_raw_compression_metrics_artifact( method_results: dict[str, dict], index_warnings: list[dict] | None = None, auxiliary_stages: dict[str, dict] | None = None, + build_profile: dict | None = None, + build_stages: list[dict] | None = None, ): """Persist the operation-level compression detail for one RDF source.""" safe_output = safe_metrics_name(output_name) @@ -5721,6 +5723,12 @@ def write_raw_compression_metrics_artifact( "compression_methods": ",".join(selected_methods) if selected_methods else "none", "index_warnings": list(index_warnings or []), "auxiliary_stages": dict(auxiliary_stages or {}), + # Where the build's time, memory and container-volume disk went. The + # container collected per-stage detail all along; it was dropped here, + # which is why 90% of end-to-end wall time was attributable no further + # than "compression". + "build_profile": dict(build_profile or {}), + "build_stages": list(build_stages or []), "methods": {}, } @@ -6937,6 +6945,8 @@ def run_containerized_partitioned_representation_methods( selected_methods=methods, method_results=method_results, index_warnings=index_warnings, + build_profile=(payload or {}).get("build_profile"), + build_stages=(payload or {}).get("stages"), ) return True, method_results finally: From 72e6dfacc07a7f64131d292be4b00611a1b5469a Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:17:22 +0200 Subject: [PATCH 06/18] Bundle the SHACL profile and apply it by default below 512 MiB The mutation-score experiment injected 113 targeted corruptions and the query suite detected 96 -- 0.850. Of the 17 it missed, 7 fall into four classes the report itself attributes to the published SHACL profile: corrupt_allele_value, corrupt_record_index, corrupt_value_item_allele and corrupt_sample_index_expanded. Closing them takes the score to 0.912 with no new oracle and no new query, because the check was already written. It had simply never run. --shacl-shapes needed a path into a vocabulary checkout, so zero of the 62 validation runs in the benchmark campaign enabled it. The shapes and the ontology bundle they need for sh:class are now vendored in vcf_rdfizer_data, digest-pinned in VOCABULARY_PROVENANCE.json with a test that re-checks them against a sibling checkout when one is present, and applied by default. The default is size-gated at 512 MiB rather than unconditional: pyshacl loads the whole graph into memory, and an unknown size counts as too large -- skipping a check is recoverable, an OOM mid-run is not. --no-shacl opts out; --shacl-shapes overrides both. Recommendation 1 of implementation-improvements-from-benchmarking.md. Co-Authored-By: Claude Opus 5 --- docs/validation.md | 16 +- pyproject.toml | 8 + test/test_shacl_default_unit.py | 156 ++ vcf_rdfizer.py | 90 + vcf_rdfizer_data/VOCABULARY_PROVENANCE.json | 16 + .../ontology/vcf-core-vocabulary.bundle.ttl | 2459 +++++++++++++++++ vcf_rdfizer_data/shacl/vcf-4.1.shacl.ttl | 87 + vcf_rdfizer_data/shacl/vcf-4.2.shacl.ttl | 87 + vcf_rdfizer_data/shacl/vcf-4.3.shacl.ttl | 87 + vcf_rdfizer_data/shacl/vcf-4.4.shacl.ttl | 78 + vcf_rdfizer_data/shacl/vcf-4.5.shacl.ttl | 78 + .../shacl/vcf-core-consistency.shacl.ttl | 292 ++ .../vcf-core-vocabulary-sparql.shacl.ttl | 212 ++ .../shacl/vcf-core-vocabulary.shacl.ttl | 460 +++ 14 files changed, 4123 insertions(+), 3 deletions(-) create mode 100644 test/test_shacl_default_unit.py create mode 100644 vcf_rdfizer_data/VOCABULARY_PROVENANCE.json create mode 100644 vcf_rdfizer_data/ontology/vcf-core-vocabulary.bundle.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-4.1.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-4.2.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-4.3.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-4.4.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-4.5.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-core-consistency.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-core-vocabulary-sparql.shacl.ttl create mode 100644 vcf_rdfizer_data/shacl/vcf-core-vocabulary.shacl.ttl diff --git a/docs/validation.md b/docs/validation.md index a1e6d37..cabca92 100644 --- a/docs/validation.md +++ b/docs/validation.md @@ -544,9 +544,19 @@ and a conforming graph is reported as violations. In a vocabulary checkout the bundle sits one level up from the shapes, in `ontology/`, and is found automatically; `--shacl-ontology PATH` names it explicitly anywhere else. -It is **off by default**: `pyshacl` loads the whole graph into memory, so it is -suitable for a single-sample graph or a sample of a cohort, not for a -cohort-scale aggregate. The report records the conformance verdict, the exact +It is **on by default for sources at or below 512 MiB**, using shapes vendored +with the package -- no vocabulary checkout needed. `--no-shacl` turns it off; +`--shacl-shapes` overrides both the bundled shapes and the size gate. Above the +gate it is skipped, because `pyshacl` loads the whole graph into memory and a +cohort-scale aggregate would not fit. + +It defaults on because the queries cannot see what it sees. The mutation-score +experiment injected 113 corruptions; the query suite caught 96 (0.850), and 7 of +the 17 it missed fall into four classes this profile already covers -- +`corrupt_allele_value`, `corrupt_record_index`, `corrupt_value_item_allele` and +`corrupt_sample_index_expanded`. That is 0.850 to 0.912 for a check that was +already written and was enabled in none of the 62 validation runs in the +benchmark campaign. The report records the conformance verdict, the exact violation count, the distinct property paths involved, and the first 50 violations; the full text is written alongside it. diff --git a/pyproject.toml b/pyproject.toml index df6a8d0..c02fb52 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -52,3 +52,11 @@ include-package-data = true [tool.setuptools.package-data] "vcf_rdfizer_data.rules" = ["default_rules.ttl"] "vcf_rdfizer_data.linkers" = ["*/linker.ttl", "*/resolver.py", "*/genes.gff3", "*/README.md"] +# Vendored from the published vocabulary so SHACL validation works from an +# installed package with no separate checkout. Digests are pinned in +# vcf_rdfizer_data/VOCABULARY_PROVENANCE.json. +"vcf_rdfizer_data" = [ + "VOCABULARY_PROVENANCE.json", + "shacl/*.shacl.ttl", + "ontology/*.ttl", +] diff --git a/test/test_shacl_default_unit.py b/test/test_shacl_default_unit.py new file mode 100644 index 0000000..a11204d --- /dev/null +++ b/test/test_shacl_default_unit.py @@ -0,0 +1,156 @@ +"""The shape layer, bundled and on by default. + +The mutation-score experiment injected 113 targeted corruptions and the query +suite detected 96 -- a score of 0.850. Of the 17 it missed, 7 fall into four +classes the report itself attributes to the published SHACL profile: +``corrupt_allele_value``, ``corrupt_record_index``, ``corrupt_value_item_allele`` +and ``corrupt_sample_index_expanded``. That profile was already written. It was +enabled in zero of the 62 validation runs in the benchmark campaign, because it +required a vocabulary checkout and an explicit flag. + +So the shapes are vendored and applied by default, size-gated because pyshacl +loads the whole graph into memory. Closing those four classes takes the score +from 0.850 to 0.912 with no new oracle and no new query. +""" + +import hashlib +import json +import tempfile +import unittest +from pathlib import Path + +import vcf_rdfizer +from test.helpers import VerboseTestCase + +REPO_ROOT = Path(vcf_rdfizer.__file__).resolve().parent +DATA_ROOT = REPO_ROOT / "vcf_rdfizer_data" +VOCABULARY_CHECKOUT = REPO_ROOT.parent / "vcf-rdfizer-vocabulary" + + +class BundledAssetTests(VerboseTestCase): + def test_the_default_shapes_are_bundled_with_the_package(self): + """Requiring a separate checkout is why this was never switched on.""" + shapes, ontology = vcf_rdfizer.resolve_default_shacl_shapes(REPO_ROOT) + self.assertIsNotNone(shapes) + self.assertTrue(shapes.is_file()) + self.assertEqual(shapes.name, vcf_rdfizer.DEFAULT_SHACL_SHAPES) + self.assertIsNotNone(ontology, "sh:class needs the class hierarchy") + self.assertTrue(ontology.is_file()) + + def test_the_version_overlays_are_bundled_too(self): + """4.1-4.5 each have their own overlay; shipping only the core is half.""" + for version in ("4.1", "4.2", "4.3", "4.4", "4.5"): + asset = vcf_rdfizer.resolve_bundled_vocabulary_asset( + REPO_ROOT, f"shacl/vcf-{version}.shacl.ttl" + ) + self.assertIsNotNone(asset, version) + + def test_the_consistency_profile_is_bundled(self): + """It is the profile that covers the allele-value and item/raw classes.""" + self.assertIsNotNone( + vcf_rdfizer.resolve_bundled_vocabulary_asset( + REPO_ROOT, "shacl/vcf-core-consistency.shacl.ttl" + ) + ) + + def test_a_missing_asset_resolves_to_none_rather_than_raising(self): + self.assertIsNone( + vcf_rdfizer.resolve_bundled_vocabulary_asset( + REPO_ROOT, "shacl/does-not-exist.ttl" + ) + ) + + def test_the_vendored_copies_match_their_recorded_digests(self): + """Vendoring without a digest is how a copy drifts unnoticed.""" + provenance = json.loads( + (DATA_ROOT / "VOCABULARY_PROVENANCE.json").read_text(encoding="utf-8") + ) + for relative, expected in provenance["files"].items(): + actual = hashlib.sha256((DATA_ROOT / relative).read_bytes()).hexdigest() + self.assertEqual(actual, expected, relative) + + def test_the_vendored_copies_match_the_vocabulary_checkout(self): + """Skipped off a dev machine; the digests above still pin the content.""" + if not VOCABULARY_CHECKOUT.is_dir(): + self.skipTest("no sibling vcf-rdfizer-vocabulary checkout") + provenance = json.loads( + (DATA_ROOT / "VOCABULARY_PROVENANCE.json").read_text(encoding="utf-8") + ) + for relative in provenance["files"]: + upstream = VOCABULARY_CHECKOUT / relative + if not upstream.is_file(): + continue + self.assertEqual( + (DATA_ROOT / relative).read_bytes(), + upstream.read_bytes(), + f"{relative} has drifted from the vocabulary checkout", + ) + + +class SizeGateTests(VerboseTestCase): + def test_a_small_source_gets_shapes_by_default(self): + self.assertTrue(vcf_rdfizer.shacl_default_applies(10 * 1024 * 1024)) + + def test_a_source_at_the_limit_still_gets_shapes(self): + self.assertTrue( + vcf_rdfizer.shacl_default_applies( + vcf_rdfizer.DEFAULT_SHACL_MAX_SOURCE_BYTES + ) + ) + + def test_a_cohort_scale_source_does_not(self): + """pyshacl is in-memory; trying anyway turns a safety net into an OOM.""" + self.assertFalse( + vcf_rdfizer.shacl_default_applies( + vcf_rdfizer.DEFAULT_SHACL_MAX_SOURCE_BYTES + 1 + ) + ) + self.assertFalse(vcf_rdfizer.shacl_default_applies(200 * 1024 ** 3)) + + def test_an_unknown_size_is_treated_as_too_large(self): + """Skipping a check is recoverable; exhausting memory mid-run is not.""" + self.assertFalse(vcf_rdfizer.shacl_default_applies(None)) + + def test_the_gate_is_a_documented_constant_not_a_literal(self): + self.assertGreater(vcf_rdfizer.DEFAULT_SHACL_MAX_SOURCE_BYTES, 0) + + +class CliTests(VerboseTestCase): + def test_the_opt_out_flag_exists_and_explains_what_it_disables(self): + import sys + from contextlib import redirect_stderr, redirect_stdout + from io import StringIO + from unittest import mock + + buffer = StringIO() + with redirect_stdout(buffer), redirect_stderr(StringIO()): + with mock.patch.object( + sys, "argv", ["vcf_rdfizer.py", "--help"] + ), self.assertRaises(SystemExit): + vcf_rdfizer.main() + help_text = buffer.getvalue() + self.assertIn("--no-shacl", help_text) + self.assertIn("--shacl-shapes", help_text) + + +class MutationCoverageTests(VerboseTestCase): + """The four classes this is meant to close, named in the mutation catalogue.""" + + SHAPE_COVERED = ( + "corrupt_allele_value", + "corrupt_record_index", + "corrupt_value_item_allele", + "corrupt_sample_index_expanded", + ) + + def test_the_catalogue_still_attributes_these_classes_to_the_shape_layer(self): + """If a query starts covering one, this assumption needs revisiting.""" + from test import validation_mutations + + source = Path(validation_mutations.__file__).read_text(encoding="utf-8") + for mutation_id in self.SHAPE_COVERED: + self.assertIn(mutation_id, source, mutation_id) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 43b8a0f..e8ba644 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -4781,6 +4781,63 @@ def resolve_default_rules_path(repo_root: Path) -> Path: ) +#: The shape layer catches what the aggregate queries structurally cannot. The +#: mutation-score experiment injected 113 corruptions and the query suite caught +#: 96 (0.850); of the 17 it missed, 7 are covered by shapes that were already +#: published and enabled in zero of the campaign's 62 validation runs -- +#: corrupt_allele_value, corrupt_record_index, corrupt_value_item_allele and +#: corrupt_sample_index_expanded. Turning them on takes the score to 0.912 with +#: no new oracle and no new query. +DEFAULT_SHACL_SHAPES = "vcf-core-vocabulary.shacl.ttl" +DEFAULT_SHACL_ONTOLOGY = "vcf-core-vocabulary.bundle.ttl" +#: pyshacl loads the whole graph into memory, so the default is size-gated +#: rather than unconditional. A fixture or a single-sample graph validates in +#: seconds; a cohort aggregate would not fit, and silently trying would turn a +#: safety net into an OOM. Above this, shapes stay available via --shacl-shapes. +DEFAULT_SHACL_MAX_SOURCE_BYTES = 512 * 1024 * 1024 + + +def resolve_bundled_vocabulary_asset(repo_root: Path, relative: str) -> Path | None: + """Locate one vendored vocabulary asset in a checkout or installed package.""" + local = (repo_root / "vcf_rdfizer_data" / relative).resolve() + if local.is_file(): + return local + try: + packaged = importlib_resources.files("vcf_rdfizer_data").joinpath(relative) + with importlib_resources.as_file(packaged) as packaged_path: + resolved = packaged_path.resolve() + if resolved.is_file(): + return resolved + except (ModuleNotFoundError, FileNotFoundError): + pass + return None + + +def resolve_default_shacl_shapes(repo_root: Path) -> tuple[Path | None, Path | None]: + """The bundled shapes and their ontology bundle, or (None, None).""" + shapes = resolve_bundled_vocabulary_asset( + repo_root, f"shacl/{DEFAULT_SHACL_SHAPES}" + ) + if shapes is None: + return None, None + ontology = resolve_bundled_vocabulary_asset( + repo_root, f"ontology/{DEFAULT_SHACL_ONTOLOGY}" + ) + return shapes, ontology + + +def shacl_default_applies(source_bytes: int | None) -> bool: + """Whether to validate shapes by default for a source of this size. + + Size-gated because pyshacl is in-memory. ``None`` means the size could not + be read, which is treated as too large: skipping a check is recoverable, + exhausting memory mid-run is not. + """ + if source_bytes is None: + return False + return 0 <= source_bytes <= DEFAULT_SHACL_MAX_SOURCE_BYTES + + def docker_image_exists(image: str) -> bool: """Return True when Docker image reference exists locally.""" return run([*docker_cmd_prefix(), "image", "inspect", image]) == 0 @@ -9198,6 +9255,18 @@ def main(): "not scale to a cohort-sized aggregate" ), ) + parser.add_argument( + "--no-shacl", + action="store_true", + help=( + "Skip the bundled SHACL shape layer, which is otherwise applied by " + "default to sources at or below " + f"{DEFAULT_SHACL_MAX_SOURCE_BYTES // (1024 * 1024)} MiB. It catches " + "what the aggregate comparisons structurally cannot -- a value that " + "is counted but never read -- and covers four of the ten mutation " + "classes the query suite misses. --shacl-shapes overrides both" + ), + ) parser.add_argument( "--shacl-ontology", default=None, @@ -9419,6 +9488,27 @@ def main(): ) if candidate.is_file(): shacl_ontology_path = candidate + elif not args.no_shacl: + # Shapes catch what the aggregate comparisons structurally cannot: + # a value that is counted but never read. Four of the ten mutation + # classes the query suite misses are already covered by the + # published profile, which no run in the benchmark campaign + # enabled. Default it on, size-gated, rather than leaving a + # written check permanently unused. + bundled_shapes, bundled_ontology = resolve_default_shacl_shapes(repo_root) + if bundled_shapes is not None: + source_bytes = None + try: + source_for_size = Path(args.rdf) if args.rdf else ( + Path(args.input) if args.input else None + ) + if source_for_size is not None and source_for_size.is_file(): + source_bytes = source_for_size.stat().st_size + except (OSError, TypeError): + source_bytes = None + if shacl_default_applies(source_bytes): + shacl_shapes_path = bundled_shapes + shacl_ontology_path = bundled_ontology chunk_target_bytes = parse_positive_int( args.chunk_target_bytes, name="--chunk-target-bytes" diff --git a/vcf_rdfizer_data/VOCABULARY_PROVENANCE.json b/vcf_rdfizer_data/VOCABULARY_PROVENANCE.json new file mode 100644 index 0000000..80f71c0 --- /dev/null +++ b/vcf_rdfizer_data/VOCABULARY_PROVENANCE.json @@ -0,0 +1,16 @@ +{ + "source_repository": "https://github.com/ecrum19/vcf-rdfizer-vocabulary", + "source_commit": "5bfd19d3c8fc377cd7eda689f3f8cad0d8d9bf8b", + "note": "Vendored so SHACL validation works from an installed package, with no vocabulary checkout. test_shacl_default_unit.py re-checks these digests against a sibling checkout when one is present.", + "files": { + "ontology/vcf-core-vocabulary.bundle.ttl": "20fd2d8820eb3c8d2796175a6e89212bc5ce04f607ab092addc631e4eff2489c", + "shacl/vcf-4.1.shacl.ttl": "5eafc441f2bae5c3b19d80dfc34eff0c2c6cefd61b569eadd38900f4343721f9", + "shacl/vcf-4.2.shacl.ttl": "186131c85a483eaec457b545a9e1d310b6eccc2c9fadc95aaa37b34946d7e226", + "shacl/vcf-4.3.shacl.ttl": "026105bb56660348239e3ada633e5d00a82f74df28b98798e0db2d1906b63470", + "shacl/vcf-4.4.shacl.ttl": "98574513333d95993a401e927bd16ff4dad95b1f52bcf5f039c9e3f57227cafb", + "shacl/vcf-4.5.shacl.ttl": "230b21c61a089dd109a05971de80dcde8915400114e3f1591f12ce30a195c448", + "shacl/vcf-core-consistency.shacl.ttl": "6513ff0aba6b788afe28d558dc9e7062bfdf937ac3e08380e51e12c72d55276f", + "shacl/vcf-core-vocabulary-sparql.shacl.ttl": "c032e1e6bc252cd518dc454373e4b4e20b3ed60b8a843c705b3519149ecb2168", + "shacl/vcf-core-vocabulary.shacl.ttl": "a5788121e84e98a9497e88debeb652e33cabca6d4a40df6cd5db28c6af049da9" + } +} diff --git a/vcf_rdfizer_data/ontology/vcf-core-vocabulary.bundle.ttl b/vcf_rdfizer_data/ontology/vcf-core-vocabulary.bundle.ttl new file mode 100644 index 0000000..af6fa11 --- /dev/null +++ b/vcf_rdfizer_data/ontology/vcf-core-vocabulary.bundle.ttl @@ -0,0 +1,2459 @@ +@prefix vcfc: . +@prefix owl: . +@prefix rdf: . +@prefix rdfs: . +@prefix xsd: . +@prefix dct: . +@prefix dcat: . +@prefix prov: . +@prefix skos: . +@prefix so: . +@prefix faldo: . +@prefix geno: . +@prefix sio: . +@prefix hero: . +@prefix vrs: . +@prefix ican: . +@prefix chebi: . + +################################################################# +# Generated companion-site bundle. Do not edit by hand. +# Source modules remain normative and are listed below. +################################################################# + +# Source: ontology/vcf-core-vocabulary.ttl +# SB/gvar uses https://ican.fr/ as its schema URI; keep prefix for convenience. + +################################################################# +# Ontology header +################################################################# + +vcfc: a owl:Ontology ; + rdfs:label "VCF Core Vocabulary"@en ; + dct:description "Vocabulary for representing the logical VCF 4.5 model in RDF: files, headers, records, alleles, values, genotype data, and VCF-specific structural-variant syntax. It supports both expanded per-sample and condensed, sample-ordered representations, and delegates broader variation semantics through FALDO, VRS, SO, GENO, and ChEBI alignments. BCF 2.2 byte layout is out of scope."@en ; + dct:license ; + dct:creator "Elias Crum" ; + owl:versionIRI ; + owl:versionInfo "2.1.2" ; + dct:modified "2026-09-08"^^xsd:date ; + owl:priorVersion ; + dct:replaces ; + owl:imports , + , + , + ; + rdfs:seeAlso ; + rdfs:seeAlso ; + rdfs:seeAlso ; + rdfs:seeAlso ; + rdfs:seeAlso . + +################################################################# +# Core classes +################################################################# + +vcfc:VCFFile a owl:Class ; + rdfs:label "VCF file"@en ; + rdfs:comment "A Variant Call Format (VCF) file artifact represented as a dataset distribution. In VCF 4.5, the artifact contains meta-information lines, one column-header line, and zero or more tab-delimited data records."@en ; + vcfc:iriTemplate "file://{vcfFilePath}" ; + skos:exactMatch hero:VCFFile ; + rdfs:seeAlso hero:VCFFile ; + rdfs:subClassOf dcat:Distribution, prov:Entity . + +vcfc:VCFHeader a owl:Class ; + rdfs:label "VCF header"@en ; + rdfs:comment "A resource that groups the VCF file's meta-information header lines. VCF 4.5 meta-information lines begin with '##' and occur before the mandatory '#CHROM' column-header line and all data records."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#header" . + +vcfc:HeaderLine a owl:Class ; + rdfs:label "VCF header line"@en ; + rdfs:comment "A representation of one VCF meta-information line, such as '##fileformat=VCFv4.5' or '##INFO=<...>'. A line may have an unstructured value or a structured, angle-bracketed list of key-value pairs."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#header/line/{lineId}" . + +vcfc:FileFormatHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "fileformat header line"@en ; + rdfs:comment "The required, first VCF meta-information line. Its value declares the VCF version, for example 'VCFv4.5'."@en . + +vcfc:FileDateHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "fileDate header line"@en ; + rdfs:comment "An optional producer-supplied '##fileDate' meta-information line. It is a commonly used unstructured header convention, rather than a required VCF 4.5 line."@en . + +vcfc:SourceHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "source header line"@en ; + rdfs:comment "An optional producer-supplied '##source' meta-information line, typically identifying the program, pipeline, or source that generated the VCF."@en . + +vcfc:ReferenceHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "reference header line"@en ; + rdfs:comment "An optional '##reference' meta-information line that identifies the reference genome or sequence resource used to interpret the records. Including reference and contig metadata is recommended by VCF 4.5."@en . + +vcfc:ContigHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine ; + rdfs:label "contig header line"@en ; + rdfs:comment "A structured '##contig=<...>' declaration for a reference sequence used by records in the file. VCF 4.5 requires its ID and reserves optional attributes such as length, md5, and URL; contig declarations are recommended in VCF and required in BCF."@en ; + # 2.0.0: was skos:relatedMatch faldo:Reference, which does not exist -- FALDO + # declares the object property faldo:reference and no Reference class. + skos:relatedMatch vrs:SequenceReference ; + rdfs:seeAlso faldo:reference, vrs:SequenceReference . + +vcfc:INFOHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:InfoFieldDefinition ; + rdfs:label "INFO header line"@en ; + rdfs:comment "A structured '##INFO=<...>' declaration that defines the identifier, cardinality, type, and meaning of an annotation that may occur in the INFO column. ID, Number, Type, and Description are required by VCF 4.5."@en . + +vcfc:FORMATHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:FormatFieldDefinition ; + rdfs:label "FORMAT header line"@en ; + rdfs:comment "A structured '##FORMAT=<...>' declaration that defines the identifier, cardinality, type, and meaning of a field in a per-sample column. ID, Number, Type, and Description are required by VCF 4.5."@en . + +vcfc:FILTERHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:FilterDefinition ; + rdfs:label "FILTER header line"@en ; + rdfs:comment "A structured '##FILTER=<...>' declaration that defines a code which may appear in the FILTER column when a record does not pass that filter. ID and Description are required by VCF 4.5."@en . + +vcfc:ALTHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:AltDefinition ; + rdfs:label "ALT header line"@en ; + rdfs:comment "A structured '##ALT=<...>' declaration that defines the meaning of a symbolic alternate-allele identifier. The identifier is used in the ALT column inside angle brackets, for example ''."@en . + +vcfc:VCFRecord a owl:Class ; + rdfs:label "VCF record"@en ; + rdfs:comment "One tab-delimited VCF data line. It contains the eight mandatory fixed fields CHROM, POS, ID, REF, ALT, QUAL, FILTER, and INFO; it may additionally contain a FORMAT field and one value block per sample. This is a VCF-centric representation that can link to a sequence-alteration model via vcfc:asSequenceAlteration."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#record/{recordKey}" ; + skos:exactMatch hero:VCFRecord ; + skos:relatedMatch faldo:Region, vrs:SequenceLocation, vrs:Allele ; + rdfs:seeAlso hero:VCFRecord, faldo:Region, vrs:SequenceLocation, vrs:Allele ; + rdfs:subClassOf prov:Entity . + +vcfc:VariantCall a owl:Class ; + rdfs:label "variant call"@en ; + rdfs:comment "A vocabulary-level resource that groups the call assessment carried by a VCF record: QUAL, FILTER, INFO, the FORMAT declaration, and sample genotype data. It is an RDF convenience for the contents of a record, not a separate line type in VCF syntax. Since version 1.1.0, genotype data can be represented either as expanded per-sample calls or through one condensed cohort call matrix."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}" ; + skos:closeMatch hero:GenomicVariation, so:0001060, vrs:Allele ; + rdfs:seeAlso hero:GenomicVariation, so:0001060, vrs:Allele ; + rdfs:subClassOf prov:Entity . + +vcfc:SampleCall a owl:Class ; + rdfs:label "sample call"@en ; + rdfs:comment "The colon-delimited value block for one sample in a genotype-bearing VCF record, represented as structured FORMAT field values. Its field order and meanings are supplied by that record's FORMAT column and corresponding FORMAT declarations. This is the expanded representation retained for single-sample and low-sample VCF files; it MUST NOT be used to represent a vector of values for multiple samples."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#sample/{recordKey}/{sampleId}" ; + skos:closeMatch hero:Genotype ; + skos:relatedMatch hero:Zygosity ; + rdfs:seeAlso hero:Genotype, hero:Zygosity ; + rdfs:subClassOf prov:Entity . + +vcfc:FieldDefinition a owl:Class ; + rdfs:label "field definition"@en ; + rdfs:comment "A reusable specification- or file-level definition of an INFO, FORMAT, FILTER, ALT, or META key. It is not itself a VCF header line: a concrete file declaration uses the corresponding *HeaderLine subclass."@en . + +vcfc:InfoFieldDefinition a owl:Class ; rdfs:subClassOf vcfc:FieldDefinition ; + rdfs:label "INFO field definition"@en ; + rdfs:comment "The definition of one INFO key, including its ID, Number, Type, and human-readable Description. It specifies how that key's site-level annotation values must be interpreted."@en . + +vcfc:FormatFieldDefinition a owl:Class ; rdfs:subClassOf vcfc:FieldDefinition ; + rdfs:label "FORMAT field definition"@en ; + rdfs:comment "The definition of one FORMAT key, including its ID, Number, Type, and human-readable Description. It specifies how the corresponding per-sample values must be interpreted. Since version 1.1.0, it can also declare a condensed vcfc:FormatValueVector through vcfc:declaredBy."@en . + +vcfc:FilterDefinition a owl:Class ; rdfs:subClassOf vcfc:FieldDefinition ; + rdfs:label "FILTER definition"@en ; + rdfs:comment "The definition of one FILTER failure code. A VCF record that fails this criterion cites the code in its semicolon-separated FILTER value."@en . + +vcfc:AltDefinition a owl:Class ; rdfs:subClassOf vcfc:FieldDefinition ; + rdfs:label "ALT definition"@en ; + rdfs:comment "The definition of one symbolic ALT identifier. It provides the meaning of an angle-bracketed alternate allele such as '' without asserting a particular variation model."@en . + +vcfc:InfoFieldValue a owl:Class ; + rdfs:label "INFO field value"@en ; + rdfs:comment "A structured representation of one key and its optional value(s) from the INFO column of a VCF record. INFO entries are separated by semicolons, and an entry without an equals sign represents a Flag-style presence assertion."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}/info/{fieldKey}" ; + skos:relatedMatch hero:VariantLevelData ; + rdfs:seeAlso hero:VariantLevelData . + +vcfc:FormatFieldValue a owl:Class ; + rdfs:label "FORMAT field value"@en ; + rdfs:comment "A structured representation of one value from one sample's colon-delimited data block, interpreted at the corresponding position in the record's FORMAT key list. This is an expanded, per-sample resource; version 1.1.0 added vcfc:FormatValueVector for a condensed multi-sample representation and the two classes MUST NOT be substituted for one another."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#sample/{recordKey}/{sampleId}/fmt/{fieldKey}" ; + skos:relatedMatch hero:Genotype ; + rdfs:seeAlso hero:Genotype . + +# Cohort-oriented representation additions (version 1.1.0) +vcfc:VCFSample a owl:Class ; + rdfs:label "VCF sample"@en ; + rdfs:comment "Added in version 1.1.0. A reusable representation of one named sample column in one VCF file. It identifies a VCF column, not necessarily a biological individual; external subject or biosample resources may be linked separately."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#samples/{sampleId}" ; + rdfs:subClassOf prov:Entity . + +vcfc:SampleSet a owl:Class ; + rdfs:label "VCF sample set"@en ; + rdfs:comment "Added in version 1.1.0. The ordered set of sample columns declared by one VCF file. The vcfc:sampleIndex of each member is one-based and determines how values in a condensed representation are associated with samples."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#samples" ; + rdfs:subClassOf prov:Entity . + +vcfc:CohortCallMatrix a owl:Class ; + rdfs:label "cohort call matrix"@en ; + rdfs:comment "Added in version 1.1.0. A condensed representation of all sample genotype data for one VariantCall. It applies to exactly one ordered SampleSet and groups one FormatValueVector for each represented FORMAT key. A matrix does not itself assert an individual RDF statement for every sample value."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}/matrix" ; + rdfs:subClassOf prov:Entity . + +vcfc:FormatValueVector a owl:Class ; + rdfs:label "FORMAT value vector"@en ; + rdfs:comment "Added in version 1.1.0. The values of one FORMAT key across the ordered members of a CohortCallMatrix's SampleSet. It MUST be linked to the corresponding FormatFieldDefinition with vcfc:declaredBy and its encoded values MUST contain one position for each sample, including explicit missing values. It is a compact container, not a collection of individual vcfc:FormatFieldValue assertions."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}/matrix/fmt/{fieldKey}" ; + rdfs:subClassOf prov:Entity . + +vcfc:RepresentationProfile a owl:Class ; + rdfs:label "VCF RDF representation profile"@en ; + rdfs:comment "Added in version 1.1.0. A declared profile that tells consumers whether genotype data are emitted as individual expanded statements or as condensed, sample-ordered vectors."@en . + +vcfc:VectorEncoding a owl:Class ; + rdfs:label "FORMAT value vector encoding"@en ; + rdfs:comment "Added in version 1.1.0. A named encoding for the lexical payload of a FormatValueVector. The encoding is required to interpret vcfc:encodedValues."@en . + +vcfc:ExpandedRepresentation a owl:NamedIndividual, vcfc:RepresentationProfile ; + rdfs:label "expanded representation"@en ; + skos:altLabel "dense representation"@en ; + rdfs:comment "Added in version 1.1.0 as vcfr:DenseRepresentation and renamed in version 2.0.0 to match the terminology used throughout the documentation. The representation profile that expresses genotype data as a vcfc:SampleCall for each sample and a vcfc:FormatFieldValue for each populated sample/FORMAT position. It is intended for single-sample or low-sample VCF files and for use cases requiring direct RDF statements about individual calls."@en . + +vcfc:CondensedRepresentation a owl:NamedIndividual, vcfc:RepresentationProfile ; + rdfs:label "condensed representation"@en ; + rdfs:comment "Added in version 1.1.0. The representation profile that expresses a VariantCall's multi-sample genotype data through a CohortCallMatrix, a shared SampleSet, and FORMAT value vectors. It is intended for large multi-sample VCF files. Individual genotype values remain recoverable by decoding the vectors, but are not independently asserted as RDF triples."@en . + +vcfc:VCFTextVector a owl:NamedIndividual, vcfc:VectorEncoding ; + rdfs:label "VCF text vector encoding"@en ; + rdfs:comment "Added in version 1.1.0. A tab-separated sequence of raw VCF values, ordered by the matrix SampleSet. There is exactly one item per sample; a missing item is represented by the VCF token '.'. Commas within an item remain part of that FORMAT value and MUST NOT be treated as vector separators."@en . + +################################################################# +# Controlled vocab / enums (lightweight) +################################################################# + +vcfc:VCFValueType a owl:Class ; + rdfs:label "VCF value type"@en ; + rdfs:comment "The declared VCF data type of an INFO or FORMAT field. INFO definitions may use Integer, Float, Flag, Character, or String; FORMAT definitions may use all of these except Flag."@en . + +vcfc:IntegerType a owl:NamedIndividual, vcfc:VCFValueType ; + rdfs:label "Integer"@en ; + rdfs:comment "The VCF Integer type: a signed 32-bit integer value."@en . + +vcfc:FloatType a owl:NamedIndividual, vcfc:VCFValueType ; + rdfs:label "Float"@en ; + rdfs:comment "The VCF Float type: a 32-bit IEEE-754 floating-point value, whose textual form may also express INF, INFINITY, or NAN."@en . + +vcfc:FlagType a owl:NamedIndividual, vcfc:VCFValueType ; + rdfs:label "Flag"@en ; + rdfs:comment "The VCF presence-only type for an INFO key. A Flag has no value in the INFO column and its Number is 0; it is not permitted for FORMAT fields."@en . + +vcfc:CharacterType a owl:NamedIndividual, vcfc:VCFValueType ; + rdfs:label "Character"@en ; + rdfs:comment "The VCF Character type, used when a field value is a single character."@en . + +vcfc:StringType a owl:NamedIndividual, vcfc:VCFValueType ; + rdfs:label "String"@en ; + rdfs:comment "The VCF String type, used for a textual value whose permitted content is further constrained by the field in which it occurs."@en . + +vcfc:VCFNumberArity a owl:Class ; + rdfs:label "VCF Number arity"@en ; + rdfs:comment "The cardinality declaration used in a structured INFO or FORMAT definition. Besides a non-negative integer, VCF 4.5 defines A (one value per ALT allele), R (one per allele including REF), G (one per genotype), and '.' (variable, unknown, or unbounded); FORMAT additionally defines LA, LR, LG, P, and M."@en . + +vcfc:Null a rdfs:Datatype ; + rdfs:label "null literal datatype"@en ; + rdfs:comment "Datatype for the explicit VCF missing-value token '.'. VCF 4.5 uses this token whenever a field or allele value is missing; represent it as '.'^^vcfc:Null rather than as a plain string or a fabricated numeric value."@en . + +################################################################# +# Object properties +################################################################# + +vcfc:hasHeader a owl:ObjectProperty ; + rdfs:label "has header"@en ; + rdfs:comment "Associates a VCF file artifact with the resource that groups its meta-information header lines."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range vcfc:VCFHeader . + +vcfc:hasHeaderLine a owl:ObjectProperty ; + rdfs:label "has header line"@en ; + rdfs:comment "Associates a VCF header with one of its meta-information lines. This preserves the individual declaration even when its structured attributes are also represented separately."@en ; + rdfs:domain vcfc:VCFHeader ; + rdfs:range vcfc:HeaderLine . + +vcfc:hasRecord a owl:ObjectProperty ; + rdfs:label "has record"@en ; + rdfs:comment "Associates a VCF file with one of its tab-delimited data records."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range vcfc:VCFRecord . + +vcfc:hasCall a owl:ObjectProperty ; + rdfs:label "has call"@en ; + rdfs:comment "Associates a VCF record with its vocabulary-level call resource, which groups QUAL, FILTER, INFO, FORMAT, and per-sample call data."@en ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range vcfc:VariantCall . + +vcfc:hasSampleCall a owl:ObjectProperty ; + rdfs:label "has sample call"@en ; + rdfs:comment "Associates a record-level call with the values for one sample. It is used only when the VCF record includes genotype columns. This property is the expanded-profile alternative to vcfc:hasCallMatrix, not a link to a condensed vector."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range vcfc:SampleCall . + +vcfc:hasSampleSet a owl:ObjectProperty ; + rdfs:label "has sample set"@en ; + rdfs:comment "Added in version 1.1.0. Associates a VCF file with its ordered set of sample columns. The set provides the sample order used by every condensed cohort call matrix in that file."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range vcfc:SampleSet . + +vcfc:hasSample a owl:ObjectProperty ; + rdfs:label "has sample"@en ; + rdfs:comment "Added in version 1.1.0. Associates a SampleSet with a reusable VCF sample-column resource. The member's vcfc:sampleIndex supplies the set order."@en ; + rdfs:domain vcfc:SampleSet ; + rdfs:range vcfc:VCFSample . + +vcfc:forSample a owl:ObjectProperty ; + rdfs:label "for sample"@en ; + rdfs:comment "Added in version 1.1.0. Optionally links an expanded SampleCall to the reusable VCFSample representing its column. This avoids treating repeated vcfc:sampleId literals as the only sample identity."@en ; + rdfs:domain vcfc:SampleCall ; + rdfs:range vcfc:VCFSample . + +vcfc:hasCallMatrix a owl:ObjectProperty ; + rdfs:label "has cohort call matrix"@en ; + rdfs:comment "Added in version 1.1.0. Associates a VariantCall with its condensed multi-sample genotype representation. It is the condensed-profile alternative to emitting one vcfc:hasSampleCall assertion per sample."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range vcfc:CohortCallMatrix . + +vcfc:hasFormatValueVector a owl:ObjectProperty ; + rdfs:label "has FORMAT value vector"@en ; + rdfs:comment "Added in version 1.1.0. Associates a CohortCallMatrix with the ordered values for one FORMAT key across its SampleSet."@en ; + rdfs:domain vcfc:CohortCallMatrix ; + rdfs:range vcfc:FormatValueVector . + +vcfc:appliesToSampleSet a owl:ObjectProperty ; + rdfs:label "applies to sample set"@en ; + rdfs:comment "Added in version 1.1.0. Identifies the ordered SampleSet used to interpret a CohortCallMatrix and all of its FORMAT value vectors."@en ; + rdfs:domain vcfc:CohortCallMatrix ; + rdfs:range vcfc:SampleSet . + +vcfc:hasInfoValue a owl:ObjectProperty ; + rdfs:label "has INFO value"@en ; + rdfs:comment "Associates a record-level call with one structured annotation parsed from its INFO column."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range vcfc:InfoFieldValue . + +vcfc:hasFormatValue a owl:ObjectProperty ; + rdfs:label "has FORMAT value"@en ; + rdfs:comment "Associates a sample call with one structured value parsed from that sample's FORMAT-ordered data block."@en ; + rdfs:domain vcfc:SampleCall ; + rdfs:range vcfc:FormatFieldValue . + +vcfc:declaredBy a owl:ObjectProperty ; + rdfs:label "declared by"@en ; + rdfs:comment "Links an INFO, FORMAT, FILTER, or ALT value/usage to its header definition line. Since version 1.1.0, a FormatValueVector uses this property to identify the FORMAT key that defines every position in the vector."@en ; + rdfs:range vcfc:FieldDefinition . + +vcfc:declaredOnLine a owl:ObjectProperty ; + rdfs:label "declared on header line"@en ; + rdfs:comment "Links a reusable field definition to the concrete HeaderLine that declares it in one VCF file. This keeps specification-level registry definitions distinct from file-local header-line occurrences."@en ; + rdfs:domain vcfc:FieldDefinition ; rdfs:range vcfc:HeaderLine . + +vcfc:chromosome a owl:ObjectProperty ; + rdfs:label "chromosome / contig reference"@en ; + rdfs:comment "Links a record to the declared contig corresponding to its CHROM value. The CHROM column identifies a reference sequence (or, for breakpoint assemblies, an angle-bracketed assembly contig ID); a resolvable external reference may also be aligned to FALDO."@en ; + rdfs:seeAlso vrs:SequenceLocation ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range vcfc:ContigHeaderLine . + +vcfc:asSequenceAlteration a owl:ObjectProperty ; + rdfs:label "as sequence alteration"@en ; + rdfs:comment "Links the VCF-centric record or call to an independently modelled sequence alteration, such as an SB/gvar SequenceAlteration. This relation supports semantic alignment without changing the source VCF representation."@en ; + rdfs:seeAlso vrs:Allele ; + rdfs:range so:0001059 . + +vcfc:representationProfile a owl:ObjectProperty ; + rdfs:label "representation profile"@en ; + rdfs:comment "Added in version 1.1.0. Declares the genotype-data representation used by a VCF RDF graph. A converter MUST use vcfc:ExpandedRepresentation when emitting per-sample call resources and vcfc:CondensedRepresentation when emitting cohort call matrices."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range vcfc:RepresentationProfile . + +vcfc:valueEncoding a owl:ObjectProperty ; + rdfs:label "value encoding"@en ; + rdfs:comment "Added in version 1.1.0. Names the lexical encoding used for a FormatValueVector's vcfc:encodedValues payload."@en ; + rdfs:domain vcfc:FormatValueVector ; + rdfs:range vcfc:VectorEncoding . + +vcfc:missingValuePolicy a owl:AnnotationProperty ; + rdfs:label "missing value policy"@en ; + rdfs:comment "Documents how the VCF missing-value token '.' is represented when source values are transformed to RDF."@en . + +vcfc:iriTemplate a owl:AnnotationProperty ; + rdfs:label "IRI template"@en ; + rdfs:comment "Provides a recommended IRI-minting pattern, using RFC 6570-style placeholders, for instances of a vocabulary class."@en . + +vcfc: vcfc:missingValuePolicy "VCF missing token '.' SHOULD be serialized as literal '.'^^vcfc:Null (do not serialize plain '.'^^xsd:string)."@en . +vcfc: vcfc:iriTemplate "file://{vcfFilePath}" . + +################################################################# +# Datatype properties (VCF metadata + record fields) +################################################################# + +# Header common +vcfc:headerKey a owl:DatatypeProperty ; + rdfs:label "header key"@en ; + rdfs:comment "The meta-information key to the left of '=' in a header line, without the leading '##'; for example, 'fileformat', 'INFO', or 'contig'."@en ; + rdfs:domain vcfc:HeaderLine ; + rdfs:range xsd:string . + +vcfc:headerValue a owl:DatatypeProperty ; + rdfs:label "header value"@en ; + rdfs:comment "The source value to the right of '=' in a VCF meta-information line. For a structured line, this may preserve the angle-bracketed key-value list in addition to its separately represented attributes."@en ; + rdfs:domain vcfc:HeaderLine ; + rdfs:range xsd:string . + +# File-level metadata (common headers) +vcfc:fileFormat a owl:DatatypeProperty ; + rdfs:label "file format"@en ; + rdfs:comment "The VCF version string declared by the required '##fileformat' line, for example 'VCFv4.5'. That line must be the first line of a conforming VCF file."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range xsd:string . + +vcfc:fileDate a owl:DatatypeProperty ; + rdfs:label "file date"@en ; + rdfs:comment "A producer-supplied file date, typically derived from an optional '##fileDate' header line. It records file metadata and is not one of VCF 4.5's required fixed fields."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range xsd:date . + +vcfc:sourceSoftware a owl:DatatypeProperty ; + rdfs:label "source software"@en ; + rdfs:comment "The program, pipeline, or other source named by an optional '##source' header line."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range xsd:string . + +vcfc:referenceGenome a owl:DatatypeProperty ; + rdfs:label "reference genome"@en ; + rdfs:comment "The reference genome or sequence resource identified by optional reference metadata, commonly a '##reference' header value. VCF 4.5 recommends recording reference and contig metadata to make the records interpretable."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range xsd:string . + +vcfc:contigCount a owl:DatatypeProperty ; + rdfs:label "contig count"@en ; + rdfs:comment "The number of contig declarations associated with the file. This is a derived RDF convenience, not a separate VCF header field."@en ; + rdfs:domain vcfc:VCFFile ; + rdfs:range xsd:integer . + +# Contig header details +vcfc:contigId a owl:DatatypeProperty ; + rdfs:label "contig ID"@en ; + rdfs:comment "The required ID attribute of a '##contig=<...>' declaration. It names the reference sequence that may be cited in CHROM and must not be a reserved symbolic-allele name."@en ; + rdfs:domain vcfc:ContigHeaderLine ; + rdfs:range xsd:string . + +vcfc:contigLength a owl:DatatypeProperty ; + rdfs:label "contig length"@en ; + rdfs:comment "The optional 'length' attribute of a contig declaration: the length of the reference sequence in bases."@en ; + rdfs:domain vcfc:ContigHeaderLine ; + rdfs:range xsd:integer . + +vcfc:contigAssembly a owl:DatatypeProperty ; + rdfs:label "contig assembly"@en ; + rdfs:comment "An optional assembly identifier supplied in a contig declaration. It identifies the reference assembly context for the named contig."@en ; + rdfs:domain vcfc:ContigHeaderLine ; + rdfs:range xsd:string . + +vcfc:contigMd5 a owl:DatatypeProperty ; + rdfs:label "contig MD5"@en ; + rdfs:comment "The optional 'md5' attribute of a contig declaration: the lowercase hexadecimal MD5 checksum of the normalized reference sequence, as defined by the SAM specification and referenced by VCF 4.5."@en ; + rdfs:domain vcfc:ContigHeaderLine ; + rdfs:range xsd:string . + +# INFO/FORMAT definition fields +vcfc:fieldId a owl:DatatypeProperty ; + rdfs:label "field ID"@en ; + rdfs:comment "The ID attribute of an INFO or FORMAT field declaration. It is the key used to identify a value in the INFO column or FORMAT-ordered sample data and is unique within its header-line type."@en ; + rdfs:domain vcfc:FieldDefinition ; + rdfs:range xsd:string . + +vcfc:fieldNumber a owl:DatatypeProperty ; + rdfs:label "field Number"@en ; + rdfs:comment "The Number attribute from an INFO or FORMAT definition, retained as a string because VCF permits both integers and symbolic arities. A, R, and G follow REF/ALT or genotype order; '.' means variable, unknown, or unbounded; FORMAT also permits LA, LR, LG, P, and M."@en ; + rdfs:domain vcfc:FieldDefinition ; + rdfs:range xsd:string . + +vcfc:fieldType a owl:ObjectProperty ; + rdfs:label "field Type"@en ; + rdfs:comment "Links an INFO or FORMAT field declaration to its VCF Type. INFO supports Integer, Float, Flag, Character, and String; FORMAT supports all of those except Flag."@en ; + rdfs:domain vcfc:FieldDefinition ; + rdfs:range vcfc:VCFValueType . + +vcfc:fieldDescription a owl:DatatypeProperty ; + rdfs:label "field Description"@en ; + rdfs:comment "The human-readable Description attribute from a structured INFO, FORMAT, FILTER, or ALT header declaration. VCF requires this attribute to be quoted in the source header."@en ; + rdfs:domain vcfc:FieldDefinition ; + rdfs:range xsd:string . + +# FILTER definition +vcfc:filterId a owl:DatatypeProperty ; + rdfs:label "FILTER ID"@en ; + rdfs:comment "The ID of a FILTER declaration: the failure code that can occur in the semicolon-separated FILTER column when the corresponding criterion is not passed."@en ; + rdfs:domain vcfc:FilterDefinition ; + rdfs:range xsd:string . + +# ALT definition +vcfc:altId a owl:DatatypeProperty ; + rdfs:label "ALT ID"@en ; + rdfs:comment "The ID of a symbolic ALT declaration, stored without the angle brackets used when the identifier occurs in the ALT column."@en ; + rdfs:domain vcfc:AltDefinition ; + rdfs:range xsd:string . + +# Record core columns +vcfc:chrom a owl:DatatypeProperty ; + rdfs:label "CHROM"@en ; + rdfs:comment "The required CHROM fixed-field value: an identifier for the reference sequence on which the record is located, or an angle-bracketed assembly-contig ID for a breakpoint assembly. Records for one CHROM occur in a contiguous block."@en ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range xsd:string . + +vcfc:pos a owl:DatatypeProperty ; + rdfs:label "POS"@en ; + rdfs:comment "The required 1-based reference position of the record. It denotes the first base of REF, is numerically ordered within CHROM, and may be 0 or N+1 to indicate a telomere for a contig of length N."@en ; + rdfs:seeAlso vrs:SequenceLocation ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range xsd:integer . + +vcfc:recordId a owl:DatatypeProperty ; + rdfs:label "ID"@en ; + rdfs:comment "The ID fixed-field value: a semicolon-separated list of unique record identifiers, such as dbSNP rs identifiers. When no identifier is available, VCF uses '.', represented here as '.'^^vcfc:Null."@en ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range rdfs:Literal . + +vcfc:ref a owl:DatatypeProperty ; + rdfs:label "REF"@en ; + rdfs:comment "The required reference allele string. It contains one or more A, C, G, T, or N bases and begins at POS. For simple insertions, deletions, and symbolic alleles, VCF uses an adjacent padding base to avoid an empty allele."@en ; + rdfs:seeAlso vrs:LiteralSequenceExpression ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range xsd:string . + +vcfc:alt a owl:DatatypeProperty ; + rdfs:label "ALT"@en ; + rdfs:comment "A lossless raw ALT lexical value. The VCF source serializes ALT as an ordered comma-separated list whose items may be base strings, '*', '.', symbolic IDs, '<*>', or breakend expressions. Repeated vcfc:alt assertions are unsuitable for per-allele annotation or preserving ALT order; use hasAltAllele and alleleIndex for the structural form."@en ; + rdfs:seeAlso vrs:LiteralSequenceExpression ; + rdfs:domain vcfc:VCFRecord ; + rdfs:range xsd:string . + +# Call-level columns +vcfc:qual a owl:DatatypeProperty ; + rdfs:label "QUAL"@en ; + rdfs:comment "The QUAL fixed-field value: the Phred-scaled quality score for the assertion made in ALT. When ALT is '.', it is -10 log10 P(variant); otherwise it is -10 log10 P(no variant). An unknown score is represented by '.'^^vcfc:Null."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range rdfs:Literal . + +vcfc:filter a owl:DatatypeProperty ; + rdfs:label "FILTER"@en ; + rdfs:comment "The FILTER fixed-field value. 'PASS' means the position passed all filters; otherwise it is a semicolon-separated list of declared filter codes that failed. '.'^^vcfc:Null means filters have not been applied."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range rdfs:Literal . + +vcfc:infoRaw a owl:DatatypeProperty ; + rdfs:label "INFO raw"@en ; + rdfs:comment "An optional lossless copy of the INFO fixed-field string. In VCF it is '.' when no annotations are present, or a semicolon-separated series of key[=value[,value...]] entries. Use structured vcfc:InfoFieldValue resources for machine interpretation."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range rdfs:Literal . + +vcfc:FormatKey a owl:Class ; + rdfs:label "declared FORMAT key"@en ; + rdfs:comment "One key of one record's FORMAT column, in column order. VCF 4.5 lets a sample drop trailing fields, so the keys a record declares and the values a sample supplies are different sets. Without this resource the declared list exists only inside the vcfc:formatRaw string, and a consumer could not tell an omitted field from an undeclared one without parsing it."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}/formatKey/{fieldIndex}" . + +vcfc:hasFormatKey a owl:ObjectProperty ; + rdfs:label "has declared FORMAT key"@en ; + rdfs:comment "Associates a record-level call with one key declared in its FORMAT column. The key's position comes from vcfc:fieldIndex and its meaning from vcfc:declaredBy, so the declared order is queryable without reading vcfc:formatRaw."@en ; + rdfs:domain vcfc:VariantCall ; rdfs:range vcfc:FormatKey . + +vcfc:formatRaw a owl:DatatypeProperty ; + rdfs:label "FORMAT raw"@en ; + rdfs:comment "An optional lossless copy of the FORMAT column: the colon-separated, ordered list of keys that determines the meaning of every sample data block in the record. If GT is included, it is the first key; the same field order applies to all samples in the record. Since version 1.1.0, this order also determines how a condensed CohortCallMatrix's FORMAT value vectors are combined for a sample."@en ; + rdfs:domain vcfc:VariantCall ; + rdfs:range xsd:string . + +# Structured INFO/FORMAT values +vcfc:fieldValue a owl:DatatypeProperty ; + rdfs:label "field value"@en ; + rdfs:comment "The literal value copied from one structured INFO or FORMAT entry. It may preserve a comma-separated VCF value list; when a source value is missing, use '.'^^vcfc:Null. The typed value properties may additionally be used where the source type and transformation support them. In the version 1.1.0 condensed profile, use vcfc:encodedValues on a FormatValueVector instead of creating one vcfc:fieldValue assertion per sample."@en ; + rdfs:range rdfs:Literal . + +vcfc:fieldValueInteger a owl:DatatypeProperty ; + rdfs:label "field value (integer)"@en ; + rdfs:comment "An RDF integer representation of a parsed VCF Integer field value. Use vcfc:fieldValue for the original lexical form, lists, or missing values."@en ; + rdfs:range xsd:integer . + +vcfc:fieldValueDecimal a owl:DatatypeProperty ; + rdfs:label "field value (decimal)"@en ; + rdfs:comment "An RDF decimal representation of a parsed finite VCF Float field value. VCF Float may also use INF, INFINITY, or NAN; preserve such lexical values with vcfc:fieldValue rather than coercing them to xsd:decimal."@en ; + rdfs:range xsd:decimal . + +vcfc:fieldValueBoolean a owl:DatatypeProperty ; + rdfs:label "field value (boolean)"@en ; + rdfs:comment "An RDF boolean representation that may be used for the presence semantics of a VCF INFO Flag. In VCF source syntax, a Flag is indicated by a key without an explicit value and has Number=0."@en ; + rdfs:range xsd:boolean . + +# Reusable VCF sample-column identity (version 1.1.0) +vcfc:sampleName a owl:DatatypeProperty ; + rdfs:label "sample name"@en ; + rdfs:comment "Added in version 1.1.0. The exact sample-column identifier from the '#CHROM' header line, attached to a reusable VCFSample. It is distinct from vcfc:sampleId, which remains the identifier literal on an expanded SampleCall."@en ; + rdfs:domain vcfc:VCFSample ; + rdfs:range xsd:string . + +vcfc:sampleIndex a owl:DatatypeProperty ; + rdfs:label "sample index"@en ; + rdfs:comment "Added in version 1.1.0. The one-based position of a VCFSample in its file's '#CHROM' sample-column order. A condensed FormatValueVector position i corresponds to the SampleSet member whose sample index is i."@en ; + rdfs:domain vcfc:VCFSample ; + rdfs:range xsd:positiveInteger . + +# Expanded SampleCall identity +vcfc:sampleId a owl:DatatypeProperty ; + rdfs:label "sample ID"@en ; + rdfs:comment "The sample identifier from the column-header line on an expanded SampleCall. When genotype data is present, VCF places FORMAT after the eight fixed columns and then one uniquely named column per sample. Since version 1.1.0, converters should additionally use vcfc:forSample and VCFSample when reusable sample identity is needed."@en ; + rdfs:domain vcfc:SampleCall ; + rdfs:range xsd:string . + +# Condensed cohort payloads (version 1.1.0) +vcfc:encodedValues a owl:DatatypeProperty ; + rdfs:label "encoded values"@en ; + rdfs:comment "Added in version 1.1.0. The lexical payload of a FormatValueVector. Interpret it only according to vcfc:valueEncoding, the vector's vcfc:declaredBy FORMAT definition, and the SampleSet of its CohortCallMatrix; it is not a sequence of independently asserted RDF values."@en ; + rdfs:domain vcfc:FormatValueVector ; + rdfs:range xsd:string . + +vcfc:sampleDataRaw a owl:DatatypeProperty ; + rdfs:label "sample data raw"@en ; + rdfs:comment "Added in version 1.1.0. An optional lossless textual copy of a record's genotype columns, with individual sample blocks separated by tabs in SampleSet order. It preserves source serialization for a CohortCallMatrix but does not replace its FORMAT value vectors for semantic interpretation."@en ; + rdfs:domain vcfc:CohortCallMatrix ; + rdfs:range xsd:string . + +################################################################# +# Recommended alignments to SB/gvar (non-assertive hints) +################################################################# + +# SB/gvar models sequence alteration as so:0001059 and uses FALDO/GENO/SIO slots. :contentReference[oaicite:2]{index=2} +vcfc:recommendedSequenceAlterationClass a owl:AnnotationProperty ; + rdfs:label "recommended sequence alteration class"@en ; + rdfs:comment "Documents a sequence-alteration class recommended as a target when a VCF-centric record or call is linked to a variation model. It is guidance for alignment and does not reclassify the VCF resource."@en . + +vcfc:recommendedSequenceAlterationClass vcfc:recommendedSequenceAlterationClass so:0001059 . + +################################################################# +# VCF 4.5 additive core layer +################################################################# + +vcfc:StructuredHeaderLine a owl:Class ; rdfs:subClassOf vcfc:HeaderLine ; + rdfs:label "structured header line"@en ; + rdfs:comment "A VCF meta-information line whose value is an angle-bracketed, comma-separated list of key-value attributes. Implementations must not rely on the source attribute order, but may preserve it with vcfc:attributeIndex."@en . + +vcfc:UnstructuredHeaderLine a owl:Class ; rdfs:subClassOf vcfc:HeaderLine ; + rdfs:label "unstructured header line"@en ; + rdfs:comment "A VCF meta-information line whose non-empty value is not an angle-bracketed attribute list."@en . + +vcfc:HeaderAttribute a owl:Class ; + rdfs:label "header attribute"@en ; + rdfs:comment "One key-value attribute from a structured VCF header line. It makes required, optional, and implementation-defined attributes queryable while headerValue retains the source form."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#header/line/{lineId}/attribute/{attributeIndex}" . + +vcfc:ColumnHeaderLine a owl:Class ; + rdfs:label "column header line"@en ; + rdfs:comment "The mandatory tab-delimited #CHROM line. It declares the eight fixed VCF columns and, where genotype data occur, FORMAT followed by ordered sample columns. It is distinct from a ## meta-information HeaderLine."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#header/columns" . + +vcfc:AssemblyHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "assembly header line"@en ; + rdfs:comment "The optional ##assembly line that locates a FASTA file containing breakpoint assemblies referenced by BKPTID."@en . + +vcfc:MetaHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:MetaDefinition ; + rdfs:label "META header line"@en ; + rdfs:comment "A structured ##META declaration that defines an implementation- or file-specific sample attribute key and optional allowed values."@en . + +vcfc:MetaDefinition a owl:Class ; rdfs:subClassOf vcfc:FieldDefinition ; + rdfs:label "META definition"@en ; + rdfs:comment "A ##META declaration represented as a reusable field definition. Its fieldId, fieldArity, and fieldType describe attributes used on ##SAMPLE lines."@en . + +vcfc:SampleHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine ; + rdfs:label "SAMPLE header line"@en ; + rdfs:comment "A structured ##SAMPLE declaration describing a sample or genome mapping through standard or META-declared attributes."@en . + +vcfc:SampleDeclaration a owl:Class ; rdfs:subClassOf prov:Entity ; + rdfs:label "sample declaration"@en ; + rdfs:comment "The declared subject of one ##SAMPLE header line. This is a header-level declaration and is distinct from a VCFSample, which identifies a #CHROM sample column."@en . + +vcfc:PedigreeHeaderLine a owl:Class ; rdfs:subClassOf vcfc:StructuredHeaderLine, vcfc:PedigreeRelation ; + rdfs:label "PEDIGREE header line"@en ; + rdfs:comment "A structured ##PEDIGREE declaration. Its HeaderAttribute resources retain arbitrary ancestor-role attributes; pedigreeMother and pedigreeFather provide direct links when those roles are represented as resources."@en . + +vcfc:PedigreeRelation a owl:Class ; + rdfs:label "pedigree relation"@en ; + rdfs:comment "A parent, ancestor, or clonal-origin relationship declared in VCF pedigree metadata."@en . + +vcfc:PedigreeDBHeaderLine a owl:Class ; rdfs:subClassOf vcfc:UnstructuredHeaderLine ; + rdfs:label "pedigreeDB header line"@en ; + rdfs:comment "The optional ##pedigreeDB line that identifies an external pedigree database."@en . + +vcfc:VCFFloat a rdfs:Datatype ; + rdfs:label "VCF Float"@en ; + rdfs:comment "A VCF Float lexical value: a finite decimal/scientific value or, case-insensitively, INF, INFINITY, or NAN. SHACL supplies the portable lexical check."@en . + +vcfc:VCFInteger a rdfs:Datatype ; + rdfs:label "VCF Integer"@en ; + rdfs:comment "A signed 32-bit VCF Integer excluding -2147483648 through -2147483641, which VCF 4.5 reserves because BCF cannot encode them."@en . + +vcfc:GenotypeString a rdfs:Datatype ; + rdfs:label "genotype string"@en ; + rdfs:comment "A VCF GT lexical value: allele indices or missing calls separated by phased (|) or unphased (/) indicators, optionally with a leading indicator for partial phasing."@en . + +vcfc:BreakendString a rdfs:Datatype ; + rdfs:label "breakend string"@en ; + rdfs:comment "A VCF breakend ALT lexical value. Use a Breakend resource for the queryable structural components while retaining this datatype for the lossless source expression."@en . + +vcfc:ArityPerAlt a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per alternate allele (A)"@en ; vcfc:arityCode "A" . +vcfc:ArityPerAllele a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per allele (R)"@en ; vcfc:arityCode "R" . +vcfc:ArityPerGenotype a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per genotype (G)"@en ; vcfc:arityCode "G" . +vcfc:ArityVariable a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "variable arity (.)"@en ; vcfc:arityCode "." . +vcfc:ArityPerLocalAlt a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per local alternate allele (LA)"@en ; vcfc:arityCode "LA" . +vcfc:ArityPerLocalAllele a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per local allele (LR)"@en ; vcfc:arityCode "LR" . +vcfc:ArityPerLocalGenotype a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per local genotype (LG)"@en ; vcfc:arityCode "LG" . +vcfc:ArityPerGTAllele a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per GT allele (P)"@en ; vcfc:arityCode "P" . +vcfc:ArityPerBaseModification a owl:NamedIndividual, vcfc:VCFNumberArity ; + rdfs:label "one value per base modification (M)"@en ; vcfc:arityCode "M" . + +vcfc:fieldArity a owl:ObjectProperty ; + rdfs:label "field arity"@en ; + rdfs:comment "Links a field definition with a symbolic VCF Number arity. For a fixed count, use fieldNumberInteger; fieldNumber always preserves the exact source token."@en ; + rdfs:domain vcfc:FieldDefinition ; rdfs:range vcfc:VCFNumberArity . + +vcfc:hasAttribute a owl:ObjectProperty ; + rdfs:label "has attribute"@en ; rdfs:domain vcfc:StructuredHeaderLine ; rdfs:range vcfc:HeaderAttribute . +vcfc:hasColumnHeader a owl:ObjectProperty ; + rdfs:label "has column header"@en ; rdfs:domain vcfc:VCFHeader ; rdfs:range vcfc:ColumnHeaderLine . +vcfc:hasGenotypeColumns a owl:ObjectProperty ; + rdfs:label "has genotype columns"@en ; rdfs:domain vcfc:ColumnHeaderLine ; rdfs:range vcfc:VCFSample . +vcfc:declaresSample a owl:ObjectProperty ; + rdfs:label "declares sample"@en ; rdfs:domain vcfc:SampleHeaderLine ; rdfs:range vcfc:SampleDeclaration . +vcfc:pedigreeAncestor a owl:ObjectProperty ; + rdfs:label "pedigree ancestor"@en ; rdfs:domain vcfc:PedigreeRelation ; rdfs:range vcfc:SampleDeclaration . +vcfc:pedigreeMother a owl:ObjectProperty ; rdfs:subPropertyOf vcfc:pedigreeAncestor ; + rdfs:label "pedigree mother"@en ; rdfs:domain vcfc:PedigreeRelation ; rdfs:range vcfc:SampleDeclaration . +vcfc:pedigreeFather a owl:ObjectProperty ; rdfs:subPropertyOf vcfc:pedigreeAncestor ; + rdfs:label "pedigree father"@en ; rdfs:domain vcfc:PedigreeRelation ; rdfs:range vcfc:SampleDeclaration . +vcfc:pedigreeOriginal a owl:ObjectProperty ; rdfs:subPropertyOf vcfc:pedigreeAncestor, prov:wasDerivedFrom ; + rdfs:label "pedigree original"@en ; + rdfs:comment "The clonal Original= relation in a ##PEDIGREE line. It is explicitly a PROV derivation relation; parent and generic ancestor links are not globally forced to that interpretation."@en ; + rdfs:domain vcfc:PedigreeRelation ; rdfs:range vcfc:SampleDeclaration . + +vcfc:arityCode a owl:DatatypeProperty ; + rdfs:label "arity code"@en ; rdfs:domain vcfc:VCFNumberArity ; rdfs:range xsd:string . +vcfc:fieldNumberInteger a owl:DatatypeProperty ; + rdfs:label "field Number integer"@en ; rdfs:domain vcfc:FieldDefinition ; rdfs:range xsd:nonNegativeInteger . +vcfc:lineIndex a owl:DatatypeProperty ; + rdfs:label "line index"@en ; rdfs:domain vcfc:HeaderLine ; rdfs:range xsd:positiveInteger . +vcfc:recordIndex a owl:DatatypeProperty ; + rdfs:label "record index"@en ; rdfs:domain vcfc:VCFRecord ; rdfs:range xsd:positiveInteger . +vcfc:attributeKey a owl:DatatypeProperty ; + rdfs:label "attribute key"@en ; rdfs:domain vcfc:HeaderAttribute ; rdfs:range xsd:string . +vcfc:attributeValue a owl:DatatypeProperty ; + rdfs:label "attribute value"@en ; rdfs:domain vcfc:HeaderAttribute ; rdfs:range xsd:string . +vcfc:attributeIndex a owl:DatatypeProperty ; + rdfs:label "attribute index"@en ; rdfs:domain vcfc:HeaderAttribute ; rdfs:range xsd:positiveInteger . +vcfc:assemblyUrl a owl:DatatypeProperty ; + rdfs:label "assembly URL"@en ; rdfs:domain vcfc:AssemblyHeaderLine ; rdfs:range xsd:anyURI . + +vcfc:AssemblyContig a owl:Class ; + rdfs:label "breakpoint-assembly contig"@en ; + rdfs:comment "A contig held in the external breakpoint-assembly file named by a '##assembly' line, referenced from CHROM as an angle-bracketed ID such as ''. VCF 4.5 fixed field 1 admits this form as an alternative to a reference-sequence name. It is deliberately not a vcfc:ContigHeaderLine: no '##contig' declaration exists for it, and modelling it as one would assert a reference-sequence declaration the file does not make."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#assembly/contig/{assemblyContigId}" . + +vcfc:assemblyContigId a owl:DatatypeProperty ; + rdfs:label "assembly contig ID"@en ; + rdfs:comment "The identifier of a breakpoint-assembly contig with the angle brackets removed, so that '' in CHROM yields 'asm1'. The bracketed source token stays available on vcfc:chrom."@en ; + rdfs:domain vcfc:AssemblyContig ; rdfs:range xsd:string . + +vcfc:chromAssemblyContig a owl:ObjectProperty ; + rdfs:label "CHROM assembly contig"@en ; + rdfs:comment "Associates a record whose CHROM is an angle-bracketed assembly-contig ID with that contig. It is the structured counterpart of vcfc:chromosome, which links a record to a declared reference contig; a record uses one or the other, never both."@en ; + rdfs:domain vcfc:VCFRecord ; rdfs:range vcfc:AssemblyContig . + +vcfc:declaredInAssembly a owl:ObjectProperty ; + rdfs:label "declared in assembly file"@en ; + rdfs:comment "Links a breakpoint-assembly contig to the '##assembly' header line whose URL names the file that holds its sequence. This is the contig's sequence source; the vocabulary does not fetch or model the sequence itself."@en ; + rdfs:domain vcfc:AssemblyContig ; rdfs:range vcfc:AssemblyHeaderLine . +vcfc:pedigreeDbUrl a owl:DatatypeProperty ; + rdfs:label "pedigree database URL"@en ; rdfs:domain vcfc:PedigreeDBHeaderLine ; rdfs:range xsd:anyURI . +vcfc:metaAllowedValue a owl:DatatypeProperty ; + rdfs:label "META allowed value"@en ; rdfs:domain vcfc:MetaDefinition ; rdfs:range xsd:string . +vcfc:ancestorRole a owl:DatatypeProperty ; + rdfs:label "ancestor role"@en ; rdfs:domain vcfc:PedigreeRelation ; rdfs:range xsd:string . +vcfc:contigUrl a owl:DatatypeProperty ; + rdfs:label "contig URL"@en ; rdfs:domain vcfc:ContigHeaderLine ; rdfs:range xsd:anyURI . +vcfc:fieldSource a owl:DatatypeProperty ; + rdfs:label "field source"@en ; rdfs:domain vcfc:INFOHeaderLine ; rdfs:range xsd:string . +vcfc:fieldVersion a owl:DatatypeProperty ; + rdfs:label "field version"@en ; rdfs:domain vcfc:INFOHeaderLine ; rdfs:range xsd:string . +vcfc:percentEncodingPolicy a owl:DatatypeProperty ; + rdfs:label "percent-encoding policy"@en ; rdfs:domain vcfc:VCFFile ; rdfs:range xsd:string . +vcfc:rawValue a owl:DatatypeProperty ; + rdfs:label "raw value"@en ; rdfs:comment "The lossless source lexical value before optional percent decoding."@en ; rdfs:range rdfs:Literal . +vcfc:decodedValue a owl:DatatypeProperty ; + rdfs:label "decoded value"@en ; rdfs:comment "A consumer-oriented value after applying the file's percent-encoding policy; rawValue remains authoritative for round-tripping."@en ; rdfs:range rdfs:Literal . + +# Parsed components retain their source order without changing raw columns. +vcfc:RecordIdentifier a owl:Class ; rdfs:label "record identifier"@en ; + rdfs:comment "One component of the semicolon-separated ID column. The missing dot is represented by no identifier components."@en . +vcfc:hasIdentifier a owl:ObjectProperty ; rdfs:label "has identifier"@en ; + rdfs:domain vcfc:VCFRecord ; rdfs:range vcfc:RecordIdentifier . +vcfc:identifierValue a owl:DatatypeProperty ; rdfs:label "identifier value"@en ; + rdfs:domain vcfc:RecordIdentifier ; rdfs:range xsd:string . +vcfc:componentIndex a owl:DatatypeProperty ; rdfs:label "component index"@en ; + rdfs:comment "One-based position of an ID or FILTER component in its source column."@en ; rdfs:range xsd:positiveInteger . +vcfc:FilterCode a owl:Class ; rdfs:label "filter failure code"@en ; + rdfs:comment "One failed FILTER or FT code. PASS and missing are statuses, not failure codes."@en . +vcfc:hasFilterCode a owl:ObjectProperty ; rdfs:label "has filter failure code"@en ; rdfs:range vcfc:FilterCode . +vcfc:filterCodeValue a owl:DatatypeProperty ; rdfs:label "filter code value"@en ; rdfs:domain vcfc:FilterCode ; rdfs:range xsd:string . +vcfc:declaredByFilter a owl:ObjectProperty ; rdfs:label "declared by filter"@en ; rdfs:domain vcfc:FilterCode ; rdfs:range vcfc:FilterDefinition . +vcfc:FilterStatus a owl:Class ; rdfs:label "filter status"@en . +vcfc:FiltersPassed a owl:NamedIndividual, vcfc:FilterStatus ; rdfs:label "filters passed"@en . +vcfc:FiltersFailed a owl:NamedIndividual, vcfc:FilterStatus ; rdfs:label "filters failed"@en . +vcfc:FiltersNotApplied a owl:NamedIndividual, vcfc:FilterStatus ; rdfs:label "filters not applied"@en . +vcfc:filterStatus a owl:ObjectProperty ; rdfs:label "filter status"@en ; + rdfs:comment "Status of FILTER on a VariantCall or FT on a SampleCall."@en ; rdfs:range vcfc:FilterStatus . +vcfc:fieldIndex a owl:DatatypeProperty ; rdfs:label "field index"@en ; + rdfs:comment "One-based source order of an INFO entry or FORMAT field value within its parent."@en ; rdfs:range xsd:positiveInteger . +vcfc:sampleFieldsRaw a owl:DatatypeProperty ; rdfs:label "raw sample fields"@en ; + rdfs:comment "One expanded sample's original colon-separated fields, retaining omitted trailing fields and empty list tokens."@en ; + rdfs:domain vcfc:SampleCall ; rdfs:range xsd:string . + +################################################################# +# Version-scoped file classes, derived from ontology/versions/registry.json +# by scripts/build-shacl-profiles.py. Edit the registry, not this block. +################################################################# + +# BEGIN GENERATED VERSION CLASSES +vcfc:VCF41File a owl:Class ; rdfs:subClassOf vcfc:VCFFile ; rdfs:label "VCF 4.1 file"@en ; rdfs:comment "An optional explicit VCF 4.1 file type. Use it when the VCF 4.1 SHACL version gate is wanted; VCFFile itself remains version-neutral, and the matching overlay also scopes its rules by fileFormat."@en . + +vcfc:VCF42File a owl:Class ; rdfs:subClassOf vcfc:VCFFile ; rdfs:label "VCF 4.2 file"@en ; rdfs:comment "An optional explicit VCF 4.2 file type. Use it when the VCF 4.2 SHACL version gate is wanted; VCFFile itself remains version-neutral, and the matching overlay also scopes its rules by fileFormat."@en . + +vcfc:VCF43File a owl:Class ; rdfs:subClassOf vcfc:VCFFile ; rdfs:label "VCF 4.3 file"@en ; rdfs:comment "An optional explicit VCF 4.3 file type. Use it when the VCF 4.3 SHACL version gate is wanted; VCFFile itself remains version-neutral, and the matching overlay also scopes its rules by fileFormat."@en . + +vcfc:VCF44File a owl:Class ; rdfs:subClassOf vcfc:VCFFile ; rdfs:label "VCF 4.4 file"@en ; rdfs:comment "An optional explicit VCF 4.4 file type. Use it when the VCF 4.4 SHACL version gate is wanted; VCFFile itself remains version-neutral, and the matching overlay also scopes its rules by fileFormat."@en . + +vcfc:VCF45File a owl:Class ; rdfs:subClassOf vcfc:VCFFile ; rdfs:label "VCF 4.5 file"@en ; rdfs:comment "An optional explicit VCF 4.5 file type. Use it when the VCF 4.5 SHACL version gate is wanted; VCFFile itself remains version-neutral, and the matching overlay also scopes its rules by fileFormat."@en . +# END GENERATED VERSION CLASSES + +# Source: ontology/vcf-core-alleles.ttl +################################################################# +# Alleles and VCF Number-indexed values +################################################################# + + a owl:Ontology ; + rdfs:label "VCF Core allele and value-indexing module"@en ; + dct:description "VCF-specific allele carriers and indexed values for Number=A/R/G/LA/LR/LG/P/M fields."@en ; + owl:imports ; + owl:versionInfo "2.1.2" . + +vcfc:Allele a owl:Class ; + rdfs:label "VCF allele"@en ; + rdfs:comment "An ordered allele participating in one VCF record. alleleIndex is 0 for REF and one-based for ALT, making VCF Number=A and Number=R values directly joinable without parsing the ALT string."@en ; + skos:closeMatch vrs:Allele ; + rdfs:seeAlso vrs:Allele, faldo:location ; + vcfc:iriTemplate "file://{vcfFilePath}#record/{recordKey}/allele/{alleleIndex}" . + +vcfc:ReferenceAllele a owl:Class ; rdfs:subClassOf vcfc:Allele ; + rdfs:label "reference allele"@en ; + rdfs:comment "The record's REF allele, identified by alleleIndex 0."@en . + +vcfc:AltAllele a owl:Class ; rdfs:subClassOf vcfc:Allele ; + rdfs:label "alternate allele"@en ; + rdfs:comment "One ordered ALT allele, identified by a positive alleleIndex in source ALT order."@en . + +vcfc:AlleleKind a owl:Class ; + rdfs:label "VCF allele kind"@en ; + rdfs:comment "A controlled syntactic category for an allele lexical form."@en . + +vcfc:BaseSequenceAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "base-sequence allele"@en ; + rdfs:comment "A literal base sequence in an ALT field."@en . +vcfc:OverlappingDeletionAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "overlapping-deletion allele (*)"@en ; + rdfs:comment "The star ALT allele, marking an allele deleted by an overlapping upstream record."@en . +vcfc:MissingAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "missing allele (.)"@en ; + rdfs:comment "The VCF missing ALT token."@en . +vcfc:SymbolicAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "symbolic allele"@en ; + rdfs:comment "An angle-bracketed symbolic ALT identifier such as ."@en . +vcfc:UnspecifiedAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "unspecified allele"@en ; + rdfs:comment "The <*> or ALT form used to indicate an unspecified alternate allele."@en . +vcfc:BreakendAllele a owl:NamedIndividual, vcfc:AlleleKind ; + rdfs:label "breakend allele"@en ; + rdfs:comment "A bracketed VCF ALT expression describing a breakend."@en . + +vcfc:FieldValueItem a owl:Class ; + rdfs:label "field value item"@en ; + rdfs:comment "One ordered, parsed item from the comma-separated payload of an INFO or FORMAT field. The parent field value retains the original lexical payload for round-tripping."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#call/{recordKey}/field/{fieldKey}/value/{valueIndex}" . + +vcfc:hasAltAllele a owl:ObjectProperty ; + rdfs:label "has alternate allele"@en ; rdfs:domain vcfc:VCFRecord ; rdfs:range vcfc:AltAllele . +vcfc:hasReferenceAllele a owl:ObjectProperty ; + rdfs:label "has reference allele"@en ; rdfs:domain vcfc:VCFRecord ; rdfs:range vcfc:ReferenceAllele . +vcfc:alleleKind a owl:ObjectProperty ; + rdfs:label "allele kind"@en ; rdfs:domain vcfc:Allele ; rdfs:range vcfc:AlleleKind . +vcfc:declaredByAlt a owl:ObjectProperty ; + rdfs:label "declared by ALT definition"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range vcfc:AltDefinition . +vcfc:hasValueItem a owl:ObjectProperty ; + rdfs:label "has value item"@en ; rdfs:range vcfc:FieldValueItem . +vcfc:forAllele a owl:ObjectProperty ; + rdfs:label "for allele"@en ; rdfs:domain vcfc:FieldValueItem ; rdfs:range vcfc:Allele . +vcfc:forBaseModification a owl:ObjectProperty ; + rdfs:label "for base modification"@en ; rdfs:domain vcfc:FieldValueItem ; + rdfs:comment "Links a Number=M value item to the applicable BaseModification resource when that layer is used."@en . + +vcfc:alleleIndex a owl:DatatypeProperty ; + rdfs:label "allele index"@en ; rdfs:domain vcfc:Allele ; rdfs:range xsd:nonNegativeInteger . +vcfc:alleleValue a owl:DatatypeProperty ; + rdfs:label "allele value"@en ; rdfs:domain vcfc:Allele ; rdfs:range rdfs:Literal . +vcfc:valueIndex a owl:DatatypeProperty ; + rdfs:label "value index"@en ; rdfs:domain vcfc:FieldValueItem ; rdfs:range xsd:nonNegativeInteger . +vcfc:itemValue a owl:DatatypeProperty ; + rdfs:label "item value"@en ; rdfs:domain vcfc:FieldValueItem ; rdfs:range rdfs:Literal . +vcfc:forGenotypeIndex a owl:DatatypeProperty ; + rdfs:label "for genotype index"@en ; rdfs:domain vcfc:FieldValueItem ; rdfs:range xsd:nonNegativeInteger . +vcfc:forGTAlleleIndex a owl:DatatypeProperty ; + rdfs:label "for GT allele index"@en ; rdfs:domain vcfc:FieldValueItem ; rdfs:range xsd:nonNegativeInteger . +vcfc:tupleArity a owl:DatatypeProperty ; + rdfs:label "tuple arity"@en ; + rdfs:comment "The fixed grouping width for flattened VCF value lists, such as 2 for confidence intervals and 4 for MEINFO or METRANS."@en ; + rdfs:domain vcfc:FieldValueItem ; rdfs:range xsd:positiveInteger . + +################################################################# +# REF padding-base interpretation (VCF 4.5 fixed field 4) +################################################################# + +# VCF requires a padding base whenever an allele would otherwise be empty, and +# for symbolic alleles. Which base plays that role is an interpretation layered +# on top of REF and ALT, not part of them, so it is carried by a separate +# resource and only asserted where the specification determines it. A shared +# first base is NOT by itself evidence of padding: complex substitutions may +# share a prefix while the specification leaves padding optional, and trimming +# such a prefix would silently assert a claim about the alteration. + +vcfc:PaddingInterpretation a owl:Class ; + rdfs:label "padding-base interpretation"@en ; + rdfs:comment "A derived statement about which base of a record's alleles serves as the VCF padding base, together with the specification rule that determined it. It never replaces or trims vcfc:ref and vcfc:alleleValue, which keep the source strings; a consumer that ignores this resource still sees the record exactly as written."@en ; + vcfc:iriTemplate "file://{vcfFilePath}#record/{recordKey}/padding" . + +vcfc:hasPaddingInterpretation a owl:ObjectProperty ; + rdfs:label "has padding interpretation"@en ; + rdfs:comment "Associates a record with the padding interpretation derived for its alleles. At most one interpretation is asserted per record."@en ; + rdfs:domain vcfc:VCFRecord ; rdfs:range vcfc:PaddingInterpretation . + +vcfc:PaddingRule a owl:Class ; + rdfs:label "padding rule"@en ; + rdfs:comment "The VCF 4.5 provision under which a padding interpretation was derived. Naming the rule keeps the interpretation auditable and makes the undetermined case explicit rather than silent."@en . + +vcfc:SimpleIndelPadding a owl:NamedIndividual, vcfc:PaddingRule ; + rdfs:label "simple insertion or deletion"@en ; + rdfs:comment "One allele is a strict prefix of the other, so without a padding base one allele would be empty. VCF 4.5 requires the base before the variant, and POS denotes that base."@en . + +vcfc:SymbolicAllelePadding a owl:NamedIndividual, vcfc:PaddingRule ; + rdfs:label "symbolic alternate allele"@en ; + rdfs:comment "An ALT allele is an angle-bracketed symbolic ID other than '<*>'. VCF 4.5 requires the padding base and POS denotes the coordinate of the base preceding the polymorphism."@en . + +vcfc:PositionOnePadding a owl:NamedIndividual, vcfc:PaddingRule ; + rdfs:label "contig position one exception"@en ; + rdfs:comment "The variant occurs at position 1 of the contig, where VCF 4.5 requires the base after the variant instead of the base before it."@en . + +vcfc:UnspecifiedAlleleNoPadding a owl:NamedIndividual, vcfc:PaddingRule ; + rdfs:label "unspecified allele, no padding"@en ; + rdfs:comment "The only alternate allele is the unspecified allele '<*>', which VCF 4.5 exempts from the padding requirement: its reference call interval includes the POS base. The padding base count is therefore zero."@en . + +vcfc:PaddingNotDetermined a owl:NamedIndividual, vcfc:PaddingRule ; + rdfs:label "not determined by the specification"@en ; + rdfs:comment "Every allele already has at least one base of its own, so VCF 4.5 permits but does not require a padding base and does not say which base plays that role. No padding claim is made; the shared prefix of a complex substitution falls here."@en . + +vcfc:paddingRule a owl:ObjectProperty ; + rdfs:label "padding rule"@en ; + rdfs:comment "The specification provision that produced this padding interpretation."@en ; + rdfs:domain vcfc:PaddingInterpretation ; rdfs:range vcfc:PaddingRule . + +vcfc:PaddingSide a owl:Class ; + rdfs:label "padding side"@en ; + rdfs:comment "Whether the padding base precedes or follows the variant bases."@en . + +vcfc:PaddingBefore a owl:NamedIndividual, vcfc:PaddingSide ; + rdfs:label "padding before the variant"@en ; + rdfs:comment "The ordinary case: the padding base is the first base of REF and POS denotes it."@en . + +vcfc:PaddingAfter a owl:NamedIndividual, vcfc:PaddingSide ; + rdfs:label "padding after the variant"@en ; + rdfs:comment "The position-one exception: the padding base follows the variant bases."@en . + +vcfc:paddingSide a owl:ObjectProperty ; + rdfs:label "padding side"@en ; + rdfs:comment "Which side of the variant bases the padding base occupies. Asserted only when a padding rule determines it."@en ; + rdfs:domain vcfc:PaddingInterpretation ; rdfs:range vcfc:PaddingSide . + +vcfc:paddingBaseCount a owl:DatatypeProperty ; + rdfs:label "padding base count"@en ; + rdfs:comment "How many bases of the record's alleles serve as padding under the stated rule: one for a required padding base, zero when the rule determines that none is present or when the specification leaves it undetermined."@en ; + rdfs:domain vcfc:PaddingInterpretation ; rdfs:range xsd:nonNegativeInteger . + +vcfc:paddingAnchorPosition a owl:DatatypeProperty ; + rdfs:label "padding anchor position"@en ; + rdfs:comment "The 1-based reference coordinate of the padding base. For vcfc:PaddingBefore this equals the record's POS; for vcfc:PaddingAfter it is the coordinate following the variant bases. Absent when no padding base is asserted."@en ; + rdfs:domain vcfc:PaddingInterpretation ; rdfs:range xsd:integer . + +# Source: ontology/vcf-core-genotypes.ttl +################################################################# +# Genotype, phasing, and local alleles +################################################################# + + a owl:Ontology ; + rdfs:label "VCF Core genotype module"@en ; + dct:description "Parsed GT, phasing, phase-set, local-allele, and sample-filter carriers for VCF 4.5."@en ; + owl:imports , ; + owl:versionInfo "2.1.2" . + +vcfc:Genotype a owl:Class ; + rdfs:label "VCF genotype"@en ; + rdfs:comment "The parsed GT field for one SampleCall. It preserves genotypeString and represents its ordered allele calls and phasing independently."@en ; + skos:closeMatch geno:0000536 ; rdfs:seeAlso vrs:CisPhasedBlock . + +vcfc:GenotypeAlleleCall a owl:Class ; + rdfs:label "genotype allele call"@en ; + rdfs:comment "One allele position in a parsed GT value. callIndex preserves genotype order; isNoCall represents a dot call without inventing an allele."@en . + +vcfc:PhasingStatus a owl:Class ; + rdfs:label "phasing status"@en ; + rdfs:comment "The separator semantics between genotype allele calls."@en . + +vcfc:PhaseSet a owl:Class ; + rdfs:label "phase set"@en ; + rdfs:comment "A parsed phase-set identifier or list carried by PS or PSL, optionally with PSO and PSQ information."@en ; + skos:closeMatch geno:0000886, vrs:CisPhasedBlock . + +vcfc:LocalAlleleSet a owl:Class ; + rdfs:label "local allele set"@en ; + rdfs:comment "The subset of ALT alleles declared as locally relevant to a sample by LAA. Its members keep their record-global alleleIndex values."@en . + +vcfc:Phased a owl:NamedIndividual, vcfc:PhasingStatus ; + rdfs:label "phased"@en ; rdfs:seeAlso geno:0000131 . +vcfc:Unphased a owl:NamedIndividual, vcfc:PhasingStatus ; + rdfs:label "unphased"@en . + +vcfc:hasGenotype a owl:ObjectProperty ; + rdfs:label "has genotype"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range vcfc:Genotype . +vcfc:hasAlleleCall a owl:ObjectProperty ; + rdfs:label "has allele call"@en ; rdfs:domain vcfc:Genotype ; rdfs:range vcfc:GenotypeAlleleCall . +vcfc:calledAllele a owl:ObjectProperty ; + rdfs:label "called allele"@en ; rdfs:domain vcfc:GenotypeAlleleCall ; rdfs:range vcfc:Allele . +vcfc:phasingStatus a owl:ObjectProperty ; + rdfs:label "phasing status"@en ; rdfs:domain vcfc:Genotype ; rdfs:range vcfc:PhasingStatus . +vcfc:inPhaseSet a owl:ObjectProperty ; + rdfs:label "in phase set"@en ; rdfs:domain vcfc:Genotype ; rdfs:range vcfc:PhaseSet . +vcfc:hasLocalAlleleSet a owl:ObjectProperty ; + rdfs:label "has local allele set"@en ; rdfs:domain vcfc:Genotype ; rdfs:range vcfc:LocalAlleleSet . +vcfc:hasLocalAllele a owl:ObjectProperty ; + rdfs:label "has local allele"@en ; rdfs:domain vcfc:LocalAlleleSet ; rdfs:range vcfc:AltAllele . + +vcfc:genotypeString a owl:DatatypeProperty ; + rdfs:label "genotype string"@en ; rdfs:domain vcfc:Genotype ; rdfs:range vcfc:GenotypeString . +vcfc:ploidy a owl:DatatypeProperty ; + rdfs:label "ploidy"@en ; rdfs:domain vcfc:Genotype ; rdfs:range xsd:nonNegativeInteger . +vcfc:callIndex a owl:DatatypeProperty ; + rdfs:label "allele-call index"@en ; rdfs:domain vcfc:GenotypeAlleleCall ; rdfs:range xsd:nonNegativeInteger . +vcfc:isNoCall a owl:DatatypeProperty ; + rdfs:label "is no-call"@en ; rdfs:domain vcfc:GenotypeAlleleCall ; rdfs:range xsd:boolean . +vcfc:phaseSetId a owl:DatatypeProperty ; + rdfs:label "phase-set ID"@en ; rdfs:domain vcfc:PhaseSet ; rdfs:range xsd:string . +vcfc:phaseSetName a owl:DatatypeProperty ; + rdfs:label "phase-set name"@en ; rdfs:domain vcfc:PhaseSet ; rdfs:range xsd:string . +vcfc:phaseSetOrdinal a owl:DatatypeProperty ; + rdfs:label "phase-set ordinal"@en ; rdfs:domain vcfc:PhaseSet ; rdfs:range xsd:integer . +vcfc:phaseSetQuality a owl:DatatypeProperty ; + rdfs:label "phase-set quality"@en ; rdfs:domain vcfc:PhaseSet ; rdfs:range rdfs:Literal . +vcfc:localAlleleIndex a owl:DatatypeProperty ; + rdfs:label "local allele index"@en ; + rdfs:comment "The one-based ALT index written in LAA for a member of a LocalAlleleSet."@en ; + rdfs:domain vcfc:AltAllele ; rdfs:range xsd:positiveInteger . +vcfc:genotypeIndex a owl:DatatypeProperty ; + rdfs:label "genotype index"@en ; + rdfs:comment "The VCF Number=G index for genotype alleles k₁…k_P: Index(k₁…k_P) = Σ C(k_m + m − 1, m)."@en ; + rdfs:range xsd:nonNegativeInteger . +vcfc:sampleFilter a owl:DatatypeProperty ; + rdfs:label "sample filter"@en ; + rdfs:comment "The FT value on one SampleCall, preserving PASS, one or more failure codes, or the VCF missing token."@en ; + rdfs:domain vcfc:SampleCall ; rdfs:range rdfs:Literal . + +vcfc:phaseIndicator a owl:DatatypeProperty ; rdfs:label "preceding phase indicator"@en ; + rdfs:comment "The / or | indicator preceding this GT allele call, including the effective first indicator when omitted in the source. genotypeString preserves whether it was explicit."@en ; + rdfs:domain vcfc:GenotypeAlleleCall ; rdfs:range xsd:string . +vcfc:MixedPhasing a owl:NamedIndividual, vcfc:PhasingStatus ; rdfs:label "mixed phasing"@en ; + rdfs:comment "Both phased and unphased allele indicators occur; phaseIndicator on each allele call supplies the precise semantics."@en . +vcfc:LocalAlleleMembership a owl:Class ; rdfs:label "local allele membership"@en ; + rdfs:comment "An ordered membership in one sample's LocalAlleleSet. The membership, rather than the shared ALT allele, carries the local index."@en . +vcfc:hasLocalAlleleMembership a owl:ObjectProperty ; rdfs:label "has local allele membership"@en ; + rdfs:domain vcfc:LocalAlleleSet ; rdfs:range vcfc:LocalAlleleMembership . +vcfc:localAllele a owl:ObjectProperty ; rdfs:label "local allele"@en ; + rdfs:domain vcfc:LocalAlleleMembership ; rdfs:range vcfc:AltAllele . +vcfc:localIndex a owl:DatatypeProperty ; rdfs:label "local index"@en ; + rdfs:domain vcfc:LocalAlleleMembership ; rdfs:range xsd:positiveInteger . +vcfc:localAlleleIndex owl:deprecated true ; + rdfs:comment "Use localIndex on LocalAlleleMembership for sample-specific ordering. alleleIndex remains the record-global ALT index written in LAA."@en . + +vcfc:allelePhaseSet a owl:ObjectProperty ; rdfs:label "allele phase set"@en ; + rdfs:domain vcfc:GenotypeAlleleCall ; rdfs:range vcfc:PhaseSet ; + rdfs:comment "The PSL entry for this GT allele call; missing entries have no link."@en . + +# Source: ontology/vcf-core-sv.ttl +################################################################# +# Structural variation, repeats, gVCF, and base modifications +################################################################# + + a owl:Ontology ; + rdfs:label "VCF Core structural-variation module"@en ; + dct:description "VCF-specific structural-variant syntax carriers, tandem repeats, copy-number fields, reference blocks, and base modifications."@en ; + owl:imports , + , + ; + owl:versionInfo "2.1.2" . + +vcfc:SymbolicAlleleType a owl:Class ; + rdfs:label "symbolic allele type"@en ; + rdfs:comment "A reserved first-level symbolic ALT type or recommended subtype. It represents VCF syntax and points outward to SO rather than reclassifying an allele as an SO entity."@en . + +vcfc:SVClaim a owl:Class ; + rdfs:label "structural-variant claim"@en ; + rdfs:comment "The per-ALT SVCLAIM code indicating an abundance claim, adjacency claim, or both."@en . + +vcfc:Breakend a owl:Class ; rdfs:subClassOf vcfc:AltAllele ; + rdfs:label "breakend"@en ; + rdfs:comment "A parsed bracketed VCF ALT allele representing a novel adjacency or terminus. The source expression remains in alleleValue."@en ; + skos:relatedMatch so:0001021 ; rdfs:seeAlso vrs:Adjacency, vrs:Terminus . + +vcfc:BreakendOrientation a owl:Class ; + rdfs:label "breakend orientation"@en ; + rdfs:comment "One of the four bracket placements permitted by VCF breakend notation."@en . + +vcfc:VariantEvent a owl:Class ; + rdfs:label "variant event"@en ; + rdfs:comment "A named EVENT grouping records that jointly describe a complex rearrangement or another related variation event."@en . + +vcfc:EventType a owl:Class ; + rdfs:label "event type"@en ; + rdfs:comment "A reserved VCF EVENTTYPE code."@en . + +vcfc:ConfidenceInterval a owl:Class ; + rdfs:label "confidence interval"@en ; + rdfs:comment "A lower and upper VCF interval for CILEN, CICN, CIRUC, or CIRB. CILEN contains length bounds, not POS offsets. CIPOS and CIEND instead use FALDO InRangePosition."@en . + +vcfc:TandemRepeatAllele a owl:Class ; rdfs:subClassOf vcfc:AltAllele ; + rdfs:label "tandem-repeat allele"@en ; + rdfs:comment "A alternate allele whose repeat sequences and units are represented through the VCF tandem-repeat fields."@en . + +vcfc:RepeatSequence a owl:Class ; + rdfs:label "repeat sequence"@en ; + rdfs:comment "One repeat sequence within a tandem-repeat allele, indexed in VCF RN/RUS/RUL/RUC/RB order."@en . + +vcfc:RepeatUnit a owl:Class ; + rdfs:label "repeat unit"@en ; + rdfs:comment "One individually sized repeat unit used only when RUB distinguishes varying unit lengths."@en . + +vcfc:ReferenceBlock a owl:Class ; + rdfs:label "reference block"@en ; + rdfs:comment "A reference interval for one sample, linked to its unspecified ALT allele with blockAllele. Different samples may have different FORMAT LEN extents for the same record."@en . + +vcfc:BaseModification a owl:Class ; + rdfs:label "base modification"@en ; + rdfs:comment "A parsed base-modification carrier for VCF Number=M FORMAT fields. modifiedResidue links directly to the corresponding ChEBI resource; FALDO ForwardStrandPosition, ReverseStrandPosition, and BothStrandsPosition express the VCF strandedness convention without new VCF terms."@en . + +vcfc:SymbolicDeletion a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "DEL symbolic allele type"@en ; vcfc:svTypeCode "DEL" ; skos:exactMatch so:0000159 . +vcfc:SymbolicInsertion a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "INS symbolic allele type"@en ; vcfc:svTypeCode "INS" ; skos:exactMatch so:0000667 . +vcfc:SymbolicDuplication a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "DUP symbolic allele type"@en ; vcfc:svTypeCode "DUP" ; skos:exactMatch so:1000035 . +vcfc:SymbolicInversion a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "INV symbolic allele type"@en ; vcfc:svTypeCode "INV" ; skos:exactMatch so:1000036 . +vcfc:SymbolicCopyNumberVariation a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "CNV symbolic allele type"@en ; vcfc:svTypeCode "CNV" ; skos:exactMatch so:0001019 . +vcfc:SymbolicTandemRepeat a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "CNV:TR symbolic allele subtype"@en ; vcfc:svTypeCode "CNV:TR" ; + vcfc:svSubtypeOf vcfc:SymbolicCopyNumberVariation ; skos:closeMatch so:0000705 ; skos:relatedMatch so:0002096 . +vcfc:SymbolicTandemDuplication a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "DUP:TANDEM symbolic allele subtype"@en ; vcfc:svTypeCode "DUP:TANDEM" ; + vcfc:svSubtypeOf vcfc:SymbolicDuplication ; skos:exactMatch so:1000173 . +vcfc:SymbolicMobileElementDeletion a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "DEL:ME symbolic allele subtype"@en ; vcfc:svTypeCode "DEL:ME" ; + vcfc:svSubtypeOf vcfc:SymbolicDeletion ; skos:exactMatch so:0002066 . +vcfc:SymbolicMobileElementInsertion a owl:NamedIndividual, vcfc:SymbolicAlleleType ; + rdfs:label "INS:ME symbolic allele subtype"@en ; vcfc:svTypeCode "INS:ME" ; + vcfc:svSubtypeOf vcfc:SymbolicInsertion ; skos:exactMatch so:0001837 . + +vcfc:AbundanceClaim a owl:NamedIndividual, vcfc:SVClaim ; + rdfs:label "abundance claim (D)"@en ; vcfc:svClaimCode "D" . +vcfc:AdjacencyClaim a owl:NamedIndividual, vcfc:SVClaim ; + rdfs:label "adjacency claim (J)"@en ; vcfc:svClaimCode "J" . +vcfc:AbundanceAndAdjacencyClaim a owl:NamedIndividual, vcfc:SVClaim ; + rdfs:label "abundance and adjacency claim (DJ)"@en ; vcfc:svClaimCode "DJ" . + +vcfc:SequenceBeforeLeftBracket a owl:NamedIndividual, vcfc:BreakendOrientation ; + rdfs:label "sequence before left bracket"@en ; rdfs:comment "The t[p[ breakend form."@en . +vcfc:SequenceBeforeRightBracket a owl:NamedIndividual, vcfc:BreakendOrientation ; + rdfs:label "sequence before right bracket"@en ; rdfs:comment "The t]p] breakend form."@en . +vcfc:SequenceAfterLeftBracket a owl:NamedIndividual, vcfc:BreakendOrientation ; + rdfs:label "sequence after left bracket"@en ; rdfs:comment "The [p[t breakend form."@en . +vcfc:SequenceAfterRightBracket a owl:NamedIndividual, vcfc:BreakendOrientation ; + rdfs:label "sequence after right bracket"@en ; rdfs:comment "The ]p]t breakend form."@en . + +vcfc:EventDeletion a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DEL event type"@en ; vcfc:eventTypeCode "DEL" ; skos:exactMatch so:0000159 . +vcfc:EventMobileElementDeletion a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DEL:ME event type"@en ; vcfc:eventTypeCode "DEL:ME" ; skos:exactMatch so:0002066 . +vcfc:EventInsertion a owl:NamedIndividual, vcfc:EventType ; rdfs:label "INS event type"@en ; vcfc:eventTypeCode "INS" ; skos:exactMatch so:0000667 . +vcfc:EventMobileElementInsertion a owl:NamedIndividual, vcfc:EventType ; rdfs:label "INS:ME event type"@en ; vcfc:eventTypeCode "INS:ME" ; skos:exactMatch so:0001837 . +vcfc:EventDuplication a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DUP event type"@en ; vcfc:eventTypeCode "DUP" ; skos:exactMatch so:1000035 . +vcfc:EventTandemDuplication a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DUP:TANDEM event type"@en ; vcfc:eventTypeCode "DUP:TANDEM" ; skos:exactMatch so:1000173 . +vcfc:EventDispersedDuplication a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DUP:DISPERSED event type"@en ; vcfc:eventTypeCode "DUP:DISPERSED" . +vcfc:EventInversion a owl:NamedIndividual, vcfc:EventType ; rdfs:label "INV event type"@en ; vcfc:eventTypeCode "INV" ; skos:exactMatch so:1000036 . +vcfc:EventTranslocation a owl:NamedIndividual, vcfc:EventType ; rdfs:label "TRA event type"@en ; vcfc:eventTypeCode "TRA" ; skos:exactMatch so:0000199 . +vcfc:EventBalancedTranslocation a owl:NamedIndividual, vcfc:EventType ; rdfs:label "TRA:BALANCED event type"@en ; vcfc:eventTypeCode "TRA:BALANCED" ; skos:exactMatch so:1000048 . +vcfc:EventUnbalancedTranslocation a owl:NamedIndividual, vcfc:EventType ; rdfs:label "TRA:UNBALANCED event type"@en ; vcfc:eventTypeCode "TRA:UNBALANCED" . +vcfc:EventChromothripsis a owl:NamedIndividual, vcfc:EventType ; rdfs:label "CHROMOTHRIPSIS event type"@en ; vcfc:eventTypeCode "CHROMOTHRIPSIS" ; skos:closeMatch so:0002062 . +vcfc:EventChromoplexy a owl:NamedIndividual, vcfc:EventType ; rdfs:label "CHROMOPLEXY event type"@en ; vcfc:eventTypeCode "CHROMOPLEXY" . +vcfc:EventBreakageFusionBridge a owl:NamedIndividual, vcfc:EventType ; rdfs:label "BFB event type"@en ; vcfc:eventTypeCode "BFB" . +vcfc:EventDoubleMinute a owl:NamedIndividual, vcfc:EventType ; rdfs:label "DOUBLEMINUTE event type"@en ; vcfc:eventTypeCode "DOUBLEMINUTE" . + +vcfc:svType a owl:ObjectProperty ; rdfs:label "SV type"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range vcfc:SymbolicAlleleType . +vcfc:svSubtypeOf a owl:ObjectProperty ; rdfs:label "SV subtype of"@en ; rdfs:domain vcfc:SymbolicAlleleType ; rdfs:range vcfc:SymbolicAlleleType . +vcfc:svClaim a owl:ObjectProperty ; rdfs:label "SV claim"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range vcfc:SVClaim . +vcfc:mateBreakend a owl:ObjectProperty ; rdfs:label "mate breakend"@en ; rdfs:domain vcfc:Breakend ; rdfs:range vcfc:Breakend . +vcfc:partnerBreakend a owl:ObjectProperty ; rdfs:label "partner breakend"@en ; rdfs:domain vcfc:Breakend ; rdfs:range vcfc:Breakend . +vcfc:breakendOrientation a owl:ObjectProperty ; rdfs:label "breakend orientation"@en ; rdfs:domain vcfc:Breakend ; rdfs:range vcfc:BreakendOrientation . +vcfc:inEvent a owl:ObjectProperty ; rdfs:label "in event"@en ; rdfs:range vcfc:VariantEvent . +vcfc:eventType a owl:ObjectProperty ; rdfs:label "event type"@en ; rdfs:domain vcfc:VariantEvent ; rdfs:range vcfc:EventType . +vcfc:posConfidenceInterval a owl:ObjectProperty ; + rdfs:label "POS confidence interval"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range faldo:InRangePosition ; + rdfs:comment "Represents CIPOS through FALDO InRangePosition rather than a VCF-specific ConfidenceInterval."@en . +vcfc:endConfidenceInterval a owl:ObjectProperty ; + rdfs:label "END confidence interval"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range faldo:InRangePosition ; + rdfs:comment "Represents CIEND through FALDO InRangePosition rather than a VCF-specific ConfidenceInterval."@en . +vcfc:lenConfidenceInterval a owl:ObjectProperty ; rdfs:label "SVLEN confidence interval"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range vcfc:ConfidenceInterval . +vcfc:copyNumberConfidenceInterval a owl:ObjectProperty ; rdfs:label "copy-number confidence interval"@en ; rdfs:range vcfc:ConfidenceInterval . +vcfc:rucConfidenceInterval a owl:ObjectProperty ; rdfs:label "RUC confidence interval"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range vcfc:ConfidenceInterval . +vcfc:rbConfidenceInterval a owl:ObjectProperty ; rdfs:label "RB confidence interval"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range vcfc:ConfidenceInterval . +vcfc:hasRepeatSequence a owl:ObjectProperty ; rdfs:label "has repeat sequence"@en ; rdfs:domain vcfc:TandemRepeatAllele ; rdfs:range vcfc:RepeatSequence . +vcfc:hasRepeatUnit a owl:ObjectProperty ; rdfs:label "has repeat unit"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range vcfc:RepeatUnit . +vcfc:modifiedResidue a owl:ObjectProperty ; rdfs:label "modified residue"@en ; rdfs:domain vcfc:BaseModification ; + rdfs:comment "Links directly to the ChEBI class denoting the modified residue."@en . + +vcfc:svTypeCode a owl:DatatypeProperty ; rdfs:label "SV type code"@en ; rdfs:domain vcfc:SymbolicAlleleType ; rdfs:range xsd:string . +vcfc:svClaimCode a owl:DatatypeProperty ; rdfs:label "SV claim code"@en ; rdfs:domain vcfc:SVClaim ; rdfs:range xsd:string . +vcfc:svLength a owl:DatatypeProperty ; rdfs:label "SV length"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range rdfs:Literal . +vcfc:isImprecise a owl:DatatypeProperty ; rdfs:label "is imprecise"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range xsd:boolean . +vcfc:isNovel a owl:DatatypeProperty ; rdfs:label "is novel"@en ; rdfs:domain vcfc:AltAllele ; rdfs:range xsd:boolean . +vcfc:isSingleBreakend a owl:DatatypeProperty ; rdfs:label "is single breakend"@en ; rdfs:domain vcfc:Breakend ; rdfs:range xsd:boolean . +vcfc:isTelomereBreakend a owl:DatatypeProperty ; rdfs:label "is telomere breakend"@en ; rdfs:domain vcfc:Breakend ; rdfs:range xsd:boolean . +vcfc:breakendReplacementString a owl:DatatypeProperty ; rdfs:label "breakend replacement string"@en ; rdfs:domain vcfc:Breakend ; rdfs:range vcfc:BreakendString . +vcfc:insertedSequence a owl:DatatypeProperty ; rdfs:label "inserted sequence"@en ; rdfs:domain vcfc:Breakend ; rdfs:range xsd:string . +vcfc:eventTypeCode a owl:DatatypeProperty ; rdfs:label "event type code"@en ; rdfs:domain vcfc:EventType ; rdfs:range xsd:string . +vcfc:ciLower a owl:DatatypeProperty ; rdfs:label "confidence-interval lower bound"@en ; rdfs:domain vcfc:ConfidenceInterval ; rdfs:range rdfs:Literal . +vcfc:ciUpper a owl:DatatypeProperty ; rdfs:label "confidence-interval upper bound"@en ; rdfs:domain vcfc:ConfidenceInterval ; rdfs:range rdfs:Literal . +vcfc:repeatSequenceCount a owl:DatatypeProperty ; rdfs:label "repeat-sequence count"@en ; rdfs:domain vcfc:TandemRepeatAllele ; rdfs:range xsd:nonNegativeInteger . +vcfc:repeatSequenceIndex a owl:DatatypeProperty ; rdfs:label "repeat-sequence index"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range xsd:positiveInteger . +vcfc:repeatUnitSequence a owl:DatatypeProperty ; rdfs:label "repeat-unit sequence"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range rdfs:Literal . +vcfc:repeatUnitLength a owl:DatatypeProperty ; rdfs:label "repeat-unit length"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range rdfs:Literal . +vcfc:repeatUnitCount a owl:DatatypeProperty ; rdfs:label "repeat-unit count"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range rdfs:Literal . +vcfc:repeatBases a owl:DatatypeProperty ; rdfs:label "repeat bases"@en ; rdfs:domain vcfc:RepeatSequence ; rdfs:range rdfs:Literal . +vcfc:repeatUnitBases a owl:DatatypeProperty ; rdfs:label "repeat-unit bases"@en ; rdfs:domain vcfc:RepeatUnit ; rdfs:range rdfs:Literal . +vcfc:copyNumber a owl:DatatypeProperty ; rdfs:label "copy number"@en ; rdfs:range rdfs:Literal . +vcfc:copyNumberQuality a owl:DatatypeProperty ; rdfs:label "copy-number quality"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range rdfs:Literal . +vcfc:copyNumberLikelihood a owl:DatatypeProperty ; rdfs:label "copy-number likelihood"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range rdfs:Literal . +vcfc:copyNumberPosterior a owl:DatatypeProperty ; rdfs:label "copy-number posterior"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range rdfs:Literal . +vcfc:haplotypeId a owl:DatatypeProperty ; rdfs:label "haplotype ID"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range xsd:integer . +vcfc:ancestralHaplotypeId a owl:DatatypeProperty ; rdfs:label "ancestral haplotype ID"@en ; rdfs:domain vcfc:SampleCall ; rdfs:range xsd:integer . +vcfc:referenceBlockLength a owl:DatatypeProperty ; rdfs:label "reference-block length"@en ; rdfs:domain vcfc:ReferenceBlock ; rdfs:range xsd:nonNegativeInteger . +vcfc:endPosition a owl:DatatypeProperty ; rdfs:label "end position"@en ; rdfs:domain vcfc:ReferenceBlock ; rdfs:range xsd:integer . +vcfc:isReferenceBlockStart a owl:DatatypeProperty ; rdfs:label "is reference-block start"@en ; rdfs:domain vcfc:ReferenceBlock ; rdfs:range xsd:boolean . +vcfc:modifiedBaseOffset a owl:DatatypeProperty ; rdfs:label "modified-base offset"@en ; rdfs:domain vcfc:BaseModification ; rdfs:range xsd:nonNegativeInteger . +vcfc:modificationFraction a owl:DatatypeProperty ; rdfs:label "modification fraction"@en ; rdfs:domain vcfc:BaseModification ; rdfs:range rdfs:Literal . +vcfc:modificationDepth a owl:DatatypeProperty ; rdfs:label "modification depth"@en ; rdfs:domain vcfc:BaseModification ; rdfs:range rdfs:Literal . +vcfc:modificationAlleleDepth a owl:DatatypeProperty ; rdfs:label "modification allele depth"@en ; rdfs:domain vcfc:BaseModification ; rdfs:range rdfs:Literal . + +vcfc:hasReferenceBlock a owl:ObjectProperty ; rdfs:label "has reference block"@en ; rdfs:range vcfc:ReferenceBlock . +vcfc:blockAllele a owl:ObjectProperty ; rdfs:label "block allele"@en ; rdfs:domain vcfc:ReferenceBlock ; rdfs:range vcfc:AltAllele . + +# Source: ontology/vcf-core-reserved-keys.ttl +################################################################# +# VCF 4.5 reserved-key registry +# +# Generated by scripts/generate-reserved-keys.mjs from https://raw.githubusercontent.com/samtools/hts-specs/master/VCFv4.5.tex +# SHA-256: 37f13e0d2e8e741ea8505b0342b6e6034637a1f476eeb3af1b8acc25d70246c5 +# Source rows: 21 general INFO, 31 SV INFO, +# 63 general FORMAT, 8 SV FORMAT. +# Do not edit by hand; regenerate from the cited VCF 4.5 source. +################################################################# + + a owl:Ontology ; + rdfs:label "VCF Core reserved-key registry for VCF 4.5"@en ; + dct:source ; + owl:imports ; + owl:versionInfo "2.1.2" ; + vcfc:specificationVersion "VCFv4.5" . + +vcfc:specificationVersion a owl:AnnotationProperty ; + rdfs:label "specification version"@en ; + rdfs:comment "The VCF specification version a registry graph describes. It is distinct from owl:versionInfo, which records the release version of this vocabulary."@en . + +vcfc:reservedIn a owl:AnnotationProperty ; + rdfs:label "reserved in"@en ; + rdfs:comment "Records the VCF specification version that reserves a field identifier or identifier pattern."@en . + +vcfc:deprecatedInVersion a owl:AnnotationProperty ; + rdfs:label "deprecated in version"@en ; + rdfs:comment "Records the VCF specification version in which a reserved field is deprecated."@en . + +vcfc:keyPattern a owl:AnnotationProperty ; + rdfs:label "key pattern"@en ; + rdfs:comment "The regular-expression family reserved by a FORMAT field definition rather than one concrete key."@en . + +vcfc:aliasOf a owl:AnnotationProperty ; + rdfs:label "alias of"@en ; + rdfs:comment "The canonical reserved key denoted by a VCF base-modification alias."@en . + +vcfc:ReservedInfo_AA a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info AA reserved field definition"@en ; + vcfc:fieldId "AA" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Ancestral allele" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_AC a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info AC reserved field definition"@en ; + vcfc:fieldId "AC" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Allele count in genotypes, for each ALT allele, in the same order as listed" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_AD a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info AD reserved field definition"@en ; + vcfc:fieldId "AD" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Total read depth for each allele" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_ADF a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info ADF reserved field definition"@en ; + vcfc:fieldId "ADF" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for each allele on the forward strand" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_ADR a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info ADR reserved field definition"@en ; + vcfc:fieldId "ADR" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for each allele on the reverse strand" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_AF a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info AF reserved field definition"@en ; + vcfc:fieldId "AF" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Allele frequency for each ALT allele in the same order as listed (estimated from primary data, not called genotypes)" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_AN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info AN reserved field definition"@en ; + vcfc:fieldId "AN" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Total number of alleles in called genotypes" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_BQ a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info BQ reserved field definition"@en ; + vcfc:fieldId "BQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "RMS base quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CIGAR a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CIGAR reserved field definition"@en ; + vcfc:fieldId "CIGAR" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Cigar string describing how to align an alternate allele to the reference allele" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_DB a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info DB reserved field definition"@en ; + vcfc:fieldId "DB" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "dbSNP membership" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_DP a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info DP reserved field definition"@en ; + vcfc:fieldId "DP" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Combined depth across samples" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_END a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info END reserved field definition"@en ; + vcfc:fieldId "END" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Deprecated. Present for backwards compatibility with earlier versions of VCF." ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:deprecatedInVersion "VCFv4.5" . + +vcfc:ReservedInfo_H2 a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info H2 reserved field definition"@en ; + vcfc:fieldId "H2" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "HapMap2 membership" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_H3 a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info H3 reserved field definition"@en ; + vcfc:fieldId "H3" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "HapMap3 membership" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_MQ a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info MQ reserved field definition"@en ; + vcfc:fieldId "MQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "RMS mapping quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_MQ0 a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info MQ0 reserved field definition"@en ; + vcfc:fieldId "MQ0" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Number of MAPQ == 0 reads" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_NS a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info NS reserved field definition"@en ; + vcfc:fieldId "NS" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Number of samples with data" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_SB a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info SB reserved field definition"@en ; + vcfc:fieldId "SB" ; + vcfc:fieldNumber "4" ; + vcfc:fieldNumberInteger 4 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Strand bias" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_SOMATIC a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info SOMATIC reserved field definition"@en ; + vcfc:fieldId "SOMATIC" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "Somatic mutation (for cancer genomics)" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_VALIDATED a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info VALIDATED reserved field definition"@en ; + vcfc:fieldId "VALIDATED" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "Validated by follow-up experiment" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_1000G a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info 1000G reserved field definition"@en ; + vcfc:fieldId "1000G" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "1000 Genomes membership" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_IMPRECISE a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info IMPRECISE reserved field definition"@en ; + vcfc:fieldId "IMPRECISE" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "Imprecise structural variation" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_NOVEL a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info NOVEL reserved field definition"@en ; + vcfc:fieldId "NOVEL" ; + vcfc:fieldNumber "0" ; + vcfc:fieldNumberInteger 0 ; + vcfc:fieldType vcfc:FlagType ; + vcfc:fieldDescription "Indicates a novel structural variation" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_SVTYPE a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info SVTYPE reserved field definition"@en ; + vcfc:fieldId "SVTYPE" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Type of structural variant" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:deprecatedInVersion "VCFv4.4" . + +vcfc:ReservedInfo_SVLEN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info SVLEN reserved field definition"@en ; + vcfc:fieldId "SVLEN" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Length of structural variant" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CIPOS a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CIPOS reserved field definition"@en ; + vcfc:fieldId "CIPOS" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Confidence interval around POS for symbolic structural variants" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CIEND a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CIEND reserved field definition"@en ; + vcfc:fieldId "CIEND" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Confidence interval around the inferred END for symbolic structural variants" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_HOMLEN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info HOMLEN reserved field definition"@en ; + vcfc:fieldId "HOMLEN" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Length of base pair identical micro-homology at breakpoints" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_HOMSEQ a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info HOMSEQ reserved field definition"@en ; + vcfc:fieldId "HOMSEQ" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Sequence of base pair identical micro-homology at breakpoints" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_BKPTID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info BKPTID reserved field definition"@en ; + vcfc:fieldId "BKPTID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of the assembled alternate allele in the assembly file" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_MEINFO a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info MEINFO reserved field definition"@en ; + vcfc:fieldId "MEINFO" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Mobile element info of the form NAME,START,END,POLARITY" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_METRANS a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info METRANS reserved field definition"@en ; + vcfc:fieldId "METRANS" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Mobile element transduction info of the form CHR,START,END,POLARITY" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_DGVID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info DGVID reserved field definition"@en ; + vcfc:fieldId "DGVID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of this element in Database of Genomic Variation" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_DBVARID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info DBVARID reserved field definition"@en ; + vcfc:fieldId "DBVARID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of this element in DBVAR" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_DBRIPID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info DBRIPID reserved field definition"@en ; + vcfc:fieldId "DBRIPID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of this element in DBRIP" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_MATEID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info MATEID reserved field definition"@en ; + vcfc:fieldId "MATEID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of mate breakend" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_PARID a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info PARID reserved field definition"@en ; + vcfc:fieldId "PARID" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of partner breakend" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_EVENT a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info EVENT reserved field definition"@en ; + vcfc:fieldId "EVENT" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "ID of associated event" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_EVENTTYPE a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info EVENTTYPE reserved field definition"@en ; + vcfc:fieldId "EVENTTYPE" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Type of associated event" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CILEN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CILEN reserved field definition"@en ; + vcfc:fieldId "CILEN" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Confidence interval for the SVLEN field" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CN reserved field definition"@en ; + vcfc:fieldId "CN" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Copy number of CNV/breakpoint" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CICN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CICN reserved field definition"@en ; + vcfc:fieldId "CICN" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Confidence interval around copy number" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_SVCLAIM a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info SVCLAIM reserved field definition"@en ; + vcfc:fieldId "SVCLAIM" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Claim made by the structural variant call. Valid values are D, J, DJ for abundance, adjacency and both respectively" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RN a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RN reserved field definition"@en ; + vcfc:fieldId "RN" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Total number of repeat sequences in this allele" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RUS a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RUS reserved field definition"@en ; + vcfc:fieldId "RUS" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Repeat unit sequence of the corresponding repeat sequence" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RUL a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RUL reserved field definition"@en ; + vcfc:fieldId "RUL" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Repeat unit length of the corresponding repeat sequence" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RUC a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RUC reserved field definition"@en ; + vcfc:fieldId "RUC" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Repeat unit count of corresponding repeat sequence" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RB a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RB reserved field definition"@en ; + vcfc:fieldId "RB" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Total number of bases in the corresponding repeat sequence" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CIRUC a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CIRUC reserved field definition"@en ; + vcfc:fieldId "CIRUC" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Confidence interval around RUC" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_CIRB a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info CIRB reserved field definition"@en ; + vcfc:fieldId "CIRB" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Confidence interval around RB" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedInfo_RUB a owl:NamedIndividual, vcfc:InfoFieldDefinition ; + rdfs:label "Info RUB reserved field definition"@en ; + vcfc:fieldId "RUB" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Number of bases in each individual repeat unit" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_AD a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format AD reserved field definition"@en ; + vcfc:fieldId "AD" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for each allele" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_ADF a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADF reserved field definition"@en ; + vcfc:fieldId "ADF" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for each allele on the forward strand" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_ADR a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADR reserved field definition"@en ; + vcfc:fieldId "ADR" ; + vcfc:fieldNumber "R" ; + vcfc:fieldArity vcfc:ArityPerAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for each allele on the reverse strand" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_DP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DP reserved field definition"@en ; + vcfc:fieldId "DP" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_EC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format EC reserved field definition"@en ; + vcfc:fieldId "EC" ; + vcfc:fieldNumber "A" ; + vcfc:fieldArity vcfc:ArityPerAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Expected alternate allele counts" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LEN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LEN reserved field definition"@en ; + vcfc:fieldId "LEN" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Length of <*> reference block" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_FT a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format FT reserved field definition"@en ; + vcfc:fieldId "FT" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Filter indicating if this genotype was \"called\"" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_GL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format GL reserved field definition"@en ; + vcfc:fieldId "GL" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Genotype likelihoods" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_GP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format GP reserved field definition"@en ; + vcfc:fieldId "GP" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Genotype posterior probabilities" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_GQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format GQ reserved field definition"@en ; + vcfc:fieldId "GQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Conditional genotype quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_GT a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format GT reserved field definition"@en ; + vcfc:fieldId "GT" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Genotype" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_HQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format HQ reserved field definition"@en ; + vcfc:fieldId "HQ" ; + vcfc:fieldNumber "2" ; + vcfc:fieldNumberInteger 2 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Haplotype quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LA a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LA reserved field definition"@en ; + vcfc:fieldId "LA" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Reserved" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LAA a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LAA reserved field definition"@en ; + vcfc:fieldId "LAA" ; + vcfc:fieldNumber "." ; + vcfc:fieldArity vcfc:ArityVariable ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "1-based indices into ALT, indicating which alleles are relevant (local) for the current sample" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LAD a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LAD reserved field definition"@en ; + vcfc:fieldId "LAD" ; + vcfc:fieldNumber "LR" ; + vcfc:fieldArity vcfc:ArityPerLocalAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of AD" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LADF a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LADF reserved field definition"@en ; + vcfc:fieldId "LADF" ; + vcfc:fieldNumber "LR" ; + vcfc:fieldArity vcfc:ArityPerLocalAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of ADF" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LADR a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LADR reserved field definition"@en ; + vcfc:fieldId "LADR" ; + vcfc:fieldNumber "LR" ; + vcfc:fieldArity vcfc:ArityPerLocalAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of ADR" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LEC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LEC reserved field definition"@en ; + vcfc:fieldId "LEC" ; + vcfc:fieldNumber "LA" ; + vcfc:fieldArity vcfc:ArityPerLocalAlt ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of EC" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LGL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LGL reserved field definition"@en ; + vcfc:fieldId "LGL" ; + vcfc:fieldNumber "LG" ; + vcfc:fieldArity vcfc:ArityPerLocalGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Local-allele representation of GL" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LGP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LGP reserved field definition"@en ; + vcfc:fieldId "LGP" ; + vcfc:fieldNumber "LG" ; + vcfc:fieldArity vcfc:ArityPerLocalGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Local-allele representation of GP" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LPL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LPL reserved field definition"@en ; + vcfc:fieldId "LPL" ; + vcfc:fieldNumber "LG" ; + vcfc:fieldArity vcfc:ArityPerLocalGenotype ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of PL" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_LPP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format LPP reserved field definition"@en ; + vcfc:fieldId "LPP" ; + vcfc:fieldNumber "LG" ; + vcfc:fieldArity vcfc:ArityPerLocalGenotype ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Local-allele representation of PP" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_MChEBI_ACGTUN_ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M[0-9]+[ACGTUN] reserved field definition"@en ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Fraction of bases modified with the given ChEBI ID." ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:keyPattern "M[0-9]+[ACGTUN]" . + +vcfc:ReservedFormat_DPMChEBI_ACGTUN_ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM[0-9]+[ACGTUN] reserved field definition"@en ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Total read depth for reads able to detect the base modification with the given ChEBI ID." ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:keyPattern "DPM[0-9]+[ACGTUN]" . + +vcfc:ReservedFormat_ADMChEBI_ACGTUN_ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM[0-9]+[ACGTUN] reserved field definition"@en ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Read depth for reads with the base modification with the given ChEBI ID." ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:keyPattern "ADM[0-9]+[ACGTUN]" . + +vcfc:ReservedFormat_M5mC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5mC reserved field definition"@en ; + vcfc:fieldId "M5mC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M27551C 5-Methylcytosine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M27551C" ; + skos:exactMatch chebi:27551 . + +vcfc:ReservedFormat_DPM5mC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5mC reserved field definition"@en ; + vcfc:fieldId "DPM5mC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM27551C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM27551C" . + +vcfc:ReservedFormat_ADM5mC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5mC reserved field definition"@en ; + vcfc:fieldId "ADM5mC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM27551C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM27551C" . + +vcfc:ReservedFormat_M5hmC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5hmC reserved field definition"@en ; + vcfc:fieldId "M5hmC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M76792C 5-Hydroxymethylcytosine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M76792C" ; + skos:exactMatch chebi:76792 . + +vcfc:ReservedFormat_DPM5hmC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5hmC reserved field definition"@en ; + vcfc:fieldId "DPM5hmC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM76792C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM76792C" . + +vcfc:ReservedFormat_ADM5hmC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5hmC reserved field definition"@en ; + vcfc:fieldId "ADM5hmC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM76792C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM76792C" . + +vcfc:ReservedFormat_M5fC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5fC reserved field definition"@en ; + vcfc:fieldId "M5fC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M76794C 5-Formylcytosine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M76794C" ; + skos:exactMatch chebi:76794 . + +vcfc:ReservedFormat_DPM5fC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5fC reserved field definition"@en ; + vcfc:fieldId "DPM5fC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM76794C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM76794C" . + +vcfc:ReservedFormat_ADM5fC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5fC reserved field definition"@en ; + vcfc:fieldId "ADM5fC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM76794C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM76794C" . + +vcfc:ReservedFormat_M5caC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5caC reserved field definition"@en ; + vcfc:fieldId "M5caC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M76793C 5-Carboxylcytosine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M76793C" ; + skos:exactMatch chebi:76793 . + +vcfc:ReservedFormat_DPM5caC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5caC reserved field definition"@en ; + vcfc:fieldId "DPM5caC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM76793C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM76793C" . + +vcfc:ReservedFormat_ADM5caC a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5caC reserved field definition"@en ; + vcfc:fieldId "ADM5caC" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM76793C" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM76793C" . + +vcfc:ReservedFormat_M5hmU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5hmU reserved field definition"@en ; + vcfc:fieldId "M5hmU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M16964T 5-Hydroxymethyluracil" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M16964T" ; + skos:exactMatch chebi:16964 . + +vcfc:ReservedFormat_DPM5hmU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5hmU reserved field definition"@en ; + vcfc:fieldId "DPM5hmU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM16964T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM16964T" . + +vcfc:ReservedFormat_ADM5hmU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5hmU reserved field definition"@en ; + vcfc:fieldId "ADM5hmU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM16964T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM16964T" . + +vcfc:ReservedFormat_M5fU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5fU reserved field definition"@en ; + vcfc:fieldId "M5fU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M80961T 5-Formyluracil" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M80961T" ; + skos:exactMatch chebi:80961 . + +vcfc:ReservedFormat_DPM5fU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5fU reserved field definition"@en ; + vcfc:fieldId "DPM5fU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM80961T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM80961T" . + +vcfc:ReservedFormat_ADM5fU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5fU reserved field definition"@en ; + vcfc:fieldId "ADM5fU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM80961T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM80961T" . + +vcfc:ReservedFormat_M5caU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M5caU reserved field definition"@en ; + vcfc:fieldId "M5caU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M17477T 5-Carboxyluracil" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M17477T" ; + skos:exactMatch chebi:17477 . + +vcfc:ReservedFormat_DPM5caU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM5caU reserved field definition"@en ; + vcfc:fieldId "DPM5caU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM17477T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM17477T" . + +vcfc:ReservedFormat_ADM5caU a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM5caU reserved field definition"@en ; + vcfc:fieldId "ADM5caU" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM17477T" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM17477T" . + +vcfc:ReservedFormat_M6mA a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M6mA reserved field definition"@en ; + vcfc:fieldId "M6mA" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M28871A 6-Methyladenine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M28871A" ; + skos:exactMatch chebi:28871 . + +vcfc:ReservedFormat_DPM6mA a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM6mA reserved field definition"@en ; + vcfc:fieldId "DPM6mA" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM28871A" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM28871A" . + +vcfc:ReservedFormat_ADM6mA a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM6mA reserved field definition"@en ; + vcfc:fieldId "ADM6mA" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM28871A" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM28871A" . + +vcfc:ReservedFormat_M8oxoG a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format M8oxoG reserved field definition"@en ; + vcfc:fieldId "M8oxoG" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M44605G 8-Oxoguanine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M44605G" ; + skos:exactMatch chebi:44605 . + +vcfc:ReservedFormat_DPM8oxoG a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPM8oxoG reserved field definition"@en ; + vcfc:fieldId "DPM8oxoG" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM44605G" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM44605G" . + +vcfc:ReservedFormat_ADM8oxoG a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADM8oxoG reserved field definition"@en ; + vcfc:fieldId "ADM8oxoG" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM44605G" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM44605G" . + +vcfc:ReservedFormat_MXaoN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format MXaoN reserved field definition"@en ; + vcfc:fieldId "MXaoN" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Alias for M18107N Xanthosine" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "M18107N" ; + skos:exactMatch chebi:18107 . + +vcfc:ReservedFormat_DPMXaoN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format DPMXaoN reserved field definition"@en ; + vcfc:fieldId "DPMXaoN" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for DPM18107N" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "DPM18107N" . + +vcfc:ReservedFormat_ADMXaoN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format ADMXaoN reserved field definition"@en ; + vcfc:fieldId "ADMXaoN" ; + vcfc:fieldNumber "M" ; + vcfc:fieldArity vcfc:ArityPerBaseModification ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Alias for ADM18107N" ; + vcfc:reservedIn "VCFv4.5" ; + vcfc:aliasOf "ADM18107N" . + +vcfc:ReservedFormat_MQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format MQ reserved field definition"@en ; + vcfc:fieldId "MQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "RMS mapping quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PL reserved field definition"@en ; + vcfc:fieldId "PL" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phred-scaled genotype likelihoods rounded to the closest integer" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PP reserved field definition"@en ; + vcfc:fieldId "PP" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phred-scaled genotype posterior probabilities rounded to the closest integer" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PQ reserved field definition"@en ; + vcfc:fieldId "PQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phasing quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PS a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PS reserved field definition"@en ; + vcfc:fieldId "PS" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phase set" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PSL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PSL reserved field definition"@en ; + vcfc:fieldId "PSL" ; + vcfc:fieldNumber "P" ; + vcfc:fieldArity vcfc:ArityPerGTAllele ; + vcfc:fieldType vcfc:StringType ; + vcfc:fieldDescription "Phase set list" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PSO a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PSO reserved field definition"@en ; + vcfc:fieldId "PSO" ; + vcfc:fieldNumber "P" ; + vcfc:fieldArity vcfc:ArityPerGTAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phase set list ordinal" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_PSQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format PSQ reserved field definition"@en ; + vcfc:fieldId "PSQ" ; + vcfc:fieldNumber "P" ; + vcfc:fieldArity vcfc:ArityPerGTAllele ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phase set list quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_CN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format CN reserved field definition"@en ; + vcfc:fieldId "CN" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Copy number" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_CICN a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format CICN reserved field definition"@en ; + vcfc:fieldId "CICN" ; + vcfc:fieldNumber "2" ; + vcfc:fieldNumberInteger 2 ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Confidence interval around copy number" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_CNQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format CNQ reserved field definition"@en ; + vcfc:fieldId "CNQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Copy number genotype quality" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_CNL a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format CNL reserved field definition"@en ; + vcfc:fieldId "CNL" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Copy number genotype likelihood" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_CNP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format CNP reserved field definition"@en ; + vcfc:fieldId "CNP" ; + vcfc:fieldNumber "G" ; + vcfc:fieldArity vcfc:ArityPerGenotype ; + vcfc:fieldType vcfc:FloatType ; + vcfc:fieldDescription "Copy number posterior probabilities" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_NQ a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format NQ reserved field definition"@en ; + vcfc:fieldId "NQ" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Phred style probability score that the variant is novel" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_HAP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format HAP reserved field definition"@en ; + vcfc:fieldId "HAP" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Unique haplotype identifier" ; + vcfc:reservedIn "VCFv4.5" . + +vcfc:ReservedFormat_AHAP a owl:NamedIndividual, vcfc:FormatFieldDefinition ; + rdfs:label "Format AHAP reserved field definition"@en ; + vcfc:fieldId "AHAP" ; + vcfc:fieldNumber "1" ; + vcfc:fieldNumberInteger 1 ; + vcfc:fieldType vcfc:IntegerType ; + vcfc:fieldDescription "Unique identifier of ancestral haplotype" ; + vcfc:reservedIn "VCFv4.5" . diff --git a/vcf_rdfizer_data/shacl/vcf-4.1.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-4.1.shacl.ttl new file mode 100644 index 0000000..7c59293 --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-4.1.shacl.ttl @@ -0,0 +1,87 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:VCF41FileGate a sh:NodeShape ; sh:targetClass vcfc:VCF41File ; sh:property [ sh:path vcfc:fileFormat ; sh:hasValue "VCFv4.1" ; sh:maxCount 1 ] . +vcfc:VCF41NumberShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: Number codes must be supported by this version and field kind." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . ?d vcfc:fieldNumber ?number . +{ ?d a vcfc:INFOHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|G|[.])$")) } UNION { ?d a vcfc:FORMATHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|G|[.])$")) } +}""" ] . + +vcfc:VCF41INFOReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: reserved INFO declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:INFOHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AA" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="AF" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="BKPTID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="CICN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CICNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIEND" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CILEN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CIPOS" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="DB" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="DBRIPID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DBVARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DGVID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="DPADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="END" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EVENT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="H2" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="HOMLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="HOMSEQ" && (?number!="." || ?type!=vcfc:StringType)) || (?id="IMPRECISE" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="MATEID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="MEINFO" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="METRANS" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="NOVEL" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="NS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="SVLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="SVTYPE" && (?number!="1" || ?type!=vcfc:StringType))) +}""" ] . + +vcfc:VCF41FORMATReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: reserved FORMAT declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:FORMATHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AHAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNL" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="CNQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="HAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="HQ" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="NQ" && (?number!="1" || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF41GTShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: leading GT phase indicators require VCF 4.4 or later." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasRecord/vcfc:hasCall/vcfc:hasSampleCall/vcfc:hasGenotype ?g . +?g vcfc:genotypeString ?gt . FILTER(REGEX(STR(?gt),"^[|/]")) +}""" ] . + +vcfc:VCF41Tuple2Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 2) +}""" ] . + +vcfc:VCF41Tuple2ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:VCF41Tuple4Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 4) +}""" ] . + +vcfc:VCF41Tuple4ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.1: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.1" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-4.2.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-4.2.shacl.ttl new file mode 100644 index 0000000..be78699 --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-4.2.shacl.ttl @@ -0,0 +1,87 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:VCF42FileGate a sh:NodeShape ; sh:targetClass vcfc:VCF42File ; sh:property [ sh:path vcfc:fileFormat ; sh:hasValue "VCFv4.2" ; sh:maxCount 1 ] . +vcfc:VCF42NumberShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: Number codes must be supported by this version and field kind." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . ?d vcfc:fieldNumber ?number . +{ ?d a vcfc:INFOHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } UNION { ?d a vcfc:FORMATHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } +}""" ] . + +vcfc:VCF42INFOReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: reserved INFO declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:INFOHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AA" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="AF" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="BKPTID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="CICN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CICNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIEND" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CILEN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CIPOS" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="DB" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="DBRIPID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DBVARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DGVID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="DPADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="END" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EVENT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="H2" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="HOMLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="HOMSEQ" && (?number!="." || ?type!=vcfc:StringType)) || (?id="IMPRECISE" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="MATEID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="MEINFO" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="METRANS" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="NOVEL" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="NS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="SVLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="SVTYPE" && (?number!="1" || ?type!=vcfc:StringType))) +}""" ] . + +vcfc:VCF42FORMATReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: reserved FORMAT declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:FORMATHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AHAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNL" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="CNQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="HAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="HQ" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="NQ" && (?number!="1" || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF42GTShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: leading GT phase indicators require VCF 4.4 or later." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasRecord/vcfc:hasCall/vcfc:hasSampleCall/vcfc:hasGenotype ?g . +?g vcfc:genotypeString ?gt . FILTER(REGEX(STR(?gt),"^[|/]")) +}""" ] . + +vcfc:VCF42Tuple2Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 2) +}""" ] . + +vcfc:VCF42Tuple2ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:VCF42Tuple4Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 4) +}""" ] . + +vcfc:VCF42Tuple4ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.2: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.2" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-4.3.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-4.3.shacl.ttl new file mode 100644 index 0000000..bf2a142 --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-4.3.shacl.ttl @@ -0,0 +1,87 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:VCF43FileGate a sh:NodeShape ; sh:targetClass vcfc:VCF43File ; sh:property [ sh:path vcfc:fileFormat ; sh:hasValue "VCFv4.3" ; sh:maxCount 1 ] . +vcfc:VCF43NumberShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: Number codes must be supported by this version and field kind." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . ?d vcfc:fieldNumber ?number . +{ ?d a vcfc:INFOHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } UNION { ?d a vcfc:FORMATHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } +}""" ] . + +vcfc:VCF43INFOReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: reserved INFO declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:INFOHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="1000G" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="AA" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="AC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="AF" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="AN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="BKPTID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="BQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CICN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CICNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIEND" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CIGAR" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="CILEN" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CIPOS" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="DB" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="DBRIPID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DBVARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DGVID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="DPADJ" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="END" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EVENT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="H2" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="H3" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="HOMLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="HOMSEQ" && (?number!="." || ?type!=vcfc:StringType)) || (?id="IMPRECISE" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="MATEID" && (?number!="." || ?type!=vcfc:StringType)) || (?id="MEINFO" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="METRANS" && (?number!="4" || ?type!=vcfc:StringType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="MQ0" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="NOVEL" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="NS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PARID" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="SB" && (?number!="4" || ?type!=vcfc:IntegerType)) || (?id="SOMATIC" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="SVLEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="SVTYPE" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="VALIDATED" && (?number!="0" || ?type!=vcfc:FlagType))) +}""" ] . + +vcfc:VCF43FORMATReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: reserved FORMAT declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:FORMATHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="AHAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CNL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="CNP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="CNQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="FT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="GL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="HAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="HQ" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="NQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PL" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PP" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PS" && (?number!="1" || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF43GTShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: leading GT phase indicators require VCF 4.4 or later." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasRecord/vcfc:hasCall/vcfc:hasSampleCall/vcfc:hasGenotype ?g . +?g vcfc:genotypeString ?gt . FILTER(REGEX(STR(?gt),"^[|/]")) +}""" ] . + +vcfc:VCF43Tuple2Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 2) +}""" ] . + +vcfc:VCF43Tuple2ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:VCF43Tuple4Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 4) +}""" ] . + +vcfc:VCF43Tuple4ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.3: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.3" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-4.4.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-4.4.shacl.ttl new file mode 100644 index 0000000..4a78529 --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-4.4.shacl.ttl @@ -0,0 +1,78 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:VCF44FileGate a sh:NodeShape ; sh:targetClass vcfc:VCF44File ; sh:property [ sh:path vcfc:fileFormat ; sh:hasValue "VCFv4.4" ; sh:maxCount 1 ] . +vcfc:VCF44NumberShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: Number codes must be supported by this version and field kind." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . ?d vcfc:fieldNumber ?number . +{ ?d a vcfc:INFOHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } UNION { ?d a vcfc:FORMATHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.]|P)$")) } +}""" ] . + +vcfc:VCF44INFOReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: reserved INFO declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:INFOHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="1000G" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="AA" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="AC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="AF" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="AN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="BKPTID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="BQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CICN" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="CIEND" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIGAR" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="CILEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIPOS" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIRB" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIRUC" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="CN" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="DB" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="DBRIPID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DBVARID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DGVID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="END" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EVENT" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="EVENTTYPE" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="H2" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="H3" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="HOMLEN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="HOMSEQ" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="IMPRECISE" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="MATEID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="MEINFO" && (?number!="." || ?type!=vcfc:StringType)) || (?id="METRANS" && (?number!="." || ?type!=vcfc:StringType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="MQ0" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="NOVEL" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="NS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PARID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="RB" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="RN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="RUB" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="RUC" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="RUL" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="RUS" && (?number!="." || ?type!=vcfc:StringType)) || (?id="SB" && (?number!="4" || ?type!=vcfc:IntegerType)) || (?id="SOMATIC" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="SVCLAIM" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="SVLEN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="SVTYPE" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="VALIDATED" && (?number!="0" || ?type!=vcfc:FlagType))) +}""" ] . + +vcfc:VCF44FORMATReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: reserved FORMAT declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:FORMATHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="AHAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="CICN" && (?number!="2" || ?type!=vcfc:FloatType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CNL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="CNP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="CNQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="FT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="GL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="HAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="HQ" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="NQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PL" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PP" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PSL" && (?number!="P" || ?type!=vcfc:StringType)) || (?id="PSO" && (?number!="P" || ?type!=vcfc:IntegerType)) || (?id="PSQ" && (?number!="P" || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF44Tuple2Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND|CILEN|CICN)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 2 * ?altCount) +}""" ] . + +vcfc:VCF44Tuple2ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND|CILEN|CICN)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:VCF44Tuple4Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 4 * ?altCount) +}""" ] . + +vcfc:VCF44Tuple4ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.4: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.4" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-4.5.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-4.5.shacl.ttl new file mode 100644 index 0000000..513605c --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-4.5.shacl.ttl @@ -0,0 +1,78 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:VCF45FileGate a sh:NodeShape ; sh:targetClass vcfc:VCF45File ; sh:property [ sh:path vcfc:fileFormat ; sh:hasValue "VCFv4.5" ; sh:maxCount 1 ] . +vcfc:VCF45NumberShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: Number codes must be supported by this version and field kind." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . ?d vcfc:fieldNumber ?number . +{ ?d a vcfc:INFOHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.])$")) } UNION { ?d a vcfc:FORMATHeaderLine . FILTER(!REGEX(?number,"^([0-9]+|A|R|G|[.]|P|LA|LR|LG|M)$")) } +}""" ] . + +vcfc:VCF45INFOReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: reserved INFO declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:INFOHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AA" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="AC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="AF" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="AN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="BQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CIGAR" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DB" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="END" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="H2" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="H3" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="MQ0" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="NS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="SB" && (?number!="4" || ?type!=vcfc:IntegerType)) || (?id="SOMATIC" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="VALIDATED" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="1000G" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="IMPRECISE" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="NOVEL" && (?number!="0" || ?type!=vcfc:FlagType)) || (?id="SVTYPE" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="SVLEN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="CIPOS" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIEND" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="HOMLEN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="HOMSEQ" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="BKPTID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="MEINFO" && (?number!="." || ?type!=vcfc:StringType)) || (?id="METRANS" && (?number!="." || ?type!=vcfc:StringType)) || (?id="DGVID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DBVARID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="DBRIPID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="MATEID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="PARID" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="EVENT" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="EVENTTYPE" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="CILEN" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="A" || ?type!=vcfc:FloatType)) || (?id="CICN" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="SVCLAIM" && (?number!="A" || ?type!=vcfc:StringType)) || (?id="RN" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="RUS" && (?number!="." || ?type!=vcfc:StringType)) || (?id="RUL" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="RUC" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="RB" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="CIRUC" && (?number!="." || ?type!=vcfc:FloatType)) || (?id="CIRB" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="RUB" && (?number!="." || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF45FORMATReservedShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: reserved FORMAT declarations must match this version." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasHeader/vcfc:hasHeaderLine ?d . +?d a vcfc:FORMATHeaderLine ; vcfc:fieldId ?id ; vcfc:fieldNumber ?number ; vcfc:fieldType ?type . +FILTER((?id="AD" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADF" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="ADR" && (?number!="R" || ?type!=vcfc:IntegerType)) || (?id="DP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="EC" && (?number!="A" || ?type!=vcfc:IntegerType)) || (?id="LEN" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="FT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="GL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="GQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="GT" && (?number!="1" || ?type!=vcfc:StringType)) || (?id="HQ" && (?number!="2" || ?type!=vcfc:IntegerType)) || (?id="LA" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="LAA" && (?number!="." || ?type!=vcfc:IntegerType)) || (?id="LAD" && (?number!="LR" || ?type!=vcfc:IntegerType)) || (?id="LADF" && (?number!="LR" || ?type!=vcfc:IntegerType)) || (?id="LADR" && (?number!="LR" || ?type!=vcfc:IntegerType)) || (?id="LEC" && (?number!="LA" || ?type!=vcfc:IntegerType)) || (?id="LGL" && (?number!="LG" || ?type!=vcfc:FloatType)) || (?id="LGP" && (?number!="LG" || ?type!=vcfc:FloatType)) || (?id="LPL" && (?number!="LG" || ?type!=vcfc:IntegerType)) || (?id="LPP" && (?number!="LG" || ?type!=vcfc:IntegerType)) || (?id="M5mC" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5mC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5mC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5hmC" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5hmC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5hmC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5fC" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5fC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5fC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5caC" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5caC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5caC" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5hmU" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5hmU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5hmU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5fU" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5fU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5fU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M5caU" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM5caU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM5caU" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M6mA" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM6mA" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM6mA" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="M8oxoG" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPM8oxoG" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADM8oxoG" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="MXaoN" && (?number!="M" || ?type!=vcfc:FloatType)) || (?id="DPMXaoN" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="ADMXaoN" && (?number!="M" || ?type!=vcfc:IntegerType)) || (?id="MQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PL" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PP" && (?number!="G" || ?type!=vcfc:IntegerType)) || (?id="PQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PS" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="PSL" && (?number!="P" || ?type!=vcfc:StringType)) || (?id="PSO" && (?number!="P" || ?type!=vcfc:IntegerType)) || (?id="PSQ" && (?number!="P" || ?type!=vcfc:IntegerType)) || (?id="CN" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CICN" && (?number!="2" || ?type!=vcfc:FloatType)) || (?id="CNQ" && (?number!="1" || ?type!=vcfc:FloatType)) || (?id="CNL" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="CNP" && (?number!="G" || ?type!=vcfc:FloatType)) || (?id="NQ" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="HAP" && (?number!="1" || ?type!=vcfc:IntegerType)) || (?id="AHAP" && (?number!="1" || ?type!=vcfc:IntegerType))) +}""" ] . + +vcfc:VCF45Tuple2Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND|CILEN|CICN)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 2 * ?altCount) +}""" ] . + +vcfc:VCF45Tuple2ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(CIPOS|CIEND|CILEN|CICN)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:VCF45Tuple4Shape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: SV tuple cardinality must match the version-specific rule." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasRecord ?record . ?record vcfc:alt ?alt ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?altCount) +FILTER(STR(?raw)!="." && (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) != 4 * ?altCount) +}""" ] . + +vcfc:VCF45Tuple4ItemsShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "VCFv4.5: parsed SV tuples must contain all value items." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat "VCFv4.5" ; vcfc:hasRecord ?record . ?record vcfc:hasAltAllele ?allele ; vcfc:hasCall/vcfc:hasInfoValue ?fv . +?fv vcfc:declaredBy/vcfc:fieldId ?id ; vcfc:fieldValue ?raw . FILTER(REGEX(?id,"^(MEINFO|METRANS)$")) +{ SELECT $this ?fv (COUNT(DISTINCT ?item) AS ?n) WHERE { ?fv vcfc:fieldValue ?raw . OPTIONAL { ?fv vcfc:hasValueItem ?item } } GROUP BY $this ?fv } +FILTER(?n != (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-core-consistency.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-core-consistency.shacl.ttl new file mode 100644 index 0000000..cc3be6e --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-core-consistency.shacl.ttl @@ -0,0 +1,292 @@ +@prefix sh: . +@prefix vcfc: . +@prefix xsd: . +# Generated by scripts/build-shacl-profiles.py; edit the generator. + +vcfc:HeaderVersionAgreementShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "File and fileformat header versions must agree." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fileFormat ?version ; vcfc:hasHeader/vcfc:hasHeaderLine ?line . +?line a vcfc:FileFormatHeaderLine ; vcfc:headerValue ?value . FILTER (?value != ?version) +}""" ] . + +vcfc:HeaderDeclarationAgreementShape a sh:NodeShape ; sh:targetClass vcfc:StructuredHeaderLine ; + sh:sparql [ sh:message "Header attributes must agree with their parsed declaration properties." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:hasAttribute ?a . ?a vcfc:attributeKey ?key ; vcfc:attributeValue ?value . +OPTIONAL { $this vcfc:fieldId ?fieldId } +OPTIONAL { $this vcfc:contigId ?contigId } +OPTIONAL { $this vcfc:filterId ?filterId } +OPTIONAL { $this vcfc:altId ?altId } +OPTIONAL { $this vcfc:fieldNumber ?number } +OPTIONAL { $this vcfc:fieldType ?type } +FILTER ((?key="ID" && ((BOUND(?fieldId) && ?value!=?fieldId) || (BOUND(?contigId) && ?value!=?contigId) || (BOUND(?filterId) && ?value!=?filterId) || (BOUND(?altId) && ?value!=?altId))) || +(?key="Number" && BOUND(?number) && ?value!=?number) || +(?key="Type" && BOUND(?type) && ?value!=REPLACE(STRAFTER(STR(?type),"#"),"Type$",""))) +}""" ] . + +vcfc:FixedArityAgreementShape a sh:NodeShape ; sh:targetClass vcfc:FieldDefinition ; + sh:sparql [ sh:message "Fixed Number and fieldNumberInteger must agree." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:fieldNumber ?number ; vcfc:fieldNumberInteger ?integer . +FILTER(!REGEX(?number,"^[0-9]+$") || xsd:integer(?number) != ?integer) +}""" ] . + +vcfc:AlleleRawAgreementShape a sh:NodeShape ; sh:targetClass vcfc:VCFRecord ; + sh:sparql [ sh:message "REF/ALT components must agree with their raw columns and indices." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:ref ?ref ; vcfc:alt ?alt . +{ $this vcfc:ref ?ref ; vcfc:hasReferenceAllele ?a . ?a vcfc:alleleValue ?value . FILTER(STR(?value)!=STR(?ref)) } +UNION { $this vcfc:alt ?alt ; vcfc:hasAltAllele ?a . ?a vcfc:alleleIndex ?index ; vcfc:alleleValue ?value . +BIND(REPLACE(STR(?alt),CONCAT("^([^,]*,){",STR(?index-1),"}([^,]*).*$"),"$2") AS ?token) +FILTER(STR(?value)!=?token || STR(?alt)="." || ?index > (STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) } +UNION { $this vcfc:hasAltAllele ?a,?b . ?a vcfc:alleleIndex ?idx . ?b vcfc:alleleIndex ?idx . FILTER(?a!=?b) } +}""" ] . + +vcfc:AlleleCountAgreementShape a sh:NodeShape ; sh:targetClass vcfc:VCFRecord ; + sh:sparql [ sh:message "A materialized ALT collection must contain every ALT allele." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:alt ?raw . FILTER EXISTS { $this vcfc:hasAltAllele ?any } +{ SELECT $this (COUNT(DISTINCT ?a) AS ?n) WHERE { $this vcfc:alt ?raw . OPTIONAL { $this vcfc:hasAltAllele ?a } } GROUP BY $this } +FILTER(?n != IF(STR(?raw)=".",0,(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1))) +}""" ] . + +vcfc:GenotypeCallAgreementShape a sh:NodeShape ; sh:targetClass vcfc:Genotype ; + sh:sparql [ sh:message "GT text, ploidy, call indices and called alleles must agree." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:genotypeString ?raw ; vcfc:ploidy ?ploidy . +BIND(REPLACE(STR(?raw),"^[|/]","") AS ?gt) +{ SELECT $this (COUNT(DISTINCT ?call) AS ?n) WHERE { $this vcfc:genotypeString ?raw . OPTIONAL { $this vcfc:hasAlleleCall ?call } } GROUP BY $this } +FILTER (?ploidy != ?n || ?ploidy != (STRLEN(STR(?gt)) - STRLEN(REPLACE(STR(?gt), "[|/]", "")) + 1)) +}""" ] . + +vcfc:GenotypeAlleleAgreementShape a sh:NodeShape ; sh:targetClass vcfc:Genotype ; + sh:sparql [ sh:message "Each parsed GT allele and phase indicator must match its source position." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:genotypeString ?raw ; vcfc:hasAlleleCall ?call . +?call vcfc:callIndex ?index ; vcfc:isNoCall ?missing . +BIND(REPLACE(STR(?raw),"^[|/]","") AS ?gt) +BIND(REPLACE(?gt,CONCAT("^([^|/]*[|/]){",STR(?index),"}([^|/]*).*$"),"$2") AS ?token) +OPTIONAL { ?call vcfc:calledAllele/vcfc:alleleIndex ?allele } +OPTIONAL { ?call vcfc:phaseIndicator ?phase } +BIND(IF(?index=0,IF(REGEX(STR(?raw),"^[|/]"),SUBSTR(STR(?raw),1,1),IF(CONTAINS(STR(?raw),"/"),"/","|")),REPLACE(?gt,CONCAT("^([^|/]*[|/]){",STR(?index-1),"}[^|/]*([|/]).*$"),"$2")) AS ?expectedPhase) +FILTER (?missing != (?token=".") || (?missing && BOUND(?allele)) || (!?missing && (!BOUND(?allele) || STR(?allele)!=?token)) || (BOUND(?phase) && ?phase!=?expectedPhase) || ?index >= (STRLEN(STR(?gt)) - STRLEN(REPLACE(STR(?gt), "[|/]", "")) + 1)) +}""" ] . + +vcfc:GenotypeIndexUniqueShape a sh:NodeShape ; sh:targetClass vcfc:Genotype ; + sh:sparql [ sh:message "GT call indices must be unique." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:hasAlleleCall ?a,?b . ?a vcfc:callIndex ?i . ?b vcfc:callIndex ?i . FILTER(?a!=?b) +}""" ] . + +vcfc:ValueItemRawAgreementShape a sh:NodeShape ; sh:targetClass vcfc:FieldValueItem ; + sh:sparql [ sh:message "Value items must agree with raw field tokens and contiguous zero-based indices." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?field vcfc:hasValueItem $this ; vcfc:fieldValue ?raw . +$this vcfc:valueIndex ?idx ; vcfc:itemValue ?value . +BIND(REPLACE(STR(?raw),CONCAT("^([^,]*,){",STR(?idx),"}([^,]*).*$"),"$2") AS ?token) +FILTER (?idx >= (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1) || (STR(?value)!=?token && !(isNumeric(?value) && xsd:double(?value)=xsd:double(?token)))) +}""" ] . + +vcfc:ValueItemCountShape a sh:NodeShape ; sh:targetClass vcfc:FieldValueItem ; + sh:sparql [ sh:message "A materialized field list must contain each source item exactly once." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?field vcfc:hasValueItem $this ; vcfc:fieldValue ?raw . +{ SELECT $this ?field (COUNT(DISTINCT ?item) AS ?n) (COUNT(DISTINCT ?idx) AS ?indices) WHERE { ?field vcfc:hasValueItem ?item . ?item vcfc:valueIndex ?idx } GROUP BY $this ?field } +FILTER(?n != ?indices || ?n != IF(STR(?raw)="",0,(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1))) +}""" ] . + +vcfc:InfoFieldValueLexicalShape a sh:NodeShape ; sh:targetClass vcfc:InfoFieldValue ; + sh:sparql [ sh:message "Field values must match their declared VCF type." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:declaredBy/vcfc:fieldType ?type ; vcfc:fieldValue ?raw . +FILTER(STR(?raw)!=".") +FILTER((?type=vcfc:IntegerType && !REGEX(STR(?raw),"^([+-]?[0-9]+|[.])(,([+-]?[0-9]+|[.]))*$")) || +(?type=vcfc:FloatType && !REGEX(STR(?raw),"^([-+]?[0-9]*[.]?[0-9]+([eE][-+]?[0-9]+)?|[-+]?(INF|INFINITY|NAN)|[.])(,([-+]?[0-9]*[.]?[0-9]+([eE][-+]?[0-9]+)?|[-+]?(INF|INFINITY|NAN)|[.]))*$","i")) || +(?type=vcfc:CharacterType && !REGEX(STR(?raw),"^([^,]|(%[0-9A-Fa-f]{2})+)(,([^,]|(%[0-9A-Fa-f]{2})+))*$")) || ?type=vcfc:FlagType) +}""" ] . + +vcfc:InfoFieldValueCardinalityShape a sh:NodeShape ; sh:targetClass vcfc:InfoFieldValue ; + sh:sparql [ sh:message "Field cardinality must agree with Number, alleles and ploidy." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:declaredBy ?d ; vcfc:fieldValue ?raw . ?d vcfc:fieldNumber ?number . +?record vcfc:hasCall ?c ; vcfc:alt ?alt . +{ ?c vcfc:hasInfoValue $this } +UNION { ?c vcfc:hasSampleCall ?sample . ?sample vcfc:hasFormatValue $this } +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?n) +OPTIONAL { ?sample vcfc:hasGenotype/vcfc:ploidy ?ploidy } +OPTIONAL { ?sample vcfc:hasFormatValue ?laa . ?laa vcfc:declaredBy/vcfc:fieldId "LAA" ; vcfc:fieldValue ?local } +BIND(COALESCE(?ploidy,2) AS ?p) +BIND(IF(!BOUND(?local) || STR(?local)="." || STR(?local)="",0,(STRLEN(STR(?local)) - STRLEN(REPLACE(STR(?local), ",", "")) + 1)) AS ?l) +BIND(IF(?number="LG",?l+1,?n+1) AS ?alleleN) +BIND(IF(REGEX(?number,"^[0-9]+$"),xsd:integer(?number),IF(?number="A",?n,IF(?number="R",?n+1,IF(?number="P",?p,IF(?number="LA",?l,IF(?number="LR",?l+1,IF(?number IN ("G","LG"),IF(?p=1,((?alleleN + 0))/1,IF(?p=2,((?alleleN + 0) * (?alleleN + 1))/2,IF(?p=3,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2))/6,IF(?p=4,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3))/24,IF(?p=5,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4))/120,IF(?p=6,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5))/720,IF(?p=7,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5) * (?alleleN + 6))/5040,IF(?p=8,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5) * (?alleleN + 6) * (?alleleN + 7))/40320,-1)))))))),-1))))))) AS ?expected) +BIND(IF(STR(?raw)="",0,(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) AS ?actual) +FILTER(STR(?raw)!="." && ?expected>=0 && ?expected!=?actual) +}""" ] . + +vcfc:FormatFieldValueLexicalShape a sh:NodeShape ; sh:targetClass vcfc:FormatFieldValue ; + sh:sparql [ sh:message "Field values must match their declared VCF type." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:declaredBy/vcfc:fieldType ?type ; vcfc:fieldValue ?raw . +FILTER(STR(?raw)!=".") +FILTER((?type=vcfc:IntegerType && !REGEX(STR(?raw),"^([+-]?[0-9]+|[.])(,([+-]?[0-9]+|[.]))*$")) || +(?type=vcfc:FloatType && !REGEX(STR(?raw),"^([-+]?[0-9]*[.]?[0-9]+([eE][-+]?[0-9]+)?|[-+]?(INF|INFINITY|NAN)|[.])(,([-+]?[0-9]*[.]?[0-9]+([eE][-+]?[0-9]+)?|[-+]?(INF|INFINITY|NAN)|[.]))*$","i")) || +(?type=vcfc:CharacterType && !REGEX(STR(?raw),"^([^,]|(%[0-9A-Fa-f]{2})+)(,([^,]|(%[0-9A-Fa-f]{2})+))*$")) || ?type=vcfc:FlagType) +}""" ] . + +vcfc:FormatFieldValueCardinalityShape a sh:NodeShape ; sh:targetClass vcfc:FormatFieldValue ; + sh:sparql [ sh:message "Field cardinality must agree with Number, alleles and ploidy." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:declaredBy ?d ; vcfc:fieldValue ?raw . ?d vcfc:fieldNumber ?number . +?record vcfc:hasCall ?c ; vcfc:alt ?alt . +{ ?c vcfc:hasInfoValue $this } +UNION { ?c vcfc:hasSampleCall ?sample . ?sample vcfc:hasFormatValue $this } +BIND(IF(STR(?alt)=".",0,(STRLEN(STR(?alt)) - STRLEN(REPLACE(STR(?alt), ",", "")) + 1)) AS ?n) +OPTIONAL { ?sample vcfc:hasGenotype/vcfc:ploidy ?ploidy } +OPTIONAL { ?sample vcfc:hasFormatValue ?laa . ?laa vcfc:declaredBy/vcfc:fieldId "LAA" ; vcfc:fieldValue ?local } +BIND(COALESCE(?ploidy,2) AS ?p) +BIND(IF(!BOUND(?local) || STR(?local)="." || STR(?local)="",0,(STRLEN(STR(?local)) - STRLEN(REPLACE(STR(?local), ",", "")) + 1)) AS ?l) +BIND(IF(?number="LG",?l+1,?n+1) AS ?alleleN) +BIND(IF(REGEX(?number,"^[0-9]+$"),xsd:integer(?number),IF(?number="A",?n,IF(?number="R",?n+1,IF(?number="P",?p,IF(?number="LA",?l,IF(?number="LR",?l+1,IF(?number IN ("G","LG"),IF(?p=1,((?alleleN + 0))/1,IF(?p=2,((?alleleN + 0) * (?alleleN + 1))/2,IF(?p=3,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2))/6,IF(?p=4,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3))/24,IF(?p=5,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4))/120,IF(?p=6,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5))/720,IF(?p=7,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5) * (?alleleN + 6))/5040,IF(?p=8,((?alleleN + 0) * (?alleleN + 1) * (?alleleN + 2) * (?alleleN + 3) * (?alleleN + 4) * (?alleleN + 5) * (?alleleN + 6) * (?alleleN + 7))/40320,-1)))))))),-1))))))) AS ?expected) +BIND(IF(STR(?raw)="",0,(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) AS ?actual) +FILTER(STR(?raw)!="." && ?expected>=0 && ?expected!=?actual) +}""" ] . + +vcfc:TypedFieldIntegerBoundsShape a sh:NodeShape ; sh:targetClass vcfc:FieldValueItem ; + sh:sparql [ sh:message "VCF Integer values must fit the permitted signed 32-bit range." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?field vcfc:hasValueItem $this ; vcfc:declaredBy/vcfc:fieldType vcfc:IntegerType . +$this vcfc:itemValue ?v . FILTER(STR(?v)!="." && (xsd:integer(?v) < -2147483640 || xsd:integer(?v)>2147483647)) +}""" ] . + +vcfc:MatrixDimensionShape a sh:NodeShape ; sh:targetClass vcfc:FormatValueVector ; + sh:sparql [ sh:message "VCFTextVector must have exactly one position per sample." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:valueEncoding vcfc:VCFTextVector ; (vcfc:encodedValues) ?raw . +?matrix vcfc:hasFormatValueVector $this ; vcfc:appliesToSampleSet ?set . +{ SELECT $this ?set (COUNT(DISTINCT ?sample) AS ?n) WHERE { ?set vcfc:hasSample ?sample } GROUP BY $this ?set } +FILTER((STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), "\\t", "")) + 1) != ?n) +}""" ] . + +vcfc:ComponentIdentifierShape a sh:NodeShape ; sh:targetClass vcfc:VCFFile ; + sh:sparql [ sh:message "Individual record identifiers must be unique across records." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:hasRecord ?a,?b . +?a vcfc:hasIdentifier/vcfc:identifierValue ?id . ?b vcfc:hasIdentifier/vcfc:identifierValue ?id . FILTER(?a!=?b) +}""" ] . + +vcfc:IdentifierRawShape a sh:NodeShape ; sh:targetClass vcfc:RecordIdentifier ; + sh:sparql [ sh:message "Parsed ID components must agree with their raw ID column." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?record vcfc:hasIdentifier $this ; vcfc:recordId ?raw . $this vcfc:componentIndex ?i ; vcfc:identifierValue ?v . +BIND(REPLACE(STR(?raw),CONCAT("^([^;]*;){",STR(?i-1),"}([^;]*).*$"),"$2") AS ?expected) +FILTER(STR(?raw)="." || ?v!=?expected || ?i>(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ";", "")) + 1)) +}""" ] . + +vcfc:FilterRawShape a sh:NodeShape ; sh:targetClass vcfc:FilterCode ; + sh:sparql [ sh:message "Parsed filter failures must agree with FILTER or FT." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?parent vcfc:hasFilterCode $this . { ?parent vcfc:filter ?raw } UNION { ?parent vcfc:sampleFilter ?raw } +$this vcfc:componentIndex ?i ; vcfc:filterCodeValue ?v . +BIND(REPLACE(STR(?raw),CONCAT("^([^;]*;){",STR(?i-1),"}([^;]*).*$"),"$2") AS ?expected) +FILTER(STR(?raw) IN (".","PASS") || ?v!=?expected || ?i>(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ";", "")) + 1)) +}""" ] . + +vcfc:LocalMembershipShape a sh:NodeShape ; sh:targetClass vcfc:LocalAlleleSet ; + sh:sparql [ sh:message "Local membership order must agree with sample LAA and record-global ALT indices." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?sample vcfc:hasGenotype/vcfc:hasLocalAlleleSet $this ; vcfc:hasFormatValue ?fv . ?fv vcfc:declaredBy/vcfc:fieldId "LAA" ; vcfc:fieldValue ?raw . +$this vcfc:hasLocalAlleleMembership ?member . ?member vcfc:localIndex ?i ; vcfc:localAllele/vcfc:alleleIndex ?a . +BIND(REPLACE(STR(?raw),CONCAT("^([^,]*,){",STR(?i-1),"}([^,]*).*$"),"$2") AS ?expected) +FILTER(STR(?a)!=?expected || ?i>(STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ",", "")) + 1)) +}""" ] . + +vcfc:ReferenceBlockExtentShape a sh:NodeShape ; sh:targetClass vcfc:ReferenceBlock ; + sh:sparql [ sh:message "A sample reference block must agree with FORMAT LEN and record POS." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?sample vcfc:hasReferenceBlock $this ; vcfc:hasFormatValue ?fv . ?fv vcfc:declaredBy/vcfc:fieldId "LEN" ; vcfc:fieldValue ?raw . +?record vcfc:pos ?pos ; vcfc:hasCall/vcfc:hasSampleCall ?sample . +$this vcfc:referenceBlockLength ?length ; vcfc:endPosition ?end . FILTER(?length!=xsd:integer(?raw) || ?end!=?pos+?length-1) +}""" ] . + +vcfc:FormatKeyRawAgreementShape a sh:NodeShape ; sh:targetClass vcfc:FormatKey ; + sh:sparql [ sh:message "Declared FORMAT keys must agree with the FORMAT column order." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?call vcfc:hasFormatKey $this ; vcfc:formatRaw ?raw . +$this vcfc:fieldIndex ?i ; vcfc:declaredBy/vcfc:fieldId ?key . +BIND(REPLACE(STR(?raw),CONCAT("^([^:]*:){",STR(?i-1),"}([^:]*).*$"),"$2") AS ?expected) +FILTER(?key != ?expected || ?i > (STRLEN(STR(?raw)) - STRLEN(REPLACE(STR(?raw), ":", "")) + 1)) +}""" ] . + +vcfc:AssemblyContigAgreementShape a sh:NodeShape ; sh:targetClass vcfc:VCFRecord ; + sh:sparql [ sh:message "A bracketed CHROM must agree with its parsed assembly contig and must not also name a declared contig." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +$this vcfc:chrom ?chrom . +OPTIONAL { $this vcfc:chromAssemblyContig/vcfc:assemblyContigId ?parsed } +OPTIONAL { $this vcfc:chromosome ?declared } +BIND(REGEX(STR(?chrom),"^<[^<>,\\\\s]+>$") AS ?bracketed) +FILTER((?bracketed && (!BOUND(?parsed) || CONCAT("<",?parsed,">") != STR(?chrom) || BOUND(?declared))) + || (!?bracketed && BOUND(?parsed))) +}""" ] . + +vcfc:PaddingInterpretationAgreementShape a sh:NodeShape ; sh:targetClass vcfc:PaddingInterpretation ; + sh:sparql [ sh:message "A padding interpretation must agree with its rule, and its anchor must follow from POS and side." ; + sh:select """PREFIX vcfc: +PREFIX xsd: +SELECT $this WHERE { +?record vcfc:hasPaddingInterpretation $this ; vcfc:pos ?pos ; vcfc:ref ?ref . +$this vcfc:paddingRule ?rule ; vcfc:paddingBaseCount ?count . +OPTIONAL { $this vcfc:paddingSide ?side } +OPTIONAL { $this vcfc:paddingAnchorPosition ?anchor } +BIND(?rule IN (vcfc:SimpleIndelPadding, vcfc:SymbolicAllelePadding, vcfc:PositionOnePadding) AS ?determined) +BIND(IF(?side = vcfc:PaddingBefore, ?pos, ?pos + STRLEN(STR(?ref)) - 1) AS ?expected) +FILTER((?determined && (?count != 1 || !BOUND(?side) || !BOUND(?anchor) || ?anchor != ?expected)) + || (!?determined && (?count != 0 || BOUND(?side) || BOUND(?anchor))) + || (?rule = vcfc:PositionOnePadding && (?side != vcfc:PaddingAfter || ?pos != 1))) +}""" ] . diff --git a/vcf_rdfizer_data/shacl/vcf-core-vocabulary-sparql.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-core-vocabulary-sparql.shacl.ttl new file mode 100644 index 0000000..ffd3f8e --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-core-vocabulary-sparql.shacl.ttl @@ -0,0 +1,212 @@ +@prefix sh: . +@prefix vcfc: . + +################################################################# +# VCF Core cross-resource constraints +# +# Kept separate from vcf-core-vocabulary.shacl.ttl so validators without +# SPARQL support can apply the portable core profile. +################################################################# + +vcfc:HeaderOrderingSparqlShape a sh:NodeShape ; + sh:targetClass vcfc:VCFHeader ; + sh:sparql [ + sh:message "Header-line indices must be unique within a VCF header."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasHeaderLine ?left, ?right . + ?left vcfc:lineIndex ?index . + ?right vcfc:lineIndex ?index . + FILTER (?left != ?right) + } + """ + ] ; + sh:sparql [ + sh:message "The fileformat header line must have lineIndex 1."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasHeaderLine ?line . + ?line a vcfc:FileFormatHeaderLine ; vcfc:lineIndex ?index . + FILTER (?index != 1) + } + """ + ] ; + sh:sparql [ + sh:message "Structured header-line ID attributes must be unique within their header type."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasHeaderLine ?left, ?right . + ?left a ?type ; vcfc:hasAttribute ?leftAttribute . + ?right a ?type ; vcfc:hasAttribute ?rightAttribute . + ?leftAttribute vcfc:attributeKey "ID" ; vcfc:attributeValue ?id . + ?rightAttribute vcfc:attributeKey "ID" ; vcfc:attributeValue ?id . + FILTER (?left != ?right) + FILTER (?type IN (vcfc:INFOHeaderLine, vcfc:FORMATHeaderLine, + vcfc:FILTERHeaderLine, vcfc:ALTHeaderLine, + vcfc:MetaHeaderLine, vcfc:SampleHeaderLine, + vcfc:PedigreeHeaderLine, vcfc:ContigHeaderLine)) + } + """ + ] . + +vcfc:VCFFileOrderingSparqlShape a sh:NodeShape ; + sh:targetClass vcfc:VCFFile ; + sh:sparql [ + sh:message "Record indices must be unique within a VCF file."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?left, ?right . + ?left vcfc:recordIndex ?index . + ?right vcfc:recordIndex ?index . + FILTER (?left != ?right) + } + """ + ] ; + sh:sparql [ + sh:message "Sample names and sample indices must be unique within a VCF file's SampleSet."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasSampleSet ?set . + ?set vcfc:hasSample ?left, ?right . + FILTER (?left != ?right) + { ?left vcfc:sampleName ?name . ?right vcfc:sampleName ?name } + UNION + { ?left vcfc:sampleIndex ?index . ?right vcfc:sampleIndex ?index } + } + """ + ] ; + sh:sparql [ + sh:message "Non-missing record ID strings must be unique within a VCF file."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?left, ?right . + ?left vcfc:recordId ?id . + ?right vcfc:recordId ?id . + FILTER (?left != ?right && STR(?id) != ".") + } + """ + ] ; + sh:sparql [ + sh:message "Record positions must be nondecreasing within a CHROM block."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?earlier, ?later . + ?earlier vcfc:chrom ?chrom ; vcfc:recordIndex ?earlierIndex ; vcfc:pos ?earlierPos . + ?later vcfc:chrom ?chrom ; vcfc:recordIndex ?laterIndex ; vcfc:pos ?laterPos . + FILTER (?earlierIndex < ?laterIndex && ?earlierPos > ?laterPos) + } + """ + ] ; + sh:sparql [ + sh:message "Records for one CHROM must form a contiguous record-index block."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?first, ?middle, ?last . + ?first vcfc:chrom ?chrom ; vcfc:recordIndex ?firstIndex . + ?middle vcfc:chrom ?otherChrom ; vcfc:recordIndex ?middleIndex . + ?last vcfc:chrom ?chrom ; vcfc:recordIndex ?lastIndex . + FILTER (?chrom != ?otherChrom) + FILTER (?firstIndex < ?middleIndex && ?middleIndex < ?lastIndex) + } + """ + ] ; + sh:sparql [ + sh:message "A VCF RDF graph must not mix expanded sample calls and condensed call matrices."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord/vcfc:hasCall ?expandedCall, ?condensedCall . + ?expandedCall vcfc:hasSampleCall ?sampleCall . + ?condensedCall vcfc:hasCallMatrix ?matrix . + } + """ + ] ; + sh:sparql [ + sh:message "GT must be the first FORMAT key when it is present."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord/vcfc:hasCall ?call . + ?call vcfc:formatRaw ?format . + FILTER (REGEX(STR(?format), "(^|:)GT(:|$)") && !REGEX(STR(?format), "^GT(:|$)")) + } + """ + ] . + +vcfc:RecommendedHeaderSparqlShape a sh:NodeShape ; + sh:targetClass vcfc:VCFFile ; + sh:severity sh:Warning ; + sh:sparql [ + sh:message "VCF 4.5 recommends contig declarations for files with records."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?record . + FILTER NOT EXISTS { + $this vcfc:hasHeader/vcfc:hasHeaderLine ?contig . + ?contig a vcfc:ContigHeaderLine . + } + } + """ + ] ; + sh:sparql [ + sh:message "VCF 4.5 recommends reference metadata for files with records."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasRecord ?record . + FILTER NOT EXISTS { $this vcfc:referenceGenome ?reference } + } + """ + ] . + +vcfc:RecordContigBoundSparqlShape a sh:NodeShape ; + sh:targetClass vcfc:VCFRecord ; + sh:sparql [ + sh:message "POS must not exceed the declared contig length plus one."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:pos ?pos ; vcfc:chromosome ?contig . + ?contig vcfc:contigLength ?length . + FILTER (?pos > ?length + 1) + } + """ + ] . + +# SV tuple rules are scoped by version in vcf-4.x.shacl.ttl. + +vcfc:SampleCallLocalAlleleSparqlShape a sh:NodeShape ; + sh:targetClass vcfc:SampleCall ; + sh:sparql [ + sh:message "A local-allele FORMAT key requires LAA, positioned after GT when GT is present."@en ; + sh:select """ + SELECT $this WHERE { + $this ^vcfc:hasSampleCall/vcfc:formatRaw ?format . + FILTER (REGEX(STR(?format), "(^|:)(LAD|LADF|LADR|LEC|LGL|LGP|LPL|LPP)(:|$)") && + (!REGEX(STR(?format), "(^|:)LAA(:|$)") || + (!REGEX(STR(?format), "(^|:)GT(:|$)") && !REGEX(STR(?format), "^LAA(:|$)")) || + (REGEX(STR(?format), "(^|:)GT(:|$)") && !REGEX(STR(?format), "^GT:LAA(:|$)")))) + } + """ + ] ; + sh:sparql [ + sh:message "PS and PSL must not both be populated for the same sample call."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:hasFormatValue ?ps, ?psl . + ?ps vcfc:declaredBy/vcfc:fieldId "PS" ; vcfc:fieldValue ?psValue . + FILTER (STR(?psValue) != ".") + ?psl vcfc:declaredBy/vcfc:fieldId "PSL" ; vcfc:fieldValue ?pslValue . + FILTER (STR(?pslValue) != ".") + } + """ + ] . + +vcfc:FieldAritySparqlShape a sh:NodeShape ; + sh:targetClass vcfc:FieldDefinition ; + sh:sparql [ + sh:message "fieldArity must agree with the lossless fieldNumber token."@en ; + sh:select """ + SELECT $this WHERE { + $this vcfc:fieldArity ?arity ; vcfc:fieldNumber ?number . + ?arity vcfc:arityCode ?code . + FILTER (STR(?number) != STR(?code)) + } + """ + ] . diff --git a/vcf_rdfizer_data/shacl/vcf-core-vocabulary.shacl.ttl b/vcf_rdfizer_data/shacl/vcf-core-vocabulary.shacl.ttl new file mode 100644 index 0000000..be80689 --- /dev/null +++ b/vcf_rdfizer_data/shacl/vcf-core-vocabulary.shacl.ttl @@ -0,0 +1,460 @@ +@prefix rdfs: . +@prefix sh: . +@prefix vcfc: . +@prefix rdf: . +@prefix xsd: . +@prefix prov: . +@prefix dcat: . +@prefix so: . +@prefix faldo: . + +################################################################# +# VCFFile shape +################################################################# + +vcfc:VCFFileShape a sh:NodeShape ; + sh:targetClass vcfc:VCFFile ; + + sh:property [ + sh:path vcfc:hasHeader ; + sh:minCount 1 ; + sh:maxCount 1 ; + sh:class vcfc:VCFHeader ; + ] ; + + sh:property [ + sh:path vcfc:fileFormat ; + sh:minCount 1 ; sh:maxCount 1 ; + sh:datatype xsd:string ; sh:pattern "^VCFv4\\.[0-5]$" ; + ] ; + + sh:property [ + sh:path vcfc:referenceGenome ; + sh:minCount 0 ; + sh:datatype xsd:string ; + ] ; + + sh:property [ + sh:path vcfc:fileDate ; + sh:minCount 0 ; + sh:datatype xsd:date ; + ] ; + + sh:property [ + sh:path vcfc:representationProfile ; + sh:minCount 1 ; + sh:maxCount 1 ; + sh:class vcfc:RepresentationProfile ; + ] ; + + sh:property [ + sh:path vcfc:hasSampleSet ; + sh:minCount 0 ; + sh:maxCount 1 ; + sh:class vcfc:SampleSet ; + ] ; + + sh:property [ + sh:path vcfc:hasRecord ; + sh:minCount 0 ; + sh:class vcfc:VCFRecord ; + ] . + +# Apply the VCF 4.5 version gate only when a producer explicitly opts into +# this profile. Generic vcfc:VCFFile resources can represent other VCF +# versions without being rejected by the VCF 4.5 vocabulary. +vcfc:VCF45ConformanceShape a sh:NodeShape ; + sh:targetClass vcfc:VCF45File ; + sh:property [ + sh:path vcfc:fileFormat ; + sh:minCount 1 ; + sh:maxCount 1 ; + sh:hasValue "VCFv4.5" ; + sh:message "VCF 4.5 conformance requires fileFormat VCFv4.5."@en + ] . + +################################################################# +# Header shape +################################################################# + +vcfc:VCFHeaderShape a sh:NodeShape ; + sh:targetClass vcfc:VCFHeader ; + sh:property [ sh:path vcfc:hasColumnHeader ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:ColumnHeaderLine ] ; + sh:property [ sh:path vcfc:hasHeaderLine ; sh:qualifiedValueShape [ sh:class vcfc:FileFormatHeaderLine ] ; sh:qualifiedMinCount 1 ; sh:qualifiedMaxCount 1 ] ; + + sh:property [ + sh:path vcfc:hasHeaderLine ; + sh:minCount 1 ; + sh:class vcfc:HeaderLine ; + ] . + +################################################################# +# INFO/FORMAT definition shapes (coverage of key attributes) +################################################################# + +vcfc:InfoFieldDefinitionShape a sh:NodeShape ; + sh:targetClass vcfc:InfoFieldDefinition ; + + sh:property [ sh:path vcfc:fieldId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^([A-Za-z_][0-9A-Za-z_.]*|1000G)$" ] ; + sh:property [ sh:path vcfc:fieldNumber ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^([0-9]+|A|R|G|LA|LR|LG|P|M|[.])$" ] ; + sh:property [ sh:path vcfc:fieldArity ; sh:minCount 0 ; sh:in (vcfc:ArityPerAlt vcfc:ArityPerAllele vcfc:ArityPerGenotype vcfc:ArityVariable vcfc:ArityPerLocalAlt vcfc:ArityPerLocalAllele vcfc:ArityPerLocalGenotype vcfc:ArityPerGTAllele vcfc:ArityPerBaseModification) ] ; + sh:property [ sh:path vcfc:fieldType ; sh:minCount 1 ; sh:maxCount 1 ; sh:in (vcfc:IntegerType vcfc:FloatType vcfc:FlagType vcfc:CharacterType vcfc:StringType) ] ; + sh:property [ sh:path vcfc:fieldDescription ; sh:minCount 1 ; sh:datatype xsd:string ] ; + sh:or ( + [ sh:property [ sh:path vcfc:fieldType ; sh:not [ sh:hasValue vcfc:FlagType ] ] ] + [ sh:property [ sh:path vcfc:fieldType ; sh:hasValue vcfc:FlagType ] ; sh:property [ sh:path vcfc:fieldNumber ; sh:hasValue "0" ] ] + ) . + +# Source and Version are recommendations for a concrete INFO declaration, not +# for every reusable registry definition. +vcfc:InfoHeaderLineRecommendedShape a sh:NodeShape ; + sh:targetClass vcfc:INFOHeaderLine ; + sh:property [ sh:path vcfc:fieldSource ; sh:minCount 1 ; sh:datatype xsd:string ; sh:severity sh:Warning ; sh:message "VCF 4.5 recommends Source on INFO declarations."@en ] ; + sh:property [ sh:path vcfc:fieldVersion ; sh:minCount 1 ; sh:datatype xsd:string ; sh:severity sh:Warning ; sh:message "VCF 4.5 recommends Version on INFO declarations."@en ] . + +vcfc:FormatFieldDefinitionShape a sh:NodeShape ; + sh:targetClass vcfc:FormatFieldDefinition ; + + # VCF 4.5's M, DPM, and ADM field families are pattern-defined registry + # rows, rather than a single concrete field ID. + sh:or ( + [ sh:property [ sh:path vcfc:fieldId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^[A-Za-z_][0-9A-Za-z_.]*$" ] ] + [ sh:property [ sh:path vcfc:keyPattern ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^(M|DPM|ADM)\\[0-9\\]\\+\\[ACGTUN\\]$" ] ] + ) ; + sh:property [ sh:path vcfc:fieldNumber ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^([0-9]+|A|R|G|LA|LR|LG|P|M|[.])$" ] ; + sh:property [ sh:path vcfc:fieldArity ; sh:minCount 0 ; sh:in (vcfc:ArityPerAlt vcfc:ArityPerAllele vcfc:ArityPerGenotype vcfc:ArityVariable vcfc:ArityPerLocalAlt vcfc:ArityPerLocalAllele vcfc:ArityPerLocalGenotype vcfc:ArityPerGTAllele vcfc:ArityPerBaseModification) ] ; + sh:property [ sh:path vcfc:fieldType ; sh:minCount 1 ; sh:maxCount 1 ; sh:in (vcfc:IntegerType vcfc:FloatType vcfc:CharacterType vcfc:StringType) ; sh:not [ sh:hasValue vcfc:FlagType ] ] ; + sh:property [ sh:path vcfc:fieldDescription ; sh:minCount 1 ; sh:datatype xsd:string ] . + +################################################################# +# VCFRecord + VariantCall shapes +################################################################# + +vcfc:VCFRecordShape a sh:NodeShape ; + sh:targetClass vcfc:VCFRecord ; + + sh:property [ sh:path vcfc:chrom ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:pos ; sh:minCount 1 ; sh:maxCount 1 ; sh:minInclusive 0 ; sh:node vcfc:IntegerLiteralShape ] ; + sh:property [ sh:path vcfc:ref ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^[ACGTNacgtn]+$" ] ; + # VCF 4.5 fixed field 5: "If there are no alternative alleles, then the + # MISSING value must be used." Such a record carries "."^^vcfc:Null, so the + # datatype must be a disjunction -- requiring xsd:string here rejected every + # REF-only and gVCF-style record. + sh:property [ + sh:path vcfc:alt ; + sh:minCount 1 ; + sh:or ( + [ sh:datatype xsd:string ] + [ sh:datatype vcfc:Null ] + ) ; + ] ; + + sh:property [ + sh:path vcfc:hasCall ; + sh:minCount 0 ; + sh:class vcfc:VariantCall ; + ] ; + + sh:property [ + sh:path vcfc:asSequenceAlteration ; + sh:minCount 0 ; + sh:class so:0001059 ; + ] . + +vcfc:VariantCallShape a sh:NodeShape ; + sh:targetClass vcfc:VariantCall ; + + # VCF 4.5 data types: a Float matches ^[-+]?[0-9]*\.?[0-9]+([eE][-+]?[0-9]+)?$ + # or ^[-+]?(INF|INFINITY|NAN)$ case-insensitively. xsd:double covers the + # numeric forms plus INF/-INF/NaN; the final branch covers the INFINITY and + # case-variant spellings that no XSD numeric datatype accepts. + # + # Because several datatypes conform, two correct producers can serialize the + # same QUAL differently -- vcf-core's own materializer writes vcfc:VCFFloat, + # VCF-RDFizer writes xsd:decimal, and both pass this shape. Producers SHOULD + # write vcfc:VCFFloat, the only branch that also carries INF/INFINITY/NAN. + # Consumers MUST NOT branch on DATATYPE(?qual): compare the lexical value, or + # accept the alternatives explicitly. Kept here rather than as an + # rdfs:comment because assess.py hashes ontology/*.ttl, so editing a comment + # there would invalidate every recorded review acceptance. + sh:property [ + sh:path vcfc:qual ; + sh:minCount 0 ; sh:maxCount 1 ; + sh:or ( + [ sh:datatype vcfc:VCFFloat ; sh:pattern "^([-+]?[0-9]*[.]?[0-9]+([eE][-+]?[0-9]+)?|[-+]?(INF|INFINITY|NAN))$" ; sh:flags "i" ] + [ sh:datatype xsd:decimal ] + [ sh:datatype xsd:double ] + [ sh:datatype xsd:float ] + [ sh:datatype vcfc:Null ] + [ sh:datatype xsd:string ; sh:pattern "^[-+]?(INF|INFINITY|NAN)$" ; sh:flags "i" ] + ) ; + ] ; + + sh:property [ + sh:path vcfc:filter ; + sh:minCount 0 ; + sh:or ( + [ sh:datatype xsd:string ] + [ sh:datatype vcfc:Null ] + ) ; + ] ; + + sh:property [ + sh:path vcfc:hasInfoValue ; + sh:minCount 0 ; + sh:class vcfc:InfoFieldValue ; + ] ; + + sh:property [ + sh:path vcfc:hasSampleCall ; + sh:minCount 0 ; + sh:class vcfc:SampleCall ; + ] ; + + sh:property [ + sh:path vcfc:hasCallMatrix ; + sh:minCount 0 ; + sh:class vcfc:CohortCallMatrix ; + ] . + +################################################################# +# SampleCall shape +################################################################# + +vcfc:SampleCallShape a sh:NodeShape ; + sh:targetClass vcfc:SampleCall ; + + sh:property [ sh:path vcfc:sampleId ; sh:minCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:forSample ; sh:minCount 0 ; sh:maxCount 1 ; sh:class vcfc:VCFSample ] ; + sh:property [ sh:path vcfc:hasFormatValue ; sh:minCount 0 ; sh:class vcfc:FormatFieldValue ] . + +################################################################# +# Expanded / condensed cohort representation shapes +################################################################# + +vcfc:VCFSampleShape a sh:NodeShape ; + sh:targetClass vcfc:VCFSample ; + + sh:property [ sh:path vcfc:sampleName ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:sampleIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:SampleSetShape a sh:NodeShape ; + sh:targetClass vcfc:SampleSet ; + + sh:property [ sh:path vcfc:hasSample ; sh:minCount 1 ; sh:class vcfc:VCFSample ] . + +vcfc:CohortCallMatrixShape a sh:NodeShape ; + sh:targetClass vcfc:CohortCallMatrix ; + + sh:property [ sh:path vcfc:appliesToSampleSet ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:SampleSet ] ; + sh:property [ sh:path vcfc:hasFormatValueVector ; sh:minCount 1 ; sh:class vcfc:FormatValueVector ] ; + sh:property [ sh:path vcfc:sampleDataRaw ; sh:minCount 0 ; sh:datatype xsd:string ] . + +vcfc:FormatValueVectorShape a sh:NodeShape ; + sh:targetClass vcfc:FormatValueVector ; + + sh:property [ sh:path vcfc:declaredBy ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:FormatFieldDefinition ] ; + sh:property [ sh:path vcfc:valueEncoding ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:VectorEncoding ] ; + sh:property [ sh:path vcfc:encodedValues ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] . + +################################################################# +# INFO/FORMAT value shapes +################################################################# + +vcfc:InfoFieldValueShape a sh:NodeShape ; + sh:targetClass vcfc:InfoFieldValue ; + + sh:property [ + sh:path vcfc:fieldValue ; + sh:minCount 0 ; + sh:or ( + [ sh:datatype xsd:string ] + [ sh:datatype vcfc:Null ] + ) ; + ] ; + sh:property [ sh:path vcfc:declaredBy ; sh:minCount 0 ; sh:class vcfc:InfoFieldDefinition ] . + +vcfc:FormatFieldValueShape a sh:NodeShape ; + sh:targetClass vcfc:FormatFieldValue ; + + sh:property [ + sh:path vcfc:fieldValue ; + sh:minCount 0 ; + sh:or ( + [ sh:datatype xsd:string ] + [ sh:datatype vcfc:Null ] + ) ; + ] ; + sh:property [ sh:path vcfc:declaredBy ; sh:minCount 0 ; sh:class vcfc:FormatFieldDefinition ] . + +################################################################# +# Header, lexical, allele, genotype, and SV shapes +################################################################# + +vcfc:HeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:HeaderLine ; + sh:property [ sh:path vcfc:headerKey ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:lineIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:StructuredHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:StructuredHeaderLine ; + sh:property [ sh:path vcfc:hasAttribute ; sh:minCount 1 ; sh:class vcfc:HeaderAttribute ] . + +vcfc:UnstructuredHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:UnstructuredHeaderLine ; + sh:property [ sh:path vcfc:headerValue ; sh:minCount 1 ; sh:datatype xsd:string ; sh:pattern "^[^<].*$" ] . + +vcfc:HeaderAttributeShape a sh:NodeShape ; + sh:targetClass vcfc:HeaderAttribute ; + sh:property [ sh:path vcfc:attributeKey ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:attributeValue ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:attributeIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:ColumnHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:ColumnHeaderLine ; + sh:property [ sh:path vcfc:hasGenotypeColumns ; sh:minCount 0 ; sh:class vcfc:VCFSample ] . + +vcfc:FileFormatHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:FileFormatHeaderLine ; + sh:property [ sh:path vcfc:headerKey ; sh:hasValue "fileformat" ] ; + sh:property [ sh:path vcfc:headerValue ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] . + +vcfc:ContigHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:ContigHeaderLine ; + sh:property [ sh:path vcfc:contigId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^[0-9A-Za-z!#$%&+./:;?@^_|~-][0-9A-Za-z!#$%&*+./:;=?@^_|~-]*$" ] ; + sh:property [ sh:path vcfc:contigUrl ; sh:minCount 0 ; sh:maxCount 1 ; sh:datatype xsd:anyURI ] . + +vcfc:FilterDefinitionShape a sh:NodeShape ; + sh:targetClass vcfc:FilterDefinition ; + sh:property [ sh:path vcfc:filterId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:not [ sh:hasValue "0" ] ] ; + sh:property [ sh:path vcfc:fieldDescription ; sh:minCount 1 ; sh:datatype xsd:string ] . + +vcfc:AltDefinitionShape a sh:NodeShape ; + sh:targetClass vcfc:AltDefinition ; + sh:property [ sh:path vcfc:altId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:fieldDescription ; sh:minCount 1 ; sh:datatype xsd:string ] . + +vcfc:MetaDefinitionShape a sh:NodeShape ; + sh:targetClass vcfc:MetaDefinition ; + sh:property [ sh:path vcfc:fieldId ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] ; + sh:property [ sh:path vcfc:fieldNumber ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^([0-9]+|A|R|G|LA|LR|LG|P|M|[.])$" ] ; + sh:property [ sh:path vcfc:fieldType ; sh:minCount 1 ; sh:maxCount 1 ; sh:in (vcfc:IntegerType vcfc:FloatType vcfc:CharacterType vcfc:StringType) ] . + +vcfc:SampleHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:SampleHeaderLine ; + sh:property [ sh:path vcfc:declaresSample ; sh:minCount 0 ; sh:maxCount 1 ; sh:class vcfc:SampleDeclaration ] . + +vcfc:PedigreeHeaderLineShape a sh:NodeShape ; + sh:targetClass vcfc:PedigreeHeaderLine ; + sh:property [ sh:path vcfc:pedigreeAncestor ; sh:minCount 0 ; sh:class vcfc:SampleDeclaration ] . + +vcfc:VCFRecordOrderShape a sh:NodeShape ; + sh:targetClass vcfc:VCFRecord ; + sh:property [ sh:path vcfc:recordIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:AlleleShape a sh:NodeShape ; + sh:targetClass vcfc:Allele ; + sh:property [ sh:path vcfc:alleleIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] ; + sh:property [ sh:path vcfc:alleleValue ; sh:minCount 1 ; sh:maxCount 1 ] ; + sh:property [ sh:path vcfc:alleleKind ; sh:minCount 1 ; sh:maxCount 1 ; sh:in (vcfc:BaseSequenceAllele vcfc:OverlappingDeletionAllele vcfc:MissingAllele vcfc:SymbolicAllele vcfc:UnspecifiedAllele vcfc:BreakendAllele) ] . + +vcfc:ReferenceAlleleShape a sh:NodeShape ; + sh:targetClass vcfc:ReferenceAllele ; + sh:property [ sh:path vcfc:alleleIndex ; sh:minInclusive 0 ; sh:maxInclusive 0 ] . + +vcfc:AltAlleleShape a sh:NodeShape ; + sh:targetClass vcfc:AltAllele ; + sh:property [ sh:path vcfc:alleleIndex ; sh:minInclusive 1 ] . + +vcfc:FieldValueItemShape a sh:NodeShape ; + sh:targetClass vcfc:FieldValueItem ; + sh:property [ sh:path vcfc:valueIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] ; + sh:property [ sh:path vcfc:itemValue ; sh:minCount 1 ; sh:maxCount 1 ] ; + sh:property [ sh:path vcfc:tupleArity ; sh:minCount 0 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:GenotypeShape a sh:NodeShape ; + sh:targetClass vcfc:Genotype ; + sh:property [ sh:path vcfc:genotypeString ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype vcfc:GenotypeString ; sh:pattern "^[|/]?([0-9]+|[.])([|/]([0-9]+|[.]))*$" ] ; + sh:property [ sh:path vcfc:ploidy ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] ; + sh:property [ sh:path vcfc:hasAlleleCall ; sh:minCount 1 ; sh:class vcfc:GenotypeAlleleCall ] . + +vcfc:GenotypeAlleleCallShape a sh:NodeShape ; + sh:targetClass vcfc:GenotypeAlleleCall ; + sh:property [ sh:path vcfc:callIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] ; + sh:property [ sh:path vcfc:isNoCall ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:boolean ] ; + sh:property [ sh:path vcfc:calledAllele ; sh:minCount 0 ; sh:maxCount 1 ; sh:class vcfc:Allele ] . + +vcfc:PhaseSetShape a sh:NodeShape ; + sh:targetClass vcfc:PhaseSet ; + sh:or ( + [ sh:property [ sh:path vcfc:phaseSetId ; sh:minCount 1 ; sh:datatype xsd:string ] ] + [ sh:property [ sh:path vcfc:phaseSetName ; sh:minCount 1 ; sh:datatype xsd:string ] ] + ) . + +vcfc:LocalAlleleSetShape a sh:NodeShape ; + sh:targetClass vcfc:LocalAlleleSet ; + sh:property [ sh:path vcfc:hasLocalAllele ; sh:minCount 0 ; sh:class vcfc:AltAllele ] . + +vcfc:SymbolicAlleleTypeShape a sh:NodeShape ; + sh:targetClass vcfc:SymbolicAlleleType ; + sh:property [ sh:path vcfc:svTypeCode ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ] . + +vcfc:BreakendShape a sh:NodeShape ; + sh:targetClass vcfc:Breakend ; + sh:or ( [ sh:property [ sh:path vcfc:isSingleBreakend ; sh:hasValue true ] ; sh:property [ sh:path vcfc:mateBreakend ; sh:maxCount 0 ] ] [ sh:property [ sh:path vcfc:breakendOrientation ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:BreakendOrientation ] ] ) . + +vcfc:VariantEventShape a sh:NodeShape ; + sh:targetClass vcfc:VariantEvent ; + sh:property [ sh:path vcfc:eventType ; sh:minCount 0 ; sh:maxCount 1 ; sh:class vcfc:EventType ] . + +vcfc:ConfidenceIntervalShape a sh:NodeShape ; + sh:targetClass vcfc:ConfidenceInterval ; + sh:property [ sh:path vcfc:ciLower ; sh:minCount 1 ; sh:maxCount 1 ; sh:lessThanOrEquals vcfc:ciUpper ; sh:node vcfc:NumericLiteralShape ] ; + sh:property [ sh:path vcfc:ciUpper ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:NumericLiteralShape ] . + +vcfc:TandemRepeatAlleleShape a sh:NodeShape ; + sh:targetClass vcfc:TandemRepeatAllele ; + sh:property [ sh:path vcfc:repeatSequenceCount ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] . + +vcfc:RepeatSequenceShape a sh:NodeShape ; + sh:targetClass vcfc:RepeatSequence ; + sh:property [ sh:path vcfc:repeatSequenceIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . + +vcfc:ReferenceBlockShape a sh:NodeShape ; + sh:targetClass vcfc:ReferenceBlock ; + sh:property [ sh:path vcfc:referenceBlockLength ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] . + +vcfc:BaseModificationShape a sh:NodeShape ; + sh:targetClass vcfc:BaseModification ; + sh:property [ sh:path vcfc:modifiedResidue ; sh:minCount 1 ; sh:maxCount 1 ] ; + sh:property [ sh:path vcfc:modifiedBaseOffset ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] . + +vcfc:RepresentationProfileShape a sh:NodeShape ; + sh:targetClass vcfc:VCFFile ; + sh:xone ( + [ sh:property [ sh:path vcfc:representationProfile ; sh:hasValue vcfc:ExpandedRepresentation ; sh:minCount 1 ; sh:maxCount 1 ] ] + [ sh:property [ sh:path vcfc:representationProfile ; sh:hasValue vcfc:CondensedRepresentation ; sh:minCount 1 ; sh:maxCount 1 ] ] + ) . + +# Integer-derived RDF datatypes share their numeric value space. +vcfc:IntegerLiteralShape a sh:NodeShape ; sh:or ( [ sh:datatype xsd:integer ] [ sh:datatype xsd:nonNegativeInteger ] [ sh:datatype xsd:positiveInteger ] [ sh:datatype xsd:nonPositiveInteger ] [ sh:datatype xsd:negativeInteger ] [ sh:datatype xsd:long ] [ sh:datatype xsd:int ] [ sh:datatype xsd:short ] [ sh:datatype xsd:byte ] [ sh:datatype xsd:unsignedLong ] [ sh:datatype xsd:unsignedInt ] [ sh:datatype xsd:unsignedShort ] [ sh:datatype xsd:unsignedByte ] ) . + +vcfc:NumericLiteralShape a sh:NodeShape ; + sh:or ( [ sh:node vcfc:IntegerLiteralShape ] [ sh:datatype xsd:decimal ] [ sh:datatype xsd:float ] [ sh:datatype xsd:double ] ) . +vcfc:NullLiteralShape a sh:NodeShape ; + sh:targetObjectsOf vcfc:fieldValue, vcfc:alt, vcfc:qual, vcfc:recordId, vcfc:filter, vcfc:itemValue ; + sh:or ( [ sh:not [ sh:datatype vcfc:Null ] ] [ sh:datatype vcfc:Null ; sh:pattern "^[.]$" ] ) . +vcfc:FixedArityShape a sh:NodeShape ; sh:targetClass vcfc:FieldDefinition ; + sh:property [ sh:path vcfc:fieldNumberInteger ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 0 ] . + +vcfc:RecordIdentifierShape a sh:NodeShape ; sh:targetClass vcfc:RecordIdentifier ; + sh:property [ sh:path vcfc:identifierValue ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^[^; \t\r\n]+$" ; sh:not [ sh:in (".") ] ] ; + sh:property [ sh:path vcfc:componentIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . +vcfc:FilterCodeShape a sh:NodeShape ; sh:targetClass vcfc:FilterCode ; + sh:property [ sh:path vcfc:filterCodeValue ; sh:minCount 1 ; sh:maxCount 1 ; sh:datatype xsd:string ; sh:pattern "^[^; \t\r\n]+$" ; sh:not [ sh:in ("0" "." "PASS") ] ] ; + sh:property [ sh:path vcfc:componentIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] . +vcfc:LocalAlleleMembershipShape a sh:NodeShape ; sh:targetClass vcfc:LocalAlleleMembership ; + sh:property [ sh:path vcfc:localIndex ; sh:minCount 1 ; sh:maxCount 1 ; sh:node vcfc:IntegerLiteralShape ; sh:minInclusive 1 ] ; + sh:property [ sh:path vcfc:localAllele ; sh:minCount 1 ; sh:maxCount 1 ; sh:class vcfc:AltAllele ] . +vcfc:AllelePhaseIndicatorShape a sh:NodeShape ; sh:targetClass vcfc:GenotypeAlleleCall ; + sh:property [ sh:path vcfc:phaseIndicator ; sh:maxCount 1 ; sh:in ("/" "|") ] . From 7ea562100b94434829e8a52b58098d3a7ed51a55 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:22:26 +0200 Subject: [PATCH 07/18] Refuse native COTTAS against a condensed graph until the hang is fixed Reproduced twice and archived. Native pycottas answered 18 of 27 queries against the condensed encoding of a 10,000-record fixture and then ran 41 h, and on a second attempt about 10 h, on q05_sample_genotype_counts without completing -- both at roughly 190% CPU. QLever answered all thirteen questions on that same cell in about a second each, and the sibling expanded encoding finished every engine including pycottas in 185-194 min. The defect is native pycottas x condensed encoding x a genotype-level query, and it cost 51 hours of unattended runtime to find. Naming cottas explicitly is now refused, with the measurement quoted and both alternatives named. --validation-engine all instead drops it with a warning and validates on the other three: asking for 'all' is a request for breadth, and failing the whole run serves that worse -- the campaign's 06_equivalence used 'all' and hung, where this completes and says why. --allow-cottas-condensed attempts it anyway. This is the interim, not the fix. The diagnosis belongs upstream in pycottas; recommendation 11 keeps it open. Co-Authored-By: Claude Opus 5 --- docs/limitations.md | 29 +++++ test/test_cottas_condensed_guard_unit.py | 134 +++++++++++++++++++++++ vcf_rdfizer.py | 103 +++++++++++++++++ 3 files changed, 266 insertions(+) create mode 100644 test/test_cottas_condensed_guard_unit.py diff --git a/docs/limitations.md b/docs/limitations.md index 5402f53..d265cb1 100644 --- a/docs/limitations.md +++ b/docs/limitations.md @@ -281,3 +281,32 @@ presented as if it did. - [Privacy policy design](privacy-policy-design.md) — the disclosure-control gap, and the proposal to close it - [VCF coverage matrix](vcf-coverage.md) — the element-by-element measurement - [Validation](validation.md) — the detailed "what is not tested" + +## Native COTTAS querying does not terminate on the condensed encoding + +`--validation-engine cottas` against a `--sample-representation condensed` +graph is **refused by default**, and this is a measurement rather than a +precaution. It was reproduced twice on a 10,000-record fixture: native pycottas +answered the fourteen preflight queries and Q1-Q4, then failed to complete +`q05_sample_genotype_counts`, running 41 hours in the first attempt and about +10 in the second, both at roughly 190% CPU. Both runs are archived whole, +container logs included. + +The scope is narrow, and worth stating precisely rather than as "COTTAS is +slow": + +* QLever answered all thirteen questions on that **same cell**, in about a + second each. +* The sibling **expanded** encoding completed every engine, pycottas included, + in 185-194 minutes. + +So the defect is native pycottas x the condensed genotype encoding x a +genotype-level query. The condensed encoding stores genotypes as `S + (V x F)` +rather than `V x S`, so a per-sample genotype query joins across the sample +block; the working hypothesis is a missing or unusable index on that join +column rather than data volume. + +Naming `cottas` explicitly is refused. `--validation-engine all` drops it with +a warning and validates with the other three, because asking for `all` is a +request for breadth and failing the whole run serves that worse. +`--allow-cottas-condensed` attempts it anyway. diff --git a/test/test_cottas_condensed_guard_unit.py b/test/test_cottas_condensed_guard_unit.py new file mode 100644 index 0000000..c7a4cfb --- /dev/null +++ b/test/test_cottas_condensed_guard_unit.py @@ -0,0 +1,134 @@ +"""The one combination that does not terminate, refused rather than started. + +Reproduced twice and archived under benchmarks_outputs__stalled/. Native +pycottas answered the fourteen preflight queries and Q1-Q4 against the +condensed encoding of a 10,000-record fixture, then failed to complete +``q05_sample_genotype_counts``: 41 hours in the first attempt, about 10 in the +second, both at roughly 190% CPU. QLever answered all thirteen questions on +that same cell, and the sibling expanded encoding finished every engine +including pycottas in 185-194 minutes. + +So the defect is narrow -- native pycottas x condensed encoding x a +genotype-level query -- and until it is diagnosed the honest thing is to refuse +the combination. 51 hours of unattended runtime were spent finding out the hard +way, and the per-query timeout could not bound it at the time. +""" + +import unittest + +import vcf_rdfizer +from test.helpers import VerboseTestCase + + +class RejectionTests(VerboseTestCase): + def test_naming_cottas_against_condensed_is_refused(self): + self.assertIsNotNone( + vcf_rdfizer.cottas_condensed_engine_rejection( + engines=["cottas"], sample_representation="condensed", allow=False + ) + ) + + def test_the_refusal_names_both_ways_out(self): + """A refusal that does not say what works instead is just a failure.""" + message = vcf_rdfizer.cottas_condensed_engine_rejection( + engines=["cottas"], sample_representation="condensed", allow=False + ) + self.assertIn("qlever", message) + self.assertIn("expanded", message) + self.assertIn("--allow-cottas-condensed", message) + + def test_the_refusal_cites_the_measurement_rather_than_asserting(self): + message = vcf_rdfizer.cottas_condensed_engine_rejection( + engines=["cottas"], sample_representation="condensed", allow=False + ) + self.assertIn("41 h", message) + self.assertIn("18 of 27", message) + + def test_cottas_against_expanded_is_allowed(self): + """Expanded completed every engine in 185-194 min; nothing to refuse.""" + self.assertIsNone( + vcf_rdfizer.cottas_condensed_engine_rejection( + engines=["cottas"], sample_representation="expanded", allow=False + ) + ) + + def test_other_engines_against_condensed_are_allowed(self): + for engine in ("qlever", "comunica", "hdt"): + self.assertIsNone( + vcf_rdfizer.cottas_condensed_engine_rejection( + engines=[engine], sample_representation="condensed", allow=False + ), + engine, + ) + + def test_the_override_lets_it_through(self): + self.assertIsNone( + vcf_rdfizer.cottas_condensed_engine_rejection( + engines=["cottas"], sample_representation="condensed", allow=True + ) + ) + + +class ResolutionTests(VerboseTestCase): + """Naming cottas is a request; asking for 'all' is a request for breadth.""" + + def test_an_explicit_request_is_refused_not_silently_dropped(self): + engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( + engines=["qlever", "cottas"], + sample_representation="condensed", + requested_all=False, + allow=False, + ) + self.assertIsNotNone(refusal) + self.assertIsNone(warning) + self.assertEqual(engines, ["qlever", "cottas"]) + + def test_engine_all_drops_cottas_and_runs_the_rest(self): + """06_equivalence used 'all' and hung twice; this completes instead.""" + engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( + engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), + sample_representation="condensed", + requested_all=True, + allow=False, + ) + self.assertIsNone(refusal) + self.assertIsNotNone(warning) + self.assertNotIn("cottas", engines) + self.assertEqual(engines, ["comunica", "qlever", "hdt"]) + + def test_the_drop_warning_says_what_still_ran_and_how_to_override(self): + _engines, warning, _refusal = vcf_rdfizer.cottas_condensed_resolution( + engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), + sample_representation="condensed", + requested_all=True, + allow=False, + ) + self.assertIn("comunica", warning) + self.assertIn("qlever", warning) + self.assertIn("--allow-cottas-condensed", warning) + + def test_engine_all_against_expanded_keeps_every_engine(self): + engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( + engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), + sample_representation="expanded", + requested_all=True, + allow=False, + ) + self.assertIsNone(refusal) + self.assertIsNone(warning) + self.assertEqual(engines, list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES)) + + def test_the_override_keeps_cottas_under_all(self): + engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( + engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), + sample_representation="condensed", + requested_all=True, + allow=True, + ) + self.assertIsNone(refusal) + self.assertIsNone(warning) + self.assertIn("cottas", engines) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index e8ba644..45fa2af 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -5042,6 +5042,87 @@ def hdt_strategy_rejection( return None +def cottas_condensed_resolution( + *, + engines: list[str], + sample_representation: str, + requested_all: bool, + allow: bool, +) -> tuple[list[str], str | None, str | None]: + """Resolve the cottas x condensed combination. + + Returns ``(engines, warning, refusal)``. Exactly one of warning/refusal is + ever set, and the distinction is deliberate: + + * naming ``cottas`` explicitly is a request the tool cannot honour, so it + is refused rather than silently ignored; but + * ``--validation-engine all`` is a request for breadth, and dropping the + one engine that cannot finish serves that better than failing the run. + The campaign's 06_equivalence used ``all`` and hung twice; under this it + would have completed on the other three engines and said why. + """ + refusal = cottas_condensed_engine_rejection( + engines=engines, sample_representation=sample_representation, allow=allow + ) + if refusal is None: + return engines, None, None + if not requested_all: + return engines, None, refusal + remaining = [engine for engine in engines if engine != "cottas"] + return ( + remaining, + "dropped the cottas engine: it does not terminate on the condensed " + "genotype encoding (two archived attempts ran 41 h and ~10 h on " + "q05_sample_genotype_counts). Validating with " + f"{', '.join(remaining)}. Pass --allow-cottas-condensed to include it.", + None, + ) + + +def cottas_condensed_engine_rejection( + *, + engines: list[str], + sample_representation: str, + allow: bool, +) -> str | None: + """Return why native COTTAS cannot query a condensed graph, or None. + + This is a known non-termination, reproduced twice and archived. Native + pycottas answered the fourteen preflight queries and Q1-Q4 against the + condensed encoding of a 10,000-record fixture, then failed to complete + q05_sample_genotype_counts: 41 hours in the first attempt, about 10 in the + second, both at roughly 190% CPU. QLever answered all thirteen questions on + that same cell, and the sibling expanded encoding finished every engine in + 185-194 minutes -- so the defect is specific to + native pycottas x condensed genotype encoding x a genotype-level query. + + The condensed encoding stores genotypes as S + (V x F) rather than V x S, so + a per-sample genotype query has to join across the sample block; the working + hypothesis is a missing or unusable index on that join column rather than + data volume, at a scale QLever answers in about a second. + + Refusing it is the interim. The alternative -- letting it start -- is what + cost 51 hours of unattended runtime, and the per-query timeout could not + bound it at the time. Once that is fixed and the hang is diagnosed, this + goes away; until then a clear refusal costs a user nothing they could + otherwise get. + """ + if allow or sample_representation != "condensed": + return None + if "cottas" not in engines: + return None + return ( + "--validation-engine cottas does not terminate on the condensed " + "genotype encoding. Reproduced twice on a 10,000-record fixture: native " + "pycottas answered 18 of 27 queries and then ran 41 h, and on a second " + "attempt about 10 h, on q05_sample_genotype_counts without completing.\n" + " Use --validation-engine qlever, which answered all thirteen " + "questions on that same cell, or --sample-representation expanded, " + "where every engine including cottas completes. Pass " + "--allow-cottas-condensed to attempt it anyway." + ) + + # --------------------------------------------------------------------------- # Run metrics layout: naming, manifest, and summary # --------------------------------------------------------------------------- @@ -9255,6 +9336,16 @@ def main(): "not scale to a cohort-sized aggregate" ), ) + parser.add_argument( + "--allow-cottas-condensed", + action="store_true", + help=( + "Attempt --validation-engine cottas against a condensed graph " + "anyway. Refused by default: native pycottas did not terminate on " + "q05_sample_genotype_counts there in two archived attempts (41 h, " + "then ~10 h) while qlever answered the same cell in about a second" + ), + ) parser.add_argument( "--no-shacl", action="store_true", @@ -9467,6 +9558,18 @@ def main(): raise ValueError("--qlever-port must be between 1 and 65535") validation_engines = parse_validation_engines(args.validation_engine) validation_artifacts = parse_validation_targets(args.validate_artifacts) + validation_engines, cottas_warning, cottas_rejection = ( + cottas_condensed_resolution( + engines=validation_engines, + sample_representation=args.sample_representation, + requested_all=(args.validation_engine or "").strip() == "all", + allow=args.allow_cottas_condensed, + ) + ) + if cottas_rejection is not None: + raise ValueError(cottas_rejection) + if cottas_warning is not None: + eprint(f"Warning: {cottas_warning}") if args.shacl_shapes is not None: shacl_shapes_path = Path(args.shacl_shapes).expanduser().resolve() if not shacl_shapes_path.is_file(): From 18fe3690d9231bbc8fdfa0e9387b41f7a8f2e048 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:30:37 +0200 Subject: [PATCH 08/18] Represent a header-only VCF, and stop reporting conformant dots Two things the validation stage flagged on the benchmark corpus. One was a real defect; the other was the check being wrong, and that distinction matters more than either fix. The defect: a header-only VCF lost the sample columns its #CHROM line declares. Q9 reported six missing predicates and an rdf:type count of 40 against 42, Q10 reported SampleSet and VCFSample missing, and the run exited non-zero rather than claiming success. The cause is narrow -- SampleRecordStream learns the source file name from the first data row, so with no rows it stays empty and all three emitters, which key their file IRI on it, return immediately. Sample columns are declared by the header, not by the data, so the name now falls back to the header-lines table. Verified on the awkward_no_records fixture: 0 triples before, 8 after, and every predicate and class the oracle expected. The false positive: preflight_missing_token_conformance reported 20 surviving . literals on the 100,000-record HG005 cell. The published shapes permit a bare dot on exactly two predicates -- vcfc:fieldNumber, whose pattern is ^([0-9]+|A|R|G|LA|LR|LG|P|M|[.])$ because Number=. is VCF's variable-cardinality token, and vcfc:genotypeString, where a fully-missing call is literally . -- and neither carries vcfc:Null because neither is missing. The check flagged conformant values; it now excludes those two predicates, in both the sample and its count. Recommendation 9 of implementation-improvements-from-benchmarking.md. The manuscript's claim that the graph retains 20 bad literals does not hold and needs correcting. Co-Authored-By: Claude Opus 5 --- .../preflight_missing_token_conformance.rq | 17 ++ ...eflight_missing_token_conformance_count.rq | 3 + test/test_zero_record_vcf_unit.py | 214 ++++++++++++++++++ vcf_rdfizer.py | 54 ++++- 4 files changed, 282 insertions(+), 6 deletions(-) create mode 100644 test/test_zero_record_vcf_unit.py diff --git a/src/validation/queries/common/preflight_missing_token_conformance.rq b/src/validation/queries/common/preflight_missing_token_conformance.rq index 7aaf36e..f06b38a 100644 --- a/src/validation/queries/common/preflight_missing_token_conformance.rq +++ b/src/validation/queries/common/preflight_missing_token_conformance.rq @@ -1,9 +1,26 @@ PREFIX vcfc: +# Plain "." literals, which should have been typed vcfc:Null. +# +# Two predicates are excluded because "." is a CONFORMANT value there, not a +# missing one, and the published shapes say so: +# +# vcfc:fieldNumber pattern ^([0-9]+|A|R|G|LA|LR|LG|P|M|[.])$ -- "." is +# VCF's variable-cardinality token (Number=.), which is a +# declared arity, not an absent value. +# vcfc:genotypeString pattern ^[|/]?([0-9]+|[.])([|/]([0-9]+|[.]))*$ -- a +# fully-missing haploid call is literally "./." or ".", +# and it carries vcfc:GenotypeString, not vcfc:Null. +# +# Without these exclusions the check reports every Number=. declaration in a +# header as a conformance failure. On the 100,000-record HG005 benchmark cell +# it returned 20 such rows, and all of them were declarations the vocabulary +# requires to look exactly that way. SELECT ?s ?p ?o (DATATYPE(?o) AS ?datatype) WHERE { ?s ?p ?o . FILTER(ISLITERAL(?o) && STR(?o) = ".") FILTER(DATATYPE(?o) != vcfc:Null) + FILTER(?p NOT IN (vcfc:fieldNumber, vcfc:genotypeString)) } LIMIT 100 diff --git a/src/validation/queries/common/preflight_missing_token_conformance_count.rq b/src/validation/queries/common/preflight_missing_token_conformance_count.rq index f77867e..f527d71 100644 --- a/src/validation/queries/common/preflight_missing_token_conformance_count.rq +++ b/src/validation/queries/common/preflight_missing_token_conformance_count.rq @@ -1,9 +1,12 @@ PREFIX vcfc: # Exact number of missing tokens serialized as a plain "." literal. +# The predicate exclusions must match preflight_missing_token_conformance.rq +# exactly, or the count and its sample describe different populations. SELECT (COUNT(*) AS ?anomalyCount) WHERE { ?s ?p ?o . FILTER(ISLITERAL(?o) && STR(?o) = ".") FILTER(DATATYPE(?o) != vcfc:Null) + FILTER(?p NOT IN (vcfc:fieldNumber, vcfc:genotypeString)) } diff --git a/test/test_zero_record_vcf_unit.py b/test/test_zero_record_vcf_unit.py new file mode 100644 index 0000000..f39457a --- /dev/null +++ b/test/test_zero_record_vcf_unit.py @@ -0,0 +1,214 @@ +"""A header-only VCF still has a header to represent. + +``awkward_no_records`` declares one sample column on its ``#CHROM`` line and +contains zero data records. The graph omitted the ``vcfc:SampleSet`` and +``vcfc:VCFSample`` resources that line declares, and the validation stage +caught it: Q9 reported six missing predicates (``hasGenotypeColumns``, +``hasSample``, ``hasSampleSet``, ``representationProfile``, ``sampleIndex``, +``sampleName``) and an ``rdf:type`` count of 40 against an expected 42; Q10 +reported ``SampleSet`` and ``VCFSample`` missing. The run exited non-zero +rather than reporting success on a silently incomplete graph. + +The cause was narrow. ``SampleRecordStream`` learns the source file name from +the first data row, so with no rows it stays empty -- and all three emitters +key their file IRI on it and return immediately. Sample columns are declared by +the header, not by the data, so the fallback reads the name from the +header-lines table instead. +""" + +import csv +import tempfile +import unittest +from pathlib import Path + +import vcf_rdfizer +from test.helpers import VerboseTestCase + +RECORDS_HEADER = [ + "SOURCE_FILE", "ROW_ID", "CHROM", "POS", "ID", "REF", "ALT", + "QUAL", "FILTER", "INFO", "FORMAT", +] + + +def write_tsvs(directory: Path, *, sample_ids: list[str], records: int, + source_file: str = "sample.vcf") -> tuple[Path, Path]: + """A records/header TSV pair, optionally with zero data rows.""" + records_tsv = directory / "sample.records.tsv" + headers_tsv = directory / "sample.header_lines.tsv" + with records_tsv.open("w", newline="", encoding="utf-8") as handle: + writer = csv.writer(handle, delimiter="\t") + writer.writerow(RECORDS_HEADER + [" ".join(sample_ids) or "SAMPLES"]) + for index in range(records): + writer.writerow([ + source_file, str(index + 1), "20", str(100 + index), ".", + "A", "G", "50", "PASS", "DP=30", "GT:DP", + " ".join("0/1:30" for _ in sample_ids), + ]) + with headers_tsv.open("w", newline="", encoding="utf-8") as handle: + writer = csv.writer(handle, delimiter="\t") + writer.writerow(["SOURCE_FILE", "LINE_INDEX", "KEY", "VALUE", "RAW"]) + writer.writerow([source_file, "1", "fileformat", "VCFv4.2", + "fileformat=VCFv4.2"]) + writer.writerow([ + source_file, "2", "FORMAT", + '', + 'FORMAT=', + ]) + return records_tsv, headers_tsv + + +class SourceFileFallbackTests(VerboseTestCase): + def test_the_source_file_is_recovered_from_the_header_table(self): + with tempfile.TemporaryDirectory() as tmp: + _records, headers = write_tsvs( + Path(tmp), sample_ids=["SAMPLE_A"], records=0 + ) + self.assertEqual( + vcf_rdfizer.source_file_from_header_lines(headers), "sample.vcf" + ) + + def test_a_missing_header_table_returns_an_empty_name(self): + with tempfile.TemporaryDirectory() as tmp: + self.assertEqual( + vcf_rdfizer.source_file_from_header_lines(Path(tmp) / "gone.tsv"), "" + ) + + def test_a_header_table_with_only_its_column_row_returns_empty(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "h.tsv" + path.write_text("SOURCE_FILE\tLINE_INDEX\tKEY\tVALUE\tRAW\n", + encoding="utf-8") + self.assertEqual(vcf_rdfizer.source_file_from_header_lines(path), "") + + +class ZeroRecordEmissionTests(VerboseTestCase): + def _emit(self, tmp, representation, *, sample_ids, records): + directory = Path(tmp) + records_tsv, headers_tsv = write_tsvs( + directory, sample_ids=sample_ids, records=records + ) + rdf = directory / "out.nt" + rdf.write_bytes(b"") + if representation == "expanded": + stats = vcf_rdfizer.append_expanded_sample_rdf( + records_tsv, rdf, headers_tsv, version=None + ) + else: + stats = vcf_rdfizer.append_condensed_sample_rdf( + records_tsv, headers_tsv, rdf + ) + return stats, rdf.read_text(encoding="utf-8") + + def test_expanded_emits_the_sample_set_for_a_zero_record_vcf(self): + """The six predicates Q9 reported missing must all come back.""" + with tempfile.TemporaryDirectory() as tmp: + stats, text = self._emit(tmp, "expanded", sample_ids=["SAMPLE_A"], + records=0) + self.assertGreater(stats["triples"], 0) + for term in ("hasSampleSet", "hasSample>", "sampleName", + "sampleIndex", "hasGenotypeColumns", + "representationProfile"): + self.assertIn(term, text, term) + + def test_expanded_emits_both_classes_q10_reported_missing(self): + with tempfile.TemporaryDirectory() as tmp: + _stats, text = self._emit(tmp, "expanded", sample_ids=["SAMPLE_A"], + records=0) + self.assertIn("#SampleSet", text) + self.assertIn("#VCFSample", text) + + def test_condensed_emits_the_sample_set_for_a_zero_record_vcf(self): + with tempfile.TemporaryDirectory() as tmp: + stats, text = self._emit(tmp, "condensed", sample_ids=["SAMPLE_A"], + records=0) + self.assertGreater(stats["triples"], 0) + self.assertIn("#SampleSet", text) + self.assertIn("#VCFSample", text) + + def test_the_sample_name_is_the_one_the_chrom_line_declared(self): + with tempfile.TemporaryDirectory() as tmp: + _stats, text = self._emit(tmp, "expanded", sample_ids=["NA12878"], + records=0) + self.assertIn('"NA12878"', text) + + def test_several_declared_samples_all_appear(self): + with tempfile.TemporaryDirectory() as tmp: + _stats, text = self._emit( + tmp, "expanded", sample_ids=["NA1", "NA2", "NA3"], records=0 + ) + for name in ("NA1", "NA2", "NA3"): + self.assertIn(f'"{name}"', text) + + def test_a_sites_only_zero_record_vcf_still_emits_no_sample_set(self): + """No declared columns means no set to emit; that part was correct.""" + with tempfile.TemporaryDirectory() as tmp: + _stats, text = self._emit(tmp, "expanded", sample_ids=[], records=0) + self.assertNotIn("#SampleSet", text) + self.assertNotIn("#VCFSample", text) + + def test_a_file_with_records_is_unchanged_by_the_fallback(self): + """The fallback must only fire when the stream has no source file.""" + with tempfile.TemporaryDirectory() as tmp: + stats, text = self._emit(tmp, "expanded", sample_ids=["SAMPLE_A"], + records=3) + self.assertGreater(stats["triples"], 8) + self.assertIn("#SampleSet", text) + self.assertIn("file://sample.vcf", text) + + + + +class MissingTokenConformanceScopeTests(VerboseTestCase): + """"." is not always a missing value, and the check said it was. + + The published shapes constrain two predicates whose lexical space includes + a bare dot: ``vcfc:fieldNumber`` (VCF's ``Number=.`` variable-cardinality + token, an arity declaration) and ``vcfc:genotypeString`` (a fully-missing + call is literally "." or "./."). Both are conformant and neither is typed + ``vcfc:Null``, so the check reported every one of them. On the + 100,000-record HG005 benchmark cell it returned 20 such rows. + """ + + QUERY_DIR = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "queries" / "common" + ) + + def test_both_queries_exclude_the_conformant_predicates(self): + for name in ( + "preflight_missing_token_conformance.rq", + "preflight_missing_token_conformance_count.rq", + ): + text = (self.QUERY_DIR / name).read_text(encoding="utf-8") + self.assertIn("vcfc:fieldNumber", text, name) + self.assertIn("vcfc:genotypeString", text, name) + self.assertIn("NOT IN", text, name) + + def test_the_sample_and_its_count_filter_identically(self): + """Different populations would make the count meaningless.""" + def filters(name): + text = (self.QUERY_DIR / name).read_text(encoding="utf-8") + return [ + line.strip() + for line in text.splitlines() + if line.strip().startswith("FILTER(") + ] + + self.assertEqual( + filters("preflight_missing_token_conformance.rq"), + filters("preflight_missing_token_conformance_count.rq"), + ) + + def test_the_exclusions_match_what_the_bundled_shapes_permit(self): + """If a shape stops allowing a dot, this exclusion must be revisited.""" + shapes = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "vcf_rdfizer_data" / "shacl" / "vcf-core-vocabulary.shacl.ttl" + ).read_text(encoding="utf-8") + for predicate in ("vcfc:fieldNumber", "vcfc:genotypeString"): + self.assertIn(predicate, shapes, predicate) + self.assertIn("|M|[.])$", shapes) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 45fa2af..472c84c 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -2232,6 +2232,27 @@ class ParsedSampleRecord: filter_value: str = "" +def source_file_from_header_lines(header_lines_tsv: Path) -> str: + """The SOURCE_FILE of a header-lines TSV, for a VCF with no data records. + + ``SampleRecordStream`` learns the source file from the first data row, so a + header-only VCF leaves it empty -- and the emitters, which key their file + IRI on it, then emit nothing at all. The sample columns are declared on the + ``#CHROM`` line, not by the data, so a zero-record file still has a sample + set to represent. Reading the name from the header table restores it. + """ + try: + with header_lines_tsv.open(newline="", encoding="utf-8") as handle: + reader = csv.reader(handle, delimiter="\t") + next(reader, None) # column header + for row in reader: + if row and row[0]: + return row[0] + except (OSError, csv.Error): + return "" + return "" + + class SampleRecordStream: """Read a records.tsv sample block once and expose a stable sample schema.""" @@ -2819,11 +2840,18 @@ def append_expanded_sample_rdf( ) with SampleRecordStream(records_tsv) as record_stream: - if not record_stream.source_file: + # A header-only VCF has no data row to learn the source file from, but + # its #CHROM line still declares sample columns and its header still + # declares fields. Falling back to the header table keeps those + # declarations representable instead of emitting an empty graph. + source_file = record_stream.source_file or source_file_from_header_lines( + header_lines_tsv + ) + if not source_file: return stats def produce(emit): - source_component = _rml_uri_component(record_stream.source_file) + source_component = _rml_uri_component(source_file) file_uri = f"file://{source_component}" # The profile is declared even for a sites-only VCF. # vcfc:RepresentationProfileShape requires exactly one on every @@ -4154,11 +4182,18 @@ def append_record_detail_rdf( } with SampleRecordStream(records_tsv) as record_stream: - if not record_stream.source_file: + # A header-only VCF has no data row to learn the source file from, but + # its #CHROM line still declares sample columns and its header still + # declares fields. Falling back to the header table keeps those + # declarations representable instead of emitting an empty graph. + source_file = record_stream.source_file or source_file_from_header_lines( + header_lines_tsv + ) + if not source_file: return stats def produce(emit): - source_component = _rml_uri_component(record_stream.source_file) + source_component = _rml_uri_component(source_file) file_uri = f"file://{source_component}" emitted_definitions: set[str] = set() emitted_assembly_contigs: set[str] = set() @@ -4436,11 +4471,18 @@ def append_condensed_sample_rdf( definitions = _load_format_definitions(header_lines_tsv) with SampleRecordStream(records_tsv) as record_stream: - if not record_stream.source_file: + # A header-only VCF has no data row to learn the source file from, but + # its #CHROM line still declares sample columns and its header still + # declares fields. Falling back to the header table keeps those + # declarations representable instead of emitting an empty graph. + source_file = record_stream.source_file or source_file_from_header_lines( + header_lines_tsv + ) + if not source_file: return stats def produce(emit): - source_component = _rml_uri_component(record_stream.source_file) + source_component = _rml_uri_component(source_file) file_uri = f"file://{source_component}" emitted_definitions: set[str] = set() # FORMAT keys repeat on every record; encode each distinct key once. From 79e5408772a1b2ae25f37cd0e604110591d16426 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 09:34:07 +0200 Subject: [PATCH 09/18] Report what a link count means, and what the link rests on The campaign's tier-1 run reported 86 links from 86 records, and that number cannot be read. Tier 1 rewrites an identifier the VCF already carries: it cannot fail on a record with an rsID and cannot succeed on one without, so 86/86 describes the fixture, not the linker. Tier 2 made the opposite point: it ran cleanly and produced zero links, because the bundled demo intervals do not overlap the fixture, so executing correctly and linking nothing were the same row in the output. Each linker now records eligible_records (the population that produced a join key), linked_subjects, and their ratio, so a yield has a denominator. Each also records what its assertion rests on: tier 1 an identifier rewrite that is NOT position-verified, tier 2 a coordinate match against a digest-pinned reference, tier 3 a service resolution with a recorded response digest. The linkset node carries all three into the graph, so a consumer reading vcfl:sameVariantAs, a strong predicate that a stale rsID in a re-annotated call set will produce confidently and wrongly, can see its basis without leaving the data. This is the measurable half of recommendation 8. The other half is an experiment, not code: a real cohort against a released dbSNP, with recall against an independent bcftools isec overlap. Co-Authored-By: Claude Opus 5 --- test/test_linking_yield_unit.py | 109 ++++++++++++++++++++++++++++++++ vcf_rdfizer_linking/runner.py | 50 +++++++++++++++ 2 files changed, 159 insertions(+) create mode 100644 test/test_linking_yield_unit.py diff --git a/test/test_linking_yield_unit.py b/test/test_linking_yield_unit.py new file mode 100644 index 0000000..3d7e3f2 --- /dev/null +++ b/test/test_linking_yield_unit.py @@ -0,0 +1,109 @@ +"""What a link count means, and what the link actually rests on. + +The campaign's tier-1 run reported "86 links from 86 records" and that number +is uninterpretable on its own. Tier 1 rewrites an identifier the VCF already +carries: it cannot fail on a record that has an rsID and cannot succeed on one +that does not, so 86/86 is a property of the fixture, not evidence about the +linker. A reader needs the denominator -- records that produced a join key -- +and needs to know the assertion is an identifier rewrite rather than a +position-verified match, because ``vcfl:sameVariantAs`` is a strong predicate +and a stale rsID in a re-annotated call set produces a confidently wrong one. + +Tier 2 made the opposite point: it ran cleanly and produced zero links, because +the bundled demonstration intervals do not overlap the fixture. "Executes +correctly" and "links nothing" were the same row in the output. +""" + +import unittest + +from vcf_rdfizer_linking import runner +from test.helpers import VerboseTestCase + + +class AssertionBasisTests(VerboseTestCase): + def test_every_tier_declares_what_its_link_rests_on(self): + for tier in (1, 2, 3): + self.assertIn(tier, runner.ASSERTION_BASIS) + self.assertTrue(runner.ASSERTION_BASIS[tier].strip()) + + def test_tier_one_is_labelled_an_identifier_rewrite(self): + """The weakest claim in the set must be the one that says so.""" + basis = runner.ASSERTION_BASIS[1] + self.assertIn("identifier-rewrite", basis) + self.assertIn("not position-verified", basis) + + def test_tier_two_is_labelled_as_verified_against_a_pinned_reference(self): + basis = runner.ASSERTION_BASIS[2] + self.assertIn("coordinate-interval", basis) + self.assertIn("digest-pinned", basis) + + def test_tier_three_records_that_a_service_answered(self): + self.assertIn("service-resolution", runner.ASSERTION_BASIS[3]) + + def test_only_the_coordinate_tier_counts_as_verified(self): + """Tier 3's answer is recorded, not checked against the call itself.""" + self.assertEqual( + [tier for tier in (1, 2, 3) if tier == 2], [2], + ) + + +class CoverageArithmeticTests(VerboseTestCase): + """The ratio the run record now carries, checked on its own terms.""" + + @staticmethod + def coverage(linked_subjects, eligible_records): + if not eligible_records: + return None + return round(linked_subjects / eligible_records, 6) + + def test_full_coverage_of_an_eligible_population_is_one(self): + self.assertEqual(self.coverage(86, 86), 1.0) + + def test_partial_coverage_is_the_fraction_linked(self): + self.assertEqual(self.coverage(43, 86), 0.5) + + def test_zero_links_against_an_eligible_population_is_zero_not_none(self): + """Tier 2's real result: it ran, and linked nothing. That is a number.""" + self.assertEqual(self.coverage(0, 86), 0.0) + + def test_no_eligible_records_reports_none_rather_than_a_false_zero(self): + """Nothing could have linked, so no rate is defined.""" + self.assertIsNone(self.coverage(0, 0)) + + +class RunnerRecordShapeTests(VerboseTestCase): + """The fields a reported yield needs, present before any run.""" + + REQUIRED = ( + "eligible_records", + "linked_subjects", + "coverage", + "assertion_basis", + "assertion_verified", + ) + + def test_the_new_fields_are_initialised_for_every_linker(self): + import inspect + + source = inspect.getsource(runner.run_linkers) + for field in self.REQUIRED: + self.assertIn(f'"{field}"', source, field) + + def test_eligible_records_is_incremented_where_keys_are_extracted(self): + import inspect + + source = inspect.getsource(runner.run_linkers) + self.assertIn('["eligible_records"] += 1', source) + + def test_the_linkset_node_carries_the_denominator_and_the_basis(self): + """A consumer must see this without leaving the graph.""" + import inspect + + source = inspect.getsource(runner.run_linkers) + self.assertIn("VCFL.eligibleRecordCount", source) + self.assertIn("VCFL.linkedSubjectCount", source) + self.assertIn("VCFL.assertionBasis", source) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer_linking/runner.py b/vcf_rdfizer_linking/runner.py index 21dc9b0..65ac946 100644 --- a/vcf_rdfizer_linking/runner.py +++ b/vcf_rdfizer_linking/runner.py @@ -32,6 +32,24 @@ def __init__(self, message, report): self.report = report +#: What each tier's link actually rests on, recorded alongside every linkset. +#: +#: tier 1 rewrites an identifier the VCF already carried. It cannot fail on a +#: record that has one and cannot succeed on a record that does not, so +#: its hit rate is a property of the input, not of the linker -- and +#: nothing checks that the identifier is still correct for this position +#: and these alleles. A stale rsID in a re-annotated call set produces a +#: confidently wrong vcfl:sameVariantAs. +#: tier 2 matches coordinates against a digest-pinned reference, so the link is +#: verified against something. +#: tier 3 resolves against a live service, whose response digest is recorded. +ASSERTION_BASIS = { + 1: "identifier-rewrite: taken from the source ID column, not position-verified", + 2: "coordinate-interval: matched against a digest-pinned reference", + 3: "service-resolution: resolved against a live service, response digest recorded", +} + + def keys_for(record, manifest): if manifest.strategy == "token": value = record.id @@ -109,6 +127,20 @@ def run_linkers(records, manifests, output: Path | None, *, cache_dir=DEFAULT_CA "assembly": manifest.reference.assembly if manifest.reference else None, "unique_keys": 0, "skipped_records": 0, "links": 0, "requests": 0, "cache_hits": 0, "bytes_transferred": 0, "final_service_status": None, + # "86 links" says nothing without its denominator. Records that + # produced at least one join key are the population a linker + # could possibly link; subjects actually linked are the + # numerator. Their ratio is the coverage a reader needs before + # "86/86" can mean anything. + "eligible_records": 0, "linked_subjects": 0, "coverage": None, + # What the link asserts on. A tier-1 rsID link is a rewrite of + # an identifier the VCF already carried: it cannot fail to + # link a record that has an rsID and cannot link one that does + # not, and nothing checks that the identifier is still correct + # for this position and these alleles. Recording the basis + # keeps a strong predicate from reading as a verified claim. + "assertion_basis": ASSERTION_BASIS.get(manifest.tier, "unspecified"), + "assertion_verified": manifest.tier == 2, "status": "pending", "wall_seconds": 0} if manifest.tier == 3: stats["resolver_sha256"] = hashlib.sha256((manifest.directory / "resolver.py").read_bytes()).hexdigest() @@ -140,6 +172,8 @@ def run_linkers(records, manifests, output: Path | None, *, cache_dir=DEFAULT_CA keys = keys_for(record, manifest) if not keys: stats_by_id[manifest.id]["skipped_records"] += 1 + else: + stats_by_id[manifest.id]["eligible_records"] += 1 subject = record.call if manifest.subject == str(VCFL.VariantCall) else record.record absolute_iri(subject) db.executemany("INSERT OR IGNORE INTO keys VALUES (?,?,?,?)", [ @@ -185,6 +219,12 @@ def run_linkers(records, manifests, output: Path | None, *, cache_dir=DEFAULT_CA db.execute("INSERT OR IGNORE INTO links SELECT linker,source,subject,?,? FROM keys WHERE linker=? AND key=?", (manifest.predicate, obj, manifest.id, encoded[link.key])) current["links"] = db.execute("SELECT COUNT(*) FROM links WHERE linker=?", (manifest.id,)).fetchone()[0] + current["linked_subjects"] = db.execute( + "SELECT COUNT(DISTINCT subject) FROM links WHERE linker=?", + (manifest.id,)).fetchone()[0] + if current["eligible_records"]: + current["coverage"] = round( + current["linked_subjects"] / current["eligible_records"], 6) current["status"] = "success" except BaseException: current["status"] = "failed" @@ -213,6 +253,16 @@ def produce(emit): emit(triple(node, RDF.type, URIRef(VCFL.Linkset).n3())) emit(triple(node, VCFL.producedBy, URIRef(f"https://w3id.org/vcf-rdfizer/linker/{manifest.id}/{manifest.version}").n3())) emit(triple(node, VCFL.linkCount, Literal(count, datatype=XSD.integer).n3())) + linker_stats = stats_by_id[manifest.id] + emit(triple(node, VCFL.eligibleRecordCount, Literal( + linker_stats["eligible_records"], datatype=XSD.integer).n3())) + emit(triple(node, VCFL.linkedSubjectCount, Literal( + linker_stats["linked_subjects"], datatype=XSD.integer).n3())) + # A consumer reading vcfl:sameVariantAs should be + # able to see what it rests on without leaving + # the graph. + emit(triple(node, VCFL.assertionBasis, + literal(linker_stats["assertion_basis"]))) emit(triple(node, VCFL.source, URIRef(source).n3())) emit(triple(node, VCFL.manifestDigest, literal("sha256:" + stats_by_id[manifest.id]["manifest_sha256"]))) if manifest.tier == 3: From bd79a83d46a455b0dfe0bfc0e2ec3df2568ba010 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 11:40:37 +0200 Subject: [PATCH 10/18] Fail SHACL only on sh:Violation, not on recommendations Caught by the first real pipeline run on bench-1, which the local Docker-less checks could not reach: test-1k failed validation with violationCount 0 and violationPaths [vcfc:fieldSource, vcfc:fieldVersion]. Those come from InfoHeaderLineRecommendedShape, which carries sh:severity sh:Warning because VCF 4.5 *recommends* Source and Version on INFO declarations rather than requiring them. pyshacl sets conforms=False for a result at any severity, so keying the verdict on conforms alone failed a conformant graph -- and, now that shapes are on by default, would have failed almost every real VCF. The verdict is now the violation count. pyshacl's own conforms flag is kept verbatim in the report, because a graph with warnings only is conforms=False and status=PASS, which is exactly the distinction the severities encode. Advisory results are counted and named alongside. Co-Authored-By: Claude Opus 5 --- src/validation/validation_runner.py | 22 ++++++++- test/test_shacl_default_unit.py | 74 +++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 1 deletion(-) diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index 40bd61e..9d74ab7 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -1878,6 +1878,21 @@ def validate_shacl( line.strip() for line in text.splitlines() if line.strip().startswith("Constraint Violation") ] + # pyshacl reports conforms=False for ANY result, including sh:Warning and + # sh:Info. The published profile uses warnings for genuine recommendations + # -- InfoHeaderLineRecommendedShape warns when an INFO declaration omits + # Source or Version, which VCF 4.5 recommends and does not require -- so + # conforms alone would fail almost every real VCF. Only sh:Violation blocks + # a run; advisory results are reported and carried, not enforced. + advisories = sorted({ + line.strip().split(":", 1)[0].strip() + for line in text.splitlines() + if line.strip().startswith(("Constraint Warning", "Constraint Info")) + }) + advisory_count = sum( + 1 for line in text.splitlines() + if line.strip().startswith(("Constraint Warning", "Constraint Info")) + ) paths = sorted({ line.split("Result Path:", 1)[1].strip() for line in text.splitlines() if "Result Path:" in line @@ -1885,12 +1900,17 @@ def validate_shacl( log_path = results_dir / "shacl-report.txt" log_path.write_text(text, encoding="utf-8") result = { - "status": "PASS" if conforms else "FAIL", + "status": "PASS" if not violations else "FAIL", + # Kept verbatim: it is pyshacl's own verdict, and it is NOT the verdict + # this run acts on. A graph with warnings only is conforms=False here + # and status=PASS, which is the distinction the severities encode. "conforms": bool(conforms), "shapes": str(shapes), "ontology": str(ontology) if ontology is not None else None, "violationCount": len(violations), "violationPaths": paths, + "advisoryCount": advisory_count, + "advisoryKinds": advisories, "report": str(log_path), "wallSeconds": time.monotonic() - started, "sampleLimitedTo": SHACL_SAMPLE_LIMIT, diff --git a/test/test_shacl_default_unit.py b/test/test_shacl_default_unit.py index a11204d..455244e 100644 --- a/test/test_shacl_default_unit.py +++ b/test/test_shacl_default_unit.py @@ -152,5 +152,79 @@ def test_the_catalogue_still_attributes_these_classes_to_the_shape_layer(self): self.assertIn(mutation_id, source, mutation_id) + + +class SeverityTests(VerboseTestCase): + """Only sh:Violation blocks a run; warnings are recommendations. + + Caught on the VM, not here: the first real pipeline run with the new + default failed on test-1k with violationCount 0 and violationPaths + [vcfc:fieldSource, vcfc:fieldVersion]. Those come from + InfoHeaderLineRecommendedShape, which carries sh:severity sh:Warning + because VCF 4.5 *recommends* Source and Version on INFO declarations and + does not require them. pyshacl reports conforms=False for any result at any + severity, so keying the verdict on conforms alone failed a conformant graph + -- and would have failed almost every real VCF. + """ + + def _verdict(self, report_text): + import importlib.util + + runner_path = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "validation_runner.py" + ) + spec = importlib.util.spec_from_file_location("vr_shacl", runner_path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + violations = [ + line.strip() for line in report_text.splitlines() + if line.strip().startswith("Constraint Violation") + ] + return "PASS" if not violations else "FAIL" + + def test_warnings_alone_do_not_fail_a_run(self): + report = ( + "Validation Report\nConforms: False\nResults (2):\n" + "Constraint Warning in MinCountConstraintComponent\n" + "\tResult Path: vcfc:fieldSource\n" + "Constraint Warning in MinCountConstraintComponent\n" + "\tResult Path: vcfc:fieldVersion\n" + ) + self.assertEqual(self._verdict(report), "PASS") + + def test_a_real_violation_still_fails(self): + report = ( + "Validation Report\nConforms: False\nResults (1):\n" + "Constraint Violation in DatatypeConstraintComponent\n" + "\tResult Path: vcfc:pos\n" + ) + self.assertEqual(self._verdict(report), "FAIL") + + def test_a_clean_report_passes(self): + self.assertEqual(self._verdict("Validation Report\nConforms: True\n"), "PASS") + + def test_the_recommended_shape_really_is_a_warning(self): + """Pin the assumption against the vendored shapes themselves.""" + shapes = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "vcf_rdfizer_data" / "shacl" / "vcf-core-vocabulary.shacl.ttl" + ).read_text(encoding="utf-8") + block = shapes[shapes.index("InfoHeaderLineRecommendedShape"):] + block = block[:block.index(" .\n")] + self.assertIn("vcfc:fieldSource", block) + self.assertIn("vcfc:fieldVersion", block) + self.assertEqual(block.count("sh:severity sh:Warning"), 2) + + def test_the_report_records_advisories_separately_from_violations(self): + source = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "validation_runner.py" + ).read_text(encoding="utf-8") + self.assertIn('"advisoryCount"', source) + self.assertIn('"advisoryKinds"', source) + self.assertIn('"status": "PASS" if not violations else "FAIL"', source) + + if __name__ == "__main__": unittest.main() From c00da21bdf18f9fa61350bbf2d7cbc758b89e40f Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 11:45:59 +0200 Subject: [PATCH 11/18] Parse pyshacl results by severity; the old matcher never fired Found by reading the real report from the bench-1 run. pyshacl 0.30.1 writes each result as a 'Validation Result in ' block with an indented 'Severity: sh:Violation|sh:Warning|sh:Info' line. This module looked for lines starting 'Constraint Violation', which that format never produces, so violationCount was always 0 and the verdict fell through to pyshacl's conforms flag. The layer was therefore broken in both directions at once: it could not report a real violation, and conforms=False failed any graph carrying a mere recommendation. The archived campaign never noticed because SHACL was never enabled in it. parse_shacl_results splits the report into blocks and classifies each by its Severity line. Violations block the run; warnings and info are counted and sampled separately. A block with no severity line is Unknown and treated as blocking, because the safe reading of a parser bug is not 'conformant'. The pre-0.30 'Constraint Violation' spelling still parses, so a downgrade does not silently stop reporting. test/fixtures/shacl-report-warnings-only.txt is the verbatim report from that run: six sh:Warning results for vcfc:fieldSource and vcfc:fieldVersion, zero violations. Co-Authored-By: Claude Opus 5 --- src/validation/validation_runner.py | 94 ++++++++++++++++---- test/fixtures/shacl-report-warnings-only.txt | 39 ++++++++ test/test_shacl_default_unit.py | 82 +++++++++++++++++ 3 files changed, 197 insertions(+), 18 deletions(-) create mode 100644 test/fixtures/shacl-report-warnings-only.txt diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index 9d74ab7..b953d7c 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -1817,6 +1817,69 @@ def materialize_ntriples( SHACL_SAMPLE_LIMIT = 50 +#: pyshacl writes each result as a "Validation Result in " block with +#: an indented "Severity: sh:Violation|sh:Warning|sh:Info" line. It does NOT +#: write "Constraint Violation", which is what this module looked for -- so the +#: violation count was always zero and the verdict fell through to pyshacl's +#: `conforms`, which is False for a warning as readily as for a violation. +#: +#: That made the shape layer unusable in both directions at once: it could never +#: report a real violation, and it failed any graph carrying a recommendation. +#: The published profile warns whenever an INFO declaration omits Source or +#: Version, which VCF 4.5 recommends rather than requires, so that is most real +#: VCFs. +_SHACL_RESULT_START = "Validation Result in " +_SHACL_SEVERITY_PREFIX = "Severity:" +#: Older pyshacl releases used this spelling; kept so a downgrade still parses. +_SHACL_LEGACY_PREFIXES = { + "Constraint Violation": "Violation", + "Constraint Warning": "Warning", + "Constraint Info": "Info", +} + + +def parse_shacl_results(text: str) -> list[dict[str, str]]: + """Split a pyshacl text report into results tagged by severity. + + Returns one entry per result, each ``{"severity": ..., "text": ...}`` with + severity one of ``Violation``, ``Warning``, ``Info`` (or ``Unknown`` when a + block carries no severity line, which is treated as a violation by the + caller -- an unparseable result must not silently pass). + """ + results: list[dict[str, str]] = [] + block: list[str] | None = None + + def flush(lines: list[str] | None) -> None: + if not lines: + return + severity = "Unknown" + for line in lines: + stripped = line.strip() + if stripped.startswith(_SHACL_SEVERITY_PREFIX): + token = stripped.split(":", 1)[1].strip() + severity = token.rsplit(":", 1)[-1] or "Unknown" + break + results.append({"severity": severity, "text": "\n".join(lines).strip()}) + + for line in text.splitlines(): + stripped = line.strip() + legacy = next( + (value for key, value in _SHACL_LEGACY_PREFIXES.items() + if stripped.startswith(key)), + None, + ) + if stripped.startswith(_SHACL_RESULT_START) or legacy is not None: + flush(block) + block = [line] + if legacy is not None: + block.append(f" Severity: sh:{legacy}") + continue + if block is not None: + block.append(line) + flush(block) + return results + + def validate_shacl( source: Path, shapes: Path, @@ -1874,25 +1937,19 @@ def validate_shacl( write_json(report_path, result) return result + results = parse_shacl_results(text) + # Unknown counts as blocking: a result this parser could not classify is a + # parser bug, and the safe reading of a parser bug is not "conformant". violations = [ - line.strip() for line in text.splitlines() - if line.strip().startswith("Constraint Violation") + entry["text"] for entry in results + if entry["severity"] in ("Violation", "Unknown") ] - # pyshacl reports conforms=False for ANY result, including sh:Warning and - # sh:Info. The published profile uses warnings for genuine recommendations - # -- InfoHeaderLineRecommendedShape warns when an INFO declaration omits - # Source or Version, which VCF 4.5 recommends and does not require -- so - # conforms alone would fail almost every real VCF. Only sh:Violation blocks - # a run; advisory results are reported and carried, not enforced. - advisories = sorted({ - line.strip().split(":", 1)[0].strip() - for line in text.splitlines() - if line.strip().startswith(("Constraint Warning", "Constraint Info")) - }) - advisory_count = sum( - 1 for line in text.splitlines() - if line.strip().startswith(("Constraint Warning", "Constraint Info")) - ) + advisories = [ + entry for entry in results + if entry["severity"] not in ("Violation", "Unknown") + ] + advisory_kinds = sorted({entry["severity"] for entry in advisories}) + advisory_count = len(advisories) paths = sorted({ line.split("Result Path:", 1)[1].strip() for line in text.splitlines() if "Result Path:" in line @@ -1910,7 +1967,8 @@ def validate_shacl( "violationCount": len(violations), "violationPaths": paths, "advisoryCount": advisory_count, - "advisoryKinds": advisories, + "advisoryKinds": advisory_kinds, + "advisorySample": [entry["text"] for entry in advisories][:SHACL_SAMPLE_LIMIT], "report": str(log_path), "wallSeconds": time.monotonic() - started, "sampleLimitedTo": SHACL_SAMPLE_LIMIT, diff --git a/test/fixtures/shacl-report-warnings-only.txt b/test/fixtures/shacl-report-warnings-only.txt new file mode 100644 index 0000000..cfab881 --- /dev/null +++ b/test/fixtures/shacl-report-warnings-only.txt @@ -0,0 +1,39 @@ +Validation Report +Conforms: False +Results (6): +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Source on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldSource ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldSource + Message: VCF 4.5 recommends Source on INFO declarations. +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Source on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldSource ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldSource + Message: VCF 4.5 recommends Source on INFO declarations. +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Source on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldSource ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldSource + Message: VCF 4.5 recommends Source on INFO declarations. +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Version on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldVersion ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldVersion + Message: VCF 4.5 recommends Version on INFO declarations. +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Version on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldVersion ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldVersion + Message: VCF 4.5 recommends Version on INFO declarations. +Validation Result in MinCountConstraintComponent (http://www.w3.org/ns/shacl#MinCountConstraintComponent): + Severity: sh:Warning + Source Shape: [ sh:datatype xsd:string ; sh:message Literal("VCF 4.5 recommends Version on INFO declarations.", lang=en) ; sh:minCount Literal("1", datatype=xsd:integer) ; sh:path vcfc:fieldVersion ; sh:severity sh:Warning ] + Focus Node: + Result Path: vcfc:fieldVersion + Message: VCF 4.5 recommends Version on INFO declarations. diff --git a/test/test_shacl_default_unit.py b/test/test_shacl_default_unit.py index 455244e..fda3869 100644 --- a/test/test_shacl_default_unit.py +++ b/test/test_shacl_default_unit.py @@ -226,5 +226,87 @@ def test_the_report_records_advisories_separately_from_violations(self): self.assertIn('"status": "PASS" if not violations else "FAIL"', source) + + +class ReportParsingTests(VerboseTestCase): + """pyshacl's real output format, captured from a run on bench-1. + + The module looked for lines starting "Constraint Violation". pyshacl 0.30.1 + writes "Validation Result in " with an indented "Severity:" + line, so that matcher never fired: violationCount was always 0 and the + verdict fell through to pyshacl's `conforms`, which is False for a warning + as readily as for a violation. The layer was therefore broken in both + directions -- unable to report a real violation, and failing any graph that + merely carried a recommendation. + + test/fixtures/shacl-report-warnings-only.txt is the verbatim report from + that run: six sh:Warning results for vcfc:fieldSource and + vcfc:fieldVersion, zero violations. + """ + + FIXTURE = Path(__file__).resolve().parent / "fixtures" / "shacl-report-warnings-only.txt" + + def _module(self): + import importlib.util + + path = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "validation_runner.py" + ) + spec = importlib.util.spec_from_file_location("vr_parse", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + def test_the_real_report_parses_to_six_warnings_and_no_violations(self): + results = self._module().parse_shacl_results(self.FIXTURE.read_text(encoding="utf-8")) + self.assertEqual(len(results), 6) + self.assertEqual({r["severity"] for r in results}, {"Warning"}) + + def test_a_violation_block_is_classified_as_blocking(self): + text = ( + "Conforms: False\nResults (1):\n" + "Validation Result in DatatypeConstraintComponent (...):\n" + "\tSeverity: sh:Violation\n" + "\tResult Path: vcfc:pos\n" + ) + results = self._module().parse_shacl_results(text) + self.assertEqual([r["severity"] for r in results], ["Violation"]) + + def test_mixed_severities_are_separated(self): + text = ( + "Conforms: False\nResults (2):\n" + "Validation Result in MinCountConstraintComponent (...):\n" + "\tSeverity: sh:Warning\n\tResult Path: vcfc:fieldSource\n" + "Validation Result in DatatypeConstraintComponent (...):\n" + "\tSeverity: sh:Violation\n\tResult Path: vcfc:pos\n" + ) + results = self._module().parse_shacl_results(text) + self.assertEqual([r["severity"] for r in results], ["Warning", "Violation"]) + + def test_a_clean_report_parses_to_nothing(self): + self.assertEqual(self._module().parse_shacl_results("Conforms: True\n"), []) + + def test_a_block_with_no_severity_line_is_unknown_and_blocks(self): + """An unclassifiable result is a parser bug; it must not read as clean.""" + text = "Validation Result in SomeComponent (...):\n\tResult Path: vcfc:pos\n" + results = self._module().parse_shacl_results(text) + self.assertEqual([r["severity"] for r in results], ["Unknown"]) + source = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "validation_runner.py" + ).read_text(encoding="utf-8") + self.assertIn('in ("Violation", "Unknown")', source) + + def test_the_older_pyshacl_spelling_still_parses(self): + """A downgrade must not silently stop reporting violations again.""" + text = ( + "Constraint Violation in DatatypeConstraintComponent (...):\n" + "\tResult Path: vcfc:pos\n" + ) + results = self._module().parse_shacl_results(text) + self.assertEqual([r["severity"] for r in results], ["Violation"]) + + if __name__ == "__main__": unittest.main() From 223fd222ad55108747c16d3d6917ff37d000da8a Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 12:03:35 +0200 Subject: [PATCH 12/18] Expect no per-call sampleIds when a VCF has no records Found on bench-1 after the sample-set fix landed: q09 and q10 passed, but preflight_sample_gt_inventory still failed on awkward_no_records, expecting sampleIdCount 1 and finding 0. vcfc:sampleId lives on SampleCall -- one per sample per record -- so a header-only VCF has no call to carry it and the count is structurally zero. Expecting sampleCount there fails a correct graph. The file-scoped sample set is a different thing and the census queries check it; with the emitter fix those now pass. Condensed is unaffected: its sample set and sampleId literals are file-scoped, so only the per-record vectors go to zero. Co-Authored-By: Claude Opus 5 --- src/validation/validation_runner.py | 13 ++++++- test/test_zero_record_vcf_unit.py | 58 +++++++++++++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index b953d7c..ff7ae9b 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -3235,14 +3235,25 @@ def preflight( report[query_id] = {"status": "FAIL", "error": f"Expected one aggregate row, got {len(returned)}"} elif representation == "expanded": actual = {field: binding_int(returned[0], field) for field in ("sampleCallCount", "sampleIdCount", "gtValueNodeCount")} + # sampleIdCount counts vcfc:sampleId, which lives on SampleCall -- + # one per sample PER RECORD. A header-only VCF declares samples and + # has no records, so there are no calls to carry the literal and the + # count is structurally zero. Expecting sampleCount there fails a + # correct graph. The file-scoped sample set is a separate thing and + # is checked by the census queries. expected = { "sampleCallCount": parser["sampleCount"] * parser["totalRecords"], - "sampleIdCount": parser["sampleCount"], + "sampleIdCount": ( + parser["sampleCount"] if parser["totalRecords"] else 0 + ), "gtValueNodeCount": parser["sampleCount"] * parser["gtRecordCount"], } report[query_id] = {"status": "PASS" if actual == expected else "FAIL", "expected": expected, "actual": actual} else: actual = {field: binding_int(returned[0], field) for field in ("sampleCount", "sampleIdCount", "gtVectorCount")} + # The condensed profile's sample set and its sampleId literals are + # file-scoped, so both survive a zero-record file; only the + # per-record vectors go to zero, which gtRecordCount already says. expected = {"sampleCount": parser["sampleCount"], "sampleIdCount": parser["sampleCount"], "gtVectorCount": parser["gtRecordCount"]} report[query_id] = {"status": "PASS" if actual == expected else "FAIL", "expected": expected, "actual": actual} return report diff --git a/test/test_zero_record_vcf_unit.py b/test/test_zero_record_vcf_unit.py index f39457a..9e3e20f 100644 --- a/test/test_zero_record_vcf_unit.py +++ b/test/test_zero_record_vcf_unit.py @@ -210,5 +210,63 @@ def test_the_exclusions_match_what_the_bundled_shapes_permit(self): self.assertIn("|M|[.])$", shapes) + + +class SampleInventoryExpectationTests(VerboseTestCase): + """What the GT inventory should expect when there are no records. + + Found on bench-1: with the sample set restored, q09/q10 passed but + preflight_sample_gt_inventory still failed, expecting sampleIdCount 1 and + finding 0. vcfc:sampleId lives on SampleCall -- one per sample *per record* + -- so a header-only VCF has no call to carry it and the count is + structurally zero. The file-scoped sample set is a different thing, and the + census queries check that. + """ + + def _expected(self, representation, sample_count, total_records, gt_records): + import importlib.util + + path = ( + Path(vcf_rdfizer.__file__).resolve().parent + / "src" / "validation" / "validation_runner.py" + ) + spec = importlib.util.spec_from_file_location("vr_inv", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + source = path.read_text(encoding="utf-8") + self.assertIn('parser["sampleCount"] if parser["totalRecords"] else 0', source) + if representation == "expanded": + return { + "sampleCallCount": sample_count * total_records, + "sampleIdCount": sample_count if total_records else 0, + "gtValueNodeCount": sample_count * gt_records, + } + return { + "sampleCount": sample_count, + "sampleIdCount": sample_count, + "gtVectorCount": gt_records, + } + + def test_expanded_expects_no_sample_ids_when_there_are_no_records(self): + self.assertEqual( + self._expected("expanded", 1, 0, 0), + {"sampleCallCount": 0, "sampleIdCount": 0, "gtValueNodeCount": 0}, + ) + + def test_expanded_still_expects_one_sample_id_per_sample_with_records(self): + """The ordinary case must not move.""" + self.assertEqual( + self._expected("expanded", 3, 1000, 1000), + {"sampleCallCount": 3000, "sampleIdCount": 3, "gtValueNodeCount": 3000}, + ) + + def test_condensed_keeps_its_file_scoped_sample_ids(self): + """Condensed sampleIds are file-scoped, so a zero-record file keeps them.""" + self.assertEqual( + self._expected("condensed", 2, 0, 0), + {"sampleCount": 2, "sampleIdCount": 2, "gtVectorCount": 0}, + ) + + if __name__ == "__main__": unittest.main() From eca2b09d87d54bd4580cc03bcfe1441de99c0a6c Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 12:18:44 +0200 Subject: [PATCH 13/18] Measure the scratch tree, not the device it sits on The first VM run reported peak_volume_workspace_bytes of 126,956,531,712 -- 127 GB, the host disk's used bytes -- for a build whose scratch was a single 11.9 MB chunk. shutil.disk_usage on /work describes the backing device, and free-space deltas are no better because anything else on the host moves them. StageRunner now samples the size of the /work tree around each stage, and the peak is the largest tree seen. Device-level figures are kept, since a build that fills the volume needs them to say so, but they are no longer mistaken for this build's footprint. A missing workspace reports None rather than a confident 0, and symlinks are not followed so a link into a mounted output tree cannot inflate the figure. Co-Authored-By: Claude Opus 5 --- src/partitioned_compression.py | 52 +++++++++++++++++++++++++----- test/test_build_profile_unit.py | 56 +++++++++++++++++++++++++++++---- 2 files changed, 95 insertions(+), 13 deletions(-) diff --git a/src/partitioned_compression.py b/src/partitioned_compression.py index 010ae51..175d8b0 100644 --- a/src/partitioned_compression.py +++ b/src/partitioned_compression.py @@ -455,6 +455,30 @@ def number(pattern: str, integer: bool = False): } +def directory_tree_bytes(root: Path) -> int | None: + """Total size of the files under ``root``, or None if it cannot be read. + + Symlinks are not followed and their targets are not counted, so a link into + a mounted output tree cannot inflate the scratch figure. + """ + # rglob on a missing directory yields nothing and raises nothing, so an + # absent workspace would otherwise report a confident 0. + if not root.is_dir(): + return None + total = 0 + try: + for path in root.rglob("*"): + try: + if path.is_symlink() or not path.is_file(): + continue + total += path.stat().st_size + except OSError: + continue + except OSError: + return None + return total + + def children_peak_rss_kb() -> int | None: """Peak RSS of every child process reaped so far, in KB. @@ -489,17 +513,23 @@ def __init__(self, work_dir: Path, progress_path: Path | None = None): self.stages: list[dict] = [] def peak_workspace_bytes(self) -> int | None: - """Highest container-volume usage seen across every stage so far. + """Highest workspace-tree size seen across every stage so far. + + Measures the tree, not the filesystem. ``shutil.disk_usage`` on /work + reports the whole backing device: on bench-1 that read 126,956,531,712 + bytes -- 127 GB of host disk in use -- for a build whose scratch was a + single 11.9 MB chunk. Free-space deltas are equally unusable, because + anything else on the host moves them. - The host-side workspace trace cannot see this: the chunk scratch lives - in a Docker volume, and the two numbers must never be added. Recording - it here is what makes the volume half of peak disk measurable at all. + The host-side out-tree trace cannot see this directory, and the two + numbers must never be added: one is the published artifact tree, this + is the container's scratch. """ peaks = [ - stage["workspace_total_bytes"] - stage[key] + stage[key] for stage in self.stages - for key in ("workspace_free_bytes_before", "workspace_free_bytes_after") - if stage.get("workspace_total_bytes") is not None and stage.get(key) is not None + for key in ("workspace_tree_bytes_before", "workspace_tree_bytes_after") + if stage.get(key) is not None ] return max(peaks) if peaks else None @@ -528,6 +558,7 @@ def run( started = time.perf_counter() workspace_free_before = None workspace_total = None + workspace_tree_before = directory_tree_bytes(self.work_dir) try: workspace_usage = shutil.disk_usage(self.work_dir) workspace_free_before = workspace_usage.free @@ -598,6 +629,13 @@ def run( if output_path is not None and output_path.is_file() else 0, } + workspace_tree_after = directory_tree_bytes(self.work_dir) + if workspace_tree_before is not None: + result["workspace_tree_bytes_before"] = workspace_tree_before + if workspace_tree_after is not None: + result["workspace_tree_bytes_after"] = workspace_tree_after + # Device-level figures are kept because a build that fills the volume + # needs them to say so -- but they describe the HOST, not this build. try: workspace_usage = shutil.disk_usage(self.work_dir) result["workspace_free_bytes_after"] = workspace_usage.free diff --git a/test/test_build_profile_unit.py b/test/test_build_profile_unit.py index 683d481..8401090 100644 --- a/test/test_build_profile_unit.py +++ b/test/test_build_profile_unit.py @@ -14,6 +14,7 @@ """ import importlib.util +import pathlib import tempfile import unittest from pathlib import Path @@ -46,8 +47,13 @@ def peak_workspace_bytes(self): return P.StageRunner.peak_workspace_bytes(self) -def stage(name, wall, *, rss=None, free_before=None, free_after=None, total=None): +def stage(name, wall, *, rss=None, free_before=None, free_after=None, total=None, + tree_before=None, tree_after=None): record = {"name": name, "wall_seconds": wall} + if tree_before is not None: + record["workspace_tree_bytes_before"] = tree_before + if tree_after is not None: + record["workspace_tree_bytes_after"] = tree_after if rss is not None: record["max_rss_kb"] = rss if free_before is not None: @@ -188,18 +194,56 @@ def test_a_missing_plan_does_not_break_the_failure_path(self): class VolumeWorkspaceTests(VerboseTestCase): - def test_peak_volume_usage_is_derived_from_the_free_space_samples(self): - """The host trace cannot see the Docker volume; this is the only source.""" + """Measure the scratch tree, not the device it happens to sit on. + + The first VM run reported peak_volume_workspace_bytes of + 126,956,531,712 -- 127 GB, the host disk's used bytes -- for a build whose + scratch was a single 11.9 MB chunk. shutil.disk_usage on /work describes + the backing device; free-space deltas are no better, because anything else + on the host moves them. + """ + + def test_peak_is_the_largest_scratch_tree_seen(self): + runner = FakeRunner([ + stage("hdt-build-00000", 1.0, tree_before=10, tree_after=40), + stage("hdt-build-00001", 1.0, tree_before=40, tree_after=25), + ]) + self.assertEqual(P.build_profile(runner, {})["peak_volume_workspace_bytes"], 40) + + def test_device_level_numbers_are_not_used_as_the_peak(self): + """A 100-byte device with 25 free must not read as 75 bytes of scratch.""" runner = FakeRunner([ - stage("hdt-build-00000", 1.0, free_before=90, free_after=70, total=100), - stage("hdt-build-00001", 1.0, free_before=70, free_after=25, total=100), + stage("hdt-build-00000", 1.0, free_before=90, free_after=25, total=100), ]) - self.assertEqual(P.build_profile(runner, {})["peak_volume_workspace_bytes"], 75) + self.assertIsNone(P.build_profile(runner, {})["peak_volume_workspace_bytes"]) def test_stages_without_workspace_samples_report_none(self): runner = FakeRunner([stage("hdt-build-00000", 1.0)]) self.assertIsNone(P.build_profile(runner, {})["peak_volume_workspace_bytes"]) + def test_the_tree_walk_sums_file_sizes_and_ignores_symlinks(self): + import os + + with tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp) + (root / "a.nt").write_bytes(b"x" * 100) + (root / "sub").mkdir() + (root / "sub" / "b.nt").write_bytes(b"y" * 50) + outside = root.parent / "outside.bin" + try: + outside.write_bytes(b"z" * 10_000) + os.symlink(outside, root / "link.nt") + except OSError: + outside = None + self.assertEqual(P.directory_tree_bytes(root), 150) + if outside is not None: + outside.unlink(missing_ok=True) + + def test_an_unreadable_root_reports_none(self): + self.assertIsNone( + P.directory_tree_bytes(pathlib.Path("/nonexistent-workspace-xyz")) + ) + class ChildRssFallbackTests(VerboseTestCase): def test_the_fallback_returns_a_positive_reading_or_none(self): From 0988187cb1e36221c80e6ecd851edeae7438cdd5 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sat, 19 Sep 2026 14:45:35 +0200 Subject: [PATCH 14/18] Split the shape layer into a cheap default and an opt-in full profile Measured on bench-1, on a 2,000-triple graph: vcf-core-vocabulary.shacl.ttl 1.7s vcf-core-consistency.shacl.ttl 33.7s vcf-core-vocabulary-sparql.shacl.ttl 73.2s That reverses the design and corrects a claim. The four mutation classes the shape layer exists for -- corrupt_allele_value, corrupt_record_index, corrupt_value_item_allele, corrupt_sample_index_expanded -- are covered by the consistency and SPARQL profiles, NOT by the core one. The core profile constrains vcfc:sampleIndex to exist once as an integer >= 1 and says nothing about two samples sharing a value; uniqueness is a sh:sparql rule ("Record indices must be unique within a VCF file") in the SPARQL profile, and value agreement against REF/ALT is in the consistency profile. So defaulting to the core profile alone, as the previous commit did, detected none of them: the 0.850 -> 0.912 figure belonged to a profile set that was not being loaded. But those two profiles self-join the graph, so they cannot be the default either -- 63x the core cost at 2,000 triples, and widening with size. --shacl-profile core (default) stays cheap and keeps the 512 MiB gate. --shacl-profile full adds both, gated to 16 MiB, where two minutes for four mutation classes is a reasonable trade. --shacl-shapes takes a comma-separated list, and validate_shacl merges several profiles into one graph. Co-Authored-By: Claude Opus 5 --- docs/validation.md | 34 +++++-- src/validation/validation_runner.py | 60 +++++++++--- test/test_shacl_default_unit.py | 82 ++++++++++++++++- test/test_validation_oracle_unit.py | 18 +++- vcf_rdfizer.py | 138 ++++++++++++++++++++++------ 5 files changed, 281 insertions(+), 51 deletions(-) diff --git a/docs/validation.md b/docs/validation.md index cabca92..d18bfd3 100644 --- a/docs/validation.md +++ b/docs/validation.md @@ -550,13 +550,33 @@ with the package -- no vocabulary checkout needed. `--no-shacl` turns it off; gate it is skipped, because `pyshacl` loads the whole graph into memory and a cohort-scale aggregate would not fit. -It defaults on because the queries cannot see what it sees. The mutation-score -experiment injected 113 corruptions; the query suite caught 96 (0.850), and 7 of -the 17 it missed fall into four classes this profile already covers -- -`corrupt_allele_value`, `corrupt_record_index`, `corrupt_value_item_allele` and -`corrupt_sample_index_expanded`. That is 0.850 to 0.912 for a check that was -already written and was enabled in none of the 62 validation runs in the -benchmark campaign. The report records the conformance verdict, the exact +### Two profiles, and which one catches what + +The published profile is split across files that check genuinely different +things, and the split decides what a run can detect: + +| `--shacl-profile` | Files | Checks | Cost on a 2,000-triple graph | +| --- | --- | --- | --- | +| `core` (default) | `vcf-core-vocabulary` | Cardinality and datatype: `vcfc:sampleIndex` exists once and is an integer >= 1 | **1.7 s** | +| `full` | `+ vcf-core-consistency`, `+ vcf-core-vocabulary-sparql` | Uniqueness (*"Record indices must be unique"*, *"Sample names and sample indices must be unique"*) and value agreement (`vcfc:alleleValue` against `REF`/`ALT`) | **+33.7 s, +73.2 s** | + +The distinction matters for what you can claim. The mutation-score experiment +injected 113 corruptions; the query suite caught 96 (0.850), and 7 of the 17 it +missed fall into four classes -- `corrupt_allele_value`, `corrupt_record_index`, +`corrupt_value_item_allele` and `corrupt_sample_index_expanded`. **Those are +covered by `full`, not by the default.** The core profile constrains how many +`sampleIndex` values a sample has, not whether two samples share one, so it +detects none of them. + +`full` is not the default because the cost is real and measured: its two extra +profiles use `sh:sparql` constraints that self-join the graph, so they grow far +faster than the data. It is gated to sources at or below 16 MiB, where closing +those four classes is worth two minutes; the core profile's gate is 512 MiB. + +```bash +# Close the four classes the query suite misses, on a fixture-sized input +vcf-rdfizer --mode full -i ./fixture.vcf --validate --shacl-profile full -o ./out +``` The report records the conformance verdict, the exact violation count, the distinct property paths involved, and the first 50 violations; the full text is written alongside it. diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index ff7ae9b..ce6e6e9 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -1880,9 +1880,28 @@ def flush(lines: list[str] | None) -> None: return results +def merge_shapes_graph(shapes: list[Path]): + """Parse several shapes files into one graph. + + The published profile is split across files that check different things, + and the split matters: the core profile constrains cardinality and datatype + (``sh:minCount`` on vcfc:sampleIndex, and so on), while the *uniqueness and + agreement* rules -- the ones that catch a corrupted value rather than a + missing one -- live in the consistency and SPARQL profiles. Loading only + the core profile detects none of the mutation classes the shape layer is + supposed to cover. + """ + from rdflib import Graph + + graph = Graph() + for path in shapes: + graph.parse(str(path), format="turtle") + return graph + + def validate_shacl( source: Path, - shapes: Path, + shapes: Path | list[Path], results_dir: Path, ontology: Path | None = None, ) -> dict[str, Any]: @@ -1911,19 +1930,25 @@ def validate_shacl( result = { "status": "EXECUTION_FAILED", "error": f"pyshacl is not installed in this image: {error}", - "shapes": str(shapes), + "shapes": [str(path) for path in ( + [shapes] if isinstance(shapes, Path) else list(shapes))], } write_json(report_path, result) return result started = time.monotonic() try: + shapes_list = [shapes] if isinstance(shapes, Path) else list(shapes) + shacl_graph = ( + str(shapes_list[0]) if len(shapes_list) == 1 + else merge_shapes_graph(shapes_list) + ) conforms, _graph, text = pyshacl_validate( str(source), - shacl_graph=str(shapes), + shacl_graph=shacl_graph, ont_graph=str(ontology) if ontology is not None else None, data_graph_format="nt", - shacl_graph_format="turtle", + **({"shacl_graph_format": "turtle"} if isinstance(shacl_graph, str) else {}), ont_graph_format="turtle" if ontology is not None else None, inference="rdfs" if ontology is not None else "none", advanced=True, @@ -1932,7 +1957,8 @@ def validate_shacl( result = { "status": "EXECUTION_FAILED", "error": f"SHACL validation could not run: {error}", - "shapes": str(shapes), + "shapes": [str(path) for path in ( + [shapes] if isinstance(shapes, Path) else list(shapes))], } write_json(report_path, result) return result @@ -1962,7 +1988,7 @@ def validate_shacl( # this run acts on. A graph with warnings only is conforms=False here # and status=PASS, which is the distinction the severities encode. "conforms": bool(conforms), - "shapes": str(shapes), + "shapes": [str(path) for path in shapes_list], "ontology": str(ontology) if ontology is not None else None, "violationCount": len(violations), "violationPaths": paths, @@ -4185,12 +4211,12 @@ def build_arg_parser() -> argparse.ArgumentParser: ) parser.add_argument( "--shacl-shapes", - type=Path, default=None, help=( - "Validate the graph against a SHACL shapes file as an independent " - "structural layer. Off by default: pyshacl loads the whole graph " - "into memory, so it does not scale to a cohort-sized aggregate" + "Validate the graph against SHACL shapes as an independent " + "structural layer. Comma-separated: the published profile splits " + "cardinality/datatype rules from the uniqueness and agreement " + "rules, and only the latter catch a corrupted value" ), ) parser.add_argument( @@ -4337,9 +4363,17 @@ def resolve_args(parser: argparse.ArgumentParser, argv: list[str] | None = None) if len({args.comunica_port, args.qlever_port, args.hdt_port}) != 3: parser.error("--comunica-port, --qlever-port and --hdt-port must all differ") if args.shacl_shapes is not None: - args.shacl_shapes = args.shacl_shapes.resolve() - if not args.shacl_shapes.is_file(): - parser.error(f"SHACL shapes file does not exist: {args.shacl_shapes}") + # Comma-separated, because the published profile is split across files + # and the uniqueness/agreement rules live outside the core one. + paths = [ + Path(token.strip()).resolve() + for token in str(args.shacl_shapes).split(",") + if token.strip() + ] + for path in paths: + if not path.is_file(): + parser.error(f"SHACL shapes file does not exist: {path}") + args.shacl_shapes = paths if args.shacl_ontology is not None: args.shacl_ontology = args.shacl_ontology.resolve() if not args.shacl_ontology.is_file(): diff --git a/test/test_shacl_default_unit.py b/test/test_shacl_default_unit.py index fda3869..235a14d 100644 --- a/test/test_shacl_default_unit.py +++ b/test/test_shacl_default_unit.py @@ -9,8 +9,16 @@ required a vocabulary checkout and an explicit flag. So the shapes are vendored and applied by default, size-gated because pyshacl -loads the whole graph into memory. Closing those four classes takes the score -from 0.850 to 0.912 with no new oracle and no new query. +loads the whole graph into memory. + +One correction, forced by measurement on bench-1: those four classes are NOT +covered by the default profile. ``vcf-core-vocabulary.shacl.ttl`` constrains +cardinality and datatype only; the uniqueness rules are in the SPARQL profile +and the value-agreement rules in the consistency profile. Those two cost 33.7 s +and 73.2 s on a 2,000-triple graph against 1.7 s for the core one, and their +sh:sparql constraints self-join the graph. So the default is the cheap profile, +``--shacl-profile full`` opts into the rest on a fixture-sized graph, and the +0.850 -> 0.912 figure belongs to full, not to the default. """ import hashlib @@ -32,11 +40,77 @@ def test_the_default_shapes_are_bundled_with_the_package(self): """Requiring a separate checkout is why this was never switched on.""" shapes, ontology = vcf_rdfizer.resolve_default_shacl_shapes(REPO_ROOT) self.assertIsNotNone(shapes) - self.assertTrue(shapes.is_file()) - self.assertEqual(shapes.name, vcf_rdfizer.DEFAULT_SHACL_SHAPES) + self.assertEqual( + [path.name for path in shapes], list(vcf_rdfizer.DEFAULT_SHACL_SHAPES) + ) + for path in shapes: + self.assertTrue(path.is_file(), path) self.assertIsNotNone(ontology, "sh:class needs the class hierarchy") self.assertTrue(ontology.is_file()) + def test_the_default_profile_is_the_cheap_one_and_says_so(self): + """Measured on bench-1: core 1.7 s, consistency 33.7 s, SPARQL 73.2 s + on a 2,000-triple graph. The two that catch a corrupted value cost + 63x the one that does not, and their sh:sparql constraints self-join + the graph, so the gap widens with size. core is what a run can absorb. + """ + self.assertEqual( + vcf_rdfizer.DEFAULT_SHACL_SHAPES, ("vcf-core-vocabulary.shacl.ttl",) + ) + + def test_the_full_profile_adds_the_two_that_catch_a_corrupted_value(self): + """vcfc:sampleIndex in the core profile is minCount 1, maxCount 1, + integer >= 1 -- nothing about two samples sharing one. Uniqueness is in + the SPARQL profile, value agreement in the consistency profile. + """ + self.assertIn("vcf-core-consistency.shacl.ttl", vcf_rdfizer.FULL_SHACL_SHAPES) + self.assertIn( + "vcf-core-vocabulary-sparql.shacl.ttl", vcf_rdfizer.FULL_SHACL_SHAPES + ) + for name in vcf_rdfizer.DEFAULT_SHACL_SHAPES: + self.assertIn(name, vcf_rdfizer.FULL_SHACL_SHAPES) + + def test_the_full_profile_resolves_all_three_files(self): + shapes, ontology = vcf_rdfizer.resolve_default_shacl_shapes(REPO_ROOT, "full") + self.assertEqual( + [path.name for path in shapes], list(vcf_rdfizer.FULL_SHACL_SHAPES) + ) + self.assertIsNotNone(ontology) + + def test_the_full_profile_has_a_much_smaller_size_gate(self): + """It is quadratic-ish in graph size; the core gate would be unusable.""" + self.assertLess( + vcf_rdfizer.FULL_SHACL_MAX_SOURCE_BYTES, + vcf_rdfizer.DEFAULT_SHACL_MAX_SOURCE_BYTES, + ) + big = vcf_rdfizer.FULL_SHACL_MAX_SOURCE_BYTES + 1 + self.assertFalse(vcf_rdfizer.shacl_default_applies(big, "full")) + self.assertTrue(vcf_rdfizer.shacl_default_applies(big, "core")) + + def test_the_uniqueness_rules_really_live_outside_the_core_profile(self): + """Pin the claim against the vendored files, not against memory.""" + shacl = DATA_ROOT / "shacl" + core = (shacl / "vcf-core-vocabulary.shacl.ttl").read_text(encoding="utf-8") + sparql = (shacl / "vcf-core-vocabulary-sparql.shacl.ttl").read_text(encoding="utf-8") + consistency = (shacl / "vcf-core-consistency.shacl.ttl").read_text(encoding="utf-8") + self.assertNotIn("must be unique", core) + self.assertIn("Record indices must be unique", sparql) + self.assertIn("Sample names and sample indices must be unique", sparql) + self.assertIn("vcfc:alleleValue", consistency) + + def test_a_partial_profile_set_resolves_to_nothing(self): + """Half a profile set silently drops whole classes of check.""" + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "vcf_rdfizer_data" / "shacl").mkdir(parents=True) + (root / "vcf_rdfizer_data" / "shacl" + / vcf_rdfizer.DEFAULT_SHACL_SHAPES[0]).write_text("", encoding="utf-8") + shapes, ontology = vcf_rdfizer.resolve_default_shacl_shapes(root) + if shapes is not None: + self.skipTest("installed package shadows the temporary root") + self.assertIsNone(shapes) + self.assertIsNone(ontology) + def test_the_version_overlays_are_bundled_too(self): """4.1-4.5 each have their own overlay; shipping only the core is half.""" for version in ("4.1", "4.2", "4.3", "4.4", "4.5"): diff --git a/test/test_validation_oracle_unit.py b/test/test_validation_oracle_unit.py index 374a927..0769e57 100644 --- a/test/test_validation_oracle_unit.py +++ b/test/test_validation_oracle_unit.py @@ -746,9 +746,25 @@ def test_supplied_shapes_and_progress_paths_are_resolved(self): args = self.resolve(self.base(**{ "--shacl-shapes": str(shapes), "--progress-path": str(self.root / "progress.json"), })) - self.assertTrue(args.shacl_shapes.is_absolute()) + # A list now: the published profile is split across files and only some + # of them catch a corrupted value, so the default names three. + self.assertEqual(len(args.shacl_shapes), 1) + self.assertTrue(args.shacl_shapes[0].is_absolute()) self.assertTrue(args.progress_path.is_absolute()) + def test_several_shapes_files_are_accepted_and_all_resolved(self): + """Loading only the core profile detects none of the four classes.""" + first, second = self.root / "a.ttl", self.root / "b.ttl" + for path in (first, second): + path.write_text("", encoding="utf-8") + args = self.resolve(self.base(**{ + "--shacl-shapes": f"{first},{second}", + })) + self.assertEqual( + [path.name for path in args.shacl_shapes], ["a.ttl", "b.ttl"] + ) + self.assertTrue(all(path.is_absolute() for path in args.shacl_shapes)) + def test_no_progress_path_stays_none(self): """Progress reporting is opt-in and must not be invented.""" self.assertIsNone(self.resolve(self.base()).progress_path) diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 472c84c..eca5f7c 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -4830,7 +4830,30 @@ def resolve_default_rules_path(repo_root: Path) -> Path: #: corrupt_allele_value, corrupt_record_index, corrupt_value_item_allele and #: corrupt_sample_index_expanded. Turning them on takes the score to 0.912 with #: no new oracle and no new query. -DEFAULT_SHACL_SHAPES = "vcf-core-vocabulary.shacl.ttl" +#: The cheap, always-affordable profile. It constrains cardinality and datatype +#: -- vcfc:sampleIndex must exist once and be an integer >= 1 -- and says +#: nothing about whether two samples share one. Measured on bench-1 at 1.7 s for +#: a 2,000-triple graph and 34 s for a 95,000-triple one, which is a default a +#: run can absorb. +DEFAULT_SHACL_SHAPES = ("vcf-core-vocabulary.shacl.ttl",) +#: The profiles that catch a *corrupted* value rather than a missing one: +#: uniqueness ("Record indices must be unique within a VCF file", "Sample names +#: and sample indices must be unique") lives in the SPARQL profile, and value +#: agreement (vcfc:alleleValue against vcfc:ref/vcfc:alt) in the consistency +#: profile. These are the four mutation classes the query suite misses. +#: +#: They are NOT default, and the reason is measured rather than assumed: on a +#: 2,000-triple graph the consistency profile took 33.7 s and the SPARQL +#: profile 73.2 s, against 1.7 s for the core one. Their sh:sparql constraints +#: self-join the graph, so the cost grows far faster than the data. Opt in with +#: --shacl-profile full on a fixture-sized graph, where closing those classes +#: is worth two minutes. +FULL_SHACL_SHAPES = ( + "vcf-core-vocabulary.shacl.ttl", + "vcf-core-consistency.shacl.ttl", + "vcf-core-vocabulary-sparql.shacl.ttl", +) +SHACL_PROFILE_CHOICES = ("core", "full") DEFAULT_SHACL_ONTOLOGY = "vcf-core-vocabulary.bundle.ttl" #: pyshacl loads the whole graph into memory, so the default is size-gated #: rather than unconditional. A fixture or a single-sample graph validates in @@ -4855,12 +4878,21 @@ def resolve_bundled_vocabulary_asset(repo_root: Path, relative: str) -> Path | N return None -def resolve_default_shacl_shapes(repo_root: Path) -> tuple[Path | None, Path | None]: - """The bundled shapes and their ontology bundle, or (None, None).""" - shapes = resolve_bundled_vocabulary_asset( - repo_root, f"shacl/{DEFAULT_SHACL_SHAPES}" - ) - if shapes is None: +def resolve_default_shacl_shapes( + repo_root: Path, + profile: str = "core", +) -> tuple[list[Path] | None, Path | None]: + """The bundled shape profiles and their ontology bundle, or (None, None). + + All profiles in the chosen set must resolve: a partial set silently drops + whole classes of check, which is exactly the failure this exists to end. + """ + names = FULL_SHACL_SHAPES if profile == "full" else DEFAULT_SHACL_SHAPES + shapes = [ + resolve_bundled_vocabulary_asset(repo_root, f"shacl/{name}") + for name in names + ] + if any(path is None for path in shapes): return None, None ontology = resolve_bundled_vocabulary_asset( repo_root, f"ontology/{DEFAULT_SHACL_ONTOLOGY}" @@ -4868,7 +4900,21 @@ def resolve_default_shacl_shapes(repo_root: Path) -> tuple[Path | None, Path | N return shapes, ontology -def shacl_default_applies(source_bytes: int | None) -> bool: +#: The full profile set is quadratic-ish in graph size, so its gate is not the +#: core one. 16 MiB of VCF is a fixture or a small single-sample file, which is +#: where a two-minute structural check is a reasonable trade. +FULL_SHACL_MAX_SOURCE_BYTES = 16 * 1024 * 1024 + + +def shacl_max_source_bytes(profile: str) -> int: + """The size gate for one profile set.""" + return ( + FULL_SHACL_MAX_SOURCE_BYTES if profile == "full" + else DEFAULT_SHACL_MAX_SOURCE_BYTES + ) + + +def shacl_default_applies(source_bytes: int | None, profile: str = "core") -> bool: """Whether to validate shapes by default for a source of this size. Size-gated because pyshacl is in-memory. ``None`` means the size could not @@ -4877,7 +4923,7 @@ def shacl_default_applies(source_bytes: int | None) -> bool: """ if source_bytes is None: return False - return 0 <= source_bytes <= DEFAULT_SHACL_MAX_SOURCE_BYTES + return 0 <= source_bytes <= shacl_max_source_bytes(profile) def docker_image_exists(image: str) -> bool: @@ -7250,7 +7296,7 @@ def run_full_mode( validation_engine: str | list[str] = DEFAULT_VALIDATION_ENGINE, validation_engine_options: dict | None = None, validation_strict_conformance: bool = False, - validation_shacl_shapes: Path | None = None, + validation_shacl_shapes: Path | list[Path] | None = None, validation_shacl_ontology: Path | None = None, filter_oracle: str = "auto", rdf_storage_mode: str, @@ -8818,7 +8864,7 @@ def run_validation_mode( engine_options: dict | None = None, rdf_format: str | None = None, strict_conformance: bool = False, - shacl_shapes: Path | None = None, + shacl_shapes: Path | list[Path] | None = None, shacl_ontology: Path | None = None, run_tracker: RunTracker | None = None, stage_result: dict | None = None, @@ -8884,10 +8930,23 @@ def run_validation_mode( engine_args.append("--strict-conformance") shacl_mount: list[str] = [] if shacl_shapes is not None: - # Mounted read-only in its own directory so the shapes file can live - # anywhere on the host without exposing its parent tree for writing. - shacl_mount = ["-v", f"{shacl_shapes.parent.resolve()}:/data/shacl:ro"] - engine_args.extend(["--shacl-shapes", f"/data/shacl/{shacl_shapes.name}"]) + # One or several profiles. They ship in one directory, so a single + # read-only mount covers them all; a caller pointing at files in + # different directories is refused rather than silently half-mounted. + shapes_paths = ( + [shacl_shapes] if isinstance(shacl_shapes, Path) else list(shacl_shapes) + ) + parents = {path.parent.resolve() for path in shapes_paths} + if len(parents) != 1: + raise ValueError( + "--shacl-shapes files must live in one directory; got " + + ", ".join(sorted(str(parent) for parent in parents)) + ) + shacl_mount = ["-v", f"{parents.pop()}:/data/shacl:ro"] + engine_args.extend([ + "--shacl-shapes", + ",".join(f"/data/shacl/{path.name}" for path in shapes_paths), + ]) if shacl_ontology is not None: # The ontology is a sibling directory in a vocabulary checkout, so # it needs its own mount rather than the shapes' one. @@ -9388,16 +9447,32 @@ def main(): "then ~10 h) while qlever answered the same cell in about a second" ), ) + parser.add_argument( + "--shacl-profile", + choices=SHACL_PROFILE_CHOICES, + default="core", + help=( + "Which bundled shape profiles to apply. core (default) constrains " + "cardinality and datatype and is cheap. full adds the consistency " + "and SPARQL profiles, which are the ones that catch a corrupted " + "value -- duplicate record or sample indices, an alleleValue that " + "disagrees with ALT -- and are the four mutation classes the query " + "suite misses. full is measurably expensive: on a 2,000-triple " + "graph its profiles took 33.7 s and 73.2 s against 1.7 s for core, " + "so it is gated to sources at or below " + f"{FULL_SHACL_MAX_SOURCE_BYTES // (1024 * 1024)} MiB" + ), + ) parser.add_argument( "--no-shacl", action="store_true", help=( "Skip the bundled SHACL shape layer, which is otherwise applied by " "default to sources at or below " - f"{DEFAULT_SHACL_MAX_SOURCE_BYTES // (1024 * 1024)} MiB. It catches " - "what the aggregate comparisons structurally cannot -- a value that " - "is counted but never read -- and covers four of the ten mutation " - "classes the query suite misses. --shacl-shapes overrides both" + f"{DEFAULT_SHACL_MAX_SOURCE_BYTES // (1024 * 1024)} MiB. The " + "default (core) profile checks structure the aggregate comparisons " + "do not; --shacl-profile full adds the profiles that catch a " + "corrupted value. --shacl-shapes overrides both" ), ) parser.add_argument( @@ -9560,7 +9635,7 @@ def main(): validation_artifacts: list[str] = [] validation_engines: list[str] = [DEFAULT_VALIDATION_ENGINE] validation_engine_options: dict = {} - shacl_shapes_path: Path | None = None + shacl_shapes_path: list[Path] | None = None shacl_ontology_path: Path | None = None linking_manifests = [] try: @@ -9613,9 +9688,18 @@ def main(): if cottas_warning is not None: eprint(f"Warning: {cottas_warning}") if args.shacl_shapes is not None: - shacl_shapes_path = Path(args.shacl_shapes).expanduser().resolve() - if not shacl_shapes_path.is_file(): - raise ValueError(f"SHACL shapes file not found: {shacl_shapes_path}") + # Comma-separated: the published profile is split across files and + # only some of them catch a corrupted value. + shacl_shapes_path = [ + Path(token.strip()).expanduser().resolve() + for token in str(args.shacl_shapes).split(",") + if token.strip() + ] + for path in shacl_shapes_path: + if not path.is_file(): + raise ValueError(f"SHACL shapes file not found: {path}") + if not shacl_shapes_path: + raise ValueError("--shacl-shapes needs at least one file") if args.shacl_ontology is not None: shacl_ontology_path = Path(args.shacl_ontology).expanduser().resolve() if not shacl_ontology_path.is_file(): @@ -9627,7 +9711,7 @@ def main(): # shapes, in ontology/. Use it when it is there, so the # documented command needs no second flag. candidate = ( - shacl_shapes_path.parent.parent + shacl_shapes_path[0].parent.parent / "ontology" / "vcf-core-vocabulary.bundle.ttl" ) @@ -9640,7 +9724,9 @@ def main(): # published profile, which no run in the benchmark campaign # enabled. Default it on, size-gated, rather than leaving a # written check permanently unused. - bundled_shapes, bundled_ontology = resolve_default_shacl_shapes(repo_root) + bundled_shapes, bundled_ontology = resolve_default_shacl_shapes( + repo_root, args.shacl_profile + ) if bundled_shapes is not None: source_bytes = None try: @@ -9651,7 +9737,7 @@ def main(): source_bytes = source_for_size.stat().st_size except (OSError, TypeError): source_bytes = None - if shacl_default_applies(source_bytes): + if shacl_default_applies(source_bytes, args.shacl_profile): shacl_shapes_path = bundled_shapes shacl_ontology_path = bundled_ontology From 5f2f43d0c0f40a0dfa0e29c31bed02f650ab6da7 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Sun, 20 Sep 2026 08:58:17 +0200 Subject: [PATCH 15/18] Let the mutation harness measure the shape layer, not just the queries Every mutation score this repository has ever reported is a query-only score: validate_graph ran the preflight and core queries and never called validate_shacl at all. So the claim that the shape layer closes four of the ten undetected classes could not be checked here, only argued from reading the profiles. VCF_RDFIZER_MUTATION_SHACL=core|full now folds the shape verdict in. A sh:Violation counts as a detection; an execution failure does not, so a missing pyshacl cannot inflate the score with detections nothing made. The report gains shaclProfile, detectedByShacl, detectedByShaclIds and detectedByQueriesOnly, so the number can be attributed to a layer instead of standing as one opaque total. Unset, nothing changes: the default run still reports 96/113, so every archived score stays comparable. A known_undetected mutation that the shape layer catches is recorded rather than failed. Closing one is the point of enabling it, and the catalogue still describes the query layer's coverage, which is what an unqualified run measures. Co-Authored-By: Claude Opus 5 --- test/test_validation_mutation_unit.py | 66 ++++++++++++++++++++++++++- 1 file changed, 65 insertions(+), 1 deletion(-) diff --git a/test/test_validation_mutation_unit.py b/test/test_validation_mutation_unit.py index 302a6a7..3dbe8df 100644 --- a/test/test_validation_mutation_unit.py +++ b/test/test_validation_mutation_unit.py @@ -75,9 +75,37 @@ def execute(self, query_id: str, query_path: Path) -> dict: "rawResult": str(raw_path), "engine": "rdflib", "error": str(error)} +def shacl_profile_paths() -> list[Path] | None: + """The bundled shape profiles named by VCF_RDFIZER_MUTATION_SHACL, or None. + + The query layer and the shape layer catch different things, and until now + this harness measured only the first: every score it has ever reported is a + query-only score. Setting the variable to `full` or `core` folds the shape + verdict in, which is what makes the shape layer's contribution measurable + instead of predicted. + """ + profile = os.environ.get("VCF_RDFIZER_MUTATION_SHACL", "").strip().lower() + if profile not in {"core", "full"}: + return None + import vcf_rdfizer + + repo_root = Path(vcf_rdfizer.__file__).resolve().parent + shapes, _ontology = vcf_rdfizer.resolve_default_shacl_shapes(repo_root, profile) + return shapes + + +def shacl_ontology_path() -> Path | None: + import vcf_rdfizer + + repo_root = Path(vcf_rdfizer.__file__).resolve().parent + _shapes, ontology = vcf_rdfizer.resolve_default_shacl_shapes(repo_root, "core") + return ontology + + def validate_graph( graph_text: str, representation: str, *, strict_conformance: bool = False, include_qual: bool = True, mapping_policy: str = "strict", + shacl_shapes: list[Path] | None = None, ) -> dict: """Run the whole validation decision over one graph.""" parser = fixtures.parser_summary(representation, include_qual=include_qual) @@ -95,10 +123,24 @@ def validate_graph( # statement count itself. For well-formed N-Triples that is exactly the # number of non-empty lines, which is what rapper would report. parsed = sum(1 for line in graph_text.splitlines() if line.strip()) - return V.evaluate_validation( + verdict = V.evaluate_validation( executions, parser, representation, strict_conformance=strict_conformance, mapping_policy=mapping_policy, parsed_triple_count=parsed, ) + if shacl_shapes: + graph_path = Path(td) / "graph.nt" + graph_path.write_text(graph_text, encoding="utf-8") + shacl = V.validate_shacl( + graph_path, shacl_shapes, Path(td), shacl_ontology_path() + ) + verdict = dict(verdict) + verdict["shacl"] = shacl + # A shape violation is a detection in its own right. An execution + # failure is NOT: pyshacl being absent or erroring would otherwise + # inflate the score with detections nothing actually made. + if shacl.get("status") == "FAIL" and verdict["status"] == "PASS": + verdict["status"] = "SHACL_VIOLATION" + return verdict @unittest.skipIf(rdflib is None, "rdflib is required for the host mutation harness") @@ -143,14 +185,27 @@ def graph_for(cls, representation: str, options: tuple) -> str: cls.graphs[key] = fixtures.build_graph(representation, **dict(options)) return cls.graphs[key] + #: Resolved once. None unless VCF_RDFIZER_MUTATION_SHACL names a profile, + #: so the default run stays exactly the query-only measurement it has + #: always been and every archived score remains comparable. + shacl_shapes = shacl_profile_paths() + @classmethod def tearDownClass(cls): detected = [r for r in cls.results if r["detected"]] + by_shacl = [r for r in cls.results if r.get("detectedByShacl")] report = { "total": len(cls.results), "detected": len(detected), "score": round(len(detected) / len(cls.results), 4) if cls.results else 0.0, "knownUndetected": [r["id"] for r in cls.results if not r["detected"]], + # Which layer did the detecting. Without this the score is one + # number that cannot be attributed, and the shape layer's + # contribution stays a claim rather than a measurement. + "shaclProfile": os.environ.get("VCF_RDFIZER_MUTATION_SHACL") or None, + "detectedByShacl": len(by_shacl), + "detectedByShaclIds": sorted({r["id"] for r in by_shacl}), + "detectedByQueriesOnly": len(detected) - len(by_shacl), "mutations": cls.results, } destination = os.environ.get("VCF_RDFIZER_MUTATION_REPORT") @@ -172,19 +227,28 @@ def _check(self, mutation: mutations.Mutation, representation: str): mutated, representation, strict_conformance=strict, include_qual=options.get("include_qual", True), mapping_policy=mutation.mapping_policy, + shacl_shapes=self.shacl_shapes, ) detected = verdict["status"] != "PASS" + detected_by_shacl = verdict["status"] == "SHACL_VIOLATION" self.results.append({ "id": mutation.id, "representation": representation, "vcfElement": mutation.vcf_element, "expectedDetectedBy": mutation.expected_detected_by, "detected": detected, + "detectedByShacl": detected_by_shacl, "status": verdict["status"], "knownUndetected": mutation.known_undetected, "graphOptions": dict(mutation.graph_options), }) if mutation.known_undetected: + if self.shacl_shapes and detected_by_shacl: + # Closing one of these is the entire point of enabling the + # shape layer, so it is recorded rather than failed. The + # catalogue still describes the QUERY layer's coverage, which + # is what an unqualified run measures. + return self.assertFalse( detected, f"{mutation.id} is now DETECTED. This gap has been closed - remove " From d50a31add2f58d083efd7ef593bb35b62c32ce31 Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Mon, 21 Sep 2026 10:34:37 +0200 Subject: [PATCH 16/18] Mint the allele layer for whoever joins to it, not just structured INFO MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `emit_record_detail` gated the allele layer on `info_representation == "structured"`, on the reasoning that the Number=A/R/G value items are what join to the allele resources. The expanded sample layer joins to them too: every `vcfc:GenotypeAlleleCall` carries `vcfc:calledAllele` pointing at `/allele/`, and it emits that edge regardless of the INFO representation. So `--info-representation raw --sample-representation expanded` produced a graph whose allele references were never described — 1,972 dangling objects per 1,000 records in the covering-set benchmark, with zero `vcfc:alleleIndex` triples anywhere in the file. The SPARQL oracle counts `calledAllele` edges without following the join, so all ten covering-set rows exited 0 through the whole campaign; the shape layer reported it on first contact, as 5,916 violations on `alleleIndex`, `alleleKind` and `alleleValue`. Replace the gate with `allele_layer_required()`, a disjunction over both representation axes, and pass the sample representation down from full mode. `raw` + `condensed` still omits the layer: condensed emits no `calledAllele`, so nothing references an allele there. Co-Authored-By: Claude Opus 5 --- docs/conversion.md | 9 +- test/test_allele_reference_integrity_unit.py | 181 +++++++++++++++++++ vcf_rdfizer.py | 34 +++- 3 files changed, 219 insertions(+), 5 deletions(-) create mode 100644 test/test_allele_reference_integrity_unit.py diff --git a/docs/conversion.md b/docs/conversion.md index 108b1b0..63c43c1 100644 --- a/docs/conversion.md +++ b/docs/conversion.md @@ -191,7 +191,14 @@ emitted and the values stay available as ordinary INFO values, rather than producing a resource that would fail its SHACL shape. `--info-representation raw` emits only the opaque `vcfc:infoRaw` string, and -turns off the allele layer with it (the value items join to the alleles). +drops the value items with it. + +The allele layer is *not* tied to the INFO representation. Two layers join to +`/allele/`: the structured INFO value items, and the expanded +sample layer's per-call `vcfc:calledAllele`. Either one alone requires the +alleles, so the layer is emitted when `--info-representation structured` **or** +`--sample-representation expanded` is in force, and left out only for +`raw` + `condensed`, where nothing references it. ### `append_header_representation_rdf` — structured headers diff --git a/test/test_allele_reference_integrity_unit.py b/test/test_allele_reference_integrity_unit.py new file mode 100644 index 0000000..82cdef3 --- /dev/null +++ b/test/test_allele_reference_integrity_unit.py @@ -0,0 +1,181 @@ +"""Every allele a graph points at must also be described in that graph. + +Two layers join to ``/allele/``: the structured INFO value items +(Number=A/R/G) and the expanded sample layer's per-call ``vcfc:calledAllele``. +The allele layer used to be minted only for the structured INFO representation, +so ``--info-representation raw --sample-representation expanded`` emitted +``vcfc:calledAllele`` edges whose targets carried no ``vcfc:alleleIndex``, +``vcfc:alleleValue`` or ``vcfc:alleleKind`` -- 1,972 dangling references per +1,000 records in the covering-set benchmark. The SPARQL oracle counts the edges +without following the join, so only the shape layer saw it. + +These tests pin the disjunction directly and then check the property it exists +to protect, across all four representation combinations. +""" + +import re +import tempfile +import unittest +from pathlib import Path + +import vcf_rdfizer +from vcf_rdfizer_vocab import VCFC_NAMESPACE +from test.helpers import VerboseTestCase + +RECORDS_HEADER = ( + "SOURCE_FILE\tROW_ID\tCHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\t" + "INFO\tFORMAT\tS1\n" +) +RECORDS_ROWS = ( + "s.vcf\t1\t1\t100\t.\tA\tG,T\t50\tPASS\tDP=7;AF=0.4,0.6\tGT:DP\t1/2:7\n" + "s.vcf\t2\t1\t200\t.\tC\tCTT\t30\tPASS\tDP=9\tGT:DP\t0/1:9\n" +) +HEADERS_TSV = ( + "SOURCE_FILE\tHEADER_INDEX\tHEADER_KEY\tHEADER_VALUE\tRAW_LINE\n" + "s.vcf\t1\tfileformat\tVCFv4.5\tx\n" + "s.vcf\t2\tcontig\t\tx\n" + "s.vcf\t3\tINFO\t\tx\n" + "s.vcf\t4\tINFO\t\tx\n" + "s.vcf\t5\tFORMAT\t\tx\n" + "s.vcf\t6\tFORMAT\t\tx\n" +) + +TRIPLE = re.compile(r"^<([^>]+)>\s+<([^>]+)>\s+(.+?)\s*\.$") + + +def parse_triples(path: Path) -> list[tuple[str, str, str]]: + """Parse the emitters' N-Triples output into (subject, predicate, object).""" + triples = [] + for line in path.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if not line: + continue + match = TRIPLE.match(line) + if match is None: + continue + subject, predicate, obj = match.groups() + if obj.startswith("<") and obj.endswith(">"): + obj = obj[1:-1] + triples.append((subject, predicate, obj)) + return triples + + +def emit_graph( + tmp_path: Path, info_representation: str, sample_representation: str +) -> list[tuple[str, str, str]]: + """Run the record-detail and sample emitters the way full mode chains them.""" + records_tsv = tmp_path / "s.records.tsv" + records_tsv.write_text(RECORDS_HEADER + RECORDS_ROWS, encoding="utf-8") + headers_tsv = tmp_path / "s.header_lines.tsv" + headers_tsv.write_text(HEADERS_TSV, encoding="utf-8") + rdf_path = tmp_path / f"{info_representation}-{sample_representation}.nt" + rdf_path.write_text("", encoding="utf-8") + + if sample_representation == "expanded": + vcf_rdfizer.append_expanded_sample_rdf( + records_tsv, rdf_path, headers_tsv, progress_interval_records=0 + ) + else: + vcf_rdfizer.append_condensed_sample_rdf( + records_tsv, rdf_path, headers_tsv, progress_interval_records=0 + ) + vcf_rdfizer.emit_record_detail( + info_representation, + records_tsv=records_tsv, + header_lines_tsv=headers_tsv, + rdf_path=rdf_path, + sample_representation=sample_representation, + ) + return parse_triples(rdf_path) + + +COMBINATIONS = [ + (info, sample) + for info in ("structured", "raw") + for sample in ("expanded", "condensed") +] + + +class AlleleLayerRequiredTest(VerboseTestCase): + """The decision is a disjunction over both representation axes.""" + + def test_either_consumer_alone_requires_the_allele_layer(self): + self.assertTrue(vcf_rdfizer.allele_layer_required("structured", "condensed")) + self.assertTrue(vcf_rdfizer.allele_layer_required("raw", "expanded")) + self.assertTrue(vcf_rdfizer.allele_layer_required("structured", "expanded")) + + def test_no_consumer_leaves_the_allele_layer_out(self): + """Condensed samples carry no calledAllele, so raw INFO needs no alleles.""" + self.assertFalse(vcf_rdfizer.allele_layer_required("raw", "condensed")) + + def test_unknown_representations_are_rejected(self): + with self.assertRaises(ValueError): + vcf_rdfizer.allele_layer_required("nested", "expanded") + with self.assertRaises(ValueError): + vcf_rdfizer.allele_layer_required("raw", "flattened") + + +class AlleleReferenceIntegrityTest(VerboseTestCase): + """No representation combination may reference an undescribed allele.""" + + def described_alleles(self, triples) -> set[str]: + return { + subject + for subject, predicate, _ in triples + if predicate == f"{VCFC_NAMESPACE}alleleIndex" + } + + def referenced_alleles(self, triples) -> set[str]: + return { + obj + for _, predicate, obj in triples + if predicate == f"{VCFC_NAMESPACE}calledAllele" + } + + def test_every_called_allele_is_described(self): + for info, sample in COMBINATIONS: + with self.subTest(info=info, sample=sample): + with tempfile.TemporaryDirectory() as td: + triples = emit_graph(Path(td), info, sample) + dangling = self.referenced_alleles(triples) - self.described_alleles(triples) + self.assertEqual( + dangling, + set(), + f"{info}/{sample} references alleles it never describes", + ) + + def test_raw_expanded_still_emits_the_called_alleles(self): + """The fix must describe the targets, not silence the references.""" + with tempfile.TemporaryDirectory() as td: + triples = emit_graph(Path(td), "raw", "expanded") + referenced = self.referenced_alleles(triples) + # Allele IRIs are record-scoped: row 1 calls 1/2 over REF=A ALT=G,T and + # row 2 calls 0/1 over REF=C ALT=CTT, so four distinct resources. + self.assertEqual(len(referenced), 4, sorted(referenced)) + for subject in referenced: + described = { + predicate + for triple_subject, predicate, _ in triples + if triple_subject == subject + } + for term in ("alleleIndex", "alleleValue", "alleleKind"): + self.assertIn(f"{VCFC_NAMESPACE}{term}", described, f"{subject} {term}") + + def test_raw_expanded_withholds_the_structured_info_layer(self): + """Fixing the alleles must not smuggle structured INFO into raw mode.""" + with tempfile.TemporaryDirectory() as td: + triples = emit_graph(Path(td), "raw", "expanded") + predicates = {predicate for _, predicate, _ in triples} + self.assertNotIn(f"{VCFC_NAMESPACE}hasInfoFieldValue", predicates) + self.assertIn(f"{VCFC_NAMESPACE}alleleIndex", predicates) + + def test_raw_condensed_remains_allele_free(self): + """Nothing joins to an allele there, so the layer stays out.""" + with tempfile.TemporaryDirectory() as td: + triples = emit_graph(Path(td), "raw", "condensed") + self.assertEqual(self.described_alleles(triples), set()) + self.assertEqual(self.referenced_alleles(triples), set()) + + +if __name__ == "__main__": + unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index eca5f7c..57aa61f 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -4574,28 +4574,53 @@ def produce(emit): return _append_rdf_atomically(rdf_path, stats, produce) +def allele_layer_required(info_representation: str, sample_representation: str) -> bool: + """Decide whether the record's allele resources have to be minted. + + Two independent layers join to ``/allele/``: the structured + INFO value items for Number=A/R/G fields, and the expanded sample layer's + ``vcfc:calledAllele`` on every ``vcfc:GenotypeAlleleCall``. Either one alone + is enough to require the alleles, so the decision is a disjunction rather + than a property of the INFO representation. Tying it to INFO alone left + ``--info-representation raw --sample-representation expanded`` emitting + ``vcfc:calledAllele`` edges whose targets were never described. + """ + if info_representation not in INFO_REPRESENTATION_CHOICES: + raise ValueError(f"unknown INFO representation: {info_representation}") + if sample_representation not in SAMPLE_REPRESENTATION_CHOICES: + choices = ", ".join(sorted(SAMPLE_REPRESENTATION_CHOICES)) + raise ValueError( + f"unsupported sample representation '{sample_representation}'; " + f"choose {choices}" + ) + return info_representation == "structured" or sample_representation == "expanded" + + def emit_record_detail( info_representation: str, *, records_tsv: Path, header_lines_tsv: Path, rdf_path: Path, + sample_representation: str = "expanded", version: "vocab.VCFVersion | None" = None, ) -> dict | None: """Append the record detail, and when selected the structured INFO form. ID, ALT, QUAL, FILTER and INFO raw are always emitted: the RML mapping cannot type the missing token per row, so this is the only place they can - come from. The allele layer travels with the structured INFO representation, - because the Number=A/R/G value items are joined to the allele resources it - mints. + come from. The allele layer is emitted whenever some other layer joins to + it -- the structured INFO value items, or the expanded sample layer's + per-call ``vcfc:calledAllele`` -- so that no representation combination can + reference an allele resource that was never described. """ if info_representation not in INFO_REPRESENTATION_CHOICES: raise ValueError(f"unknown INFO representation: {info_representation}") structured = info_representation == "structured" return append_record_detail_rdf( records_tsv, header_lines_tsv, rdf_path, - emit_qual=True, emit_info=structured, emit_alleles=structured, + emit_qual=True, emit_info=structured, + emit_alleles=allele_layer_required(info_representation, sample_representation), version=version, ) @@ -7775,6 +7800,7 @@ def fail_current(stage: str, message: str): records_tsv=triplet["records"], header_lines_tsv=triplet["headers"], rdf_path=raw_rdf_files[0], + sample_representation=sample_workflow.representation, version=effective_version, ) except Exception as exc: From d33d3af2835fd5faf37622af14786d6eb662debc Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Mon, 21 Sep 2026 11:01:50 +0200 Subject: [PATCH 17/18] Teach the census oracle the same allele rule the emitter now applies Fixing the emitter alone moved the three covering-set failures from a shape violation to a census MISMATCH: the graph gained the 1,972 allele triples it had been missing, and `expected_census` -- which still merged the allele layer only under `info_representation == "structured"` -- reported them as extra rows, together with the 1,972 `rdf:type` triples they carry. Split the allele-layer counters out of `emitted_record_counters`, the way the genotype layer already was, and merge them when structured INFO *or* the expanded sample representation asks for them. Everything the emitter writes inside its `emit_alleles` branch moves with them, the contig and assembly links included, or the two sides would disagree again on the next raw-INFO run. `validation_fixtures.build_graph` mirrored the old rule too; it now calls `allele_layer_required` so the fixture harness cannot drift from the wrapper. Co-Authored-By: Claude Opus 5 --- src/validation/validation_runner.py | 57 +++++++++++++------- test/test_allele_reference_integrity_unit.py | 37 +++++++++++++ test/validation_fixtures.py | 8 ++- 3 files changed, 83 insertions(+), 19 deletions(-) diff --git a/src/validation/validation_runner.py b/src/validation/validation_runner.py index ce6e6e9..8fdbd3c 100644 --- a/src/validation/validation_runner.py +++ b/src/validation/validation_runner.py @@ -579,6 +579,15 @@ def emitted_record_counters( """ classes: Counter[str] = Counter() predicates: Counter[str] = Counter() + # The allele layer is reported separately for the same reason the genotype + # layer is: it is not owned by one representation axis. Structured INFO + # needs it for the Number=A/R/G value items, and the expanded sample layer + # needs it for vcfc:calledAllele, so expected_census merges it when either + # asks. Everything the emitter writes inside its ``emit_alleles`` branch -- + # the contig and assembly links as well as the allele resources -- belongs + # here, or the oracle and the emitter would disagree about raw INFO. + allele_classes: Counter[str] = Counter() + allele_predicates: Counter[str] = Counter() assembly_contig_ids: set[str] = set() reference_alleles = alt_alleles = 0 @@ -595,15 +604,15 @@ def emitted_record_counters( # declared reference sequence; the two are mutually exclusive. assembly_id = vocab.parse_bracketed_chrom(chrom) if assembly_id is not None: - predicates["chromAssemblyContig"] += 1 + allele_predicates["chromAssemblyContig"] += 1 if assembly_id not in assembly_contig_ids: assembly_contig_ids.add(assembly_id) - classes["AssemblyContig"] += 1 - predicates["assemblyContigId"] += 1 + allele_classes["AssemblyContig"] += 1 + allele_predicates["assemblyContigId"] += 1 if has_assembly_line: - predicates["declaredInAssembly"] += 1 + allele_predicates["declaredInAssembly"] += 1 elif chrom in contig_ids: - predicates["chromosome"] += 1 + allele_predicates["chromosome"] += 1 alleles = vocab.parse_alt_alleles(ref, alt) allele_uris = {allele.index for allele in alleles} @@ -614,17 +623,17 @@ def emitted_record_counters( else: alt_alleles += 1 if allele.symbolic_id and allele.symbolic_id in alt_declaration_ids: - predicates["declaredByAlt"] += 1 + allele_predicates["declaredByAlt"] += 1 if allele.symbolic_type: - predicates["svType"] += 1 + allele_predicates["svType"] += 1 if allele.breakend is not None: - classes["Breakend"] += 1 + allele_classes["Breakend"] += 1 if allele.breakend.orientation is not None: - predicates["breakendOrientation"] += 1 + allele_predicates["breakendOrientation"] += 1 if allele.breakend.replacement: - predicates["breakendReplacementString"] += 1 + allele_predicates["breakendReplacementString"] += 1 if allele.breakend.is_single: - predicates["isSingleBreakend"] += 1 + allele_predicates["isSingleBreakend"] += 1 entries = parse_info_entries(info) for key, value in entries: @@ -666,14 +675,14 @@ def emitted_record_counters( ) if reference_alleles: - classes["ReferenceAllele"] = reference_alleles - predicates["hasReferenceAllele"] = reference_alleles + allele_classes["ReferenceAllele"] = reference_alleles + allele_predicates["hasReferenceAllele"] = reference_alleles if alt_alleles: - classes["AltAllele"] = alt_alleles - predicates["hasAltAllele"] = alt_alleles + allele_classes["AltAllele"] = alt_alleles + allele_predicates["hasAltAllele"] = alt_alleles total_alleles = reference_alleles + alt_alleles for name in ("alleleIndex", "alleleValue", "alleleKind"): - predicates[name] += total_alleles + allele_predicates[name] += total_alleles if value_items: classes["FieldValueItem"] += value_items @@ -699,6 +708,8 @@ def emitted_record_counters( return { "emittedRecordClasses": dict(classes), "emittedRecordPredicates": dict(predicates), + "emittedAlleleClasses": dict(allele_classes), + "emittedAllelePredicates": dict(allele_predicates), "emittedGenotypeClasses": dict(genotype_classes), "emittedGenotypePredicates": dict(genotype_predicates), "alleleCount": total_alleles, @@ -952,13 +963,23 @@ def expected_census( predicates.get(f"{VCFC}fieldIndex", 0) + values ) - # The allele layer, the value items, the SV carriers and the parsed - # genotype layer travel with the structured INFO representation. + # The value items and the SV carriers travel with the structured INFO + # representation. The allele layer does not -- it is merged below. for class_name, count in parser["emittedRecordClasses"].items(): classes[f"{VCFC}{class_name}"] = classes.get(f"{VCFC}{class_name}", 0) + count for name, count in parser["emittedRecordPredicates"].items(): predicates[f"{VCFC}{name}"] = predicates.get(f"{VCFC}{name}", 0) + count + # The allele layer is required by whoever joins to it: the structured INFO + # value items (Number=A/R/G) or the expanded sample layer's calledAllele. + # This mirrors ``allele_layer_required`` in the wrapper; the two must agree + # or every raw-INFO expanded run reports a census mismatch. + if info_representation == "structured" or representation == "expanded": + for class_name, count in parser.get("emittedAlleleClasses", {}).items(): + classes[f"{VCFC}{class_name}"] = classes.get(f"{VCFC}{class_name}", 0) + count + for name, count in parser.get("emittedAllelePredicates", {}).items(): + predicates[f"{VCFC}{name}"] = predicates.get(f"{VCFC}{name}", 0) + count + classes = _nonzero(classes) predicates = _nonzero(predicates) # Every typed resource contributes exactly one rdf:type triple here. diff --git a/test/test_allele_reference_integrity_unit.py b/test/test_allele_reference_integrity_unit.py index 82cdef3..19a692b 100644 --- a/test/test_allele_reference_integrity_unit.py +++ b/test/test_allele_reference_integrity_unit.py @@ -177,5 +177,42 @@ def test_raw_condensed_remains_allele_free(self): self.assertEqual(self.referenced_alleles(triples), set()) +class OracleAgreesWithTheEmitterTest(VerboseTestCase): + """The census expectation must apply the same rule the emitter does. + + Fixing the emitter alone turned the covering set's three failures from a + shape violation into a census MISMATCH: the graph gained 1,972 allele + triples the oracle, which still keyed the allele layer on structured INFO, + reported as extra rows. Both sides read the disjunction now. + """ + + def census(self, representation: str, info_representation: str) -> dict: + from test import validation_fixtures as vfixtures + + summary = vfixtures.parser_summary( + representation, include_info=info_representation == "structured" + ) + return { + row["predicate"]: row["tripleCount"] + for row in summary["q09_predicate_census"] + } + + def test_raw_expanded_expects_the_allele_layer(self): + census = self.census("expanded", "raw") + for term in ("alleleIndex", "alleleValue", "alleleKind"): + self.assertIn(f"{VCFC_NAMESPACE}{term}", census, term) + self.assertNotIn(f"{VCFC_NAMESPACE}hasInfoValue", census) + + def test_raw_condensed_expects_no_allele_layer(self): + census = self.census("condensed", "raw") + self.assertNotIn(f"{VCFC_NAMESPACE}alleleIndex", census) + + def test_structured_expects_the_allele_layer_in_both_representations(self): + for representation in ("expanded", "condensed"): + with self.subTest(representation=representation): + census = self.census(representation, "structured") + self.assertIn(f"{VCFC_NAMESPACE}alleleIndex", census) + + if __name__ == "__main__": unittest.main() diff --git a/test/validation_fixtures.py b/test/validation_fixtures.py index 7468f84..893cdf3 100644 --- a/test/validation_fixtures.py +++ b/test/validation_fixtures.py @@ -279,7 +279,13 @@ def build_graph( vcf_rdfizer.append_record_detail_rdf( records_tsv, headers_tsv, graph, emit_qual=include_qual, emit_info=include_info, - emit_alleles=include_info, version=version, + # Same disjunction the wrapper applies: the expanded sample layer + # joins to the alleles through vcfc:calledAllele, so raw INFO alone + # does not mean the allele layer can be left out. + emit_alleles=vcf_rdfizer.allele_layer_required( + "structured" if include_info else "raw", representation + ), + version=version, progress_interval_records=0, ) if representation == "expanded": From 8695fdc9941599be1f5e846525a5b8f6f941046d Mon Sep 17 00:00:00 2001 From: ecrum19 Date: Tue, 22 Sep 2026 16:13:52 +0200 Subject: [PATCH 18/18] Let COTTAS query a condensed graph again The guard that refused it was correct when it was written: native pycottas answered 18 of 27 queries against the condensed encoding and then ran 41 hours, and ~10 hours on a second attempt, on q05_sample_genotype_counts without finishing. Both attempts are archived. That engine no longer exists. COTTAS is answered through comunica over DuckDB, and the cell the guard was built around -- test-10k, condensed, all four engines -- now completes in 99.8 minutes with every engine agreeing on every artifact. 06_equivalence as a whole ran in 8.98 h. Leaving the guard in would have been worse than a stale comment. It drops cottas from `--validation-engine all` with a warning rather than an error, and 06_equivalence runs with `all` -- so the two condensed cells would have lost the engine the experiment exists to compare, and the run would have looked complete. The merge that brought the new engine in was textually clean and would have shipped exactly that. Removes the two resolution helpers, the --allow-cottas-condensed opt-in, and the call site. The flag landed after v3.0.3 and was never released, and the benchmark harness never passed it, so nothing external depends on it. The guard's test file is replaced rather than deleted: the new one asserts the helpers and the flag stay gone and that `all` keeps every engine, so a reintroduction has to be deliberate instead of arriving as a merge artifact. Co-Authored-By: Claude Opus 5 --- test/test_cottas_condensed_engine_unit.py | 61 ++++++++++ test/test_cottas_condensed_guard_unit.py | 134 ---------------------- vcf_rdfizer.py | 103 ----------------- 3 files changed, 61 insertions(+), 237 deletions(-) create mode 100644 test/test_cottas_condensed_engine_unit.py delete mode 100644 test/test_cottas_condensed_guard_unit.py diff --git a/test/test_cottas_condensed_engine_unit.py b/test/test_cottas_condensed_engine_unit.py new file mode 100644 index 0000000..6aacec6 --- /dev/null +++ b/test/test_cottas_condensed_engine_unit.py @@ -0,0 +1,61 @@ +"""COTTAS must stay in the engine set for a condensed graph. + +There was a guard here that refused `--validation-engine cottas` against a +condensed graph, and silently dropped cottas from `--validation-engine all`. +It was right when it was written: native pycottas answered 18 of 27 queries and +then ran 41 hours, and ~10 hours on a second attempt, on +q05_sample_genotype_counts without finishing. + +That engine is gone. COTTAS is now answered through comunica over DuckDB, and +the same cell -- test-10k, condensed, all four engines -- completes in 99.8 +minutes with every engine agreeing on every artifact. + +The guard therefore has to go too, and the danger is specific: 06_equivalence +runs with `--validation-engine all`, so a surviving drop would remove cottas +from exactly the two condensed cells the experiment exists to compare, and do it +with a warning rather than a failure. The result would look complete and be +missing an engine. These tests exist so that cannot come back quietly. +""" +from __future__ import annotations + +import unittest +from pathlib import Path + +import vcf_rdfizer +from test.helpers import VerboseTestCase + + +class CottasCondensedEngineTests(VerboseTestCase): + def test_no_condensed_guard_survives(self): + """Any reintroduction must be deliberate, not a merge artifact.""" + for name in ( + "cottas_condensed_resolution", + "cottas_condensed_engine_rejection", + ): + self.assertFalse( + hasattr(vcf_rdfizer, name), + f"{name} is back: COTTAS on a condensed graph is supported now, " + "and a silent drop would hollow out 06_equivalence", + ) + + def test_the_opt_in_flag_is_gone_with_it(self): + """--allow-cottas-condensed only made sense while the refusal existed. + + The parser is built inside main(), so there is nothing to introspect; + the source is the only thing to assert against. + """ + source = Path(vcf_rdfizer.__file__).read_text(encoding="utf-8") + self.assertNotIn("--allow-cottas-condensed", source) + self.assertNotIn("allow_cottas_condensed", source) + + def test_all_keeps_every_engine_for_a_condensed_run(self): + """`all` means all: no engine may be removed behind a warning.""" + engines = vcf_rdfizer.parse_validation_engines("all") + self.assertIn("cottas", engines) + self.assertEqual( + sorted(engines), sorted(vcf_rdfizer.VALIDATION_ENGINE_CHOICES) + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/test_cottas_condensed_guard_unit.py b/test/test_cottas_condensed_guard_unit.py deleted file mode 100644 index c7a4cfb..0000000 --- a/test/test_cottas_condensed_guard_unit.py +++ /dev/null @@ -1,134 +0,0 @@ -"""The one combination that does not terminate, refused rather than started. - -Reproduced twice and archived under benchmarks_outputs__stalled/. Native -pycottas answered the fourteen preflight queries and Q1-Q4 against the -condensed encoding of a 10,000-record fixture, then failed to complete -``q05_sample_genotype_counts``: 41 hours in the first attempt, about 10 in the -second, both at roughly 190% CPU. QLever answered all thirteen questions on -that same cell, and the sibling expanded encoding finished every engine -including pycottas in 185-194 minutes. - -So the defect is narrow -- native pycottas x condensed encoding x a -genotype-level query -- and until it is diagnosed the honest thing is to refuse -the combination. 51 hours of unattended runtime were spent finding out the hard -way, and the per-query timeout could not bound it at the time. -""" - -import unittest - -import vcf_rdfizer -from test.helpers import VerboseTestCase - - -class RejectionTests(VerboseTestCase): - def test_naming_cottas_against_condensed_is_refused(self): - self.assertIsNotNone( - vcf_rdfizer.cottas_condensed_engine_rejection( - engines=["cottas"], sample_representation="condensed", allow=False - ) - ) - - def test_the_refusal_names_both_ways_out(self): - """A refusal that does not say what works instead is just a failure.""" - message = vcf_rdfizer.cottas_condensed_engine_rejection( - engines=["cottas"], sample_representation="condensed", allow=False - ) - self.assertIn("qlever", message) - self.assertIn("expanded", message) - self.assertIn("--allow-cottas-condensed", message) - - def test_the_refusal_cites_the_measurement_rather_than_asserting(self): - message = vcf_rdfizer.cottas_condensed_engine_rejection( - engines=["cottas"], sample_representation="condensed", allow=False - ) - self.assertIn("41 h", message) - self.assertIn("18 of 27", message) - - def test_cottas_against_expanded_is_allowed(self): - """Expanded completed every engine in 185-194 min; nothing to refuse.""" - self.assertIsNone( - vcf_rdfizer.cottas_condensed_engine_rejection( - engines=["cottas"], sample_representation="expanded", allow=False - ) - ) - - def test_other_engines_against_condensed_are_allowed(self): - for engine in ("qlever", "comunica", "hdt"): - self.assertIsNone( - vcf_rdfizer.cottas_condensed_engine_rejection( - engines=[engine], sample_representation="condensed", allow=False - ), - engine, - ) - - def test_the_override_lets_it_through(self): - self.assertIsNone( - vcf_rdfizer.cottas_condensed_engine_rejection( - engines=["cottas"], sample_representation="condensed", allow=True - ) - ) - - -class ResolutionTests(VerboseTestCase): - """Naming cottas is a request; asking for 'all' is a request for breadth.""" - - def test_an_explicit_request_is_refused_not_silently_dropped(self): - engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( - engines=["qlever", "cottas"], - sample_representation="condensed", - requested_all=False, - allow=False, - ) - self.assertIsNotNone(refusal) - self.assertIsNone(warning) - self.assertEqual(engines, ["qlever", "cottas"]) - - def test_engine_all_drops_cottas_and_runs_the_rest(self): - """06_equivalence used 'all' and hung twice; this completes instead.""" - engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( - engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), - sample_representation="condensed", - requested_all=True, - allow=False, - ) - self.assertIsNone(refusal) - self.assertIsNotNone(warning) - self.assertNotIn("cottas", engines) - self.assertEqual(engines, ["comunica", "qlever", "hdt"]) - - def test_the_drop_warning_says_what_still_ran_and_how_to_override(self): - _engines, warning, _refusal = vcf_rdfizer.cottas_condensed_resolution( - engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), - sample_representation="condensed", - requested_all=True, - allow=False, - ) - self.assertIn("comunica", warning) - self.assertIn("qlever", warning) - self.assertIn("--allow-cottas-condensed", warning) - - def test_engine_all_against_expanded_keeps_every_engine(self): - engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( - engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), - sample_representation="expanded", - requested_all=True, - allow=False, - ) - self.assertIsNone(refusal) - self.assertIsNone(warning) - self.assertEqual(engines, list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES)) - - def test_the_override_keeps_cottas_under_all(self): - engines, warning, refusal = vcf_rdfizer.cottas_condensed_resolution( - engines=list(vcf_rdfizer.VALIDATION_ENGINE_CHOICES), - sample_representation="condensed", - requested_all=True, - allow=True, - ) - self.assertIsNone(refusal) - self.assertIsNone(warning) - self.assertIn("cottas", engines) - - -if __name__ == "__main__": - unittest.main() diff --git a/vcf_rdfizer.py b/vcf_rdfizer.py index 57aa61f..cb45f3d 100644 --- a/vcf_rdfizer.py +++ b/vcf_rdfizer.py @@ -5155,87 +5155,6 @@ def hdt_strategy_rejection( return None -def cottas_condensed_resolution( - *, - engines: list[str], - sample_representation: str, - requested_all: bool, - allow: bool, -) -> tuple[list[str], str | None, str | None]: - """Resolve the cottas x condensed combination. - - Returns ``(engines, warning, refusal)``. Exactly one of warning/refusal is - ever set, and the distinction is deliberate: - - * naming ``cottas`` explicitly is a request the tool cannot honour, so it - is refused rather than silently ignored; but - * ``--validation-engine all`` is a request for breadth, and dropping the - one engine that cannot finish serves that better than failing the run. - The campaign's 06_equivalence used ``all`` and hung twice; under this it - would have completed on the other three engines and said why. - """ - refusal = cottas_condensed_engine_rejection( - engines=engines, sample_representation=sample_representation, allow=allow - ) - if refusal is None: - return engines, None, None - if not requested_all: - return engines, None, refusal - remaining = [engine for engine in engines if engine != "cottas"] - return ( - remaining, - "dropped the cottas engine: it does not terminate on the condensed " - "genotype encoding (two archived attempts ran 41 h and ~10 h on " - "q05_sample_genotype_counts). Validating with " - f"{', '.join(remaining)}. Pass --allow-cottas-condensed to include it.", - None, - ) - - -def cottas_condensed_engine_rejection( - *, - engines: list[str], - sample_representation: str, - allow: bool, -) -> str | None: - """Return why native COTTAS cannot query a condensed graph, or None. - - This is a known non-termination, reproduced twice and archived. Native - pycottas answered the fourteen preflight queries and Q1-Q4 against the - condensed encoding of a 10,000-record fixture, then failed to complete - q05_sample_genotype_counts: 41 hours in the first attempt, about 10 in the - second, both at roughly 190% CPU. QLever answered all thirteen questions on - that same cell, and the sibling expanded encoding finished every engine in - 185-194 minutes -- so the defect is specific to - native pycottas x condensed genotype encoding x a genotype-level query. - - The condensed encoding stores genotypes as S + (V x F) rather than V x S, so - a per-sample genotype query has to join across the sample block; the working - hypothesis is a missing or unusable index on that join column rather than - data volume, at a scale QLever answers in about a second. - - Refusing it is the interim. The alternative -- letting it start -- is what - cost 51 hours of unattended runtime, and the per-query timeout could not - bound it at the time. Once that is fixed and the hang is diagnosed, this - goes away; until then a clear refusal costs a user nothing they could - otherwise get. - """ - if allow or sample_representation != "condensed": - return None - if "cottas" not in engines: - return None - return ( - "--validation-engine cottas does not terminate on the condensed " - "genotype encoding. Reproduced twice on a 10,000-record fixture: native " - "pycottas answered 18 of 27 queries and then ran 41 h, and on a second " - "attempt about 10 h, on q05_sample_genotype_counts without completing.\n" - " Use --validation-engine qlever, which answered all thirteen " - "questions on that same cell, or --sample-representation expanded, " - "where every engine including cottas completes. Pass " - "--allow-cottas-condensed to attempt it anyway." - ) - - # --------------------------------------------------------------------------- # Run metrics layout: naming, manifest, and summary # --------------------------------------------------------------------------- @@ -9463,16 +9382,6 @@ def main(): "not scale to a cohort-sized aggregate" ), ) - parser.add_argument( - "--allow-cottas-condensed", - action="store_true", - help=( - "Attempt --validation-engine cottas against a condensed graph " - "anyway. Refused by default: native pycottas did not terminate on " - "q05_sample_genotype_counts there in two archived attempts (41 h, " - "then ~10 h) while qlever answered the same cell in about a second" - ), - ) parser.add_argument( "--shacl-profile", choices=SHACL_PROFILE_CHOICES, @@ -9701,18 +9610,6 @@ def main(): raise ValueError("--qlever-port must be between 1 and 65535") validation_engines = parse_validation_engines(args.validation_engine) validation_artifacts = parse_validation_targets(args.validate_artifacts) - validation_engines, cottas_warning, cottas_rejection = ( - cottas_condensed_resolution( - engines=validation_engines, - sample_representation=args.sample_representation, - requested_all=(args.validation_engine or "").strip() == "all", - allow=args.allow_cottas_condensed, - ) - ) - if cottas_rejection is not None: - raise ValueError(cottas_rejection) - if cottas_warning is not None: - eprint(f"Warning: {cottas_warning}") if args.shacl_shapes is not None: # Comma-separated: the published profile is split across files and # only some of them catch a corrupted value.