Repository navigation
Expand file tree
/
Copy pathvalidation_fixtures.py
More file actions
756 lines (672 loc) · 30.7 KB
/
Copy pathvalidation_fixtures.py
File metadata and controls
756 lines (672 loc) · 30.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
"""One declarative fixture, three derived artifacts.
The mutation harness needs a VCF, the RDF graph that VCF should convert to, and
the parser summary the validator's oracle should compute from it. Deriving all
three from a single specification below is what keeps them honest: a fixture
change cannot make the graph and the oracle disagree by accident.
``test_validation_container_unit.py`` closes the loop in CI by running the real
``parse_vcf`` over :func:`write_vcf` output and asserting it equals
:func:`parser_summary`, so the hand-derived oracle is pinned to the real one.
The records are chosen to exercise every branch the validation queries have:
a transition SNV, a transversion SNV, a deletion shape, a multi-allelic site,
a no-ALT site, PASS and non-PASS FILTER values, a missing genotype, a
non-diploid genotype, and a non-GT FORMAT field that nothing currently checks.
"""
from __future__ import annotations
from collections import Counter
from dataclasses import dataclass, field
from pathlib import Path
VCFC = "https://w3id.org/vcf-core/vocab#"
RDF_TYPE = "http://www.w3.org/1999/02/22-rdf-syntax-ns#type"
XSD_INTEGER = "http://www.w3.org/2001/XMLSchema#integer"
def _runner():
"""Load the shipped validation runner (it lives outside the package path)."""
import importlib.util
path = Path(__file__).resolve().parents[1] / "src" / "validation" / "validation_runner.py"
spec = importlib.util.spec_from_file_location("validation_runner_fixture", path)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
spec.loader.exec_module(module)
return module
SOURCE_FILE = "fixture.vcf"
SAMPLES = ("HG001", "HG002")
FILE_FORMAT = "VCFv4.2"
FILE_DATE = "20260101"
SOURCE_SOFTWARE = "vcf-rdfizer-fixture"
REFERENCE_GENOME = "GRCh38"
#: The optional vcfc subclass the mapping's version sentinel resolves to for
#: this fixture's declared version. Derived rather than hard-coded so changing
#: FILE_FORMAT above changes the expected graph with it.
VERSION_FILE_CLASS = f"VCF{FILE_FORMAT.removeprefix('VCFv').replace('.', '')}File"
#: The bare version, as vcf_rdfizer_vocab.VCF_VERSIONS keys it.
FIXTURE_VERSION = FILE_FORMAT.removeprefix("VCFv")
#: ``##`` meta-information lines, in file order. The ``#CHROM`` line is not one
#: of these; ``header_count`` below counts only the ``##`` lines.
HEADER_LINES: tuple[tuple[str, str], ...] = (
("fileformat", FILE_FORMAT),
("fileDate", FILE_DATE),
("source", SOURCE_SOFTWARE),
("reference", REFERENCE_GENOME),
("FILTER", '<ID=q10,Description="Quality below 10">'),
("INFO", '<ID=AC,Number=A,Type=Integer,Description="Allele count">'),
("INFO", '<ID=DB,Number=0,Type=Flag,Description="dbSNP membership">'),
("FORMAT", '<ID=GT,Number=1,Type=String,Description="Genotype">'),
("FORMAT", '<ID=DP,Number=1,Type=Integer,Description="Read depth">'),
# Exercise the structured header paths: a contig with every optional
# attribute, one with none, and a symbolic ALT declaration.
("contig", "<ID=20,length=64444167,md5=b0a5f3e1,assembly=GRCh38>"),
("contig", "<ID=21>"),
("ALT", '<ID=DEL,Description="Deletion relative to the reference">'),
# An unrecognized key: it must keep only the base HeaderLine type.
("phasing", "partial"),
)
@dataclass(frozen=True)
class FixtureRecord:
"""One VCF data line and everything derived from it."""
row_id: str
chrom: str
pos: int
record_id: str
ref: str
alt: str
qual: str
filter_value: str
info: str
format_keys: tuple[str, ...]
#: One ``:``-joined payload per sample, aligned to ``format_keys``.
sample_payloads: tuple[str, ...] = field(default_factory=tuple)
RECORDS: tuple[FixtureRecord, ...] = (
# Transition SNV, PASS, both samples called.
FixtureRecord("1", "20", 100, "rs100", "A", "G", "50", "PASS", "AC=1;DB",
("GT", "DP"), ("0|1:30", "0/0:28")),
# Transversion SNV in the same 1 Mb window as record 1: a POS swap between
# these two is invisible to every aggregate query.
FixtureRecord("2", "20", 300, ".", "A", "C", "99", "PASS", "AC=2",
("GT", "DP"), ("1/1:41", "0/1:35")),
# Deletion shape in a different window, failing FILTER, missing genotype.
FixtureRecord("3", "20", 2000200, "rs200", "AAT", "A", "12.5", "q10", "AC=1",
("GT", "DP"), ("0/1:15", "./.:0")),
# Multi-allelic: excluded from q06, classified MULTIALLELIC by q02.
FixtureRecord("4", "21", 500, ".", "C", "T,G", ".", "PASS", "AC=1,1",
("GT", "DP"), ("1/2:22", "0/1:19")),
# No ALT, haploid genotype: exercises NO_ALT and HAPLOID_* branches.
FixtureRecord("5", "21", 900, ".", "G", ".", "7", "q10", "DB",
("GT", "DP"), ("0:11", "1:9")),
# A second transition SNV sharing record 1's contig and 1 Mb window. A
# REF/ALT permutation between records 1 and 6 leaves every aggregate
# identical, which is the blind spot the record digest is meant to close.
FixtureRecord("6", "20", 800, ".", "C", "T", "60", "PASS", "AC=1",
("GT", "DP"), ("0/1:33", "0/1:31")),
)
TRANSITIONS = {("A", "G"), ("G", "A"), ("C", "T"), ("T", "C")}
# ---------------------------------------------------------------------------
# VCF
# ---------------------------------------------------------------------------
def vcf_text() -> str:
lines = [f"##{key}={value}" for key, value in HEADER_LINES]
lines.append(
"#" + "\t".join(
["CHROM", "POS", "ID", "REF", "ALT", "QUAL", "FILTER", "INFO", "FORMAT", *SAMPLES]
)
)
for record in RECORDS:
lines.append("\t".join([
record.chrom, str(record.pos), record.record_id, record.ref, record.alt,
record.qual, record.filter_value, record.info,
":".join(record.format_keys), *record.sample_payloads,
]))
return "\n".join(lines) + "\n"
def write_vcf(path: Path) -> Path:
path.write_text(vcf_text(), encoding="utf-8")
return path
def records_tsv_text() -> str:
"""The `vcf_as_tsv.sh` output for this fixture, used by the RDF emitters."""
header = ["SOURCE_FILE", "ROW_ID", "CHROM", "POS", "ID", "REF", "ALT", "QUAL",
"FILTER", "INFO", "FORMAT", " ".join(SAMPLES)]
rows = ["\t".join(header)]
for record in RECORDS:
rows.append("\t".join([
SOURCE_FILE, record.row_id, record.chrom, str(record.pos), record.record_id,
record.ref, record.alt, record.qual, record.filter_value, record.info,
":".join(record.format_keys), " ".join(record.sample_payloads),
]))
return "\n".join(rows) + "\n"
def header_lines_tsv_text() -> str:
rows = ["\t".join(["SOURCE_FILE", "HEADER_INDEX", "HEADER_KEY", "HEADER_VALUE", "RAW_LINE"])]
for index, (key, value) in enumerate(HEADER_LINES, start=1):
rows.append("\t".join([SOURCE_FILE, str(index), key, value, f"{key}={value}"]))
return "\n".join(rows) + "\n"
# ---------------------------------------------------------------------------
# RDF graph
# ---------------------------------------------------------------------------
def _literal(value: str) -> str:
escaped = value.replace("\\", "\\\\").replace('"', '\\"')
if value == ".":
return f'"{escaped}"^^<{VCFC}Null>'
return f'"{escaped}"'
def _info_entries(record: "FixtureRecord") -> list[tuple[str, str | None]]:
"""Split an INFO column into (key, value) pairs; a bare key is a Flag."""
entries: list[tuple[str, str | None]] = []
if record.info in ("", "."):
return entries
for item in record.info.split(";"):
if "=" in item:
key, value = item.split("=", 1)
entries.append((key, value))
else:
entries.append((item, None))
return entries
def base_triples(*, include_qual: bool = True, include_info: bool = True) -> list[str]:
"""The triples RMLStreamer produces from `default_rules.ttl`.
This has to track the shipped mapping, and the division it follows is one
rule: RML carries every field whose RDF datatype is the same for every row.
ID, ALT, QUAL, FILTER and INFO are therefore *not* here -- each may be the
VCF missing token, which the vocabulary requires as ``"."^^vcfc:Null``, so
the wrapper's own emitter produces them and ``build_graph`` appends them.
``include_qual`` and ``include_info`` are kept because the harness uses them
to model a graph missing those emitter outputs, which is how it shows that
dropping them is detected.
"""
file_uri = f"file://{SOURCE_FILE}"
header_uri = f"{file_uri}#header"
columns_uri = f"{file_uri}#header/columns"
out = [
f"<{file_uri}> <{RDF_TYPE}> <{VCFC}VCFFile> .",
# The mapping's version sentinel, resolved per input to the subclass for
# the version the file declares. The fixture declares VCFv4.2.
f"<{file_uri}> <{RDF_TYPE}> <{VCFC}{VERSION_FILE_CLASS}> .",
f"<{file_uri}> <{VCFC}fileFormat> {_literal(FILE_FORMAT)} .",
f"<{file_uri}> <{VCFC}sourceSoftware> {_literal(SOURCE_SOFTWARE)} .",
f"<{file_uri}> <{VCFC}referenceGenome> {_literal(REFERENCE_GENOME)} .",
f"<{file_uri}> <{VCFC}hasHeader> <{header_uri}> .",
f"<{header_uri}> <{RDF_TYPE}> <{VCFC}VCFHeader> .",
# The mandatory #CHROM line. Its ordered sample columns are attached by
# the wrapper, because they live in the records TSV header.
f"<{header_uri}> <{VCFC}hasColumnHeader> <{columns_uri}> .",
f"<{columns_uri}> <{RDF_TYPE}> <{VCFC}ColumnHeaderLine> .",
]
for index, (key, value) in enumerate(HEADER_LINES, start=1):
line_uri = f"{file_uri}#header/line/{index}"
out += [
f"<{header_uri}> <{VCFC}hasHeaderLine> <{line_uri}> .",
f"<{line_uri}> <{RDF_TYPE}> <{VCFC}HeaderLine> .",
f"<{line_uri}> <{VCFC}headerKey> {_literal(key)} .",
f"<{line_uri}> <{VCFC}headerValue> {_literal(value)} .",
f'<{line_uri}> <{VCFC}lineIndex> "{index}"^^<{XSD_INTEGER}> .',
]
for record in RECORDS:
record_uri = f"{file_uri}#record/{record.row_id}"
call_uri = f"{file_uri}#call/{record.row_id}"
out += [
f"<{file_uri}> <{VCFC}hasRecord> <{record_uri}> .",
f"<{record_uri}> <{RDF_TYPE}> <{VCFC}VCFRecord> .",
f"<{record_uri}> <{VCFC}chrom> {_literal(record.chrom)} .",
f'<{record_uri}> <{VCFC}pos> "{record.pos}"^^<{XSD_INTEGER}> .',
f"<{record_uri}> <{VCFC}ref> {_literal(record.ref)} .",
f'<{record_uri}> <{VCFC}recordIndex> "{record.row_id}"^^<{XSD_INTEGER}> .',
f"<{record_uri}> <{VCFC}hasCall> <{call_uri}> .",
f"<{call_uri}> <{RDF_TYPE}> <{VCFC}VariantCall> .",
f"<{call_uri}> <{VCFC}formatRaw> {_literal(':'.join(record.format_keys))} .",
]
return out
def build_graph(
representation: str = "expanded", *, include_qual: bool = True,
include_info: bool = True, include_headers: bool = True,
) -> str:
"""Return the N-Triples graph for this fixture, sample triples included.
The sample triples come from the project's own emitters rather than being
written out here, so the fixture tracks the real implementation.
"""
import tempfile
import vcf_rdfizer
if representation not in {"expanded", "condensed"}:
raise ValueError(f"unknown representation: {representation}")
with tempfile.TemporaryDirectory() as td:
tmp_path = Path(td)
records_tsv = tmp_path / f"{SOURCE_FILE}.records.tsv"
records_tsv.write_text(records_tsv_text(), encoding="utf-8")
headers_tsv = tmp_path / f"{SOURCE_FILE}.header_lines.tsv"
headers_tsv.write_text(header_lines_tsv_text(), encoding="utf-8")
graph = tmp_path / "graph.nt"
graph.write_text(
"\n".join(base_triples(include_qual=include_qual, include_info=include_info)) + "\n",
encoding="utf-8",
)
# The fixture declares a VCF version; the emitters follow it exactly as
# a real conversion does, so version-dependent behaviour is exercised
# rather than assumed.
version = vcf_rdfizer.vocab.VCF_VERSIONS[FIXTURE_VERSION]
vcf_rdfizer.append_record_detail_rdf(
records_tsv, headers_tsv, graph,
emit_qual=include_qual, emit_info=include_info,
# Same disjunction the wrapper applies: the expanded sample layer
# joins to the alleles through vcfc:calledAllele, so raw INFO alone
# does not mean the allele layer can be left out.
emit_alleles=vcf_rdfizer.allele_layer_required(
"structured" if include_info else "raw", representation
),
version=version,
progress_interval_records=0,
)
if representation == "expanded":
vcf_rdfizer.append_expanded_sample_rdf(
records_tsv, graph, headers_tsv, version=version,
progress_interval_records=0,
)
else:
vcf_rdfizer.append_condensed_sample_rdf(
records_tsv, headers_tsv, graph, progress_interval_records=0
)
if include_headers:
vcf_rdfizer.append_header_representation_rdf(headers_tsv, graph)
return graph.read_text(encoding="utf-8")
# ---------------------------------------------------------------------------
# Parser oracle
# ---------------------------------------------------------------------------
def _classify_shape(ref: str, alt: str) -> str:
import re
ref, alt = ref.upper(), alt.upper()
if alt == ".":
return "NO_ALT"
if "," in alt:
return "MULTIALLELIC"
if alt == "*" or "[" in alt or "]" in alt or (alt.startswith("<") and alt.endswith(">")):
return "SYMBOLIC_OR_BREAKEND"
if not re.fullmatch(r"[ACGTN]+", ref) or not re.fullmatch(r"[ACGTN]+", alt):
return "OTHER"
if len(ref) == len(alt) == 1:
return "SNV"
if len(ref) == len(alt):
return "MNV_OR_EQUAL_LENGTH_SUBSTITUTION"
return "INSERTION_SHAPE" if len(ref) < len(alt) else "DELETION_SHAPE"
def _classify_genotype(raw: str) -> str:
normalized = raw.replace("|", "/")
if "." in normalized:
return "MISSING"
alleles = normalized.split("/")
if len(alleles) == 1:
return "HAPLOID_REF" if alleles[0] == "0" else "HAPLOID_ALT"
if len(alleles) == 2:
if alleles[0] == alleles[1]:
return "HOM_REF" if alleles[0] == "0" else "HOM_ALT"
return "HET"
return "OTHER_PLOIDY"
def _genotype_of(record: FixtureRecord, sample_index: int) -> str:
payload = record.sample_payloads[sample_index].split(":")
return payload[record.format_keys.index("GT")]
def _format_shape() -> dict:
"""FORMAT counters the census needs, mirroring SampleRecordStream widening."""
records_with_column = records_with_keys = 0
occurrences = slots = non_empty = 0
distinct: set[str] = set()
for record in RECORDS:
if record.format_keys:
records_with_column += 1
payload_fields = [p.split(":") if p else [] for p in record.sample_payloads]
width = max([len(record.format_keys), *(len(f) for f in payload_fields)], default=0)
keys = [
record.format_keys[i] if i < len(record.format_keys) and record.format_keys[i]
else f"FIELD_{i + 1}"
for i in range(width)
]
if SAMPLES and keys:
records_with_keys += 1
distinct.update(keys)
occurrences += width
slots += width * len(SAMPLES)
non_empty += sum(
1 for f in payload_fields for i in range(width)
if i < len(f) and f[i] != ""
)
return {
"recordsWithFormatColumn": records_with_column,
"recordsWithFormatKeys": records_with_keys,
"formatKeyOccurrences": occurrences,
"formatValueSlots": slots,
"nonEmptyFormatValues": non_empty,
"distinctFormatKeyCount": len(distinct),
}
def _declared_numbers(header_key: str) -> dict[str, str]:
"""Map each declared INFO or FORMAT ID to its Number token."""
runner = _runner()
numbers: dict[str, str] = {}
for key, value in HEADER_LINES:
if key.upper() != header_key:
continue
fields = runner.parse_structured_header_fields(value)
ident = (fields.get("ID") or "").strip()
if ident:
numbers.setdefault(ident, fields.get("Number") or ".")
return numbers
def _declared_ids(header_key: str) -> set[str]:
"""The set of IDs declared by one structured header key."""
runner = _runner()
ids: set[str] = set()
for key, value in HEADER_LINES:
if key.upper() != header_key:
continue
ident = (runner.parse_structured_header_fields(value).get("ID") or "").strip()
if ident:
ids.add(ident)
return ids
def _record_rows() -> list[list[str]]:
"""The raw VCF data columns, as the runner's oracle sees them."""
return [
[record.chrom, str(record.pos), record.record_id, record.ref, record.alt,
record.qual, record.filter_value, record.info,
":".join(record.format_keys), *record.sample_payloads]
for record in RECORDS
]
def _header_shape() -> dict:
"""Header counters, derived by the shipped runner rather than restated.
The census asserts an exact inventory, so this has to cover every family the
header emitter produces. Delegating keeps the fixture and the container
oracle on one implementation.
"""
runner = _runner()
shape = dict(runner.emitted_header_counters(list(HEADER_LINES)))
shape["headerValueCount"] = sum(1 for _key, value in HEADER_LINES if value != "")
return shape
def _synthesized_definition_numbers() -> dict:
"""Delegated to the runner, so the two oracles cannot disagree."""
return dict(_runner().synthesized_definition_numbers(
_record_rows(),
declared_info=set(_declared_numbers("INFO")),
declared_format=set(_declared_numbers("FORMAT")),
))
def _record_shape() -> dict:
"""Allele, value-item, SV and genotype counters, derived the same way."""
runner = _runner()
return dict(runner.emitted_record_counters(
_record_rows(),
list(SAMPLES),
version=runner.vocab.VCF_VERSIONS[FIXTURE_VERSION],
contig_ids=_declared_ids("CONTIG"),
alt_declaration_ids=_declared_ids("ALT"),
has_assembly_line=any(
key.lower() == "assembly" and value for key, value in HEADER_LINES
),
info_numbers=_declared_numbers("INFO"),
format_numbers=_declared_numbers("FORMAT"),
))
def _declared_types(header_key: str) -> dict[str, str]:
"""Map each declared INFO or FORMAT key to its VCF Type."""
import re as _re
declared: dict[str, str] = {}
for key, value in HEADER_LINES:
if key != header_key:
continue
fields = dict(_re.findall(r'(\w+)=("[^"]*"|[^,>]*)', value))
ident = fields.get("ID", "").strip()
if ident:
declared[ident] = fields.get("Type", "String").strip('"')
return declared
def _format_typed_shape() -> dict:
"""FORMAT cells that gain a typed companion, by the runner's own rule.
The emitter types a single, non-missing Integer/Float FORMAT value exactly
as it types an INFO one, so the census has to expect those triples too.
"""
runner = _runner()
declared = _declared_types("FORMAT")
typed_int = typed_dec = 0
for row in _record_rows():
format_keys = row[8].split(":") if len(row) > 8 and row[8] else []
payload_fields = [payload.split(":") if payload else [] for payload in row[9:]]
width = max(
[len(format_keys), *(len(fields) for fields in payload_fields)], default=0
)
for index in range(width):
key = (format_keys[index]
if index < len(format_keys) and format_keys[index]
else f"FIELD_{index + 1}")
for fields in payload_fields:
cell = fields[index] if index < len(fields) else ""
if not cell:
continue
kind = runner.typed_value_kind(cell, declared.get(key, "String"))
if kind == "Integer":
typed_int += 1
elif kind == "Float":
typed_dec += 1
return {
"formatTypedIntegerCount": typed_int,
"formatTypedDecimalCount": typed_dec,
}
def _record_decomposition_shape() -> dict:
"""ID and FILTER components the emitter turns into ordered resources.
Derived from the fixture's own columns, mirroring the rule rather than
calling the emitter: PASS and the missing token are FILTER statuses, so
neither contributes a failure code.
"""
declared_filters = set(_declared_ids("FILTER"))
identifiers = codes = declared_codes = 0
for row in _record_rows():
record_id = row[2] if len(row) > 2 else "."
if record_id != ".":
identifiers += sum(1 for part in record_id.split(";") if part)
filter_column = row[6] if len(row) > 6 else "."
if filter_column not in ("PASS", "."):
for code in (part for part in filter_column.split(";") if part):
codes += 1
if code in declared_filters:
declared_codes += 1
return {
"recordIdentifierCount": identifiers,
"filterCodeCount": codes,
"declaredFilterCodeCount": declared_codes,
}
def _info_shape() -> dict:
"""INFO counters the census needs, mirroring the emitter's typing rules."""
declared = {}
for key, value in HEADER_LINES:
if key != "INFO":
continue
import re as _re
fields = dict(_re.findall(r'(\w+)=("[^"]*"|[^,>]*)', value))
ident = fields.get("ID", "").strip()
if ident:
declared[ident] = fields.get("Type", "String").strip('"')
values = flags = typed_int = typed_dec = 0
keys: set[str] = set()
for record in RECORDS:
for key, value in _info_entries(record):
values += 1
keys.add(key)
if value is None:
flags += 1
continue
if "," in value or value == ".":
continue
kind = declared.get(key, "String")
try:
if kind == "Integer":
int(value); typed_int += 1
elif kind == "Float":
float(value); typed_dec += 1
except ValueError:
pass
return {
"infoValueCount": values,
"infoDefinitionCount": len(keys),
"infoFlagCount": flags,
"infoTypedIntegerCount": typed_int,
"infoTypedDecimalCount": typed_dec,
}
def parser_summary(
representation: str = "expanded", *, include_qual: bool = True,
include_info: bool = True, include_headers: bool = True,
) -> dict:
"""The summary ``parse_vcf`` must produce for this fixture.
``representation`` selects the census expectation, which depends on which
genotype emitter ran. ``include_qual`` mirrors ``build_graph`` so a graph
built without QUAL is compared against an expectation that also omits it.
"""
import re
density: Counter[tuple[str, int]] = Counter()
shapes: Counter[str] = Counter()
filters: Counter[tuple[str, str]] = Counter()
genotypes: Counter[tuple[str, str]] = Counter()
ac_an: Counter[tuple[int, int]] = Counter()
transitions = transversions = biallelic_snvs = single_alt = q06_eligible = 0
for record in RECORDS:
density[(record.chrom, (record.pos - 1) // 1_000_000)] += 1
shapes[_classify_shape(record.ref, record.alt)] += 1
ref_upper, alt_upper = record.ref.upper(), record.alt.upper()
if (re.fullmatch(r"[ACGT]", ref_upper) and re.fullmatch(r"[ACGT]", alt_upper)
and ref_upper != alt_upper):
biallelic_snvs += 1
if (ref_upper, alt_upper) in TRANSITIONS:
transitions += 1
else:
transversions += 1
status = ("PASS" if record.filter_value == "PASS"
else "NOT_APPLIED" if record.filter_value == "." else "FAILED")
filters[(status, record.filter_value)] += 1
for index, sample in enumerate(SAMPLES):
genotypes[(sample, _classify_genotype(_genotype_of(record, index)))] += 1
if record.alt != "." and "," not in record.alt:
single_alt += 1
an = ac = 0
for index in range(len(SAMPLES)):
alleles = _genotype_of(record, index).replace("|", "/").split("/")
if any(a == "." for a in alleles):
continue
if len(alleles) not in (1, 2) or any(a not in ("0", "1") for a in alleles):
continue
an += len(alleles)
ac += sum(int(a) for a in alleles)
if an:
ac_an[(an, ac)] += 1
q06_eligible += 1
summary = {
"sampleCount": len(SAMPLES),
"samples": list(SAMPLES),
"totalRecords": len(RECORDS),
"gtRecordCount": sum(1 for record in RECORDS if "GT" in record.format_keys),
"singleAltRecordCount": single_alt,
"q06EligibleSiteCount": q06_eligible,
"headerLineCount": len(HEADER_LINES),
**_format_shape(),
**_format_typed_shape(),
**_record_decomposition_shape(),
**_info_shape(),
"fileFormat": FILE_FORMAT,
"referenceGenome": REFERENCE_GENOME,
"sourceSoftware": SOURCE_SOFTWARE,
"fileDate": FILE_DATE,
**_header_shape(),
**_record_shape(),
**_synthesized_definition_numbers(),
"q07_file_metadata": {
"fileFormat": FILE_FORMAT,
"referenceGenome": REFERENCE_GENOME,
"sourceSoftware": SOURCE_SOFTWARE,
},
"q08_header_line_census": [
{"headerKey": key, "lineCount": count}
for key, count in sorted(Counter(key for key, _ in HEADER_LINES).items())
],
"q01_record_density_1mb": [
{"chrom": chrom, "windowIndex": window, "recordCount": count}
for (chrom, window), count in sorted(density.items())
],
"q02_variant_shape_counts": [
{"variantClass": name, "recordCount": count} for name, count in sorted(shapes.items())
],
"q03_titv": {
"biallelicSnvCount": biallelic_snvs,
"transitionCount": transitions,
"transversionCount": transversions,
"tiTvRatio": transitions / transversions if transversions else None,
},
"q04_filter_distribution": [
{"filterStatus": status, "filterLexical": lexical, "recordCount": count}
for (status, lexical), count in sorted(filters.items())
],
"q05_sample_genotype_counts": [
{"sampleId": sample, "genotypeClass": genotype_class, "callCount": count}
for (sample, genotype_class), count in sorted(genotypes.items())
],
"q06_ac_an_distribution": [
{"an": an, "ac": ac, "siteCount": count, "af": ac / an}
for (an, ac), count in sorted(ac_an.items())
],
}
# The census and digest expectations come from the shipped derivations
# rather than hand-written copies, so the fixture cannot drift from them.
runner = _runner()
summary.update(runner.expected_census(
summary, representation,
info_representation="structured" if include_info else "raw",
header_representation="structured" if include_headers else "basic",
))
source_component = runner.rml_uri_component(SOURCE_FILE)
digest: Counter[str] = Counter()
for record in RECORDS:
record_iri = (
f"file://{source_component}#record/"
f"{runner.rml_uri_component(record.row_id)}"
)
digest[runner.record_digest_bucket([
record_iri, record.chrom, str(record.pos), record.record_id,
record.ref, record.alt,
runner.digest_qual(record.qual) if include_qual else "",
record.filter_value, record.info,
])] += 1
summary["q11_record_digest"] = [
{"bucket": bucket, "recordCount": count} for bucket, count in sorted(digest.items())
]
# Value-level digests, derived the same way the runner derives them.
sample_components = [
runner.rml_uri_component(uri)
for uri in runner.sample_uri_ids(list(SAMPLES))
]
info_digest: Counter[str] = Counter()
format_digest: Counter[str] = Counter()
for record in RECORDS:
row_component = runner.rml_uri_component(record.row_id)
call_iri = f"file://{source_component}#call/{row_component}"
if include_info:
for key, value in _info_entries(record):
if value is None:
continue
info_iri = f"{call_iri}/info/{runner.rml_uri_component(key)}"
info_digest[runner.record_digest_bucket([info_iri, value])] += 1
payload_fields = [p.split(":") if p else [] for p in record.sample_payloads]
width = max([len(record.format_keys), *(len(f) for f in payload_fields)], default=0)
keys = [
record.format_keys[i] if i < len(record.format_keys) and record.format_keys[i]
else f"FIELD_{i + 1}"
for i in range(width)
]
for key_index, key in enumerate(keys):
key_component = runner.rml_uri_component(key)
if representation == "expanded":
for sample_index, fields in enumerate(payload_fields):
cell = fields[key_index] if key_index < len(fields) else ""
if not cell:
continue
value_iri = (
f"file://{source_component}#sample/{row_component}"
f"/{sample_components[sample_index]}/fmt/{key_component}"
)
format_digest[runner.record_digest_bucket([value_iri, cell])] += 1
else:
encoded = "\t".join(
(fields[key_index] if key_index < len(fields) and fields[key_index]
else ".")
for fields in payload_fields
)
vector_iri = f"{call_iri}/matrix/fmt/{key_component}"
format_digest[runner.record_digest_bucket([vector_iri, encoded])] += 1
summary["q12_info_value_digest"] = [
{"bucket": b, "valueCount": c} for b, c in sorted(info_digest.items())
]
summary["q13_format_value_digest"] = [
{"bucket": b, "valueCount": c} for b, c in sorted(format_digest.items())
]
if not include_qual:
summary["q09_predicate_census"] = [
row for row in summary["q09_predicate_census"]
if not row["predicate"].endswith("#qual")
]
return summary