From 68f55e9353359dd684653740baf361c956529482 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:15:25 +0200 Subject: [PATCH 01/29] feat(duckdb): scaffold adapter package and identifier value objects Co-Authored-By: Claude Haiku 4.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- adapters/duckdb/README.md | 6 ++ .../continuo_duckdb_adapter/__init__.py | 1 + .../domain/__init__.py | 1 + .../domain/identifiers.py | 33 +++++++++++ adapters/duckdb/pyproject.toml | 56 ++++++++++++++++++ adapters/duckdb/tests/__init__.py | 0 .../tests/test_domain_duckdb_identifiers.py | 26 +++++++++ pyproject.toml | 2 +- uv.lock | 57 +++++++++++++++++++ 9 files changed, 181 insertions(+), 1 deletion(-) create mode 100644 adapters/duckdb/README.md create mode 100644 adapters/duckdb/continuo_duckdb_adapter/__init__.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/domain/__init__.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/domain/identifiers.py create mode 100644 adapters/duckdb/pyproject.toml create mode 100644 adapters/duckdb/tests/__init__.py create mode 100644 adapters/duckdb/tests/test_domain_duckdb_identifiers.py diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md new file mode 100644 index 0000000..214ecd8 --- /dev/null +++ b/adapters/duckdb/README.md @@ -0,0 +1,6 @@ +# continuo-duckdb-adapter + +DuckDB warehouse adapter for Continuo, running on a DuckLake: the catalog lives +in Postgres and table data is Parquet on S3/MinIO. Implements +`continuo_engine_contract.port.WarehouseAdapter`; registered under the +`continuo_engine.adapters` entry-point group as `duckdb`. diff --git a/adapters/duckdb/continuo_duckdb_adapter/__init__.py b/adapters/duckdb/continuo_duckdb_adapter/__init__.py new file mode 100644 index 0000000..b0bfe5d --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/__init__.py @@ -0,0 +1 @@ +"""DuckDB (DuckLake) warehouse adapter for Continuo.""" diff --git a/adapters/duckdb/continuo_duckdb_adapter/domain/__init__.py b/adapters/duckdb/continuo_duckdb_adapter/domain/__init__.py new file mode 100644 index 0000000..ea31902 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/domain/__init__.py @@ -0,0 +1 @@ +"""Pure domain rules: no duckdb, psycopg2, pyarrow or boto imports.""" diff --git a/adapters/duckdb/continuo_duckdb_adapter/domain/identifiers.py b/adapters/duckdb/continuo_duckdb_adapter/domain/identifiers.py new file mode 100644 index 0000000..9234213 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/domain/identifiers.py @@ -0,0 +1,33 @@ +"""Identifier value objects. + +They validate and carry names. Turning a name into quoted SQL is the job of the +infrastructure layer's ``DdlRenderer``, so this module stays engine-neutral. +""" +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class Identifier: + """A schema, table or column name, validated but never quoted.""" + + name: str + + def __post_init__(self) -> None: + if not isinstance(self.name, str) or not self.name: + raise ValueError(f"identifier must be a non-empty string, got {self.name!r}") + if "\x00" in self.name: + raise ValueError("identifier must not contain a NUL character") + + +@dataclass(frozen=True) +class QualifiedTable: + """A table addressed by schema and name.""" + + schema: Identifier + table: Identifier + + @classmethod + def of(cls, schema: str, table: str) -> "QualifiedTable": + return cls(Identifier(schema), Identifier(table)) diff --git a/adapters/duckdb/pyproject.toml b/adapters/duckdb/pyproject.toml new file mode 100644 index 0000000..310f095 --- /dev/null +++ b/adapters/duckdb/pyproject.toml @@ -0,0 +1,56 @@ +[project] +version = "0.1.0" +name = "continuo-duckdb-adapter" +description = "DuckDB (DuckLake) warehouse adapter for Continuo: validation and python-node runtime." +authors = [{ name = "Simone Carolini" }] +maintainers = [{ name = "Simone Carolini" }] +readme = "README.md" +requires-python = ">= 3.14" +dependencies = [ + # Pinned exactly for the same reason as the postgres adapter: the + # [tool.uv.sources] redirect below is uv-only and never reaches built + # metadata, and this adapter subclasses the contract's port directly. + "continuo-engine-contract==0.7.3", + "duckdb==1.5.6", + "pyarrow==25.0.1", +] +classifiers = [ + "Development Status :: 4 - Beta", + "Intended Audience :: Developers", + "Programming Language :: Python :: 3.14", +] + +[dependency-groups] +dev = [ + "ruff==0.16.5", + "mypy==2.3.1", +] +test = [ + "pytest==9.1.1", + "pytest-cov==7.1.0", + # Integration tests read the DuckLake catalog (postgres) and the data + # files (MinIO) directly to prove layout was really applied. + "psycopg2-binary==2.9.12", + "boto3==1.43.85", +] + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +packages = ["continuo_duckdb_adapter"] + +[tool.ruff] +target-version = "py312" + +[tool.ruff.lint] +# Kept in lockstep with the root pyproject's select -- see the note there. +select = ["E4", "E7", "E9", "F"] + +[tool.mypy] +python_version = "3.14" +strict = false + +[tool.uv.sources] +continuo-engine-contract = { workspace = true } diff --git a/adapters/duckdb/tests/__init__.py b/adapters/duckdb/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/adapters/duckdb/tests/test_domain_duckdb_identifiers.py b/adapters/duckdb/tests/test_domain_duckdb_identifiers.py new file mode 100644 index 0000000..44abe5f --- /dev/null +++ b/adapters/duckdb/tests/test_domain_duckdb_identifiers.py @@ -0,0 +1,26 @@ +"""Identifier value objects: validation only; quoting is an infrastructure concern.""" +import pytest + +from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable + + +@pytest.mark.parametrize("name", ["orders", "Order Table", 'we"ird', "50%", "select", "lake", "ünï"]) +def test_identifier_accepts_any_non_empty_text(name): + assert Identifier(name).name == name + + +@pytest.mark.parametrize("bad", ["", None, 7, "a\x00b"]) +def test_identifier_rejects_empty_non_string_and_nul(bad): + with pytest.raises(ValueError): + Identifier(bad) + + +def test_identifiers_are_value_objects(): + assert Identifier("a") == Identifier("a") + assert hash(Identifier("a")) == hash(Identifier("a")) + + +def test_qualified_table_of_builds_both_parts(): + table = QualifiedTable.of("analytics", "orders") + assert table.schema == Identifier("analytics") + assert table.table == Identifier("orders") diff --git a/pyproject.toml b/pyproject.toml index bdcb1cd..9a5efde 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -72,7 +72,7 @@ python_version = "3.14" strict = false [tool.uv.workspace] -members = ["contract", "adapters/postgres", "adapters/trino"] +members = ["contract", "adapters/postgres", "adapters/trino", "adapters/duckdb"] [tool.uv.sources] continuo-engine-contract = { workspace = true } diff --git a/uv.lock b/uv.lock index b2332a8..1328c32 100644 --- a/uv.lock +++ b/uv.lock @@ -8,6 +8,7 @@ resolution-markers = [ [manifest] members = [ + "continuo-duckdb-adapter", "continuo-engine-contract", "continuo-postgres-adapter", "continuo-python-runtime", @@ -159,6 +160,47 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335, upload-time = "2022-10-25T02:36:20.889Z" }, ] +[[package]] +name = "continuo-duckdb-adapter" +version = "0.1.0" +source = { editable = "adapters/duckdb" } +dependencies = [ + { name = "continuo-engine-contract" }, + { name = "duckdb" }, + { name = "pyarrow" }, +] + +[package.dev-dependencies] +dev = [ + { name = "mypy" }, + { name = "ruff" }, +] +test = [ + { name = "boto3" }, + { name = "psycopg2-binary" }, + { name = "pytest" }, + { name = "pytest-cov" }, +] + +[package.metadata] +requires-dist = [ + { name = "continuo-engine-contract", editable = "contract" }, + { name = "duckdb", specifier = "==1.5.6" }, + { name = "pyarrow", specifier = "==25.0.1" }, +] + +[package.metadata.requires-dev] +dev = [ + { name = "mypy", specifier = "==2.3.1" }, + { name = "ruff", specifier = "==0.16.5" }, +] +test = [ + { name = "boto3", specifier = "==1.43.85" }, + { name = "psycopg2-binary", specifier = "==2.9.12" }, + { name = "pytest", specifier = "==9.1.1" }, + { name = "pytest-cov", specifier = "==7.1.0" }, +] + [[package]] name = "continuo-engine-contract" version = "0.7.3" @@ -326,6 +368,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ec/82/32e3bd191d498e64f6f911ad55d14006a0861e54869d2d32452326399e65/coverage-7.15.2-py3-none-any.whl", hash = "sha256:eb6bcae8d1a9d305351ecb108232441d11c5cfe9de840a04388ba5d2db8d735c", size = 213375, upload-time = "2026-07-15T18:56:17.305Z" }, ] +[[package]] +name = "duckdb" +version = "1.5.6" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/59/0b/d65ea3be00ea79aa276a8388bec588a9cbf409ce637c6d306e5316210d15/duckdb-1.5.6.tar.gz", hash = "sha256:166a91dbfacfc0c9f08cc76c0243cb6d3d4296bfab5bad72a3cfb63140a5b7c8", size = 18032957, upload-time = "2026-09-28T13:38:37.978Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/62/a8a30a4c6b94c0861d348ed5633b963f6745a5525527530f02f3c1a7c931/duckdb-1.5.6-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:aa21d2ad803b2524326e8622d7d96b2bb1ff1d5b60368e1978ee805df9c21fb3", size = 32828003, upload-time = "2026-09-28T13:38:21.414Z" }, + { url = "https://files.pythonhosted.org/packages/71/b7/1dcca0005eb8c67adf9fc06bf0cbb1d2bf4ea1974cc89e7a7c2ad66aac28/duckdb-1.5.6-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:8a1b2ad27d414068cbca06c55cfa802eece10f86ea4812ff082f8ab4cb25fc85", size = 17413912, upload-time = "2026-09-28T13:38:23.915Z" }, + { url = "https://files.pythonhosted.org/packages/93/b0/e3ac175443550f3464f2d95731a8b0aae9b4dc3875c3a186c352262b43c2/duckdb-1.5.6-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:c79c6d222b1d015cde73b5139087186b00db65357fb4e2c94c2308fbbf465a72", size = 15543122, upload-time = "2026-09-28T13:38:26.317Z" }, + { url = "https://files.pythonhosted.org/packages/9d/08/cc510a7952aba69d5cdca17f3ef61c95713d86143f2ee9aa3e097d38f50b/duckdb-1.5.6-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1052b8050ef5696e2c0d8c836949c72f3dd11f0690466acbea739613e8e2750b", size = 19457946, upload-time = "2026-09-28T13:38:28.877Z" }, + { url = "https://files.pythonhosted.org/packages/ef/a5/6f8099d9a5a02ddff89e5c85875df3465054845b0920fb0703fbdf8dd2ec/duckdb-1.5.6-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:19c5e485e59613b8878d1670bcaa7a010f53c5a4da5ae8e08863e5e529ca6182", size = 21575132, upload-time = "2026-09-28T13:38:31.231Z" }, + { url = "https://files.pythonhosted.org/packages/9f/58/762f7159662d7859e201fa05ca29f306795daeabf84f3e087215a966b001/duckdb-1.5.6-cp314-cp314-win_amd64.whl", hash = "sha256:ebcbd09cd8578ab1093393e9b16289cda0e8f1791ac595bf00eb5bad75c3cf00", size = 13713963, upload-time = "2026-09-28T13:38:33.543Z" }, + { url = "https://files.pythonhosted.org/packages/46/69/64d165db322de13f5c3e75d377b6b9694df1821155ad1fa4b14b04601abc/duckdb-1.5.6-cp314-cp314-win_arm64.whl", hash = "sha256:820a8384faef11cd86068ea48c5da57ce2d8f1c7b3d2bdb9be3398317a7c3728", size = 14514368, upload-time = "2026-09-28T13:38:35.676Z" }, +] + [[package]] name = "idna" version = "3.18" From 0610e3a7c2e5562c45ad67b2e31377ca2d8be2a6 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:17:58 +0200 Subject: [PATCH 02/29] feat(duckdb): column definitions validated against the contract grammar Co-Authored-By: Claude Haiku 4.5 Signed-off-by: Simone Carolini --- .../continuo_duckdb_adapter/domain/columns.py | 46 ++++++++++++++++ .../tests/test_domain_duckdb_columns.py | 55 +++++++++++++++++++ 2 files changed, 101 insertions(+) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/domain/columns.py create mode 100644 adapters/duckdb/tests/test_domain_duckdb_columns.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/domain/columns.py b/adapters/duckdb/continuo_duckdb_adapter/domain/columns.py new file mode 100644 index 0000000..9e831d6 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/domain/columns.py @@ -0,0 +1,46 @@ +"""Typed column definitions, validated against the contract's SQL type grammar.""" +from __future__ import annotations + +import re +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from typing import Any + +from continuo_engine_contract.types import validate_column_type # type: ignore[import-untyped] + +from .identifiers import Identifier + +_TEMPORAL = re.compile(r"^(TIMESTAMP|DATE)\Z", re.IGNORECASE | re.ASCII) + + +def is_temporal_type(type_str: str) -> bool: + """True for the contract types that support year/month/day/hour partitioning.""" + return _TEMPORAL.match(type_str) is not None + + +@dataclass(frozen=True) +class ColumnDefinition: + """One declared output column. ``type`` is validated on construction because + the text is interpolated into DDL; the name is quoted by the renderer.""" + + name: Identifier + type: str + nullable: bool = True + + def __post_init__(self) -> None: + validate_column_type(self.type) + + @property + def is_temporal(self) -> bool: + return is_temporal_type(self.type) + + @classmethod + def from_mapping(cls, raw: Mapping[str, Any]) -> "ColumnDefinition": + if not isinstance(raw, Mapping) or "name" not in raw or "type" not in raw: + raise ValueError(f"column must be a mapping with 'name' and 'type', got {raw!r}") + return cls(Identifier(raw["name"]), raw["type"], bool(raw.get("nullable", True))) + + +def column_types(columns: Sequence[ColumnDefinition]) -> dict[str, str]: + """Declared column name -> contract type, for layout validation.""" + return {column.name.name: column.type for column in columns} diff --git a/adapters/duckdb/tests/test_domain_duckdb_columns.py b/adapters/duckdb/tests/test_domain_duckdb_columns.py new file mode 100644 index 0000000..934badf --- /dev/null +++ b/adapters/duckdb/tests/test_domain_duckdb_columns.py @@ -0,0 +1,55 @@ +"""ColumnDefinition: contract type grammar, nullability default, temporal detection.""" +import pytest + +from continuo_duckdb_adapter.domain.columns import ( + ColumnDefinition, + column_types, + is_temporal_type, +) +from continuo_duckdb_adapter.domain.identifiers import Identifier + + +@pytest.mark.parametrize( + "type_str", + ["BIGINT", "INT", "INTEGER", "DOUBLE PRECISION", "TEXT", "TIMESTAMP", "DATE", "BOOLEAN", + "NUMERIC(10,2)", "NUMERIC(10, 2)", "DECIMAL(5,0)", "VARCHAR(255)", "CHAR(1)", "bigint"], +) +def test_accepts_the_contract_grammar(type_str): + assert ColumnDefinition(Identifier("c"), type_str).type == type_str + + +@pytest.mark.parametrize( + "bad", ["INTEGER; DROP TABLE x", "JSON", "VARCHAR", "INT\n", "NUMERIC(1,2)", ""] +) +def test_rejects_anything_outside_the_grammar(bad): + with pytest.raises(ValueError): + ColumnDefinition(Identifier("c"), bad) + + +def test_from_mapping_defaults_nullable_to_true(): + col = ColumnDefinition.from_mapping({"name": "id", "type": "INTEGER"}) + assert col == ColumnDefinition(Identifier("id"), "INTEGER", True) + + +def test_from_mapping_honours_nullable_false(): + assert ColumnDefinition.from_mapping({"name": "id", "type": "INT", "nullable": False}).nullable is False + + +@pytest.mark.parametrize("raw", [{}, {"name": "id"}, {"type": "INT"}, "id", None]) +def test_from_mapping_rejects_malformed_entries(raw): + with pytest.raises(ValueError): + ColumnDefinition.from_mapping(raw) + + +@pytest.mark.parametrize("type_str,expected", [ + ("TIMESTAMP", True), ("timestamp", True), ("DATE", True), + ("TEXT", False), ("BIGINT", False), ("VARCHAR(3)", False), +]) +def test_temporal_detection(type_str, expected): + assert is_temporal_type(type_str) is expected + + +def test_column_types_maps_name_to_type(): + cols = [ColumnDefinition.from_mapping({"name": "a", "type": "DATE"}), + ColumnDefinition.from_mapping({"name": "b", "type": "TEXT"})] + assert column_types(cols) == {"a": "DATE", "b": "TEXT"} From 1d760f5ff425f03d8397dfacd3689a59f55a8995 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:20:14 +0200 Subject: [PATCH 03/29] feat(duckdb): TableLayout with fail-closed partitioned_by/sorted_by validation Co-Authored-By: Claude Haiku 4.5 Signed-off-by: Simone Carolini --- .../continuo_duckdb_adapter/domain/layout.py | 154 ++++++++++++++++++ .../duckdb/tests/test_domain_duckdb_layout.py | 116 +++++++++++++ 2 files changed, 270 insertions(+) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/domain/layout.py create mode 100644 adapters/duckdb/tests/test_domain_duckdb_layout.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py b/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py new file mode 100644 index 0000000..09412ff --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py @@ -0,0 +1,154 @@ +"""Physical layout of a DuckLake table: partitioning and sort order. + +This is the duckdb adapter's ``config`` vocabulary. Every key and value is +validated before any DDL exists (fail closed): an unrecognized key is an +authoring error, never silently dropped. Sort keys are declared columns only, +no free-form expressions, so nothing author-written reaches DDL unquoted. +""" +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + +from continuo_engine_contract.config import ensure_known_keys # type: ignore[import-untyped] + +from .columns import is_temporal_type +from .identifiers import Identifier + +ENGINE = "duckdb" +KNOWN_CONFIG_KEYS: tuple[str, ...] = ("partitioned_by", "sorted_by") +_PARTITION_ENTRY_KEYS: tuple[str, ...] = ("column", "transform", "buckets") +_SORT_ENTRY_KEYS: tuple[str, ...] = ("column", "direction", "nulls") +_TRANSFORMS: tuple[str, ...] = ("identity", "bucket", "year", "month", "day", "hour") +_TEMPORAL_TRANSFORMS = frozenset({"year", "month", "day", "hour"}) +_DIRECTIONS: tuple[str, ...] = ("asc", "desc") +_NULL_ORDERS: tuple[str, ...] = ("first", "last") + + +@dataclass(frozen=True) +class PartitionKey: + column: Identifier + transform: str = "identity" + buckets: int | None = None + + +@dataclass(frozen=True) +class SortKey: + column: Identifier + descending: bool = False + nulls_first: bool | None = None # None: the engine's default null ordering + + +@dataclass(frozen=True) +class TableLayout: + partition_keys: tuple[PartitionKey, ...] = () + sort_keys: tuple[SortKey, ...] = () + + @property + def is_empty(self) -> bool: + return not self.partition_keys and not self.sort_keys + + @classmethod + def empty(cls) -> "TableLayout": + return cls() + + @classmethod + def from_config( + cls, config: Mapping[str, Any] | None, column_types: Mapping[str, str | None] + ) -> "TableLayout": + """Validate *config* and build the layout. + + *column_types* maps each declared column to its contract type, or to + ``None`` when only the names are known (the harness's early hook): the + time-transform type check is then deferred to ``ensure_table``. + """ + if config is None: + return cls.empty() + ensure_known_keys(config, KNOWN_CONFIG_KEYS, ENGINE) + partition_keys: tuple[PartitionKey, ...] = () + sort_keys: tuple[SortKey, ...] = () + if "partitioned_by" in config: + partition_keys = _partition_keys(config["partitioned_by"], column_types) + if "sorted_by" in config: + sort_keys = _sort_keys(config["sorted_by"], column_types) + return cls(partition_keys, sort_keys) + + +def _declared_column(value: Any, column_types: Mapping[str, str | None], where: str) -> str: + if not isinstance(value, str) or not value: + raise ValueError(f"{where} 'column' must be a non-empty column name, got {value!r}") + if value not in column_types: + raise ValueError( + f"{where} names undeclared column {value!r}; declared columns: " + f"{sorted(column_types)!r}" + ) + return value + + +def _partition_keys(raw: Any, column_types: Mapping[str, str | None]) -> tuple[PartitionKey, ...]: + if not isinstance(raw, list) or not raw: + raise ValueError(f"config 'partitioned_by' must be a non-empty list, got {raw!r}") + keys: list[PartitionKey] = [] + for entry in raw: + key = _partition_key(entry, column_types) + if key in keys: + raise ValueError(f"duplicate partition key: {entry!r}") + keys.append(key) + return tuple(keys) + + +def _partition_key(entry: Any, column_types: Mapping[str, str | None]) -> PartitionKey: + if isinstance(entry, str): + entry = {"column": entry} + where = "config 'partitioned_by' entry" + ensure_known_keys(entry, _PARTITION_ENTRY_KEYS, ENGINE, where=where) + column = _declared_column(entry.get("column"), column_types, where) + transform = entry.get("transform", "identity") + if transform not in _TRANSFORMS: + raise ValueError( + f"unsupported partition 'transform' {transform!r}; supported: {', '.join(_TRANSFORMS)}" + ) + buckets = entry.get("buckets") + if transform == "bucket": + if isinstance(buckets, bool) or not isinstance(buckets, int) or buckets < 1: + raise ValueError( + f"partition transform 'bucket' requires 'buckets' as a positive integer, got {buckets!r}" + ) + elif buckets is not None: + raise ValueError("'buckets' is only valid with partition transform 'bucket'") + if transform in _TEMPORAL_TRANSFORMS: + column_type = column_types[column] + if column_type is not None and not is_temporal_type(column_type): + raise ValueError( + f"partition transform {transform!r} needs a DATE or TIMESTAMP column, " + f"but {column!r} is {column_type}" + ) + return PartitionKey(Identifier(column), transform, buckets) + + +def _sort_keys(raw: Any, column_types: Mapping[str, str | None]) -> tuple[SortKey, ...]: + if not isinstance(raw, list) or not raw: + raise ValueError(f"config 'sorted_by' must be a non-empty list, got {raw!r}") + keys: list[SortKey] = [] + seen: set[str] = set() + for entry in raw: + if isinstance(entry, str): + entry = {"column": entry} + where = "config 'sorted_by' entry" + ensure_known_keys(entry, _SORT_ENTRY_KEYS, ENGINE, where=where) + column = _declared_column(entry.get("column"), column_types, where) + if column in seen: + raise ValueError(f"duplicate sort column: {column!r}") + seen.add(column) + direction = _choice(entry.get("direction", "asc"), _DIRECTIONS, "direction") + nulls = entry.get("nulls") + nulls_first = None if nulls is None else _choice(nulls, _NULL_ORDERS, "nulls") == "first" + keys.append(SortKey(Identifier(column), direction == "desc", nulls_first)) + return tuple(keys) + + +def _choice(value: Any, allowed: tuple[str, ...], key: str) -> str: + if not isinstance(value, str) or value.lower() not in allowed: + raise ValueError(f"sort '{key}' must be one of {', '.join(allowed)}, got {value!r}") + return value.lower() diff --git a/adapters/duckdb/tests/test_domain_duckdb_layout.py b/adapters/duckdb/tests/test_domain_duckdb_layout.py new file mode 100644 index 0000000..7e5d0f9 --- /dev/null +++ b/adapters/duckdb/tests/test_domain_duckdb_layout.py @@ -0,0 +1,116 @@ +"""TableLayout: fail-closed validation of the duckdb physical-layout vocabulary.""" +import pytest + +from continuo_duckdb_adapter.domain.identifiers import Identifier +from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey, TableLayout + +TYPES = {"id": "INTEGER", "ts": "TIMESTAMP", "day_col": "DATE", "name": "TEXT"} + + +def _layout(config, types=TYPES): + return TableLayout.from_config(config, types) + + +def test_none_and_empty_config_give_an_empty_layout(): + assert _layout(None).is_empty + assert _layout({}).is_empty + assert TableLayout.empty().is_empty + + +def test_unknown_top_level_key_is_rejected(): + with pytest.raises(ValueError, match="indexes"): + _layout({"indexes": []}) + + +def test_config_must_be_a_mapping(): + with pytest.raises(ValueError, match="mapping"): + _layout(["partitioned_by"]) + + +def test_string_entry_is_identity_partitioning(): + layout = _layout({"partitioned_by": ["name"]}) + assert layout.partition_keys == (PartitionKey(Identifier("name")),) + + +def test_transform_entries(): + layout = _layout({"partitioned_by": [ + {"column": "id", "transform": "bucket", "buckets": 8}, + {"column": "ts", "transform": "month"}, + {"column": "day_col", "transform": "day"}, + {"column": "ts", "transform": "hour"}, + {"column": "ts", "transform": "year"}, + ]}) + assert layout.partition_keys == ( + PartitionKey(Identifier("id"), "bucket", 8), + PartitionKey(Identifier("ts"), "month"), + PartitionKey(Identifier("day_col"), "day"), + PartitionKey(Identifier("ts"), "hour"), + PartitionKey(Identifier("ts"), "year"), + ) + + +@pytest.mark.parametrize("bad", [ + "name", # not a list + [], # empty list + [7], # entry neither str nor mapping + [{"column": "missing"}], # undeclared column + [{"column": ""}], # empty column + [{}], # no column + [{"column": "name", "transform": "week"}], + [{"column": "name", "transform": "bucket"}], # no buckets + [{"column": "id", "transform": "bucket", "buckets": 0}], + [{"column": "id", "transform": "bucket", "buckets": True}], + [{"column": "id", "transform": "bucket", "buckets": "4"}], + [{"column": "id", "transform": "identity", "buckets": 4}], # buckets w/o bucket + [{"column": "id", "transform": "month"}], # time transform on INTEGER + [{"column": "name", "transform": "year"}], # time transform on TEXT + [{"column": "id", "extra": 1}], # unknown entry key + ["name", "name"], # duplicate key +]) +def test_bad_partitioned_by_is_rejected(bad): + with pytest.raises(ValueError): + _layout({"partitioned_by": bad}) + + +def test_time_transform_type_check_is_skipped_when_the_type_is_unknown(): + layout = _layout({"partitioned_by": [{"column": "id", "transform": "month"}]}, {"id": None}) + assert layout.partition_keys == (PartitionKey(Identifier("id"), "month"),) + + +def test_column_existence_is_checked_even_when_types_are_unknown(): + with pytest.raises(ValueError, match="nope"): + _layout({"partitioned_by": ["nope"]}, {"id": None}) + + +def test_sorted_by_defaults_and_options(): + layout = _layout({"sorted_by": [ + "id", + {"column": "ts", "direction": "DESC", "nulls": "First"}, + {"column": "name", "direction": "asc", "nulls": "last"}, + ]}) + assert layout.sort_keys == ( + SortKey(Identifier("id")), + SortKey(Identifier("ts"), descending=True, nulls_first=True), + SortKey(Identifier("name"), descending=False, nulls_first=False), + ) + + +@pytest.mark.parametrize("bad", [ + "id", [], [7], [{"column": "missing"}], [{}], + [{"column": "id", "direction": "sideways"}], + [{"column": "id", "direction": 1}], + [{"column": "id", "nulls": "middle"}], + [{"column": "id", "nulls": 0}], + [{"column": "id", "expr": "id + 1"}], # free-form expressions are not allowed + ["id", "id"], # duplicate sort column + [{"column": "id; DROP TABLE x"}], # not a declared column +]) +def test_bad_sorted_by_is_rejected(bad): + with pytest.raises(ValueError): + _layout({"sorted_by": bad}) + + +def test_both_keys_together(): + layout = _layout({"partitioned_by": ["name"], "sorted_by": ["id"]}) + assert not layout.is_empty + assert len(layout.partition_keys) == 1 and len(layout.sort_keys) == 1 From 21eff7733e8d395bfbbccb9f125126793077cf27 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:23:09 +0200 Subject: [PATCH 04/29] feat(duckdb): LakeGateway port and validation use cases Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../application/__init__.py | 1 + .../application/ports.py | 77 +++++++++++ .../application/warehouse.py | 127 +++++++++++++++++ adapters/duckdb/tests/conftest.py | 108 +++++++++++++++ .../tests/test_adapter_duckdb_validation.py | 128 ++++++++++++++++++ 5 files changed, 441 insertions(+) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/application/__init__.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/application/ports.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py create mode 100644 adapters/duckdb/tests/conftest.py create mode 100644 adapters/duckdb/tests/test_adapter_duckdb_validation.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/__init__.py b/adapters/duckdb/continuo_duckdb_adapter/application/__init__.py new file mode 100644 index 0000000..f1584bf --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/application/__init__.py @@ -0,0 +1 @@ +"""Use cases. Depends on the LakeGateway port, never on infrastructure.""" diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/ports.py b/adapters/duckdb/continuo_duckdb_adapter/application/ports.py new file mode 100644 index 0000000..49a74d6 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/application/ports.py @@ -0,0 +1,77 @@ +"""The port the use cases need from a DuckLake. Implemented by infrastructure.""" +from __future__ import annotations + +from abc import ABC, abstractmethod +from collections.abc import Sequence +from contextlib import AbstractContextManager +from typing import TYPE_CHECKING + +from ..domain.columns import ColumnDefinition +from ..domain.identifiers import Identifier, QualifiedTable +from ..domain.layout import TableLayout + +if TYPE_CHECKING: # pragma: no cover + import pyarrow # type: ignore[import-untyped] + + +class LakeConflictError(Exception): + """A concurrent DuckLake transaction won a snapshot conflict; safe to retry.""" + + +class LakeGateway(ABC): + """Everything the warehouse use cases ask of one DuckLake connection.""" + + @abstractmethod + def transaction(self) -> AbstractContextManager[None]: + """BEGIN ... COMMIT; ROLLBACK (logged, never masking) if the body raises.""" + + @abstractmethod + def create_schema_if_not_exists(self, schema: Identifier) -> None: ... + + @abstractmethod + def drop_schema_cascade(self, schema: Identifier) -> None: + """Drop the schema and everything in it; a no-op when absent.""" + + @abstractmethod + def table_exists(self, table: QualifiedTable) -> bool: ... + + @abstractmethod + def drop_table_if_exists(self, table: QualifiedTable) -> None: ... + + @abstractmethod + def create_table( + self, + table: QualifiedTable, + columns: Sequence[ColumnDefinition], + *, + if_not_exists: bool = False, + ) -> None: ... + + @abstractmethod + def create_empty_table_as(self, table: QualifiedTable, select_sql: str) -> None: + """Create *table* empty, shaped by the SELECT (no terminator, single read).""" + + @abstractmethod + def create_empty_clone(self, target: QualifiedTable, source: QualifiedTable) -> None: + """Create *target* empty with *source*'s shape.""" + + @abstractmethod + def apply_layout(self, table: QualifiedTable, layout: TableLayout) -> None: + """Apply partitioning and sort order to an existing table.""" + + @abstractmethod + def explain_read(self, sql: str) -> None: + """Bind-check one read without scanning data; raise if it does not bind.""" + + @abstractmethod + def fetch_arrow(self, sql: str) -> "pyarrow.Table": ... + + @abstractmethod + def delete_all(self, table: QualifiedTable) -> None: ... + + @abstractmethod + def insert_arrow(self, table: QualifiedTable, data: "pyarrow.Table") -> None: + """Append *data* in its own column order.""" + + @abstractmethod + def close(self) -> None: ... diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py new file mode 100644 index 0000000..e7bd39b --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py @@ -0,0 +1,127 @@ +"""The warehouse use cases: one class for both roles the port serves. + +Validation DDL (this task) and python-node data-plane I/O (next task) share one +connection, so they are one implementation, as in the postgres and trino +adapters. The class depends only on the LakeGateway port; ``required_env`` and +``from_env`` stay abstract and are supplied by the composition root +(``continuo_duckdb_adapter.adapter.DuckDBAdapter``). +""" +from __future__ import annotations + +import logging +import time +from collections.abc import Callable + +from continuo_engine_contract.port import WarehouseAdapter # type: ignore[import-untyped] +from continuo_engine_contract.sql import ensure_single_read # type: ignore[import-untyped] + +from ..domain.columns import ColumnDefinition, column_types +from ..domain.identifiers import Identifier, QualifiedTable +from ..domain.layout import TableLayout +from .ports import LakeConflictError, LakeGateway + +logger = logging.getLogger("continuo_duckdb_adapter") + +# DuckLake has no advisory lock. Parallel validation nodes race on creation and +# the loser fails its COMMIT with a snapshot conflict; the retry then finds the +# object exists. Bounded so a genuine, persistent conflict still surfaces. +_CONFLICT_ATTEMPTS = 5 +_CONFLICT_BACKOFF_SECONDS = 0.05 + + +def _columns(raw: list[dict]) -> list[ColumnDefinition]: + return [ColumnDefinition.from_mapping(entry) for entry in raw] + + +class LakeWarehouse(WarehouseAdapter): + """WarehouseAdapter implemented over a LakeGateway.""" + + def __init__( + self, gateway: LakeGateway, *, sleep: Callable[[float], None] = time.sleep + ) -> None: + self._gateway = gateway + self._sleep = sleep + + def _retrying(self, operation: Callable[[], None]) -> None: + for attempt in range(1, _CONFLICT_ATTEMPTS + 1): + try: + operation() + return + except LakeConflictError: + if attempt == _CONFLICT_ATTEMPTS: + raise + logger.info("concurrent change conflicted (attempt %d); retrying", attempt) + self._sleep(_CONFLICT_BACKOFF_SECONDS * attempt) + + # --- Schema lifecycle --------------------------------------------------- + + def ensure_schema(self, schema: str) -> None: + """Idempotently create *schema*; safe under concurrent callers.""" + target = Identifier(schema) + logger.info("ensuring schema %s exists", schema) + self._retrying(lambda: self._gateway.create_schema_if_not_exists(target)) + + def drop_schema(self, schema: str) -> None: + """Idempotently drop *schema* and everything in it; no-op if absent.""" + logger.info("dropping candidate schema %s", schema) + self._gateway.drop_schema_cascade(Identifier(schema)) + + # --- Validation builds -------------------------------------------------- + + def build_empty_from_sql(self, schema: str, table: str, compiled_sql: str) -> None: + """Create ``schema.table`` empty, shaped by the compiled SELECT.""" + inner = compiled_sql.strip().rstrip(";").strip() + target = QualifiedTable.of(schema, table) + with self._gateway.transaction(): + self._gateway.drop_table_if_exists(target) + self._gateway.create_empty_table_as(target, inner) + + def clone_empty_from_prod(self, candidate_schema: str, prod_schema: str, table: str) -> None: + """Create ``candidate_schema.table`` empty, shaped like ``prod_schema.table``.""" + target = QualifiedTable.of(candidate_schema, table) + with self._gateway.transaction(): + self._gateway.drop_table_if_exists(target) + self._gateway.create_empty_clone(target, QualifiedTable.of(prod_schema, table)) + + def build_empty_from_columns( + self, schema: str, table: str, columns: list[dict], config: dict + ) -> None: + """Create ``schema.table`` empty from declared typed columns (drop-then-create). + + Types and layout are validated before the first statement; the rebuild is + one transaction, so a failure leaves the prior table in place. + """ + defs = _columns(columns) + layout = TableLayout.from_config(config, column_types(defs)) + target = QualifiedTable.of(schema, table) + with self._gateway.transaction(): + self._gateway.drop_table_if_exists(target) + self._gateway.create_table(target, defs) + if not layout.is_empty: + self._gateway.apply_layout(target, layout) + + def check_binds(self, sql: str) -> None: + """Verify *sql* binds against current schema state; scans no data. + + The parse gate runs first: DuckDB executes every ``;``-separated + statement handed to it, so a stacked statement would run for real. + """ + ensure_single_read(sql, dialect="duckdb") + inner = sql.strip().rstrip(";").strip() + logger.info("bind-checking read via EXPLAIN") + self._gateway.explain_read(inner) + + # --- Python-node data plane: implemented in task 5 ----------------------- + + def fetch(self, sql: str): + raise NotImplementedError("implemented in task 5") + + def ensure_table(self, schema: str, table: str, columns: list[dict], *, config: dict) -> None: + raise NotImplementedError("implemented in task 5") + + def load(self, schema: str, table: str, data) -> None: + raise NotImplementedError("implemented in task 5") + + def close(self) -> None: + """Release the underlying connection.""" + self._gateway.close() diff --git a/adapters/duckdb/tests/conftest.py b/adapters/duckdb/tests/conftest.py new file mode 100644 index 0000000..27fd438 --- /dev/null +++ b/adapters/duckdb/tests/conftest.py @@ -0,0 +1,108 @@ +"""Shared fixtures for adapters/duckdb/tests. + +This directory is its own top-level pytest package, so it cannot see the root +``tests/conftest.py``. The FakeLakeGateway lives here (not in an importable +module) because two ``tests`` packages cannot both be imported by name under +importlib mode; tests receive it through the ``gateway`` fixture. +""" +import contextlib + +import pytest + +from continuo_duckdb_adapter.application.ports import LakeConflictError, LakeGateway +from continuo_duckdb_adapter.application.warehouse import LakeWarehouse + + +class FakeLakeGateway(LakeGateway): + """Records every call as ``(method_name, *args)``; can raise on demand.""" + + def __init__(self) -> None: + self.calls: list[tuple] = [] + self.existing: set[tuple[str, str]] = set() + self.conflicts: dict[str, int] = {} + self.fail_on: dict[str, Exception] = {} + self.fetch_result = None + + def names(self) -> list[str]: + return [call[0] for call in self.calls] + + def _do(self, name: str, *args) -> None: + self.calls.append((name, *args)) + if self.conflicts.get(name, 0) > 0: + self.conflicts[name] -= 1 + raise LakeConflictError(name) + if name in self.fail_on: + raise self.fail_on[name] + + @contextlib.contextmanager + def transaction(self): + self.calls.append(("begin",)) + try: + yield + except BaseException: + self.calls.append(("rollback",)) + raise + self.calls.append(("commit",)) + + def create_schema_if_not_exists(self, schema): + self._do("create_schema_if_not_exists", schema) + + def drop_schema_cascade(self, schema): + self._do("drop_schema_cascade", schema) + + def table_exists(self, table): + self._do("table_exists", table) + return (table.schema.name, table.table.name) in self.existing + + def drop_table_if_exists(self, table): + self._do("drop_table_if_exists", table) + + def create_table(self, table, columns, *, if_not_exists=False): + self._do("create_table", table, tuple(columns), if_not_exists) + + def create_empty_table_as(self, table, select_sql): + self._do("create_empty_table_as", table, select_sql) + + def create_empty_clone(self, target, source): + self._do("create_empty_clone", target, source) + + def apply_layout(self, table, layout): + self._do("apply_layout", table, layout) + + def explain_read(self, sql): + self._do("explain_read", sql) + + def fetch_arrow(self, sql): + self._do("fetch_arrow", sql) + return self.fetch_result + + def delete_all(self, table): + self._do("delete_all", table) + + def insert_arrow(self, table, data): + self._do("insert_arrow", table, data) + + def close(self): + self._do("close") + + +class _UnitWarehouse(LakeWarehouse): + """LakeWarehouse with the composition-root hooks stubbed for unit tests.""" + + @classmethod + def required_env(cls) -> list[str]: + return [] + + @classmethod + def from_env(cls) -> "_UnitWarehouse": + raise NotImplementedError + + +@pytest.fixture +def gateway() -> FakeLakeGateway: + return FakeLakeGateway() + + +@pytest.fixture +def warehouse(gateway) -> _UnitWarehouse: + return _UnitWarehouse(gateway, sleep=lambda _seconds: None) diff --git a/adapters/duckdb/tests/test_adapter_duckdb_validation.py b/adapters/duckdb/tests/test_adapter_duckdb_validation.py new file mode 100644 index 0000000..64555ff --- /dev/null +++ b/adapters/duckdb/tests/test_adapter_duckdb_validation.py @@ -0,0 +1,128 @@ +"""Validation use cases, against the recording FakeLakeGateway (no engine).""" +import pytest + +from continuo_duckdb_adapter.application.ports import LakeConflictError +from continuo_duckdb_adapter.domain.columns import ColumnDefinition +from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable +from continuo_duckdb_adapter.domain.layout import PartitionKey, TableLayout + +T = QualifiedTable.of("s", "t") +COLS = [{"name": "id", "type": "INTEGER", "nullable": False}, {"name": "ts", "type": "TIMESTAMP"}] + + +def test_ensure_schema_creates_the_schema(warehouse, gateway): + warehouse.ensure_schema("analytics") + assert gateway.calls == [("create_schema_if_not_exists", Identifier("analytics"))] + + +def test_ensure_schema_retries_conflicts_then_succeeds(warehouse, gateway): + gateway.conflicts["create_schema_if_not_exists"] = 2 + warehouse.ensure_schema("analytics") + assert gateway.names() == ["create_schema_if_not_exists"] * 3 + + +def test_ensure_schema_gives_up_after_bounded_attempts(warehouse, gateway): + gateway.conflicts["create_schema_if_not_exists"] = 99 + with pytest.raises(LakeConflictError): + warehouse.ensure_schema("analytics") + assert gateway.names() == ["create_schema_if_not_exists"] * 5 + + +def test_ensure_schema_does_not_retry_other_errors(warehouse, gateway): + gateway.fail_on["create_schema_if_not_exists"] = RuntimeError("boom") + with pytest.raises(RuntimeError): + warehouse.ensure_schema("analytics") + assert gateway.names() == ["create_schema_if_not_exists"] + + +def test_drop_schema_cascades(warehouse, gateway): + warehouse.drop_schema("analytics") + assert gateway.calls == [("drop_schema_cascade", Identifier("analytics"))] + + +def test_build_empty_from_sql_strips_the_terminator_and_rebuilds_in_one_transaction(warehouse, gateway): + warehouse.build_empty_from_sql("s", "t", " select 1 as a ; \n") + assert gateway.calls == [ + ("begin",), + ("drop_table_if_exists", T), + ("create_empty_table_as", T, "select 1 as a"), + ("commit",), + ] + + +def test_build_empty_from_sql_rolls_back_on_failure(warehouse, gateway): + gateway.fail_on["create_empty_table_as"] = RuntimeError("bind error") + with pytest.raises(RuntimeError): + warehouse.build_empty_from_sql("s", "t", "select 1") + assert gateway.names()[-1] == "rollback" + + +def test_clone_empty_from_prod_rebuilds_in_one_transaction(warehouse, gateway): + warehouse.clone_empty_from_prod("cand", "prod", "t") + assert gateway.calls == [ + ("begin",), + ("drop_table_if_exists", QualifiedTable.of("cand", "t")), + ("create_empty_clone", QualifiedTable.of("cand", "t"), QualifiedTable.of("prod", "t")), + ("commit",), + ] + + +def test_build_empty_from_columns_without_config_emits_no_layout(warehouse, gateway): + warehouse.build_empty_from_columns("s", "t", COLS, {}) + assert gateway.names() == ["begin", "drop_table_if_exists", "create_table", "commit"] + assert gateway.calls[2] == ( + "create_table", T, + (ColumnDefinition(Identifier("id"), "INTEGER", False), + ColumnDefinition(Identifier("ts"), "TIMESTAMP", True)), + False, + ) + + +def test_build_empty_from_columns_applies_layout_inside_the_transaction(warehouse, gateway): + warehouse.build_empty_from_columns("s", "t", COLS, {"partitioned_by": [{"column": "ts", "transform": "month"}]}) + assert gateway.names() == ["begin", "drop_table_if_exists", "create_table", "apply_layout", "commit"] + assert gateway.calls[3] == ( + "apply_layout", T, TableLayout(partition_keys=(PartitionKey(Identifier("ts"), "month"),)), + ) + + +@pytest.mark.parametrize("config", [ + {"sortkey": ["id"]}, + {"partitioned_by": ["missing"]}, + {"partitioned_by": [{"column": "id", "transform": "month"}]}, + {"sorted_by": "id"}, +]) +def test_bad_config_is_rejected_before_any_statement(warehouse, gateway, config): + with pytest.raises(ValueError): + warehouse.build_empty_from_columns("s", "t", COLS, config) + assert gateway.calls == [] + + +def test_bad_column_type_is_rejected_before_any_statement(warehouse, gateway): + with pytest.raises(ValueError): + warehouse.build_empty_from_columns("s", "t", [{"name": "id", "type": "INT; DROP TABLE x"}], {}) + assert gateway.calls == [] + + +def test_check_binds_runs_the_gate_before_touching_the_gateway(warehouse, gateway): + for bad in ("SELECT 1; DROP TABLE t", "DELETE FROM t", "SELECT 1) AS x; DROP TABLE t; SELECT * FROM (SELECT 1"): + with pytest.raises(ValueError): + warehouse.check_binds(bad) + assert gateway.calls == [] + + +@pytest.mark.parametrize("sql,inner", [ + ("SELECT id FROM s.t;", "SELECT id FROM s.t"), + ("SELECT ';' AS semi", "SELECT ';' AS semi"), + ("SELECT 1 -- trailing comment", "SELECT 1 -- trailing comment"), + ("WITH c AS (SELECT 1 AS a) SELECT a FROM c", "WITH c AS (SELECT 1 AS a) SELECT a FROM c"), + ("VALUES (1), (2)", "VALUES (1), (2)"), +]) +def test_check_binds_passes_a_single_read_to_explain(warehouse, gateway, sql, inner): + warehouse.check_binds(sql) + assert gateway.calls == [("explain_read", inner)] + + +def test_close_closes_the_gateway(warehouse, gateway): + warehouse.close() + assert gateway.calls == [("close",)] From 88bbf378c5e377b2ab41e80b68d6471eea232823 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:25:05 +0200 Subject: [PATCH 05/29] feat(duckdb): runtime use cases (fetch, ensure_table, load, validate_config) Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../application/warehouse.py | 76 +++++++++++-- .../tests/test_adapter_duckdb_runtime.py | 106 ++++++++++++++++++ 2 files changed, 175 insertions(+), 7 deletions(-) create mode 100644 adapters/duckdb/tests/test_adapter_duckdb_runtime.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py index e7bd39b..56287c4 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py +++ b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py @@ -11,6 +11,7 @@ import logging import time from collections.abc import Callable +from typing import TYPE_CHECKING, Any from continuo_engine_contract.port import WarehouseAdapter # type: ignore[import-untyped] from continuo_engine_contract.sql import ensure_single_read # type: ignore[import-untyped] @@ -20,6 +21,9 @@ from ..domain.layout import TableLayout from .ports import LakeConflictError, LakeGateway +if TYPE_CHECKING: # pragma: no cover + import pyarrow as pa # type: ignore[import-untyped] + logger = logging.getLogger("continuo_duckdb_adapter") # DuckLake has no advisory lock. Parallel validation nodes race on creation and @@ -111,16 +115,74 @@ def check_binds(self, sql: str) -> None: logger.info("bind-checking read via EXPLAIN") self._gateway.explain_read(inner) - # --- Python-node data plane: implemented in task 5 ----------------------- + # --- Python-node data plane --------------------------------------------- + + def fetch(self, sql: str) -> "pa.Table": + """Execute one declared read and return the result as an Arrow table.""" + data = self._gateway.fetch_arrow(sql) + seen: set[str] = set() + duplicates: set[str] = set() + for name in data.schema.names: + if name in seen: + duplicates.add(name) + seen.add(name) + if duplicates: + raise ValueError(f"duplicate column name(s) in SELECT result: {sorted(duplicates)!r}") + return data + + @classmethod + def validate_config(cls, config: dict[str, Any] | None, column_names: list[str]) -> None: + """Validate *config* against this engine's vocabulary, without connecting. + + The harness calls this right after selecting the node so a malformed + ``config`` fails in the first second, not after the script has run. It + only knows column names, so the DATE/TIMESTAMP check for time transforms + is deferred; ``ensure_table`` runs the complete check and remains the + enforcement point (the harness skips adapters without this method; see + docs/boundary-contract.md section 13.4). + """ + TableLayout.from_config(config, {name: None for name in column_names}) + + def ensure_table( + self, + schema: str, + table: str, + columns: list[dict[str, Any]], + *, + config: dict[str, Any], + ) -> None: + """Create the table if absent; layout is applied only when this call creates it. - def fetch(self, sql: str): - raise NotImplementedError("implemented in task 5") + Validation comes first, so a bad config or type emits no statement. The + existence check and the create share one transaction; a concurrent + creator makes one side fail with a conflict, and the retry then finds + the table and returns. + """ + defs = _columns(columns) + layout = TableLayout.from_config(config, column_types(defs)) + self.ensure_schema(schema) + target = QualifiedTable.of(schema, table) + self._retrying(lambda: self._create_if_absent(target, defs, layout)) - def ensure_table(self, schema: str, table: str, columns: list[dict], *, config: dict) -> None: - raise NotImplementedError("implemented in task 5") + def _create_if_absent( + self, target: QualifiedTable, defs: list[ColumnDefinition], layout: TableLayout + ) -> None: + with self._gateway.transaction(): + if self._gateway.table_exists(target): + return + logger.info("creating table %s.%s", target.schema.name, target.table.name) + self._gateway.create_table(target, defs) + if not layout.is_empty: + self._gateway.apply_layout(target, layout) - def load(self, schema: str, table: str, data) -> None: - raise NotImplementedError("implemented in task 5") + def load(self, schema: str, table: str, data: "pa.Table") -> None: + """Atomically replace the table's contents with *data* (one transaction).""" + target = QualifiedTable.of(schema, table) + with self._gateway.transaction(): + self._gateway.delete_all(target) + if data.num_rows: + logger.info("loading %d row(s) into %s.%s", data.num_rows, schema, table) + self._gateway.insert_arrow(target, data) def close(self) -> None: """Release the underlying connection.""" diff --git a/adapters/duckdb/tests/test_adapter_duckdb_runtime.py b/adapters/duckdb/tests/test_adapter_duckdb_runtime.py new file mode 100644 index 0000000..0ab23e8 --- /dev/null +++ b/adapters/duckdb/tests/test_adapter_duckdb_runtime.py @@ -0,0 +1,106 @@ +"""Runtime use cases, against the recording FakeLakeGateway (no engine).""" +import pyarrow as pa +import pytest + +from continuo_duckdb_adapter.application.ports import LakeConflictError +from continuo_duckdb_adapter.domain.columns import ColumnDefinition +from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable +from continuo_duckdb_adapter.domain.layout import PartitionKey, TableLayout + +T = QualifiedTable.of("s", "t") +COLS = [{"name": "id", "type": "INTEGER", "nullable": False}, {"name": "ts", "type": "TIMESTAMP", "nullable": True}] +DEFS = (ColumnDefinition(Identifier("id"), "INTEGER", False), ColumnDefinition(Identifier("ts"), "TIMESTAMP", True)) + + +def test_fetch_returns_the_gateways_arrow_table(warehouse, gateway): + gateway.fetch_result = pa.table({"a": [1, 2]}) + assert warehouse.fetch("SELECT a FROM s.t") is gateway.fetch_result + assert gateway.calls == [("fetch_arrow", "SELECT a FROM s.t")] + + +def test_fetch_rejects_duplicate_output_columns(warehouse, gateway): + gateway.fetch_result = pa.Table.from_arrays([pa.array([1]), pa.array([2])], names=["id", "id"]) + with pytest.raises(ValueError, match="duplicate column name"): + warehouse.fetch("SELECT 1 AS id, 2 AS id") + + +def test_fetch_accepts_an_empty_result(warehouse, gateway): + gateway.fetch_result = pa.table({"a": pa.array([], pa.int32())}) + assert warehouse.fetch("SELECT a FROM s.t WHERE false").num_rows == 0 + + +def test_validate_config_needs_no_connection(warehouse): + type(warehouse).validate_config({"partitioned_by": ["id"], "sorted_by": ["id"]}, ["id", "ts"]) + type(warehouse).validate_config(None, ["id"]) + type(warehouse).validate_config({}, ["id"]) + + +@pytest.mark.parametrize("config", [ + {"indexes": []}, {"partitioned_by": ["missing"]}, {"sorted_by": [{"column": "id", "direction": "up"}]}, +]) +def test_validate_config_rejects_what_ensure_table_rejects(warehouse, config): + with pytest.raises(ValueError): + type(warehouse).validate_config(config, ["id"]) + + +def test_validate_config_defers_the_time_transform_type_check(warehouse): + type(warehouse).validate_config({"partitioned_by": [{"column": "id", "transform": "month"}]}, ["id"]) + + +def test_ensure_table_creates_a_missing_table_once(warehouse, gateway): + warehouse.ensure_table("s", "t", COLS, config={}) + assert gateway.calls == [ + ("create_schema_if_not_exists", Identifier("s")), + ("begin",), ("table_exists", T), ("create_table", T, DEFS, False), ("commit",), + ] + + +def test_ensure_table_applies_layout_only_when_it_creates_the_table(warehouse, gateway): + config = {"partitioned_by": [{"column": "ts", "transform": "month"}]} + layout = TableLayout(partition_keys=(PartitionKey(Identifier("ts"), "month"),)) + warehouse.ensure_table("s", "t", COLS, config=config) + assert ("apply_layout", T, layout) in gateway.calls + gateway.calls.clear() + gateway.existing.add(("s", "t")) + warehouse.ensure_table("s", "t", COLS, config=config) + assert gateway.names() == ["create_schema_if_not_exists", "begin", "table_exists", "commit"] + + +def test_ensure_table_bad_config_or_type_emits_nothing(warehouse, gateway): + with pytest.raises(ValueError): + warehouse.ensure_table("s", "t", COLS, config={"nope": 1}) + with pytest.raises(ValueError): + warehouse.ensure_table("s", "t", [{"name": "id", "type": "INT; DROP"}], config={}) + assert gateway.calls == [] + + +def test_ensure_table_retries_a_creation_conflict(warehouse, gateway): + gateway.conflicts["create_table"] = 1 + warehouse.ensure_table("s", "t", COLS, config={}) + assert gateway.names().count("create_table") == 2 + assert gateway.names().count("rollback") == 1 + + +def test_ensure_table_gives_up_after_bounded_conflicts(warehouse, gateway): + gateway.conflicts["create_table"] = 99 + with pytest.raises(LakeConflictError): + warehouse.ensure_table("s", "t", COLS, config={}) + assert gateway.names().count("create_table") == 5 + + +def test_load_replaces_contents_in_one_transaction(warehouse, gateway): + data = pa.table({"id": [1, 2]}) + warehouse.load("s", "t", data) + assert gateway.calls == [("begin",), ("delete_all", T), ("insert_arrow", T, data), ("commit",)] + + +def test_load_of_zero_rows_only_clears(warehouse, gateway): + warehouse.load("s", "t", pa.table({"id": pa.array([], pa.int32())})) + assert gateway.names() == ["begin", "delete_all", "commit"] + + +def test_load_rolls_back_when_the_insert_fails(warehouse, gateway): + gateway.fail_on["insert_arrow"] = RuntimeError("NOT NULL violated") + with pytest.raises(RuntimeError): + warehouse.load("s", "t", pa.table({"id": [1]})) + assert gateway.names() == ["begin", "delete_all", "insert_arrow", "rollback"] From 23c59834878bd070f31930afa2b1b7f294368f90 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:27:50 +0200 Subject: [PATCH 06/29] feat(duckdb): DuckLake settings and SQL renderer Co-Authored-By: Claude Haiku 4.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../infrastructure/__init__.py | 1 + .../infrastructure/ddl.py | 158 +++++++++++++++++ .../infrastructure/settings.py | 79 +++++++++ .../tests/test_infra_duckdb_settings_ddl.py | 163 ++++++++++++++++++ 4 files changed, 401 insertions(+) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/__init__.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py create mode 100644 adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/__init__.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/__init__.py new file mode 100644 index 0000000..d104691 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/__init__.py @@ -0,0 +1 @@ +"""Engine I/O: DuckDB/DuckLake connection, settings and SQL rendering.""" diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py new file mode 100644 index 0000000..8a8aba3 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -0,0 +1,158 @@ +"""Render domain objects into DuckDB/DuckLake SQL. + +Every identifier goes through ``quote_identifier`` and every string value through +``sql_literal``; types, transforms, directions and null orders come from +validated domain values, never raw input. Own DDL is fully catalog-qualified +(``"lake"."schema"."table"``) so a schema named like an attached database cannot +be mis-resolved. +""" +from __future__ import annotations + +import re +from collections.abc import Sequence + +from ..domain.columns import ColumnDefinition +from ..domain.identifiers import Identifier, QualifiedTable +from ..domain.layout import PartitionKey, SortKey +from .settings import DuckLakeSettings + +_LIBPQ_BARE = re.compile(r"[A-Za-z0-9_.:/\-]+") + + +def quote_identifier(name: str) -> str: + return '"' + name.replace('"', '""') + '"' + + +def sql_literal(value: str) -> str: + return "'" + value.replace("'", "''") + "'" + + +def _libpq_value(value: str) -> str: + """Quote one libpq ``keyword=value`` value only when it needs it.""" + if _LIBPQ_BARE.fullmatch(value): + return value + escaped = value.replace("\\", "\\\\").replace("'", "\\'") + return f"'{escaped}'" + + +def attach_statement(settings: DuckLakeSettings, catalog: str) -> str: + conninfo = " ".join( + f"{key}={_libpq_value(value)}" + for key, value in ( + ("host", settings.catalog_host), + ("port", settings.catalog_port), + ("dbname", settings.catalog_db), + ("user", settings.catalog_user), + ("password", settings.catalog_password), + ) + ) + options = [f"DATA_PATH {sql_literal(settings.data_path)}"] + if settings.data_inlining_row_limit is not None: + options.append(f"DATA_INLINING_ROW_LIMIT {settings.data_inlining_row_limit}") + return ( + f"ATTACH {sql_literal('ducklake:postgres:' + conninfo)} " + f"AS {quote_identifier(catalog)} ({', '.join(options)})" + ) + + +def s3_secret_statement(settings: DuckLakeSettings) -> str: + parts = ["TYPE S3"] + if settings.s3_access_key_id and settings.s3_secret_access_key: + parts.append(f"KEY_ID {sql_literal(settings.s3_access_key_id)}") + parts.append(f"SECRET {sql_literal(settings.s3_secret_access_key)}") + else: + parts.append("PROVIDER credential_chain") + if settings.s3_endpoint: + parts.append(f"ENDPOINT {sql_literal(settings.s3_endpoint)}") + parts.append(f"REGION {sql_literal(settings.s3_region)}") + parts.append(f"URL_STYLE {sql_literal(settings.s3_url_style)}") + parts.append(f"USE_SSL {'true' if settings.s3_use_ssl else 'false'}") + return f"CREATE OR REPLACE SECRET continuo_s3 ({', '.join(parts)})" + + +class DdlRenderer: + """SQL text for every LakeGateway operation, against one catalog alias.""" + + TABLE_EXISTS_QUERY = ( + "SELECT count(*) FROM duckdb_tables() " + "WHERE database_name = ? AND schema_name = ? AND table_name = ?" + ) + + def __init__(self, catalog: str) -> None: + self._catalog = catalog + + @property + def catalog_name(self) -> str: + return self._catalog + + def schema_ref(self, schema: Identifier) -> str: + return f"{quote_identifier(self._catalog)}.{quote_identifier(schema.name)}" + + def table_ref(self, table: QualifiedTable) -> str: + return f"{self.schema_ref(table.schema)}.{quote_identifier(table.table.name)}" + + def create_schema_if_not_exists(self, schema: Identifier) -> str: + return f"CREATE SCHEMA IF NOT EXISTS {self.schema_ref(schema)}" + + def drop_schema_cascade(self, schema: Identifier) -> str: + return f"DROP SCHEMA IF EXISTS {self.schema_ref(schema)} CASCADE" + + def drop_table_if_exists(self, table: QualifiedTable) -> str: + return f"DROP TABLE IF EXISTS {self.table_ref(table)}" + + def create_table( + self, + table: QualifiedTable, + columns: Sequence[ColumnDefinition], + *, + if_not_exists: bool = False, + ) -> str: + defs = ", ".join( + f"{quote_identifier(c.name.name)} {c.type}" + ("" if c.nullable else " NOT NULL") + for c in columns + ) + clause = " IF NOT EXISTS" if if_not_exists else "" + return f"CREATE TABLE{clause} {self.table_ref(table)} ({defs})" + + def create_empty_table_as(self, table: QualifiedTable, select_sql: str) -> str: + # The newline before ")" keeps a trailing "-- comment" in the read from + # swallowing the closing parenthesis. + return f"CREATE TABLE {self.table_ref(table)} AS (\n{select_sql}\n) WITH NO DATA" + + def create_empty_clone(self, target: QualifiedTable, source: QualifiedTable) -> str: + return ( + f"CREATE TABLE {self.table_ref(target)} AS " + f"SELECT * FROM {self.table_ref(source)} WHERE false" + ) + + def set_partitioned_by(self, table: QualifiedTable, keys: Sequence[PartitionKey]) -> str: + return f"ALTER TABLE {self.table_ref(table)} SET PARTITIONED BY ({', '.join(map(self._partition, keys))})" + + def set_sorted_by(self, table: QualifiedTable, keys: Sequence[SortKey]) -> str: + return f"ALTER TABLE {self.table_ref(table)} SET SORTED BY ({', '.join(map(self._sort, keys))})" + + def explain_read(self, sql: str) -> str: + return f"EXPLAIN SELECT * FROM (\n{sql}\n) AS __check_binds__" + + def delete_all(self, table: QualifiedTable) -> str: + return f"DELETE FROM {self.table_ref(table)}" + + def insert_select(self, table: QualifiedTable, column_names: Sequence[str], source: str) -> str: + cols = ", ".join(quote_identifier(name) for name in column_names) + return f"INSERT INTO {self.table_ref(table)} ({cols}) SELECT {cols} FROM {quote_identifier(source)}" + + @staticmethod + def _partition(key: PartitionKey) -> str: + column = quote_identifier(key.column.name) + if key.transform == "identity": + return column + if key.transform == "bucket": + return f"bucket({key.buckets}, {column})" + return f"{key.transform}({column})" + + @staticmethod + def _sort(key: SortKey) -> str: + text = f"{quote_identifier(key.column.name)} {'DESC' if key.descending else 'ASC'}" + if key.nulls_first is not None: + text += " NULLS FIRST" if key.nulls_first else " NULLS LAST" + return text diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py new file mode 100644 index 0000000..ac92a19 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py @@ -0,0 +1,79 @@ +"""Connection settings for one DuckLake, read from DUCKDB_* environment variables.""" +from __future__ import annotations + +import os +from collections.abc import Mapping +from dataclasses import dataclass + +REQUIRED_ENV: tuple[str, ...] = ( + "DUCKDB_CATALOG_HOST", + "DUCKDB_CATALOG_DB", + "DUCKDB_CATALOG_USER", + "DUCKDB_DATA_PATH", +) +_URL_STYLES = ("path", "vhost") + + +@dataclass(frozen=True) +class DuckLakeSettings: + catalog_host: str + catalog_port: str + catalog_db: str + catalog_user: str + catalog_password: str + data_path: str + s3_endpoint: str | None + s3_access_key_id: str | None + s3_secret_access_key: str | None + s3_region: str + s3_url_style: str + s3_use_ssl: bool + extension_directory: str | None + data_inlining_row_limit: int | None + + @property + def uses_s3(self) -> bool: + return self.data_path.startswith("s3://") + + @classmethod + def from_env(cls, env: Mapping[str, str] | None = None) -> "DuckLakeSettings": + env = os.environ if env is None else env + + def need(name: str) -> str: + value = env.get(name, "") + if not value: + raise ValueError(f"missing required env var {name}") + return value + + def optional(name: str) -> str | None: + return env.get(name) or None + + port = env.get("DUCKDB_CATALOG_PORT") or "5432" + if not port.isdigit(): + raise ValueError(f"DUCKDB_CATALOG_PORT must be a port number, got {port!r}") + limit_raw = optional("DUCKDB_DATA_INLINING_ROW_LIMIT") + if limit_raw is not None and not limit_raw.isdigit(): + raise ValueError( + f"DUCKDB_DATA_INLINING_ROW_LIMIT must be a non-negative integer, got {limit_raw!r}" + ) + endpoint = optional("DUCKDB_S3_ENDPOINT") + url_style = env.get("DUCKDB_S3_URL_STYLE") or ("path" if endpoint else "vhost") + if url_style not in _URL_STYLES: + raise ValueError(f"DUCKDB_S3_URL_STYLE must be one of {_URL_STYLES}, got {url_style!r}") + use_ssl = (env.get("DUCKDB_S3_USE_SSL") or "true").lower() not in ("0", "false", "no") + return cls( + catalog_host=need("DUCKDB_CATALOG_HOST"), + catalog_port=port, + catalog_db=need("DUCKDB_CATALOG_DB"), + catalog_user=need("DUCKDB_CATALOG_USER"), + catalog_password=env.get("DUCKDB_CATALOG_PASSWORD", ""), + data_path=need("DUCKDB_DATA_PATH"), + s3_endpoint=endpoint, + s3_access_key_id=optional("DUCKDB_S3_ACCESS_KEY_ID"), + s3_secret_access_key=optional("DUCKDB_S3_SECRET_ACCESS_KEY"), + s3_region=env.get("DUCKDB_S3_REGION") or "us-east-1", + s3_url_style=url_style, + s3_use_ssl=use_ssl, + extension_directory=optional("DUCKDB_EXTENSION_DIRECTORY"), + data_inlining_row_limit=None if limit_raw is None else int(limit_raw), + ) diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py new file mode 100644 index 0000000..7671c9a --- /dev/null +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -0,0 +1,163 @@ +"""Settings parsing and SQL rendering: pure string work, no engine.""" +import pytest + +from continuo_duckdb_adapter.domain.columns import ColumnDefinition +from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable +from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey +from continuo_duckdb_adapter.infrastructure.ddl import ( + DdlRenderer, attach_statement, quote_identifier, s3_secret_statement, sql_literal, +) +from continuo_duckdb_adapter.infrastructure.settings import REQUIRED_ENV, DuckLakeSettings + +ENV = { + "DUCKDB_CATALOG_HOST": "localhost", "DUCKDB_CATALOG_DB": "catalog", + "DUCKDB_CATALOG_USER": "continuo", "DUCKDB_DATA_PATH": "s3://warehouse/lake/", +} + + +def test_required_env_lists_the_four_mandatory_vars(): + assert list(REQUIRED_ENV) == [ + "DUCKDB_CATALOG_HOST", "DUCKDB_CATALOG_DB", "DUCKDB_CATALOG_USER", "DUCKDB_DATA_PATH", + ] + + +@pytest.mark.parametrize("missing", REQUIRED_ENV) +def test_missing_required_var_is_named(missing): + env = {k: v for k, v in ENV.items() if k != missing} + with pytest.raises(ValueError, match=missing): + DuckLakeSettings.from_env(env) + + +def test_defaults(): + s = DuckLakeSettings.from_env(ENV) + assert (s.catalog_port, s.catalog_password) == ("5432", "") + assert (s.s3_region, s.s3_use_ssl, s.s3_url_style) == ("us-east-1", True, "vhost") + assert s.s3_endpoint is None and s.s3_access_key_id is None + assert s.extension_directory is None and s.data_inlining_row_limit is None + assert s.uses_s3 + + +def test_endpoint_defaults_to_path_style_and_overrides_apply(): + s = DuckLakeSettings.from_env({ + **ENV, "DUCKDB_S3_ENDPOINT": "localhost:19100", "DUCKDB_S3_USE_SSL": "false", + "DUCKDB_S3_ACCESS_KEY_ID": "k", "DUCKDB_S3_SECRET_ACCESS_KEY": "s", + "DUCKDB_EXTENSION_DIRECTORY": "/opt/x", "DUCKDB_DATA_INLINING_ROW_LIMIT": "0", + "DUCKDB_CATALOG_PORT": "15599", "DUCKDB_CATALOG_PASSWORD": "pw", + }) + assert (s.s3_url_style, s.s3_use_ssl, s.data_inlining_row_limit) == ("path", False, 0) + assert (s.catalog_port, s.catalog_password, s.extension_directory) == ("15599", "pw", "/opt/x") + + +@pytest.mark.parametrize("name,value", [ + ("DUCKDB_DATA_INLINING_ROW_LIMIT", "many"), ("DUCKDB_S3_URL_STYLE", "sideways"), + ("DUCKDB_CATALOG_PORT", "five"), +]) +def test_invalid_values_name_the_variable(name, value): + with pytest.raises(ValueError, match=name): + DuckLakeSettings.from_env({**ENV, name: value}) + + +def test_local_data_path_does_not_use_s3(): + assert not DuckLakeSettings.from_env({**ENV, "DUCKDB_DATA_PATH": "/data/lake/"}).uses_s3 + + +def test_quoting_helpers(): + assert quote_identifier('we"ird') == '"we""ird"' + assert quote_identifier("50%") == '"50%"' + assert sql_literal("it's") == "'it''s'" + + +def test_attach_statement_plain(): + s = DuckLakeSettings.from_env({**ENV, "DUCKDB_CATALOG_PORT": "15599", "DUCKDB_CATALOG_PASSWORD": "continuo"}) + assert attach_statement(s, "lake") == ( + "ATTACH 'ducklake:postgres:host=localhost port=15599 dbname=catalog user=continuo " + "password=continuo' AS \"lake\" (DATA_PATH 's3://warehouse/lake/')" + ) + + +def test_attach_statement_with_inlining_limit(): + s = DuckLakeSettings.from_env({**ENV, "DUCKDB_DATA_INLINING_ROW_LIMIT": "0"}) + assert attach_statement(s, "lake").endswith("(DATA_PATH 's3://warehouse/lake/', DATA_INLINING_ROW_LIMIT 0)") + + +def test_attach_statement_escapes_awkward_passwords(): + s = DuckLakeSettings.from_env({**ENV, "DUCKDB_CATALOG_PASSWORD": "p w'd\\x"}) + statement = attach_statement(s, "lake") + # undo the SQL-literal doubling, then the libpq quoting must be intact + assert "password='p w\\'d\\\\x'" in statement.replace("''", "'") + # an empty password renders as libpq '' which the SQL literal doubles to '''' + empty = DuckLakeSettings.from_env({**ENV, "DUCKDB_CATALOG_PASSWORD": ""}) + assert "password=''''" in attach_statement(empty, "lake") + + +def test_s3_secret_with_static_credentials(): + s = DuckLakeSettings.from_env({ + **ENV, "DUCKDB_S3_ENDPOINT": "localhost:19100", "DUCKDB_S3_USE_SSL": "false", + "DUCKDB_S3_ACCESS_KEY_ID": "k'1", "DUCKDB_S3_SECRET_ACCESS_KEY": "s", + }) + assert s3_secret_statement(s) == ( + "CREATE OR REPLACE SECRET continuo_s3 (TYPE S3, KEY_ID 'k''1', SECRET 's', " + "ENDPOINT 'localhost:19100', REGION 'us-east-1', URL_STYLE 'path', USE_SSL false)" + ) + + +def test_s3_secret_falls_back_to_the_credential_chain(): + statement = s3_secret_statement(DuckLakeSettings.from_env(ENV)) + assert statement == ( + "CREATE OR REPLACE SECRET continuo_s3 (TYPE S3, PROVIDER credential_chain, " + "REGION 'us-east-1', URL_STYLE 'vhost', USE_SSL true)" + ) + + +R = DdlRenderer("lake") +T = QualifiedTable.of("s", "t") + + +def test_references_are_catalog_qualified_and_quoted(): + assert R.schema_ref(Identifier("s")) == '"lake"."s"' + assert R.table_ref(T) == '"lake"."s"."t"' + # a schema or table named like the catalog alias must not be ambiguous + assert R.table_ref(QualifiedTable.of("lake", "lake")) == '"lake"."lake"."lake"' + assert R.table_ref(QualifiedTable.of('we"ird', "50%")) == '"lake"."we""ird"."50%"' + assert R.catalog_name == "lake" + + +def test_schema_and_table_ddl(): + assert R.create_schema_if_not_exists(Identifier("s")) == 'CREATE SCHEMA IF NOT EXISTS "lake"."s"' + assert R.drop_schema_cascade(Identifier("s")) == 'DROP SCHEMA IF EXISTS "lake"."s" CASCADE' + assert R.drop_table_if_exists(T) == 'DROP TABLE IF EXISTS "lake"."s"."t"' + + +def test_create_table_renders_types_and_not_null(): + cols = [ColumnDefinition(Identifier("id"), "INTEGER", False), ColumnDefinition(Identifier("ts"), "TIMESTAMP")] + assert R.create_table(T, cols) == 'CREATE TABLE "lake"."s"."t" ("id" INTEGER NOT NULL, "ts" TIMESTAMP)' + assert R.create_table(T, cols, if_not_exists=True).startswith('CREATE TABLE IF NOT EXISTS "lake"."s"."t" (') + + +def test_empty_builds(): + assert R.create_empty_table_as(T, "SELECT 1 AS a -- c") == ( + 'CREATE TABLE "lake"."s"."t" AS (\nSELECT 1 AS a -- c\n) WITH NO DATA' + ) + assert R.create_empty_clone(T, QualifiedTable.of("p", "t")) == ( + 'CREATE TABLE "lake"."s"."t" AS SELECT * FROM "lake"."p"."t" WHERE false' + ) + + +def test_layout_ddl(): + keys = (PartitionKey(Identifier("name")), PartitionKey(Identifier("id"), "bucket", 4), + PartitionKey(Identifier("ts"), "month")) + assert R.set_partitioned_by(T, keys) == ( + 'ALTER TABLE "lake"."s"."t" SET PARTITIONED BY ("name", bucket(4, "id"), month("ts"))' + ) + sort = (SortKey(Identifier("id"), True, False), SortKey(Identifier("name")), SortKey(Identifier("ts"), False, True)) + assert R.set_sorted_by(T, sort) == ( + 'ALTER TABLE "lake"."s"."t" SET SORTED BY ("id" DESC NULLS LAST, "name" ASC, "ts" ASC NULLS FIRST)' + ) + + +def test_read_and_write_statements(): + assert R.explain_read("SELECT 1 -- c") == "EXPLAIN SELECT * FROM (\nSELECT 1 -- c\n) AS __check_binds__" + assert R.delete_all(T) == 'DELETE FROM "lake"."s"."t"' + assert R.insert_select(T, ["a", "b%"], "src") == ( + 'INSERT INTO "lake"."s"."t" ("a", "b%") SELECT "a", "b%" FROM "src"' + ) From 1ca1c25e442df38fb7a0fdaa91b06eba9b774e52 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:31:29 +0200 Subject: [PATCH 07/29] feat(duckdb): DuckLake session, composition root, entry point and compose stack Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../duckdb/continuo_duckdb_adapter/adapter.py | 25 +++ .../infrastructure/session.py | 158 ++++++++++++++++++ adapters/duckdb/pyproject.toml | 3 + adapters/duckdb/tests/conftest.py | 153 +++++++++++++++++ .../tests/test_adapter_duckdb_wiring.py | 25 +++ .../tests/test_integration_duckdb_stack.py | 26 +++ tests/smoke/duckdb-stack/docker-compose.yml | 51 ++++++ tests/test_adapter_naming.py | 4 +- tests/test_contract_pin_consistency.py | 1 + 9 files changed, 444 insertions(+), 2 deletions(-) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/adapter.py create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py create mode 100644 adapters/duckdb/tests/test_adapter_duckdb_wiring.py create mode 100644 adapters/duckdb/tests/test_integration_duckdb_stack.py create mode 100644 tests/smoke/duckdb-stack/docker-compose.yml diff --git a/adapters/duckdb/continuo_duckdb_adapter/adapter.py b/adapters/duckdb/continuo_duckdb_adapter/adapter.py new file mode 100644 index 0000000..f6348a5 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/adapter.py @@ -0,0 +1,25 @@ +"""Composition root: wires settings, session and use cases. Entry-point target. + +The only module that knows every layer. The use cases live in +``application.warehouse.LakeWarehouse``; this class supplies the two hooks the +port leaves to the concrete engine, ``required_env`` and ``from_env``. +""" +from __future__ import annotations + +from .application.warehouse import LakeWarehouse +from .infrastructure.session import DuckLakeSession +from .infrastructure.settings import REQUIRED_ENV, DuckLakeSettings + + +class DuckDBAdapter(LakeWarehouse): + """WarehouseAdapter for DuckDB on a DuckLake (Postgres catalog, S3 data).""" + + @classmethod + def required_env(cls) -> list[str]: + """Vars that must be non-empty before connecting.""" + return list(REQUIRED_ENV) + + @classmethod + def from_env(cls) -> "DuckDBAdapter": + """Connect from DUCKDB_* env (see DuckLakeSettings for every variable).""" + return cls(DuckLakeSession.connect(DuckLakeSettings.from_env())) diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py new file mode 100644 index 0000000..3df3126 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -0,0 +1,158 @@ +"""DuckLakeSession: the LakeGateway implementation over one DuckDB connection. + +The local DuckDB instance is in-memory and holds nothing durable: the catalog +(Postgres) and the data (S3/MinIO Parquet) are the warehouse. Extensions are +LOADed first and only INSTALLed when missing, so an image that baked them in at +build time starts offline and as a non-root user. + +SQL is never logged: the ATTACH and secret statements carry credentials. +""" +from __future__ import annotations + +import contextlib +import logging +import uuid +from collections.abc import Iterator, Sequence +from typing import TYPE_CHECKING + +import duckdb + +from ..application.ports import LakeConflictError, LakeGateway +from ..domain.columns import ColumnDefinition +from ..domain.identifiers import Identifier, QualifiedTable +from ..domain.layout import TableLayout +from .ddl import DdlRenderer, attach_statement, quote_identifier, s3_secret_statement, sql_literal +from .settings import DuckLakeSettings + +if TYPE_CHECKING: # pragma: no cover + import pyarrow as pa # type: ignore[import-untyped] + +logger = logging.getLogger("continuo_duckdb_adapter") + +CATALOG_ALIAS = "lake" +_EXTENSIONS = ("ducklake", "postgres", "httpfs") + + +def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: + try: + con.execute(f"LOAD {name}") + except duckdb.Error: + logger.info("installing duckdb extension %s", name) + con.execute(f"INSTALL {name}") + con.execute(f"LOAD {name}") + + +class DuckLakeSession(LakeGateway): + def __init__(self, connection: "duckdb.DuckDBPyConnection", renderer: DdlRenderer) -> None: + self._con = connection + self._renderer = renderer + + @classmethod + def connect(cls, settings: DuckLakeSettings) -> "DuckLakeSession": + con = duckdb.connect() + try: + if settings.extension_directory: + con.execute(f"SET extension_directory = {sql_literal(settings.extension_directory)}") + for extension in _EXTENSIONS: + _load_extension(con, extension) + if settings.uses_s3: + con.execute(s3_secret_statement(settings)) + con.execute(attach_statement(settings, CATALOG_ALIAS)) + con.execute(f"USE {quote_identifier(CATALOG_ALIAS)}") + except BaseException: + con.close() + raise + return cls(con, DdlRenderer(CATALOG_ALIAS)) + + # --- plumbing ----------------------------------------------------------- + + def _run(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConnection": + try: + if parameters is None: + return self._con.execute(sql) + return self._con.execute(sql, parameters) + except duckdb.TransactionException as exc: + raise LakeConflictError(str(exc)) from exc + + def _rollback(self) -> None: + try: + self._con.execute("ROLLBACK") + except duckdb.Error as exc: + # Never mask the failure that got us here. + logger.warning("rollback failed: %s", exc) + + @contextlib.contextmanager + def transaction(self) -> Iterator[None]: + self._con.execute("BEGIN") + try: + yield + self._run("COMMIT") + except BaseException: + self._rollback() + raise + + # --- LakeGateway -------------------------------------------------------- + + def create_schema_if_not_exists(self, schema: Identifier) -> None: + self._run(self._renderer.create_schema_if_not_exists(schema)) + + def drop_schema_cascade(self, schema: Identifier) -> None: + self._run(self._renderer.drop_schema_cascade(schema)) + + def table_exists(self, table: QualifiedTable) -> bool: + row = self._run( + DdlRenderer.TABLE_EXISTS_QUERY, + [self._renderer.catalog_name, table.schema.name, table.table.name], + ).fetchone() + return bool(row and row[0]) + + def drop_table_if_exists(self, table: QualifiedTable) -> None: + self._run(self._renderer.drop_table_if_exists(table)) + + def create_table( + self, + table: QualifiedTable, + columns: Sequence[ColumnDefinition], + *, + if_not_exists: bool = False, + ) -> None: + self._run(self._renderer.create_table(table, columns, if_not_exists=if_not_exists)) + + def create_empty_table_as(self, table: QualifiedTable, select_sql: str) -> None: + self._run(self._renderer.create_empty_table_as(table, select_sql)) + + def create_empty_clone(self, target: QualifiedTable, source: QualifiedTable) -> None: + self._run(self._renderer.create_empty_clone(target, source)) + + def apply_layout(self, table: QualifiedTable, layout: TableLayout) -> None: + if layout.partition_keys: + logger.info("partitioning %s.%s", table.schema.name, table.table.name) + self._run(self._renderer.set_partitioned_by(table, layout.partition_keys)) + if layout.sort_keys: + logger.info("sorting %s.%s", table.schema.name, table.table.name) + self._run(self._renderer.set_sorted_by(table, layout.sort_keys)) + + def explain_read(self, sql: str) -> None: + # EXPLAIN binds without scanning; the transaction is always rolled back. + self._con.execute("BEGIN") + try: + self._run(self._renderer.explain_read(sql)) + finally: + self._rollback() + + def fetch_arrow(self, sql: str) -> "pa.Table": + return self._run(sql).to_arrow_table() + + def delete_all(self, table: QualifiedTable) -> None: + self._run(self._renderer.delete_all(table)) + + def insert_arrow(self, table: QualifiedTable, data: "pa.Table") -> None: + view = f"__continuo_load_{uuid.uuid4().hex}" + self._con.register(view, data) + try: + self._run(self._renderer.insert_select(table, data.schema.names, view)) + finally: + self._con.unregister(view) + + def close(self) -> None: + self._con.close() diff --git a/adapters/duckdb/pyproject.toml b/adapters/duckdb/pyproject.toml index 310f095..cf40088 100644 --- a/adapters/duckdb/pyproject.toml +++ b/adapters/duckdb/pyproject.toml @@ -20,6 +20,9 @@ classifiers = [ "Programming Language :: Python :: 3.14", ] +[project.entry-points."continuo_engine.adapters"] +duckdb = "continuo_duckdb_adapter.adapter:DuckDBAdapter" + [dependency-groups] dev = [ "ruff==0.16.5", diff --git a/adapters/duckdb/tests/conftest.py b/adapters/duckdb/tests/conftest.py index 27fd438..6ef2fd1 100644 --- a/adapters/duckdb/tests/conftest.py +++ b/adapters/duckdb/tests/conftest.py @@ -6,9 +6,19 @@ importlib mode; tests receive it through the ``gateway`` fixture. """ import contextlib +import os +import socket +import uuid +from contextlib import suppress +import boto3 +import psycopg2 +import pyarrow as pa import pytest +from continuo_duckdb_adapter.adapter import DuckDBAdapter +from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession +from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings from continuo_duckdb_adapter.application.ports import LakeConflictError, LakeGateway from continuo_duckdb_adapter.application.warehouse import LakeWarehouse @@ -106,3 +116,146 @@ def gateway() -> FakeLakeGateway: @pytest.fixture def warehouse(gateway) -> _UnitWarehouse: return _UnitWarehouse(gateway, sleep=lambda _seconds: None) + + +CATALOG_PORT = int(os.environ.get("VR_IT_DUCKDB_CATALOG_PORT", "15599")) +S3_PORT = int(os.environ.get("VR_IT_DUCKDB_S3_PORT", "19100")) +BUCKET = "warehouse" +DATA_PATH = f"s3://{BUCKET}/lake/" +STACK_HINT = "docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait" + + +def _require_open(port: int) -> None: + """Fail loudly (never skip) when the compose stack is not running.""" + try: + with socket.create_connection(("127.0.0.1", port), timeout=2): + return + except OSError as exc: + pytest.fail(f"nothing listening on 127.0.0.1:{port}; start the stack: {STACK_HINT} ({exc})") + + +@pytest.fixture(scope="session") +def lake_env() -> dict[str, str]: + """DUCKDB_* env for the compose stack; initialises the catalog exactly once. + + DuckLake creates its metadata tables on the first ATTACH, so that must not + race with the concurrency tests: do it here, once, before any test runs. + """ + _require_open(CATALOG_PORT) + _require_open(S3_PORT) + env = { + "DUCKDB_CATALOG_HOST": "localhost", + "DUCKDB_CATALOG_PORT": str(CATALOG_PORT), + "DUCKDB_CATALOG_DB": "catalog", + "DUCKDB_CATALOG_USER": "continuo", + "DUCKDB_CATALOG_PASSWORD": "continuo", + "DUCKDB_DATA_PATH": DATA_PATH, + "DUCKDB_S3_ENDPOINT": f"localhost:{S3_PORT}", + "DUCKDB_S3_ACCESS_KEY_ID": "minioadmin", + "DUCKDB_S3_SECRET_ACCESS_KEY": "minioadmin", + "DUCKDB_S3_URL_STYLE": "path", + "DUCKDB_S3_USE_SSL": "false", + } + DuckLakeSession.connect(DuckLakeSettings.from_env(env)).close() + return env + + +@pytest.fixture +def adapter_factory(lake_env, monkeypatch): + """Build real adapters against the stack; all are closed at teardown.""" + made: list[DuckDBAdapter] = [] + + def make(**extra_env: str) -> DuckDBAdapter: + for key, value in {**lake_env, **extra_env}.items(): + monkeypatch.setenv(key, value) + adapter = DuckDBAdapter.from_env() + made.append(adapter) + return adapter + + yield make + for adapter in made: + with suppress(Exception): + adapter.close() + + +@pytest.fixture +def adapter(adapter_factory) -> DuckDBAdapter: + return adapter_factory() + + +@pytest.fixture +def parquet_adapter(adapter_factory) -> DuckDBAdapter: + """Inlining off: every insert becomes a Parquet file, so layout is observable.""" + return adapter_factory(DUCKDB_DATA_INLINING_ROW_LIMIT="0") + + +@pytest.fixture +def schema(adapter) -> str: + name = f"it_{uuid.uuid4().hex[:10]}" + yield name + adapter.drop_schema(name) + + +@pytest.fixture +def prod_table(adapter): + """A seeded ``(schema, 'src_table')`` with two rows, dropped afterwards.""" + name = f"prod_{uuid.uuid4().hex[:10]}" + adapter.ensure_table( + name, "src_table", + [{"name": "id", "type": "INTEGER", "nullable": True}, + {"name": "name", "type": "VARCHAR(20)", "nullable": True}], + config={}, + ) + adapter.load(name, "src_table", pa.table({"id": pa.array([1, 2], pa.int32()), "name": ["a", "b"]})) + yield name, "src_table" + adapter.drop_schema(name) + + +@pytest.fixture +def scalar(adapter): + def run(sql: str): + table = adapter.fetch(sql) + return table.to_pylist()[0][table.schema.names[0]] + return run + + +@pytest.fixture +def columns_of(adapter): + def run(schema: str, table: str) -> list[tuple[str, str, str]]: + rows = adapter.fetch( + "SELECT column_name, data_type, is_nullable FROM information_schema.columns " + f"WHERE table_catalog = 'lake' AND table_schema = '{schema}' " + f"AND table_name = '{table}' ORDER BY ordinal_position" + ).to_pylist() + return [(r["column_name"], r["data_type"], r["is_nullable"]) for r in rows] + return run + + +@pytest.fixture +def tables_in(adapter): + def run(schema: str) -> list[str]: + rows = adapter.fetch( + "SELECT table_name FROM information_schema.tables " + f"WHERE table_catalog = 'lake' AND table_schema = '{schema}' ORDER BY table_name" + ).to_pylist() + return [r["table_name"] for r in rows] + return run + + +@pytest.fixture +def catalog_db(): + """A read-only-by-convention cursor on the DuckLake catalog (postgres).""" + conn = psycopg2.connect( + host="localhost", port=CATALOG_PORT, dbname="catalog", user="continuo", password="continuo" + ) + conn.autocommit = True + yield conn.cursor() + conn.close() + + +@pytest.fixture +def s3(): + return boto3.client( + "s3", endpoint_url=f"http://localhost:{S3_PORT}", + aws_access_key_id="minioadmin", aws_secret_access_key="minioadmin", region_name="us-east-1", + ) diff --git a/adapters/duckdb/tests/test_adapter_duckdb_wiring.py b/adapters/duckdb/tests/test_adapter_duckdb_wiring.py new file mode 100644 index 0000000..8f5673c --- /dev/null +++ b/adapters/duckdb/tests/test_adapter_duckdb_wiring.py @@ -0,0 +1,25 @@ +"""Composition root wiring that needs no running warehouse.""" +from importlib.metadata import entry_points + +import pytest + +from continuo_duckdb_adapter.adapter import DuckDBAdapter + + +def test_required_env_names_connection_vars(): + assert DuckDBAdapter.required_env() == [ + "DUCKDB_CATALOG_HOST", "DUCKDB_CATALOG_DB", "DUCKDB_CATALOG_USER", "DUCKDB_DATA_PATH", + ] + + +def test_entry_point_registered_and_loads_adapter(): + eps = [ep for ep in entry_points(group="continuo_engine.adapters") if ep.name == "duckdb"] + assert len(eps) == 1 + assert eps[0].load() is DuckDBAdapter + + +def test_from_env_names_the_missing_variable(monkeypatch): + for name in DuckDBAdapter.required_env(): + monkeypatch.delenv(name, raising=False) + with pytest.raises(ValueError, match="DUCKDB_CATALOG_HOST"): + DuckDBAdapter.from_env() diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py new file mode 100644 index 0000000..96d3885 --- /dev/null +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -0,0 +1,26 @@ +"""The adapter connects to the real DuckLake stack, and fails cleanly when it cannot.""" +import duckdb +import pytest + +pytestmark = pytest.mark.integration + + +def test_from_env_connects_and_serves_a_read(adapter): + assert adapter.fetch("SELECT 41 + 1 AS answer").to_pylist() == [{"answer": 42}] + + +def test_ensure_schema_creates_a_schema_visible_to_another_connection(adapter, adapter_factory, schema): + adapter.ensure_schema(schema) + other = adapter_factory() + other.ensure_table(schema, "t", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + assert other.fetch(f'SELECT count(*) AS n FROM "{schema}"."t"').to_pylist() == [{"n": 0}] + + +def test_unreachable_catalog_raises_a_clear_error(adapter_factory): + with pytest.raises(duckdb.Error): + adapter_factory(DUCKDB_CATALOG_PORT="1") + + +def test_wrong_catalog_password_raises_a_clear_error(adapter_factory): + with pytest.raises(duckdb.Error): + adapter_factory(DUCKDB_CATALOG_PASSWORD="definitely-wrong") diff --git a/tests/smoke/duckdb-stack/docker-compose.yml b/tests/smoke/duckdb-stack/docker-compose.yml new file mode 100644 index 0000000..c72cb6d --- /dev/null +++ b/tests/smoke/duckdb-stack/docker-compose.yml @@ -0,0 +1,51 @@ +# Tier-2 integration stack for adapters/duckdb: a DuckLake is a Postgres catalog +# plus Parquet data on S3, so this is postgres + minio. Used by the +# adapters/duckdb integration suite (`pytest adapters/duckdb/tests -m +# integration`) and by the Dockerfile.duckdb image smoke test. Non-default host +# ports (15599 catalog, 19100 S3) to avoid colliding with other local stacks. +# The catalog records its DATA_PATH on first attach, so `down -v` between runs +# that change it. +services: + catalog: + image: postgres:16 + environment: + POSTGRES_USER: continuo + POSTGRES_PASSWORD: continuo + POSTGRES_DB: catalog + ports: + - "15599:5432" + healthcheck: + test: ["CMD-SHELL", "pg_isready -U continuo -d catalog"] + interval: 2s + timeout: 2s + retries: 30 + + minio: + # ghcr mirror of Bitnami's minio image; its ENTRYPOINT is Bitnami's setup + # script, so it is overridden to reach `minio` directly (see the trino + # stack for the full reasoning). + image: ghcr.io/carolsimone/continuo-minio:2025.7.23 + entrypoint: minio + command: server /bitnami/minio/data + environment: + MINIO_ROOT_USER: minioadmin + MINIO_ROOT_PASSWORD: minioadmin + ports: + - "19100:9000" + healthcheck: + test: ["CMD", "mc", "ready", "local"] + interval: 2s + timeout: 2s + retries: 30 + + createbuckets: + image: ghcr.io/carolsimone/continuo-mc:2025.7.21 + depends_on: + minio: + condition: service_healthy + entrypoint: > + /bin/sh -c " + mc alias set local http://minio:9000 minioadmin minioadmin && + mc mb --ignore-existing local/warehouse && + echo bucket-ready + " diff --git a/tests/test_adapter_naming.py b/tests/test_adapter_naming.py index 5a11b2c..0026b60 100644 --- a/tests/test_adapter_naming.py +++ b/tests/test_adapter_naming.py @@ -15,9 +15,9 @@ def _entry_points(): return list(md.entry_points(group=ENTRY_POINT_GROUP)) -def test_at_least_the_two_known_engines_are_installed(): +def test_the_known_engines_are_installed(): names = {ep.name for ep in _entry_points()} - assert {"postgres", "trino"} <= names, f"missing engines; found {sorted(names)}" + assert {"postgres", "trino", "duckdb"} <= names, f"missing engines; found {sorted(names)}" @pytest.mark.parametrize("ep", _entry_points(), ids=lambda ep: ep.name) diff --git a/tests/test_contract_pin_consistency.py b/tests/test_contract_pin_consistency.py index 868da56..5760bc4 100644 --- a/tests/test_contract_pin_consistency.py +++ b/tests/test_contract_pin_consistency.py @@ -20,6 +20,7 @@ "pyproject.toml", "adapters/postgres/pyproject.toml", "adapters/trino/pyproject.toml", + "adapters/duckdb/pyproject.toml", ] _PIN = re.compile(r"continuo-engine-contract==([0-9][0-9a-z.]*)") From 1fc726f42776b560dc2d9fe394651dd2eb1096ff Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:33:36 +0200 Subject: [PATCH 08/29] test(duckdb): validation behaviours against the real DuckLake stack Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../test_integration_duckdb_validation.py | 171 ++++++++++++++++++ 1 file changed, 171 insertions(+) create mode 100644 adapters/duckdb/tests/test_integration_duckdb_validation.py diff --git a/adapters/duckdb/tests/test_integration_duckdb_validation.py b/adapters/duckdb/tests/test_integration_duckdb_validation.py new file mode 100644 index 0000000..65c6b99 --- /dev/null +++ b/adapters/duckdb/tests/test_integration_duckdb_validation.py @@ -0,0 +1,171 @@ +"""Validation behaviours against a real DuckLake (postgres catalog + minio data).""" +import concurrent.futures + +import duckdb +import pyarrow as pa +import pytest + +pytestmark = pytest.mark.integration + + +def _schema_exists(adapter, schema: str) -> bool: + rows = adapter.fetch( + "SELECT count(*) AS n FROM information_schema.schemata " + f"WHERE catalog_name = 'lake' AND schema_name = '{schema}'" + ).to_pylist() + return rows[0]["n"] == 1 + + +def test_ensure_schema_creates_and_is_idempotent(adapter, schema): + assert not _schema_exists(adapter, schema) + adapter.ensure_schema(schema) + adapter.ensure_schema(schema) + assert _schema_exists(adapter, schema) + + +def test_ensure_schema_race_all_callers_succeed(adapter_factory, adapter, schema): + def call(_): + adapter_factory().ensure_schema(schema) + + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(call, range(8))) # any failure propagates here + assert _schema_exists(adapter, schema) + + +def test_drop_schema_removes_a_schema_containing_tables(adapter, schema, tables_in): + adapter.ensure_table(schema, "a", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + adapter.drop_schema(schema) + assert not _schema_exists(adapter, schema) + assert tables_in(schema) == [] + + +def test_drop_schema_on_an_absent_schema_is_a_noop(adapter, schema): + adapter.drop_schema(schema) + adapter.drop_schema(schema) + + +def test_build_empty_from_sql_creates_an_empty_table_with_the_reads_shape( + adapter, schema, prod_table, columns_of, scalar +): + prod, src = prod_table + adapter.ensure_schema(schema) + adapter.build_empty_from_sql(schema, "m", f'SELECT id, name FROM "{prod}"."{src}";') + assert columns_of(schema, "m") == [("id", "INTEGER", "YES"), ("name", "VARCHAR", "YES")] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."m"') == 0 + + +def test_build_is_rerun_idempotent(adapter, schema, prod_table, columns_of): + prod, src = prod_table + adapter.ensure_schema(schema) + sql = f'SELECT id FROM "{prod}"."{src}"' + adapter.build_empty_from_sql(schema, "m", sql) + adapter.build_empty_from_sql(schema, "m", sql) + assert columns_of(schema, "m") == [("id", "INTEGER", "YES")] + + +def test_build_empty_from_sql_keeps_a_read_ending_in_a_line_comment(adapter, schema, prod_table, columns_of): + prod, src = prod_table + adapter.ensure_schema(schema) + adapter.build_empty_from_sql(schema, "m", f'SELECT id FROM "{prod}"."{src}" -- trailing') + assert columns_of(schema, "m") == [("id", "INTEGER", "YES")] + + +def test_clone_empty_from_prod_copies_the_shape_not_the_rows(adapter, schema, prod_table, columns_of, scalar): + prod, src = prod_table + adapter.ensure_schema(schema) + adapter.clone_empty_from_prod(schema, prod, src) + assert columns_of(schema, src) == [("id", "INTEGER", "YES"), ("name", "VARCHAR", "YES")] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."{src}"') == 0 + + +def test_build_empty_from_columns_creates_a_typed_empty_table(adapter, schema, columns_of, scalar): + adapter.ensure_schema(schema) + adapter.build_empty_from_columns( + schema, "typed", + [{"name": "id", "type": "BIGINT", "nullable": False}, + {"name": "label", "type": "VARCHAR(10)", "nullable": True}, + {"name": "amount", "type": "NUMERIC(10,2)", "nullable": False}], + {}, + ) + assert columns_of(schema, "typed") == [ + ("id", "BIGINT", "NO"), ("label", "VARCHAR", "YES"), ("amount", "DECIMAL(10,2)", "NO"), + ] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."typed"') == 0 + + +def test_not_null_is_enforced_on_load(adapter, schema): + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", [{"name": "id", "type": "INTEGER", "nullable": False}], {}) + with pytest.raises(duckdb.Error): + adapter.load(schema, "t", pa.table({"id": pa.array([None], pa.int32())})) + + +def test_build_empty_from_columns_rebuilds_on_rerun(adapter, schema, columns_of): + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", [{"name": "a", "type": "INTEGER"}], {}) + adapter.build_empty_from_columns(schema, "t", [{"name": "b", "type": "TEXT"}, {"name": "c", "type": "DATE"}], {}) + assert columns_of(schema, "t") == [("b", "VARCHAR", "YES"), ("c", "DATE", "YES")] + + +def test_an_engine_rejected_rebuild_rolls_back_and_keeps_the_prior_table(adapter, schema, columns_of, scalar): + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", [{"name": "keep", "type": "INTEGER"}], {}) + adapter.load(schema, "t", pa.table({"keep": pa.array([1], pa.int32())})) + with pytest.raises(duckdb.Error): # duplicate column names: rejected by the engine after the DROP + adapter.build_empty_from_columns( + schema, "t", [{"name": "dup", "type": "INTEGER"}, {"name": "dup", "type": "INTEGER"}], {} + ) + assert columns_of(schema, "t") == [("keep", "INTEGER", "YES")] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."t"') == 1 + adapter.fetch("SELECT 1 AS still_usable") # the connection is not left in an aborted transaction + + +@pytest.mark.parametrize("read", [ + "SELECT id, name FROM {prod}.src_table", + "SELECT id FROM {prod}.src_table -- trailing comment", + "SELECT ';' AS semi", + "WITH c AS (SELECT id FROM {prod}.src_table) SELECT id FROM c", + "VALUES (1), (2)", +]) +def test_check_binds_passes_a_valid_single_read(adapter, prod_table, read): + prod, _ = prod_table + adapter.check_binds(read.format(prod=f'"{prod}"')) + + +def test_check_binds_raises_on_a_missing_column(adapter, prod_table): + prod, _ = prod_table + with pytest.raises(duckdb.Error): + adapter.check_binds(f'SELECT nope FROM "{prod}".src_table') + + +def test_check_binds_raises_on_a_missing_table(adapter, prod_table): + prod, _ = prod_table + with pytest.raises(duckdb.Error): + adapter.check_binds(f'SELECT id FROM "{prod}".does_not_exist') + + +def test_check_binds_scans_no_data(adapter, prod_table, scalar): + prod, src = prod_table + adapter.check_binds(f'SELECT id FROM "{prod}"."{src}"') + assert scalar(f'SELECT count(*) AS n FROM "{prod}"."{src}"') == 2 # untouched + + +@pytest.mark.parametrize("attack", [ + 'SELECT 1; DROP TABLE "{s}"."victim"', + 'SELECT 1) AS x; DROP TABLE "{s}"."victim"; SELECT * FROM (SELECT 1', + 'DELETE FROM "{s}"."victim"', + 'DROP TABLE "{s}"."victim"', +]) +def test_check_binds_rejects_stacked_and_non_read_statements_and_executes_nothing(adapter, schema, tables_in, attack): + adapter.ensure_table(schema, "victim", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + with pytest.raises(ValueError): + adapter.check_binds(attack.format(s=schema)) + assert tables_in(schema) == ["victim"] + + +def test_check_binds_leaves_the_connection_usable_after_a_failed_read(adapter, prod_table): + prod, _ = prod_table + with pytest.raises(duckdb.Error): + adapter.check_binds(f'SELECT nope FROM "{prod}".src_table') + assert adapter.fetch("SELECT 1 AS ok").to_pylist() == [{"ok": 1}] + adapter.check_binds(f'SELECT id FROM "{prod}".src_table') From 9701ff08b7d6658a7f69cb4adfd0e762482d08bb Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:35:51 +0200 Subject: [PATCH 09/29] test(duckdb): make the check_binds safety tests able to fail Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../test_integration_duckdb_validation.py | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/adapters/duckdb/tests/test_integration_duckdb_validation.py b/adapters/duckdb/tests/test_integration_duckdb_validation.py index 65c6b99..05c69d6 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_validation.py +++ b/adapters/duckdb/tests/test_integration_duckdb_validation.py @@ -24,6 +24,8 @@ def test_ensure_schema_creates_and_is_idempotent(adapter, schema): def test_ensure_schema_race_all_callers_succeed(adapter_factory, adapter, schema): + assert not _schema_exists(adapter, schema) + def call(_): adapter_factory().ensure_schema(schema) @@ -93,11 +95,12 @@ def test_build_empty_from_columns_creates_a_typed_empty_table(adapter, schema, c assert scalar(f'SELECT count(*) AS n FROM "{schema}"."typed"') == 0 -def test_not_null_is_enforced_on_load(adapter, schema): +def test_not_null_is_enforced_on_load(adapter, schema, scalar): adapter.ensure_schema(schema) adapter.build_empty_from_columns(schema, "t", [{"name": "id", "type": "INTEGER", "nullable": False}], {}) - with pytest.raises(duckdb.Error): + with pytest.raises(duckdb.Error, match="(?i)not null"): adapter.load(schema, "t", pa.table({"id": pa.array([None], pa.int32())})) + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."t"') == 0 def test_build_empty_from_columns_rebuilds_on_rerun(adapter, schema, columns_of): @@ -146,21 +149,28 @@ def test_check_binds_raises_on_a_missing_table(adapter, prod_table): def test_check_binds_scans_no_data(adapter, prod_table, scalar): prod, src = prod_table - adapter.check_binds(f'SELECT id FROM "{prod}"."{src}"') + # error() fires only if a row is evaluated (id is 1 and 2, so the branch is taken at + # runtime); EXPLAIN binds the expression without running it. + adapter.check_binds(f'SELECT CASE WHEN id > 0 THEN error(\'scanned\') ELSE 0 END AS v FROM "{prod}"."{src}"') assert scalar(f'SELECT count(*) AS n FROM "{prod}"."{src}"') == 2 # untouched @pytest.mark.parametrize("attack", [ 'SELECT 1; DROP TABLE "{s}"."victim"', + 'SELECT \';\'; DROP TABLE "{s}"."victim"', 'SELECT 1) AS x; DROP TABLE "{s}"."victim"; SELECT * FROM (SELECT 1', 'DELETE FROM "{s}"."victim"', 'DROP TABLE "{s}"."victim"', ]) -def test_check_binds_rejects_stacked_and_non_read_statements_and_executes_nothing(adapter, schema, tables_in, attack): +def test_check_binds_rejects_stacked_and_non_read_statements_and_executes_nothing( + adapter, schema, tables_in, scalar, attack +): adapter.ensure_table(schema, "victim", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + adapter.load(schema, "victim", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) with pytest.raises(ValueError): adapter.check_binds(attack.format(s=schema)) assert tables_in(schema) == ["victim"] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."victim"') == 3 # no row deleted def test_check_binds_leaves_the_connection_usable_after_a_failed_read(adapter, prod_table): From 8237a53e97c18e84a0573ebff1990b60b516d6e7 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:38:54 +0200 Subject: [PATCH 10/29] test(duckdb): runtime behaviours and csv node end to end against the real stack Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../tests/test_integration_duckdb_runtime.py | 202 ++++++++++++++++++ 1 file changed, 202 insertions(+) create mode 100644 adapters/duckdb/tests/test_integration_duckdb_runtime.py diff --git a/adapters/duckdb/tests/test_integration_duckdb_runtime.py b/adapters/duckdb/tests/test_integration_duckdb_runtime.py new file mode 100644 index 0000000..d0962d3 --- /dev/null +++ b/adapters/duckdb/tests/test_integration_duckdb_runtime.py @@ -0,0 +1,202 @@ +"""Python-node data-plane behaviours against a real DuckLake, plus the csv node end to end.""" +import concurrent.futures +from datetime import date, datetime +from decimal import Decimal + +import duckdb +import pyarrow as pa +import pytest +import yaml + +from continuo_python_runtime.harness import run_node + +pytestmark = pytest.mark.integration + +BUCKET = "warehouse" +ID = [{"name": "id", "type": "INTEGER", "nullable": True}] + + +def test_ensure_table_creates_a_typed_table_with_not_null(adapter, schema, columns_of): + adapter.ensure_table( + schema, "typed", + [{"name": "id", "type": "BIGINT", "nullable": False}, + {"name": "label", "type": "VARCHAR(10)", "nullable": True}, + {"name": "amount", "type": "NUMERIC(10,2)", "nullable": False}], + config={}, + ) + assert columns_of(schema, "typed") == [ + ("id", "BIGINT", "NO"), ("label", "VARCHAR", "YES"), ("amount", "DECIMAL(10,2)", "NO"), + ] + + +def test_ensure_table_is_idempotent(adapter, schema, columns_of): + adapter.ensure_table(schema, "again", ID, config={}) + adapter.ensure_table(schema, "again", ID, config={}) + assert columns_of(schema, "again") == [("id", "INTEGER", "YES")] + + +def test_ensure_table_does_not_touch_an_existing_tables_data(adapter, schema, scalar): + adapter.ensure_table(schema, "keep", ID, config={}) + adapter.load(schema, "keep", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) + adapter.ensure_table(schema, "keep", ID, config={}) + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."keep"') == 3 + + +def test_concurrent_ensure_table_all_callers_succeed(adapter_factory, schema, columns_of): + def call(_): + adapter_factory().ensure_table(schema, "race", ID, config={}) + + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(call, range(8))) + assert columns_of(schema, "race") == [("id", "INTEGER", "YES")] + + +def test_fetch_round_trips_every_contract_type(adapter, schema): + adapter.ensure_table( + schema, "typed", + [{"name": "i", "type": "BIGINT", "nullable": False}, + {"name": "d", "type": "NUMERIC(10,2)", "nullable": True}, + {"name": "dt", "type": "DATE", "nullable": True}, + {"name": "ts", "type": "TIMESTAMP", "nullable": True}, + {"name": "b", "type": "BOOLEAN", "nullable": True}, + {"name": "s", "type": "TEXT", "nullable": True}], + config={}, + ) + data = pa.table({ + "i": pa.array([1, 2], pa.int64()), + "d": pa.array([Decimal("1.50"), None], pa.decimal128(10, 2)), + "dt": pa.array([date(2026, 1, 2), None], pa.date32()), + "ts": pa.array([datetime(2026, 1, 2, 3, 4, 5), None], pa.timestamp("us")), + "b": pa.array([True, None]), + "s": pa.array(["x", None]), + }) + adapter.load(schema, "typed", data) + out = adapter.fetch(f'SELECT * FROM "{schema}"."typed" ORDER BY i') + assert out.to_pylist() == data.to_pylist() + + +def test_fetch_rejects_duplicate_select_columns(adapter): + with pytest.raises(ValueError, match="duplicate column name"): + adapter.fetch("SELECT 1 AS id, 2 AS id") + + +def test_fetch_of_an_empty_result_keeps_the_declared_shape(adapter, schema): + adapter.ensure_table(schema, "empty", ID, config={}) + out = adapter.fetch(f'SELECT id FROM "{schema}"."empty"') + assert out.num_rows == 0 and out.schema.names == ["id"] + + +def test_load_replaces_contents_atomically(adapter, adapter_factory, schema, scalar): + adapter.ensure_table(schema, "t", ID, config={}) + adapter.load(schema, "t", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) + + # Observe the table from a second connection in the window between the delete + # and the insert: a transactional replace must still show the old contents there. + observer = adapter_factory() + seen_mid_load: list[list[dict]] = [] + gateway = adapter._gateway + real_insert = gateway.insert_arrow + + def insert_then_peek(table, data): + seen_mid_load.append(observer.fetch(f'SELECT id FROM "{schema}"."t" ORDER BY id').to_pylist()) + real_insert(table, data) + + gateway.insert_arrow = insert_then_peek + try: + adapter.load(schema, "t", pa.table({"id": pa.array([9], pa.int32())})) + finally: + gateway.insert_arrow = real_insert + + assert seen_mid_load == [[{"id": 1}, {"id": 2}, {"id": 3}]] + assert adapter.fetch(f'SELECT id FROM "{schema}"."t"').to_pylist() == [{"id": 9}] + + +def test_load_of_zero_rows_just_clears(adapter, schema, scalar): + adapter.ensure_table(schema, "t", ID, config={}) + adapter.load(schema, "t", pa.table({"id": pa.array([1], pa.int32())})) + adapter.load(schema, "t", pa.table({"id": pa.array([], pa.int32())})) + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."t"') == 0 + + +def test_a_failed_load_leaves_prior_contents_intact(adapter, schema): + adapter.ensure_table(schema, "t", [{"name": "id", "type": "INTEGER", "nullable": False}], config={}) + adapter.load(schema, "t", pa.table({"id": pa.array([1, 2], pa.int32())})) + with pytest.raises(duckdb.Error): # NOT NULL violated part-way through the insert + adapter.load(schema, "t", pa.table({"id": pa.array([5, None], pa.int32())})) + assert adapter.fetch(f'SELECT id FROM "{schema}"."t" ORDER BY id').to_pylist() == [{"id": 1}, {"id": 2}] + adapter.load(schema, "t", pa.table({"id": pa.array([7], pa.int32())})) # connection still usable + + +def test_load_accepts_an_all_null_column(adapter, schema): + adapter.ensure_table( + schema, "t", [{"name": "id", "type": "INTEGER", "nullable": True}, {"name": "note", "type": "TEXT", "nullable": True}], + config={}, + ) + adapter.load(schema, "t", pa.table({"id": pa.array([1, 2], pa.int32()), "note": pa.nulls(2)})) + assert adapter.fetch(f'SELECT id, note FROM "{schema}"."t" ORDER BY id').to_pylist() == [ + {"id": 1, "note": None}, {"id": 2, "note": None}, + ] + + +def test_awkward_identifiers_round_trip(adapter, schema): + cols = [{"name": 'order"id', "type": "BIGINT", "nullable": False}, + {"name": "pct%", "type": "TEXT", "nullable": True}, + {"name": "order", "type": "TEXT", "nullable": True}] + for table in ("50% order table", "select"): + adapter.ensure_table(schema, table, cols, config={}) + adapter.load(schema, table, pa.table({ + 'order"id': pa.array([7], pa.int64()), "pct%": ["x"], "order": ["y"], + })) + rows = adapter.fetch( + f'SELECT "order""id" AS oid, "pct%" AS p, "order" AS o FROM "{schema}"."{table}"' + ).to_pylist() + assert rows == [{"oid": 7, "p": "x", "o": "y"}] + + +def test_a_schema_named_like_the_catalog_alias_is_not_ambiguous(adapter): + try: + adapter.ensure_table("lake", "lake", ID, config={}) + adapter.load("lake", "lake", pa.table({"id": pa.array([1], pa.int32())})) + assert adapter.fetch('SELECT id FROM "lake"."lake"."lake"').to_pylist() == [{"id": 1}] + finally: + adapter.drop_schema("lake") + + +def _csv_contract_dir(tmp_path, schema): + (tmp_path / "contracts").mkdir() + (tmp_path / "contracts" / "t.yml").write_text(yaml.safe_dump({"nodes": [{ + "schema": schema, "table": "orders_csv", "owner": "m", "schedule": "daily", + "criticality": "SECONDARY", "kind": "python-csv", + "reads": {"csv": f"s3://{BUCKET}/drops/orders.csv"}, + "output_columns": [ + {"name": "order_id", "type": "INTEGER", "nullable": False}, + {"name": "amount", "type": "DOUBLE PRECISION"}, + ], + }]})) + return tmp_path + + +def test_run_node_csv_kind_loads_a_minio_csv_into_duckdb( + adapter, adapter_factory, schema, s3, columns_of, monkeypatch, tmp_path, +): + """The full production path: real minio csv -> harness -> DuckDBAdapter -> real DuckLake.""" + s3.put_object(Bucket=BUCKET, Key="drops/orders.csv", Body=b"order_id,amount\n1,10.5\n2,20.0\n3,5.25\n") + monkeypatch.setenv("S3_ENDPOINT_URL", s3.meta.endpoint_url) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "minioadmin") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "minioadmin") + monkeypatch.setenv("AWS_DEFAULT_REGION", "us-east-1") + repo = _csv_contract_dir(tmp_path, schema) + env = { + "NODE_ID": f"python-csv.svc.{schema}.orders_csv", + "TABLE_NAME": "orders_csv", + "TARGET_SCHEMA": schema, + "CONTRACT_DIR": str(repo / "contracts"), + "APP_ROOT": str(repo), + } + # run_node closes the adapter it is given, so hand it its own and verify through `adapter`. + assert run_node(env, adapter=adapter_factory()) == 0 + assert columns_of(schema, "orders_csv") == [("order_id", "INTEGER", "NO"), ("amount", "DOUBLE", "YES")] + rows = adapter.fetch(f'SELECT order_id, amount FROM "{schema}"."orders_csv" ORDER BY order_id').to_pylist() + assert rows == [ + {"order_id": 1, "amount": 10.5}, {"order_id": 2, "amount": 20.0}, {"order_id": 3, "amount": 5.25}, + ] From 99360022078062882ad91f07e043419da1288bfa Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:41:48 +0200 Subject: [PATCH 11/29] test(duckdb): prove partitioned_by and sorted_by are applied (catalog metadata and Parquet files) Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../tests/test_integration_duckdb_layout.py | 242 ++++++++++++++++++ 1 file changed, 242 insertions(+) create mode 100644 adapters/duckdb/tests/test_integration_duckdb_layout.py diff --git a/adapters/duckdb/tests/test_integration_duckdb_layout.py b/adapters/duckdb/tests/test_integration_duckdb_layout.py new file mode 100644 index 0000000..dbeabc3 --- /dev/null +++ b/adapters/duckdb/tests/test_integration_duckdb_layout.py @@ -0,0 +1,242 @@ +"""partitioned_by / sorted_by are really applied: catalog metadata AND Parquet files.""" +import io + +import pyarrow as pa +import pyarrow.parquet as pq +import pytest + +pytestmark = pytest.mark.integration + +COLS = [ + {"name": "id", "type": "INTEGER", "nullable": True}, + {"name": "ts", "type": "TIMESTAMP", "nullable": True}, + {"name": "name", "type": "VARCHAR(20)", "nullable": True}, +] + +_PARTITION_COLUMNS = """ +SELECT pc.partition_key_index, c.column_name, pc.transform +FROM ducklake_partition_column pc +JOIN ducklake_partition_info pi ON pi.partition_id = pc.partition_id AND pi.table_id = pc.table_id +JOIN ducklake_table t ON t.table_id = pc.table_id +JOIN ducklake_schema s ON s.schema_id = t.schema_id +JOIN ducklake_column c ON c.table_id = pc.table_id AND c.column_id = pc.column_id +WHERE s.schema_name = %s AND t.table_name = %s AND pi.end_snapshot IS NULL + AND t.end_snapshot IS NULL AND c.end_snapshot IS NULL +ORDER BY pc.partition_key_index +""" +_ACTIVE_PARTITION_IDS = """ +SELECT pi.partition_id FROM ducklake_partition_info pi +JOIN ducklake_table t ON t.table_id = pi.table_id +JOIN ducklake_schema s ON s.schema_id = t.schema_id +WHERE s.schema_name = %s AND t.table_name = %s AND pi.end_snapshot IS NULL AND t.end_snapshot IS NULL +""" +_SORT_EXPRESSIONS = """ +SELECT se.expression, se.sort_direction, se.null_order +FROM ducklake_sort_expression se +JOIN ducklake_sort_info si ON si.sort_id = se.sort_id AND si.table_id = se.table_id +JOIN ducklake_table t ON t.table_id = se.table_id +JOIN ducklake_schema s ON s.schema_id = t.schema_id +WHERE s.schema_name = %s AND t.table_name = %s AND si.end_snapshot IS NULL AND t.end_snapshot IS NULL +ORDER BY se.sort_key_index +""" + + +def _partition_keys(cur, schema, table): + cur.execute(_PARTITION_COLUMNS, (schema, table)) + return [(row[1], row[2]) for row in cur.fetchall()] + + +def _active_partition_ids(cur, schema, table): + cur.execute(_ACTIVE_PARTITION_IDS, (schema, table)) + return sorted(row[0] for row in cur.fetchall()) + + +def _sort_keys(cur, schema, table): + cur.execute(_SORT_EXPRESSIONS, (schema, table)) + return [(row[0].strip('"'), row[1], row[2]) for row in cur.fetchall()] + + +def _parquet_files(s3, schema, table) -> dict[str, pa.Table]: + """Every Parquet data file DuckLake wrote for the table, read back from MinIO.""" + listing = s3.list_objects_v2(Bucket="warehouse", Prefix=f"lake/{schema}/{table}/") + files = {} + for item in listing.get("Contents", []): + if item["Key"].endswith(".parquet"): + body = s3.get_object(Bucket="warehouse", Key=item["Key"])["Body"].read() + files[item["Key"]] = pq.read_table(io.BytesIO(body)) + return files + + +# --- partitioned_by: catalog metadata --------------------------------------- + + +@pytest.mark.parametrize("build", ["build_empty_from_columns", "ensure_table"]) +def test_partition_keys_reach_the_catalog(adapter, schema, catalog_db, build): + config = {"partitioned_by": [ + "name", + {"column": "id", "transform": "bucket", "buckets": 4}, + {"column": "ts", "transform": "month"}, + ]} + if build == "ensure_table": + adapter.ensure_table(schema, "t", COLS, config=config) + else: + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", COLS, config) + assert _partition_keys(catalog_db, schema, "t") == [ + ("name", "identity"), ("id", "bucket(4)"), ("ts", "month"), + ] + + +@pytest.mark.parametrize("transform", ["year", "month", "day", "hour"]) +def test_every_time_transform_reaches_the_catalog(adapter, schema, catalog_db, transform): + adapter.ensure_table(schema, "t", COLS, config={"partitioned_by": [{"column": "ts", "transform": transform}]}) + assert _partition_keys(catalog_db, schema, "t") == [("ts", transform)] + + +def test_time_transforms_work_on_a_date_column(adapter, schema, catalog_db): + adapter.ensure_table( + schema, "t", [{"name": "d", "type": "DATE", "nullable": True}], + config={"partitioned_by": [{"column": "d", "transform": "month"}]}, + ) + assert _partition_keys(catalog_db, schema, "t") == [("d", "month")] + + +# --- partitioned_by: physical files ----------------------------------------- + + +def test_partitioned_data_lands_in_one_directory_per_value(parquet_adapter, schema, s3): + parquet_adapter.ensure_table(schema, "events", COLS, config={"partitioned_by": ["name"]}) + parquet_adapter.load(schema, "events", pa.table({ + "id": pa.array([1, 2, 3, 4, 5], pa.int32()), + "ts": pa.array([None] * 5, pa.timestamp("us")), + "name": ["a", "a", "a", "b", "b"], + })) + files = _parquet_files(s3, schema, "events") + rows_per_directory: dict[str, int] = {} + for key, table in files.items(): + directory = key.split("/")[-2] + rows_per_directory[directory] = rows_per_directory.get(directory, 0) + table.num_rows + assert rows_per_directory == {"name=a": 3, "name=b": 2} + + +# --- sorted_by: catalog metadata -------------------------------------------- + + +@pytest.mark.parametrize("build", ["build_empty_from_columns", "ensure_table"]) +def test_sort_keys_reach_the_catalog(adapter, schema, catalog_db, build): + config = {"sorted_by": [{"column": "id", "direction": "desc", "nulls": "last"}, "name"]} + if build == "ensure_table": + adapter.ensure_table(schema, "t", COLS, config=config) + else: + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", COLS, config) + assert _sort_keys(catalog_db, schema, "t") == [ + ("id", "DESC", "NULLS_LAST"), ("name", "ASC", "NULLS_LAST"), + ] + + +def test_explicit_nulls_first_reaches_the_catalog(adapter, schema, catalog_db): + adapter.ensure_table(schema, "t", COLS, config={"sorted_by": [{"column": "id", "nulls": "first"}]}) + assert _sort_keys(catalog_db, schema, "t") == [("id", "ASC", "NULLS_FIRST")] + + +# --- sorted_by: physical files ---------------------------------------------- + + +@pytest.mark.parametrize("direction,nulls,expected", [ + ("asc", "first", [None, 1, 2, 3]), + ("asc", "last", [1, 2, 3, None]), + ("desc", "first", [None, 3, 2, 1]), + ("desc", "last", [3, 2, 1, None]), +]) +def test_rows_are_ordered_inside_the_parquet_file(parquet_adapter, schema, s3, direction, nulls, expected): + parquet_adapter.ensure_table( + schema, "t", COLS, config={"sorted_by": [{"column": "id", "direction": direction, "nulls": nulls}]} + ) + parquet_adapter.load(schema, "t", pa.table({ + "id": pa.array([2, None, 1, 3], pa.int32()), + "ts": pa.array([None] * 4, pa.timestamp("us")), + "name": ["a"] * 4, + })) + files = list(_parquet_files(s3, schema, "t").values()) + assert len(files) == 1 + assert files[0].column("id").to_pylist() == expected + + +def test_partitioned_and_sorted_together(parquet_adapter, schema, catalog_db, s3): + parquet_adapter.ensure_table(schema, "t", COLS, config={ + "partitioned_by": ["name"], "sorted_by": [{"column": "id", "direction": "desc"}], + }) + parquet_adapter.load(schema, "t", pa.table({ + "id": pa.array([1, 3, 2, 8, 9], pa.int32()), + "ts": pa.array([None] * 5, pa.timestamp("us")), + "name": ["a", "a", "a", "b", "b"], + })) + assert _partition_keys(catalog_db, schema, "t") == [("name", "identity")] + assert _sort_keys(catalog_db, schema, "t") == [("id", "DESC", "NULLS_LAST")] + by_directory = {key.split("/")[-2]: table.column("id").to_pylist() + for key, table in _parquet_files(s3, schema, "t").items()} + assert by_directory == {"name=a": [3, 2, 1], "name=b": [9, 8]} + + +# --- fail closed ------------------------------------------------------------ + + +@pytest.mark.parametrize("config", [ + {"indexes": [{"columns": ["id"]}]}, # postgres vocabulary + {"sortkey": ["id"]}, + {"partitioned_by": ["nope"]}, + {"partitioned_by": [{"column": "id", "transform": "month"}]}, # time transform on INTEGER + {"partitioned_by": [{"column": "id", "transform": "week"}]}, + {"partitioned_by": [{"column": "id", "transform": "bucket"}]}, + {"sorted_by": [{"column": "id", "direction": "sideways"}]}, + {"sorted_by": [{"column": "id + 1"}]}, + {"sorted_by": ["id", "id"]}, +]) +@pytest.mark.parametrize("build", ["build_empty_from_columns", "ensure_table"]) +def test_a_bad_layout_raises_and_leaves_no_table(adapter, schema, tables_in, config, build): + adapter.ensure_schema(schema) + with pytest.raises(ValueError): + if build == "ensure_table": + adapter.ensure_table(schema, "t", COLS, config=config) + else: + adapter.build_empty_from_columns(schema, "t", COLS, config) + assert tables_in(schema) == [] + + +def test_an_empty_config_applies_no_layout(adapter, schema, catalog_db): + adapter.ensure_table(schema, "t", COLS, config={}) + assert _partition_keys(catalog_db, schema, "t") == [] + assert _sort_keys(catalog_db, schema, "t") == [] + + +# --- idempotence and rebuild ------------------------------------------------ + + +def test_a_second_ensure_table_with_the_same_layout_changes_nothing(adapter, schema, catalog_db): + config = {"partitioned_by": ["name"], "sorted_by": ["id"]} + adapter.ensure_table(schema, "t", COLS, config=config) + before = _active_partition_ids(catalog_db, schema, "t") + adapter.ensure_table(schema, "t", COLS, config=config) + assert _active_partition_ids(catalog_db, schema, "t") == before + assert len(before) == 1 + + +def test_ensure_table_does_not_change_the_layout_of_an_existing_table(adapter, schema, catalog_db): + adapter.ensure_table(schema, "t", COLS, config={"partitioned_by": ["name"]}) + adapter.ensure_table(schema, "t", COLS, config={"partitioned_by": [{"column": "ts", "transform": "month"}]}) + assert _partition_keys(catalog_db, schema, "t") == [("name", "identity")] + + +def test_a_rebuild_replaces_the_previous_layout(adapter, schema, catalog_db): + adapter.ensure_schema(schema) + adapter.build_empty_from_columns(schema, "t", COLS, {"partitioned_by": ["name"]}) + adapter.build_empty_from_columns( + schema, "t", COLS, {"partitioned_by": [{"column": "ts", "transform": "month"}], "sorted_by": ["id"]} + ) + assert _partition_keys(catalog_db, schema, "t") == [("ts", "month")] + assert len(_active_partition_ids(catalog_db, schema, "t")) == 1 + assert _sort_keys(catalog_db, schema, "t") == [("id", "ASC", "NULLS_LAST")] + adapter.build_empty_from_columns(schema, "t", COLS, {}) + assert _partition_keys(catalog_db, schema, "t") == [] + assert _sort_keys(catalog_db, schema, "t") == [] From e0aee4922e873224c598ddba6834e619ae996b36 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:47:02 +0200 Subject: [PATCH 12/29] feat(duckdb): engine image, version guard and image smoke job Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .github/workflows/images.yml | 85 ++++++++++++++++++++++++++- Dockerfile.duckdb | 41 +++++++++++++ image-requirements-duckdb.txt | 2 + scripts/bake_duckdb_extensions.py | 32 ++++++++++ scripts/check_version_bumps.py | 1 + tests/test_image_requirements_sync.py | 6 ++ 6 files changed, 164 insertions(+), 3 deletions(-) create mode 100644 Dockerfile.duckdb create mode 100644 image-requirements-duckdb.txt create mode 100644 scripts/bake_duckdb_extensions.py diff --git a/.github/workflows/images.yml b/.github/workflows/images.yml index 85e4497..02ab8a8 100644 --- a/.github/workflows/images.yml +++ b/.github/workflows/images.yml @@ -17,7 +17,7 @@ jobs: build: strategy: matrix: - engine: [postgres, trino] + engine: [postgres, trino, duckdb] runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 @@ -156,6 +156,85 @@ jobs: if: always() run: docker compose -f tests/smoke/trino-stack/docker-compose.yml down -v + smoke-duckdb: + needs: build + runs-on: ubuntu-latest + env: + DUCKDB_CATALOG_HOST: localhost + DUCKDB_CATALOG_PORT: "15599" + DUCKDB_CATALOG_DB: catalog + DUCKDB_CATALOG_USER: continuo + DUCKDB_CATALOG_PASSWORD: continuo + DUCKDB_DATA_PATH: s3://warehouse/lake/ + DUCKDB_S3_ENDPOINT: localhost:19100 + DUCKDB_S3_ACCESS_KEY_ID: minioadmin + DUCKDB_S3_SECRET_ACCESS_KEY: minioadmin + DUCKDB_S3_URL_STYLE: path + DUCKDB_S3_USE_SSL: "false" + steps: + - uses: actions/checkout@v7 + - uses: actions/download-artifact@v8 + with: + name: cpr-smoke-duckdb-image + path: /tmp + - name: Load image + run: docker load -i /tmp/cpr-smoke-duckdb.tar + - uses: astral-sh/setup-uv@v7 + - name: Sync + run: uv sync --all-packages --all-groups + - name: Image smoke tests (duckdb) + env: + VALIDATION_IMAGE_UNDER_TEST: cpr-smoke-duckdb + VALIDATION_IMAGE_ENGINE: duckdb + # DuckDBAdapter.required_env()[0] + VALIDATION_IMAGE_REQUIRED_ENV: DUCKDB_CATALOG_HOST + run: uv run pytest tests/test_image_smoke_validation.py -m image -v + - name: Extensions load offline as the non-root user + run: | + docker run --rm --network none --entrypoint python cpr-smoke-duckdb -c \ + "import os, duckdb; c = duckdb.connect(); c.execute('SET extension_directory = ' + repr(os.environ['DUCKDB_EXTENSION_DIRECTORY'])); [c.execute('LOAD ' + e) for e in ('ducklake', 'postgres', 'httpfs')]" + - name: Start duckdb stack + run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait + - name: Seed source table + run: | + uv run python - <<'PY' + import pyarrow as pa + from continuo_duckdb_adapter.adapter import DuckDBAdapter + + a = DuckDBAdapter.from_env() + a.ensure_table("analytics", "src", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + a.load("analytics", "src", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) + a.close() + PY + - name: Run smoke node + run: | + docker run --rm --network host \ + -v "$PWD/tests/smoke/node_smoke:/app" \ + -e NODE_ID=python-node.smoke.analytics.smoke \ + -e TABLE_NAME=smoke \ + -e TARGET_SCHEMA=analytics \ + -e DUCKDB_CATALOG_HOST -e DUCKDB_CATALOG_PORT -e DUCKDB_CATALOG_DB \ + -e DUCKDB_CATALOG_USER -e DUCKDB_CATALOG_PASSWORD -e DUCKDB_DATA_PATH \ + -e DUCKDB_S3_ENDPOINT -e DUCKDB_S3_ACCESS_KEY_ID -e DUCKDB_S3_SECRET_ACCESS_KEY \ + -e DUCKDB_S3_URL_STYLE -e DUCKDB_S3_USE_SSL \ + cpr-smoke-duckdb | tee /tmp/smoke-duckdb.out + test "${PIPESTATUS[0]}" -eq 0 + shell: bash + - name: Assert success sentinel + run: grep -q '"status":"success"' /tmp/smoke-duckdb.out + - name: Assert row count + run: | + uv run python - <<'PY' + from continuo_duckdb_adapter.adapter import DuckDBAdapter + + a = DuckDBAdapter.from_env() + assert a.fetch("SELECT count(*) AS n FROM analytics.smoke").to_pylist() == [{"n": 3}] + a.close() + PY + - name: Tear down duckdb stack + if: always() + run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml down -v + # Runs on this same tag push, after the smoke jobs above pass. It builds both # engine images for linux/amd64 and linux/arm64 installing the pinned # versions FROM PyPI (WHEEL_SOURCE=pypi, the Dockerfile default), then @@ -179,11 +258,11 @@ jobs: # skips them: there is no real-PyPI installable version to build an image # from for those. publish: - needs: [smoke-postgres, smoke-trino] + needs: [smoke-postgres, smoke-trino, smoke-duckdb] if: startsWith(github.ref, 'refs/tags/v') && !contains(github.ref_name, '-test') strategy: matrix: - engine: [postgres, trino] + engine: [postgres, trino, duckdb] runs-on: ubuntu-latest permissions: contents: read diff --git a/Dockerfile.duckdb b/Dockerfile.duckdb new file mode 100644 index 0000000..f304756 --- /dev/null +++ b/Dockerfile.duckdb @@ -0,0 +1,41 @@ +FROM python:3.14-slim +# One dual-role image per engine: the python-node runtime harness (default +# command `run`) and the blue/green validation runner (`validation-op`). It +# installs the PUBLISHED, versioned libraries, never the repo source: +# WHEEL_SOURCE=pypi (release) installs the pins from PyPI. +# WHEEL_SOURCE=wheelhouse (CI/PR) installs the same pins from ./wheelhouse +# (built by CI) so an unreleased change is testable. +# The pinned versions live in image-requirements-duckdb.txt, kept equal to the +# repo's pyproject versions by tests/test_image_requirements_sync.py. See +# Dockerfile.postgres for the full reasoning behind the two-step wheelhouse +# install (exact local files first, --no-index --no-deps, then PyPI for the +# third-party dependencies). +ARG WHEEL_SOURCE=pypi +COPY image-requirements-duckdb.txt /tmp/req.txt +COPY wheelhous[e] /tmp/wheelhouse +RUN set -eu; \ + if [ "$WHEEL_SOURCE" = "wheelhouse" ]; then \ + pip install --no-cache-dir --no-index --no-deps \ + /tmp/wheelhouse/continuo_engine_contract-*.whl \ + /tmp/wheelhouse/continuo_python_runtime-*.whl \ + /tmp/wheelhouse/continuo_duckdb_adapter-*.whl; \ + pip install --no-cache-dir -r /tmp/req.txt; \ + else \ + pip install --no-cache-dir -r /tmp/req.txt; \ + fi; \ + rm -rf /tmp/req.txt /tmp/wheelhouse +# The adapter LOADs ducklake/postgres/httpfs from this directory. They are +# installed now, as root, because the runtime user below has no home directory +# and the container may have no network. HOME=/tmp gives DuckDB a writable +# home for its own bookkeeping. +ENV DUCKDB_EXTENSION_DIRECTORY=/opt/duckdb/extensions HOME=/tmp +COPY scripts/bake_duckdb_extensions.py /tmp/bake_duckdb_extensions.py +RUN python /tmp/bake_duckdb_extensions.py && rm /tmp/bake_duckdb_extensions.py \ + && chmod -R a+rX /opt/duckdb +ENV CONTRACT_DIR=/app/contracts APP_ROOT=/app PYTHONPATH=/app +WORKDIR /app +# uid 65532 matches continuo's executor securityContext expectation. +RUN useradd --uid 65532 --no-create-home --shell /usr/sbin/nologin nonroot +USER 65532:65532 +ENTRYPOINT ["continuo-runtime"] +CMD ["run"] diff --git a/image-requirements-duckdb.txt b/image-requirements-duckdb.txt new file mode 100644 index 0000000..5e6e38f --- /dev/null +++ b/image-requirements-duckdb.txt @@ -0,0 +1,2 @@ +continuo-python-runtime==0.7.0 +continuo-duckdb-adapter==0.1.0 diff --git a/scripts/bake_duckdb_extensions.py b/scripts/bake_duckdb_extensions.py new file mode 100644 index 0000000..5fa45e5 --- /dev/null +++ b/scripts/bake_duckdb_extensions.py @@ -0,0 +1,32 @@ +"""Install the DuckDB extensions the duckdb adapter needs into the image. + +Run once at image build time, as root, into DUCKDB_EXTENSION_DIRECTORY. The +runtime container is non-root and must start without network access, so the +adapter only LOADs these, it never has to INSTALL them. +""" +import logging +import os +import sys + +import duckdb + +logger = logging.getLogger("bake_duckdb_extensions") + +EXTENSIONS = ("ducklake", "postgres", "httpfs") + + +def main() -> int: + logging.basicConfig(stream=sys.stderr, level=logging.INFO) + directory = os.environ["DUCKDB_EXTENSION_DIRECTORY"] + os.makedirs(directory, exist_ok=True) + con = duckdb.connect() + con.execute("SET extension_directory = '" + directory.replace("'", "''") + "'") + for name in EXTENSIONS: + logger.info("installing %s into %s", name, directory) + con.execute(f"INSTALL {name}") + con.execute(f"LOAD {name}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/check_version_bumps.py b/scripts/check_version_bumps.py index ee73b18..3c5a114 100644 --- a/scripts/check_version_bumps.py +++ b/scripts/check_version_bumps.py @@ -17,6 +17,7 @@ "continuo-engine-contract": ("contract/pyproject.toml", ["contract"]), "continuo-postgres-adapter": ("adapters/postgres/pyproject.toml", ["adapters/postgres"]), "continuo-trino-adapter": ("adapters/trino/pyproject.toml", ["adapters/trino"]), + "continuo-duckdb-adapter": ("adapters/duckdb/pyproject.toml", ["adapters/duckdb"]), } diff --git a/tests/test_image_requirements_sync.py b/tests/test_image_requirements_sync.py index 6a57203..3ff2f01 100644 --- a/tests/test_image_requirements_sync.py +++ b/tests/test_image_requirements_sync.py @@ -34,3 +34,9 @@ def test_trino_image_requirements_match_pyproject(): pins = _pins("image-requirements-trino.txt") assert pins["continuo-python-runtime"] == _v("pyproject.toml") assert pins["continuo-trino-adapter"] == _v("adapters/trino/pyproject.toml") + + +def test_duckdb_image_requirements_match_pyproject(): + pins = _pins("image-requirements-duckdb.txt") + assert pins["continuo-python-runtime"] == _v("pyproject.toml") + assert pins["continuo-duckdb-adapter"] == _v("adapters/duckdb/pyproject.toml") From 1ee4ea420e07a05863ec9a119988ae9621a8a24a Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:49:54 +0200 Subject: [PATCH 13/29] chore(duckdb): wire CI, publish, docs and changelog Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .github/ISSUE_TEMPLATE/bug_report.yml | 1 + .github/dependabot.yml | 3 +- .github/workflows/ci.yml | 18 ++++++++++ .github/workflows/publish-pypi.yml | 15 ++++---- CHANGELOG.md | 13 +++++++ CLAUDE.md | 4 +-- CONTRIBUTING.md | 12 +++++-- README.md | 13 +++---- adapters/duckdb/README.md | 52 +++++++++++++++++++++++++-- docs/boundary-contract.md | 13 +++++-- template/contracts/example.yml | 3 +- 11 files changed, 124 insertions(+), 23 deletions(-) diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 048ccea..0d24f30 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -36,6 +36,7 @@ body: - contract (continuo-engine-contract) - adapters/postgres - adapters/trino + - adapters/duckdb - template - CI (.github/workflows) - Not sure diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 14e0478..5c29ead 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -16,6 +16,7 @@ updates: - /contract - /adapters/postgres - /adapters/trino + - /adapters/duckdb schedule: interval: weekly open-pull-requests-limit: 5 @@ -46,7 +47,7 @@ updates: commit-message: prefix: chore(ci) - # Dockerfile.postgres and Dockerfile.trino, the published engine images. Most + # Dockerfile.postgres, Dockerfile.trino and Dockerfile.duckdb, the published engine images. Most # base-image CVEs are fixed by moving to a newer tag. - package-ecosystem: docker directory: / diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a1ad70b..94bf44b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,6 +22,8 @@ jobs: run: uv run --package continuo-postgres-adapter mypy adapters/postgres/continuo_postgres_adapter - name: Types (trino adapter) run: uv run --package continuo-trino-adapter mypy adapters/trino/continuo_trino_adapter + - name: Types (duckdb adapter) + run: uv run --package continuo-duckdb-adapter mypy adapters/duckdb/continuo_duckdb_adapter # `-m "not image"` deselects tests/test_image_smoke_validation.py, which # needs a built engine image and the env naming it. Those tests run in # images.yml's smoke jobs, where an image actually exists. `and not @@ -37,6 +39,8 @@ jobs: run: uv run pytest contract/tests -v - name: Tests (adapter units) run: uv run pytest adapters/postgres/tests adapters/trino/tests -m "not integration" -v + - name: Tests (duckdb adapter units) + run: uv run pytest adapters/duckdb/tests -m "not integration" -v integration-postgres: runs-on: ubuntu-latest @@ -65,3 +69,17 @@ jobs: - name: Tear down trino stack if: always() run: docker compose -f tests/smoke/trino-stack/docker-compose.yml down -v + + integration-duckdb: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: astral-sh/setup-uv@v7 + - run: uv sync --all-packages --all-groups + - name: Start duckdb stack + run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait + - name: Tests (duckdb integration) + run: uv run pytest adapters/duckdb/tests -m integration -v + - name: Tear down duckdb stack + if: always() + run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml down -v diff --git a/.github/workflows/publish-pypi.yml b/.github/workflows/publish-pypi.yml index 0c753b1..b064cbf 100644 --- a/.github/workflows/publish-pypi.yml +++ b/.github/workflows/publish-pypi.yml @@ -1,9 +1,9 @@ name: publish-pypi -# Publishes all four PyPI distributions this repo owns — continuo-python-runtime +# Publishes all five PyPI distributions this repo owns — continuo-python-runtime # (the harness), continuo-engine-contract (the port, result-block format, and -# shared guards), and the two engine adapters (continuo-postgres-adapter, -# continuo-trino-adapter) — via PyPI Trusted Publishing (OIDC, no stored +# shared guards), and the three engine adapters (continuo-postgres-adapter, +# continuo-trino-adapter, continuo-duckdb-adapter) — via PyPI Trusted Publishing (OIDC, no stored # token). They publish together on a single v* tag, each at its own # pyproject version, so a tag that only bumps the runtime still carries an # unchanged adapter version along for the ride; skip-existing (below) makes @@ -18,9 +18,9 @@ name: publish-pypi # claims every tag beginning with "v"; no other pattern may be introduced. # Tag `v-test` publishes to TestPyPI; `v` publishes to real PyPI. # The GitHub environment name is what the PyPI "pending publisher" is -# registered against — all four project names need one on each index. +# registered against — all five project names need one on each index. # -# All four distributions are built into a single `dist/` and uploaded in one +# All five distributions are built into a single `dist/` and uploaded in one # publish call, so there is no ordering constraint between them. The runtime # wheel declares continuo-engine-contract as a dependency and resolves it from # the index at install time ([tool.uv.sources] is dev-only and is not embedded @@ -48,7 +48,7 @@ jobs: - name: Refuse a tag that changed a package without bumping its version run: python scripts/check_version_bumps.py - name: Test before publishing - # Gate all four published projects, including both adapter suites — an + # Gate all five published projects, including every adapter suite — an # adapter regression must block its own immutable PyPI upload, and # images.yml runs in parallel so its smoke failure cannot. Each suite is # a SEPARATE pytest invocation: every workspace member's tests/ is its @@ -63,6 +63,7 @@ jobs: uv run pytest tests contract/tests -m "not image and not integration" -q uv run pytest adapters/postgres/tests -m "not image and not integration" -q uv run pytest adapters/trino/tests -m "not image and not integration" -q + uv run pytest adapters/duckdb/tests -m "not image and not integration" -q - name: Build the contract sdist + wheel run: uv build --package continuo-engine-contract -o dist - name: Build the runtime sdist + wheel @@ -71,6 +72,8 @@ jobs: run: uv build --package continuo-postgres-adapter -o dist - name: Build the trino adapter sdist + wheel run: uv build --package continuo-trino-adapter -o dist + - name: Build the duckdb adapter sdist + wheel + run: uv build --package continuo-duckdb-adapter -o dist # Adapters change rarely, so most v* tags carry an unchanged adapter # version alongside a bumped runtime/contract version. Without # skip-existing, re-uploading that unchanged version 400s and fails the diff --git a/CHANGELOG.md b/CHANGELOG.md index 2939c51..a998b1b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,19 @@ follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). ## [Unreleased] +### Added + +- `continuo-duckdb-adapter` 0.1.0: a DuckDB engine adapter on a DuckLake + (Postgres catalog, Parquet data on S3/MinIO) with the same behaviour as the + postgres and trino adapters: validation DDL, `check_binds`, and the + python-node `fetch` / `ensure_table` / `load`. Physical layout `config` + accepts `partitioned_by` (identity, `bucket`, `year`/`month`/`day`/`hour`) + and `sorted_by`; any other key is rejected. Configured with `DUCKDB_*` + environment variables (catalog, data path, S3). +- `Dockerfile.duckdb` (engine image, DuckDB extensions baked in for offline, + non-root start), the `tests/smoke/duckdb-stack` compose stack, and CI jobs + that run the adapter's integration suite and the image smoke test against it. + ## [0.7.0] - 2026-09-29 ### Added diff --git a/CLAUDE.md b/CLAUDE.md index dbf4c41..14eda13 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -3,7 +3,7 @@ The validation/execution runtime that continuo's executor runs as a Kubernetes Job. Python 3.12, uv workspace: the runtime (`continuo_python_runtime`), the engine contract (`contract/`), and the engine adapters (`adapters/postgres`, -`adapters/trino`). +`adapters/trino`, `adapters/duckdb`). ## CHANGELOG is not optional @@ -26,7 +26,7 @@ merged PRs before adding anything new. A release is one `chore(release):` commit that bumps versions and updates the changelog, then a `vX.Y.Z` tag on it. Bump the version of every package whose source changed since the last tag — `scripts/check_version_bumps.py` fails the -PyPI publish otherwise. The tag builds and publishes the `-postgres` / `-trino` +PyPI publish otherwise. The tag builds and publishes the `-postgres` / `-trino` / `-duckdb` images and the PyPI packages. See `CONTRIBUTING.md` for the full steps. ## Conventions diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 12171f0..c9bddba 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -47,7 +47,7 @@ uv sync --all-packages --all-groups ``` This is a uv workspace: the root package (`continuo_python_runtime/`, the harness), the -port (`contract/`), and the two engine adapters (`adapters/postgres/`, `adapters/trino/`) +port (`contract/`), and the three engine adapters (`adapters/postgres/`, `adapters/trino/`, `adapters/duckdb/`) are separate packages sharing one lockfile. ## Before you open a pull request @@ -59,16 +59,22 @@ uv run mypy continuo_python_runtime uv run mypy contract/continuo_engine_contract uv run --package continuo-postgres-adapter mypy adapters/postgres/continuo_postgres_adapter uv run --package continuo-trino-adapter mypy adapters/trino/continuo_trino_adapter +uv run --package continuo-duckdb-adapter mypy adapters/duckdb/continuo_duckdb_adapter uv run pytest --cov=continuo_python_runtime -m "not image and not integration" -v uv run pytest tests/test_csv_readers_integration.py tests/test_validation_runner.py -m integration -v uv run pytest contract/tests -v uv run pytest adapters/postgres/tests adapters/trino/tests -m "not integration" -v +uv run pytest adapters/duckdb/tests -m "not integration" -v ``` These are exactly what `.github/workflows/ci.yml` runs. Integration tests against a real -Postgres/Trino stack, or against the csv-reader/validation-runner minio backend, need +Postgres/Trino/DuckLake stack, or against the csv-reader/validation-runner minio backend, need Docker and are not required for most changes — see `.github/workflows/ci.yml` for how CI -stands them up if you want to run them locally. +stands them up if you want to run them locally. The duckdb suite, for example, runs +against `tests/smoke/duckdb-stack/docker-compose.yml`: +`docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait`, then +`uv run pytest adapters/duckdb/tests -m integration -v`, then the same compose file with +`down -v`. Also run the security scan before opening a pull request that touches dependencies or anything that could carry a credential: diff --git a/README.md b/README.md index b7076f4..67266a5 100644 --- a/README.md +++ b/README.md @@ -21,8 +21,8 @@ Five artifacts come out of this repository: - **The `continuo-engine-contract` PyPI package** — the `WarehouseAdapter` port, the contract schema, the shared SQL/type/config guards, and the sentinel result-block format. Adapter authors outside this repo pin it. -- **The two engine-adapter PyPI packages** (`continuo-postgres-adapter`, - `continuo-trino-adapter`) — one `WarehouseAdapter` implementation per +- **The three engine-adapter PyPI packages** (`continuo-postgres-adapter`, + `continuo-trino-adapter`, `continuo-duckdb-adapter`) — one `WarehouseAdapter` implementation per warehouse engine, each published independently under the same tag. A domain repo normally never installs these directly (the engine image already has the matching one baked in); they exist as standalone PyPI @@ -60,12 +60,13 @@ validation-side port, adapter class, entry-point group, or image. One | `continuo-engine-contract` | `continuo_engine_contract` | this repo, `contract/` | The `WarehouseAdapter` port, contract schema, the SQL/type/config guards adapters must run, and the result-block format. Published to PyPI. | | `continuo-postgres-adapter` | `continuo_postgres_adapter` | this repo, `adapters/postgres/` | `PostgresAdapter` — one class, both roles. Published to PyPI. | | `continuo-trino-adapter` | `continuo_trino_adapter` | this repo, `adapters/trino/` | `TrinoAdapter` — one class, both roles, for Trino/Iceberg. Published to PyPI. | +| `continuo-duckdb-adapter` | `continuo_duckdb_adapter` | this repo, `adapters/duckdb/` | `DuckDBAdapter` — one class, both roles, for DuckDB on a DuckLake (Postgres catalog, Parquet on S3). Published to PyPI. | -All four are uv workspace members (`[tool.uv.workspace]` in the root +All five are uv workspace members (`[tool.uv.workspace]` in the root `pyproject.toml`), so `uv sync --all-packages --all-groups` at the repo root installs everything for local development. -**All four packages in the table above are published to PyPI**, under the +**All five packages in the table above are published to PyPI**, under the same `vX.Y.Z` tag. The two engine images then **install the matching pinned adapter version from PyPI** — `Dockerfile.postgres` installs `continuo-postgres-adapter==X.Y.Z`, `Dockerfile.trino` installs @@ -129,7 +130,7 @@ the pre-flight check once before your next release; it reports every affected read at once: ```bash -continuo-runtime validate contracts/ --dialect postgres # or trino +continuo-runtime validate contracts/ --dialect postgres # or trino, duckdb ``` The runtime image does not re-run this gate, so a read that passes here is @@ -274,7 +275,7 @@ pinned PyPI version of the `continuo-postgres-adapter` or `continuo-trino-adapter` package built from this repo's `adapters/postgres/` or `adapters/trino/` source (see the table above) — registered under the `continuo_engine.adapters` entry-point group (entry names `postgres` / -`trino`). The runtime discovers it via `discover_adapter()` at run time, so a +`trino` / `duckdb`). The runtime discovers it via `discover_adapter()` at run time, so a single image serves every node in the service and the release-time validation Job for it. The executor injects the warehouse connection as environment variables (engine-native, e.g. diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md index 214ecd8..193bb2e 100644 --- a/adapters/duckdb/README.md +++ b/adapters/duckdb/README.md @@ -1,6 +1,54 @@ # continuo-duckdb-adapter -DuckDB warehouse adapter for Continuo, running on a DuckLake: the catalog lives -in Postgres and table data is Parquet on S3/MinIO. Implements +DuckDB warehouse adapter for Continuo, running on a **DuckLake**: the catalog +lives in Postgres and table data is Parquet on S3/MinIO, so every Kubernetes +Job shares one transactional warehouse. Implements `continuo_engine_contract.port.WarehouseAdapter`; registered under the `continuo_engine.adapters` entry-point group as `duckdb`. + +## Environment + +| Variable | Required | Default | Meaning | +|---|---|---|---| +| `DUCKDB_CATALOG_HOST` | yes | | Postgres host holding the DuckLake catalog | +| `DUCKDB_CATALOG_PORT` | no | `5432` | | +| `DUCKDB_CATALOG_DB` | yes | | | +| `DUCKDB_CATALOG_USER` | yes | | | +| `DUCKDB_CATALOG_PASSWORD` | no | empty | | +| `DUCKDB_DATA_PATH` | yes | | Data location, e.g. `s3://bucket/lake/` (fixed once the catalog exists) | +| `DUCKDB_S3_ENDPOINT` | no | AWS | `host:port`, for MinIO and other S3-compatible stores | +| `DUCKDB_S3_ACCESS_KEY_ID` / `DUCKDB_S3_SECRET_ACCESS_KEY` | no | credential chain | Static credentials; omit to use the AWS credential chain | +| `DUCKDB_S3_REGION` | no | `us-east-1` | | +| `DUCKDB_S3_URL_STYLE` | no | `path` with an endpoint, else `vhost` | `path` or `vhost` | +| `DUCKDB_S3_USE_SSL` | no | `true` | | +| `DUCKDB_EXTENSION_DIRECTORY` | no | DuckDB default | Where `ducklake`, `postgres`, `httpfs` are loaded from (set in the image) | +| `DUCKDB_DATA_INLINING_ROW_LIMIT` | no | DuckLake default | `0` writes every insert as a Parquet file instead of inlining small ones in the catalog | + +## Physical layout (`config`) + +- `partitioned_by`: non-empty list of a column name, or + `{column, transform, buckets}` with `transform` in `identity`, `bucket` + (`buckets` required), `year`, `month`, `day`, `hour` (time transforms need a + `DATE` or `TIMESTAMP` column). +- `sorted_by`: non-empty list of a column name, or + `{column, direction: asc|desc, nulls: first|last}`. +- Columns must be declared in `output_columns`. Any other key, including + postgres's `indexes`, is rejected before any DDL runs. +- Layout is applied when `ensure_table` creates the table; changing it on an + existing table is a no-op. `build_empty_from_columns` (the release gate) + always rebuilds. + +## Layout of the code + +`domain/` (pure rules) <- `application/` (use cases, `LakeGateway` port) <- +`infrastructure/` (DuckDB/DuckLake I/O, SQL rendering); `adapter.py` is the +composition root and the entry-point target. + +## Tests + +```bash +uv run pytest adapters/duckdb/tests -m "not integration" -v +docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait +uv run pytest adapters/duckdb/tests -m integration -v +docker compose -f tests/smoke/duckdb-stack/docker-compose.yml down -v +``` diff --git a/docs/boundary-contract.md b/docs/boundary-contract.md index 1c2df53..dd56cf6 100644 --- a/docs/boundary-contract.md +++ b/docs/boundary-contract.md @@ -72,7 +72,7 @@ s3://///contract.yaml own dialect — a read that's valid against postgres can fail `InvalidCompiledSql` on an install whose warehouse is Trino, and vice versa. `continuo-runtime validate|merge|hash` accept an optional - `--dialect ` flag (e.g. `postgres`, `trino`) so a domain repo can + `--dialect ` flag (e.g. `postgres`, `trino`, `duckdb`) so a domain repo can check its reads against that dialect locally, catching the failure in its own CI instead of at Continuo's parser. Reads are always parsed (via `continuo_engine_contract.sql.ensure_single_read`, sqlglot-backed) even @@ -126,6 +126,15 @@ parse time. non-empty list of non-empty strings — column names or Iceberg partition transforms like `day(event_ts)`), and `format` (one of `PARQUET`/`ORC`/`AVRO`, case-insensitive). + - **duckdb (DuckLake)** → `partitioned_by` and `sorted_by`. + `partitioned_by` is a non-empty list whose entries are a declared column + name (identity partitioning) or `{column, transform, buckets}` with + `transform` one of `identity`, `bucket` (needs `buckets`, a positive + integer), `year`, `month`, `day`, `hour` (these four need a `DATE` or + `TIMESTAMP` column). `sorted_by` is a non-empty list of declared column + names or `{column, direction: asc|desc, nulls: first|last}`; free-form + expressions are rejected. DuckLake has no indexes, so postgres's `indexes` + is an unrecognized key on this engine. - Worked example (from `template/contracts/example.yml`): ```yaml config: @@ -142,7 +151,7 @@ parse time. `name`, since that changes the derived default name. Iceberg's properties instead all ride on a single `WITH (...)` clause attached to the `CREATE TABLE`'s own `IF NOT EXISTS`, so once the table exists - nothing in `config` is applied at all. Neither engine is a migration + nothing in `config` is applied at all. DuckLake behaves like Iceberg here: `partitioned_by` and `sorted_by` are applied only when `ensure_table` creates the table, so changing them on an existing table is a silent no-op; the release gate's `build_empty_from_columns` always rebuilds with the new layout. Neither engine is a migration mechanism for an index/property that already exists under the same name: flipping `unique: false → true` under a fixed index `name` is a **silent no-op** on postgres (the name already resolves, so `IF NOT diff --git a/template/contracts/example.yml b/template/contracts/example.yml index 3c3ef92..7c94676 100644 --- a/template/contracts/example.yml +++ b/template/contracts/example.yml @@ -18,7 +18,7 @@ nodes: # like psycopg2's `%(name)s`) now fails validate, where it previously did # not. `ctx.read(name)` never takes params, so such a read was already # broken at run time; this only moves the failure earlier, to CI. Add - # --dialect (e.g. postgres, trino) to check against the + # --dialect (e.g. postgres, trino, duckdb) to check against the # warehouse dialect Continuo will actually use, instead of only # sqlglot's dialect-neutral default. reads: @@ -32,6 +32,7 @@ nodes: # engine's adapter rejects any key it does not recognize. # postgres: indexes # trino (iceberg): partitioning, sorted_by, format + # duckdb (ducklake): partitioned_by, sorted_by # config: # indexes: # - columns: [order_id] From 4633839faf4650fc11152ac736570229ce1f61ed Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 20:52:45 +0200 Subject: [PATCH 14/29] chore: add the DDD architecture auditor agent Co-Authored-By: Claude Haiku 4.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .claude/agents/ddd-architecture-auditor.md | 117 +++++++++++++++++++++ .gitignore | 1 + 2 files changed, 118 insertions(+) create mode 100644 .claude/agents/ddd-architecture-auditor.md diff --git a/.claude/agents/ddd-architecture-auditor.md b/.claude/agents/ddd-architecture-auditor.md new file mode 100644 index 0000000..188bfcb --- /dev/null +++ b/.claude/agents/ddd-architecture-auditor.md @@ -0,0 +1,117 @@ +--- +name: ddd-architecture-auditor +description: >- + Use to audit code for DDD and Clean Architecture compliance in this repo's + engine adapters: layer dependency direction, domain purity, port placement, + thin use cases, SQL/identifier discipline, and repo conventions (changelog, + pins, test naming). Dispatch before merging a branch or whenever asked to + review architecture. Read-only: it writes a findings document the main agent + can act on; it never edits source. +tools: Read, Grep, Glob, Bash, Write +model: inherit +--- + +# Role + +You are a Domain-Driven Design and Clean Architecture auditor for the +`continuo-python-runtime` repo. You inspect code, find architecture violations, +and write a structured report so the main agent can fix them. You do not modify +source code. Your only output artifact is the report. + +# Scope + +Default to the current branch's diff against `main`: + +``` +git diff main...HEAD --name-only +git diff main...HEAD +``` + +Audit the whole tree only when the caller asks for a full sweep. State the +scope you chose at the top of the report. + +# What to check + +Report a finding only when you can point at a real `file:line` you have opened +and confirmed. Do not flag hypotheticals. + +## Layer 1: generic DDD / Clean Architecture + +- **Dependency direction.** The arrow runs infrastructure -> application -> + domain. Flag any inward-pointing import. +- **Domain purity.** Domain code has no infrastructure concerns: no database, + object-store or dataframe clients, no serialization or framework types. +- **Ports and adapters.** A port is declared by the layer that consumes it; + implementations live outside it. Flag an implementation that leaks + engine-specific types through the port. +- **Thin use cases.** Use cases validate, then orchestrate through ports. Flag + business rules hiding in infrastructure, or SQL/engine detail in use cases. +- **SOLID.** Flag a class with more than one reason to change, a use case that + switches on engine type instead of depending on an abstraction, and an + interface too wide for its consumers. + +## Layer 2: rules for `adapters/duckdb` (and any new adapter) + +1. **Layering by import.** `domain/` imports none of `duckdb`, `psycopg2`, + `pyarrow`, `boto3`, `application`, `infrastructure`. `application/` imports + neither `infrastructure` nor `duckdb`. `infrastructure/` imports only + `domain` and `application.ports`. Only the top-level `adapter.py` + (composition root) may import every layer. Verify with grep over each + layer's files. +2. **Port placement.** The `LakeGateway` port and `LakeConflictError` live in + `application/ports.py`. `infrastructure` only implements them. +3. **Validate before acting.** Every public use case validates its inputs + (types, layout, single-read gate) before the first gateway call. +4. **SQL discipline.** Identifiers go through `quote_identifier`, strings + through `sql_literal`, in `infrastructure/ddl.py` only. Flag any SQL built + by string formatting elsewhere, and any author-supplied text that reaches + SQL unquoted. +5. **Logging.** Standard `logging` only, never `print`; no SQL that can carry + a secret is logged (the ATTACH and secret statements). +6. **Repo conventions (CLAUDE.md).** `CHANGELOG.md` has an `[Unreleased]` + entry for user-facing change; in-repo and third-party pins are exact; the + package is registered in the workspace, `scripts/check_version_bumps.py`, + `tests/test_image_requirements_sync.py` and the CI/image workflows; test + module basenames are unique across the repo. +7. **Doc currency.** `docs/boundary-contract.md` documents every `config` key + the adapter accepts. + +# Report + +Write to `docs/ddd-violations/-ddd-violations.md` (create the +directory; timestamp from `date +%Y-%m-%d-%H%M`). + +```markdown +# DDD / Clean Architecture Audit: + +**Scope:** +**Branch / commit:** @ +**Summary:** blockers, should-fix, nits + +## Findings + +### [BLOCKER] +- **Where:** `path/to/file.py:123` +- **Rule:** +- **Problem:** +- **Suggested fix:** + +### [SHOULD-FIX] ... +### [NIT] ... + +## Clean +- +``` + +Severity: **BLOCKER** = breaks the dependency arrow, domain purity, or a +CI-enforced rule; **SHOULD-FIX** = boundary erosion or convention drift that +will bite later; **NIT** = stylistic. + +# Hard constraints + +- Never edit, create, or scaffold source code. The report is your only write. +- Never claim a violation is fixed; you only diagnose. +- Every finding cites a real `file:line` you have opened and confirmed. +- If you find nothing, still write the report with an empty Findings section and + a populated Clean list. +- End your reply with the report's path and the summary counts. diff --git a/.gitignore b/.gitignore index 8a96f53..1b449eb 100644 --- a/.gitignore +++ b/.gitignore @@ -5,6 +5,7 @@ __pycache__/ dist/ .superpowers/ docs/superpowers/ +docs/ddd-violations/ .pytest_cache/ .mypy_cache/ .ruff_cache/ From d97f8d1abfd8512aeb7ffa3eab0f4fc1b0bd4880 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:26:58 +0200 Subject: [PATCH 15/29] fix(duckdb): gate build_empty_from_sql and validate layout keys on construction Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../application/warehouse.py | 24 ++++++++++--- .../continuo_duckdb_adapter/domain/layout.py | 34 ++++++++++++------- .../tests/test_adapter_duckdb_validation.py | 22 ++++++++++++ .../duckdb/tests/test_domain_duckdb_layout.py | 19 +++++++++++ .../test_integration_duckdb_validation.py | 15 ++++++++ 5 files changed, 96 insertions(+), 18 deletions(-) diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py index 56287c4..6c02118 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py +++ b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py @@ -33,6 +33,11 @@ _CONFLICT_BACKOFF_SECONDS = 0.05 +def _bare_read(sql: str) -> str: + """*sql* without surrounding whitespace or one trailing terminator.""" + return sql.strip().rstrip(";").strip() + + def _columns(raw: list[dict]) -> list[ColumnDefinition]: return [ColumnDefinition.from_mapping(entry) for entry in raw] @@ -73,8 +78,13 @@ def drop_schema(self, schema: str) -> None: # --- Validation builds -------------------------------------------------- def build_empty_from_sql(self, schema: str, table: str, compiled_sql: str) -> None: - """Create ``schema.table`` empty, shaped by the compiled SELECT.""" - inner = compiled_sql.strip().rstrip(";").strip() + """Create ``schema.table`` empty, shaped by the compiled SELECT. + + The parse gate runs first: DuckDB executes every ``;``-separated + statement handed to it, so a stacked statement would run for real. + """ + ensure_single_read(compiled_sql, dialect="duckdb") + inner = _bare_read(compiled_sql) target = QualifiedTable.of(schema, table) with self._gateway.transaction(): self._gateway.drop_table_if_exists(target) @@ -83,9 +93,10 @@ def build_empty_from_sql(self, schema: str, table: str, compiled_sql: str) -> No def clone_empty_from_prod(self, candidate_schema: str, prod_schema: str, table: str) -> None: """Create ``candidate_schema.table`` empty, shaped like ``prod_schema.table``.""" target = QualifiedTable.of(candidate_schema, table) + source = QualifiedTable.of(prod_schema, table) with self._gateway.transaction(): self._gateway.drop_table_if_exists(target) - self._gateway.create_empty_clone(target, QualifiedTable.of(prod_schema, table)) + self._gateway.create_empty_clone(target, source) def build_empty_from_columns( self, schema: str, table: str, columns: list[dict], config: dict @@ -111,14 +122,17 @@ def check_binds(self, sql: str) -> None: statement handed to it, so a stacked statement would run for real. """ ensure_single_read(sql, dialect="duckdb") - inner = sql.strip().rstrip(";").strip() + inner = _bare_read(sql) logger.info("bind-checking read via EXPLAIN") self._gateway.explain_read(inner) # --- Python-node data plane --------------------------------------------- def fetch(self, sql: str) -> "pa.Table": - """Execute one declared read and return the result as an Arrow table.""" + """Execute one declared read and return the result as an Arrow table. + + Not re-gated here: reads are single-read checked earlier, at contract load. + """ data = self._gateway.fetch_arrow(sql) seen: set[str] = set() duplicates: set[str] = set() diff --git a/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py b/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py index 09412ff..e7bf965 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py +++ b/adapters/duckdb/continuo_duckdb_adapter/domain/layout.py @@ -32,6 +32,25 @@ class PartitionKey: transform: str = "identity" buckets: int | None = None + def __post_init__(self) -> None: + if self.transform not in _TRANSFORMS: + raise ValueError( + f"unsupported partition 'transform' {self.transform!r}; " + f"supported: {', '.join(_TRANSFORMS)}" + ) + if self.transform == "bucket": + if ( + isinstance(self.buckets, bool) + or not isinstance(self.buckets, int) + or self.buckets < 1 + ): + raise ValueError( + "partition transform 'bucket' requires 'buckets' as a positive integer, " + f"got {self.buckets!r}" + ) + elif self.buckets is not None: + raise ValueError("'buckets' is only valid with partition transform 'bucket'") + @dataclass(frozen=True) class SortKey: @@ -105,18 +124,7 @@ def _partition_key(entry: Any, column_types: Mapping[str, str | None]) -> Partit ensure_known_keys(entry, _PARTITION_ENTRY_KEYS, ENGINE, where=where) column = _declared_column(entry.get("column"), column_types, where) transform = entry.get("transform", "identity") - if transform not in _TRANSFORMS: - raise ValueError( - f"unsupported partition 'transform' {transform!r}; supported: {', '.join(_TRANSFORMS)}" - ) - buckets = entry.get("buckets") - if transform == "bucket": - if isinstance(buckets, bool) or not isinstance(buckets, int) or buckets < 1: - raise ValueError( - f"partition transform 'bucket' requires 'buckets' as a positive integer, got {buckets!r}" - ) - elif buckets is not None: - raise ValueError("'buckets' is only valid with partition transform 'bucket'") + key = PartitionKey(Identifier(column), transform, entry.get("buckets")) # validates itself if transform in _TEMPORAL_TRANSFORMS: column_type = column_types[column] if column_type is not None and not is_temporal_type(column_type): @@ -124,7 +132,7 @@ def _partition_key(entry: Any, column_types: Mapping[str, str | None]) -> Partit f"partition transform {transform!r} needs a DATE or TIMESTAMP column, " f"but {column!r} is {column_type}" ) - return PartitionKey(Identifier(column), transform, buckets) + return key def _sort_keys(raw: Any, column_types: Mapping[str, str | None]) -> tuple[SortKey, ...]: diff --git a/adapters/duckdb/tests/test_adapter_duckdb_validation.py b/adapters/duckdb/tests/test_adapter_duckdb_validation.py index 64555ff..1ba75d2 100644 --- a/adapters/duckdb/tests/test_adapter_duckdb_validation.py +++ b/adapters/duckdb/tests/test_adapter_duckdb_validation.py @@ -126,3 +126,25 @@ def test_check_binds_passes_a_single_read_to_explain(warehouse, gateway, sql, in def test_close_closes_the_gateway(warehouse, gateway): warehouse.close() assert gateway.calls == [("close",)] + + +def test_build_empty_from_sql_runs_the_gate_before_touching_the_gateway(warehouse, gateway): + for bad in ("SELECT 1; DROP TABLE t", "DELETE FROM t", "SELECT 1) AS x; DROP TABLE t; SELECT * FROM (SELECT 1"): + with pytest.raises(ValueError): + warehouse.build_empty_from_sql("s", "t", bad) + assert gateway.calls == [] + + +@pytest.mark.parametrize("sql,inner", [ + ("SELECT 1 AS a;", "SELECT 1 AS a"), + ("SELECT 1 AS a -- trailing comment", "SELECT 1 AS a -- trailing comment"), +]) +def test_build_empty_from_sql_still_accepts_single_reads(warehouse, gateway, sql, inner): + warehouse.build_empty_from_sql("s", "t", sql) + assert ("create_empty_table_as", T, inner) in gateway.calls + + +def test_clone_empty_from_prod_validates_both_tables_before_any_call(warehouse, gateway): + with pytest.raises(ValueError): + warehouse.clone_empty_from_prod("cand", "", "t") + assert gateway.calls == [] diff --git a/adapters/duckdb/tests/test_domain_duckdb_layout.py b/adapters/duckdb/tests/test_domain_duckdb_layout.py index 7e5d0f9..fe34560 100644 --- a/adapters/duckdb/tests/test_domain_duckdb_layout.py +++ b/adapters/duckdb/tests/test_domain_duckdb_layout.py @@ -114,3 +114,22 @@ def test_both_keys_together(): layout = _layout({"partitioned_by": ["name"], "sorted_by": ["id"]}) assert not layout.is_empty assert len(layout.partition_keys) == 1 and len(layout.sort_keys) == 1 + + +def test_partition_key_defaults_are_valid(): + assert PartitionKey(Identifier("c")).transform == "identity" + + +@pytest.mark.parametrize("transform,buckets", [ + ("drop table", None), + ("x); DROP TABLE t; --", None), + ("bucket", None), + ("bucket", 0), + ("bucket", True), + ("bucket", "4"), + ("identity", 4), + ("month", 4), +]) +def test_partition_key_rejects_invalid_construction(transform, buckets): + with pytest.raises(ValueError): + PartitionKey(Identifier("c"), transform, buckets) diff --git a/adapters/duckdb/tests/test_integration_duckdb_validation.py b/adapters/duckdb/tests/test_integration_duckdb_validation.py index 05c69d6..5719084 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_validation.py +++ b/adapters/duckdb/tests/test_integration_duckdb_validation.py @@ -179,3 +179,18 @@ def test_check_binds_leaves_the_connection_usable_after_a_failed_read(adapter, p adapter.check_binds(f'SELECT nope FROM "{prod}".src_table') assert adapter.fetch("SELECT 1 AS ok").to_pylist() == [{"ok": 1}] adapter.check_binds(f'SELECT id FROM "{prod}".src_table') + + +@pytest.mark.parametrize("attack", [ + 'SELECT 1; DROP TABLE "{s}"."victim"', + 'SELECT 1) AS x; DROP TABLE "{s}"."victim"; SELECT * FROM (SELECT 1', +]) +def test_build_empty_from_sql_rejects_stacked_statements_and_executes_nothing( + adapter, schema, tables_in, scalar, attack +): + adapter.ensure_table(schema, "victim", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + adapter.load(schema, "victim", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) + with pytest.raises(ValueError): + adapter.build_empty_from_sql(schema, "built", attack.format(s=schema)) + assert tables_in(schema) == ["victim"] + assert scalar(f'SELECT count(*) AS n FROM "{schema}"."victim"') == 3 From ca4035ed15ba9908d53d345c23108f2786a4b03e Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:30:43 +0200 Subject: [PATCH 16/29] refactor(duckdb): keep all SQL in ddl.py and single-source the extension list Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .github/workflows/images.yml | 2 +- .../infrastructure/ddl.py | 25 +++++++++ .../infrastructure/session.py | 51 ++++++++++++++----- .../tests/test_infra_duckdb_settings_ddl.py | 30 ++++++++++- scripts/bake_duckdb_extensions.py | 17 ++++--- 5 files changed, 105 insertions(+), 20 deletions(-) diff --git a/.github/workflows/images.yml b/.github/workflows/images.yml index 02ab8a8..e73d35a 100644 --- a/.github/workflows/images.yml +++ b/.github/workflows/images.yml @@ -192,7 +192,7 @@ jobs: - name: Extensions load offline as the non-root user run: | docker run --rm --network none --entrypoint python cpr-smoke-duckdb -c \ - "import os, duckdb; c = duckdb.connect(); c.execute('SET extension_directory = ' + repr(os.environ['DUCKDB_EXTENSION_DIRECTORY'])); [c.execute('LOAD ' + e) for e in ('ducklake', 'postgres', 'httpfs')]" + "import os; from continuo_duckdb_adapter.infrastructure.session import check_offline; check_offline(os.environ['DUCKDB_EXTENSION_DIRECTORY'])" - name: Start duckdb stack run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait - name: Seed source table diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index 8a8aba3..c5b461b 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -18,6 +18,15 @@ _LIBPQ_BARE = re.compile(r"[A-Za-z0-9_.:/\-]+") +# The one list of DuckDB extensions the adapter needs: the session LOADs them, +# the image bake script INSTALLs them, the offline image check LOADs them. +EXTENSIONS: tuple[str, ...] = ("ducklake", "postgres", "httpfs") + + +BEGIN = "BEGIN" +COMMIT = "COMMIT" +ROLLBACK = "ROLLBACK" + def quote_identifier(name: str) -> str: return '"' + name.replace('"', '""') + '"' @@ -27,6 +36,22 @@ def sql_literal(value: str) -> str: return "'" + value.replace("'", "''") + "'" +def load_extension(name: str) -> str: + return f"LOAD {quote_identifier(name)}" + + +def install_extension(name: str) -> str: + return f"INSTALL {quote_identifier(name)}" + + +def set_extension_directory(directory: str) -> str: + return f"SET extension_directory = {sql_literal(directory)}" + + +def use_catalog(catalog: str) -> str: + return f"USE {quote_identifier(catalog)}" + + def _libpq_value(value: str) -> str: """Quote one libpq ``keyword=value`` value only when it needs it.""" if _LIBPQ_BARE.fullmatch(value): diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index 3df3126..6143563 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -21,7 +21,19 @@ from ..domain.columns import ColumnDefinition from ..domain.identifiers import Identifier, QualifiedTable from ..domain.layout import TableLayout -from .ddl import DdlRenderer, attach_statement, quote_identifier, s3_secret_statement, sql_literal +from .ddl import ( + BEGIN, + COMMIT, + EXTENSIONS, + ROLLBACK, + DdlRenderer, + attach_statement, + install_extension, + load_extension, + s3_secret_statement, + set_extension_directory, + use_catalog, +) from .settings import DuckLakeSettings if TYPE_CHECKING: # pragma: no cover @@ -30,16 +42,31 @@ logger = logging.getLogger("continuo_duckdb_adapter") CATALOG_ALIAS = "lake" -_EXTENSIONS = ("ducklake", "postgres", "httpfs") def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: try: - con.execute(f"LOAD {name}") + con.execute(load_extension(name)) except duckdb.Error: logger.info("installing duckdb extension %s", name) - con.execute(f"INSTALL {name}") - con.execute(f"LOAD {name}") + con.execute(install_extension(name)) + con.execute(load_extension(name)) + + +def check_offline(directory: str) -> None: + """LOAD every extension from *directory* without ever INSTALLing. + + What a baked, offline, non-root image must be able to do; raises + ``duckdb.Error`` naming the first extension that is missing. + """ + con = duckdb.connect() + try: + con.execute(set_extension_directory(directory)) + for name in EXTENSIONS: + con.execute(load_extension(name)) + logger.info("loaded %s offline from %s", name, directory) + finally: + con.close() class DuckLakeSession(LakeGateway): @@ -52,13 +79,13 @@ def connect(cls, settings: DuckLakeSettings) -> "DuckLakeSession": con = duckdb.connect() try: if settings.extension_directory: - con.execute(f"SET extension_directory = {sql_literal(settings.extension_directory)}") - for extension in _EXTENSIONS: + con.execute(set_extension_directory(settings.extension_directory)) + for extension in EXTENSIONS: _load_extension(con, extension) if settings.uses_s3: con.execute(s3_secret_statement(settings)) con.execute(attach_statement(settings, CATALOG_ALIAS)) - con.execute(f"USE {quote_identifier(CATALOG_ALIAS)}") + con.execute(use_catalog(CATALOG_ALIAS)) except BaseException: con.close() raise @@ -76,17 +103,17 @@ def _run(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConn def _rollback(self) -> None: try: - self._con.execute("ROLLBACK") + self._con.execute(ROLLBACK) except duckdb.Error as exc: # Never mask the failure that got us here. logger.warning("rollback failed: %s", exc) @contextlib.contextmanager def transaction(self) -> Iterator[None]: - self._con.execute("BEGIN") + self._con.execute(BEGIN) try: yield - self._run("COMMIT") + self._run(COMMIT) except BaseException: self._rollback() raise @@ -134,7 +161,7 @@ def apply_layout(self, table: QualifiedTable, layout: TableLayout) -> None: def explain_read(self, sql: str) -> None: # EXPLAIN binds without scanning; the transaction is always rolled back. - self._con.execute("BEGIN") + self._con.execute(BEGIN) try: self._run(self._renderer.explain_read(sql)) finally: diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py index 7671c9a..86b4a99 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -1,12 +1,15 @@ """Settings parsing and SQL rendering: pure string work, no engine.""" +import duckdb import pytest from continuo_duckdb_adapter.domain.columns import ColumnDefinition from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey from continuo_duckdb_adapter.infrastructure.ddl import ( - DdlRenderer, attach_statement, quote_identifier, s3_secret_statement, sql_literal, + EXTENSIONS, DdlRenderer, attach_statement, install_extension, load_extension, + quote_identifier, s3_secret_statement, set_extension_directory, sql_literal, use_catalog, ) +from continuo_duckdb_adapter.infrastructure.session import check_offline from continuo_duckdb_adapter.infrastructure.settings import REQUIRED_ENV, DuckLakeSettings ENV = { @@ -109,6 +112,31 @@ def test_s3_secret_falls_back_to_the_credential_chain(): ) +def test_extension_statements_quote_the_name(): + assert load_extension("ducklake") == 'LOAD "ducklake"' + assert install_extension("ducklake") == 'INSTALL "ducklake"' + assert load_extension('we"ird') == 'LOAD "we""ird"' + + +def test_set_extension_directory_escapes_the_path(): + assert set_extension_directory("/opt/duckdb") == "SET extension_directory = '/opt/duckdb'" + assert set_extension_directory("/o'pt") == "SET extension_directory = '/o''pt'" + + +def test_use_catalog_quotes_the_alias(): + assert use_catalog("lake") == 'USE "lake"' + assert use_catalog('la"ke') == 'USE "la""ke"' + + +def test_extensions_are_the_three_the_adapter_needs(): + assert EXTENSIONS == ("ducklake", "postgres", "httpfs") + + +def test_check_offline_fails_when_the_extensions_are_not_in_the_directory(tmp_path): + with pytest.raises(duckdb.Error): + check_offline(str(tmp_path)) + + R = DdlRenderer("lake") T = QualifiedTable.of("s", "t") diff --git a/scripts/bake_duckdb_extensions.py b/scripts/bake_duckdb_extensions.py index 5fa45e5..622f13b 100644 --- a/scripts/bake_duckdb_extensions.py +++ b/scripts/bake_duckdb_extensions.py @@ -2,29 +2,34 @@ Run once at image build time, as root, into DUCKDB_EXTENSION_DIRECTORY. The runtime container is non-root and must start without network access, so the -adapter only LOADs these, it never has to INSTALL them. +adapter only LOADs these, it never has to INSTALL them. The extension list and +the statements come from the adapter package, so they cannot drift from it. """ import logging import os import sys import duckdb +from continuo_duckdb_adapter.infrastructure.ddl import ( + EXTENSIONS, + install_extension, + load_extension, + set_extension_directory, +) logger = logging.getLogger("bake_duckdb_extensions") -EXTENSIONS = ("ducklake", "postgres", "httpfs") - def main() -> int: logging.basicConfig(stream=sys.stderr, level=logging.INFO) directory = os.environ["DUCKDB_EXTENSION_DIRECTORY"] os.makedirs(directory, exist_ok=True) con = duckdb.connect() - con.execute("SET extension_directory = '" + directory.replace("'", "''") + "'") + con.execute(set_extension_directory(directory)) for name in EXTENSIONS: logger.info("installing %s into %s", name, directory) - con.execute(f"INSTALL {name}") - con.execute(f"LOAD {name}") + con.execute(install_extension(name)) + con.execute(load_extension(name)) return 0 From 6ee1574f144fb8c15031a08497f47f66a47f758f Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:32:01 +0200 Subject: [PATCH 17/29] fix(duckdb): harden settings parsing and keep secrets out of repr and errors Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../infrastructure/ddl.py | 16 ++++++ .../infrastructure/session.py | 17 +++++- .../infrastructure/settings.py | 36 ++++++++++--- .../tests/test_infra_duckdb_settings_ddl.py | 54 ++++++++++++++++++- .../tests/test_integration_duckdb_stack.py | 7 +++ 5 files changed, 120 insertions(+), 10 deletions(-) diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index c5b461b..de700a3 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -60,6 +60,22 @@ def _libpq_value(value: str) -> str: return f"'{escaped}'" +def secret_texts(settings: DuckLakeSettings) -> tuple[str, ...]: + """Every spelling of the credentials that can appear in an engine error message. + + The raw value, its libpq-quoted form (as written into the ATTACH conninfo) + and the SQL-literal-doubled form of that, longest first so redaction of a + longer spelling is not pre-empted by a shorter one. + """ + texts: set[str] = set() + for secret in (settings.catalog_password, settings.s3_secret_access_key): + if not secret: + continue + quoted = _libpq_value(secret) + texts.update((secret, quoted, quoted.replace("'", "''"), secret.replace("'", "''"))) + return tuple(sorted(texts, key=len, reverse=True)) + + def attach_statement(settings: DuckLakeSettings, catalog: str) -> str: conninfo = " ".join( f"{key}={_libpq_value(value)}" diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index 6143563..a6213c7 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -31,6 +31,7 @@ install_extension, load_extension, s3_secret_statement, + secret_texts, set_extension_directory, use_catalog, ) @@ -53,6 +54,16 @@ def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: con.execute(load_extension(name)) +def _redacted(exc: duckdb.Error, secrets: Sequence[str]) -> duckdb.Error: + message = str(exc) + for secret in secrets: + message = message.replace(secret, "***") + try: + return type(exc)(message) + except Exception: # an exception type with a non-standard constructor + return duckdb.Error(message) + + def check_offline(directory: str) -> None: """LOAD every extension from *directory* without ever INSTALLing. @@ -84,7 +95,11 @@ def connect(cls, settings: DuckLakeSettings) -> "DuckLakeSession": _load_extension(con, extension) if settings.uses_s3: con.execute(s3_secret_statement(settings)) - con.execute(attach_statement(settings, CATALOG_ALIAS)) + try: + con.execute(attach_statement(settings, CATALOG_ALIAS)) + except duckdb.Error as exc: + # The engine echoes the ATTACH conninfo, password included. + raise _redacted(exc, secret_texts(settings)) from None con.execute(use_catalog(CATALOG_ALIAS)) except BaseException: con.close() diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py index ac92a19..8d826ad 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py @@ -3,7 +3,7 @@ import os from collections.abc import Mapping -from dataclasses import dataclass +from dataclasses import dataclass, field REQUIRED_ENV: tuple[str, ...] = ( "DUCKDB_CATALOG_HOST", @@ -12,6 +12,13 @@ "DUCKDB_DATA_PATH", ) _URL_STYLES = ("path", "vhost") +_TRUE_WORDS = ("true", "1", "yes") +_FALSE_WORDS = ("false", "0", "no") + + +def _is_ascii_digits(value: str) -> bool: + # str.isdigit() alone accepts e.g. superscripts, which int() then rejects. + return value.isascii() and value.isdigit() @dataclass(frozen=True) @@ -20,11 +27,11 @@ class DuckLakeSettings: catalog_port: str catalog_db: str catalog_user: str - catalog_password: str + catalog_password: str = field(repr=False) data_path: str s3_endpoint: str | None s3_access_key_id: str | None - s3_secret_access_key: str | None + s3_secret_access_key: str | None = field(repr=False) s3_region: str s3_url_style: str s3_use_ssl: bool @@ -49,10 +56,10 @@ def optional(name: str) -> str | None: return env.get(name) or None port = env.get("DUCKDB_CATALOG_PORT") or "5432" - if not port.isdigit(): + if not _is_ascii_digits(port): raise ValueError(f"DUCKDB_CATALOG_PORT must be a port number, got {port!r}") limit_raw = optional("DUCKDB_DATA_INLINING_ROW_LIMIT") - if limit_raw is not None and not limit_raw.isdigit(): + if limit_raw is not None and not _is_ascii_digits(limit_raw): raise ValueError( f"DUCKDB_DATA_INLINING_ROW_LIMIT must be a non-negative integer, got {limit_raw!r}" ) @@ -60,7 +67,20 @@ def optional(name: str) -> str | None: url_style = env.get("DUCKDB_S3_URL_STYLE") or ("path" if endpoint else "vhost") if url_style not in _URL_STYLES: raise ValueError(f"DUCKDB_S3_URL_STYLE must be one of {_URL_STYLES}, got {url_style!r}") - use_ssl = (env.get("DUCKDB_S3_USE_SSL") or "true").lower() not in ("0", "false", "no") + ssl_word = (env.get("DUCKDB_S3_USE_SSL") or "true").lower() + if ssl_word not in _TRUE_WORDS + _FALSE_WORDS: + raise ValueError( + "DUCKDB_S3_USE_SSL must be one of " + f"{', '.join(_TRUE_WORDS + _FALSE_WORDS)}, got {env.get('DUCKDB_S3_USE_SSL')!r}" + ) + use_ssl = ssl_word in _TRUE_WORDS + access_key_id = optional("DUCKDB_S3_ACCESS_KEY_ID") + secret_access_key = optional("DUCKDB_S3_SECRET_ACCESS_KEY") + if (access_key_id is None) != (secret_access_key is None): + raise ValueError( + "DUCKDB_S3_ACCESS_KEY_ID and DUCKDB_S3_SECRET_ACCESS_KEY: " + "set both or neither" + ) return cls( catalog_host=need("DUCKDB_CATALOG_HOST"), catalog_port=port, @@ -69,8 +89,8 @@ def optional(name: str) -> str | None: catalog_password=env.get("DUCKDB_CATALOG_PASSWORD", ""), data_path=need("DUCKDB_DATA_PATH"), s3_endpoint=endpoint, - s3_access_key_id=optional("DUCKDB_S3_ACCESS_KEY_ID"), - s3_secret_access_key=optional("DUCKDB_S3_SECRET_ACCESS_KEY"), + s3_access_key_id=access_key_id, + s3_secret_access_key=secret_access_key, s3_region=env.get("DUCKDB_S3_REGION") or "us-east-1", s3_url_style=url_style, s3_use_ssl=use_ssl, diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py index 86b4a99..a16ccc8 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -7,7 +7,8 @@ from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey from continuo_duckdb_adapter.infrastructure.ddl import ( EXTENSIONS, DdlRenderer, attach_statement, install_extension, load_extension, - quote_identifier, s3_secret_statement, set_extension_directory, sql_literal, use_catalog, + quote_identifier, s3_secret_statement, secret_texts, set_extension_directory, sql_literal, + use_catalog, ) from continuo_duckdb_adapter.infrastructure.session import check_offline from continuo_duckdb_adapter.infrastructure.settings import REQUIRED_ENV, DuckLakeSettings @@ -60,6 +61,57 @@ def test_invalid_values_name_the_variable(name, value): DuckLakeSettings.from_env({**ENV, name: value}) +def test_repr_hides_the_credentials(): + s = DuckLakeSettings.from_env({ + **ENV, "DUCKDB_CATALOG_PASSWORD": "pw-Hidden-1", + "DUCKDB_S3_ACCESS_KEY_ID": "key-id", "DUCKDB_S3_SECRET_ACCESS_KEY": "s3-Hidden-2", + }) + text = repr(s) + assert "pw-Hidden-1" not in text and "s3-Hidden-2" not in text + assert "catalog_host='localhost'" in text + + +@pytest.mark.parametrize("name", ["DUCKDB_CATALOG_PORT", "DUCKDB_DATA_INLINING_ROW_LIMIT"]) +@pytest.mark.parametrize("value", ["²", "٣", "-1", "1.5"]) +def test_numeric_vars_accept_only_ascii_digits(name, value): + with pytest.raises(ValueError, match=name): + DuckLakeSettings.from_env({**ENV, name: value}) + + +@pytest.mark.parametrize("value,expected", [ + ("true", True), ("TRUE", True), ("1", True), ("yes", True), ("Yes", True), + ("false", False), ("False", False), ("0", False), ("no", False), ("NO", False), +]) +def test_s3_use_ssl_accepts_boolean_words(value, expected): + assert DuckLakeSettings.from_env({**ENV, "DUCKDB_S3_USE_SSL": value}).s3_use_ssl is expected + + +@pytest.mark.parametrize("value", ["flase", "on", "2", "enabled"]) +def test_s3_use_ssl_rejects_anything_else(value): + with pytest.raises(ValueError, match="DUCKDB_S3_USE_SSL"): + DuckLakeSettings.from_env({**ENV, "DUCKDB_S3_USE_SSL": value}) + + +@pytest.mark.parametrize("only", ["DUCKDB_S3_ACCESS_KEY_ID", "DUCKDB_S3_SECRET_ACCESS_KEY"]) +def test_half_configured_static_s3_credentials_are_rejected(only): + with pytest.raises(ValueError) as caught: + DuckLakeSettings.from_env({**ENV, only: "x"}) + message = str(caught.value) + assert "DUCKDB_S3_ACCESS_KEY_ID" in message and "DUCKDB_S3_SECRET_ACCESS_KEY" in message + assert "both or neither" in message + + +def test_secret_texts_cover_every_spelling_of_the_credentials(): + s = DuckLakeSettings.from_env({ + **ENV, "DUCKDB_CATALOG_PASSWORD": "p w'd", "DUCKDB_S3_ACCESS_KEY_ID": "k", + "DUCKDB_S3_SECRET_ACCESS_KEY": "s3cret", + }) + texts = secret_texts(s) + assert {"p w'd", "'p w\\'d'", "p w''d", "s3cret"} <= set(texts) + assert list(texts) == sorted(texts, key=len, reverse=True) + assert secret_texts(DuckLakeSettings.from_env(ENV)) == () + + def test_local_data_path_does_not_use_s3(): assert not DuckLakeSettings.from_env({**ENV, "DUCKDB_DATA_PATH": "/data/lake/"}).uses_s3 diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py index 96d3885..fd18505 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_stack.py +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -24,3 +24,10 @@ def test_unreachable_catalog_raises_a_clear_error(adapter_factory): def test_wrong_catalog_password_raises_a_clear_error(adapter_factory): with pytest.raises(duckdb.Error): adapter_factory(DUCKDB_CATALOG_PASSWORD="definitely-wrong") + + +def test_failed_attach_does_not_echo_the_password(adapter_factory): + wrong = "wrong-pw-Zx81" + with pytest.raises(duckdb.Error) as caught: + adapter_factory(DUCKDB_CATALOG_PASSWORD=wrong) + assert wrong not in str(caught.value) From 0365d3e6669ef97e97bc31a102dbbb8a07633bbe Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:32:44 +0200 Subject: [PATCH 18/29] test(duckdb): keep unit-test imports free of infrastructure Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- adapters/duckdb/tests/conftest.py | 29 ++++++++++++++++++++--------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/adapters/duckdb/tests/conftest.py b/adapters/duckdb/tests/conftest.py index 6ef2fd1..375d1e4 100644 --- a/adapters/duckdb/tests/conftest.py +++ b/adapters/duckdb/tests/conftest.py @@ -10,18 +10,20 @@ import socket import uuid from contextlib import suppress +from typing import TYPE_CHECKING -import boto3 -import psycopg2 import pyarrow as pa import pytest -from continuo_duckdb_adapter.adapter import DuckDBAdapter -from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession -from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings from continuo_duckdb_adapter.application.ports import LakeConflictError, LakeGateway from continuo_duckdb_adapter.application.warehouse import LakeWarehouse +if TYPE_CHECKING: # pragma: no cover + from continuo_duckdb_adapter.adapter import DuckDBAdapter + +# Infrastructure, boto3 and psycopg2 are imported inside the integration +# fixtures below, so the pure domain/application unit tests never load them. + class FakeLakeGateway(LakeGateway): """Records every call as ``(method_name, *args)``; can raise on demand.""" @@ -143,6 +145,9 @@ def lake_env() -> dict[str, str]: """ _require_open(CATALOG_PORT) _require_open(S3_PORT) + from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession + from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings + env = { "DUCKDB_CATALOG_HOST": "localhost", "DUCKDB_CATALOG_PORT": str(CATALOG_PORT), @@ -163,6 +168,8 @@ def lake_env() -> dict[str, str]: @pytest.fixture def adapter_factory(lake_env, monkeypatch): """Build real adapters against the stack; all are closed at teardown.""" + from continuo_duckdb_adapter.adapter import DuckDBAdapter + made: list[DuckDBAdapter] = [] def make(**extra_env: str) -> DuckDBAdapter: @@ -179,12 +186,12 @@ def make(**extra_env: str) -> DuckDBAdapter: @pytest.fixture -def adapter(adapter_factory) -> DuckDBAdapter: +def adapter(adapter_factory) -> "DuckDBAdapter": return adapter_factory() @pytest.fixture -def parquet_adapter(adapter_factory) -> DuckDBAdapter: +def parquet_adapter(adapter_factory) -> "DuckDBAdapter": """Inlining off: every insert becomes a Parquet file, so layout is observable.""" return adapter_factory(DUCKDB_DATA_INLINING_ROW_LIMIT="0") @@ -243,8 +250,10 @@ def run(schema: str) -> list[str]: @pytest.fixture -def catalog_db(): +def catalog_db(lake_env): """A read-only-by-convention cursor on the DuckLake catalog (postgres).""" + import psycopg2 + conn = psycopg2.connect( host="localhost", port=CATALOG_PORT, dbname="catalog", user="continuo", password="continuo" ) @@ -254,7 +263,9 @@ def catalog_db(): @pytest.fixture -def s3(): +def s3(lake_env): + import boto3 + return boto3.client( "s3", endpoint_url=f"http://localhost:{S3_PORT}", aws_access_key_id="minioadmin", aws_secret_access_key="minioadmin", region_name="us-east-1", From 57d3ba75004e5e6d54acfaa88a6fd0dd322f4712 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:33:06 +0200 Subject: [PATCH 19/29] docs: fix stale adapter counts and engine wording Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .github/dependabot.yml | 4 ++-- docs/boundary-contract.md | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 5c29ead..61d63e8 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -7,8 +7,8 @@ version: 2 updates: - # The uv workspace: this package (root), contract/, and the two adapters. All - # four share one lockfile, but Dependabot still needs one entry per pyproject.toml + # The uv workspace: this package (root), contract/, and the three adapters. All + # five share one lockfile, but Dependabot still needs one entry per pyproject.toml # directory to see each package's own dependency declarations. - package-ecosystem: uv directories: diff --git a/docs/boundary-contract.md b/docs/boundary-contract.md index dd56cf6..13f2cb7 100644 --- a/docs/boundary-contract.md +++ b/docs/boundary-contract.md @@ -151,13 +151,13 @@ parse time. `name`, since that changes the derived default name. Iceberg's properties instead all ride on a single `WITH (...)` clause attached to the `CREATE TABLE`'s own `IF NOT EXISTS`, so once the table exists - nothing in `config` is applied at all. DuckLake behaves like Iceberg here: `partitioned_by` and `sorted_by` are applied only when `ensure_table` creates the table, so changing them on an existing table is a silent no-op; the release gate's `build_empty_from_columns` always rebuilds with the new layout. Neither engine is a migration + nothing in `config` is applied at all. DuckLake behaves like Iceberg here: `partitioned_by` and `sorted_by` are applied only when `ensure_table` creates the table, so changing them on an existing table is a silent no-op; the release gate's `build_empty_from_columns` always rebuilds with the new layout. None of the three engines is a migration mechanism for an index/property that already exists under the same - name: flipping `unique: false → true` under a fixed index `name` is a + name (on DuckLake, see above): flipping `unique: false → true` under a fixed index `name` is a **silent no-op** on postgres (the name already resolves, so `IF NOT EXISTS` skips it), and changing `partitioning` on a trino table that already exists is a silent no-op for the reason above — not an applied - change and not an error in either case. The same vocabulary is checked + change and not an error in any of these cases. The same vocabulary is checked ahead of runtime by `build_empty_from_columns`, so a malformed config fails the release gate rather than surfacing in production. - Trino's `format` is case-normalized to uppercase in the emitted DDL, so From 39a15aae92aa8922ad983e66c4d95e95c8d5a0a8 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:53:01 +0200 Subject: [PATCH 20/29] fix(duckdb): redact credentials from every DuckDB error path Every DuckDB error now leaves DuckLakeSession through one choke point (connect, _run, BEGIN/COMMIT/ROLLBACK, fetch, insert, the rollback-failure log) that masks every spelling of the catalog password and S3 secret, keeps the exception type and the LakeConflictError mapping, and raises from None. The catalog password is additionally moved out of the ATTACH conninfo into a private 0600 libpq passfile (passfile= keyword, no PGPASSWORD), so it is absent from duckdb_databases().path and from DuckDB's own connection errors. A failed COMMIT is no longer followed by a spurious ROLLBACK. Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../infrastructure/ddl.py | 12 +- .../infrastructure/passfile.py | 44 ++++ .../infrastructure/session.py | 124 +++++++--- adapters/duckdb/tests/conftest.py | 110 +++++++++ .../duckdb/tests/test_infra_duckdb_session.py | 226 ++++++++++++++++++ .../tests/test_integration_duckdb_stack.py | 62 +++++ 6 files changed, 538 insertions(+), 40 deletions(-) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py create mode 100644 adapters/duckdb/tests/test_infra_duckdb_session.py diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index de700a3..cb1e415 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -76,7 +76,15 @@ def secret_texts(settings: DuckLakeSettings) -> tuple[str, ...]: return tuple(sorted(texts, key=len, reverse=True)) -def attach_statement(settings: DuckLakeSettings, catalog: str) -> str: +def attach_statement( + settings: DuckLakeSettings, catalog: str, passfile: str | None = None +) -> str: + """The ATTACH for the DuckLake catalog. + + With *passfile* the conninfo names that libpq passfile and carries no + password at all; without it the password is written inline. + """ + credential = ("passfile", passfile) if passfile else ("password", settings.catalog_password) conninfo = " ".join( f"{key}={_libpq_value(value)}" for key, value in ( @@ -84,7 +92,7 @@ def attach_statement(settings: DuckLakeSettings, catalog: str) -> str: ("port", settings.catalog_port), ("dbname", settings.catalog_db), ("user", settings.catalog_user), - ("password", settings.catalog_password), + credential, ) ) options = [f"DATA_PATH {sql_literal(settings.data_path)}"] diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py new file mode 100644 index 0000000..46a8e8b --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py @@ -0,0 +1,44 @@ +"""A private libpq passfile that keeps the catalog password out of the DSN. + +DuckDB echoes the ATTACH conninfo in its own error strings and exposes it in +``duckdb_databases().path``, so a ``password=`` keyword would leak there. A +``passfile=`` keyword points libpq at a mode-0600 file instead. Unlike the +``PGPASSWORD`` environment variable this is scoped to the one connection: it +cannot silently authenticate unrelated postgres connections made elsewhere in +the same process (for example by a node script). +""" +from __future__ import annotations + +import atexit +import contextlib +import os +import tempfile + + +def _escape(field: str) -> str: + return field.replace("\\", "\\\\").replace(":", "\\:") + + +def can_store(password: str) -> bool: + """A passfile is line-oriented: it cannot carry a newline in the password.""" + return bool(password) and "\n" not in password and "\r" not in password + + +class Passfile: + """One temporary ``*:*:*:*:`` file, created 0600, removed on close.""" + + def __init__(self, password: str) -> None: + fd, self.path = tempfile.mkstemp(prefix="cpr-pgpass-") # mode 0600 + try: + with os.fdopen(fd, "w") as handle: + handle.write(f"*:*:*:*:{_escape(password)}\n") + except BaseException: + self.close() + raise + # A session that is never closed must not leave the password on disk. + atexit.register(self.close) + + def close(self) -> None: + atexit.unregister(self.close) + with contextlib.suppress(FileNotFoundError): + os.unlink(self.path) diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index a6213c7..d777c7a 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -5,7 +5,9 @@ LOADed first and only INSTALLed when missing, so an image that baked them in at build time starts offline and as a non-root user. -SQL is never logged: the ATTACH and secret statements carry credentials. +SQL is never logged: the secret statement carries credentials. The catalog +password is kept out of the ATTACH conninfo in a private libpq passfile, and +every engine error is redacted on its way out of the session. """ from __future__ import annotations @@ -35,6 +37,7 @@ set_extension_directory, use_catalog, ) +from .passfile import Passfile, can_store from .settings import DuckLakeSettings if TYPE_CHECKING: # pragma: no cover @@ -54,16 +57,6 @@ def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: con.execute(load_extension(name)) -def _redacted(exc: duckdb.Error, secrets: Sequence[str]) -> duckdb.Error: - message = str(exc) - for secret in secrets: - message = message.replace(secret, "***") - try: - return type(exc)(message) - except Exception: # an exception type with a non-standard constructor - return duckdb.Error(message) - - def check_offline(directory: str) -> None: """LOAD every extension from *directory* without ever INSTALLing. @@ -80,58 +73,105 @@ def check_offline(directory: str) -> None: con.close() +def _translate(exc: duckdb.Error, secrets: Sequence[str]) -> Exception: + """*exc* with every spelling of *secrets* masked; type and conflict mapping kept. + + A ``TransactionException`` becomes a ``LakeConflictError``; any other engine + error keeps its type, or falls back to ``duckdb.Error`` when its constructor + does not take a single message. + """ + message = str(exc) + for secret in secrets: + if secret: + message = message.replace(secret, "***") + if isinstance(exc, duckdb.TransactionException): + return LakeConflictError(message) + try: + return type(exc)(message) + except Exception: # an exception type with a non-standard constructor + return duckdb.Error(message) + + +def _redact(exc: BaseException, secrets: Sequence[str]) -> str: + """The text of *exc* with *secrets* masked, for log lines.""" + return str(_translate(exc, secrets)) if isinstance(exc, duckdb.Error) else str(exc) + + class DuckLakeSession(LakeGateway): - def __init__(self, connection: "duckdb.DuckDBPyConnection", renderer: DdlRenderer) -> None: + """Every DuckDB error leaves this class through ``_guarded``: DuckDB echoes + the catalog conninfo (and so the password) in connection errors, and the + runner and harness write ``str(exc)`` to the result block and the pod logs.""" + + def __init__( + self, + connection: "duckdb.DuckDBPyConnection", + renderer: DdlRenderer, + *, + secrets: Sequence[str] = (), + passfile: Passfile | None = None, + ) -> None: self._con = connection self._renderer = renderer + self._secrets = tuple(secrets) + self._passfile = passfile @classmethod def connect(cls, settings: DuckLakeSettings) -> "DuckLakeSession": + secrets = secret_texts(settings) + passfile = Passfile(settings.catalog_password) if can_store(settings.catalog_password) else None con = duckdb.connect() try: - if settings.extension_directory: - con.execute(set_extension_directory(settings.extension_directory)) - for extension in EXTENSIONS: - _load_extension(con, extension) - if settings.uses_s3: - con.execute(s3_secret_statement(settings)) try: - con.execute(attach_statement(settings, CATALOG_ALIAS)) + if settings.extension_directory: + con.execute(set_extension_directory(settings.extension_directory)) + for extension in EXTENSIONS: + _load_extension(con, extension) + if settings.uses_s3: + con.execute(s3_secret_statement(settings)) + con.execute(attach_statement(settings, CATALOG_ALIAS, passfile.path if passfile else None)) + con.execute(use_catalog(CATALOG_ALIAS)) except duckdb.Error as exc: - # The engine echoes the ATTACH conninfo, password included. - raise _redacted(exc, secret_texts(settings)) from None - con.execute(use_catalog(CATALOG_ALIAS)) + raise _translate(exc, secrets) from None except BaseException: con.close() + if passfile: + passfile.close() raise - return cls(con, DdlRenderer(CATALOG_ALIAS)) + return cls(con, DdlRenderer(CATALOG_ALIAS), secrets=secrets, passfile=passfile) # --- plumbing ----------------------------------------------------------- - def _run(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConnection": + @contextlib.contextmanager + def _guarded(self) -> Iterator[None]: try: + yield + except duckdb.Error as exc: + raise _translate(exc, self._secrets) from None + + def _run(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConnection": + with self._guarded(): if parameters is None: return self._con.execute(sql) return self._con.execute(sql, parameters) - except duckdb.TransactionException as exc: - raise LakeConflictError(str(exc)) from exc def _rollback(self) -> None: try: self._con.execute(ROLLBACK) except duckdb.Error as exc: # Never mask the failure that got us here. - logger.warning("rollback failed: %s", exc) + logger.warning("rollback failed: %s", _redact(exc, self._secrets)) @contextlib.contextmanager def transaction(self) -> Iterator[None]: - self._con.execute(BEGIN) + self._run(BEGIN) try: yield - self._run(COMMIT) except BaseException: self._rollback() raise + # Outside the try: a COMMIT that fails has already ended the transaction + # in DuckDB, so there is nothing to roll back. + self._run(COMMIT) # --- LakeGateway -------------------------------------------------------- @@ -142,10 +182,11 @@ def drop_schema_cascade(self, schema: Identifier) -> None: self._run(self._renderer.drop_schema_cascade(schema)) def table_exists(self, table: QualifiedTable) -> bool: - row = self._run( - DdlRenderer.TABLE_EXISTS_QUERY, - [self._renderer.catalog_name, table.schema.name, table.table.name], - ).fetchone() + with self._guarded(): + row = self._run( + DdlRenderer.TABLE_EXISTS_QUERY, + [self._renderer.catalog_name, table.schema.name, table.table.name], + ).fetchone() return bool(row and row[0]) def drop_table_if_exists(self, table: QualifiedTable) -> None: @@ -176,25 +217,32 @@ def apply_layout(self, table: QualifiedTable, layout: TableLayout) -> None: def explain_read(self, sql: str) -> None: # EXPLAIN binds without scanning; the transaction is always rolled back. - self._con.execute(BEGIN) + self._run(BEGIN) try: self._run(self._renderer.explain_read(sql)) finally: self._rollback() def fetch_arrow(self, sql: str) -> "pa.Table": - return self._run(sql).to_arrow_table() + with self._guarded(): + return self._run(sql).to_arrow_table() def delete_all(self, table: QualifiedTable) -> None: self._run(self._renderer.delete_all(table)) def insert_arrow(self, table: QualifiedTable, data: "pa.Table") -> None: view = f"__continuo_load_{uuid.uuid4().hex}" - self._con.register(view, data) + with self._guarded(): + self._con.register(view, data) try: self._run(self._renderer.insert_select(table, data.schema.names, view)) finally: - self._con.unregister(view) + with contextlib.suppress(duckdb.Error): + self._con.unregister(view) def close(self) -> None: - self._con.close() + try: + self._con.close() + finally: + if self._passfile: + self._passfile.close() diff --git a/adapters/duckdb/tests/conftest.py b/adapters/duckdb/tests/conftest.py index 375d1e4..0aec7c2 100644 --- a/adapters/duckdb/tests/conftest.py +++ b/adapters/duckdb/tests/conftest.py @@ -8,6 +8,7 @@ import contextlib import os import socket +import threading import uuid from contextlib import suppress from typing import TYPE_CHECKING @@ -270,3 +271,112 @@ def s3(lake_env): "s3", endpoint_url=f"http://localhost:{S3_PORT}", aws_access_key_id="minioadmin", aws_secret_access_key="minioadmin", region_name="us-east-1", ) + + +def _admin_connection(): + import psycopg2 + + conn = psycopg2.connect( + host="localhost", port=CATALOG_PORT, dbname="catalog", user="continuo", password="continuo" + ) + conn.autocommit = True + return conn + + +@pytest.fixture +def fresh_catalog(lake_env): + """Factory for a brand-new catalog database (optionally owned by its own role). + + Returns the DUCKDB_* env for it. Each gets a private ``DATA_PATH`` prefix, so + it never touches the shared lake. Databases and roles are dropped at teardown. + """ + created: list[tuple[str, str | None]] = [] + + def make(password: str | None = None) -> dict[str, str]: + suffix = uuid.uuid4().hex[:10] + database, role = f"fresh_{suffix}", None + user, secret = lake_env["DUCKDB_CATALOG_USER"], lake_env["DUCKDB_CATALOG_PASSWORD"] + conn = _admin_connection() + try: + with conn.cursor() as cur: + if password is not None: + role = f"cpr_role_{suffix}" + cur.execute(f'CREATE ROLE "{role}" LOGIN PASSWORD %s', (password,)) + user, secret = role, password + owner = f' OWNER "{role}"' if role else "" + cur.execute(f'CREATE DATABASE "{database}"{owner}') + finally: + conn.close() + created.append((database, role)) + return { + **lake_env, + "DUCKDB_CATALOG_DB": database, + "DUCKDB_CATALOG_USER": user, + "DUCKDB_CATALOG_PASSWORD": secret, + "DUCKDB_DATA_PATH": f"s3://{BUCKET}/fresh-{suffix}/", + } + + yield make + conn = _admin_connection() + try: + with conn.cursor() as cur: + for database, role in created: + cur.execute(f'DROP DATABASE IF EXISTS "{database}" WITH (FORCE)') + if role: + cur.execute(f'DROP ROLE IF EXISTS "{role}"') + finally: + conn.close() + + +class CatalogProxy: + """A TCP proxy in front of the catalog whose connections can be cut on demand.""" + + def __init__(self) -> None: + self._listener = socket.socket() + self._listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self._listener.bind(("127.0.0.1", 0)) + self._listener.listen(16) + self.port: int = self._listener.getsockname()[1] + self._stopped = threading.Event() + self._sockets: list[socket.socket] = [] + threading.Thread(target=self._accept, daemon=True).start() + + def _accept(self) -> None: + while not self._stopped.is_set(): + try: + client, _ = self._listener.accept() + upstream = socket.create_connection(("127.0.0.1", CATALOG_PORT)) + except OSError: + return + self._sockets.extend([client, upstream]) + threading.Thread(target=self._pipe, args=(client, upstream), daemon=True).start() + threading.Thread(target=self._pipe, args=(upstream, client), daemon=True).start() + + @staticmethod + def _pipe(source: socket.socket, sink: socket.socket) -> None: + try: + while data := source.recv(65536): + sink.sendall(data) + except OSError: + pass + finally: + for sock in (source, sink): + with suppress(OSError): + sock.close() + + def cut(self) -> None: + """Stop accepting and drop every open connection: a catalog outage.""" + self._stopped.set() + self._listener.close() + for sock in self._sockets: + with suppress(OSError): + sock.shutdown(socket.SHUT_RDWR) + with suppress(OSError): + sock.close() + + +@pytest.fixture +def catalog_proxy(): + proxy = CatalogProxy() + yield proxy + proxy.cut() diff --git a/adapters/duckdb/tests/test_infra_duckdb_session.py b/adapters/duckdb/tests/test_infra_duckdb_session.py new file mode 100644 index 0000000..36d1477 --- /dev/null +++ b/adapters/duckdb/tests/test_infra_duckdb_session.py @@ -0,0 +1,226 @@ +"""DuckLakeSession over a fake connection: the redaction choke point, the +transaction protocol and the passfile. No DuckDB instance is opened.""" +import logging +import os +import stat + +import duckdb +import pytest + +from continuo_duckdb_adapter.application.ports import LakeConflictError +from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable +from continuo_duckdb_adapter.infrastructure.ddl import DdlRenderer, attach_statement +from continuo_duckdb_adapter.infrastructure.passfile import Passfile, can_store +from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession +from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings + +PASSWORD = "p w'd\\x" +# raw, libpq-quoted, SQL-literal form of the libpq-quoted value, SQL-doubled raw +SPELLINGS = [PASSWORD, "'p w\\'d\\\\x'", "''p w\\''d\\\\x''", "p w''d\\x"] +S3_SECRET = "s3-Secret-Zq" +ENV = { + "DUCKDB_CATALOG_HOST": "localhost", "DUCKDB_CATALOG_DB": "catalog", + "DUCKDB_CATALOG_USER": "continuo", "DUCKDB_DATA_PATH": "s3://warehouse/lake/", + "DUCKDB_CATALOG_PASSWORD": PASSWORD, + "DUCKDB_S3_ACCESS_KEY_ID": "key", "DUCKDB_S3_SECRET_ACCESS_KEY": S3_SECRET, +} +T = QualifiedTable.of("s", "t") + + +class FakeConnection: + """Records statements; raises the error queued for a statement prefix.""" + + def __init__(self) -> None: + self.statements: list[str] = [] + self.errors: dict[str, Exception] = {} + self.closed = False + + def execute(self, sql, parameters=None): + self.statements.append(sql) + for prefix, error in self.errors.items(): + if sql.startswith(prefix): + raise error + return self + + def register(self, name, data): + self.statements.append(f"register {name}") + for prefix, error in self.errors.items(): + if "register".startswith(prefix): + raise error + + def unregister(self, name): + self.statements.append(f"unregister {name}") + + def fetchone(self): + return (1,) + + def to_arrow_table(self): + return "arrow" + + def close(self): + self.closed = True + + +def make_session(connection=None): + settings = DuckLakeSettings.from_env(ENV) + from continuo_duckdb_adapter.infrastructure.ddl import secret_texts + + connection = connection or FakeConnection() + return DuckLakeSession(connection, DdlRenderer("lake"), secrets=secret_texts(settings)), connection + + +def leaky(kind, text): + return kind(f'Unable to connect to Postgres at "host=h dbname=d password={text}": refused') + + +@pytest.mark.parametrize("spelling", SPELLINGS + [S3_SECRET]) +def test_every_operation_redacts_every_spelling(spelling): + operations = { + "create_schema_if_not_exists": lambda s: s.create_schema_if_not_exists(Identifier("s")), + "drop_schema_cascade": lambda s: s.drop_schema_cascade(Identifier("s")), + "table_exists": lambda s: s.table_exists(T), + "delete_all": lambda s: s.delete_all(T), + "fetch_arrow": lambda s: s.fetch_arrow("SELECT 1"), + "explain_read": lambda s: s.explain_read("SELECT 1"), + } + prefixes = { + "create_schema_if_not_exists": "CREATE SCHEMA", "drop_schema_cascade": "DROP SCHEMA", + "table_exists": "SELECT count", "delete_all": "DELETE", "fetch_arrow": "SELECT 1", + "explain_read": "EXPLAIN", + } + for name, operation in operations.items(): + session, connection = make_session() + connection.errors[prefixes[name]] = leaky(duckdb.IOException, spelling) + with pytest.raises(duckdb.IOException) as caught: + operation(session) + assert spelling not in str(caught.value), name + assert "***" in str(caught.value), name + assert "host=h dbname=d" in str(caught.value), name + assert caught.value.__cause__ is None and caught.value.__suppress_context__, name + + +def test_insert_arrow_and_begin_are_redacted_too(): + import pyarrow as pa + + session, connection = make_session() + connection.errors["INSERT"] = leaky(duckdb.IOException, PASSWORD) + with pytest.raises(duckdb.IOException) as caught: + session.insert_arrow(T, pa.table({"a": [1]})) + assert PASSWORD not in str(caught.value) + session, connection = make_session() + connection.errors["BEGIN"] = leaky(duckdb.IOException, PASSWORD) + with pytest.raises(duckdb.IOException) as caught, session.transaction(): + pass + assert PASSWORD not in str(caught.value) + + +def test_exception_type_is_preserved_with_a_duckdb_error_fallback(): + session, connection = make_session() + connection.errors["DELETE"] = leaky(duckdb.CatalogException, PASSWORD) + with pytest.raises(duckdb.CatalogException): + session.delete_all(T) + + class Odd(duckdb.Error): + def __init__(self, a, b): # non-standard constructor + super().__init__(f"{a}{b}") + + connection.errors["DELETE"] = Odd("password=", PASSWORD) + with pytest.raises(duckdb.Error) as caught: + session.delete_all(T) + assert type(caught.value) is duckdb.Error + assert PASSWORD not in str(caught.value) + + +def test_transaction_exception_maps_to_a_redacted_conflict(): + session, connection = make_session() + connection.errors["DELETE"] = leaky(duckdb.TransactionException, PASSWORD) + with pytest.raises(LakeConflictError) as caught: + session.delete_all(T) + assert PASSWORD not in str(caught.value) and "***" in str(caught.value) + + +def test_an_empty_secret_is_never_redacted(): + settings = DuckLakeSettings.from_env({**ENV, "DUCKDB_CATALOG_PASSWORD": ""}) + from continuo_duckdb_adapter.infrastructure.ddl import secret_texts + + connection = FakeConnection() + session = DuckLakeSession(connection, DdlRenderer("lake"), secrets=(*secret_texts(settings), "")) + connection.errors["DELETE"] = duckdb.IOException("plain message stays intact") + with pytest.raises(duckdb.IOException, match="plain message stays intact"): + session.delete_all(T) + + +def test_a_failed_commit_is_redacted_and_not_rolled_back(): + session, connection = make_session() + connection.errors["COMMIT"] = leaky(duckdb.TransactionException, PASSWORD) + with pytest.raises(LakeConflictError) as caught, session.transaction(): + pass + assert PASSWORD not in str(caught.value) + assert connection.statements == ["BEGIN", "COMMIT"] + + +def test_a_failing_body_rolls_back_exactly_once(): + session, connection = make_session() + with pytest.raises(ValueError), session.transaction(): + raise ValueError("body") + assert connection.statements == ["BEGIN", "ROLLBACK"] + + +def test_a_clean_body_commits_without_rollback(): + session, connection = make_session() + with session.transaction(): + pass + assert connection.statements == ["BEGIN", "COMMIT"] + + +def test_rollback_failure_is_logged_redacted_and_never_masks_the_body_error(caplog): + session, connection = make_session() + connection.errors["ROLLBACK"] = leaky(duckdb.IOException, PASSWORD) + with caplog.at_level(logging.WARNING, logger="continuo_duckdb_adapter"): + with pytest.raises(ValueError, match="body"), session.transaction(): + raise ValueError("body") + messages = [r.getMessage() for r in caplog.records] + assert any("rollback failed" in m for m in messages) + assert all(PASSWORD not in m for m in messages) + + +def test_explain_read_rolls_back_even_when_the_bind_fails(): + session, connection = make_session() + connection.errors["EXPLAIN"] = duckdb.BinderException("no such column") + with pytest.raises(duckdb.BinderException): + session.explain_read("SELECT nope") + assert connection.statements[0] == "BEGIN" + assert connection.statements[-1] == "ROLLBACK" + + +def test_attach_statement_with_a_passfile_carries_no_password(): + settings = DuckLakeSettings.from_env(ENV) + statement = attach_statement(settings, "lake", "/tmp/continuo pg/pass") + assert "passfile='/tmp/continuo pg/pass'" in statement.replace("''", "'") + assert "password" not in statement and PASSWORD not in statement + + +def test_passfile_is_private_holds_an_escaped_entry_and_is_removed(): + passfile = Passfile("a:b\\c") + try: + assert stat.S_IMODE(os.stat(passfile.path).st_mode) == 0o600 + assert open(passfile.path).read() == "*:*:*:*:a\\:b\\\\c\n" + finally: + passfile.close() + assert not os.path.exists(passfile.path) + passfile.close() # idempotent + + +@pytest.mark.parametrize("password,storable", [ + ("pw", True), ("", False), ("two\nlines", False), ("cr\rhere", False), +]) +def test_can_store(password, storable): + assert can_store(password) is storable + + +def test_close_closes_the_connection_and_removes_the_passfile(): + passfile = Passfile("pw") + connection = FakeConnection() + session = DuckLakeSession(connection, DdlRenderer("lake"), passfile=passfile) + session.close() + assert connection.closed and not os.path.exists(passfile.path) diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py index fd18505..ab87d84 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_stack.py +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -1,4 +1,8 @@ """The adapter connects to the real DuckLake stack, and fails cleanly when it cannot.""" +import logging +import traceback +import uuid + import duckdb import pytest @@ -31,3 +35,61 @@ def test_failed_attach_does_not_echo_the_password(adapter_factory): with pytest.raises(duckdb.Error) as caught: adapter_factory(DUCKDB_CATALOG_PASSWORD=wrong) assert wrong not in str(caught.value) + + +def _outage_operations(): + import pyarrow as pa + + return { + "ensure_schema": lambda a: a.ensure_schema(f"outage_{uuid.uuid4().hex[:8]}"), + "fetch": lambda a: a.fetch("SELECT count(*) FROM information_schema.tables"), + "load": lambda a: a.load("no_such_schema", "t", pa.table({"a": [1]})), + "check_binds": lambda a: a.check_binds("SELECT * FROM information_schema.tables"), + } + + +def _assert_outage_is_clean(adapter, catalog_proxy, caplog, password, *, bare): + """Cut the catalog mid-session; no error text and no log record may carry *password*. + + *bare* also forbids the password token on its own (only sound for a password + distinctive enough not to occur in paths or in the user name). + """ + adapter.fetch("SELECT count(*) FROM information_schema.schemata") + catalog_proxy.cut() + texts: list[str] = [] + with caplog.at_level(logging.DEBUG): + for operation in _outage_operations().values(): + try: + operation(adapter) + except Exception as exc: # noqa: BLE001 - any error type must be clean + texts.append(f"{type(exc).__name__}: {exc}") + texts.append("".join(traceback.format_exception(exc))) + texts.extend(record.getMessage() for record in caplog.records) + assert any("Unable to connect to Postgres" in text for text in texts), texts + for text in texts: + assert f"password={password}" not in text, text + if bare: + assert password not in text, text + + +def test_catalog_outage_mid_session_does_not_leak_the_password(adapter_factory, catalog_proxy, caplog): + adapter = adapter_factory(DUCKDB_CATALOG_HOST="127.0.0.1", DUCKDB_CATALOG_PORT=str(catalog_proxy.port)) + _assert_outage_is_clean(adapter, catalog_proxy, caplog, "continuo", bare=False) + + +def test_redaction_alone_covers_a_password_a_passfile_cannot_hold( + fresh_catalog, adapter_factory, catalog_proxy, caplog +): + """A newline cannot live in a passfile, so this password travels inline in the DSN.""" + env = fresh_catalog(password="Zx-leak-9\nprobe'q") + adapter = adapter_factory( + **{**env, "DUCKDB_CATALOG_HOST": "127.0.0.1", "DUCKDB_CATALOG_PORT": str(catalog_proxy.port)} + ) + _assert_outage_is_clean(adapter, catalog_proxy, caplog, "Zx-leak-9", bare=True) + + +def test_the_password_is_not_visible_in_the_attached_database_path(adapter): + path = adapter.fetch( + "SELECT path FROM duckdb_databases() WHERE database_name = 'lake'" + ).to_pylist()[0]["path"] + assert "password=" not in path From 59887aeb41650e4d15b2855bced6257a1bf28d38 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:56:05 +0200 Subject: [PATCH 21/29] fix(duckdb): retry the first ATTACH race and bake the aws extension Concurrent first ATTACH of a fresh catalog failed for all but one Job on DuckLake's duplicate metadata CREATE: connect() now retries exactly that failure class (bounded, backoff, fresh connection each time). The aws extension (credential_chain S3 secret) joins the single extension list, which moves with check_offline into infrastructure/extensions.py; check_offline now disables autoinstall so it can never silently download an extension. Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .github/workflows/images.yml | 2 +- Dockerfile.duckdb | 2 +- adapters/duckdb/README.md | 2 +- .../infrastructure/ddl.py | 9 +- .../infrastructure/extensions.py | 40 ++++++++ .../infrastructure/session.py | 83 +++++++++------- .../duckdb/tests/test_infra_duckdb_session.py | 94 +++++++++++++++++++ .../tests/test_infra_duckdb_settings_ddl.py | 27 ++++-- .../tests/test_integration_duckdb_stack.py | 21 +++++ scripts/bake_duckdb_extensions.py | 2 +- 10 files changed, 232 insertions(+), 50 deletions(-) create mode 100644 adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py diff --git a/.github/workflows/images.yml b/.github/workflows/images.yml index e73d35a..48e00c5 100644 --- a/.github/workflows/images.yml +++ b/.github/workflows/images.yml @@ -192,7 +192,7 @@ jobs: - name: Extensions load offline as the non-root user run: | docker run --rm --network none --entrypoint python cpr-smoke-duckdb -c \ - "import os; from continuo_duckdb_adapter.infrastructure.session import check_offline; check_offline(os.environ['DUCKDB_EXTENSION_DIRECTORY'])" + "import os; from continuo_duckdb_adapter.infrastructure.extensions import check_offline; check_offline(os.environ['DUCKDB_EXTENSION_DIRECTORY'])" - name: Start duckdb stack run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait - name: Seed source table diff --git a/Dockerfile.duckdb b/Dockerfile.duckdb index f304756..3900bd7 100644 --- a/Dockerfile.duckdb +++ b/Dockerfile.duckdb @@ -24,7 +24,7 @@ RUN set -eu; \ pip install --no-cache-dir -r /tmp/req.txt; \ fi; \ rm -rf /tmp/req.txt /tmp/wheelhouse -# The adapter LOADs ducklake/postgres/httpfs from this directory. They are +# The adapter LOADs ducklake/postgres/httpfs/aws from this directory. They are # installed now, as root, because the runtime user below has no home directory # and the container may have no network. HOME=/tmp gives DuckDB a writable # home for its own bookkeeping. diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md index 193bb2e..e4d7c42 100644 --- a/adapters/duckdb/README.md +++ b/adapters/duckdb/README.md @@ -21,7 +21,7 @@ Job shares one transactional warehouse. Implements | `DUCKDB_S3_REGION` | no | `us-east-1` | | | `DUCKDB_S3_URL_STYLE` | no | `path` with an endpoint, else `vhost` | `path` or `vhost` | | `DUCKDB_S3_USE_SSL` | no | `true` | | -| `DUCKDB_EXTENSION_DIRECTORY` | no | DuckDB default | Where `ducklake`, `postgres`, `httpfs` are loaded from (set in the image) | +| `DUCKDB_EXTENSION_DIRECTORY` | no | DuckDB default | Where `ducklake`, `postgres`, `httpfs` and `aws` (the credential chain) are loaded from (set in the image) | | `DUCKDB_DATA_INLINING_ROW_LIMIT` | no | DuckLake default | `0` writes every insert as a Parquet file instead of inlining small ones in the catalog | ## Physical layout (`config`) diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index cb1e415..3b086e6 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -18,11 +18,6 @@ _LIBPQ_BARE = re.compile(r"[A-Za-z0-9_.:/\-]+") -# The one list of DuckDB extensions the adapter needs: the session LOADs them, -# the image bake script INSTALLs them, the offline image check LOADs them. -EXTENSIONS: tuple[str, ...] = ("ducklake", "postgres", "httpfs") - - BEGIN = "BEGIN" COMMIT = "COMMIT" ROLLBACK = "ROLLBACK" @@ -48,6 +43,10 @@ def set_extension_directory(directory: str) -> str: return f"SET extension_directory = {sql_literal(directory)}" +def disable_extension_autoinstall() -> str: + return "SET autoinstall_known_extensions = false" + + def use_catalog(catalog: str) -> str: return f"USE {quote_identifier(catalog)}" diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py new file mode 100644 index 0000000..a052be9 --- /dev/null +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py @@ -0,0 +1,40 @@ +"""The DuckDB extensions the adapter needs, and the offline check for them. + +The session LOADs them, the image bake script INSTALLs them and the offline +image check LOADs them, so the list and the check live in this one module. The +SQL statements themselves are rendered by ``ddl``. +""" +from __future__ import annotations + +import logging + +import duckdb + +from .ddl import ( + disable_extension_autoinstall, + load_extension, + set_extension_directory, +) + +logger = logging.getLogger("continuo_duckdb_adapter") + +# ``aws`` backs ``CREATE SECRET ... PROVIDER credential_chain``. +EXTENSIONS: tuple[str, ...] = ("ducklake", "postgres", "httpfs", "aws") + + +def check_offline(directory: str) -> None: + """LOAD every extension from *directory* without ever installing one. + + What a baked, offline, non-root image must be able to do; raises + ``duckdb.Error`` naming the first extension that is missing. Autoinstall is + disabled first, otherwise a missing extension would be silently downloaded. + """ + con = duckdb.connect() + try: + con.execute(set_extension_directory(directory)) + con.execute(disable_extension_autoinstall()) + for name in EXTENSIONS: + con.execute(load_extension(name)) + logger.info("loaded %s offline from %s", name, directory) + finally: + con.close() diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index d777c7a..08380a4 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -13,8 +13,9 @@ import contextlib import logging +import time import uuid -from collections.abc import Iterator, Sequence +from collections.abc import Callable, Iterator, Sequence from typing import TYPE_CHECKING import duckdb @@ -26,7 +27,6 @@ from .ddl import ( BEGIN, COMMIT, - EXTENSIONS, ROLLBACK, DdlRenderer, attach_statement, @@ -37,6 +37,7 @@ set_extension_directory, use_catalog, ) +from .extensions import EXTENSIONS from .passfile import Passfile, can_store from .settings import DuckLakeSettings @@ -47,6 +48,21 @@ CATALOG_ALIAS = "lake" +# DuckLake creates its metadata tables on the first ATTACH of a fresh catalog +# and does not guard that against a concurrent first ATTACH (several Jobs of one +# run starting together): all but one fail on the duplicate CREATE. The loser +# retries on a fresh connection and then finds the catalog initialised. Bounded, +# and only this failure class is retried. +_ATTACH_ATTEMPTS = 5 +_ATTACH_BACKOFF_SECONDS = 0.2 + + +def _is_first_attach_race(exc: duckdb.Error) -> bool: + message = str(exc) + return "Failed to initialize DuckLake" in message and ( + "duplicate" in message.lower() or "already exists" in message.lower() + ) + def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: try: @@ -57,22 +73,6 @@ def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: con.execute(load_extension(name)) -def check_offline(directory: str) -> None: - """LOAD every extension from *directory* without ever INSTALLing. - - What a baked, offline, non-root image must be able to do; raises - ``duckdb.Error`` naming the first extension that is missing. - """ - con = duckdb.connect() - try: - con.execute(set_extension_directory(directory)) - for name in EXTENSIONS: - con.execute(load_extension(name)) - logger.info("loaded %s offline from %s", name, directory) - finally: - con.close() - - def _translate(exc: duckdb.Error, secrets: Sequence[str]) -> Exception: """*exc* with every spelling of *secrets* masked; type and conflict mapping kept. @@ -116,28 +116,45 @@ def __init__( self._passfile = passfile @classmethod - def connect(cls, settings: DuckLakeSettings) -> "DuckLakeSession": + def connect( + cls, settings: DuckLakeSettings, *, sleep: Callable[[float], None] = time.sleep + ) -> "DuckLakeSession": secrets = secret_texts(settings) passfile = Passfile(settings.catalog_password) if can_store(settings.catalog_password) else None - con = duckdb.connect() try: - try: - if settings.extension_directory: - con.execute(set_extension_directory(settings.extension_directory)) - for extension in EXTENSIONS: - _load_extension(con, extension) - if settings.uses_s3: - con.execute(s3_secret_statement(settings)) - con.execute(attach_statement(settings, CATALOG_ALIAS, passfile.path if passfile else None)) - con.execute(use_catalog(CATALOG_ALIAS)) - except duckdb.Error as exc: - raise _translate(exc, secrets) from None + for attempt in range(1, _ATTACH_ATTEMPTS + 1): + try: + con = cls._open(settings, passfile) + except duckdb.Error as exc: + if attempt < _ATTACH_ATTEMPTS and _is_first_attach_race(exc): + logger.info("first attach of a fresh catalog raced (attempt %d); retrying", attempt) + sleep(_ATTACH_BACKOFF_SECONDS * attempt) + continue + raise _translate(exc, secrets) from None + return cls(con, DdlRenderer(CATALOG_ALIAS), secrets=secrets, passfile=passfile) except BaseException: - con.close() if passfile: passfile.close() raise - return cls(con, DdlRenderer(CATALOG_ALIAS), secrets=secrets, passfile=passfile) + raise AssertionError("unreachable") # pragma: no cover + + @staticmethod + def _open(settings: DuckLakeSettings, passfile: Passfile | None) -> "duckdb.DuckDBPyConnection": + """One fresh connection, extensions loaded, secret created, catalog attached.""" + con = duckdb.connect() + try: + if settings.extension_directory: + con.execute(set_extension_directory(settings.extension_directory)) + for extension in EXTENSIONS: + _load_extension(con, extension) + if settings.uses_s3: + con.execute(s3_secret_statement(settings)) + con.execute(attach_statement(settings, CATALOG_ALIAS, passfile.path if passfile else None)) + con.execute(use_catalog(CATALOG_ALIAS)) + except BaseException: + con.close() + raise + return con # --- plumbing ----------------------------------------------------------- diff --git a/adapters/duckdb/tests/test_infra_duckdb_session.py b/adapters/duckdb/tests/test_infra_duckdb_session.py index 36d1477..b80890c 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_session.py +++ b/adapters/duckdb/tests/test_infra_duckdb_session.py @@ -224,3 +224,97 @@ def test_close_closes_the_connection_and_removes_the_passfile(): session = DuckLakeSession(connection, DdlRenderer("lake"), passfile=passfile) session.close() assert connection.closed and not os.path.exists(passfile.path) + + +# --- the first-ATTACH initialisation race ----------------------------------- + +RACE = ( + 'Failed to initialize DuckLake: Failed to execute query "CREATE TABLE ' + '"public"."ducklake_metadata"(...)": ERROR: duplicate key value violates ' + 'unique constraint "pg_type_typname_nsp_index"' +) + + +class ConnectScript: + """duckdb.connect() replacement: hands out fake connections whose ATTACH + raises the next scripted error (None = succeed).""" + + def __init__(self, attach_errors): + self.attach_errors = list(attach_errors) + self.connections: list[FakeConnection] = [] + + def __call__(self): + connection = FakeConnection() + error = self.attach_errors.pop(0) if self.attach_errors else None + if error is not None: + connection.errors["ATTACH"] = error + self.connections.append(connection) + return connection + + +@pytest.fixture +def connect_with(monkeypatch): + def run(attach_errors, env=ENV): + script = ConnectScript(attach_errors) + monkeypatch.setattr("continuo_duckdb_adapter.infrastructure.session.duckdb.connect", script) + sleeps: list[float] = [] + settings = DuckLakeSettings.from_env({**env, "DUCKDB_EXTENSION_DIRECTORY": "/ext"}) + return script, sleeps, lambda: DuckLakeSession.connect(settings, sleep=sleeps.append) + + return run + + +def test_the_init_race_is_retried_on_a_fresh_connection(connect_with): + script, sleeps, connect = connect_with([duckdb.Error(RACE), duckdb.Error(RACE), None]) + session = connect() + assert len(script.connections) == 3 + assert [c.closed for c in script.connections] == [True, True, False] + assert len(sleeps) == 2 and sleeps == sorted(sleeps) and all(s > 0 for s in sleeps) + session.close() + + +def test_a_persistent_init_race_surfaces_after_bounded_attempts(connect_with): + from continuo_duckdb_adapter.infrastructure import session as session_module + + script, _, connect = connect_with([duckdb.Error(RACE)] * 50) + with pytest.raises(duckdb.Error, match="Failed to initialize DuckLake"): + connect() + assert len(script.connections) == session_module._ATTACH_ATTEMPTS > 1 + assert all(c.closed for c in script.connections) + + +@pytest.mark.parametrize("error", [ + duckdb.IOException('Unable to connect to Postgres at "host=h": Connection refused'), + duckdb.IOException('Failed to attach DuckLake: FATAL: password authentication failed'), + duckdb.Error("Failed to initialize DuckLake: disk full"), + duckdb.CatalogException("something else"), +]) +def test_every_other_attach_error_is_not_retried(connect_with, error): + script, sleeps, connect = connect_with([error]) + with pytest.raises(type(error)): + connect() + assert len(script.connections) == 1 and sleeps == [] + + +def test_a_retried_race_error_is_still_redacted(connect_with): + from continuo_duckdb_adapter.infrastructure import session as session_module + + leaking = duckdb.Error(f"{RACE} password={PASSWORD}") + _, _, connect = connect_with([leaking] * session_module._ATTACH_ATTEMPTS) + with pytest.raises(duckdb.Error) as caught: + connect() + assert PASSWORD not in str(caught.value) + + +def test_a_failed_connect_removes_the_passfile(connect_with, monkeypatch): + made: list[Passfile] = [] + + def recording(password): + made.append(Passfile(password)) + return made[-1] + + monkeypatch.setattr("continuo_duckdb_adapter.infrastructure.session.Passfile", recording) + _, _, connect = connect_with([duckdb.IOException("Connection refused")]) + with pytest.raises(duckdb.IOException): + connect() + assert made and not any(os.path.exists(p.path) for p in made) diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py index a16ccc8..a5a41a7 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -1,4 +1,7 @@ -"""Settings parsing and SQL rendering: pure string work, no engine.""" +"""Settings parsing, SQL rendering and the extension list. + +Mostly pure string work; the one check that opens a (local, in-memory) DuckDB is +the offline check, which must fail without touching the network.""" import duckdb import pytest @@ -6,11 +9,11 @@ from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey from continuo_duckdb_adapter.infrastructure.ddl import ( - EXTENSIONS, DdlRenderer, attach_statement, install_extension, load_extension, - quote_identifier, s3_secret_statement, secret_texts, set_extension_directory, sql_literal, - use_catalog, + DdlRenderer, attach_statement, disable_extension_autoinstall, install_extension, + load_extension, quote_identifier, s3_secret_statement, secret_texts, + set_extension_directory, sql_literal, use_catalog, ) -from continuo_duckdb_adapter.infrastructure.session import check_offline +from continuo_duckdb_adapter.infrastructure.extensions import EXTENSIONS, check_offline from continuo_duckdb_adapter.infrastructure.settings import REQUIRED_ENV, DuckLakeSettings ENV = { @@ -180,13 +183,21 @@ def test_use_catalog_quotes_the_alias(): assert use_catalog('la"ke') == 'USE "la""ke"' -def test_extensions_are_the_three_the_adapter_needs(): - assert EXTENSIONS == ("ducklake", "postgres", "httpfs") +def test_disable_extension_autoinstall_statement(): + assert disable_extension_autoinstall() == "SET autoinstall_known_extensions = false" + + +def test_extensions_are_the_four_the_adapter_needs(): + # aws backs the credential_chain S3 secret + assert EXTENSIONS == ("ducklake", "postgres", "httpfs", "aws") -def test_check_offline_fails_when_the_extensions_are_not_in_the_directory(tmp_path): +def test_check_offline_fails_without_installing_anything(tmp_path): with pytest.raises(duckdb.Error): check_offline(str(tmp_path)) + # Autoinstall would have downloaded the first extension into the directory + # (and let the check pass wherever there is network); nothing may be written. + assert list(tmp_path.iterdir()) == [] R = DdlRenderer("lake") diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py index ab87d84..d564ef2 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_stack.py +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -93,3 +93,24 @@ def test_the_password_is_not_visible_in_the_attached_database_path(adapter): "SELECT path FROM duckdb_databases() WHERE database_name = 'lake'" ).to_pylist()[0]["path"] assert "password=" not in path + + +def test_concurrent_first_attach_of_a_fresh_catalog_all_succeed(fresh_catalog): + """Eight Jobs starting together against a never-used catalog must all attach.""" + import concurrent.futures + + from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession + from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings + + settings = DuckLakeSettings.from_env(fresh_catalog()) + + def attach(_: int) -> str: + try: + DuckLakeSession.connect(settings).close() + return "ok" + except Exception as exc: # noqa: BLE001 - report every failure + return f"{type(exc).__name__}: {str(exc)[:200]}" + + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: + results = list(pool.map(attach, range(8))) + assert results == ["ok"] * 8 diff --git a/scripts/bake_duckdb_extensions.py b/scripts/bake_duckdb_extensions.py index 622f13b..5006224 100644 --- a/scripts/bake_duckdb_extensions.py +++ b/scripts/bake_duckdb_extensions.py @@ -11,11 +11,11 @@ import duckdb from continuo_duckdb_adapter.infrastructure.ddl import ( - EXTENSIONS, install_extension, load_extension, set_extension_directory, ) +from continuo_duckdb_adapter.infrastructure.extensions import EXTENSIONS logger = logging.getLogger("bake_duckdb_extensions") From 3b6356890909a381b99c6c93f7f46970282758ae Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:57:03 +0200 Subject: [PATCH 22/29] fix(duckdb): read-only bind check and writable temp directory explain_read now runs in BEGIN TRANSACTION READ ONLY (parity with the postgres adapter's backstop). DUCKDB_TEMP_DIRECTORY (set to /tmp/duckdb-tmp in the image) lets in-memory DuckDB spill as the non-root user. The fetch docstring now states that reads are not re-gated there (the harness loads with check_reads=False; check_binds is the gated path). Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- Dockerfile.duckdb | 5 +++++ adapters/duckdb/README.md | 1 + .../application/warehouse.py | 4 +++- .../infrastructure/ddl.py | 5 +++++ .../infrastructure/session.py | 11 +++++++++-- .../infrastructure/settings.py | 2 ++ .../duckdb/tests/test_infra_duckdb_session.py | 12 +++++++++++- .../tests/test_infra_duckdb_settings_ddl.py | 18 ++++++++++++++++-- .../tests/test_integration_duckdb_stack.py | 6 ++++++ 9 files changed, 58 insertions(+), 6 deletions(-) diff --git a/Dockerfile.duckdb b/Dockerfile.duckdb index 3900bd7..e70cee9 100644 --- a/Dockerfile.duckdb +++ b/Dockerfile.duckdb @@ -32,6 +32,11 @@ ENV DUCKDB_EXTENSION_DIRECTORY=/opt/duckdb/extensions HOME=/tmp COPY scripts/bake_duckdb_extensions.py /tmp/bake_duckdb_extensions.py RUN python /tmp/bake_duckdb_extensions.py && rm /tmp/bake_duckdb_extensions.py \ && chmod -R a+rX /opt/duckdb +# In-memory DuckDB spills larger-than-memory sorts/joins to temp_directory, +# which defaults to ./.tmp under the root-owned /app: point it at a directory +# the non-root runtime user can write. +ENV DUCKDB_TEMP_DIRECTORY=/tmp/duckdb-tmp +RUN mkdir -p "$DUCKDB_TEMP_DIRECTORY" && chmod 1777 "$DUCKDB_TEMP_DIRECTORY" ENV CONTRACT_DIR=/app/contracts APP_ROOT=/app PYTHONPATH=/app WORKDIR /app # uid 65532 matches continuo's executor securityContext expectation. diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md index e4d7c42..78d1f08 100644 --- a/adapters/duckdb/README.md +++ b/adapters/duckdb/README.md @@ -22,6 +22,7 @@ Job shares one transactional warehouse. Implements | `DUCKDB_S3_URL_STYLE` | no | `path` with an endpoint, else `vhost` | `path` or `vhost` | | `DUCKDB_S3_USE_SSL` | no | `true` | | | `DUCKDB_EXTENSION_DIRECTORY` | no | DuckDB default | Where `ducklake`, `postgres`, `httpfs` and `aws` (the credential chain) are loaded from (set in the image) | +| `DUCKDB_TEMP_DIRECTORY` | no | `.tmp` in the working directory | Where DuckDB spills larger-than-memory work; must be writable by the runtime user (the image sets `/tmp/duckdb-tmp`) | | `DUCKDB_DATA_INLINING_ROW_LIMIT` | no | DuckLake default | `0` writes every insert as a Parquet file instead of inlining small ones in the catalog | ## Physical layout (`config`) diff --git a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py index 6c02118..6ef39e3 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py +++ b/adapters/duckdb/continuo_duckdb_adapter/application/warehouse.py @@ -131,7 +131,9 @@ def check_binds(self, sql: str) -> None: def fetch(self, sql: str) -> "pa.Table": """Execute one declared read and return the result as an Arrow table. - Not re-gated here: reads are single-read checked earlier, at contract load. + Not re-gated here, matching the postgres adapter: the harness loads the + contract with ``check_reads=False``, so ``check_binds`` is the gated path + (single-read parse gate, then a bind check). """ data = self._gateway.fetch_arrow(sql) seen: set[str] = set() diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index 3b086e6..7207b28 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -19,6 +19,7 @@ _LIBPQ_BARE = re.compile(r"[A-Za-z0-9_.:/\-]+") BEGIN = "BEGIN" +BEGIN_READ_ONLY = "BEGIN TRANSACTION READ ONLY" COMMIT = "COMMIT" ROLLBACK = "ROLLBACK" @@ -43,6 +44,10 @@ def set_extension_directory(directory: str) -> str: return f"SET extension_directory = {sql_literal(directory)}" +def set_temp_directory(directory: str) -> str: + return f"SET temp_directory = {sql_literal(directory)}" + + def disable_extension_autoinstall() -> str: return "SET autoinstall_known_extensions = false" diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index 08380a4..c21aaaa 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -26,6 +26,7 @@ from ..domain.layout import TableLayout from .ddl import ( BEGIN, + BEGIN_READ_ONLY, COMMIT, ROLLBACK, DdlRenderer, @@ -35,6 +36,7 @@ s3_secret_statement, secret_texts, set_extension_directory, + set_temp_directory, use_catalog, ) from .extensions import EXTENSIONS @@ -145,6 +147,10 @@ def _open(settings: DuckLakeSettings, passfile: Passfile | None) -> "duckdb.Duck try: if settings.extension_directory: con.execute(set_extension_directory(settings.extension_directory)) + if settings.temp_directory: + # In-memory DuckDB spills here; its default is under the + # (root-owned) working directory, which a non-root uid cannot use. + con.execute(set_temp_directory(settings.temp_directory)) for extension in EXTENSIONS: _load_extension(con, extension) if settings.uses_s3: @@ -233,8 +239,9 @@ def apply_layout(self, table: QualifiedTable, layout: TableLayout) -> None: self._run(self._renderer.set_sorted_by(table, layout.sort_keys)) def explain_read(self, sql: str) -> None: - # EXPLAIN binds without scanning; the transaction is always rolled back. - self._run(BEGIN) + # EXPLAIN binds without scanning; the transaction is read-only (a backstop, + # as in the postgres adapter) and always rolled back. + self._run(BEGIN_READ_ONLY) try: self._run(self._renderer.explain_read(sql)) finally: diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py index 8d826ad..acfcf42 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/settings.py @@ -36,6 +36,7 @@ class DuckLakeSettings: s3_url_style: str s3_use_ssl: bool extension_directory: str | None + temp_directory: str | None data_inlining_row_limit: int | None @property @@ -95,5 +96,6 @@ def optional(name: str) -> str | None: s3_url_style=url_style, s3_use_ssl=use_ssl, extension_directory=optional("DUCKDB_EXTENSION_DIRECTORY"), + temp_directory=optional("DUCKDB_TEMP_DIRECTORY"), data_inlining_row_limit=None if limit_raw is None else int(limit_raw), ) diff --git a/adapters/duckdb/tests/test_infra_duckdb_session.py b/adapters/duckdb/tests/test_infra_duckdb_session.py index b80890c..0e92b62 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_session.py +++ b/adapters/duckdb/tests/test_infra_duckdb_session.py @@ -189,10 +189,20 @@ def test_explain_read_rolls_back_even_when_the_bind_fails(): connection.errors["EXPLAIN"] = duckdb.BinderException("no such column") with pytest.raises(duckdb.BinderException): session.explain_read("SELECT nope") - assert connection.statements[0] == "BEGIN" + # read-only: the bind check can never write, whatever the read contains + assert connection.statements[0] == "BEGIN TRANSACTION READ ONLY" assert connection.statements[-1] == "ROLLBACK" +def test_connect_applies_the_temp_directory_only_when_set(connect_with): + script, _, connect = connect_with([None], env={**ENV, "DUCKDB_TEMP_DIRECTORY": "/tmp/spill"}) + connect().close() + assert "SET temp_directory = '/tmp/spill'" in script.connections[0].statements + script, _, connect = connect_with([None]) + connect().close() + assert not any("temp_directory" in s for s in script.connections[0].statements) + + def test_attach_statement_with_a_passfile_carries_no_password(): settings = DuckLakeSettings.from_env(ENV) statement = attach_statement(settings, "lake", "/tmp/continuo pg/pass") diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py index a5a41a7..17473b0 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -9,9 +9,9 @@ from continuo_duckdb_adapter.domain.identifiers import Identifier, QualifiedTable from continuo_duckdb_adapter.domain.layout import PartitionKey, SortKey from continuo_duckdb_adapter.infrastructure.ddl import ( - DdlRenderer, attach_statement, disable_extension_autoinstall, install_extension, + BEGIN_READ_ONLY, DdlRenderer, attach_statement, disable_extension_autoinstall, install_extension, load_extension, quote_identifier, s3_secret_statement, secret_texts, - set_extension_directory, sql_literal, use_catalog, + set_extension_directory, set_temp_directory, sql_literal, use_catalog, ) from continuo_duckdb_adapter.infrastructure.extensions import EXTENSIONS, check_offline from continuo_duckdb_adapter.infrastructure.settings import REQUIRED_ENV, DuckLakeSettings @@ -41,6 +41,7 @@ def test_defaults(): assert (s.s3_region, s.s3_use_ssl, s.s3_url_style) == ("us-east-1", True, "vhost") assert s.s3_endpoint is None and s.s3_access_key_id is None assert s.extension_directory is None and s.data_inlining_row_limit is None + assert s.temp_directory is None assert s.uses_s3 @@ -173,6 +174,18 @@ def test_extension_statements_quote_the_name(): assert load_extension('we"ird') == 'LOAD "we""ird"' +def test_temp_directory_is_optional_and_read_from_env(): + assert DuckLakeSettings.from_env(ENV).temp_directory is None + assert DuckLakeSettings.from_env({**ENV, "DUCKDB_TEMP_DIRECTORY": ""}).temp_directory is None + s = DuckLakeSettings.from_env({**ENV, "DUCKDB_TEMP_DIRECTORY": "/tmp/duckdb-tmp"}) + assert s.temp_directory == "/tmp/duckdb-tmp" + + +def test_set_temp_directory_escapes_the_path(): + assert set_temp_directory("/tmp/duckdb-tmp") == "SET temp_directory = '/tmp/duckdb-tmp'" + assert set_temp_directory("/t'mp") == "SET temp_directory = '/t''mp'" + + def test_set_extension_directory_escapes_the_path(): assert set_extension_directory("/opt/duckdb") == "SET extension_directory = '/opt/duckdb'" assert set_extension_directory("/o'pt") == "SET extension_directory = '/o''pt'" @@ -247,6 +260,7 @@ def test_layout_ddl(): def test_read_and_write_statements(): + assert BEGIN_READ_ONLY == "BEGIN TRANSACTION READ ONLY" assert R.explain_read("SELECT 1 -- c") == "EXPLAIN SELECT * FROM (\nSELECT 1 -- c\n) AS __check_binds__" assert R.delete_all(T) == 'DELETE FROM "lake"."s"."t"' assert R.insert_select(T, ["a", "b%"], "src") == ( diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py index d564ef2..5399c8c 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_stack.py +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -95,6 +95,12 @@ def test_the_password_is_not_visible_in_the_attached_database_path(adapter): assert "password=" not in path +def test_temp_directory_setting_reaches_the_engine(adapter_factory, tmp_path): + adapter = adapter_factory(DUCKDB_TEMP_DIRECTORY=str(tmp_path)) + setting = adapter.fetch("SELECT current_setting('temp_directory') AS d").to_pylist()[0]["d"] + assert setting == str(tmp_path) + + def test_concurrent_first_attach_of_a_fresh_catalog_all_succeed(fresh_catalog): """Eight Jobs starting together against a never-used catalog must all attach.""" import concurrent.futures From cd3850cf474832efd1068090456cafc408f8a5bb Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:58:22 +0200 Subject: [PATCH 23/29] test(duckdb): drop_schema with views, failure recovery and file-metadata evidence Real-stack tests: drop_schema removes a schema that contains a view; the connection stays usable after a call rejected before the engine and one rejected by it; the partitioned-write test also checks DuckLake's own file metadata (one partition value per live data file, matching the directory). Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .../tests/test_integration_duckdb_layout.py | 29 +++++++++++++++++- .../test_integration_duckdb_validation.py | 30 +++++++++++++++++++ 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/adapters/duckdb/tests/test_integration_duckdb_layout.py b/adapters/duckdb/tests/test_integration_duckdb_layout.py index dbeabc3..c5f4324 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_layout.py +++ b/adapters/duckdb/tests/test_integration_duckdb_layout.py @@ -56,6 +56,24 @@ def _sort_keys(cur, schema, table): return [(row[0].strip('"'), row[1], row[2]) for row in cur.fetchall()] +_FILE_PARTITION_VALUES = """ +SELECT df.path, fpv.partition_key_index, fpv.partition_value +FROM ducklake_data_file df +JOIN ducklake_file_partition_value fpv + ON fpv.data_file_id = df.data_file_id AND fpv.table_id = df.table_id +JOIN ducklake_table t ON t.table_id = df.table_id +JOIN ducklake_schema s ON s.schema_id = t.schema_id +WHERE s.schema_name = %s AND t.table_name = %s + AND df.end_snapshot IS NULL AND t.end_snapshot IS NULL +ORDER BY df.path, fpv.partition_key_index +""" + + +def _file_partition_values(cur, schema, table) -> list[tuple[str, int, str]]: + cur.execute(_FILE_PARTITION_VALUES, (schema, table)) + return [(row[0], row[1], row[2]) for row in cur.fetchall()] + + def _parquet_files(s3, schema, table) -> dict[str, pa.Table]: """Every Parquet data file DuckLake wrote for the table, read back from MinIO.""" listing = s3.list_objects_v2(Bucket="warehouse", Prefix=f"lake/{schema}/{table}/") @@ -104,7 +122,7 @@ def test_time_transforms_work_on_a_date_column(adapter, schema, catalog_db): # --- partitioned_by: physical files ----------------------------------------- -def test_partitioned_data_lands_in_one_directory_per_value(parquet_adapter, schema, s3): +def test_partitioned_data_lands_in_one_directory_per_value(parquet_adapter, schema, s3, catalog_db): parquet_adapter.ensure_table(schema, "events", COLS, config={"partitioned_by": ["name"]}) parquet_adapter.load(schema, "events", pa.table({ "id": pa.array([1, 2, 3, 4, 5], pa.int32()), @@ -117,6 +135,15 @@ def test_partitioned_data_lands_in_one_directory_per_value(parquet_adapter, sche directory = key.split("/")[-2] rows_per_directory[directory] = rows_per_directory.get(directory, 0) + table.num_rows assert rows_per_directory == {"name=a": 3, "name=b": 2} + # The catalog's own file metadata agrees: exactly one partition value per + # live data file, and it is the value of the directory the file sits in. + recorded = _file_partition_values(catalog_db, schema, "events") + assert len(recorded) == len(files) + assert all(any(key.endswith(path) for key in files) for path, _, _ in recorded) + assert {index for _, index, _ in recorded} == {0} + assert sorted(value for _, _, value in recorded) == ["a", "b"] + for path, _, value in recorded: + assert f"name={value}/" in path # --- sorted_by: catalog metadata -------------------------------------------- diff --git a/adapters/duckdb/tests/test_integration_duckdb_validation.py b/adapters/duckdb/tests/test_integration_duckdb_validation.py index 5719084..40ec346 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_validation.py +++ b/adapters/duckdb/tests/test_integration_duckdb_validation.py @@ -41,6 +41,36 @@ def test_drop_schema_removes_a_schema_containing_tables(adapter, schema, tables_ assert tables_in(schema) == [] +def test_drop_schema_removes_a_schema_containing_a_view(adapter, lake_env, schema): + # The adapter cannot create a view, so a second session (its private _run, + # the only DDL path available to a test) makes one, as a user's own SQL would. + from continuo_duckdb_adapter.infrastructure.session import DuckLakeSession + from continuo_duckdb_adapter.infrastructure.settings import DuckLakeSettings + + adapter.ensure_table(schema, "a", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + other = DuckLakeSession.connect(DuckLakeSettings.from_env(lake_env)) + try: + other._run(f'CREATE VIEW "lake"."{schema}"."v" AS SELECT id FROM "lake"."{schema}"."a"') + finally: + other.close() + views = f"SELECT count(*) AS n FROM duckdb_views() WHERE database_name = 'lake' AND schema_name = '{schema}'" + assert adapter.fetch(views).to_pylist() == [{"n": 1}] + adapter.drop_schema(schema) + assert not _schema_exists(adapter, schema) + assert adapter.fetch(views).to_pylist() == [{"n": 0}] + + +def test_a_rejected_call_leaves_the_connection_usable(adapter, schema, scalar): + with pytest.raises(ValueError): + adapter.ensure_schema("") # rejected before any statement + adapter.ensure_schema(schema) + with pytest.raises(duckdb.Error): + # rejected by the engine, inside the build transaction + adapter.build_empty_from_sql(schema, "t", "SELECT * FROM no_such_table_anywhere") + adapter.ensure_table(schema, "t", [{"name": "id", "type": "INTEGER", "nullable": True}], config={}) + assert scalar(f'SELECT count(*) FROM "{schema}"."t"') == 0 + + def test_drop_schema_on_an_absent_schema_is_a_noop(adapter, schema): adapter.drop_schema(schema) adapter.drop_schema(schema) From 4d4dcc14b7368d6808d98b5d4099b53a62a6f8f8 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 21:59:53 +0200 Subject: [PATCH 24/29] chore: release readiness, compose bucket gate and wording images.yml: publish matrix fail-fast off, scripts/** in the PR path filter, offline extension check runs as uid 65532. CONTRIBUTING: register the PyPI/TestPyPI pending publisher for a new adapter before its first tag. Wording: three engines / three images in README, CONTRIBUTING and the workflow comments. duckdb stack: a bucket-gate service makes up --wait return only once the warehouse bucket exists. Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- .github/workflows/images.yml | 7 +++++-- .github/workflows/publish-pypi.yml | 2 +- .github/workflows/release.yml | 2 +- CONTRIBUTING.md | 11 ++++++++++- README.md | 16 +++++++++------- tests/smoke/duckdb-stack/docker-compose.yml | 20 ++++++++++++++++++++ 6 files changed, 46 insertions(+), 12 deletions(-) diff --git a/.github/workflows/images.yml b/.github/workflows/images.yml index 48e00c5..570ae65 100644 --- a/.github/workflows/images.yml +++ b/.github/workflows/images.yml @@ -7,6 +7,7 @@ on: - "continuo_python_runtime/**" - "contract/**" - "adapters/**" + - "scripts/**" - "pyproject.toml" - ".github/workflows/images.yml" push: @@ -191,7 +192,7 @@ jobs: run: uv run pytest tests/test_image_smoke_validation.py -m image -v - name: Extensions load offline as the non-root user run: | - docker run --rm --network none --entrypoint python cpr-smoke-duckdb -c \ + docker run --rm --network none --user 65532:65532 --entrypoint python cpr-smoke-duckdb -c \ "import os; from continuo_duckdb_adapter.infrastructure.extensions import check_offline; check_offline(os.environ['DUCKDB_EXTENSION_DIRECTORY'])" - name: Start duckdb stack run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml up -d --wait @@ -235,7 +236,7 @@ jobs: if: always() run: docker compose -f tests/smoke/duckdb-stack/docker-compose.yml down -v - # Runs on this same tag push, after the smoke jobs above pass. It builds both + # Runs on this same tag push, after the smoke jobs above pass. It builds all three # engine images for linux/amd64 and linux/arm64 installing the pinned # versions FROM PyPI (WHEEL_SOURCE=pypi, the Dockerfile default), then # pushes them to ghcr.io//continuo-python-runtime-:. The @@ -261,6 +262,8 @@ jobs: needs: [smoke-postgres, smoke-trino, smoke-duckdb] if: startsWith(github.ref, 'refs/tags/v') && !contains(github.ref_name, '-test') strategy: + # One engine's failed push must not cancel the other engines' pushes. + fail-fast: false matrix: engine: [postgres, trino, duckdb] runs-on: ubuntu-latest diff --git a/.github/workflows/publish-pypi.yml b/.github/workflows/publish-pypi.yml index b064cbf..fd1edbd 100644 --- a/.github/workflows/publish-pypi.yml +++ b/.github/workflows/publish-pypi.yml @@ -13,7 +13,7 @@ name: publish-pypi # ship stale adapter code under an unchanged version number. # # Tag glob note: `v*` is the repository's single release pattern — the same tag -# that images.yml builds and pushes both engine images from, so one tag ships +# that images.yml builds and pushes all three engine images from, so one tag ships # the whole release. GitHub Actions tag globs anchor at character 1, so `v*` # claims every tag beginning with "v"; no other pattern may be introduced. # Tag `v-test` publishes to TestPyPI; `v` publishes to real PyPI. diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 1241192..b1665b2 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -3,7 +3,7 @@ name: release # Creates a GitHub Release once BOTH tag-triggered publish workflows # (publish-pypi.yml, images.yml) have finished successfully for this exact # tag. A Release exists here beyond what PyPI already shows because a tag -# also ships two ghcr.io images that PyPI knows nothing about — the Release +# also ships three ghcr.io images that PyPI knows nothing about — the Release # is the one place that says "this tag = these packages + these images." # # Triggered by the same tag push as its two siblings, rather than by diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index c9bddba..69b8f75 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -39,7 +39,7 @@ repository; please do not add headers to new files. ## Development setup Prerequisites: Python 3.14+, [uv](https://docs.astral.sh/uv/), and Docker (only needed -for the Postgres/Trino integration tests and the csv-reader/validation-runner +for the Postgres/Trino/DuckLake integration tests and the csv-reader/validation-runner integration tests, which start a real minio backend via `docker run`). ```bash @@ -98,6 +98,15 @@ scripts/security-scan.sh - **Exact-pinned dependencies.** `continuo-engine-contract` and other in-repo packages are pinned exactly, not with a range — see the comment in `pyproject.toml` for why. +## Before the first release that ships a new adapter + +A new adapter package needs a PyPI and a TestPyPI *pending trusted publisher* +registered **before** its first tag: for `continuo-duckdb-adapter`, workflow file +`publish-pypi.yml` and the same GitHub environments (`pypi`, `testpypi`) as the +other packages. `publish-pypi.yml` uploads every package in one call, so a project +with no publisher registered fails the upload for the whole tag. This is done by +hand on pypi.org / test.pypi.org; nothing in this repository can do it. + ## Code of conduct Participation in this project is governed by our diff --git a/README.md b/README.md index 67266a5..472c731 100644 --- a/README.md +++ b/README.md @@ -29,7 +29,8 @@ Five artifacts come out of this repository: packages for the "build your own container" shape (see below) and for third-party adapter authors to reference. - **Per-engine base images**, one per warehouse engine - (`continuo-python-runtime-postgres`, `continuo-python-runtime-trino`), that + (`continuo-python-runtime-postgres`, `continuo-python-runtime-trino`, + `continuo-python-runtime-duckdb`), that domain repos build `FROM`. Each image bakes in the runtime and a single `WarehouseAdapter` for that engine, and serves both roles that adapter has: the node harness (`ENTRYPOINT ["continuo-runtime"]`, `CMD ["run"]`) and the @@ -37,15 +38,15 @@ Five artifacts come out of this repository: - **`template/`** — a copy-ready domain repo: `Dockerfile`, `contracts/`, `scripts/`, and the `release.yml` CI/CD workflow. -One `vX.Y.Z` git tag releases all of it: `publish-pypi.yml` builds all four +One `vX.Y.Z` git tag releases all of it: `publish-pypi.yml` builds all five PyPI distributions into a single `dist/` and publishes them together, and -`images.yml` builds and pushes both engine images — each installing its +`images.yml` builds and pushes all three engine images — each installing its matching pinned adapter version from that same release — multi-arch under the same tag. ### What this repo owns -This repository owns the entire python-node surface: the engine contract, both +This repository owns the entire python-node surface: the engine contract, the three engine adapters, the validation runner, and the node harness. The former `continuo-validation` repository was merged in — there is no longer a separate validation-side port, adapter class, entry-point group, or image. One @@ -67,10 +68,11 @@ All five are uv workspace members (`[tool.uv.workspace]` in the root installs everything for local development. **All five packages in the table above are published to PyPI**, under the -same `vX.Y.Z` tag. The two engine images then **install the matching pinned +same `vX.Y.Z` tag. The three engine images then **install the matching pinned adapter version from PyPI** — `Dockerfile.postgres` installs `continuo-postgres-adapter==X.Y.Z`, `Dockerfile.trino` installs -`continuo-trino-adapter==X.Y.Z` — rather than building it from this repo's +`continuo-trino-adapter==X.Y.Z` and `Dockerfile.duckdb` installs +`continuo-duckdb-adapter==X.Y.Z` — rather than building it from this repo's source tree, so each image still ships exactly one adapter and the runtime still discovers it through the `continuo_engine.adapters` entry-point group at run time. The image **name** (`continuo-python-runtime-`) and the @@ -78,7 +80,7 @@ adapter's pip **distribution** name (`continuo--adapter`) are two different artifacts of the same adapter — same engine, same version, same runtime behavior, different packaging; see "Build your own container" below for a build shape that installs the pip package directly instead of `FROM` -the image. All four packages are still built, type-checked, and tested by CI +the image. All five packages are still built, type-checked, and tested by CI on every change. ### The result block is a frozen wire contract diff --git a/tests/smoke/duckdb-stack/docker-compose.yml b/tests/smoke/duckdb-stack/docker-compose.yml index c72cb6d..7569503 100644 --- a/tests/smoke/duckdb-stack/docker-compose.yml +++ b/tests/smoke/duckdb-stack/docker-compose.yml @@ -49,3 +49,23 @@ services: mc mb --ignore-existing local/warehouse && echo bucket-ready " + + # `up --wait` waits for healthchecks and for one-shot services to exit, but it + # does not make anything wait for `createbuckets`. This idle service depends on + # it completing and only turns healthy once the bucket can really be stat'ed, + # so `--wait` returns only when the `warehouse` bucket exists. + bucket-gate: + image: ghcr.io/carolsimone/continuo-mc:2025.7.21 + depends_on: + createbuckets: + condition: service_completed_successfully + entrypoint: > + /bin/sh -c " + mc alias set local http://minio:9000 minioadmin minioadmin && + exec sleep infinity + " + healthcheck: + test: ["CMD", "mc", "stat", "local/warehouse"] + interval: 2s + timeout: 2s + retries: 30 From 2d0f5dc571a98930e45db15cb68b5a6073c05dd9 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 22:00:34 +0200 Subject: [PATCH 25/29] docs(duckdb): document parsing rules, limits and parity README: strict settings parsing, credential handling and failure modes, VARCHAR(n) length drop, DuckLake inlining vs layout, PIVOT/UNPIVOT under the single-read gate, and a parity table against the postgres and trino adapters. CHANGELOG: note the new behaviour on the unreleased adapter entry. Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- CHANGELOG.md | 6 +++++ adapters/duckdb/README.md | 52 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 58 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index a998b1b..494bd09 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,12 @@ follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). accepts `partitioned_by` (identity, `bucket`, `year`/`month`/`day`/`hour`) and `sorted_by`; any other key is rejected. Configured with `DUCKDB_*` environment variables (catalog, data path, S3). + The catalog password is kept out of the connection string (private libpq + passfile) and redacted, with the S3 secret, from every engine error; a first + attach of a fresh catalog from concurrent Jobs is retried; `check_binds` runs + in a read-only transaction; `DUCKDB_TEMP_DIRECTORY` (set in the image) gives + DuckDB a writable spill directory; the `aws` extension is baked for the + credential-chain S3 path. - `Dockerfile.duckdb` (engine image, DuckDB extensions baked in for offline, non-root start), the `tests/smoke/duckdb-stack` compose stack, and CI jobs that run the adapter's integration suite and the image smoke test against it. diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md index 78d1f08..cadf9aa 100644 --- a/adapters/duckdb/README.md +++ b/adapters/duckdb/README.md @@ -25,6 +25,30 @@ Job shares one transactional warehouse. Implements | `DUCKDB_TEMP_DIRECTORY` | no | `.tmp` in the working directory | Where DuckDB spills larger-than-memory work; must be writable by the runtime user (the image sets `/tmp/duckdb-tmp`) | | `DUCKDB_DATA_INLINING_ROW_LIMIT` | no | DuckLake default | `0` writes every insert as a Parquet file instead of inlining small ones in the catalog | +Settings are parsed strictly and fail fast with the variable named: + +- `DUCKDB_S3_USE_SSL` accepts `true`/`false`/`1`/`0`/`yes`/`no` (any case); + anything else is an error rather than silently meaning "true". +- `DUCKDB_S3_ACCESS_KEY_ID` and `DUCKDB_S3_SECRET_ACCESS_KEY` are set together or + not at all; one without the other is rejected (it would otherwise fall back to + the credential chain without saying so). +- `DUCKDB_CATALOG_PORT` and `DUCKDB_DATA_INLINING_ROW_LIMIT` take ASCII digits + only. + +## Credentials and failure modes + +The catalog password is handed to libpq through a private (mode 0600) temporary +passfile, not in the connection string, so it does not appear in DuckDB's error +text or in `duckdb_databases()`. Every DuckDB error also passes through one +redaction point that masks the catalog password and the S3 secret (every +spelling) before the error reaches the result block or the pod logs; host, port +and database stay visible. A password containing a line break cannot live in a +passfile and travels inline instead, still redacted from errors. + +A first attach to a brand-new catalog from several Jobs at once can race on +DuckLake's metadata creation; the adapter retries exactly that failure a few +times, and surfaces every other connection error immediately. + ## Physical layout (`config`) - `partitioned_by`: non-empty list of a column name, or @@ -39,6 +63,34 @@ Job shares one transactional warehouse. Implements existing table is a no-op. `build_empty_from_columns` (the release gate) always rebuilds. +## Engine behaviour to know + +- DuckDB drops the length of `VARCHAR(n)` / `CHAR(n)`: the catalog shows plain + `VARCHAR`. Length is enforced by `conform()` for python nodes only, not by the + table. +- DuckLake inlines small inserts into the catalog (see + `DUCKDB_DATA_INLINING_ROW_LIMIT`), so partitioning and sorting are applied to + Parquet files only after a flush; set the limit to `0` when files must be + laid out per write. +- A read is one single query: top-level `PIVOT` / `UNPIVOT` statements are + rejected by the single-read gate. Wrap them: `SELECT * FROM (PIVOT ...)`. + +## Parity with the postgres and trino adapters + +| Behaviour | postgres | trino | duckdb | +|---|---|---|---| +| `drop_schema` | `DROP SCHEMA IF EXISTS ... CASCADE` | same | same (tables and views go too) | +| `ensure_schema` under concurrency | session advisory lock | `IF NOT EXISTS`, tolerating a concurrent creation | `IF NOT EXISTS`, bounded retry on a DuckLake snapshot conflict | +| `check_binds` | `EXPLAIN` in `BEGIN READ ONLY` | `EXPLAIN (TYPE VALIDATE)` | `EXPLAIN` in `BEGIN TRANSACTION READ ONLY`, always rolled back | +| `ensure_table` | creates if absent, layout on create | same | same, in one transaction with the existence check | +| `load` (replace contents) | one transaction | atomic table swap (no multi-statement transactions) | one transaction: `DELETE` then `INSERT` | +| Physical layout keys | `indexes` | `partitioning`, `sorted_by`, `format`, `format_version` | `partitioned_by`, `sorted_by` | + +Not applicable here: postgres `indexes`; trino `format` and `format_version` +(DuckLake always writes Parquet). The postgres advisory lock has no DuckLake +counterpart and is replaced by the bounded conflict retry; the trino table swap +is replaced by the single `DELETE` + `INSERT` transaction. + ## Layout of the code `domain/` (pure rules) <- `application/` (use cases, `LakeGateway` port) <- From dcbbe75ac28dc3c80a5c5383061b18bd2d538824 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Wed, 30 Sep 2026 22:56:35 +0200 Subject: [PATCH 26/29] fix(duckdb): keep PGPASSWORD working with the passfile and narrow conflict mapping libpq fills the password from PGPASSWORD before reading a passfile, so the password now goes inline (still redacted) when PGPASSWORD is set, and also when no passfile can be created. TransactionException maps to LakeConflictError only on the DDL/DML paths; connect and fetch_arrow keep the engine exception type. aws is loaded only for the credential-chain S3 path. Co-Authored-By: Claude Sonnet 5.5 Signed-off-by: Simone Carolini --- adapters/duckdb/README.md | 6 +- .../infrastructure/extensions.py | 11 +++ .../infrastructure/passfile.py | 23 +++++ .../infrastructure/session.py | 39 ++++---- .../duckdb/tests/test_infra_duckdb_session.py | 88 ++++++++++++++++++- .../tests/test_integration_duckdb_stack.py | 11 +++ 6 files changed, 157 insertions(+), 21 deletions(-) diff --git a/adapters/duckdb/README.md b/adapters/duckdb/README.md index cadf9aa..bd54148 100644 --- a/adapters/duckdb/README.md +++ b/adapters/duckdb/README.md @@ -42,8 +42,10 @@ passfile, not in the connection string, so it does not appear in DuckDB's error text or in `duckdb_databases()`. Every DuckDB error also passes through one redaction point that masks the catalog password and the S3 secret (every spelling) before the error reaches the result block or the pod logs; host, port -and database stay visible. A password containing a line break cannot live in a -passfile and travels inline instead, still redacted from errors. +and database stay visible. The password travels inline instead (still redacted from errors) when it +contains a line break, when `PGPASSWORD` is set in the environment (libpq would +prefer it to any passfile), or when no temp file can be created. The `aws` +extension is only loaded for the AWS credential-chain S3 path. A first attach to a brand-new catalog from several Jobs at once can race on DuckLake's metadata creation; the adapter retries exactly that failure a few diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py index a052be9..860e69f 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/extensions.py @@ -10,6 +10,7 @@ import duckdb +from .settings import DuckLakeSettings from .ddl import ( disable_extension_autoinstall, load_extension, @@ -22,6 +23,16 @@ EXTENSIONS: tuple[str, ...] = ("ducklake", "postgres", "httpfs", "aws") +def extensions_to_load(settings: DuckLakeSettings) -> tuple[str, ...]: + """The extensions a session needs: ``aws`` only for the credential-chain S3 path. + + With static S3 credentials (or local data) ``aws`` is not needed, so a + setup that baked only the other extensions never tries to install it. + """ + chain = settings.uses_s3 and not (settings.s3_access_key_id and settings.s3_secret_access_key) + return tuple(name for name in EXTENSIONS if name != "aws" or chain) + + def check_offline(directory: str) -> None: """LOAD every extension from *directory* without ever installing one. diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py index 46a8e8b..218f64c 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/passfile.py @@ -11,9 +11,12 @@ import atexit import contextlib +import logging import os import tempfile +logger = logging.getLogger("continuo_duckdb_adapter") + def _escape(field: str) -> str: return field.replace("\\", "\\\\").replace(":", "\\:") @@ -42,3 +45,23 @@ def close(self) -> None: atexit.unregister(self.close) with contextlib.suppress(FileNotFoundError): os.unlink(self.path) + + +def open_passfile(password: str) -> Passfile | None: + """A passfile for *password*, or None when the password must go inline. + + Inline is the fallback when the password cannot be stored (empty, or with a + line break), when ``PGPASSWORD`` is set (libpq fills the password from it + before it reads any passfile, so the passfile would be ignored), or when no + temp file can be created (for example a read-only root filesystem). + """ + if not can_store(password): + return None + if "PGPASSWORD" in os.environ: + logger.info("PGPASSWORD is set; passing the catalog password inline instead of a passfile") + return None + try: + return Passfile(password) + except OSError as exc: + logger.info("could not create a passfile (%s); passing the catalog password inline", type(exc).__name__) + return None diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index c21aaaa..8c83ddb 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -39,8 +39,8 @@ set_temp_directory, use_catalog, ) -from .extensions import EXTENSIONS -from .passfile import Passfile, can_store +from .extensions import extensions_to_load +from .passfile import Passfile, open_passfile from .settings import DuckLakeSettings if TYPE_CHECKING: # pragma: no cover @@ -75,18 +75,19 @@ def _load_extension(con: "duckdb.DuckDBPyConnection", name: str) -> None: con.execute(load_extension(name)) -def _translate(exc: duckdb.Error, secrets: Sequence[str]) -> Exception: - """*exc* with every spelling of *secrets* masked; type and conflict mapping kept. +def _translate(exc: duckdb.Error, secrets: Sequence[str], *, conflicts: bool = False) -> Exception: + """*exc* with every spelling of *secrets* masked, keeping its type. - A ``TransactionException`` becomes a ``LakeConflictError``; any other engine - error keeps its type, or falls back to ``duckdb.Error`` when its constructor - does not take a single message. + With *conflicts* a ``TransactionException`` becomes a ``LakeConflictError`` + (only the DDL/DML paths mean "retry" by it). Otherwise the type is kept, or + falls back to ``duckdb.Error`` when the constructor does not take a single + message. """ message = str(exc) for secret in secrets: if secret: message = message.replace(secret, "***") - if isinstance(exc, duckdb.TransactionException): + if conflicts and isinstance(exc, duckdb.TransactionException): return LakeConflictError(message) try: return type(exc)(message) @@ -122,7 +123,7 @@ def connect( cls, settings: DuckLakeSettings, *, sleep: Callable[[float], None] = time.sleep ) -> "DuckLakeSession": secrets = secret_texts(settings) - passfile = Passfile(settings.catalog_password) if can_store(settings.catalog_password) else None + passfile = open_passfile(settings.catalog_password) try: for attempt in range(1, _ATTACH_ATTEMPTS + 1): try: @@ -151,7 +152,7 @@ def _open(settings: DuckLakeSettings, passfile: Passfile | None) -> "duckdb.Duck # In-memory DuckDB spills here; its default is under the # (root-owned) working directory, which a non-root uid cannot use. con.execute(set_temp_directory(settings.temp_directory)) - for extension in EXTENSIONS: + for extension in extensions_to_load(settings): _load_extension(con, extension) if settings.uses_s3: con.execute(s3_secret_statement(settings)) @@ -165,17 +166,20 @@ def _open(settings: DuckLakeSettings, passfile: Passfile | None) -> "duckdb.Duck # --- plumbing ----------------------------------------------------------- @contextlib.contextmanager - def _guarded(self) -> Iterator[None]: + def _guarded(self, *, conflicts: bool = True) -> Iterator[None]: try: yield except duckdb.Error as exc: - raise _translate(exc, self._secrets) from None + raise _translate(exc, self._secrets, conflicts=conflicts) from None + + def _raw(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConnection": + if parameters is None: + return self._con.execute(sql) + return self._con.execute(sql, parameters) def _run(self, sql: str, parameters: list | None = None) -> "duckdb.DuckDBPyConnection": with self._guarded(): - if parameters is None: - return self._con.execute(sql) - return self._con.execute(sql, parameters) + return self._raw(sql, parameters) def _rollback(self) -> None: try: @@ -248,8 +252,9 @@ def explain_read(self, sql: str) -> None: self._rollback() def fetch_arrow(self, sql: str) -> "pa.Table": - with self._guarded(): - return self._run(sql).to_arrow_table() + # A read: a TransactionException keeps its type instead of meaning "retry". + with self._guarded(conflicts=False): + return self._raw(sql).to_arrow_table() def delete_all(self, table: QualifiedTable) -> None: self._run(self._renderer.delete_all(table)) diff --git a/adapters/duckdb/tests/test_infra_duckdb_session.py b/adapters/duckdb/tests/test_infra_duckdb_session.py index 0e92b62..34fa44f 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_session.py +++ b/adapters/duckdb/tests/test_infra_duckdb_session.py @@ -317,14 +317,98 @@ def test_a_retried_race_error_is_still_redacted(connect_with): def test_a_failed_connect_removes_the_passfile(connect_with, monkeypatch): + monkeypatch.delenv("PGPASSWORD", raising=False) made: list[Passfile] = [] + original = Passfile def recording(password): - made.append(Passfile(password)) + made.append(original(password)) return made[-1] - monkeypatch.setattr("continuo_duckdb_adapter.infrastructure.session.Passfile", recording) + monkeypatch.setattr("continuo_duckdb_adapter.infrastructure.passfile.Passfile", recording) _, _, connect = connect_with([duckdb.IOException("Connection refused")]) with pytest.raises(duckdb.IOException): connect() assert made and not any(os.path.exists(p.path) for p in made) + + +# --- passfile vs PGPASSWORD, fallbacks, exception types, aws on demand ------- + + +def _attach_of(script): + return next(s for s in script.connections[0].statements if s.startswith("ATTACH")) + + +def test_passfile_is_used_when_pgpassword_is_not_set(connect_with, monkeypatch): + monkeypatch.delenv("PGPASSWORD", raising=False) + script, _, connect = connect_with([None]) + session = connect() + attach = _attach_of(script) + assert "passfile=" in attach and "password=" not in attach + assert session._passfile is not None + session.close() + + +def test_inline_password_is_used_when_pgpassword_is_set(connect_with, monkeypatch): + # libpq fills the password from PGPASSWORD before it ever reads a passfile, + # so the passfile would be ignored and authentication would fail. + monkeypatch.setenv("PGPASSWORD", "some-other-db-password") + script, _, connect = connect_with([None]) + session = connect() + attach = _attach_of(script) + assert "password=" in attach and "passfile=" not in attach + assert session._passfile is None + session.close() + + +def test_inline_password_is_used_when_the_passfile_cannot_be_created(connect_with, monkeypatch, caplog): + monkeypatch.delenv("PGPASSWORD", raising=False) + + def unwritable(*args, **kwargs): + raise FileNotFoundError("no writable temp directory") + + monkeypatch.setattr("continuo_duckdb_adapter.infrastructure.passfile.tempfile.mkstemp", unwritable) + script, _, connect = connect_with([None]) + with caplog.at_level(logging.INFO, logger="continuo_duckdb_adapter"): + session = connect() + assert "password=" in _attach_of(script) and session._passfile is None + messages = [r.getMessage() for r in caplog.records] + assert any("passfile" in m for m in messages) + assert all(PASSWORD not in m for m in messages) + session.close() + + +def test_connect_and_fetch_keep_the_engine_exception_type(connect_with): + _, _, connect = connect_with([leaky(duckdb.TransactionException, PASSWORD)]) + with pytest.raises(duckdb.TransactionException) as caught: + connect() + assert not isinstance(caught.value, LakeConflictError) and PASSWORD not in str(caught.value) + session, connection = make_session() + connection.errors["SELECT 1"] = leaky(duckdb.TransactionException, PASSWORD) + with pytest.raises(duckdb.TransactionException) as caught: + session.fetch_arrow("SELECT 1") + assert not isinstance(caught.value, LakeConflictError) and PASSWORD not in str(caught.value) + + +def test_ddl_and_dml_paths_still_map_to_a_conflict(): + session, connection = make_session() + connection.errors["DELETE"] = leaky(duckdb.TransactionException, PASSWORD) + with pytest.raises(LakeConflictError): + session.delete_all(T) + + +def _loaded(script): + return [s for s in script.connections[0].statements if s.startswith("LOAD")] + + +def test_aws_is_loaded_only_for_the_credential_chain(connect_with): + chain_env = {k: v for k, v in ENV.items() if "S3_" not in k} + script, _, connect = connect_with([None], env=chain_env) + connect().close() + assert 'LOAD "aws"' in _loaded(script) + script, _, connect = connect_with([None]) # static key/secret pair + connect().close() + assert 'LOAD "aws"' not in _loaded(script) and 'LOAD "httpfs"' in _loaded(script) + script, _, connect = connect_with([None], env={**chain_env, "DUCKDB_DATA_PATH": "/data/lake/"}) + connect().close() + assert 'LOAD "aws"' not in _loaded(script) diff --git a/adapters/duckdb/tests/test_integration_duckdb_stack.py b/adapters/duckdb/tests/test_integration_duckdb_stack.py index 5399c8c..223a748 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_stack.py +++ b/adapters/duckdb/tests/test_integration_duckdb_stack.py @@ -88,6 +88,17 @@ def test_redaction_alone_covers_a_password_a_passfile_cannot_hold( _assert_outage_is_clean(adapter, catalog_proxy, caplog, "Zx-leak-9", bare=True) +def test_a_pgpassword_in_the_environment_does_not_break_the_catalog_attach( + adapter_factory, catalog_proxy, caplog, monkeypatch +): + """libpq prefers PGPASSWORD to a passfile, so the password must go inline then.""" + monkeypatch.setenv("PGPASSWORD", "some-other-db-password") + adapter = adapter_factory(DUCKDB_CATALOG_HOST="127.0.0.1", DUCKDB_CATALOG_PORT=str(catalog_proxy.port)) + assert adapter._gateway._passfile is None # inline mode: no passfile was made + assert adapter.fetch("SELECT 41 + 1 AS answer").to_pylist() == [{"answer": 42}] + _assert_outage_is_clean(adapter, catalog_proxy, caplog, "continuo", bare=False) + + def test_the_password_is_not_visible_in_the_attached_database_path(adapter): path = adapter.fetch( "SELECT path FROM duckdb_databases() WHERE database_name = 'lake'" From e1c2f21d51d4fabfbcf2c3820fc055f4097061fa Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Thu, 1 Oct 2026 07:58:59 +0200 Subject: [PATCH 27/29] fix(duckdb): make the catalog-outage test proxy refuse connections on Linux On Linux, close() on a listening socket does not release it while another thread is blocked in accept(), so the proxy kept accepting and serving reconnects after the 'cut'. DuckDB reconnected through the supposed outage, so the three mid-session outage tests passed on macOS but failed on the ubuntu CI runner (they assert the outage error). shutdown() before close() wakes accept() and makes the port refuse; a connection accepted in the race window is closed unserved. A stack-free regression test pins the proxy contract and was shown RED (old cut) and GREEN (fixed cut) in a Linux container. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- adapters/duckdb/tests/conftest.py | 25 +++++++++++++++++-- .../duckdb/tests/test_adapter_duckdb_proxy.py | 25 +++++++++++++++++++ 2 files changed, 48 insertions(+), 2 deletions(-) create mode 100644 adapters/duckdb/tests/test_adapter_duckdb_proxy.py diff --git a/adapters/duckdb/tests/conftest.py b/adapters/duckdb/tests/conftest.py index 0aec7c2..2548653 100644 --- a/adapters/duckdb/tests/conftest.py +++ b/adapters/duckdb/tests/conftest.py @@ -338,6 +338,7 @@ def __init__(self) -> None: self._listener.listen(16) self.port: int = self._listener.getsockname()[1] self._stopped = threading.Event() + self._lock = threading.Lock() self._sockets: list[socket.socket] = [] threading.Thread(target=self._accept, daemon=True).start() @@ -345,10 +346,21 @@ def _accept(self) -> None: while not self._stopped.is_set(): try: client, _ = self._listener.accept() + except OSError: + return + # A connection accepted in the instant between the cut and the + # listener closing must not be served: that would be a reconnect + # through an outage that is supposed to be total. + if self._stopped.is_set(): + client.close() + return + try: upstream = socket.create_connection(("127.0.0.1", CATALOG_PORT)) except OSError: + client.close() return - self._sockets.extend([client, upstream]) + with self._lock: + self._sockets.extend([client, upstream]) threading.Thread(target=self._pipe, args=(client, upstream), daemon=True).start() threading.Thread(target=self._pipe, args=(upstream, client), daemon=True).start() @@ -367,8 +379,17 @@ def _pipe(source: socket.socket, sink: socket.socket) -> None: def cut(self) -> None: """Stop accepting and drop every open connection: a catalog outage.""" self._stopped.set() + # shutdown() before close(): on Linux, close() alone does not release a + # listening socket another thread is blocked in accept() on, so the + # kernel kept accepting (and the accept loop kept serving) new + # connections after the "cut". shutdown() wakes accept() and makes the + # port refuse connections on every platform. + with suppress(OSError): + self._listener.shutdown(socket.SHUT_RDWR) self._listener.close() - for sock in self._sockets: + with self._lock: + sockets = list(self._sockets) + for sock in sockets: with suppress(OSError): sock.shutdown(socket.SHUT_RDWR) with suppress(OSError): diff --git a/adapters/duckdb/tests/test_adapter_duckdb_proxy.py b/adapters/duckdb/tests/test_adapter_duckdb_proxy.py new file mode 100644 index 0000000..bd2dc2c --- /dev/null +++ b/adapters/duckdb/tests/test_adapter_duckdb_proxy.py @@ -0,0 +1,25 @@ +"""The catalog-outage test double must really be an outage, on every platform. + +The mid-session outage integration tests put a TCP proxy in front of the +catalog and "cut" it. On Linux, closing a listening socket does not release it +while another thread is blocked in ``accept()``: the kernel kept accepting, the +accept loop kept proxying, DuckDB reconnected through the "outage", and the +tests passed on macOS but failed (correctly: they assert the outage error) on +the Linux CI runner. This test pins the proxy's own contract, with no stack. +""" +import socket +import time + +import pytest + + +def test_the_catalog_proxy_refuses_new_connections_after_a_cut(catalog_proxy): + time.sleep(0.2) # let the accept thread block in accept(): the Linux failure needs it + catalog_proxy.cut() + with pytest.raises(OSError): + socket.create_connection(("127.0.0.1", catalog_proxy.port), timeout=2) + + +def test_the_catalog_proxy_can_be_cut_twice(catalog_proxy): + catalog_proxy.cut() + catalog_proxy.cut() From 7cb79b1220f09e5815f664d0c510783f286cf915 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Thu, 1 Oct 2026 07:59:05 +0200 Subject: [PATCH 28/29] fix(duckdb): match schema and table names as DuckDB does in the existence check Review comment on #67 (P2): table_exists compared schema and table names with case-sensitive equality, but DuckDB resolves identifiers case-insensitively (quoted or not). ensure_table("analytics", "orders") against an existing "Analytics"."Orders" therefore fell through to CREATE TABLE and failed with 'already exists', so the harness could never load its data. DuckDB folds ASCII case only ("e-acute" does not resolve its upper-case form) while SQL lower() is Unicode-aware, so the query pre-filters with lower() (a superset) and each returned row is confirmed with same_identifier(), which implements the engine's ASCII-only rule. Real-engine tests reproduce the reported error (RED) and cover the other casing, the no-reapply-of-layout rule and distinct non-ASCII names. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .../infrastructure/ddl.py | 22 +++++++- .../infrastructure/session.py | 10 ++-- .../duckdb/tests/test_infra_duckdb_session.py | 46 ++++++++++++++++- .../tests/test_infra_duckdb_settings_ddl.py | 27 ++++++++++ .../tests/test_integration_duckdb_runtime.py | 51 +++++++++++++++++++ 5 files changed, 150 insertions(+), 6 deletions(-) diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py index 7207b28..7bbb4f4 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/ddl.py @@ -28,6 +28,20 @@ def quote_identifier(name: str) -> str: return '"' + name.replace('"', '""') + '"' +_ASCII_LOWER = str.maketrans("ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz") + + +def same_identifier(left: str, right: str) -> bool: + """Whether DuckDB resolves *left* and *right* to the same schema or table name. + + DuckDB matches identifiers case-insensitively, quoted or not, but only for + ASCII letters (``"é"`` does not resolve ``"É"``), and it keeps the casing a + name was created with. SQL ``lower()`` is Unicode-aware, so it cannot stand + in for this comparison. + """ + return left.translate(_ASCII_LOWER) == right.translate(_ASCII_LOWER) + + def sql_literal(value: str) -> str: return "'" + value.replace("'", "''") + "'" @@ -126,9 +140,13 @@ def s3_secret_statement(settings: DuckLakeSettings) -> str: class DdlRenderer: """SQL text for every LakeGateway operation, against one catalog alias.""" + # lower() is only a cheap pre-filter: it matches a superset of what DuckDB + # treats as the same name, so every returned row is confirmed with + # ``same_identifier`` before it counts. TABLE_EXISTS_QUERY = ( - "SELECT count(*) FROM duckdb_tables() " - "WHERE database_name = ? AND schema_name = ? AND table_name = ?" + "SELECT schema_name, table_name FROM duckdb_tables() " + "WHERE lower(database_name) = lower(?) AND lower(schema_name) = lower(?) " + "AND lower(table_name) = lower(?)" ) def __init__(self, catalog: str) -> None: diff --git a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py index 8c83ddb..99f3f31 100644 --- a/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py +++ b/adapters/duckdb/continuo_duckdb_adapter/infrastructure/session.py @@ -34,6 +34,7 @@ install_extension, load_extension, s3_secret_statement, + same_identifier, secret_texts, set_extension_directory, set_temp_directory, @@ -210,11 +211,14 @@ def drop_schema_cascade(self, schema: Identifier) -> None: def table_exists(self, table: QualifiedTable) -> bool: with self._guarded(): - row = self._run( + rows = self._run( DdlRenderer.TABLE_EXISTS_QUERY, [self._renderer.catalog_name, table.schema.name, table.table.name], - ).fetchone() - return bool(row and row[0]) + ).fetchall() + return any( + same_identifier(schema, table.schema.name) and same_identifier(name, table.table.name) + for schema, name in rows + ) def drop_table_if_exists(self, table: QualifiedTable) -> None: self._run(self._renderer.drop_table_if_exists(table)) diff --git a/adapters/duckdb/tests/test_infra_duckdb_session.py b/adapters/duckdb/tests/test_infra_duckdb_session.py index 34fa44f..f3f0ce2 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_session.py +++ b/adapters/duckdb/tests/test_infra_duckdb_session.py @@ -33,6 +33,7 @@ class FakeConnection: def __init__(self) -> None: self.statements: list[str] = [] self.errors: dict[str, Exception] = {} + self.rows: list[tuple] = [] self.closed = False def execute(self, sql, parameters=None): @@ -54,6 +55,9 @@ def unregister(self, name): def fetchone(self): return (1,) + def fetchall(self): + return self.rows + def to_arrow_table(self): return "arrow" @@ -85,7 +89,7 @@ def test_every_operation_redacts_every_spelling(spelling): } prefixes = { "create_schema_if_not_exists": "CREATE SCHEMA", "drop_schema_cascade": "DROP SCHEMA", - "table_exists": "SELECT count", "delete_all": "DELETE", "fetch_arrow": "SELECT 1", + "table_exists": "SELECT schema_name", "delete_all": "DELETE", "fetch_arrow": "SELECT 1", "explain_read": "EXPLAIN", } for name, operation in operations.items(): @@ -412,3 +416,43 @@ def test_aws_is_loaded_only_for_the_credential_chain(connect_with): script, _, connect = connect_with([None], env={**chain_env, "DUCKDB_DATA_PATH": "/data/lake/"}) connect().close() assert 'LOAD "aws"' not in _loaded(script) + + +# --- table_exists follows DuckDB's identifier semantics --------------------- +# +# DuckDB resolves schema and table names case-insensitively (quoted or not) but +# only for ASCII letters, and keeps the stored casing. The existence check must +# agree, or ensure_table on an existing "Analytics"."Orders" would fall through +# to CREATE TABLE and fail with "already exists". + + +@pytest.mark.parametrize("schema,table", [ + ("analytics", "orders"), ("ANALYTICS", "ORDERS"), ("Analytics", "Orders"), ("aNaLyTiCs", "oRdErS"), +]) +def test_table_exists_ignores_ascii_case_like_duckdb(schema, table): + connection = FakeConnection() + connection.rows = [("Analytics", "Orders")] + session, _ = make_session(connection) + assert session.table_exists(QualifiedTable.of(schema, table)) + + +@pytest.mark.parametrize("schema,table", [("analytics", "other"), ("other", "orders")]) +def test_table_exists_is_false_for_a_different_name(schema, table): + connection = FakeConnection() + connection.rows = [("Analytics", "Orders")] + session, _ = make_session(connection) + assert not session.table_exists(QualifiedTable.of(schema, table)) + + +def test_table_exists_does_not_fold_non_ascii_case(): + """SQL lower() is Unicode-aware but DuckDB's catalog is not: "é" is not "É".""" + connection = FakeConnection() + connection.rows = [("s", "É")] + session, _ = make_session(connection) + assert session.table_exists(QualifiedTable.of("s", "É")) + assert not session.table_exists(QualifiedTable.of("s", "é")) + + +def test_table_exists_is_false_when_the_catalog_returns_nothing(): + session, _ = make_session(FakeConnection()) + assert not session.table_exists(T) diff --git a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py index 17473b0..29d8a46 100644 --- a/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py +++ b/adapters/duckdb/tests/test_infra_duckdb_settings_ddl.py @@ -266,3 +266,30 @@ def test_read_and_write_statements(): assert R.insert_select(T, ["a", "b%"], "src") == ( 'INSERT INTO "lake"."s"."t" ("a", "b%") SELECT "a", "b%" FROM "src"' ) + + +# --- identifier equality as DuckDB defines it -------------------------------- + + +def test_same_identifier_is_ascii_case_insensitive(): + from continuo_duckdb_adapter.infrastructure.ddl import same_identifier + + assert same_identifier("Orders", "orders") + assert same_identifier("ORDERS", "oRdErS") + assert same_identifier("order table", "ORDER TABLE") + assert not same_identifier("orders", "order") + + +def test_same_identifier_does_not_fold_non_ascii(): + from continuo_duckdb_adapter.infrastructure.ddl import same_identifier + + assert not same_identifier("É", "é") + assert same_identifier("É", "É") + + +def test_table_exists_query_prefilters_with_lower_and_binds_every_name(): + assert DdlRenderer.TABLE_EXISTS_QUERY == ( + "SELECT schema_name, table_name FROM duckdb_tables() " + "WHERE lower(database_name) = lower(?) AND lower(schema_name) = lower(?) " + "AND lower(table_name) = lower(?)" + ) diff --git a/adapters/duckdb/tests/test_integration_duckdb_runtime.py b/adapters/duckdb/tests/test_integration_duckdb_runtime.py index d0962d3..ecb1b78 100644 --- a/adapters/duckdb/tests/test_integration_duckdb_runtime.py +++ b/adapters/duckdb/tests/test_integration_duckdb_runtime.py @@ -200,3 +200,54 @@ def test_run_node_csv_kind_loads_a_minio_csv_into_duckdb( assert rows == [ {"order_id": 1, "amount": 10.5}, {"order_id": 2, "amount": 20.0}, {"order_id": 3, "amount": 5.25}, ] + + +# --- identifier casing (PR review): DuckDB resolves names case-insensitively -- + + +def _tables_named(adapter, schema: str, table: str) -> int: + rows = adapter.fetch( + "SELECT count(*) AS n FROM duckdb_tables() " + f"WHERE database_name = 'lake' AND schema_name ILIKE '{schema}' AND table_name ILIKE '{table}'" + ).to_pylist() + return rows[0]["n"] + + +def test_ensure_table_finds_an_existing_table_whatever_the_casing(adapter, schema): + """An existing "IT_X"."Orders" must satisfy ensure_table("it_x", "orders"): without + that, ensure_table falls through to CREATE TABLE and fails with "already exists", + and the harness never loads its data.""" + adapter.ensure_table(schema.upper(), "Orders", ID, config={}) + adapter.load(schema.upper(), "Orders", pa.table({"id": pa.array([1, 2, 3], pa.int32())})) + + adapter.ensure_table(schema, "orders", ID, config={}) # must not raise + adapter.ensure_table(schema.upper(), "ORDERS", ID, config={}) # nor this + + assert _tables_named(adapter, schema, "orders") == 1 + assert adapter.fetch(f'SELECT count(*) AS n FROM "{schema}"."orders"').to_pylist() == [{"n": 3}] + + +def test_ensure_table_with_another_casing_does_not_reapply_the_layout(adapter, schema, catalog_db): + cols = [{"name": "id", "type": "INTEGER", "nullable": True}, {"name": "name", "type": "TEXT", "nullable": True}] + adapter.ensure_table(schema.upper(), "Orders", cols, config={"partitioned_by": ["name"]}) + adapter.ensure_table(schema, "orders", cols, config={"sorted_by": ["id"]}) + catalog_db.execute( + "SELECT count(*) FROM ducklake_sort_info si JOIN ducklake_table t ON t.table_id = si.table_id " + "JOIN ducklake_schema s ON s.schema_id = t.schema_id " + "WHERE lower(s.schema_name) = lower(%s) AND lower(t.table_name) = 'orders' " + "AND si.end_snapshot IS NULL AND t.end_snapshot IS NULL", + (schema,), + ) + assert catalog_db.fetchone()[0] == 0 # the table existed: layout is applied only on creation + + +def test_non_ascii_names_that_differ_only_in_case_are_two_tables(adapter, schema): + """DuckDB's catalog folds ASCII case only, so "É" and "é" are different tables and the + existence check must not conflate them (SQL lower() alone would).""" + adapter.ensure_table(schema, "É", ID, config={}) + adapter.ensure_table(schema, "é", ID, config={}) # a different table: must be created + names = adapter.fetch( + f"SELECT table_name FROM duckdb_tables() WHERE database_name = 'lake' AND schema_name = '{schema}' " + "ORDER BY table_name" + ).to_pylist() + assert sorted(row["table_name"] for row in names) == ["É", "é"] From 4cb79fb364e18d4159b25dea150539ccf64ca709 Mon Sep 17 00:00:00 2001 From: Simone Carolini Date: Thu, 1 Oct 2026 07:59:12 +0200 Subject: [PATCH 29/29] ci: select root integration tests by marker and guard that every test is run by a workflow Every collected test (root 569 incl. 30 integration and 3 image, contract 121, postgres 139, trino 176, duckdb 340) is selected by some workflow today, but the root integration step named two files, so any new integration-marked test elsewhere in tests/ would never have run. It now selects by marker (pytest tests -m integration). The images workflow also now triggers on changes to the image tests and the image-requirements pins, which are exactly what its smoke jobs verify. tests/test_ci_test_coverage.py keeps it that way: it fails if a test file lives outside a test root CI runs, if a test root or adapter has no matching pytest step (units and integration), if the root integration step goes back to a file list, if an engine has no image smoke job, or if a new custom marker is used that no workflow selects. Mutation-checked: a stray test file and an unrouted marker both fail it. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_0131xsRkatq65f5x1AW6pahH Signed-off-by: Simone Carolini --- .github/workflows/ci.yml | 5 +- .github/workflows/images.yml | 2 + CONTRIBUTING.md | 2 +- tests/test_ci_test_coverage.py | 134 +++++++++++++++++++++++++++++++++ 4 files changed, 141 insertions(+), 2 deletions(-) create mode 100644 tests/test_ci_test_coverage.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 94bf44b..78d5b56 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,10 +31,13 @@ jobs: # need a real minio backend (started via `docker run` by the # `minio_container` fixture, not docker-compose) -- those run in the # dedicated step below, on the same runner, where docker is available. + # That step selects BY MARKER, never by file name: an explicit file list + # would silently drop every new integration test added elsewhere in + # tests/ (tests/test_ci_test_coverage.py enforces this). - name: Tests (runtime) run: uv run pytest --cov=continuo_python_runtime -m "not image and not integration" -v - name: Tests (runtime, integration) - run: uv run pytest tests/test_csv_readers_integration.py tests/test_validation_runner.py -m integration -v + run: uv run pytest tests -m integration -v - name: Tests (contract) run: uv run pytest contract/tests -v - name: Tests (adapter units) diff --git a/.github/workflows/images.yml b/.github/workflows/images.yml index 570ae65..b37805f 100644 --- a/.github/workflows/images.yml +++ b/.github/workflows/images.yml @@ -4,6 +4,8 @@ on: paths: - "Dockerfile.*" - "tests/smoke/**" + - "tests/test_image_smoke_validation.py" + - "image-requirements-*.txt" - "continuo_python_runtime/**" - "contract/**" - "adapters/**" diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 69b8f75..f1e2472 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -61,7 +61,7 @@ uv run --package continuo-postgres-adapter mypy adapters/postgres/continuo_postg uv run --package continuo-trino-adapter mypy adapters/trino/continuo_trino_adapter uv run --package continuo-duckdb-adapter mypy adapters/duckdb/continuo_duckdb_adapter uv run pytest --cov=continuo_python_runtime -m "not image and not integration" -v -uv run pytest tests/test_csv_readers_integration.py tests/test_validation_runner.py -m integration -v +uv run pytest tests -m integration -v uv run pytest contract/tests -v uv run pytest adapters/postgres/tests adapters/trino/tests -m "not integration" -v uv run pytest adapters/duckdb/tests -m "not integration" -v diff --git a/tests/test_ci_test_coverage.py b/tests/test_ci_test_coverage.py new file mode 100644 index 0000000..08e3612 --- /dev/null +++ b/tests/test_ci_test_coverage.py @@ -0,0 +1,134 @@ +"""Guard: every test in this repo is selected by a CI workflow. + +A test that no workflow runs is a test that cannot fail. The risk is not today's +tests but tomorrow's: a new test directory, a new marker, or a new file with a +marker that an explicit file list in a workflow does not name. This guard fails +when that happens, so CI coverage cannot silently shrink. + +It reads the workflow files as text on purpose: the assertions are about the +commands CI runs, and a YAML round trip would add nothing to them. +""" +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +CI = (ROOT / ".github/workflows/ci.yml").read_text() +IMAGES = (ROOT / ".github/workflows/images.yml").read_text() + +# Directories that never hold this repo's own tests. +_SKIP_DIRS = {".git", ".venv", ".claude", ".worktrees", ".superpowers", "node_modules", + "wheelhouse", "dist", "build", "__pycache__", ".mypy_cache", ".ruff_cache", + ".pytest_cache"} +# pytest's own markers: they never route a test to a different CI step. +_BUILTIN_MARKS = {"parametrize", "skip", "skipif", "xfail", "usefixtures", "filterwarnings"} +# The custom markers pyproject declares, each of which a workflow must select. +_ROUTED_MARKS = {"integration", "image"} + + +def _test_files() -> list[Path]: + found = [] + for pattern in ("test_*.py", "*_test.py"): + for path in ROOT.rglob(pattern): + if not any(part in _SKIP_DIRS for part in path.relative_to(ROOT).parts): + found.append(path) + return sorted(set(found)) + + +def _test_roots() -> list[str]: + """Each test root: the top-level tests/, contract/tests and adapters//tests.""" + roots = ["tests", "contract/tests"] + roots += sorted(f"adapters/{p.name}/tests" for p in (ROOT / "adapters").iterdir() if (p / "tests").is_dir()) + return roots + + +def test_every_test_file_lives_under_a_known_test_root(): + roots = tuple(f"{root}/" for root in _test_roots()) + stray = [str(p.relative_to(ROOT)) for p in _test_files() + if not str(p.relative_to(ROOT)).startswith(roots)] + assert not stray, ( + f"test files outside every test root CI runs: {stray}; move them under one of " + f"{_test_roots()} or add their directory to .github/workflows/ci.yml and this guard" + ) + + +def _ci_pytest_commands() -> list[str]: + return [line.strip() for line in CI.splitlines() if "uv run pytest" in line] + + +def _runs(root: str, *, marker_expr: str | None) -> bool: + """Whether some ci.yml pytest command names *root* and (optionally) selects *marker_expr*.""" + for command in _ci_pytest_commands(): + if root in command.split() and (marker_expr is None or f"-m {marker_expr}" in command): + return True + return False + + +def test_every_test_root_is_run_by_ci(): + # The top-level tests/ is run bare (`pytest ...` uses testpaths = ["tests"]). + for root in _test_roots(): + if root != "tests": + assert _runs(root, marker_expr=None), ( + f"{root} is not named in any pytest command of .github/workflows/ci.yml" + ) + + +def test_root_unit_tests_are_run_by_ci(): + assert 'pytest --cov=continuo_python_runtime -m "not image and not integration"' in CI + + +def test_root_integration_tests_are_selected_by_marker_not_by_file_name(): + """An explicit file list silently drops every new integration test at the repo root.""" + assert "pytest tests -m integration" in CI, ( + "the root integration step must select by marker (`pytest tests -m integration`), " + "not by a hard-coded list of files" + ) + assert "tests/test_csv_readers_integration.py" not in CI + + +def test_every_adapter_runs_its_integration_tests_in_ci(): + for root in _test_roots(): + if not root.startswith("adapters/"): + continue + marked = any( + re.search(r"pytest\.mark\.integration", p.read_text()) + for p in (ROOT / root).glob("test_*.py") + ) + if marked: + assert _runs(root, marker_expr="integration"), ( + f"{root} has integration-marked tests but no `pytest {root} -m integration` step in ci.yml" + ) + + +def test_every_adapter_runs_its_unit_tests_in_ci(): + for root in _test_roots(): + if root.startswith("adapters/"): + assert _runs(root, marker_expr='"not integration"'), ( + f'{root} has no `pytest {root} -m "not integration"` step in ci.yml' + ) + + +def test_image_tests_run_in_the_image_workflow_for_every_engine(): + engines = sorted(p.name for p in (ROOT / "adapters").iterdir() if (p / "tests").is_dir()) + for engine in engines: + assert f"VALIDATION_IMAGE_ENGINE: {engine}" in IMAGES, ( + f"images.yml has no image smoke job for engine {engine!r}" + ) + assert IMAGES.count("-m image") >= len(engines) + + +def test_the_image_workflow_runs_when_the_image_tests_or_pins_change(): + for path in ("tests/test_image_smoke_validation.py", "image-requirements-*.txt", "scripts/**"): + assert f'"{path}"' in IMAGES, f"images.yml pull_request paths do not include {path}" + + +def test_every_custom_marker_in_use_is_routed_to_a_workflow(): + """A new marker would deselect its tests from the `not integration`/`not image` steps + without any workflow selecting them: fail until CI knows about it.""" + used = set() + for path in _test_files(): + used |= set(re.findall(r"pytest\.mark\.(\w+)", path.read_text())) + custom = used - _BUILTIN_MARKS + assert custom <= _ROUTED_MARKS, ( + f"markers {sorted(custom - _ROUTED_MARKS)} are used by tests but no CI step selects them; " + f"route them in .github/workflows and add them to _ROUTED_MARKS here" + )