diff --git a/.changeset/column-expression-kinds.md b/.changeset/column-expression-kinds.md
new file mode 100644
index 00000000..9d5fa574
--- /dev/null
+++ b/.changeset/column-expression-kinds.md
@@ -0,0 +1,15 @@
+---
+"@chkit/core": minor
+"@chkit/codegen": patch
+"@chkit/clickhouse": minor
+"chkit": minor
+"@chkit/plugin-pull": minor
+"@chkit/plugin-codegen": minor
+"@chkit/plugin-backfill": patch
+---
+
+Support `MATERIALIZED`, `ALIAS`, and `EPHEMERAL` column expressions with `defaultKind`, preserving kinds through SQL rendering, pull, snapshots, and drift. Keep existing defaults and snapshots stable; use the existing `fn:` prefix for SQL expressions and allow expressionless `EPHEMERAL` columns.
+
+Generate separate row and insert shapes for tables with special column kinds. Exclude generated columns from automatic backfill insert projections and reject automatic backfills that cannot reconstruct ephemeral inputs. Emit explicit removal of stored expressions, reject automatic storage-kind conversions involving `ALIAS` or `EPHEMERAL`, and never automatically rewrite historical materialized values.
+
+Compare defaults with quote-aware SQL tokens, preserving literal whitespace and distinguishing SQL expressions from string literals. Generate `Row` for default `SELECT *`, `RowExplicit` for all readable columns, and `RowInsert` for writes. Verify live column kinds during backfill planning and local execution, and fail closed on unavailable metadata. Surface historical-value warnings in CLI output and migration files.
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index dca86cbb..1b5eebad 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -118,6 +118,11 @@ jobs:
- name: Test Python text indexes
working-directory: chkit_python
run: python -m pytest tests/test_text_index.py tests/test_text_index_e2e.py -q
+ - name: Test TypeScript column expressions
+ run: bun test packages/cli/src/test/column-expression-safety.test.ts packages/core/src/column-expressions.test.ts packages/clickhouse/src/column-expressions.test.ts packages/cli/src/test/column-expressions.e2e.test.ts
+ - name: Test Python column expressions
+ working-directory: chkit_python
+ run: python -m pytest tests/test_column_expressions.py tests/test_column_expressions_e2e.py -q
obsessiondb:
runs-on: blacksmith-8vcpu-ubuntu-2404
diff --git a/apps/docs/src/content/docs/schema/dsl-reference.mdx b/apps/docs/src/content/docs/schema/dsl-reference.mdx
index 510f2c21..f68e9a63 100644
--- a/apps/docs/src/content/docs/schema/dsl-reference.mdx
+++ b/apps/docs/src/content/docs/schema/dsl-reference.mdx
@@ -281,6 +281,84 @@ Default value for the column.
+### `defaultKind` (optional)
+
+Choose `DEFAULT` (the implicit default), `MATERIALIZED`, `ALIAS`, or `EPHEMERAL`.
+Python also accepts `default_kind`. The `default` field holds the value or expression
+for every kind: strings remain SQL literals, and `fn:` marks raw SQL expressions.
+
+| Kind | Behavior |
+| --- | --- |
+| `DEFAULT` | Stored; the expression applies when the insert omits the value. |
+| `MATERIALIZED` | Computed on insert and stored; cannot be supplied in a normal insert. |
+| `ALIAS` | Computed when explicitly selected; neither stored nor insertable. |
+| `EPHEMERAL` | Input for other column expressions; neither stored nor selectable. |
+
+
+
+ ```ts
+ columns: [
+ { name: 'ts', type: 'DateTime' },
+ { name: 'day', type: 'Date', defaultKind: 'MATERIALIZED', default: 'fn:toDate(ts)' },
+ { name: 'label', type: 'String', defaultKind: 'ALIAS', default: 'fn:toString(day)' },
+ { name: 'raw', type: 'String', defaultKind: 'EPHEMERAL' },
+ { name: 'size', type: 'UInt64', default: 'fn:length(raw)' },
+ ]
+ ```
+
+
+ ```python
+ columns=[
+ {"name": "ts", "type": "DateTime"},
+ {"name": "day", "type": "Date", "default_kind": "MATERIALIZED", "default": "fn:toDate(ts)"},
+ {"name": "label", "type": "String", "default_kind": "ALIAS", "default": "fn:toString(day)"},
+ {"name": "raw", "type": "String", "default_kind": "EPHEMERAL"},
+ {"name": "size", "type": "UInt64", "default": "fn:length(raw)"},
+ ]
+ ```
+
+
+
+`MATERIALIZED` and `ALIAS` require a value or expression. `EPHEMERAL` may omit it;
+supply these inputs with an explicit insert column list. ClickHouse normally
+excludes all three special kinds from `SELECT *`.
+
+`pull`, snapshots, and drift preserve the column kind. Omitted kind and explicit
+`DEFAULT` compare identically, so existing snapshots need no migration. Keep the
+base type in `type`; do not embed `MATERIALIZED ...` in the type string.
+
+Changing a stored expression emits `MODIFY COLUMN`, without rewriting historical
+values. The planner reports this explicitly in human and JSON output and includes
+the warning in migration SQL. Use a separately reviewed `ALTER TABLE ... MATERIALIZE
+COLUMN ...` only when you need that rewrite and the required inputs still exist.
+Values derived from discarded `EPHEMERAL` inputs cannot be reconstructed this way.
+Removing a `DEFAULT` or `MATERIALIZED` expression emits an explicit `REMOVE` clause.
+
+Converting an existing column to or from `ALIAS` or `EPHEMERAL` is not supported
+by the migration generator. This release does not provide snapshot adoption for
+manually applied conversions. The planner rejects these changes instead of dropping
+and recreating columns.
+
+Generated `Row` models describe the default `SELECT *` result, excluding
+`MATERIALIZED`, `ALIAS`, and `EPHEMERAL`. Tables using special kinds also get:
+
+- `RowExplicit`, containing all readable columns, for queries that name those columns explicitly.
+- `RowInsert`, excluding `MATERIALIZED` and `ALIAS` and including `EPHEMERAL` inputs.
+
+These names are suffixes: for example, `DefaultEventsRow`, `DefaultEventsRowExplicit`,
+and `DefaultEventsRowInsert`. TypeScript emits matching Zod schemas when enabled;
+Python emits Pydantic models. Generated TypeScript ingest helpers use `RowInsert`.
+Field requiredness is unchanged: insert models require their declared input fields,
+including defaults. Non-default `asterisk_include_materialized_columns` or
+`asterisk_include_alias_columns` settings change the result shape; use an explicit
+projection and its model when selecting those columns.
+
+Automatic backfill projections omit `MATERIALIZED` and `ALIAS` columns. Planning
+and local execution both check live target column kinds, including when the local
+schema is missing or cannot load. Missing, unknown, or unreadable metadata blocks
+the backfill. Targets with `EPHEMERAL` columns require explicit SQL with an input
+mapping because these inputs cannot be recovered from stored rows.
+
### `comment` (string, optional)
Column-level comment rendered in SQL.
diff --git a/bun.lock b/bun.lock
index 918afc4c..0e7431ee 100644
--- a/bun.lock
+++ b/bun.lock
@@ -59,6 +59,9 @@
"fast-glob": "^3.3.2",
"p-retry": "^7.1.1",
},
+ "devDependencies": {
+ "@clickhouse/client": "^1.11.0",
+ },
"optionalDependencies": {
"@chkit/plugin-obsessiondb": "workspace:*",
},
diff --git a/chkit_python/CHANGELOG.md b/chkit_python/CHANGELOG.md
index 2bbf2b55..e3545f76 100644
--- a/chkit_python/CHANGELOG.md
+++ b/chkit_python/CHANGELOG.md
@@ -3,6 +3,11 @@
## Unreleased
### Added
+- Compare column defaults with quote-aware SQL tokens and correctly escape literal backslashes.
+- Generate separate `Row` (default `SELECT *`), `RowExplicit`, and `RowInsert` models.
+- Check live column metadata before backfill planning and local execution; block unknown metadata and unrecoverable `EPHEMERAL` inputs.
+- Warn about unchanged historical values in migration output and SQL.
+
- Add `SkipIndexText` for full-text index generation, introspection, pull, and drift.
Preserve quoted SQL literals, normalize ClickHouse’s fixed granularity, and reject
malformed or unsupported metadata. Exercise adversarial round trips and actual
@@ -18,6 +23,11 @@
- Preserve quoted clause names, delimiters, whitespace, and escaped trailing
backslashes in table introspection and migration statement splitting.
+
+- Support `default_kind` / `defaultKind` for `DEFAULT`, `MATERIALIZED`, `ALIAS`, and `EPHEMERAL` columns through SQL rendering, introspection, pull, snapshots, and drift. Existing defaults and snapshots remain compatible. Use `fn:` for SQL expressions; expressionless `EPHEMERAL` is supported.
+- Generate separate read/insert models for tables with special column kinds, and make backfill projections respect generated columns. Automatic backfills with ephemeral inputs require explicit SQL input mappings.
+- Emit explicit removal of stored column expressions; reject automatic kind conversions involving `ALIAS` or `EPHEMERAL`. Expression changes never automatically materialize historical data.
+
## 0.2.0 — 2026-08-10
**Full parity with the TypeScript chkit.** Every remaining gap is closed;
diff --git a/chkit_python/src/chkit/__init__.py b/chkit_python/src/chkit/__init__.py
index fe1fe444..e1cebfdb 100644
--- a/chkit_python/src/chkit/__init__.py
+++ b/chkit_python/src/chkit/__init__.py
@@ -7,6 +7,7 @@
ChxUserClickHouseConfig,
ChxUserConfig,
ChxValidationError,
+ ColumnDefaultKind,
ColumnDefinition,
DictionaryAttribute,
DictionaryDefinition,
@@ -60,6 +61,7 @@
"ChxUserClickHouseConfig",
"ChxUserConfig",
"ChxValidationError",
+ "ColumnDefaultKind",
"ColumnDefinition",
"DictionaryAttribute",
"DictionaryDefinition",
diff --git a/chkit_python/src/chkit/cli/commands/drift_compare.py b/chkit_python/src/chkit/cli/commands/drift_compare.py
index 71869ecb..829f60fa 100644
--- a/chkit_python/src/chkit/cli/commands/drift_compare.py
+++ b/chkit_python/src/chkit/cli/commands/drift_compare.py
@@ -29,8 +29,10 @@
TableDefinition,
)
from chkit.core.projection import is_index_projection, normalize_projection_index
+from chkit.core.sql import render_default
from chkit.core.sql_normalizer import normalize_engine, normalize_sql_fragment
from chkit.core.text_index import render_text_index_type, text_index_fingerprint
+from chkit.core.text_index_sql import text_expression_fingerprint, text_sql_fingerprint
_MIN_QUOTED_LEN = 2
@@ -216,30 +218,16 @@ def summarize_drift_reasons(
def _normalize_column_shape(column: ColumnDefinition) -> str:
- def _normalize_default_value(value: str) -> str:
- normalized = normalize_sql_fragment(value)
- if (
- len(normalized) >= _MIN_QUOTED_LEN
- and normalized[0] == "'"
- and normalized[-1] == "'"
- ):
- inner = normalized[1:-1]
- return inner.replace("''", "'")
- return normalized
-
- if column.default is None:
- normalized_default = ""
- else:
- as_string = str(column.default)
- if as_string.startswith("fn:"):
- normalized_default = _normalize_default_value(as_string[3:])
- else:
- normalized_default = _normalize_default_value(as_string)
+ normalized_default = (
+ "" if column.default is None
+ else text_sql_fingerprint(text_expression_fingerprint(render_default(column.default)))
+ )
parts = [
f"type={str(column.type).strip()}",
f"nullable={'1' if column.nullable else '0'}",
f"default={normalized_default}",
+ f"defaultKind={column.default_kind or 'DEFAULT'}",
f"comment={(column.comment or '').strip()}",
]
return "|".join(parts)
@@ -337,7 +325,12 @@ def compare_table_shape( # noqa: PLR0912, PLR0915
"""Compare every shape-bearing field on the table. Returns None if identical."""
column_diff = diff_by_name(
expected.columns,
- actual.columns,
+ [
+ column.model_copy(update={"default": f"fn:{column.default}"})
+ if isinstance(column.default, str) and not column.default.startswith("fn:")
+ else column
+ for column in actual.columns
+ ],
lambda c: c.name,
_normalize_column_shape,
)
diff --git a/chkit_python/src/chkit/cli/commands/generate.py b/chkit_python/src/chkit/cli/commands/generate.py
index e3e95b6a..de54792a 100644
--- a/chkit_python/src/chkit/cli/commands/generate.py
+++ b/chkit_python/src/chkit/cli/commands/generate.py
@@ -429,7 +429,9 @@ def run( # noqa: PLR0911, PLR0912, PLR0915, PLR0917
plan, config.clickhouse.cluster if config.clickhouse else None
)
- dictionary_password_warnings = detect_dictionary_password_warnings(plan)
+ dictionary_password_warnings = detect_dictionary_password_warnings(plan) + [
+ op.warning for op in plan.operations if op.warning
+ ]
if not plan.operations:
if output_json:
diff --git a/chkit_python/src/chkit/cli/commands/pull.py b/chkit_python/src/chkit/cli/commands/pull.py
index 9f2ed103..689e29e3 100644
--- a/chkit_python/src/chkit/cli/commands/pull.py
+++ b/chkit_python/src/chkit/cli/commands/pull.py
@@ -107,7 +107,11 @@ def _introspected_table_to_definition(
database=item.database,
name=item.name,
engine=item.engine or "MergeTree",
- columns=list(item.columns),
+ columns=[
+ column.model_copy(update={"default": f"fn:{column.default}"})
+ if isinstance(column.default, str) else column
+ for column in item.columns
+ ],
primary_key=[] if kafka else _split_clause(item.primary_key) or [item.columns[0].name],
order_by=[] if kafka else _split_clause(item.order_by) or [item.columns[0].name],
unique_key=_split_clause(item.unique_key) or None,
diff --git a/chkit_python/src/chkit/cli/commands/pull_render.py b/chkit_python/src/chkit/cli/commands/pull_render.py
index c723cca5..0d4f1656 100644
--- a/chkit_python/src/chkit/cli/commands/pull_render.py
+++ b/chkit_python/src/chkit/cli/commands/pull_render.py
@@ -219,6 +219,8 @@ def _render_column(column: ColumnDefinition) -> str:
]
if column.nullable:
parts.append("nullable=True")
+ if column.default_kind and column.default_kind != "DEFAULT":
+ parts.append(f"default_kind={_render_string(column.default_kind)}")
if column.default is not None:
parts.append(f"default={_render_literal(column.default)}")
if column.comment:
diff --git a/chkit_python/src/chkit/cli/migration_store.py b/chkit_python/src/chkit/cli/migration_store.py
index 0acd9756..82df6207 100644
--- a/chkit_python/src/chkit/cli/migration_store.py
+++ b/chkit_python/src/chkit/cli/migration_store.py
@@ -150,7 +150,9 @@ def _build_migration_content(
for s in plan.rename_suggestions
]
body_blocks = [
- f"-- operation: {op.type} key={op.key} risk={op.risk}\n{op.sql}"
+ f"-- operation: {op.type} key={op.key} risk={op.risk}\n"
+ + ("-- Warning: " + " ".join(op.warning.splitlines()) + "\n" if op.warning else "")
+ + op.sql
for op in plan.operations
]
body = "\n\n".join(body_blocks)
diff --git a/chkit_python/src/chkit/clickhouse/introspect.py b/chkit_python/src/chkit/clickhouse/introspect.py
index 7d5c6ce3..d76d3f1e 100644
--- a/chkit_python/src/chkit/clickhouse/introspect.py
+++ b/chkit_python/src/chkit/clickhouse/introspect.py
@@ -48,9 +48,10 @@
SkipIndexText,
SkipIndexTokenBF,
)
+from chkit.core.sql import render_default
from chkit.core.sql_normalizer import normalize_sql_fragment
from chkit.core.text_index import parse_text_index_params
-from chkit.core.text_index_sql import normalize_text_index_sql
+from chkit.core.text_index_sql import normalize_text_index_sql, text_sql_fingerprint
SchemaObjectKind: TypeAlias = Literal["table", "view", "materialized_view", "dictionary"]
@@ -148,20 +149,30 @@ def normalize_column_from_system_row(row: SystemColumnRow) -> ColumnDefinition:
nullable = bool(inner)
default_value: str | None = None
- if row.default_expression and row.default_kind == "DEFAULT":
- default_value = normalize_sql_fragment(row.default_expression)
-
+ kind = row.default_kind
+ if kind and kind not in {"DEFAULT", "MATERIALIZED", "ALIAS", "EPHEMERAL"}:
+ raise ValueError(f"Unsupported column default kind: {kind}")
+ if row.default_expression and kind:
+ # Preserve whitespace inside SQL string literals when pulling expressions.
+ default_value = row.default_expression.strip()
+
+ if kind == "EPHEMERAL" and default_value is not None and (
+ text_sql_fingerprint(str(default_value))
+ == text_sql_fingerprint(f"defaultValueOfTypeName({render_default(row.type)})")
+ ):
+ default_value = None
codec_steps = parse_codec(row.compression_codec)
comment = row.comment.strip() if row.comment is not None else None
- return ColumnDefinition(
- name=row.name,
- type=type_,
- nullable=nullable or None,
- default=default_value,
- comment=comment or None,
- codec=codec_steps,
- )
+ return ColumnDefinition.model_validate({
+ "name": row.name,
+ "type": type_,
+ "nullable": nullable or None,
+ "default": default_value,
+ "defaultKind": kind if kind and kind != "DEFAULT" else None,
+ "comment": comment or None,
+ "codec": codec_steps,
+ })
def _split_int_args(args: str | None) -> list[int]:
diff --git a/chkit_python/src/chkit/core/__init__.py b/chkit_python/src/chkit/core/__init__.py
index ae9262ff..67420ce8 100644
--- a/chkit_python/src/chkit/core/__init__.py
+++ b/chkit_python/src/chkit/core/__init__.py
@@ -36,6 +36,7 @@
ChxValidationError,
ColumnCodec,
ColumnCodecSpec,
+ ColumnDefaultKind,
ColumnDefinition,
DictionaryAttribute,
DictionaryDefinition,
@@ -99,6 +100,7 @@
"ChxValidationError",
"ColumnCodec",
"ColumnCodecSpec",
+ "ColumnDefaultKind",
"ColumnDefinition",
"DictionaryAttribute",
"DictionaryDefinition",
diff --git a/chkit_python/src/chkit/core/canonical.py b/chkit_python/src/chkit/core/canonical.py
index 2fe9bf9d..1d3c7d57 100644
--- a/chkit_python/src/chkit/core/canonical.py
+++ b/chkit_python/src/chkit/core/canonical.py
@@ -52,6 +52,7 @@ def _canonicalize_column(column: ColumnDefinition) -> ColumnDefinition:
canon_type = type_value.strip() if isinstance(type_value, str) else type_value
return column.model_copy(
update={
+ "default_kind": None if column.default_kind == "DEFAULT" else column.default_kind,
"name": column.name.strip(),
"renamed_from": column.renamed_from.strip()
if column.renamed_from is not None
diff --git a/chkit_python/src/chkit/core/model.py b/chkit_python/src/chkit/core/model.py
index b93a80bf..18e4ee55 100644
--- a/chkit_python/src/chkit/core/model.py
+++ b/chkit_python/src/chkit/core/model.py
@@ -124,12 +124,16 @@ class RawColumnCodec(_StrictModel):
ColumnType: TypeAlias = PrimitiveColumnType | str
+ColumnDefaultKind: TypeAlias = Literal["DEFAULT", "MATERIALIZED", "ALIAS", "EPHEMERAL"]
+
+
class ColumnDefinition(_StrictModel):
name: str
type: ColumnType
renamed_from: str | None = Field(default=None, alias="renamedFrom")
nullable: bool | None = None
default: str | int | float | bool | None = None
+ default_kind: ColumnDefaultKind | None = Field(default=None, alias="defaultKind")
comment: str | None = None
codec: ColumnCodecSpec | None = None
@@ -624,6 +628,7 @@ class MigrationOperation(_StrictModel):
key: str
risk: RiskLevel
sql: str
+ warning: str | None = None
class ColumnRenameSuggestion(_StrictModel):
@@ -692,6 +697,8 @@ class MigrationPlan(_StrictModel):
"codec_chain_must_end_with_general",
"codec_chain_multiple_general",
"codec_chain_empty",
+ "column_default_kind_invalid",
+ "column_expression_required",
"dictionary_missing_primary_key",
"dictionary_primary_key_missing_attribute",
"dictionary_missing_source",
diff --git a/chkit_python/src/chkit/core/planner.py b/chkit_python/src/chkit/core/planner.py
index 93c69e83..f5e86378 100644
--- a/chkit_python/src/chkit/core/planner.py
+++ b/chkit_python/src/chkit/core/planner.py
@@ -439,10 +439,19 @@ def _diff_tables(
)
)
for column_change in column_diff.changed:
+ old_kind = column_change.old_item.default_kind or "DEFAULT"
+ new_kind = column_change.new_item.default_kind or "DEFAULT"
+ if old_kind != new_kind and {old_kind, new_kind} & {"ALIAS", "EPHEMERAL"}:
+ raise ValueError(
+ f"Cannot automatically change column {new.database}.{new.name}."
+ f"{column_change.name} "
+ f"from {old_kind} to {new_kind}; "
+ "storage-kind conversions involving ALIAS or EPHEMERAL are not supported"
+ )
sql = (
render_alter_remove_codec(new, column_change.name)
if _is_codec_removal(column_change.old_item, column_change.new_item)
- else render_alter_modify_column(new, column_change.new_item)
+ else render_alter_modify_column(new, column_change.new_item, column_change.old_item)
)
ops.append(
MigrationOperation(
@@ -450,6 +459,19 @@ def _diff_tables(
key=f"table:{new.database}.{new.name}:column:{column_change.name}",
risk="caution",
sql=sql,
+ warning=(
+ f"Changing the expression for {new.database}.{new.name}.{column_change.name} "
+ "does not rewrite stored historical values. "
+ "Review a separate MATERIALIZE COLUMN "
+ "migration if a rewrite is required; never reconstruct values "
+ "from discarded EPHEMERAL inputs."
+ if {old_kind, new_kind} & {"DEFAULT", "MATERIALIZED"}
+ and (
+ column_change.old_item.default != column_change.new_item.default
+ or old_kind != new_kind
+ )
+ else None
+ ),
)
)
for column in column_diff.removed:
diff --git a/chkit_python/src/chkit/core/sql.py b/chkit_python/src/chkit/core/sql.py
index 78a8863f..5ca0e02b 100644
--- a/chkit_python/src/chkit/core/sql.py
+++ b/chkit_python/src/chkit/core/sql.py
@@ -58,11 +58,11 @@ def _normalize_projection(projection: ProjectionInput) -> ProjectionDefinition:
return projection
-def _render_default(value: str | int | float | bool) -> str:
+def render_default(value: str | int | float | bool) -> str:
if isinstance(value, str):
if value.startswith("fn:"):
return value[3:]
- escaped = value.replace("'", "''")
+ escaped = value.replace("\\", "\\\\").replace("'", "''")
return f"'{escaped}'"
if isinstance(value, bool):
return "true" if value else "false"
@@ -73,7 +73,9 @@ def _render_column(col: ColumnDefinition) -> str:
type_text = f"Nullable({col.type})" if col.nullable else f"{col.type}"
out = f"`{col.name}` {type_text}"
if col.default is not None:
- out += f" DEFAULT {_render_default(col.default)}"
+ out += f" {col.default_kind or 'DEFAULT'} {render_default(col.default)}"
+ elif col.default_kind == "EPHEMERAL":
+ out += " EPHEMERAL"
if col.comment is not None and len(col.comment) > 0:
escaped = col.comment.replace("'", "''")
out += f" COMMENT '{escaped}'"
@@ -252,7 +254,7 @@ def _render_dictionary_attribute(attr: DictionaryAttribute) -> str:
if attr.expression is not None:
out += f" EXPRESSION {attr.expression}"
elif attr.default is not None:
- out += f" DEFAULT {_render_default(attr.default)}"
+ out += f" DEFAULT {render_default(attr.default)}"
if attr.hierarchical:
out += " HIERARCHICAL"
if attr.bidirectional:
@@ -330,11 +332,19 @@ def render_alter_add_column(definition: TableDefinition, column: ColumnInput) ->
)
-def render_alter_modify_column(definition: TableDefinition, column: ColumnInput) -> str:
+def render_alter_modify_column(
+ definition: TableDefinition, column: ColumnInput, previous: ColumnDefinition | None = None
+) -> str:
normalized = _normalize_column(column)
+ remove = ""
+ if (
+ previous is not None and previous.default is not None
+ and normalized.default is None and normalized.default_kind != "EPHEMERAL"
+ ):
+ remove = f", MODIFY COLUMN `{normalized.name}` REMOVE {previous.default_kind or 'DEFAULT'}"
return (
f"ALTER TABLE {definition.database}.{definition.name} "
- f"MODIFY COLUMN {_render_column(normalized)};"
+ f"MODIFY COLUMN {_render_column(normalized)}{remove};"
)
diff --git a/chkit_python/src/chkit/core/validate.py b/chkit_python/src/chkit/core/validate.py
index d279a614..8ff41e21 100644
--- a/chkit_python/src/chkit/core/validate.py
+++ b/chkit_python/src/chkit/core/validate.py
@@ -180,6 +180,25 @@ def _validate_kafka_table(definition: TableDefinition, issues: list[ValidationIs
)
+def _validate_column_expression(
+ definition: TableDefinition, column: ColumnDefinition, issues: list[ValidationIssue]
+) -> None:
+ if (
+ column.default_kind in {"MATERIALIZED", "ALIAS"} and column.default is None
+ ) or (
+ isinstance(column.default, str)
+ and column.default.startswith("fn:")
+ and not column.default[3:].strip()
+ ):
+ _push(
+ issues,
+ definition,
+ "column_expression_required",
+ f'Column "{column.name}" requires a non-empty expression; '
+ "use fn: for SQL expressions",
+ )
+
+
def _validate_table(definition: TableDefinition, issues: list[ValidationIssue]) -> None:
_validate_kafka_table(definition, issues)
column_seen: set[str] = set()
@@ -196,6 +215,7 @@ def _validate_table(definition: TableDefinition, issues: list[ValidationIssue])
continue
column_seen.add(column.name)
column_set.add(column.name)
+ _validate_column_expression(definition, column, issues)
_validate_column_codec(definition, column, issues)
_validate_indexes(definition, issues)
diff --git a/chkit_python/src/chkit_plugin_backfill/planner.py b/chkit_python/src/chkit_plugin_backfill/planner.py
index 32028c0a..1fc1933b 100644
--- a/chkit_python/src/chkit_plugin_backfill/planner.py
+++ b/chkit_python/src/chkit_plugin_backfill/planner.py
@@ -71,8 +71,6 @@ def _detect_backfill_strategy(
try:
definitions = load_schema_definitions(schema, cwd=config_dir)
mvs = find_mvs_for_target(definitions, database, table)
- if len(mvs) == 0:
- return _BackfillStrategy(mvs=[])
table_def = next(
(
@@ -84,20 +82,63 @@ def _detect_backfill_strategy(
),
None,
)
+ if table_def is not None and any(
+ column.default_kind == "EPHEMERAL" for column in table_def.columns
+ ):
+ raise BackfillConfigError(
+ "Automatic backfill cannot reconstruct EPHEMERAL inputs; "
+ "use an explicit INSERT with an input column mapping."
+ )
+ if len(mvs) == 0:
+ return _BackfillStrategy(mvs=[])
return _BackfillStrategy(
mvs=mvs,
mv_replay_queries=[mv.as_ for mv in mvs],
target_columns=(
- [column.name for column in table_def.columns]
+ [
+ column.name for column in table_def.columns
+ if column.default_kind in {None, "DEFAULT"}
+ ]
if table_def is not None
else None
),
)
+ except BackfillConfigError:
+ raise
except Exception:
# Schema load failed, fall back to direct copy.
return _BackfillStrategy(mvs=[])
+def assert_backfill_target_safe(
+ *, database: str, table: str, query: PlannerQuery,
+ query_settings: QuerySettings | None = None,
+) -> None:
+ """Fail closed when live metadata cannot establish safe input semantics."""
+ def quote(value: str) -> str:
+ return "'" + value.replace("\\", "\\\\").replace("'", "\\'") + "'"
+
+ rows = query(
+ "SELECT name, default_kind FROM system.columns "
+ f"WHERE database = {quote(database)} AND table = {quote(table)} ORDER BY position",
+ query_settings,
+ )
+ if not rows or any(
+ not column.get("name")
+ or column.get("default_kind") not in {"", "DEFAULT", "MATERIALIZED", "ALIAS", "EPHEMERAL"}
+ for column in rows
+ ):
+ raise BackfillConfigError(
+ "Cannot verify live target column kinds; automatic backfill is blocked. "
+ "Check metadata access and use an explicit INSERT if needed."
+ )
+ if any(column["default_kind"] == "EPHEMERAL" for column in rows):
+ raise BackfillConfigError(
+ "Automatic backfill cannot reconstruct EPHEMERAL inputs; "
+ "use an explicit INSERT with an input column mapping."
+ )
+
+
@dataclass(frozen=True)
class BuildBackfillPlanOutput:
plan: BackfillPlanState
@@ -124,13 +165,16 @@ def build_backfill_plan(
# backfill sizes its chunks against the MV *source* (the table its SELECT
# reads), because the injected chunk conditions run against that source —
# not the target, which is legitimately empty when bootstrapping an
- # aggregate. Only the copy path introspects the target itself.
+ # aggregate. Target column safety is checked separately for both paths.
strategy = _detect_backfill_strategy(
schema=config.schema_,
config_dir=Path(config_path).resolve().parent,
database=database,
table=table,
)
+ assert_backfill_target_safe(
+ database=database, table=table, query=clickhouse_query, query_settings=query_settings,
+ )
replay_source = (
resolve_mv_replay_source(strategy.mvs)
if strategy.mv_replay_queries is not None
diff --git a/chkit_python/src/chkit_plugin_backfill/plugin.py b/chkit_python/src/chkit_plugin_backfill/plugin.py
index f8a4b66a..f316853f 100644
--- a/chkit_python/src/chkit_plugin_backfill/plugin.py
+++ b/chkit_python/src/chkit_plugin_backfill/plugin.py
@@ -53,7 +53,7 @@
plan_payload,
status_payload,
)
-from chkit_plugin_backfill.planner import build_backfill_plan
+from chkit_plugin_backfill.planner import assert_backfill_target_safe, build_backfill_plan
from chkit_plugin_backfill.queries import (
cancel_backfill_run,
get_backfill_doctor_report,
@@ -154,6 +154,13 @@ def query_status(
def query(self, statement: str) -> object:
return self._client().query(statement)
+ def query_rows(
+ self, statement: str, settings: QuerySettings | None = None,
+ ) -> list[dict[str, object]]:
+ return self._client().query(
+ statement, dict(settings) if settings is not None else None
+ ).rows
+
def close(self) -> None:
with self._clients_lock:
clients = list(self._clients)
@@ -230,6 +237,8 @@ def _run_backfill( # noqa: PLR0915 — mirrors TS runBackfill
db = _ThreadLocalExecutor(clickhouse)
try:
+ database, table = plan.target.split(".")
+ assert_backfill_target_safe(database=database, table=table, query=db.query_rows)
run_state = BackfillRunState(
plan_id=plan.plan_id,
target=plan.target,
diff --git a/chkit_python/src/chkit_plugin_codegen/type_artifacts.py b/chkit_python/src/chkit_plugin_codegen/type_artifacts.py
index cba1ffe5..49d82e01 100644
--- a/chkit_python/src/chkit_plugin_codegen/type_artifacts.py
+++ b/chkit_python/src/chkit_plugin_codegen/type_artifacts.py
@@ -302,9 +302,29 @@ def _render_table_model(
options: CodegenOptions,
) -> tuple[list[str], list[CodegenFinding], set[str]]:
"""Render the lines for a single table → Pydantic model."""
- return _render_fields_model(
- list(table.columns), class_name, f"{table.database}.{table.name}", options
+ lines, findings, imports = _render_fields_model(
+ [column for column in table.columns if column.default_kind in {None, "DEFAULT"}],
+ class_name, f"{table.database}.{table.name}", options
)
+ if any(column.default_kind not in {None, "DEFAULT"} for column in table.columns):
+ explicit_lines, explicit_findings, explicit_imports = _render_fields_model(
+ [column for column in table.columns if column.default_kind != "EPHEMERAL"],
+ f"{class_name}Explicit", f"{table.database}.{table.name}", options
+ )
+ lines.extend(explicit_lines)
+ findings.extend(explicit_findings)
+ imports.update(explicit_imports)
+ insert_lines, insert_findings, insert_imports = _render_fields_model(
+ [
+ column for column in table.columns
+ if column.default_kind not in {"MATERIALIZED", "ALIAS"}
+ ],
+ f"{class_name}Insert", f"{table.database}.{table.name}", options
+ )
+ lines.extend(insert_lines)
+ findings.extend(insert_findings)
+ imports.update(insert_imports)
+ return lines, findings, imports
def _render_dictionary_model(
diff --git a/chkit_python/tests/test_backfill_planner.py b/chkit_python/tests/test_backfill_planner.py
index 582d5edc..9343d173 100644
--- a/chkit_python/tests/test_backfill_planner.py
+++ b/chkit_python/tests/test_backfill_planner.py
@@ -79,6 +79,8 @@ def _create_mock_query(
def query(sql: str, settings: QuerySettings | None) -> list[dict[str, object]]:
_ = settings
+ if "SELECT name, default_kind" in sql:
+ return [{"name": "id", "default_kind": ""}]
if "SELECT 1 FROM" in sql:
return [{"ok": 1}]
if "FROM system.parts" in sql:
@@ -124,6 +126,8 @@ def _create_source_scoped_mock_query(
def query(sql: str, settings: QuerySettings | None) -> list[dict[str, object]]:
_ = settings
+ if "SELECT name, default_kind" in sql:
+ return [{"name": "id", "default_kind": ""}]
if "SELECT 1 FROM" in sql:
return [{"ok": 1}]
if "FROM system.parts" in sql and f"table = '{source_table}'" in sql:
diff --git a/chkit_python/tests/test_column_expressions.py b/chkit_python/tests/test_column_expressions.py
new file mode 100644
index 00000000..2215b9df
--- /dev/null
+++ b/chkit_python/tests/test_column_expressions.py
@@ -0,0 +1,290 @@
+"""Column expression lifecycle and legacy snapshot compatibility (#206)."""
+
+from __future__ import annotations
+
+from pathlib import Path
+from typing import Any
+
+import pytest
+
+from chkit import ColumnDefinition, table
+from chkit.cli.commands.drift_compare import compare_table_shape
+from chkit.cli.commands.pull import _introspected_table_to_definition
+from chkit.cli.commands.pull_render import render_schema_file
+from chkit.clickhouse.introspect import (
+ IntrospectedTable,
+ SystemColumnRow,
+ normalize_column_from_system_row,
+)
+from chkit.core.model import Snapshot, TableDefinition
+from chkit.core.planner import plan_diff
+from chkit.core.snapshot import create_snapshot
+from chkit.core.sql import to_create_sql
+from chkit.core.validate import validate_definitions
+from chkit_plugin_backfill.planner import _detect_backfill_strategy, assert_backfill_target_safe
+from chkit_plugin_codegen import generate_type_artifacts
+
+
+def definition(**column: Any) -> TableDefinition:
+ return table(
+ database="default",
+ name="events",
+ engine="MergeTree()",
+ primary_key=["ts"],
+ order_by=["ts"],
+ columns=[
+ {"name": "ts", "type": "DateTime"},
+ {"name": "day", "type": "Date", **column},
+ ],
+ )
+
+
+def actual_table(columns: list[ColumnDefinition]) -> IntrospectedTable:
+ return IntrospectedTable(
+ database="default",
+ name="events",
+ columns=columns,
+ settings={},
+ indexes=[],
+ projections=[],
+ engine="MergeTree()",
+ primary_key="ts",
+ order_by="ts",
+ )
+
+
+@pytest.mark.parametrize("kind", ["DEFAULT", "MATERIALIZED", "ALIAS", "EPHEMERAL"])
+def test_roundtrip_kind_and_expression(kind: str) -> None:
+ expected = definition(default_kind=kind, default="fn:toDate(ts)")
+ actual = normalize_column_from_system_row(
+ SystemColumnRow(
+ database="default",
+ table="events",
+ name="day",
+ type="Date",
+ position=2,
+ default_kind=kind,
+ default_expression="toDate(ts)",
+ )
+ )
+ live = actual_table([expected.columns[0], actual])
+ assert compare_table_shape(expected, live) is None
+ pulled = _introspected_table_to_definition(live)
+ assert pulled is not None
+ source = render_schema_file([pulled])
+ namespace: dict[str, Any] = {}
+ exec(source, namespace)
+ reloaded = namespace["definitions"][0]
+ assert plan_diff([expected], [reloaded]).operations == []
+ assert f"`day` Date {kind} toDate(ts)" in to_create_sql(reloaded)
+ assert plan_diff([reloaded], [reloaded]).operations == []
+
+
+def test_bare_ephemeral_synthetic_expression_is_normalized() -> None:
+ expected = definition(default_kind="EPHEMERAL")
+ actual = normalize_column_from_system_row(
+ SystemColumnRow(
+ database="default",
+ table="events",
+ name="day",
+ type="Date",
+ position=2,
+ default_kind="EPHEMERAL",
+ default_expression="defaultValueOfTypeName('Date')",
+ )
+ )
+ assert actual.default is None
+ assert actual.default_kind == "EPHEMERAL"
+ assert compare_table_shape(expected, actual_table([expected.columns[0], actual])) is None
+ assert "`day` Date EPHEMERAL" in to_create_sql(expected)
+
+
+def test_legacy_snapshot_stability() -> None:
+ for value in (None, 0, False, "", "fn:toDate(ts)"):
+ old = definition(default=value)
+ explicit = definition(default_kind="DEFAULT", default=value)
+ payload = create_snapshot([old]).model_dump(mode="json", by_alias=True, exclude_none=True)
+ assert "defaultKind" not in str(payload)
+ legacy = Snapshot.model_validate(payload)
+ assert plan_diff(list(legacy.definitions), [explicit]).operations == []
+ assert "defaultKind" not in str(
+ create_snapshot([explicit]).model_dump(exclude_none=True, by_alias=True)
+ )
+
+
+def test_drift_detects_kind_only_changes() -> None:
+ materialized = definition(default_kind="MATERIALIZED", default="fn:toDate(ts)")
+ normal = definition(default="fn:toDate(ts)")
+ result = compare_table_shape(normal, actual_table(materialized.columns))
+ assert result is not None
+ assert result.changed_columns == ["day"]
+ assert len(plan_diff([normal], [materialized]).operations) == 1
+
+
+@pytest.mark.parametrize("kind", ["DEFAULT", "MATERIALIZED"])
+def test_remove_expression_with_type_change(kind: str) -> None:
+ operations = plan_diff(
+ [definition(default_kind=kind, default="fn:toDate(ts)")], [definition(type="Date32")]
+ ).operations
+ assert len(operations) == 1
+ assert (
+ operations[0].sql
+ == f"ALTER TABLE default.events MODIFY COLUMN `day` Date32, MODIFY COLUMN `day` REMOVE {kind};"
+ )
+
+
+@pytest.mark.parametrize("kind", ["ALIAS", "EPHEMERAL"])
+def test_storage_kind_changes_are_rejected(kind: str) -> None:
+ virtual = definition(default_kind=kind, default="fn:toDate(ts)")
+ with pytest.raises(ValueError, match="storage-kind conversions involving ALIAS or EPHEMERAL are not supported"):
+ plan_diff([definition()], [virtual])
+ with pytest.raises(ValueError, match="storage-kind conversions involving ALIAS or EPHEMERAL are not supported"):
+ plan_diff([virtual], [definition()])
+
+
+def test_validation_and_alias() -> None:
+ col = ColumnDefinition.model_validate(
+ {"name": "day", "type": "Date", "defaultKind": "MATERIALIZED", "default": "fn:toDate(ts)"}
+ )
+ assert col.default_kind == "MATERIALIZED"
+ for kind in ("MATERIALIZED", "ALIAS"):
+ assert any(
+ issue.code == "column_expression_required"
+ for issue in validate_definitions([definition(default_kind=kind)])
+ )
+ assert any(
+ issue.code == "column_expression_required"
+ for issue in validate_definitions([definition(default="fn:")])
+ )
+
+
+def test_codegen_read_and_insert_models() -> None:
+ expected = definition(default_kind="MATERIALIZED", default="fn:toDate(ts)")
+ expected = expected.model_copy(
+ update={
+ "columns": [
+ *expected.columns,
+ ColumnDefinition(name="raw", type="String", defaultKind="EPHEMERAL"),
+ ColumnDefinition(
+ name="label", type="String", default_kind="ALIAS", default="fn:toString(day)"
+ ),
+ ]
+ }
+ )
+ output = generate_type_artifacts(definitions=[expected])
+ namespace: dict[str, Any] = {"__name__": "generated_columns"}
+ exec(output.content, namespace)
+ models = [value for key, value in namespace.items() if key.endswith(("Row", "RowInsert"))]
+ read = next(model for model in models if model.__name__.endswith("Row"))
+ insert = next(model for model in models if model.__name__.endswith("RowInsert"))
+ assert set(read.model_fields) == {"ts"}
+ explicit = namespace[read.__name__ + "Explicit"]
+ assert set(explicit.model_fields) == {"ts", "day", "label"}
+ read.model_validate({"ts": "2026-01-01 00:00:00"})
+ assert set(insert.model_fields) == {"ts", "raw"}
+
+
+def test_backfill_uses_implicit_insert_columns(tmp_path: Path) -> None:
+
+ (tmp_path / "schema.py").write_text("""from chkit import table, materialized_view
+result = table(database="default", name="events", engine="MergeTree()", order_by=["id"], primary_key=["id"], columns=[
+ {"name": "id", "type": "UInt32"},
+ {"name": "size", "type": "UInt64", "default_kind": "MATERIALIZED", "default": "fn:length(raw)"},
+ {"name": "label", "type": "String", "default_kind": "ALIAS", "default": "fn:toString(size)"},
+])
+mv = materialized_view(database="default", name="mv", to={"database": "default", "name": "events"}, as_="SELECT id FROM default.source")
+""")
+ strategy = _detect_backfill_strategy(
+ schema=["schema.py"], config_dir=tmp_path, database="default", table="events"
+ )
+ assert strategy.target_columns == ["id"]
+ path = tmp_path / "ephemeral.py"
+ path.write_text(
+ (tmp_path / "schema.py")
+ .read_text()
+ .replace(
+ '{"name": "id", "type": "UInt32"}',
+ '{"name": "id", "type": "UInt32"}, {"name": "raw", "type": "String", "default_kind": "EPHEMERAL"}',
+ )
+ )
+ with pytest.raises(Exception, match="cannot reconstruct EPHEMERAL inputs"):
+ _detect_backfill_strategy(
+ schema=["ephemeral.py"], config_dir=tmp_path, database="default", table="events"
+ )
+
+
+def test_introspection_preserves_sql_literal_whitespace() -> None:
+ column = normalize_column_from_system_row(
+ SystemColumnRow(
+ database="default",
+ table="events",
+ name="label",
+ type="String",
+ position=1,
+ default_kind="ALIAS",
+ default_expression=" concat('a b', toString(id)) ",
+ )
+ )
+ assert column.default == "concat('a b', toString(id))"
+
+
+@pytest.mark.parametrize(
+ ("expected", "actual", "equal"),
+ [
+ ("fn:concat('a b', toString(ts))", "concat('a b', toString(ts))", False),
+ ("fn:toString(ts+1)", "toString(ts + 1)", True),
+ ("toString(ts)", "toString(ts)", False),
+ ("toString(ts)", "'toString(ts)'", True),
+ (" a b ", "' a b '", True),
+ ("O'Reilly", "'O\\'Reilly'", True),
+ ("\\n", "'\\\\n'", True),
+ ("\\n", "'\\n'", False),
+ ("", None, False),
+ (False, "false", True),
+ (0, "0", True),
+ ("fn:concat('a', `ts`)", "concat('a', ts)", True),
+ ("fn:toString(ts /* comment */ +1)", "toString(ts + 1)", True),
+ ("fn:concat('/* a */', ts)", "concat('/* b */', ts)", False),
+ ],
+)
+def test_expression_comparison_preserves_literals(expected: Any, actual: Any, equal: bool) -> None:
+ result = compare_table_shape(
+ definition(default=expected), actual_table(definition(default=actual).columns)
+ )
+ assert (result is None) == equal
+
+
+def test_stored_expression_warning() -> None:
+ plan = plan_diff([definition(default="fn:toDate(ts)")], [definition(default="fn:today()")])
+ assert "does not rewrite stored historical values" in (plan.operations[0].warning or "")
+ plan = plan_diff(
+ [definition(default="fn:toDate(ts)", default_kind="ALIAS")],
+ [definition(default="fn:today()", default_kind="ALIAS")],
+ )
+ assert plan.operations[0].warning is None
+
+
+@pytest.mark.parametrize(
+ ("rows", "message"),
+ [
+ ([{"name": "raw", "default_kind": "EPHEMERAL"}], "cannot reconstruct EPHEMERAL"),
+ ([], "Cannot verify live target column kinds"),
+ ([{"name": "raw"}], "Cannot verify live target column kinds"),
+ ([{"name": "raw", "default_kind": "FUTURE"}], "Cannot verify live target column kinds"),
+ ],
+)
+def test_backfill_live_safety_gate(rows: list[dict[str, object]], message: str) -> None:
+
+ with pytest.raises(Exception, match=message):
+ assert_backfill_target_safe(
+ database="default", table="events", query=lambda sql, settings: rows
+ )
+
+
+def test_synthetic_ephemeral_default_escaped_type() -> None:
+ col = normalize_column_from_system_row(SystemColumnRow(
+ database="default", table="events", name="raw", type="Enum8('a\\b' = 1)",
+ position=1, default_kind="EPHEMERAL",
+ default_expression="defaultValueOfTypeName('Enum8(\\'a\\\\b\\' = 1)')",
+ ))
+ assert col.default is None
diff --git a/chkit_python/tests/test_column_expressions_e2e.py b/chkit_python/tests/test_column_expressions_e2e.py
new file mode 100644
index 00000000..b025ef12
--- /dev/null
+++ b/chkit_python/tests/test_column_expressions_e2e.py
@@ -0,0 +1,125 @@
+"""Execute expression-column DDL, inserts, pull and ALTER against ClickHouse."""
+
+from __future__ import annotations
+
+from typing import Any
+from uuid import uuid4
+
+import pytest
+
+from chkit import table
+from chkit.cli.commands.drift_compare import compare_table_shape
+from chkit.cli.commands.pull import _introspected_table_to_definition
+from chkit.cli.commands.pull_render import render_schema_file
+from chkit.clickhouse.introspect import (
+ IntrospectedTable,
+ SystemColumnRow,
+ normalize_column_from_system_row,
+)
+from chkit.core.planner import plan_diff
+from chkit.core.sql import to_create_sql
+from chkit_plugin_backfill.planner import assert_backfill_target_safe
+from chkit_plugin_codegen import generate_type_artifacts
+
+
+def test_expression_column_lifecycle(ch_client: Any) -> None:
+ client = ch_client._client
+ name = f"column_expr_py_{uuid4().hex}"
+ database = client.database
+ definition = table(
+ database=database,
+ name=name,
+ engine="MergeTree()",
+ order_by=["id"],
+ primary_key=["id"],
+ columns=[
+ {"name": "id", "type": "UInt32"},
+ {"name": "raw", "type": "String", "default_kind": "EPHEMERAL"},
+ {
+ "name": "size",
+ "type": "UInt64",
+ "default_kind": "MATERIALIZED",
+ "default": "fn:length(raw)",
+ },
+ {
+ "name": "label",
+ "type": "String",
+ "default_kind": "ALIAS",
+ "default": "fn:toString(size)",
+ },
+ ],
+ )
+ target = f"{database}.{name}"
+ try:
+ client.command(to_create_sql(definition))
+ rows = client.query(
+ f"SELECT database, table, name, type, position, default_kind, default_expression FROM system.columns WHERE database='{database}' AND table='{name}' ORDER BY position"
+ ).named_results()
+ columns = [normalize_column_from_system_row(SystemColumnRow(**row)) for row in rows]
+ actual = IntrospectedTable(
+ database=database,
+ name=name,
+ columns=columns,
+ settings={},
+ indexes=[],
+ projections=[],
+ engine="MergeTree()",
+ primary_key="id",
+ order_by="id",
+ )
+ assert compare_table_shape(definition, actual) is None
+ with pytest.raises(Exception, match="cannot reconstruct EPHEMERAL"):
+ assert_backfill_target_safe(database=database, table=name,
+ query=lambda sql, settings: list(client.query(sql).named_results()))
+ pulled = _introspected_table_to_definition(actual)
+ assert pulled is not None
+ namespace: dict[str, Any] = {}
+ exec(render_schema_file([pulled]), namespace)
+ assert plan_diff([definition], namespace["definitions"]).operations == []
+ client.command(f"INSERT INTO {target} (id, raw) VALUES (1, 'abc')")
+ assert client.query(f"SELECT size, label FROM {target}").result_rows == [(3, "3")]
+ models: dict[str, Any] = {"__name__": "generated_live"}
+ exec(generate_type_artifacts(definitions=[definition]).content, models)
+ read = next(value for key, value in models.items() if key.endswith("Row"))
+ explicit = models[read.__name__ + "Explicit"]
+ read.model_validate(next(client.query(f"SELECT * FROM {target}").named_results()))
+ # UInt64 models use the same string encoding as ClickHouse JSON output.
+ explicit.model_validate({"id": 1, "size": "3", "label": "3"})
+ with pytest.raises(Exception, match="MATERIALIZED"):
+ client.command(f"INSERT INTO {target} (id, size) VALUES (2, 10)")
+ with pytest.raises(Exception, match="raw"):
+ client.query(f"SELECT raw FROM {target}")
+ # Changing the expression leaves the already stored value intact.
+ changed = definition.model_copy(
+ update={
+ "columns": [
+ column.model_copy(update={"default": "fn:toUInt64(7)"})
+ if column.name == "size"
+ else column
+ for column in definition.columns
+ ]
+ }
+ )
+ for operation in plan_diff([definition], [changed]).operations:
+ assert "MATERIALIZE COLUMN" not in operation.sql
+ client.command(operation.sql)
+ assert client.query(f"SELECT size FROM {target}").result_rows == [(3,)]
+ client.command(f"INSERT INTO {target} (id, raw) VALUES (2, 'abcd')")
+ assert client.query(f"SELECT size FROM {target} ORDER BY id").result_rows == [(3,), (7,)]
+ plain = changed.model_copy(
+ update={
+ "columns": [
+ column.model_copy(update={"default": None, "default_kind": None})
+ if column.name == "size"
+ else column
+ for column in changed.columns
+ ]
+ }
+ )
+ for operation in plan_diff([changed], [plain]).operations:
+ client.command(operation.sql)
+ assert client.query(
+ f"SELECT default_kind FROM system.columns WHERE database='{database}' AND table='{name}' AND name='size'"
+ ).result_rows == [("",)]
+ finally:
+ client.command(f"DROP TABLE IF EXISTS {target} SYNC")
diff --git a/chkit_python/tests/test_introspect.py b/chkit_python/tests/test_introspect.py
index bb5e3c1e..05a68784 100644
--- a/chkit_python/tests/test_introspect.py
+++ b/chkit_python/tests/test_introspect.py
@@ -93,7 +93,7 @@ def test_normalize_column_picks_up_default_when_kind_is_default() -> None:
assert column.default == "now()"
-def test_normalize_column_ignores_default_when_kind_is_materialized() -> None:
+def test_normalize_column_preserves_materialized_expression() -> None:
row = SystemColumnRow(
database="db",
table="t",
@@ -104,7 +104,8 @@ def test_normalize_column_ignores_default_when_kind_is_materialized() -> None:
default_expression="now()",
)
column = normalize_column_from_system_row(row)
- assert column.default is None
+ assert column.default == "now()"
+ assert column.default_kind == "MATERIALIZED"
def test_normalize_column_preserves_comment_and_codec() -> None:
diff --git a/packages/cli/package.json b/packages/cli/package.json
index e1816efb..a7c7dbfa 100644
--- a/packages/cli/package.json
+++ b/packages/cli/package.json
@@ -48,5 +48,8 @@
},
"optionalDependencies": {
"@chkit/plugin-obsessiondb": "workspace:*"
+ },
+ "devDependencies": {
+ "@clickhouse/client": "^1.11.0"
}
}
diff --git a/packages/cli/src/commands/drift/compare.ts b/packages/cli/src/commands/drift/compare.ts
index 97adfc24..3d4b6c20 100644
--- a/packages/cli/src/commands/drift/compare.ts
+++ b/packages/cli/src/commands/drift/compare.ts
@@ -6,6 +6,8 @@ import {
isIndexProjection,
normalizeProjectionIndex,
normalizeSQLFragment,
+ renderDefault,
+ sqlExpressionFingerprint,
type ColumnDefinition,
type ProjectionDefinition,
type SkipIndexDefinition,
@@ -185,23 +187,14 @@ export function summarizeDriftReasons(input: {
}
function normalizeColumnShape(column: ColumnDefinition): string {
- const normalizeDefaultValue = (value: string): string => {
- const normalized = normalizeSQLFragment(value)
- const quoted = normalized.match(/^'(.*)'$/)
- if (!quoted) return normalized
- return (quoted[1] ?? '').replace(/''/g, "'")
- }
-
- const normalizedDefault = (() => {
- if (column.default === undefined) return ''
- const asString = String(column.default)
- if (asString.startsWith('fn:')) return normalizeDefaultValue(asString.slice(3))
- return normalizeDefaultValue(asString)
- })()
+ const normalizedDefault = column.default === undefined
+ ? ''
+ : sqlExpressionFingerprint(renderDefault(column.default))
const parts = [
`type=${String(column.type).trim()}`,
`nullable=${column.nullable ? '1' : '0'}`,
`default=${normalizedDefault}`,
+ `defaultKind=${column.defaultKind ?? 'DEFAULT'}`,
`comment=${column.comment?.trim() ?? ''}`,
]
return parts.join('|')
@@ -274,7 +267,13 @@ function normalizeEngine(value: string | undefined): string {
export function compareTableShape(expected: TableDefinition, actual: ActualTableShape): TableDriftDetail | null {
const columnDiff = diffByName(
expected.columns,
- actual.columns,
+ // system.columns stores SQL, whereas schema strings are literals unless fn:-prefixed.
+ actual.columns.map((column) => ({
+ ...column,
+ default: typeof column.default === 'string' && !column.default.startsWith('fn:')
+ ? `fn:${column.default}`
+ : column.default,
+ })),
(column: ColumnDefinition) => column.name,
normalizeColumnShape
)
diff --git a/packages/cli/src/commands/generate/command.ts b/packages/cli/src/commands/generate/command.ts
index 5bdbdbeb..d28b62cd 100644
--- a/packages/cli/src/commands/generate/command.ts
+++ b/packages/cli/src/commands/generate/command.ts
@@ -204,7 +204,10 @@ async function cmdGenerate(ctx: import('../../plugins.js').ChxPluginCommandConte
// post-pass, after all plan transforms (renames, plugins, scope filtering).
plan = applyOnClusterToPlan(plan, config.clickhouse?.cluster)
- const dictionaryPasswordWarnings = detectDictionaryPasswordWarnings(plan)
+ const dictionaryPasswordWarnings = [
+ ...detectDictionaryPasswordWarnings(plan),
+ ...plan.operations.flatMap((operation) => operation.warning ? [operation.warning] : []),
+ ]
if (planMode) {
emitGeneratePlanOutput(plan, jsonMode, resolvedScope, dictionaryPasswordWarnings)
diff --git a/packages/cli/src/test/column-expression-safety.test.ts b/packages/cli/src/test/column-expression-safety.test.ts
new file mode 100644
index 00000000..9b1559bb
--- /dev/null
+++ b/packages/cli/src/test/column-expression-safety.test.ts
@@ -0,0 +1,75 @@
+import { expect, test } from 'bun:test'
+import { planDiff, table } from '@chkit/core'
+import { compareTableShape } from '../commands/drift/compare.js'
+
+const definition = (value?: string | number | boolean) =>
+ table({
+ database: 'default',
+ name: 'events',
+ engine: 'MergeTree()',
+ primaryKey: ['id'],
+ orderBy: ['id'],
+ columns: [
+ { name: 'id', type: 'UInt32' },
+ { name: 'value', type: 'String', default: value },
+ ],
+ })
+
+for (const [expected, actual, equal] of [
+ ["fn:concat('a b', toString(id))", "concat('a b', toString(id))", false],
+ ['fn:toString(id+1)', 'toString(id + 1)', true],
+ ['toString(id)', 'toString(id)', false],
+ ['toString(id)', "'toString(id)'", true],
+ [' a b ', "' a b '", true],
+ ["O'Reilly", "'O\\'Reilly'", true],
+ ['\\n', "'\\\\n'", true],
+ ['\\n', "'\\n'", false],
+ ['', undefined, false],
+ [false, 'false', true],
+ [0, '0', true],
+ ["fn:concat('a', `id`)", "concat('a', id)", true],
+ ['fn:toString(id /* comment */ +1)', 'toString(id + 1)', true],
+ ["fn:concat('/* a */', id)", "concat('/* b */', id)", false],
+] as const) {
+ test(`default comparison ${JSON.stringify(expected)} vs ${JSON.stringify(actual)}`, () => {
+ const def = definition(expected)
+ const result = compareTableShape(def, {
+ columns: definition(actual).columns,
+ engine: 'MergeTree()',
+ primaryKey: 'id',
+ orderBy: 'id',
+ settings: {},
+ indexes: [],
+ projections: [],
+ })
+ expect(result === null).toBe(equal)
+ })
+}
+
+test('stored expression changes warn about historical values; computed aliases do not', () => {
+ const before = definition('old')
+ const after = definition('new')
+ expect(planDiff([before], [after]).operations[0]?.warning).toContain(
+ 'does not rewrite stored historical values',
+ )
+ const asAlias = (def: ReturnType) => ({
+ ...def,
+ columns: def.columns.map((col) =>
+ col.name === 'value' ? { ...col, defaultKind: 'ALIAS' as const } : col,
+ ),
+ })
+ expect(
+ planDiff([asAlias(before)], [asAlias(after)]).operations[0]?.warning,
+ ).toBeUndefined()
+})
+
+test('historical value warnings are persisted in migration SQL', async () => {
+ const { generateArtifacts } = await import('@chkit/codegen')
+ const { mkdtemp, readFile, rm } = await import('node:fs/promises')
+ const dir = await mkdtemp('/tmp/chkit-expression-warning-')
+ try {
+ const after = definition('new')
+ const result = await generateArtifacts({ definitions: [after], migrationsDir: `${dir}/migrations`, metaDir: `${dir}/meta`, plan: planDiff([definition('old')], [after]) })
+ expect(await readFile(result.migrationFile ?? '', 'utf8')).toContain('-- Warning: Changing the expression')
+ } finally { await rm(dir, { recursive: true, force: true }) }
+})
diff --git a/packages/cli/src/test/column-expressions.e2e.test.ts b/packages/cli/src/test/column-expressions.e2e.test.ts
new file mode 100644
index 00000000..b03a669a
--- /dev/null
+++ b/packages/cli/src/test/column-expressions.e2e.test.ts
@@ -0,0 +1,277 @@
+import { expect, test } from 'bun:test'
+import { mkdtemp, rm, writeFile } from 'node:fs/promises'
+import { tmpdir } from 'node:os'
+import { join } from 'node:path'
+import { createClient } from '@clickhouse/client'
+import { planDiff, table, toCreateSQL, type TableDefinition } from '@chkit/core'
+import {
+ normalizeColumnFromSystemRow,
+ type SystemColumnRow,
+} from '@chkit/clickhouse'
+import { compareTableShape } from '../commands/drift/compare.js'
+import { renderSchemaFile } from '../../../plugin-pull/src/render-schema.js'
+import {
+ generateTypeArtifacts,
+ generateIngestArtifacts,
+} from '../../../plugin-codegen/src/index.js'
+import { buildBackfillPlan } from '../../../plugin-backfill/src/planner.js'
+import { PlanSchema } from '../../../plugin-backfill/src/options.js'
+import { getRequiredEnv } from './e2e-testkit.js'
+
+test('column expressions survive create, pull, drift, inserts and ALTER on live ClickHouse', async () => {
+ const env = getRequiredEnv()
+ const client = createClient({
+ url: env.clickhouseUrl,
+ username: env.clickhouseUser,
+ password: env.clickhousePassword,
+ database: env.clickhouseDatabase,
+ })
+ const dir = await mkdtemp(join(tmpdir(), 'chkit-expression-pull-'))
+ const name = `column_expr_${Date.now()}_${Math.random().toString(16).slice(2)}`
+ let def = table({
+ database: env.clickhouseDatabase,
+ name,
+ engine: 'MergeTree()',
+ primaryKey: ['id'],
+ orderBy: ['id'],
+ columns: [
+ { name: 'id', type: 'UInt32' },
+ { name: 'ts', type: 'DateTime' },
+ { name: 'raw', type: 'String', defaultKind: 'EPHEMERAL' },
+ {
+ name: 'day',
+ type: 'Date',
+ defaultKind: 'MATERIALIZED',
+ default: 'fn:toDate(ts)',
+ },
+ {
+ name: 'label',
+ type: 'String',
+ defaultKind: 'ALIAS',
+ default: 'fn:toString(day)',
+ },
+ { name: 'size', type: 'UInt64', default: 'fn:length(raw)' },
+ ],
+ })
+ const query = async (sql: string) =>
+ (
+ await client.query({
+ query: sql,
+ format: 'JSONEachRow',
+ clickhouse_settings: { output_format_json_quote_64bit_integers: 1 },
+ })
+ ).json()
+ const columns = async () =>
+ (
+ await query(
+ `SELECT database, table, name, type, position, default_kind, default_expression FROM system.columns WHERE database='${def.database}' AND table='${name}' ORDER BY position`,
+ )
+ ).map(normalizeColumnFromSystemRow)
+ const actual = async () => ({
+ columns: await columns(),
+ settings: {},
+ indexes: [],
+ projections: [],
+ engine: 'MergeTree()',
+ primaryKey: 'id',
+ orderBy: 'id',
+ })
+ const migrate = async (next: TableDefinition) => {
+ const plan = planDiff([def], [next])
+ for (const operation of plan.operations) {
+ expect(operation.sql).not.toContain('MATERIALIZE COLUMN')
+ await client.command({ query: operation.sql })
+ }
+ def = next
+ expect(compareTableShape(def, await actual())).toBeNull()
+ }
+ try {
+ await client.command({ query: toCreateSQL(def) })
+ expect(compareTableShape(def, await actual())).toBeNull()
+ await expect(buildBackfillPlan({
+ opts: PlanSchema.parse({ target: `${def.database}.${name}` }),
+ configPath: join(dir, 'config.ts'),
+ config: { metaDir: join(dir, 'meta'), schema: [join(dir, 'missing.ts')] },
+ clickhouseQuery: query,
+ })).rejects.toThrow('cannot reconstruct EPHEMERAL inputs')
+ const pulled = {
+ ...def,
+ columns: (await columns()).map((column) => ({
+ ...column,
+ default:
+ typeof column.default === 'string'
+ ? `fn:${column.default}`
+ : column.default,
+ })),
+ }
+ const source = renderSchemaFile([pulled]).replace(
+ "'@chkit/core'",
+ JSON.stringify(
+ new URL('../../../core/src/index.ts', import.meta.url).href,
+ ),
+ )
+ const path = join(dir, 'pulled.ts')
+ await writeFile(path, source)
+ const reloaded = (await import(path)).default
+ expect(planDiff(reloaded, [pulled]).operations).toEqual([])
+ expect(toCreateSQL(reloaded[0])).toContain('MATERIALIZED toDate(ts)')
+ expect(toCreateSQL(reloaded[0])).toContain('`raw` String EPHEMERAL')
+
+ await client.command({
+ query: `INSERT INTO ${def.database}.${name} (id, ts, raw) VALUES (1, '2026-01-01 12:00:00', 'abc')`,
+ })
+ expect(
+ await query(
+ `SELECT day, label, toString(size) AS size FROM ${def.database}.${name}`,
+ ),
+ ).toEqual([{ day: '2026-01-01', label: '2026-01-01', size: '3' }])
+ const generatedPath = join(dir, 'types.ts')
+ await writeFile(
+ generatedPath,
+ generateTypeArtifacts({
+ definitions: [def],
+ options: { emitZod: true },
+ }).content.replace("'zod'", JSON.stringify(import.meta.resolve('zod'))),
+ )
+ const models = await import(generatedPath)
+ const readSchema =
+ models[Object.keys(models).find((key) => key.endsWith('RowSchema')) ?? '']
+ const explicitSchema =
+ models[
+ Object.keys(models).find((key) => key.endsWith('RowExplicitSchema')) ??
+ ''
+ ]
+ const star = (await query(`SELECT * FROM ${def.database}.${name}`))[0]
+ const full = (
+ await query(
+ `SELECT id, ts, day, label, size FROM ${def.database}.${name}`,
+ )
+ )[0]
+ expect(readSchema.parse(star)).toBeDefined()
+ expect(explicitSchema.safeParse(star).success).toBe(false)
+ expect(explicitSchema.safeParse(full).success).toBe(true)
+ await expect(
+ client.command({
+ query: `INSERT INTO ${def.database}.${name} (id, ts, day) VALUES (2, '2026-01-01 12:00:00', '2000-01-01')`,
+ }),
+ ).rejects.toThrow()
+ await expect(
+ query(`SELECT raw FROM ${def.database}.${name}`),
+ ).rejects.toThrow()
+
+ await migrate({
+ ...def,
+ columns: def.columns.map((column) =>
+ column.name === 'day'
+ ? { ...column, default: 'fn:addDays(toDate(ts), 1)' }
+ : column,
+ ),
+ })
+ expect(
+ await query(`SELECT day FROM ${def.database}.${name} WHERE id=1`),
+ ).toEqual([{ day: '2026-01-01' }])
+ await client.command({
+ query: `INSERT INTO ${def.database}.${name} (id, ts, raw) VALUES (2, '2026-01-01 12:00:00', 'abcd')`,
+ })
+ expect(
+ await query(`SELECT day FROM ${def.database}.${name} WHERE id=2`),
+ ).toEqual([{ day: '2026-01-02' }])
+ await migrate({
+ ...def,
+ columns: [
+ ...def.columns,
+ {
+ name: 'copy',
+ type: 'UInt32',
+ defaultKind: 'MATERIALIZED',
+ default: 'fn:id',
+ },
+ ],
+ })
+ await migrate({
+ ...def,
+ columns: def.columns.map((column) =>
+ column.name === 'copy' ? { ...column, defaultKind: 'DEFAULT' } : column,
+ ),
+ })
+ await migrate({
+ ...def,
+ columns: def.columns.map((column) =>
+ column.name === 'copy' || column.name === 'day'
+ ? { name: column.name, type: column.type }
+ : column,
+ ),
+ })
+ await migrate({
+ ...def,
+ columns: def.columns.map((column) =>
+ column.name === 'raw' ? { ...column, default: 'seed' } : column,
+ ),
+ })
+ await migrate({
+ ...def,
+ columns: def.columns.map((column) =>
+ column.name === 'raw' ? { ...column, default: undefined } : column,
+ ),
+ })
+ } finally {
+ await client.command({
+ query: `DROP TABLE IF EXISTS ${def.database}.${name} SYNC`,
+ })
+ await client.close()
+ await rm(dir, { recursive: true, force: true })
+ }
+}, 30_000)
+
+test('generated ingest helpers use insert shapes, while rows exclude ephemeral inputs', () => {
+ const def = table({
+ database: 'default',
+ name: 'events',
+ engine: 'MergeTree()',
+ primaryKey: ['id'],
+ orderBy: ['id'],
+ columns: [
+ { name: 'id', type: 'UInt32' },
+ { name: 'raw', type: 'String', defaultKind: 'EPHEMERAL' },
+ {
+ name: 'computed',
+ type: 'UInt32',
+ defaultKind: 'MATERIALIZED',
+ default: 'fn:length(raw)',
+ },
+ {
+ name: 'label',
+ type: 'String',
+ defaultKind: 'ALIAS',
+ default: 'fn:toString(id)',
+ },
+ ],
+ })
+ const types = generateTypeArtifacts({
+ definitions: [def],
+ options: { emitZod: true },
+ }).content
+ const read = types.split('export type DefaultEventsRow = {')[1]?.split('}')[0]
+ const insert = types
+ .split('export type DefaultEventsRowInsert = {')[1]
+ ?.split('}')[0]
+ expect(read).not.toContain('computed:')
+ expect(read).not.toContain('label:')
+ const explicit = types
+ .split('export type DefaultEventsRowExplicit = {')[1]
+ ?.split('}')[0]
+ expect(explicit).toContain('computed: number')
+ expect(explicit).toContain('label: string')
+ expect(explicit).not.toContain('raw:')
+ expect(read).not.toContain('raw:')
+ expect(insert).toContain('raw: string')
+ expect(insert).not.toContain('computed:')
+ expect(insert).not.toContain('label:')
+ const ingest = generateIngestArtifacts({
+ definitions: [def],
+ options: { emitZod: true },
+ }).content
+ expect(ingest).toContain('rows: DefaultEventsRowInsert[]')
+ expect(ingest).toContain('DefaultEventsRowInsertSchema.parse(row)')
+ expect(ingest).toContain('function ingestDefaultEvents(')
+})
diff --git a/packages/clickhouse/src/column-expressions.test.ts b/packages/clickhouse/src/column-expressions.test.ts
new file mode 100644
index 00000000..9a6a5445
--- /dev/null
+++ b/packages/clickhouse/src/column-expressions.test.ts
@@ -0,0 +1,58 @@
+import { expect, test } from 'bun:test'
+import { normalizeColumnFromSystemRow } from './index.js'
+
+for (const kind of ['DEFAULT', 'MATERIALIZED', 'ALIAS', 'EPHEMERAL'] as const) {
+ test(`introspection preserves ${kind} expressions including literal whitespace`, () => {
+ const column = normalizeColumnFromSystemRow({
+ database: 'default',
+ table: 'events',
+ name: 'label',
+ type: 'String',
+ position: 1,
+ default_kind: kind,
+ default_expression: " concat('a b', toString(id)) ",
+ })
+ expect(column.default).toBe("concat('a b', toString(id))")
+ expect(column.defaultKind).toBe(kind === 'DEFAULT' ? undefined : kind)
+ })
+}
+
+test('expressionless nullable EPHEMERAL columns preserve kind without synthetic defaults', () => {
+ const column = normalizeColumnFromSystemRow({
+ database: 'default',
+ table: 'events',
+ name: 'raw',
+ type: 'Nullable(String)',
+ position: 1,
+ default_kind: 'EPHEMERAL',
+ default_expression: "defaultValueOfTypeName('Nullable(String)')",
+ })
+ expect(column).toMatchObject({
+ name: 'raw',
+ type: 'String',
+ nullable: true,
+ defaultKind: 'EPHEMERAL',
+ })
+ expect(column.default).toBeUndefined()
+})
+
+test('unknown expression kinds fail explicitly instead of losing metadata', () => {
+ expect(() =>
+ normalizeColumnFromSystemRow({
+ database: 'default',
+ table: 'events',
+ name: 'x',
+ type: 'String',
+ position: 1,
+ default_kind: 'FUTURE_KIND',
+ default_expression: 'someExpression()',
+ }),
+ ).toThrow('Unsupported column default kind')
+})
+
+test('synthetic EPHEMERAL defaults compare SQL literals with quotes and backslashes', () => {
+ const type = "Enum8('a\\b' = 1)"
+ const expression = "defaultValueOfTypeName('Enum8(\\'a\\\\b\\' = 1)')"
+ const column = normalizeColumnFromSystemRow({ database: 'default', table: 'events', name: 'raw', type, position: 1, default_kind: 'EPHEMERAL', default_expression: expression })
+ expect(column.default).toBeUndefined()
+})
diff --git a/packages/clickhouse/src/index.ts b/packages/clickhouse/src/index.ts
index 29c14dc5..5f1eff2c 100644
--- a/packages/clickhouse/src/index.ts
+++ b/packages/clickhouse/src/index.ts
@@ -3,6 +3,8 @@ import {
type ChxConfig,
type ColumnDefinition,
normalizeSQLFragment,
+ renderDefault,
+ sqlExpressionFingerprint,
type ProjectionDefinition,
parseCodec,
type SkipIndexDefinition,
@@ -192,8 +194,21 @@ export function normalizeColumnFromSystemRow(
const type = nullableMatch?.[1] ? nullableMatch[1] : row.type
const nullable = Boolean(nullableMatch?.[1])
let defaultValue: ColumnDefinition['default'] | undefined
- if (row.default_expression && row.default_kind === 'DEFAULT') {
- defaultValue = normalizeSQLFragment(row.default_expression)
+ const defaultKind = row.default_kind
+ if (defaultKind && !['DEFAULT', 'MATERIALIZED', 'ALIAS', 'EPHEMERAL'].includes(defaultKind)) {
+ throw new Error(`Unsupported column default kind: ${defaultKind}`)
+ }
+ if (row.default_expression && defaultKind) {
+ // Preserve whitespace inside SQL string literals when pulling expressions.
+ defaultValue = row.default_expression.trim()
+ }
+ // ClickHouse synthesizes this expression for a bare EPHEMERAL column.
+ if (
+ defaultKind === 'EPHEMERAL' && defaultValue !== undefined &&
+ sqlExpressionFingerprint(String(defaultValue)) ===
+ sqlExpressionFingerprint(`defaultValueOfTypeName(${renderDefault(row.type)})`)
+ ) {
+ defaultValue = undefined
}
const codecSteps = parseCodec(row.compression_codec)
return {
@@ -201,6 +216,9 @@ export function normalizeColumnFromSystemRow(
type,
nullable: nullable || undefined,
default: defaultValue,
+ defaultKind: defaultKind && defaultKind !== 'DEFAULT'
+ ? defaultKind as ColumnDefinition['defaultKind']
+ : undefined,
comment: row.comment?.trim() || undefined,
codec: codecSteps,
}
diff --git a/packages/codegen/src/index.ts b/packages/codegen/src/index.ts
index 1a46c106..6c8cfa1a 100644
--- a/packages/codegen/src/index.ts
+++ b/packages/codegen/src/index.ts
@@ -126,7 +126,11 @@ function buildMigrationContent(input: {
)
const body = input.plan.operations
- .map((op) => [`-- operation: ${op.type} key=${op.key} risk=${op.risk}`, op.sql].join('\n'))
+ .map((op) => [
+ `-- operation: ${op.type} key=${op.key} risk=${op.risk}`,
+ ...(op.warning ? [`-- Warning: ${op.warning.replace(/[\r\n]/g, ' ')}`] : []),
+ op.sql,
+ ].join('\n'))
.join('\n\n')
const withHints = [...header, ...renameHints]
diff --git a/packages/core/src/canonical.ts b/packages/core/src/canonical.ts
index 46313b9d..84ceb2aa 100644
--- a/packages/core/src/canonical.ts
+++ b/packages/core/src/canonical.ts
@@ -28,8 +28,10 @@ function sortKind(kind: SchemaDefinition['kind']): number {
}
function canonicalizeColumn(column: ColumnDefinition): ColumnDefinition {
+ const { defaultKind, ...rest } = column
return {
- ...column,
+ ...rest,
+ defaultKind: defaultKind === 'DEFAULT' ? undefined : defaultKind,
name: column.name.trim(),
renamedFrom: column.renamedFrom?.trim(),
type: typeof column.type === 'string' ? column.type.trim() : column.type,
diff --git a/packages/core/src/column-expressions.test.ts b/packages/core/src/column-expressions.test.ts
new file mode 100644
index 00000000..664b4bb0
--- /dev/null
+++ b/packages/core/src/column-expressions.test.ts
@@ -0,0 +1,127 @@
+import { describe, expect, test } from 'bun:test'
+import {
+ createSnapshot,
+ planDiff,
+ table,
+ toCreateSQL,
+ validateDefinitions,
+ type ColumnDefinition,
+} from './index.js'
+
+const definition = (column: Partial = {}) =>
+ table({
+ database: 'default',
+ name: 'events',
+ engine: 'MergeTree()',
+ primaryKey: ['ts'],
+ orderBy: ['ts'],
+ columns: [
+ { name: 'ts', type: 'DateTime' },
+ { name: 'day', type: 'Date', ...column },
+ ],
+ })
+
+describe('column expressions', () => {
+ for (const defaultKind of [
+ 'DEFAULT',
+ 'MATERIALIZED',
+ 'ALIAS',
+ 'EPHEMERAL',
+ ] as const) {
+ test(`renders ${defaultKind} expressions and retains literal quoting`, () => {
+ const def = definition({ defaultKind, default: 'fn:toDate(ts)' })
+ expect(toCreateSQL(def)).toContain(
+ `\`day\` Date ${defaultKind} toDate(ts)`,
+ )
+ expect(
+ toCreateSQL(
+ definition({ defaultKind, type: 'String', default: "it's literal" }),
+ ),
+ ).toContain(`${defaultKind} 'it''s literal'`)
+ expect(planDiff([def], [def]).operations).toEqual([])
+ })
+ }
+
+ test('supports EPHEMERAL without an expression', () => {
+ expect(toCreateSQL(definition({ defaultKind: 'EPHEMERAL' }))).toContain(
+ '`day` Date EPHEMERAL',
+ )
+ })
+
+ test('keeps legacy and explicit DEFAULT snapshots equivalent', () => {
+ for (const value of [undefined, 0, false, '', 'fn:toDate(ts)']) {
+ const old = definition({ default: value })
+ const explicit = definition({ defaultKind: 'DEFAULT', default: value })
+ const legacy = JSON.parse(JSON.stringify(createSnapshot([old])))
+ expect(planDiff(legacy.definitions, [explicit]).operations).toEqual([])
+ expect(JSON.stringify(createSnapshot([explicit]))).not.toContain(
+ 'defaultKind',
+ )
+ }
+ })
+
+ test('detects kind-only and expression changes', () => {
+ const old = definition({ default: 'fn:toDate(ts)' })
+ const materialized = definition({
+ defaultKind: 'MATERIALIZED',
+ default: 'fn:toDate(ts)',
+ })
+ const plan = planDiff([old], [materialized])
+ expect(plan.operations).toHaveLength(1)
+ expect(plan.operations[0]?.sql).toContain('MATERIALIZED toDate(ts)')
+ expect(plan.operations[0]?.risk).toBe('caution')
+ expect(
+ planDiff(
+ [materialized],
+ [definition({ defaultKind: 'MATERIALIZED', default: 'fn:today()' })],
+ ).operations[0]?.sql,
+ ).toContain('MATERIALIZED today()')
+ })
+
+ for (const defaultKind of ['DEFAULT', 'MATERIALIZED'] as const) {
+ test(`explicitly removes ${defaultKind}, including simultaneous type changes`, () => {
+ const plan = planDiff(
+ [definition({ defaultKind, default: 'fn:toDate(ts)' })],
+ [definition({ type: 'Date32' })],
+ )
+ expect(plan.operations[0]?.sql).toBe(
+ `ALTER TABLE default.events MODIFY COLUMN \`day\` Date32, MODIFY COLUMN \`day\` REMOVE ${defaultKind};`,
+ )
+ expect(plan.operations).toHaveLength(1)
+ })
+ }
+
+ test('rejects automatic storage-kind changes in both directions', () => {
+ for (const defaultKind of ['ALIAS', 'EPHEMERAL'] as const) {
+ const virtual = definition({ defaultKind, default: 'fn:toDate(ts)' })
+ expect(() => planDiff([definition()], [virtual])).toThrow(
+ 'storage-kind conversions involving ALIAS or EPHEMERAL are not supported',
+ )
+ expect(() => planDiff([virtual], [definition()])).toThrow(
+ 'storage-kind conversions involving ALIAS or EPHEMERAL are not supported',
+ )
+ }
+ })
+
+ test('validates kind and missing/empty expressions', () => {
+ for (const defaultKind of ['MATERIALIZED', 'ALIAS'] as const) {
+ expect(
+ validateDefinitions([definition({ defaultKind })]).map(
+ (issue) => issue.code,
+ ),
+ ).toContain('column_expression_required')
+ }
+ expect(
+ validateDefinitions([definition({ default: 'fn: ' })]).map(
+ (issue) => issue.code,
+ ),
+ ).toContain('column_expression_required')
+ expect(
+ validateDefinitions([
+ definition({
+ defaultKind: 'invalid' as ColumnDefinition['defaultKind'],
+ }),
+ ]).map((issue) => issue.code),
+ ).toContain('column_default_kind_invalid')
+ })
+})
diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts
index 2af4c46e..77cd4c05 100644
--- a/packages/core/src/index.ts
+++ b/packages/core/src/index.ts
@@ -20,8 +20,8 @@ export {
} from './text-index.js'
export { splitTopLevelComma } from './key-clause.js'
export { isIndexProjection, normalizeProjectionIndex } from './projection.js'
-export { normalizeEngine, normalizeSQLFragment } from './sql-normalizer.js'
-export { renderDictionarySQL, toCreateSQL } from './sql.js'
+export { normalizeEngine, normalizeSQLFragment, sqlExpressionFingerprint } from './sql-normalizer.js'
+export { renderDefault, renderDictionarySQL, toCreateSQL } from './sql.js'
export { applyOnClusterToPlan, onClusterClause } from './on-cluster.js'
export {
canonicalizeCodec,
diff --git a/packages/core/src/model-types.ts b/packages/core/src/model-types.ts
index c0e9f438..19a1a051 100644
--- a/packages/core/src/model-types.ts
+++ b/packages/core/src/model-types.ts
@@ -52,12 +52,16 @@ export type ColumnCodec = GeneralColumnCodec | PreprocessingColumnCodec | RawCol
/** Single codec or a chain (preprocessors then exactly one general codec). */
export type ColumnCodecSpec = ColumnCodec | ColumnCodec[]
+export type ColumnDefaultKind = 'DEFAULT' | 'MATERIALIZED' | 'ALIAS' | 'EPHEMERAL'
+
export interface ColumnDefinition {
name: string
type: PrimitiveColumnType | string
renamedFrom?: string
nullable?: boolean
+ /** Strings are literals; prefix SQL expressions with `fn:`. */
default?: string | number | boolean
+ defaultKind?: ColumnDefaultKind
comment?: string
codec?: ColumnCodecSpec
}
@@ -382,6 +386,7 @@ export interface MigrationOperation {
key: string
risk: RiskLevel
sql: string
+ warning?: string
}
export interface ColumnRenameSuggestion {
@@ -426,6 +431,8 @@ export type ValidationIssueCode =
| 'codec_chain_must_end_with_general'
| 'codec_chain_multiple_general'
| 'codec_chain_empty'
+ | 'column_default_kind_invalid'
+ | 'column_expression_required'
| 'dictionary_missing_primary_key'
| 'dictionary_primary_key_missing_attribute'
| 'dictionary_missing_source'
diff --git a/packages/core/src/planner.ts b/packages/core/src/planner.ts
index 5d180f2c..4c47ed5a 100644
--- a/packages/core/src/planner.ts
+++ b/packages/core/src/planner.ts
@@ -409,14 +409,22 @@ function diffTables(oldDef: TableDefinition, newDef: TableDefinition): TableDiff
})
}
for (const { name, oldItem, newItem } of columnDiff.changed) {
+ const oldKind = oldItem.defaultKind ?? 'DEFAULT'
+ const newKind = newItem.defaultKind ?? 'DEFAULT'
+ if (oldKind !== newKind && [oldKind, newKind].some((kind) => kind === 'ALIAS' || kind === 'EPHEMERAL')) {
+ throw new Error(`Cannot automatically change column ${newDef.database}.${newDef.name}.${name} from ${oldKind} to ${newKind}; storage-kind conversions involving ALIAS or EPHEMERAL are not supported`)
+ }
const sql = isCodecRemoval(oldItem, newItem)
? renderAlterRemoveCodec(newDef, name)
- : renderAlterModifyColumn(newDef, newItem)
+ : renderAlterModifyColumn(newDef, newItem, oldItem)
ops.push( {
type: 'alter_table_modify_column',
key: `table:${newDef.database}.${newDef.name}:column:${name}`,
risk: 'caution',
sql,
+ ...((oldKind === 'DEFAULT' || oldKind === 'MATERIALIZED' || newKind === 'DEFAULT' || newKind === 'MATERIALIZED') && (oldItem.default !== newItem.default || oldKind !== newKind)
+ ? { warning: `Changing the expression for ${newDef.database}.${newDef.name}.${name} does not rewrite stored historical values. Review a separate MATERIALIZE COLUMN migration if a rewrite is required; never reconstruct values from discarded EPHEMERAL inputs.` }
+ : {}),
})
}
for (const column of columnDiff.removed) {
diff --git a/packages/core/src/sql-normalizer.ts b/packages/core/src/sql-normalizer.ts
index 6a107164..83b29445 100644
--- a/packages/core/src/sql-normalizer.ts
+++ b/packages/core/src/sql-normalizer.ts
@@ -1,4 +1,10 @@
import { isKafkaEngine, normalizeKafkaEngine } from './kafka.js'
+import { textExpressionFingerprint, textSQLFingerprint } from './text-index-sql.js'
+
+/** Compare expression tokens while preserving quoted values and identifier case. */
+export function sqlExpressionFingerprint(value: string): string {
+ return textSQLFingerprint(textExpressionFingerprint(value))
+}
export function normalizeSQLFragment(value: string): string {
return value.replace(/\s+/g, ' ').trim()
diff --git a/packages/core/src/sql.ts b/packages/core/src/sql.ts
index 8ac3d364..02bc7600 100644
--- a/packages/core/src/sql.ts
+++ b/packages/core/src/sql.ts
@@ -17,17 +17,18 @@ import { renderProjectionBody } from './projection.js'
import { TEXT_INDEX_GRANULARITY, renderTextIndexType } from './text-index.js'
import { assertValidDefinitions } from './validate.js'
-function renderDefault(value: string | number | boolean): string {
+export function renderDefault(value: string | number | boolean): string {
if (typeof value === 'string') {
if (value.startsWith('fn:')) return value.slice(3)
- return `'${value.replace(/'/g, "''")}'`
+ return `'${value.replace(/\\/g, "\\\\").replace(/'/g, "''")}'`
}
return String(value)
}
function renderColumn(col: ColumnDefinition): string {
let out = `\`${col.name}\` ${col.nullable ? `Nullable(${col.type})` : col.type}`
- if (col.default !== undefined) out += ` DEFAULT ${renderDefault(col.default)}`
+ if (col.default !== undefined) out += ` ${col.defaultKind ?? 'DEFAULT'} ${renderDefault(col.default)}`
+ else if (col.defaultKind === 'EPHEMERAL') out += ' EPHEMERAL'
if (col.comment) out += ` COMMENT '${col.comment.replace(/'/g, "''")}'`
if (col.codec) out += ` ${renderCodec(col.codec)}`
return out
@@ -211,8 +212,15 @@ export function renderAlterAddColumn(def: TableDefinition, column: ColumnDefinit
return `ALTER TABLE ${def.database}.${def.name} ADD COLUMN IF NOT EXISTS ${renderColumn(column)};`
}
-export function renderAlterModifyColumn(def: TableDefinition, column: ColumnDefinition): string {
- return `ALTER TABLE ${def.database}.${def.name} MODIFY COLUMN ${renderColumn(column)};`
+export function renderAlterModifyColumn(
+ def: TableDefinition,
+ column: ColumnDefinition,
+ previous?: ColumnDefinition
+): string {
+ const remove = previous?.default !== undefined && column.default === undefined && column.defaultKind !== 'EPHEMERAL'
+ ? `, MODIFY COLUMN \`${column.name}\` REMOVE ${previous.defaultKind ?? 'DEFAULT'}`
+ : ''
+ return `ALTER TABLE ${def.database}.${def.name} MODIFY COLUMN ${renderColumn(column)}${remove};`
}
export function renderAlterDropColumn(def: TableDefinition, columnName: string): string {
diff --git a/packages/core/src/validate.ts b/packages/core/src/validate.ts
index ecc2b68b..db0dc8ff 100644
--- a/packages/core/src/validate.ts
+++ b/packages/core/src/validate.ts
@@ -123,6 +123,21 @@ function validateTableDefinition(def: TableDefinition, issues: ValidationIssue[]
}
columnSeen.add(column.name)
columnSet.add(column.name)
+ const kind = column.defaultKind
+ if (kind !== undefined && !['DEFAULT', 'MATERIALIZED', 'ALIAS', 'EPHEMERAL'].includes(kind)) {
+ pushValidationIssue(
+ issues, def, 'column_default_kind_invalid', `Invalid defaultKind on column "${column.name}"`
+ )
+ }
+ const missingExpression = (kind === 'MATERIALIZED' || kind === 'ALIAS') && column.default === undefined
+ const emptyExpression = typeof column.default === 'string'
+ && column.default.startsWith('fn:') && !column.default.slice(3).trim()
+ if (missingExpression || emptyExpression) {
+ pushValidationIssue(
+ issues, def, 'column_expression_required',
+ `Column "${column.name}" requires a non-empty expression; use fn: for SQL expressions`
+ )
+ }
validateColumnCodec(def, column, issues)
}
diff --git a/packages/plugin-backfill/src/planner.test.ts b/packages/plugin-backfill/src/planner.test.ts
index a9717f5d..83fe6b98 100644
--- a/packages/plugin-backfill/src/planner.test.ts
+++ b/packages/plugin-backfill/src/planner.test.ts
@@ -37,6 +37,7 @@ function createMockQuery(opts: {
const columnRows = opts.columnRows ?? [{ name: 'event_time', type: 'DateTime' }]
return async (sql: string) => {
+ if (sql.includes('SELECT name, default_kind')) return [{ name: 'id', default_kind: '' }] as T[]
if (sql.includes('SELECT 1 FROM')) return [{ ok: 1 }] as T[]
if (sql.includes('FROM system.parts')) return partitions as T[]
if (sql.includes('FROM system.tables')) return [{ sorting_key: sortingKey }] as T[]
@@ -79,6 +80,7 @@ function createSourceScopedMockQuery(opts: {
const table = opts.sourceTable
return async (sql: string) => {
+ if (sql.includes('SELECT name, default_kind')) return [{ name: 'id', default_kind: '' }] as T[]
if (sql.includes('SELECT 1 FROM')) return [{ ok: 1 }] as T[]
if (sql.includes('FROM system.parts')) {
return (sql.includes(`table = '${table}'`) ? partitions : []) as T[]
@@ -548,3 +550,52 @@ export const api_mv = {
}
})
})
+
+test('MV replay omits computed columns and rejects unrecoverable ephemeral inputs', async () => {
+ const dir = await mkdtemp(join(tmpdir(), 'chkit-backfill-expressions-'))
+ try {
+ await writeFile(join(dir, 'schema.ts'), `
+ export const target = { kind: 'table', database: 'app', name: 'events_agg', engine: 'MergeTree()',
+ primaryKey: ['event_time'], orderBy: ['event_time'], columns: [
+ { name: 'event_time', type: 'DateTime' }, { name: 'count', type: 'UInt64' },
+ { name: 'day', type: 'Date', defaultKind: 'MATERIALIZED', default: 'fn:toDate(event_time)' },
+ { name: 'label', type: 'String', defaultKind: 'ALIAS', default: 'fn:toString(count)' }
+ ] }
+ export const mv = { kind: 'materialized_view', database: 'app', name: 'events_mv',
+ to: { database: 'app', name: 'events_agg' }, as: 'SELECT event_time, count() AS count FROM app.events GROUP BY event_time' }
+ `)
+ const output = await buildBackfillPlan({ opts: PlanSchema.parse({ target: 'app.events_agg' }),
+ configPath: join(dir, 'clickhouse.config.ts'), config: resolveConfig({ schema: './schema.ts', metaDir: './chkit/meta' }),
+ clickhouseQuery: createMockQuery(),
+ })
+ expect(output.plan.execution.targetColumns).toEqual(['event_time', 'count'])
+ const ephemeralPath = join(dir, 'ephemeral.ts')
+ const source = await readFile(join(dir, 'schema.ts'), 'utf8')
+ await writeFile(ephemeralPath, source.replace("{ name: 'count', type: 'UInt64' }", "{ name: 'raw', type: 'String', defaultKind: 'EPHEMERAL' }"))
+ await expect(buildBackfillPlan({ opts: PlanSchema.parse({ target: 'app.events_agg' }),
+ configPath: join(dir, 'clickhouse.config.ts'), config: resolveConfig({ schema: './ephemeral.ts', metaDir: './chkit/meta' }),
+ clickhouseQuery: createMockQuery(),
+ })).rejects.toThrow('cannot reconstruct EPHEMERAL inputs')
+ } finally {
+ await rm(dir, { recursive: true, force: true })
+ }
+})
+
+for (const [rows, message] of [
+ [[{ name: 'raw', default_kind: 'EPHEMERAL' }], 'cannot reconstruct EPHEMERAL'],
+ [[], 'Cannot verify live target column kinds'],
+ [[{ name: 'raw' }], 'Cannot verify live target column kinds'],
+ [[{ name: 'raw', default_kind: 'FUTURE' }], 'Cannot verify live target column kinds'],
+] as const) {
+ test(`live target metadata blocks unsafe schema fallback: ${JSON.stringify(rows)}`, async () => {
+ const dir = await mkdtemp(join(tmpdir(), 'chkit-backfill-unsafe-'))
+ try {
+ await expect(buildBackfillPlan({
+ opts: PlanSchema.parse({ target: 'app.events' }),
+ configPath: join(dir, 'config.ts'),
+ config: { metaDir: join(dir, 'meta'), schema: [join(dir, 'missing.ts')] },
+ clickhouseQuery: async () => [...rows] as T[],
+ })).rejects.toThrow(message)
+ } finally { await rm(dir, { recursive: true, force: true }) }
+ })
+}
diff --git a/packages/plugin-backfill/src/planner.ts b/packages/plugin-backfill/src/planner.ts
index e5d40c2f..1658160c 100644
--- a/packages/plugin-backfill/src/planner.ts
+++ b/packages/plugin-backfill/src/planner.ts
@@ -38,7 +38,6 @@ async function detectBackfillStrategy(input: {
try {
const definitions = await loadSchemaDefinitions(input.schema, { cwd: input.configDir })
const mvs = findMvsForTarget(definitions, input.database, input.table)
- if (mvs.length === 0) return { mvs: [] }
const tableDef = definitions.find(
(definition) =>
@@ -46,17 +45,46 @@ async function detectBackfillStrategy(input: {
definition.database === input.database &&
definition.name === input.table
)
+ if (tableDef?.kind === 'table' && tableDef.columns.some((column) => column.defaultKind === 'EPHEMERAL')) {
+ throw new BackfillConfigError('Automatic backfill cannot reconstruct EPHEMERAL inputs; use an explicit INSERT with an input column mapping.')
+ }
+ if (mvs.length === 0) return { mvs: [] }
return {
mvs,
mvReplayQueries: mvs.map((mv) => mv.as),
- targetColumns: tableDef?.kind === 'table' ? tableDef.columns.map((column) => column.name) : undefined,
+ targetColumns: tableDef?.kind === 'table'
+ ? tableDef.columns
+ .filter((column) => !column.defaultKind || column.defaultKind === 'DEFAULT')
+ .map((column) => column.name)
+ : undefined,
}
- } catch {
+ } catch (error) {
+ if (error instanceof BackfillConfigError) throw error
// Schema load failed, fall back to direct copy.
return { mvs: [] }
}
}
+/** Missing, unreadable or unsupported metadata must never bypass the safety gate. */
+export async function assertBackfillTargetSafe(input: {
+ database: string
+ table: string
+ query: (sql: string, settings?: Record) => Promise
+ querySettings?: Record
+}): Promise {
+ const quote = (value: string) => `'${value.replaceAll('\\', '\\\\').replaceAll("'", "\\'")}'`
+ const rows = await input.query<{ name: string; default_kind: string }>(
+ `SELECT name, default_kind FROM system.columns WHERE database = ${quote(input.database)} AND table = ${quote(input.table)} ORDER BY position`,
+ input.querySettings
+ )
+ if (!rows.length || rows.some((column) => !column.name || !['', 'DEFAULT', 'MATERIALIZED', 'ALIAS', 'EPHEMERAL'].includes(column.default_kind))) {
+ throw new BackfillConfigError('Cannot verify live target column kinds; automatic backfill is blocked. Check metadata access and use an explicit INSERT if needed.')
+ }
+ if (rows.some((column) => column.default_kind === 'EPHEMERAL')) {
+ throw new BackfillConfigError('Automatic backfill cannot reconstruct EPHEMERAL inputs; use an explicit INSERT with an input column mapping.')
+ }
+}
+
export async function buildBackfillPlan(input: {
opts: PlanOptions
configPath: string
@@ -74,14 +102,16 @@ export async function buildBackfillPlan(input: {
// Detect the execution strategy before chunk planning: an mv_replay backfill
// sizes its chunks against the MV *source* (the table its SELECT reads),
// because the injected chunk conditions run against that source — not the
- // target, which is legitimately empty when bootstrapping an aggregate. Only
- // the copy path introspects the target itself.
+ // target, which is legitimately empty when bootstrapping an aggregate. Target column safety is checked separately for both paths.
const strategy = await detectBackfillStrategy({
schema: input.config.schema,
configDir: dirname(input.configPath),
database,
table,
})
+ await assertBackfillTargetSafe({
+ database, table, query: input.clickhouseQuery, querySettings: input.querySettings,
+ })
const replaySource = strategy.mvReplayQueries ? resolveMvReplaySource(strategy.mvs) : undefined
const chunkSource = replaySource ?? { database, table }
diff --git a/packages/plugin-backfill/src/plugin.ts b/packages/plugin-backfill/src/plugin.ts
index 4fca65be..c519ad9c 100644
--- a/packages/plugin-backfill/src/plugin.ts
+++ b/packages/plugin-backfill/src/plugin.ts
@@ -30,7 +30,7 @@ import {
type StatusOptions,
} from './options.js'
import { planPayload, statusPayload, cancelPayload, doctorPayload } from './payload.js'
-import { buildBackfillPlan } from './planner.js'
+import { assertBackfillTargetSafe, buildBackfillPlan } from './planner.js'
import { evaluateBackfillCheck } from './check.js'
import { cancelBackfillRun, getBackfillDoctorReport, getBackfillStatus } from './queries.js'
import {
@@ -116,6 +116,9 @@ async function runBackfill(input: {
const db = createClickHouseExecutor(input.clickhouse)
try {
+ const [database, table] = plan.target.split('.')
+ if (!database || !table) throw new BackfillConfigError('Invalid backfill target.')
+ await assertBackfillTargetSafe({ database, table, query: (sql, settings) => db.query(sql, settings) })
const runState: BackfillRunState = {
planId: plan.planId,
target: plan.target,
diff --git a/packages/plugin-codegen/src/generators/ingest-artifacts.ts b/packages/plugin-codegen/src/generators/ingest-artifacts.ts
index b15d8586..aa25465a 100644
--- a/packages/plugin-codegen/src/generators/ingest-artifacts.ts
+++ b/packages/plugin-codegen/src/generators/ingest-artifacts.ts
@@ -11,7 +11,7 @@ import type {
} from '../types.js'
import { normalizeCodegenOptions } from '../options.js'
import { resolveTableNames } from '../naming.js'
-import { renderHeader } from './shared.js'
+import { insertTypeName, renderHeader } from './shared.js'
function computeRelativeImportPath(fromFile: string, toFile: string): string {
const fromDir = dirname(fromFile)
@@ -34,21 +34,22 @@ function renderIngestFunction(
): string[] {
const funcName = `ingest${stripRowSuffix(interfaceName)}`
const tableFqn = `${table.database}.${table.name}`
+ const inputType = insertTypeName(table, interfaceName)
const lines: string[] = []
if (emitZod) {
lines.push(`export async function ${funcName}(`)
lines.push(` ingestor: Ingestor,`)
- lines.push(` rows: ${interfaceName}[],`)
+ lines.push(` rows: ${inputType}[],`)
lines.push(` options?: IngestOptions`)
lines.push(`): Promise {`)
- lines.push(` const data = options?.validate ? rows.map(row => ${interfaceName}Schema.parse(row)) : rows`)
+ lines.push(` const data = options?.validate ? rows.map(row => ${inputType}Schema.parse(row)) : rows`)
lines.push(` await ingestor.insert({ table: '${tableFqn}', values: data, compressed: options?.compressed ?? true })`)
lines.push(`}`)
} else {
lines.push(`export async function ${funcName}(`)
lines.push(` ingestor: Ingestor,`)
- lines.push(` rows: ${interfaceName}[],`)
+ lines.push(` rows: ${inputType}[],`)
lines.push(` options?: IngestOptions`)
lines.push(`): Promise {`)
lines.push(` await ingestor.insert({ table: '${tableFqn}', values: rows, compressed: options?.compressed ?? true })`)
@@ -76,9 +77,12 @@ export function generateIngestArtifacts(
const typeImports: string[] = []
const valueImports: string[] = []
for (const entry of resolved) {
- typeImports.push(entry.interfaceName)
+ const name = entry.definition.kind === 'table'
+ ? insertTypeName(entry.definition, entry.interfaceName)
+ : entry.interfaceName
+ typeImports.push(name)
if (normalized.emitZod) {
- valueImports.push(`${entry.interfaceName}Schema`)
+ valueImports.push(`${name}Schema`)
}
}
diff --git a/packages/plugin-codegen/src/generators/shared.ts b/packages/plugin-codegen/src/generators/shared.ts
index 8f8751ac..b3688553 100644
--- a/packages/plugin-codegen/src/generators/shared.ts
+++ b/packages/plugin-codegen/src/generators/shared.ts
@@ -1,3 +1,5 @@
+import type { TableDefinition } from '@chkit/core'
+
export function renderHeader(toolVersion: string): string[] {
const lines = [
'// This file is auto-generated by chkit codegen — do not edit manually.',
@@ -5,3 +7,10 @@ export function renderHeader(toolVersion: string): string[] {
lines.push(`// chkit-codegen-version: ${toolVersion}`)
return lines
}
+
+/** Preserve existing names for ordinary tables; special columns need an insert shape. */
+export function insertTypeName(table: TableDefinition, rowType: string): string {
+ return table.columns.some((column) => column.defaultKind && column.defaultKind !== 'DEFAULT')
+ ? `${rowType}Insert`
+ : rowType
+}
diff --git a/packages/plugin-codegen/src/generators/type-artifacts.ts b/packages/plugin-codegen/src/generators/type-artifacts.ts
index 90eaece0..d3dacc8d 100644
--- a/packages/plugin-codegen/src/generators/type-artifacts.ts
+++ b/packages/plugin-codegen/src/generators/type-artifacts.ts
@@ -17,7 +17,7 @@ import type {
import { UnsupportedTypeError } from '../errors.js'
import { normalizeCodegenOptions } from '../options.js'
import { renderPropertyName, resolveTableNames } from '../naming.js'
-import { renderHeader } from './shared.js'
+import { insertTypeName, renderHeader } from './shared.js'
const LARGE_INTEGER_TYPES = new Set([
'Int64',
@@ -268,7 +268,25 @@ function renderTableInterface(
interfaceName: string,
options: Required
): { lines: string[]; findings: CodegenFinding[] } {
- return renderFieldsInterface(table.columns, interfaceName, `${table.database}.${table.name}`, options)
+ const path = `${table.database}.${table.name}`
+ const read = renderFieldsInterface(
+ table.columns.filter((column) => !column.defaultKind || column.defaultKind === 'DEFAULT'),
+ interfaceName, path, options
+ )
+ const insertName = insertTypeName(table, interfaceName)
+ if (insertName === interfaceName) return read
+ const insert = renderFieldsInterface(
+ table.columns.filter((column) => column.defaultKind !== 'MATERIALIZED' && column.defaultKind !== 'ALIAS'),
+ insertName, path, options
+ )
+ const explicit = renderFieldsInterface(
+ table.columns.filter((column) => column.defaultKind !== 'EPHEMERAL'),
+ `${interfaceName}Explicit`, path, options
+ )
+ return {
+ lines: [...read.lines, '', ...explicit.lines, '', ...insert.lines],
+ findings: [...read.findings, ...explicit.findings, ...insert.findings],
+ }
}
function renderDictionaryInterface(
diff --git a/packages/plugin-pull/src/render-schema.ts b/packages/plugin-pull/src/render-schema.ts
index 3afaa063..1b85fb43 100644
--- a/packages/plugin-pull/src/render-schema.ts
+++ b/packages/plugin-pull/src/render-schema.ts
@@ -193,6 +193,7 @@ function renderColumn(column: ColumnDefinition): string {
`type: ${renderString(column.type)}`,
]
if (column.nullable) parts.push('nullable: true')
+ if (column.defaultKind && column.defaultKind !== 'DEFAULT') parts.push(`defaultKind: ${renderString(column.defaultKind)}`)
if (column.default !== undefined) parts.push(`default: ${renderLiteral(column.default)}`)
if (column.comment) parts.push(`comment: ${renderString(column.comment)}`)
if (column.codec) parts.push(`codec: ${renderCodecSource(column.codec)}`)