Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .changeset/drift-index-arguments.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
"@chkit/clickhouse": patch
"@chkit/plugin-obsessiondb": patch
"chkit": patch
---

Stop `chkit drift` from reporting `index_mismatch` for skip indexes it just created. Introspection read `system.data_skipping_indices.type`, which holds only the index name (`ngrambf_v1`), so every argument parsed as 0; it now reads `type_full` (`ngrambf_v1(3, 4096, 2, 0)`). chkit renders `INDEX name (expr)` and ClickHouse keeps those parentheses in `expr`, so the comparison now drops one pair when it encloses the whole expression. chkit-py introspection reads `type_full` as well.
8 changes: 8 additions & 0 deletions .changeset/text-skip-index.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
---
"@chkit/core": patch
"@chkit/clickhouse": patch
"@chkit/plugin-pull": patch
"chkit": patch
---

Add the ClickHouse `text` skip index (GA in 26.2) as `type: 'text'`, with a required `tokenizer` and the optional `preprocessor`, `postprocessor`, `supportPhraseSearch`, `dictionaryBlockSize`, `dictionaryBlockFrontcodingCompression`, `postingListBlockSize`, and `postingListCodec`. chkit renders the parameters in a fixed order, and introspection parses `type_full` by key, so `chkit drift` stays clean whatever order the DDL used. `chkit pull` writes the fields back, validation reports `text_index_missing_tokenizer`, and chkit-py has the same support as `SkipIndexText`.
3 changes: 2 additions & 1 deletion apps/docs/src/content/docs/schema/dsl-reference.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -358,7 +358,7 @@ Each entry in the `indexes` array is a `SkipIndexDefinition`. The shared base fi
|-------|------|-------------|
| `name` | `string` | Index name |
| `expression` | `string` | Indexed expression |
| `type` | `'minmax' \| 'set' \| 'bloom_filter' \| 'tokenbf_v1' \| 'ngrambf_v1'` | Index type |
| `type` | `'minmax' \| 'set' \| 'bloom_filter' \| 'tokenbf_v1' \| 'ngrambf_v1' \| 'text'` | Index type |
| `granularity` | `number` | Index granularity |

Type-specific fields:
Expand All @@ -370,6 +370,7 @@ Type-specific fields:
| `bloom_filter` | — | `falsePositiveRate: number` | Defaults to `0.025` when omitted |
| `tokenbf_v1` | `sizeBytes`, `hashFunctions`, `randomSeed` (all `number`) | — | Maps to `tokenbf_v1(size_bytes, n_hash, seed)` |
| `ngrambf_v1` | `ngramSize`, `sizeBytes`, `hashFunctions`, `randomSeed` (all `number`) | — | Maps to `ngrambf_v1(n, size_bytes, n_hash, seed)` |
| `text` | `tokenizer: string` (SQL, e.g. `'splitByNonAlpha'`, `'ngrams(3)'`) | `preprocessor`, `postprocessor` (SQL `string`); `supportPhraseSearch`, `dictionaryBlockFrontcodingCompression` (`boolean`); `dictionaryBlockSize`, `postingListBlockSize` (`number`); `postingListCodec: 'none' \| 'bitpacking'` | Maps to `text(tokenizer = …, …)` in a fixed parameter order. Needs ClickHouse 26.2 or newer; `supportPhraseSearch` also needs the MergeTree setting `allow_experimental_text_index_phrase_search` |

<Tabs syncKey="lang">
<TabItem label="TypeScript">
Expand Down
2 changes: 2 additions & 0 deletions chkit_python/src/chkit/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@
SkipIndexMinmax,
SkipIndexNgramBF,
SkipIndexSet,
SkipIndexText,
SkipIndexTokenBF,
)

Expand Down Expand Up @@ -75,6 +76,7 @@
"SkipIndexMinmax",
"SkipIndexNgramBF",
"SkipIndexSet",
"SkipIndexText",
"SkipIndexTokenBF",
"TableDefinition",
"TableRef",
Expand Down
24 changes: 23 additions & 1 deletion chkit_python/src/chkit/cli/commands/drift_compare.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@
)
from chkit.core.projection import is_index_projection, normalize_projection_index
from chkit.core.sql_normalizer import normalize_engine, normalize_sql_fragment
from chkit.core.text_index import render_text_index_type

_MIN_QUOTED_LEN = 2

Expand Down Expand Up @@ -259,16 +260,37 @@ def _render_index_type_fingerprint(index: SkipIndexDefinition) -> str:
f"tokenbf_v1({index.size_bytes}, "
f"{index.hash_functions}, {index.random_seed})"
)
if index.type == "text":
return render_text_index_type(index)
return (
f"ngrambf_v1({index.ngram_size}, {index.size_bytes}, "
f"{index.hash_functions}, {index.random_seed})"
)


def _strip_enclosing_parens(value: str) -> str:
"""Drop one pair of parentheses only when it encloses the whole value.

chkit renders ``INDEX name (expr)`` and ClickHouse keeps those parentheses
in ``system.data_skipping_indices.expr``; ``(a) || (b)`` stays as written.
"""
if not (value.startswith("(") and value.endswith(")")):
return value
depth = 0
for i, ch in enumerate(value):
if ch == "(":
depth += 1
elif ch == ")":
depth -= 1
if depth == 0 and i < len(value) - 1:
return value
return value[1:-1].strip()


def _normalize_index_shape(index: SkipIndexDefinition) -> str:
return "|".join(
[
f"expr={normalize_sql_fragment(index.expression)}",
f"expr={_strip_enclosing_parens(normalize_sql_fragment(index.expression))}",
f"type={_render_index_type_fingerprint(index)}",
f"granularity={index.granularity}",
]
Expand Down
2 changes: 2 additions & 0 deletions chkit_python/src/chkit/cli/commands/pull.py
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,7 @@
SkipIndexMinmax,
SkipIndexNgramBF,
SkipIndexSet,
SkipIndexText,
SkipIndexTokenBF,
TableDefinition,
TableRef,
Expand Down Expand Up @@ -92,6 +93,7 @@ def _introspected_table_to_definition(
| SkipIndexBloomFilter
| SkipIndexTokenBF
| SkipIndexNgramBF
| SkipIndexText
| dict[str, object]
] = list(item.indexes)

Expand Down
17 changes: 17 additions & 0 deletions chkit_python/src/chkit/cli/commands/pull_render.py
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,7 @@ def render_schema_file( # noqa: PLR0912, PLR0915
"bloom_filter": "SkipIndexBloomFilter",
"tokenbf_v1": "SkipIndexTokenBF",
"ngrambf_v1": "SkipIndexNgramBF",
"text": "SkipIndexText",
}[idx.type]
)

Expand Down Expand Up @@ -245,13 +246,29 @@ def _render_index(index: SkipIndexDefinition) -> str:
parts.append(f"size_bytes={index.size_bytes}")
parts.append(f"hash_functions={index.hash_functions}")
parts.append(f"random_seed={index.random_seed}")
elif index.type == "text":
parts.append(f"tokenizer={_render_string(index.tokenizer)}")
for field in ("preprocessor", "postprocessor", "posting_list_codec"):
value = getattr(index, field)
if value is not None:
parts.append(f"{field}={_render_string(value)}")
for field in (
"support_phrase_search",
"dictionary_block_size",
"dictionary_block_frontcoding_compression",
"posting_list_block_size",
):
value = getattr(index, field)
if value is not None:
parts.append(f"{field}={value}")
parts.append(f"granularity={index.granularity}")
type_class = {
"minmax": "SkipIndexMinmax",
"set": "SkipIndexSet",
"bloom_filter": "SkipIndexBloomFilter",
"tokenbf_v1": "SkipIndexTokenBF",
"ngrambf_v1": "SkipIndexNgramBF",
"text": "SkipIndexText",
}[index.type]
return f"{type_class}({', '.join(parts)})"

Expand Down
11 changes: 9 additions & 2 deletions chkit_python/src/chkit/clickhouse/introspect.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,9 +45,11 @@
SkipIndexMinmax,
SkipIndexNgramBF,
SkipIndexSet,
SkipIndexText,
SkipIndexTokenBF,
)
from chkit.core.sql_normalizer import normalize_sql_fragment
from chkit.core.text_index import parse_text_index_params

SchemaObjectKind: TypeAlias = Literal["table", "view", "materialized_view", "dictionary"]

Expand Down Expand Up @@ -122,7 +124,7 @@ class IntrospectedTable:


_NULLABLE_RE = re.compile(r"^Nullable\((.+)\)$")
_INDEX_TYPE_RE = re.compile(r"^(\w+)\((.+)\)$")
_INDEX_TYPE_RE = re.compile(r"^(\w+)\((.+)\)$", re.DOTALL)


def infer_schema_kind_from_engine(engine: str) -> SchemaObjectKind | None:
Expand Down Expand Up @@ -215,6 +217,11 @@ def normalize_index_from_system_row(row: SystemSkippingIndexRow) -> SkipIndexDef
if base_name == "minmax":
return SkipIndexMinmax(**base_payload)

if base_name == "text":
return SkipIndexText.model_validate(
{**base_payload, **parse_text_index_params(args_str or "")}
)

if base_name == "bloom_filter":
floats = _split_float_args(args_str)
rate = floats[0] if floats else None
Expand Down Expand Up @@ -370,7 +377,7 @@ def list_table_details(client: Any, databases: list[str]) -> list[IntrospectedTa
f"FROM system.columns WHERE database IN ({quoted})"
).rows
index_rows_raw = client.query(
f"SELECT database, table, name, expr, type, granularity "
f"SELECT database, table, name, expr, type_full AS type, granularity "
f"FROM system.data_skipping_indices WHERE database IN ({quoted})"
).rows

Expand Down
9 changes: 8 additions & 1 deletion chkit_python/src/chkit/core/canonical.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,14 @@ def _canonicalize_column(column: ColumnDefinition) -> ColumnDefinition:


def _canonicalize_index(index: SkipIndexDefinition) -> SkipIndexDefinition:
return index.model_copy(update={"expression": normalize_sql_fragment(index.expression)})
update: dict[str, object] = {"expression": normalize_sql_fragment(index.expression)}
if index.type == "text":
update["tokenizer"] = normalize_sql_fragment(index.tokenizer)
for field in ("preprocessor", "postprocessor"):
value = getattr(index, field)
if value is not None:
update[field] = normalize_sql_fragment(value)
return index.model_copy(update=update)


def _sorted_settings(
Expand Down
30 changes: 29 additions & 1 deletion chkit_python/src/chkit/core/model.py
Original file line number Diff line number Diff line change
Expand Up @@ -209,12 +209,39 @@ class SkipIndexNgramBF(_SkipIndexBase):
)


class SkipIndexText(_SkipIndexBase):
"""ClickHouse ``text(...)`` index (26.2+); ``tokenizer`` is required."""

type: Literal["text"] = "text"
tokenizer: str
preprocessor: str | None = None
postprocessor: str | None = None
support_phrase_search: bool | None = Field(default=None, alias="supportPhraseSearch")
dictionary_block_size: int | None = Field(default=None, alias="dictionaryBlockSize")
dictionary_block_frontcoding_compression: bool | None = Field(
default=None, alias="dictionaryBlockFrontcodingCompression"
)
posting_list_block_size: int | None = Field(default=None, alias="postingListBlockSize")
posting_list_codec: Literal["none", "bitpacking"] | None = Field(
default=None, alias="postingListCodec"
)

model_config = ConfigDict(
frozen=True,
extra="forbid",
strict=True,
validate_assignment=True,
populate_by_name=True,
)


SkipIndexDefinition: TypeAlias = Annotated[
SkipIndexMinmax
| SkipIndexSet
| SkipIndexBloomFilter
| SkipIndexTokenBF
| SkipIndexNgramBF,
| SkipIndexNgramBF
| SkipIndexText,
Field(discriminator="type"),
]

Expand Down Expand Up @@ -652,6 +679,7 @@ class MigrationPlan(_StrictModel):
"duplicate_object_name",
"duplicate_column_name",
"duplicate_index_name",
"text_index_missing_tokenizer",
"duplicate_projection_name",
"projection_ambiguous_kind",
"projection_empty_index",
Expand Down
9 changes: 6 additions & 3 deletions chkit_python/src/chkit/core/sql.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,12 +21,14 @@
SkipIndexDefinition,
SkipIndexMinmax,
SkipIndexSet,
SkipIndexText,
SkipIndexTokenBF,
TableDefinition,
TableRef,
ViewDefinition,
)
from chkit.core.projection import render_projection_body
from chkit.core.text_index import render_text_index_type
from chkit.core.validate import assert_valid_definitions

_COLUMN_ADAPTER: TypeAdapter[ColumnDefinition] = TypeAdapter(ColumnDefinition)
Expand Down Expand Up @@ -96,11 +98,12 @@ def _render_index_type(idx: SkipIndexDefinition) -> str:
if isinstance(idx, SkipIndexSet):
return f"set({idx.max_rows})"
if isinstance(idx, SkipIndexBloomFilter):
if idx.false_positive_rate is not None:
return f"bloom_filter({idx.false_positive_rate})"
return "bloom_filter"
rate = idx.false_positive_rate
return "bloom_filter" if rate is None else f"bloom_filter({rate})"
if isinstance(idx, SkipIndexTokenBF):
return f"tokenbf_v1({idx.size_bytes}, {idx.hash_functions}, {idx.random_seed})"
if isinstance(idx, SkipIndexText):
return render_text_index_type(idx)
# SkipIndexNgramBF is the only remaining variant in the discriminated union.
return (
f"ngrambf_v1({idx.ngram_size}, {idx.size_bytes}, {idx.hash_functions}, "
Expand Down
75 changes: 75 additions & 0 deletions chkit_python/src/chkit/core/text_index.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,75 @@
"""Render and parse the ClickHouse ``text(...)`` skip index type.

Mirrors ``packages/core/src/text-index.ts``. ClickHouse keeps parameters in
the order the DDL was written (``system.data_skipping_indices.type_full``), so
comparisons go through parse + render rather than string equality.
"""

from __future__ import annotations

from typing import TYPE_CHECKING, Literal

from chkit.core.key_clause import split_top_level_comma
from chkit.core.sql_normalizer import normalize_sql_fragment

if TYPE_CHECKING:
from chkit.core.model import SkipIndexText

_Kind = Literal["sql", "flag", "number", "string"]

# (model field, DDL key, kind) in render order.
PARAMS: tuple[tuple[str, str, _Kind], ...] = (
("tokenizer", "tokenizer", "sql"),
("preprocessor", "preprocessor", "sql"),
("postprocessor", "postprocessor", "sql"),
("support_phrase_search", "support_phrase_search", "flag"),
("dictionary_block_size", "dictionary_block_size", "number"),
(
"dictionary_block_frontcoding_compression",
"dictionary_block_frontcoding_compression",
"flag",
),
("posting_list_block_size", "posting_list_block_size", "number"),
("posting_list_codec", "posting_list_codec", "string"),
)


def render_text_index_type(index: SkipIndexText) -> str:
parts: list[str] = []
for field, key, kind in PARAMS:
value = getattr(index, field)
if value is None:
continue
if kind == "sql":
parts.append(f"{key} = {normalize_sql_fragment(str(value))}")
elif kind == "flag":
parts.append(f"{key} = {1 if value else 0}")
elif kind == "number":
parts.append(f"{key} = {value}")
else:
parts.append(f"{key} = '{value}'")
return f"text({', '.join(parts)})"


def parse_text_index_params(args: str) -> dict[str, str | bool | int]:
"""Parse the argument list of a ``text(...)`` index type into model fields."""
values: dict[str, str] = {}
for part in split_top_level_comma(args):
key, sep, raw = part.partition("=")
if sep:
values[key.strip()] = raw.strip()
params: dict[str, str | bool | int] = {"tokenizer": ""}
for field, key, kind in PARAMS:
value = values.get(key)
if value is None:
continue
if kind == "sql":
params[field] = normalize_sql_fragment(value)
elif kind == "flag":
params[field] = value == "1" or value.lower() == "true"
elif kind == "number":
params[field] = int(value)
else:
quoted = len(value) > 1 and value[0] == value[-1] == "'"
params[field] = value[1:-1] if quoted else value
return params
Loading