From e1089b7b8cc5f82994a4c74868ff8513a77151be Mon Sep 17 00:00:00 2001 From: sdairs Date: Fri, 2 Oct 2026 16:21:04 +0100 Subject: [PATCH 1/2] Add semantic notes with real CPU embeddings and revisioned pgvector storage --- .github/workflows/semantic-notes.yml | 27 + applications/semantic-notes/.env.example | 11 + applications/semantic-notes/.gitignore | 9 + applications/semantic-notes/README.md | 179 +++++++ applications/semantic-notes/alembic.ini | 3 + applications/semantic-notes/migrations/env.py | 10 + .../migrations/versions/0001_notes.py | 44 ++ applications/semantic-notes/model-spec.json | 9 + applications/semantic-notes/pyproject.toml | 5 + .../semantic-notes/requirements-dev.txt | 4 + applications/semantic-notes/requirements.in | 14 + applications/semantic-notes/requirements.txt | 59 +++ .../semantic-notes/samples/notes.json | 5 + .../semantic-notes/scripts/bootstrap.sql | 14 + .../semantic-notes/scripts/grants.sql | 7 + .../semantic-notes/scripts/model_preflight.py | 72 +++ applications/semantic-notes/scripts/seed.py | 29 ++ .../semantic-notes/semantic_notes/__init__.py | 0 .../semantic-notes/semantic_notes/app.py | 260 ++++++++++ .../semantic-notes/semantic_notes/database.py | 43 ++ .../semantic_notes/embeddings.py | 107 ++++ .../semantic-notes/semantic_notes/models.py | 33 ++ .../semantic-notes/semantic_notes/service.py | 125 +++++ .../semantic-notes/semantic_notes/spec.py | 26 + .../semantic_notes/validation.py | 63 +++ .../semantic-notes/semantic_notes/worker.py | 39 ++ applications/semantic-notes/static/style.css | 1 + .../semantic-notes/templates/base.html | 3 + .../semantic-notes/templates/edit.html | 1 + .../semantic-notes/templates/error.html | 1 + .../semantic-notes/templates/index.html | 5 + .../semantic-notes/templates/search.html | 1 + .../tests/browser_acceptance.py | 125 +++++ .../semantic-notes/tests/cloud_acceptance.py | 487 ++++++++++++++++++ .../semantic-notes/tests/test_boundaries.py | 57 ++ .../semantic-notes/tests/test_worker.py | 50 ++ 36 files changed, 1928 insertions(+) create mode 100644 .github/workflows/semantic-notes.yml create mode 100644 applications/semantic-notes/.env.example create mode 100644 applications/semantic-notes/.gitignore create mode 100644 applications/semantic-notes/README.md create mode 100644 applications/semantic-notes/alembic.ini create mode 100644 applications/semantic-notes/migrations/env.py create mode 100644 applications/semantic-notes/migrations/versions/0001_notes.py create mode 100644 applications/semantic-notes/model-spec.json create mode 100644 applications/semantic-notes/pyproject.toml create mode 100644 applications/semantic-notes/requirements-dev.txt create mode 100644 applications/semantic-notes/requirements.in create mode 100644 applications/semantic-notes/requirements.txt create mode 100644 applications/semantic-notes/samples/notes.json create mode 100644 applications/semantic-notes/scripts/bootstrap.sql create mode 100644 applications/semantic-notes/scripts/grants.sql create mode 100644 applications/semantic-notes/scripts/model_preflight.py create mode 100644 applications/semantic-notes/scripts/seed.py create mode 100644 applications/semantic-notes/semantic_notes/__init__.py create mode 100644 applications/semantic-notes/semantic_notes/app.py create mode 100644 applications/semantic-notes/semantic_notes/database.py create mode 100644 applications/semantic-notes/semantic_notes/embeddings.py create mode 100644 applications/semantic-notes/semantic_notes/models.py create mode 100644 applications/semantic-notes/semantic_notes/service.py create mode 100644 applications/semantic-notes/semantic_notes/spec.py create mode 100644 applications/semantic-notes/semantic_notes/validation.py create mode 100644 applications/semantic-notes/semantic_notes/worker.py create mode 100644 applications/semantic-notes/static/style.css create mode 100644 applications/semantic-notes/templates/base.html create mode 100644 applications/semantic-notes/templates/edit.html create mode 100644 applications/semantic-notes/templates/error.html create mode 100644 applications/semantic-notes/templates/index.html create mode 100644 applications/semantic-notes/templates/search.html create mode 100644 applications/semantic-notes/tests/browser_acceptance.py create mode 100644 applications/semantic-notes/tests/cloud_acceptance.py create mode 100644 applications/semantic-notes/tests/test_boundaries.py create mode 100644 applications/semantic-notes/tests/test_worker.py diff --git a/.github/workflows/semantic-notes.yml b/.github/workflows/semantic-notes.yml new file mode 100644 index 00000000..62e71fb0 --- /dev/null +++ b/.github/workflows/semantic-notes.yml @@ -0,0 +1,27 @@ +name: Semantic notes +on: + push: + paths: ['applications/semantic-notes/**', '.github/workflows/semantic-notes.yml'] + pull_request: + paths: ['applications/semantic-notes/**', '.github/workflows/semantic-notes.yml'] +permissions: + contents: read +jobs: + checks: + runs-on: ubuntu-24.04 + defaults: + run: + working-directory: applications/semantic-notes + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + cache: pip + cache-dependency-path: | + applications/semantic-notes/requirements.txt + applications/semantic-notes/requirements-dev.txt + - run: pip install -r requirements-dev.txt && pip check + - run: ruff check semantic_notes scripts tests migrations + - run: ruff format --check semantic_notes scripts tests migrations + - run: python -m unittest discover -s tests -v diff --git a/applications/semantic-notes/.env.example b/applications/semantic-notes/.env.example new file mode 100644 index 00000000..b11578b1 --- /dev/null +++ b/applications/semantic-notes/.env.example @@ -0,0 +1,11 @@ +PGHOST=your-cloud-postgres-host +PGPORT=5432 +PGDATABASE=postgres +PGUSER=semantic_app +PGPASSWORD=replace-with-runtime-role-password +PGSSLMODE=verify-full +PGSSLROOTCERT=/absolute/path/to/ca.pem +MODEL_CACHE_DIR=/absolute/path/to/model-cache +HF_HUB_OFFLINE=1 +TOKENIZERS_PARALLELISM=false +APP_ORIGIN=http://127.0.0.1:8000 diff --git a/applications/semantic-notes/.gitignore b/applications/semantic-notes/.gitignore new file mode 100644 index 00000000..ddee2c54 --- /dev/null +++ b/applications/semantic-notes/.gitignore @@ -0,0 +1,9 @@ +.venv/ +.local/ +.private/ +.env +.env.* +!.env.example +__pycache__/ +*.pyc +.ruff_cache/ diff --git a/applications/semantic-notes/README.md b/applications/semantic-notes/README.md new file mode 100644 index 00000000..ffb0a8a6 --- /dev/null +++ b/applications/semantic-notes/README.md @@ -0,0 +1,179 @@ +# Semantic notes + +A trusted local notebook using FastAPI, SQLModel, Sentence Transformers and pgvector on **ClickHouse Managed Postgres (public beta)**. Add short titled notes, edit them with an observed revision, and retrieve excerpts by meaning. The application returns stored notes and cosine distances; it does not generate answers. + +The CPU model is `sentence-transformers/all-MiniLM-L6-v2`, fixed at public revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`. Only allowlisted model files and safetensors weights are downloaded; remote custom code is disabled. After the explicit download, application startup is offline. A full collection specification fixes model ID, revision, 384 dimensions, float32 values, normalized embeddings, 256 tokens including special tokens, and document formatting (`title + two newlines + body`). Source metadata and database metadata must match at startup. + +## Behavior and limits + +- At most 100 notes, enforced by a collection row lock and database quota trigger. Listing shows 12 notes per page. +- Title: 1–120 characters. Body: 1–4,096 characters. Both together must fit the actual tokenizer's 256-token limit, including special tokens. Inputs are rejected before encoding rather than silently truncated. Controls and lone Unicode surrogates are rejected; body permits newline/tab. +- Search text: 1–512 characters and the same tokenizer bound. Results: 1–10. Optional literal title substring: at most 80 characters. `%` and `_` are escaped rather than interpreted as filter wildcards. +- Exact cosine search over the small collection, ordered by distance then UUID. There is no approximate index or measured recall/latency claim. +- Two admitted embedding jobs per process; shared model access is serialized because `encode()` changes model state. Excess jobs receive 429. Cancellation keeps the permit until the actual worker completes. CPU inference runs outside the event loop and before database acquisition; blocking database work also runs in a thread. +- Edit saves text, vector and incremented revision in one transaction. A stale revision receives 409 and the form keeps submitted text until explicit reload. Slow older embedding work cannot overwrite a newer edit. +- Requests are bounded to 16 KiB including streamed bodies. Unsafe browser/API requests require the fixed `Origin: http://127.0.0.1:8000`. Numeric form fields have digit limits before conversion; duplicate and unknown fields are rejected. + +This is a single trusted operator's loopback workbench, without application authentication or user isolation. Bind to `127.0.0.1:8000`, use one Uvicorn worker, and open that exact address. Do not expose it through a public proxy. The shared runtime database role is trusted: PostgreSQL enforces shape, nonzero vectors, collection metadata and quota, but cannot establish that a direct writer used the matching text/model or observed revision. + +## 1. Native Linux CPU setup + +Tested on Ubuntu 24.04 ARM64 with Python 3.12.3. Install Python virtual-environment support and the PostgreSQL client using your Linux package manager. Run commands from this application directory: + +```bash +python3 -m venv .venv +.venv/bin/pip install -r requirements.txt +.venv/bin/pip check +export MODEL_CACHE_DIR="$PWD/.local/model-cache" +.venv/bin/python -m scripts.model_preflight --download +``` + +The lock pins the full runtime dependency graph and `torch==2.14.1+cpu` from the official CPU wheel index. `requirements.in` records direct dependencies; `requirements.txt` records the resolved versions. The preflight downloads the pinned public model, runs real CPU inference, checks finite/nonzero 384-dimensional values, compares related/unrelated synthetic texts, and proves 256 tokens accepted / 257 rejected. Complete this before allocating a billable service. Keep the cache: application startup does not download missing weights. + +## 2. Create a dedicated Cloud service + +Install/configure [clickhousectl](https://github.com/ClickHouse/clickhousectl) with your Cloud API credentials. Confirm supported region/size in current [Managed Postgres documentation](https://clickhouse.com/docs/products/managed-postgres/). The tested fixture used `c6gd.large`, AWS `us-east-1`, PostgreSQL 18 and no HA. Service creation is billable; there is no free-running-service assumption. + +```bash +mkdir -p .private +chmod 700 .private +clickhousectl cloud postgres create \ + --name semantic-notes-demo --provider aws --region us-east-1 \ + --size c6gd.large --pg-version 18 --ha-type none \ + --tag purpose=semantic-notes --org-id YOUR_ORG_ID \ + --json > .private/create.json +chmod 600 .private/create.json +``` + +The creation receipt contains credentials once. Keep it private. Read its `id`, then repeat the following status command until `state` is `running`; do not start migrations while it is `creating`: + +```bash +clickhousectl cloud postgres get YOUR_SERVICE_ID --org-id YOUR_ORG_ID --json +clickhousectl cloud postgres certs get YOUR_SERVICE_ID \ + --org-id YOUR_ORG_ID --output .private/ca.pem +``` + +Create four separate private environment files, using the receipt without printing credentials: + +```bash +python3 - <<'PY' +import json, secrets, shlex +from pathlib import Path +p = Path('.private') +r = json.loads((p / 'create.json').read_text()) +base = {'PGHOST': r['hostname'], 'PGPORT': '5432', 'PGDATABASE': 'postgres', + 'PGSSLROOTCERT': str((p / 'ca.pem').resolve()), 'PGSSLMODE': 'verify-full'} +owner_password, app_password = secrets.token_urlsafe(32), secrets.token_urlsafe(32) +files = { + 'admin.env': {**base, 'PGUSER': r['username'], 'PGPASSWORD': r['password'], + 'PG_MIGRATION_PASSWORD': owner_password, 'PG_APP_PASSWORD': app_password}, + 'migration.env': {**base, 'PGUSER': 'semantic_owner', 'PGPASSWORD': owner_password}, + 'runtime.env': {**base, 'PGUSER': 'semantic_app', 'PGPASSWORD': app_password, + 'MODEL_CACHE_DIR': str(Path('.local/model-cache').resolve()), + 'HF_HUB_OFFLINE': '1', 'TOKENIZERS_PARALLELISM': 'false', + 'APP_ORIGIN': 'http://127.0.0.1:8000'}, + 'test.env': {'TEST_OWNER_USER': 'semantic_owner', 'TEST_OWNER_PASSWORD': owner_password}, +} +for name, values in files.items(): + f = p / name + f.write_text(''.join(k + '=' + shlex.quote(v) + '\n' for k, v in values.items())) + f.chmod(0o600) +PY +``` + +`.env.example` lists the runtime fields. The application does not load environment files implicitly. `set -a` below exports values to child processes. + +## 3. Administrator bootstrap and owner migrations + +Use a dedicated service. Bootstrap creates two login roles, installs `vector`, creates the owned schema and removes public schema/database creation privileges. It is intentionally a one-time script, not an idempotent role reset. + +```bash +# Setup shell only; close it before starting the application. +set -a +source .private/admin.env +set +a +psql -X -v ON_ERROR_STOP=1 -f scripts/bootstrap.sql + +set -a +source .private/migration.env +set +a +.venv/bin/alembic upgrade head +psql -X -v ON_ERROR_STOP=1 -f scripts/grants.sql +``` + +The runtime role receives schema usage, reads, note inserts and selected note updates. It receives `UPDATE(id)` on the fixed singleton collection solely to authorize `FOR UPDATE`; its CHECK permits only ID 1. It cannot modify model metadata, delete notes, install extensions or alter schema. The SQLAlchemy pool registers pgvector types on every actual psycopg connection and requires `sslmode=verify-full` with the Cloud CA. + +Migration lifecycle verification on an empty test notebook: + +```bash +# Owner shell; downgrade deletes all notes. +.venv/bin/alembic downgrade base +.venv/bin/alembic upgrade head +psql -X -v ON_ERROR_STOP=1 -f scripts/grants.sql +``` + +## 4. Run and use the notebook + +Open a **fresh shell** so administrator/test passwords are absent from the application process: + +```bash +set -a +source .private/runtime.env +set +a +.venv/bin/python -m scripts.seed # optional, run once: adds three actually embedded notes +.venv/bin/python -m uvicorn semantic_notes.app:app --host 127.0.0.1 --port 8000 +``` + +The seed is additive: repeating it adds three more notes. + +Visit [the local notebook](http://127.0.0.1:8000). Add a note, search for a paraphrase, edit the note, then reload. Opening the same note in two tabs demonstrates the explicit stale-save conflict. API documentation is at `/docs`; unsafe API clients must send the fixed Origin header: + +```bash +curl --fail-with-body http://127.0.0.1:8000/api/search \ + -H 'Origin: http://127.0.0.1:8000' -H 'Content-Type: application/json' \ + -d '{"q":"A journal helps me remember the books I read","k":3}' +``` + +Embedding happens before each mutation transaction. If inference fails, no note is saved. If the database transaction fails, text/vector/revision all roll back. A database error produces a static 503 response without another rendering query. There is no upload, arbitrary URL ingestion, model training or generated response. + +## 5. Verification + +Install helper tools natively; Python Playwright 1.56.0 is pinned: + +```bash +.venv/bin/pip install -r requirements-dev.txt +.venv/bin/pip check +.venv/bin/ruff check semantic_notes scripts tests migrations +.venv/bin/ruff format --check semantic_notes scripts tests migrations +.venv/bin/python -m unittest discover -s tests -v +.venv/bin/python -m playwright install --with-deps chromium +``` + +Five local tests cover strict input/revision/vector/model-format boundaries and actual thread cancellation capacity. CI runs these and formatting without Cloud credentials or a model download. + +Run Cloud acceptance **before seeding**, against a fresh empty schema with the real runtime server running. In a separate test shell export `.private/runtime.env` and `.private/test.env`, then: + +```bash +.venv/bin/python -m tests.cloud_acceptance +EVIDENCE_DIR=/tmp/semantic-notes-evidence \ + .venv/bin/python -m tests.browser_acceptance +``` + +The Cloud helper creates three real-model fixtures and has 11 cases: independent float64 cosine comparison, revision/competing HTTP row locks, delayed older CPU work, owner-induced deferred commit failure, Origin/token/body/form/Unicode bounds, role/vector/spec checks, database quota, CA/hostname negatives, stable ties and paging. Quota/paging controls reuse actual model vectors and make no semantic-quality claim. Tests mutate their dedicated notebook; they are not production probes. The browser helper adds/edits a real note, checks two-tab stale behavior, searches, and saves desktop/mobile evidence outside the repository. + +On 2 October 2026: a clean locked install, five units, ten full Cloud cases plus the final added input-shape case, browser workflow, fresh migration up/down/up and cached-model process replacement passed. Tested FastAPI 0.142.2, SQLModel 0.0.47, SQLAlchemy 2.0.54, psycopg 3.3.6, pgvector-python 0.5.0, Sentence Transformers 6.1.0, Transformers 5.18.0, PyTorch 2.14.1+cpu, PostgreSQL 18.6 and pgvector extension 0.8.6. The reading fixtures ranked ahead of the unrelated garden fixture; these observations establish fixture behavior, not general retrieval accuracy. + +For restart verification, stop the original Uvicorn process, wait for it to exit and confirm port 8000 has no listener before starting a replacement from a fresh runtime-only shell. Keep the same model cache and database. Compare saved text, revisions, vector bytes and exact-search results before/after. Cold download/preflight and cached startup are separate observations, not speed benchmarks. + +## 6. Cleanup + +Stop the application. Delete only the dedicated service you created and verify its ID is absent from `list`: + +```bash +clickhousectl cloud postgres delete YOUR_SERVICE_ID --org-id YOUR_ORG_ID +clickhousectl cloud postgres list --org-id YOUR_ORG_ID --json +``` + +If retaining the service but removing this example, restore administrator credentials from the saved receipt in a setup shell first (`source .private/admin.env` with `set -a`), then drop the example schema/roles after stopping all app/test connections. Do not run role cleanup with runtime credentials. Preserve source/evidence locally; keep receipts, CA files and model cache out of Git. + +Sources: [fixed model](https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2/tree/1110a243fdf4706b3f48f1d95db1a4f5529b4d41), [pgvector Python integration](https://github.com/pgvector/pgvector-python), [pgvector distance operators](https://github.com/pgvector/pgvector), [Cloud extensions](https://clickhouse.com/docs/products/managed-postgres/extensions). diff --git a/applications/semantic-notes/alembic.ini b/applications/semantic-notes/alembic.ini new file mode 100644 index 00000000..400d2827 --- /dev/null +++ b/applications/semantic-notes/alembic.ini @@ -0,0 +1,3 @@ +[alembic] +script_location = migrations +prepend_sys_path = . diff --git a/applications/semantic-notes/migrations/env.py b/applications/semantic-notes/migrations/env.py new file mode 100644 index 00000000..540f0726 --- /dev/null +++ b/applications/semantic-notes/migrations/env.py @@ -0,0 +1,10 @@ +from alembic import context + +from semantic_notes.database import make_engine + +engine = make_engine() +with engine.connect() as connection: + context.configure(connection=connection, version_table_schema="semantic_notes") + with context.begin_transaction(): + context.run_migrations() +engine.dispose() diff --git a/applications/semantic-notes/migrations/versions/0001_notes.py b/applications/semantic-notes/migrations/versions/0001_notes.py new file mode 100644 index 00000000..1fcc11cd --- /dev/null +++ b/applications/semantic-notes/migrations/versions/0001_notes.py @@ -0,0 +1,44 @@ +from alembic import op + +revision = "0001_notes" +down_revision = None + + +def upgrade(): + op.execute( + """ + CREATE TABLE semantic_notes.collection ( + id integer PRIMARY KEY CHECK (id = 1), + spec jsonb NOT NULL CHECK (spec = '__SPEC__'::jsonb) + ); + INSERT INTO semantic_notes.collection VALUES (1, '__SPEC__'::jsonb); + CREATE TABLE semantic_notes.notes ( + id uuid PRIMARY KEY, + collection_id integer NOT NULL DEFAULT 1 CHECK (collection_id = 1) REFERENCES semantic_notes.collection(id), + title text NOT NULL CHECK (length(btrim(title)) BETWEEN 1 AND 120), + body text NOT NULL CHECK (length(btrim(body)) BETWEEN 1 AND 4096), + embedding public.vector(384) NOT NULL CHECK (public.vector_norm(embedding) > 0), + revision integer NOT NULL DEFAULT 1 CHECK (revision > 0), + created_at timestamptz NOT NULL DEFAULT now(), + updated_at timestamptz NOT NULL DEFAULT now() + ); + CREATE FUNCTION semantic_notes.note_quota() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + PERFORM 1 FROM semantic_notes.collection WHERE id = 1 FOR UPDATE; + IF (SELECT count(*) FROM semantic_notes.notes) >= 100 THEN + RAISE EXCEPTION 'Notebook supports at most 100 notes' USING ERRCODE = '23514'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER note_quota BEFORE INSERT ON semantic_notes.notes FOR EACH ROW EXECUTE FUNCTION semantic_notes.note_quota(); + """.replace( + "__SPEC__", + '{"dimensions": 384, "document_format": "title + two newlines + body", "dtype": "float32", "max_tokens_including_special": 256, "model_id": "sentence-transformers/all-MiniLM-L6-v2", "normalize_embeddings": true, "revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41"}', + ) + ) + + +def downgrade(): + op.execute( + "DROP TABLE semantic_notes.notes; DROP FUNCTION semantic_notes.note_quota(); DROP TABLE semantic_notes.collection;" + ) diff --git a/applications/semantic-notes/model-spec.json b/applications/semantic-notes/model-spec.json new file mode 100644 index 00000000..7503a928 --- /dev/null +++ b/applications/semantic-notes/model-spec.json @@ -0,0 +1,9 @@ +{ + "model_id": "sentence-transformers/all-MiniLM-L6-v2", + "revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", + "dimensions": 384, + "max_tokens_including_special": 256, + "normalize_embeddings": true, + "dtype": "float32", + "document_format": "title + two newlines + body" +} diff --git a/applications/semantic-notes/pyproject.toml b/applications/semantic-notes/pyproject.toml new file mode 100644 index 00000000..28b65af8 --- /dev/null +++ b/applications/semantic-notes/pyproject.toml @@ -0,0 +1,5 @@ +[tool.ruff] +line-length = 88 + +[tool.ruff.lint] +select = ["E4", "E7", "E9", "F", "I"] diff --git a/applications/semantic-notes/requirements-dev.txt b/applications/semantic-notes/requirements-dev.txt new file mode 100644 index 00000000..1cd2c50e --- /dev/null +++ b/applications/semantic-notes/requirements-dev.txt @@ -0,0 +1,4 @@ +-r requirements.txt +playwright==1.56.0 +pyee==13.0.1 +ruff==0.16.10 diff --git a/applications/semantic-notes/requirements.in b/applications/semantic-notes/requirements.in new file mode 100644 index 00000000..e60a5171 --- /dev/null +++ b/applications/semantic-notes/requirements.in @@ -0,0 +1,14 @@ +--extra-index-url https://download.pytorch.org/whl/cpu +fastapi==0.142.2 +uvicorn==0.54.0 +sqlmodel==0.0.47 +SQLAlchemy==2.0.54 +psycopg[binary]==3.3.6 +pgvector==0.5.0 +alembic==1.20.0 +sentence-transformers==6.1.0 +transformers==5.18.0 +torch==2.14.1+cpu +numpy==2.5.3 +Jinja2==3.1.6 +python-multipart==0.0.32 diff --git a/applications/semantic-notes/requirements.txt b/applications/semantic-notes/requirements.txt new file mode 100644 index 00000000..6fae5b73 --- /dev/null +++ b/applications/semantic-notes/requirements.txt @@ -0,0 +1,59 @@ +--extra-index-url https://download.pytorch.org/whl/cpu +alembic==1.20.0 +annotated-doc==0.0.5 +annotated-types==0.8.0 +anyio==4.15.1 +certifi==2026.7.22 +click==8.5.0 +cloudpickle==3.1.2 +fastapi==0.142.2 +filelock==4.0.9 +fsspec==2026.9.0 +greenlet==3.5.6 +h11==0.16.0 +hf-xet==1.6.0 +httpcore==1.0.9 +httpx==0.28.1 +huggingface_hub==1.33.0 +idna==3.20 +Jinja2==3.1.6 +joblib==1.6.0 +Mako==1.4.3 +markdown-it-py==4.2.0 +MarkupSafe==3.0.3 +mdurl==0.1.2 +mpmath==1.3.0 +narwhals==2.26.0 +networkx==3.7 +numpy==2.5.3 +opentelemetry-api==1.45.0 +packaging==26.3 +pgvector==0.5.0 +psycopg==3.3.6 +psycopg-binary==3.3.6 +pydantic==2.13.5 +pydantic_core==2.46.5 +Pygments==2.21.0 +python-multipart==0.0.32 +PyYAML==6.0.3 +regex==2026.9.29 +rich==15.0.0 +safetensors==0.8.0 +scikit-learn==1.9.1 +scipy==1.18.1 +sentence-transformers==6.1.0 +setuptools==84.0.0 +shellingham==1.5.4 +SQLAlchemy==2.0.54 +sqlmodel==0.0.47 +starlette==1.7.0 +sympy==1.14.0 +threadpoolctl==3.7.0 +tokenizers==0.23.2 +torch==2.14.1+cpu +tqdm==4.70.1 +transformers==5.18.0 +typer==0.27.2 +typing-inspection==0.4.4 +typing_extensions==4.16.0 +uvicorn==0.54.0 diff --git a/applications/semantic-notes/samples/notes.json b/applications/semantic-notes/samples/notes.json new file mode 100644 index 00000000..27d710d3 --- /dev/null +++ b/applications/semantic-notes/samples/notes.json @@ -0,0 +1,5 @@ +[ + {"title": "Reading habit", "body": "Read a book for twenty minutes every evening and write a short note about what you learned."}, + {"title": "Remembering books", "body": "Keep a daily reading journal with a few sentences summarizing the ideas in each chapter."}, + {"title": "Garden watering", "body": "Water tomato seedlings early in the morning and check whether the soil is dry before adding more water."} +] diff --git a/applications/semantic-notes/scripts/bootstrap.sql b/applications/semantic-notes/scripts/bootstrap.sql new file mode 100644 index 00000000..25c14fd8 --- /dev/null +++ b/applications/semantic-notes/scripts/bootstrap.sql @@ -0,0 +1,14 @@ +-- Administrator only, on a dedicated service. Install pgvector before owner migrations. +\getenv owner_password PG_MIGRATION_PASSWORD +\getenv app_password PG_APP_PASSWORD +CREATE EXTENSION IF NOT EXISTS vector; +CREATE ROLE semantic_owner LOGIN PASSWORD :'owner_password'; +CREATE ROLE semantic_app LOGIN PASSWORD :'app_password'; +CREATE SCHEMA semantic_notes AUTHORIZATION semantic_owner; +REVOKE ALL ON SCHEMA semantic_notes FROM PUBLIC; +REVOKE CREATE ON SCHEMA public FROM PUBLIC; +REVOKE CREATE, TEMP ON DATABASE postgres FROM PUBLIC; +GRANT CONNECT ON DATABASE postgres TO semantic_owner, semantic_app; +ALTER ROLE semantic_owner SET search_path TO semantic_notes, public; +ALTER ROLE semantic_app SET search_path TO semantic_notes, public; +SELECT extversion AS pgvector_version FROM pg_extension WHERE extname = 'vector'; diff --git a/applications/semantic-notes/scripts/grants.sql b/applications/semantic-notes/scripts/grants.sql new file mode 100644 index 00000000..3344fda0 --- /dev/null +++ b/applications/semantic-notes/scripts/grants.sql @@ -0,0 +1,7 @@ +-- Schema owner only, after migration. +GRANT USAGE ON SCHEMA semantic_notes TO semantic_app; +GRANT SELECT ON semantic_notes.collection, semantic_notes.notes TO semantic_app; +-- FOR UPDATE needs UPDATE privilege; the singleton id is fixed to 1 by CHECK. +GRANT UPDATE (id) ON semantic_notes.collection TO semantic_app; +GRANT INSERT ON semantic_notes.notes TO semantic_app; +GRANT UPDATE (title, body, embedding, revision, updated_at) ON semantic_notes.notes TO semantic_app; diff --git a/applications/semantic-notes/scripts/model_preflight.py b/applications/semantic-notes/scripts/model_preflight.py new file mode 100644 index 00000000..92c0d7bc --- /dev/null +++ b/applications/semantic-notes/scripts/model_preflight.py @@ -0,0 +1,72 @@ +"""Download the exact public CPU model once; no database is needed.""" + +import argparse +import json +import time +from importlib.metadata import version + +import numpy as np + +from semantic_notes.embeddings import MODEL_ID, MODEL_REVISION, Encoder, TokenLimitError + +parser = argparse.ArgumentParser() +parser.add_argument("--download", action="store_true") +args = parser.parse_args() +start = time.monotonic() +encoder = Encoder(download=args.download) +texts = [ + "Postgres row locks protect concurrent updates to the same record.", + "Locking a database row coordinates competing edits.", + "The garden needs watering before the tomatoes ripen.", +] +vectors = [encoder.embed(text) for text in texts] +assert all( + v.shape == (384,) and np.isfinite(v).all() and np.linalg.norm(v) > 0 + for v in vectors +) +related = float(np.dot(vectors[0], vectors[1])) +unrelated = float(np.dot(vectors[0], vectors[2])) +assert related > unrelated +assert encoder.token_count("hello " * 254) == 256 +encoder.embed("hello " * 254) +assert encoder.token_count("hello " * 255) == 257 +try: + encoder.embed("hello " * 255) +except TokenLimitError: + pass +else: + raise AssertionError("257-token input must be rejected before silent truncation") +print( + json.dumps( + { + "model": MODEL_ID, + "revision": MODEL_REVISION, + "cpu": str(encoder.model.device), + "safetensors": True, + "remote_custom_code": False, + "dimensions": 384, + "finite_nonzero": True, + "inclusive_token_boundary": [256, 257], + "related_fixture_cosine": related, + "unrelated_fixture_cosine": unrelated, + "mode": "cold-download" if args.download else "cached-only", + "elapsed_seconds_observed_not_benchmark": round( + time.monotonic() - start, 3 + ), + "versions": { + p: version(p) + for p in [ + "torch", + "sentence-transformers", + "transformers", + "sqlmodel", + "SQLAlchemy", + "fastapi", + "psycopg", + "pgvector", + ] + }, + }, + indent=2, + ) +) diff --git a/applications/semantic-notes/scripts/seed.py b/applications/semantic-notes/scripts/seed.py new file mode 100644 index 00000000..0f21dd0a --- /dev/null +++ b/applications/semantic-notes/scripts/seed.py @@ -0,0 +1,29 @@ +"""Add synthetic sample notes through the actual CPU encoder and restricted role.""" + +import json +from pathlib import Path + +from semantic_notes.database import make_engine +from semantic_notes.embeddings import Encoder +from semantic_notes.service import create_note, verify_collection +from semantic_notes.validation import NoteInput + + +def main(): + encoder = Encoder() + engine = make_engine() + try: + verify_collection(engine) + for raw in json.loads( + (Path(__file__).resolve().parent.parent / "samples/notes.json").read_text() + ): + fields = NoteInput.model_validate(raw) + vector = encoder.embed(encoder.document_text(fields.title, fields.body)) + note = create_note(engine, fields, vector) + print(note["id"], note["title"]) + finally: + engine.dispose() + + +if __name__ == "__main__": + main() diff --git a/applications/semantic-notes/semantic_notes/__init__.py b/applications/semantic-notes/semantic_notes/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/applications/semantic-notes/semantic_notes/app.py b/applications/semantic-notes/semantic_notes/app.py new file mode 100644 index 00000000..5ee3e2d2 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/app.py @@ -0,0 +1,260 @@ +import os +from contextlib import asynccontextmanager +from pathlib import Path +from uuid import UUID + +import anyio +from fastapi import FastAPI, Request +from fastapi.exceptions import RequestValidationError +from fastapi.responses import JSONResponse, RedirectResponse +from fastapi.staticfiles import StaticFiles +from fastapi.templating import Jinja2Templates +from pydantic import ValidationError +from sqlalchemy.exc import SQLAlchemyError + +from .database import make_engine +from .embeddings import Encoder, TokenLimitError +from .service import ( + DomainError, + create_note, + list_notes, + read_note, + search_notes, + update_note, + verify_collection, +) +from .validation import NoteInput, SearchInput, UpdateInput +from .worker import EmbeddingWorker + +ROOT = Path(__file__).resolve().parent.parent + + +class BrowserBoundary: + def __init__(self, app): + self.app = app + self.origin = os.environ.get("APP_ORIGIN", "http://127.0.0.1:8000") + if self.origin != "http://127.0.0.1:8000": + raise ValueError( + "This example uses the fixed loopback origin http://127.0.0.1:8000." + ) + + async def __call__(self, scope, receive, send): + if scope["type"] != "http" or scope["method"] in {"GET", "HEAD", "OPTIONS"}: + return await self.app(scope, receive, send) + origins = [ + value.decode("latin1") + for name, value in scope["headers"] + if name == b"origin" + ] + if origins != [self.origin]: + return await JSONResponse( + {"detail": "A request from the notebook origin is required."}, 403 + )(scope, receive, send) + body = bytearray() + while True: + message = await receive() + if message["type"] == "http.disconnect": + return + body.extend(message.get("body", b"")) + if len(body) > 16384: + return await JSONResponse({"detail": "Request exceeds 16 KiB."}, 413)( + scope, receive, send + ) + if not message.get("more_body", False): + break + delivered = False + + async def bounded_receive(): + nonlocal delivered + if not delivered: + delivered = True + return {"type": "http.request", "body": bytes(body), "more_body": False} + return await receive() + + await self.app(scope, bounded_receive, send) + + +@asynccontextmanager +async def lifespan(app): + engine = make_engine() + try: + encoder = await anyio.to_thread.run_sync(Encoder) + await anyio.to_thread.run_sync(verify_collection, engine) + app.state.engine = engine + app.state.worker = EmbeddingWorker(encoder) + yield + finally: + engine.dispose() + + +app = FastAPI(title="Semantic notes", lifespan=lifespan) +app.add_middleware(BrowserBoundary) +app.mount("/static", StaticFiles(directory=ROOT / "static"), name="static") +templates = Jinja2Templates(directory=ROOT / "templates") + + +@app.exception_handler(RequestValidationError) +async def invalid_request(request, error): + # Do not echo rejected input: lone JSON surrogates cannot be UTF-8 encoded. + return JSONResponse( + {"detail": "Invalid input. Check field types, bounds, and Unicode."}, 422 + ) + + +@app.exception_handler(DomainError) +async def domain_error(request, error): + return JSONResponse({"detail": str(error)}, error.status) + + +@app.exception_handler(TokenLimitError) +async def token_error(request, error): + return JSONResponse({"detail": str(error)}, 422) + + +@app.exception_handler(SQLAlchemyError) +async def database_error(request, error): + # Do not query an unavailable database to construct an error response. + return JSONResponse( + { + "detail": "The notebook could not save or read this request. Try again shortly." + }, + 503, + ) + + +async def encode_document(fields): + return await app.state.worker.embed( + Encoder.document_text(fields.title, fields.body) + ) + + +@app.get("/api/notes") +async def api_list(page: int = 1): + return await anyio.to_thread.run_sync(list_notes, app.state.engine, page) + + +@app.get("/api/notes/{note_id}") +async def api_read(note_id: UUID): + return await anyio.to_thread.run_sync(read_note, app.state.engine, note_id) + + +@app.post("/api/notes", status_code=201) +async def api_create(fields: NoteInput): + vector = await encode_document(fields) + return await anyio.to_thread.run_sync(create_note, app.state.engine, fields, vector) + + +@app.put("/api/notes/{note_id}") +async def api_update(note_id: UUID, fields: UpdateInput): + vector = await encode_document(fields) + return await anyio.to_thread.run_sync( + update_note, app.state.engine, note_id, fields, vector + ) + + +@app.post("/api/search") +async def api_search(fields: SearchInput): + vector = await app.state.worker.embed(fields.q) + results = await anyio.to_thread.run_sync( + search_notes, app.state.engine, fields, vector + ) + return {"results": results, "k": fields.k, "query": fields.q} + + +@app.get("/") +async def home(request: Request, page: int = 1): + board = await anyio.to_thread.run_sync(list_notes, app.state.engine, page) + return templates.TemplateResponse(request, "index.html", {"board": board}) + + +@app.get("/notes/{note_id}") +async def edit(request: Request, note_id: UUID): + note = await anyio.to_thread.run_sync(read_note, app.state.engine, note_id) + return templates.TemplateResponse(request, "edit.html", {"note": note, "error": ""}) + + +async def form_fields(request, model): + form = await request.form(max_fields=5, max_files=0) + values = dict(form) + if len(form.multi_items()) != len(values): + raise DomainError("Duplicate form fields are not allowed.") + for key in ("revision", "k"): + if key in values: + if ( + not isinstance(values[key], str) + or not values[key].isascii() + or not values[key].isdigit() + or len(values[key]) > (10 if key == "revision" else 2) + ): + raise DomainError(f"{key} must be an integer.") + values[key] = int(values[key]) + return model.model_validate(values) + + +@app.post("/notes") +async def add(request: Request): + try: + fields = await form_fields(request, NoteInput) + vector = await encode_document(fields) + note = await anyio.to_thread.run_sync( + create_note, app.state.engine, fields, vector + ) + except (ValidationError, DomainError, TokenLimitError) as error: + return templates.TemplateResponse( + request, + "error.html", + {"error": str(error)}, + status_code=getattr(error, "status", 422), + ) + return RedirectResponse(f"/notes/{note['id']}?saved=1", status_code=303) + + +@app.post("/notes/{note_id}") +async def save(request: Request, note_id: UUID): + form = await request.form(max_fields=5, max_files=0) + values = dict(form) + try: + revision = values.get("revision", "") + if len(form.multi_items()) != len(values): + raise DomainError("Duplicate form fields are not allowed.") + if ( + not isinstance(revision, str) + or not revision.isascii() + or not revision.isdigit() + or len(revision) > 10 + ): + raise DomainError("Revision must be an integer.") + fields = UpdateInput.model_validate({**values, "revision": int(revision)}) + vector = await encode_document(fields) + await anyio.to_thread.run_sync( + update_note, app.state.engine, note_id, fields, vector + ) + except (ValidationError, DomainError, TokenLimitError) as error: + # Keep the submitted revision and text. Only an explicit reload fetches new values. + return templates.TemplateResponse( + request, + "edit.html", + {"note": {**values, "id": str(note_id)}, "error": str(error)}, + status_code=getattr(error, "status", 422), + ) + return RedirectResponse(f"/notes/{note_id}?saved=1", status_code=303) + + +@app.post("/search") +async def search(request: Request): + try: + fields = await form_fields(request, SearchInput) + vector = await app.state.worker.embed(fields.q) + results = await anyio.to_thread.run_sync( + search_notes, app.state.engine, fields, vector + ) + except (ValidationError, DomainError, TokenLimitError) as error: + return templates.TemplateResponse( + request, + "error.html", + {"error": str(error)}, + status_code=getattr(error, "status", 422), + ) + return templates.TemplateResponse( + request, "search.html", {"fields": fields, "results": results} + ) diff --git a/applications/semantic-notes/semantic_notes/database.py b/applications/semantic-notes/semantic_notes/database.py new file mode 100644 index 00000000..4c99e3a1 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/database.py @@ -0,0 +1,43 @@ +import os +from pathlib import Path + +from pgvector.psycopg import register_vector +from sqlalchemy import URL, event +from sqlmodel import create_engine + + +def make_engine(): + for key in ["PGHOST", "PGUSER", "PGPASSWORD", "PGDATABASE", "PGSSLROOTCERT"]: + if not os.getenv(key): + raise ValueError(f"{key} is required.") + certificate = Path(os.environ["PGSSLROOTCERT"]) + if not certificate.is_file(): + raise ValueError("The Cloud CA file must exist.") + url = URL.create( + "postgresql+psycopg", + username=os.environ["PGUSER"], + password=os.environ["PGPASSWORD"], + host=os.environ["PGHOST"], + port=int(os.getenv("PGPORT", "5432")), + database=os.environ["PGDATABASE"], + ) + engine = create_engine( + url, + pool_size=4, + max_overflow=0, + pool_timeout=5, + pool_pre_ping=True, + connect_args={ + "sslmode": "verify-full", + "sslrootcert": str(certificate), + "connect_timeout": 10, + "options": "-csearch_path=semantic_notes,public -cstatement_timeout=15000", + "application_name": "semantic-notes", + }, + ) + + @event.listens_for(engine, "connect") + def register(dbapi_connection, _connection_record): + register_vector(dbapi_connection) + + return engine diff --git a/applications/semantic-notes/semantic_notes/embeddings.py b/applications/semantic-notes/semantic_notes/embeddings.py new file mode 100644 index 00000000..2c2e78b7 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/embeddings.py @@ -0,0 +1,107 @@ +"""Fixed CPU encoder. Model/tokenizer work happens before database acquisition.""" + +from __future__ import annotations + +import os +import threading +from pathlib import Path + +import numpy as np +import torch +from huggingface_hub import snapshot_download +from sentence_transformers import SentenceTransformer + +from .spec import DIMENSIONS, MAX_TOKENS, MODEL_ID, MODEL_REVISION, check_source_spec + +MODEL_FILES = [ + "modules.json", + "config.json", + "config_sentence_transformers.json", + "sentence_bert_config.json", + "special_tokens_map.json", + "tokenizer.json", + "tokenizer_config.json", + "vocab.txt", + "1_Pooling/config.json", + "model.safetensors", +] + + +class TokenLimitError(ValueError): + pass + + +def checked_vector(value: object) -> np.ndarray: + vector = np.asarray(value, dtype=np.float32) + if vector.shape != (DIMENSIONS,) or not np.isfinite(vector).all(): + raise ValueError("Embedding must have exactly 384 finite dimensions.") + norm = float(np.linalg.norm(vector)) + if not np.isfinite(norm) or norm <= 0: + raise ValueError("Embedding must be nonzero with a finite norm.") + return vector + + +class Encoder: + def __init__(self, *, download: bool = False): + check_source_spec() + if ( + os.getenv("MODEL_ID", MODEL_ID) != MODEL_ID + or os.getenv("MODEL_REVISION", MODEL_REVISION) != MODEL_REVISION + ): + raise ValueError( + "This collection requires its fixed model ID and revision." + ) + torch.set_num_threads(1) + self._lock = threading.Lock() + self.path = snapshot_download( + MODEL_ID, + revision=MODEL_REVISION, + token=False, + cache_dir=os.getenv("MODEL_CACHE_DIR"), + allow_patterns=MODEL_FILES, + local_files_only=not download, + ) + self.model = SentenceTransformer( + self.path, + device="cpu", + trust_remote_code=False, + local_files_only=True, + model_kwargs={"use_safetensors": True}, + ) + if ( + self.model.get_embedding_dimension() != DIMENSIONS + or self.model.max_seq_length != MAX_TOKENS + ): + raise ValueError( + "Cached model shape/token specification does not match the collection." + ) + if not (Path(self.path) / "model.safetensors").is_file(): + raise ValueError("Pinned safetensors weights are required.") + + def token_count(self, text: str) -> int: + return len( + self.model.tokenizer(text, add_special_tokens=True, truncation=False)[ + "input_ids" + ] + ) + + def embed(self, text: str) -> np.ndarray: + # SentenceTransformer.encode changes module/device state; serialize shared-model access. + # HTTP admission is separately bounded and rejects excess work instead of queuing jobs. + with self._lock: + count = self.token_count(text) + if count > MAX_TOKENS: + raise TokenLimitError( + f"Input contains {count} tokens; limit is 256 including special tokens." + ) + value = self.model.encode( + text, + convert_to_numpy=True, + normalize_embeddings=True, + show_progress_bar=False, + ) + return checked_vector(value) + + @staticmethod + def document_text(title: str, body: str) -> str: + return title + "\n\n" + body diff --git a/applications/semantic-notes/semantic_notes/models.py b/applications/semantic-notes/semantic_notes/models.py new file mode 100644 index 00000000..3cf95b9f --- /dev/null +++ b/applications/semantic-notes/semantic_notes/models.py @@ -0,0 +1,33 @@ +from datetime import datetime, timezone +from uuid import UUID, uuid4 + +from pgvector.sqlalchemy import VECTOR +from sqlalchemy import Column, DateTime +from sqlalchemy.dialects.postgresql import JSONB +from sqlmodel import Field, SQLModel + + +class Collection(SQLModel, table=True): + __tablename__ = "collection" + __table_args__ = {"schema": "semantic_notes"} + id: int = Field(primary_key=True) + spec: dict = Field(sa_column=Column(JSONB, nullable=False)) + + +class Note(SQLModel, table=True): + __tablename__ = "notes" + __table_args__ = {"schema": "semantic_notes"} + id: UUID = Field(default_factory=uuid4, primary_key=True) + collection_id: int = Field(default=1, foreign_key="semantic_notes.collection.id") + title: str + body: str + embedding: list[float] = Field(sa_type=VECTOR(384)) + revision: int = Field(default=1) + created_at: datetime = Field( + default_factory=lambda: datetime.now(timezone.utc), + sa_column=Column(DateTime(timezone=True), nullable=False), + ) + updated_at: datetime = Field( + default_factory=lambda: datetime.now(timezone.utc), + sa_column=Column(DateTime(timezone=True), nullable=False), + ) diff --git a/applications/semantic-notes/semantic_notes/service.py b/applications/semantic-notes/semantic_notes/service.py new file mode 100644 index 00000000..1351cf36 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/service.py @@ -0,0 +1,125 @@ +from datetime import datetime, timezone +from uuid import UUID + +from sqlalchemy import func +from sqlmodel import Session, select + +from .embeddings import checked_vector +from .models import Collection, Note +from .spec import SPEC + + +class DomainError(Exception): + def __init__(self, message, status=422): + super().__init__(message) + self.status = status + + +def verify_collection(engine): + with Session(engine) as session: + collection = session.get(Collection, 1) + if collection is None or collection.spec != SPEC: + raise ValueError("Database collection and full model specification differ.") + + +def view(note): + return { + "id": str(note.id), + "title": note.title, + "body": note.body, + "revision": note.revision, + "created_at": note.created_at.isoformat(), + "updated_at": note.updated_at.isoformat(), + } + + +def create_note(engine, fields, vector): + vector = checked_vector(vector) + with Session(engine) as session, session.begin(): + collection = session.exec( + select(Collection).where(Collection.id == 1).with_for_update() + ).one() + if collection.spec != SPEC: + raise DomainError("Collection specification mismatch.", 503) + count = session.exec(select(func.count()).select_from(Note)).one() + if count >= 100: + raise DomainError("This notebook holds at most 100 notes.", 409) + note = Note(title=fields.title, body=fields.body, embedding=vector.tolist()) + session.add(note) + session.flush() + result = view(note) + return result + + +def update_note(engine, note_id: UUID, fields, vector): + vector = checked_vector(vector) + with Session(engine) as session, session.begin(): + note = session.exec( + select(Note).where(Note.id == note_id).with_for_update() + ).first() + if note is None: + raise DomainError("Note not found.", 404) + if note.revision != fields.revision: + raise DomainError( + "This note changed while you were editing. Reload before saving.", 409 + ) + note.title, note.body, note.embedding = ( + fields.title, + fields.body, + vector.tolist(), + ) + note.revision += 1 + note.updated_at = datetime.now(timezone.utc) + session.add(note) + session.flush() + result = view(note) + return result + + +def read_note(engine, note_id): + with Session(engine) as session: + note = session.get(Note, note_id) + if note is None: + raise DomainError("Note not found.", 404) + return view(note) + + +def list_notes(engine, page=1): + if not isinstance(page, int) or not 1 <= page <= 10: + raise DomainError("Page must be 1–10.") + with Session(engine) as session, session.begin(): + session.connection().exec_driver_sql( + "SET TRANSACTION ISOLATION LEVEL REPEATABLE READ READ ONLY" + ) + total = session.exec(select(func.count()).select_from(Note)).one() + notes = session.exec( + select(Note) + .order_by(Note.updated_at.desc(), Note.id) + .limit(12) + .offset((page - 1) * 12) + ).all() + return { + "notes": [view(n) for n in notes], + "total": total, + "page": page, + "page_size": 12, + } + + +def search_notes(engine, fields, vector): + vector = checked_vector(vector) + distance = Note.embedding.cosine_distance(vector) + statement = select(Note, distance.label("distance")) + if fields.title_filter: + literal = ( + fields.title_filter.replace("\\", "\\\\") + .replace("%", "\\%") + .replace("_", "\\_") + ) + statement = statement.where(Note.title.ilike(f"%{literal}%", escape="\\")) + statement = statement.order_by(distance, Note.id).limit(fields.k) + with Session(engine) as session: + return [ + {**view(note), "distance": float(value), "excerpt": note.body[:240]} + for note, value in session.exec(statement).all() + ] diff --git a/applications/semantic-notes/semantic_notes/spec.py b/applications/semantic-notes/semantic_notes/spec.py new file mode 100644 index 00000000..9b94cb64 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/spec.py @@ -0,0 +1,26 @@ +import json +from pathlib import Path + +MODEL_ID = "sentence-transformers/all-MiniLM-L6-v2" +MODEL_REVISION = "1110a243fdf4706b3f48f1d95db1a4f5529b4d41" +DIMENSIONS = 384 +MAX_TOKENS = 256 +SPEC = { + "model_id": MODEL_ID, + "revision": MODEL_REVISION, + "dimensions": DIMENSIONS, + "max_tokens_including_special": MAX_TOKENS, + "normalize_embeddings": True, + "dtype": "float32", + "document_format": "title + two newlines + body", +} + + +def check_source_spec(): + actual = json.loads( + (Path(__file__).resolve().parent.parent / "model-spec.json").read_text() + ) + if actual != SPEC: + raise ValueError( + "The complete fixed model specification must match this collection." + ) diff --git a/applications/semantic-notes/semantic_notes/validation.py b/applications/semantic-notes/semantic_notes/validation.py new file mode 100644 index 00000000..cdfe6295 --- /dev/null +++ b/applications/semantic-notes/semantic_notes/validation.py @@ -0,0 +1,63 @@ +import unicodedata +from typing import Annotated + +from pydantic import BaseModel, ConfigDict, Field, StringConstraints, field_validator + +Title = Annotated[ + str, + StringConstraints(strict=True, strip_whitespace=True, min_length=1, max_length=120), +] +Body = Annotated[ + str, + StringConstraints( + strict=True, strip_whitespace=True, min_length=1, max_length=4096 + ), +] + + +class NoteInput(BaseModel): + model_config = ConfigDict(extra="forbid") + title: Title + body: Body + + @field_validator("title", "body", mode="before") + @classmethod + def controls(cls, value, info): + if not isinstance(value, str): + return value + allowed = {"\n", "\t"} if info.field_name == "body" else set() + if any( + unicodedata.category(char) in {"Cc", "Cs"} and char not in allowed + for char in value + ): + raise ValueError( + "Controls and invalid Unicode are not allowed (body permits newline/tab)." + ) + return value + + +class UpdateInput(NoteInput): + revision: Annotated[int, Field(strict=True, ge=1, le=2147483646)] + + +class SearchInput(BaseModel): + model_config = ConfigDict(extra="forbid") + q: Annotated[ + str, + StringConstraints( + strict=True, strip_whitespace=True, min_length=1, max_length=512 + ), + ] + k: Annotated[int, Field(strict=True, ge=1, le=10)] = 5 + title_filter: Annotated[ + str, StringConstraints(strict=True, strip_whitespace=True, max_length=80) + ] = "" + + @field_validator("q", "title_filter", mode="before") + @classmethod + def controls(cls, value): + if not isinstance(value, str): + return value + if any(unicodedata.category(c) in {"Cc", "Cs"} for c in value): + raise ValueError("Search cannot contain controls or invalid Unicode.") + return value diff --git a/applications/semantic-notes/semantic_notes/worker.py b/applications/semantic-notes/semantic_notes/worker.py new file mode 100644 index 00000000..f6f78bed --- /dev/null +++ b/applications/semantic-notes/semantic_notes/worker.py @@ -0,0 +1,39 @@ +import asyncio + +import anyio + +from .service import DomainError + + +class EmbeddingWorker: + def __init__(self, encoder): + self.encoder = encoder + self.slots = asyncio.BoundedSemaphore(2) + self.pending = set() + + async def embed(self, text): + if self.slots.locked(): + raise DomainError("The encoder is busy. Please try again shortly.", 429) + await self.slots.acquire() + try: + task = asyncio.create_task( + anyio.to_thread.run_sync( + self.encoder.embed, text, abandon_on_cancel=False + ) + ) + except BaseException: + self.slots.release() + raise + self.pending.add(task) + + def finished(done): + self.pending.discard(done) + self.slots.release() + # Retrieve errors even if the requesting task was cancelled. + if not done.cancelled(): + done.exception() + + task.add_done_callback(finished) + # A directly cancelled HTTP task cannot cancel the inner worker or release + # its permit. The completion callback owns capacity until CPU work ends. + return await asyncio.shield(task) diff --git a/applications/semantic-notes/static/style.css b/applications/semantic-notes/static/style.css new file mode 100644 index 00000000..a3f0589f --- /dev/null +++ b/applications/semantic-notes/static/style.css @@ -0,0 +1 @@ +*{box-sizing:border-box}body{margin:0;background:#f4f1ea;color:#20312b;font:16px/1.55 system-ui,-apple-system,sans-serif}header,main,footer{max-width:1100px;margin:auto;padding:24px}header{display:flex;align-items:center;justify-content:space-between;border-bottom:1px solid #d8d9d0}.brand{font-size:21px;font-weight:750;text-decoration:none}header span,.eyebrow{font-size:11px;letter-spacing:.15em;font-weight:750;color:#657466}a{color:#245847}h1,h2,h3,p{margin-top:0}h1{font-size:clamp(34px,5vw,58px);line-height:1.1;letter-spacing:-.035em;margin-bottom:20px}h2{font-size:21px;line-height:1.3}h3{font-size:18px;margin-bottom:6px}.intro{padding:36px 0 26px;max-width:650px}.intro>p:last-child{color:#637066;font-size:18px}.compact h1{font-size:36px}.grid{display:grid;grid-template-columns:1fr 1fr;gap:24px}.panel{background:#fffdf8;border:1px solid #d9ded3;padding:28px;border-radius:16px}label{display:block;font-weight:650;font-size:13px;margin:15px 0 6px}input,textarea,select{width:100%;font:inherit;color:inherit;border:1px solid #b9c5b9;border-radius:7px;padding:10px;background:white}textarea{resize:vertical}input:focus,textarea:focus,select:focus{outline:3px solid #cce3ca;outline-offset:1px}button{background:#244f3e;color:white;border:0;border-radius:7px;padding:12px 18px;font:inherit;font-weight:650;cursor:pointer;margin-top:8px}.row{display:grid;grid-template-columns:1fr 95px;gap:14px}.hint{font-size:12px;color:#637066;margin:12px 0}.notes{margin-top:36px}.section-head{display:flex;align-items:center;justify-content:space-between}.section-head span{color:#637066}.note{display:flex;gap:20px;justify-content:space-between;padding:22px 0;border-top:1px solid #d4d9d0}.note p{max-width:740px;margin:0;color:#647168;overflow-wrap:anywhere}.note h2,.note h3{overflow-wrap:anywhere}.note a{text-decoration:none}.pill{background:#e4ecdf;color:#496149;border-radius:20px;padding:5px 11px;font-size:11px;white-space:nowrap;align-self:start}.editor{max-width:760px;margin:24px auto}.editor h1{font-size:38px}.back{display:inline-block;margin:12px 0}.alert,.success{border-radius:8px;padding:15px;margin-bottom:20px;overflow-wrap:anywhere}.alert{background:#fff0d4;color:#714b14}.success{background:#e3f1df;color:#2e5831}.pages{display:flex;gap:20px;justify-content:center;padding:24px}.empty{padding:24px;color:#637066}footer{font-size:12px;color:#68766a;margin-top:30px} @media(max-width:680px){header,main,footer{padding:18px}.grid{grid-template-columns:1fr}.panel{padding:22px}.intro{padding-top:26px}header span{display:none}.note{display:block}.pill{display:inline-block;margin-top:12px}.row{grid-template-columns:1fr 85px}} diff --git a/applications/semantic-notes/templates/base.html b/applications/semantic-notes/templates/base.html new file mode 100644 index 00000000..d53826e6 --- /dev/null +++ b/applications/semantic-notes/templates/base.html @@ -0,0 +1,3 @@ + +Semantic notes +
◈ Semantic notesLOCAL NOTEBOOK
{% block content %}{% endblock %}
diff --git a/applications/semantic-notes/templates/edit.html b/applications/semantic-notes/templates/edit.html new file mode 100644 index 00000000..2300e0a5 --- /dev/null +++ b/applications/semantic-notes/templates/edit.html @@ -0,0 +1 @@ +{% extends "base.html" %}{% block content %}← Your notebook

KEEP THE IDEA CURRENT

Edit note

{% if error %}{% elif request.query_params.get('saved') %}
Note saved.
{% endif %}

Revision {{note.revision}} · 256 tokens maximum including title and special tokens.

{% endblock %} diff --git a/applications/semantic-notes/templates/error.html b/applications/semantic-notes/templates/error.html new file mode 100644 index 00000000..e9ee47e7 --- /dev/null +++ b/applications/semantic-notes/templates/error.html @@ -0,0 +1 @@ +{% extends "base.html" %}{% block content %}

Could not complete this request

Return to your notebook
{% endblock %} diff --git a/applications/semantic-notes/templates/index.html b/applications/semantic-notes/templates/index.html new file mode 100644 index 00000000..6ca21d36 --- /dev/null +++ b/applications/semantic-notes/templates/index.html @@ -0,0 +1,5 @@ +{% extends "base.html" %}{% block content %} +

YOUR IDEAS, WITH CONTEXT

Find the thought.
Keep the words.

Save short notes, then find them by meaning—even when you use different words.

+

Search your notebook

Returns matching notes, with cosine distance. It does not generate answers.

+

Add a note

Keep it short: title and note together must fit 256 model tokens, including special tokens.

+

Your notes

{{board.total}} / 100
{% for note in board.notes %}

{{note.title}}

{{note.body[:180]}}{% if note.body|length>180 %}…{% endif %}

Revision {{note.revision}}
{% else %}

Your first note starts here.

{% endfor %}
{% endblock %} diff --git a/applications/semantic-notes/templates/search.html b/applications/semantic-notes/templates/search.html new file mode 100644 index 00000000..5d2afb55 --- /dev/null +++ b/applications/semantic-notes/templates/search.html @@ -0,0 +1 @@ +{% extends "base.html" %}{% block content %}← Your notebook

SEARCH BY MEANING

“{{fields.q}}”

{{results|length}} result(s){% if fields.title_filter %}, titles containing “{{fields.title_filter}}”{% endif %}. Lower cosine distance means closer vectors.

{% for note in results %}

{{note.title}}

{{note.excerpt}}

Distance {{'%.4f'|format(note.distance)}}
{% else %}

No notes match this title filter.

{% endfor %}
{% endblock %} diff --git a/applications/semantic-notes/tests/browser_acceptance.py b/applications/semantic-notes/tests/browser_acceptance.py new file mode 100644 index 00000000..668bb661 --- /dev/null +++ b/applications/semantic-notes/tests/browser_acceptance.py @@ -0,0 +1,125 @@ +"""Actual form workflow; evidence stays outside the source checkout.""" + +import asyncio +import json +import os +from pathlib import Path + +from playwright.async_api import async_playwright, expect + +ORIGIN = "http://127.0.0.1:8000" +ARTIFACTS = Path(os.getenv("EVIDENCE_DIR", "/tmp/semantic-notes-evidence")) +ARTIFACTS.mkdir(parents=True, exist_ok=True) + + +async def main(): + async with async_playwright() as playwright: + browser = await playwright.chromium.launch() + context = await browser.new_context(viewport={"width": 1440, "height": 1100}) + other = await browser.new_context(viewport={"width": 1280, "height": 900}) + page, stale = await context.new_page(), await other.new_page() + errors = [] + page.on("pageerror", lambda error: errors.append(str(error))) + await page.goto(ORIGIN) + await expect( + page.get_by_role("heading", name="Find the thought.") + ).to_be_visible(timeout=30000) + await page.get_by_label("Title", exact=True).fill("Project memory") + await page.get_by_label("Note", exact=True).fill( + "Keep a brief record of decisions after each project meeting so the team can revisit the reasons." + ) + await page.get_by_role("button", name="Save note", exact=True).click() + await expect(page.get_by_role("status")).to_have_text( + "Note saved.", timeout=30000 + ) + note_url = page.url.split("?")[0] + await stale.goto(note_url) + await expect(stale.get_by_label("Title", exact=True)).to_have_value( + "Project memory", timeout=30000 + ) + await page.get_by_label("Note", exact=True).fill( + "Write the decision and its reasoning immediately after each meeting. Review this record before making the next choice." + ) + await page.get_by_role("button", name="Save changes").click() + await expect(page.get_by_role("status")).to_have_text( + "Note saved.", timeout=30000 + ) + await stale.get_by_label("Note", exact=True).fill( + "An older draft must not replace the current decision record." + ) + await stale.get_by_role("button", name="Save changes").click() + await expect(stale.get_by_role("alert")).to_contain_text( + "changed while you were editing", timeout=30000 + ) + await expect(stale.get_by_label("Note", exact=True)).to_have_value( + "An older draft must not replace the current decision record." + ) + await stale.get_by_role("link", name="Reload latest note").click() + await expect(stale.get_by_label("Note", exact=True)).to_have_value( + "Write the decision and its reasoning immediately after each meeting. Review this record before making the next choice.", + timeout=30000, + ) + await page.goto(ORIGIN) + await page.get_by_label("What are you looking for?").fill( + "Remember why we made a choice in a meeting" + ) + await page.get_by_label("Title contains (optional)").fill("Project") + await page.get_by_role("button", name="Search by meaning").click() + await expect( + page.get_by_role("heading", name="Project memory", exact=True) + ).to_be_visible(timeout=30000) + await expect(page.locator(".pill").filter(has_text="Distance")).to_be_visible() + await page.screenshot( + path=str(ARTIFACTS / "semantic-search.png"), full_page=True + ) + foreign = await page.request.post( + ORIGIN + "/api/notes", + headers={"Origin": "https://foreign.invalid"}, + data={"title": "Foreign", "body": "Rejected"}, + ) + assert foreign.status == 403 + await page.goto(ORIGIN) + await expect( + page.get_by_role("heading", name="Project memory", exact=True) + ).to_be_visible(timeout=30000) + await page.screenshot( + path=str(ARTIFACTS / "semantic-desktop.png"), full_page=True + ) + mobile = await browser.new_context( + viewport={"width": 390, "height": 844}, + is_mobile=True, + device_scale_factor=1, + ) + phone = await mobile.new_page() + await phone.goto(ORIGIN) + await expect( + phone.get_by_role("heading", name="Find the thought.") + ).to_be_visible(timeout=30000) + assert await phone.evaluate( + "document.documentElement.scrollWidth <= window.innerWidth" + ) + await phone.screenshot( + path=str(ARTIFACTS / "semantic-mobile.png"), full_page=True + ) + assert not errors, errors + report = { + "browser_version": browser.version, + "note_url_path": note_url.removeprefix(ORIGIN), + "checks": [ + "actual add/save", + "actual correction durable after reload", + "two-browser stale form 409 retains input", + "explicit reload latest", + "actual model search with literal title filter", + "foreign-origin 403", + "desktop/mobile viewport", + "no page errors", + ], + } + (ARTIFACTS / "browser-result.json").write_text(json.dumps(report, indent=2)) + print(json.dumps(report, indent=2)) + await browser.close() + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/applications/semantic-notes/tests/cloud_acceptance.py b/applications/semantic-notes/tests/cloud_acceptance.py new file mode 100644 index 00000000..945026bc --- /dev/null +++ b/applications/semantic-notes/tests/cloud_acceptance.py @@ -0,0 +1,487 @@ +"""Real Cloud/model controls. Run only against your dedicated example service.""" + +import concurrent.futures +import os +import socket +import threading +import time +import unittest +from uuid import uuid4 + +import httpx +import numpy as np +import psycopg +from pgvector.psycopg import register_vector + +from semantic_notes.database import make_engine +from semantic_notes.embeddings import Encoder +from semantic_notes.service import DomainError, update_note +from semantic_notes.spec import SPEC +from semantic_notes.validation import UpdateInput + +ORIGIN = "http://127.0.0.1:8000" + + +def connection(*, owner=False, **overrides): + values = { + "host": os.environ["PGHOST"], + "port": os.getenv("PGPORT", "5432"), + "dbname": os.environ["PGDATABASE"], + "user": os.environ["PGUSER"], + "password": os.environ["PGPASSWORD"], + "sslmode": "verify-full", + "sslrootcert": os.environ["PGSSLROOTCERT"], + "connect_timeout": 10, + } + if owner: + values.update( + user=os.environ["TEST_OWNER_USER"], + password=os.environ["TEST_OWNER_PASSWORD"], + ) + values.update(overrides) + conn = psycopg.connect(**values) + register_vector(conn) + return conn + + +def stored(note_id): + with connection() as conn: + return conn.execute( + "SELECT title,body,embedding,revision FROM semantic_notes.notes WHERE id=%s", + (note_id,), + ).fetchone() + + +class CloudAcceptance(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.client = httpx.Client( + base_url=ORIGIN, headers={"Origin": ORIGIN}, timeout=40 + ) + cls.encoder = Encoder() + cls.ids = [] + examples = [ + ( + "Reading habit", + "Read a book for twenty minutes every evening and write a short note about what you learned.", + ), + ( + "Remembering books", + "Keep a daily reading journal with a few sentences summarizing the ideas in each chapter.", + ), + ( + "Garden watering", + "Water tomato seedlings early in the morning and check whether the soil is dry before adding more water.", + ), + ] + for title, body in examples: + response = cls.client.post( + "/api/notes", json={"title": title, "body": body} + ) + if response.status_code != 201: + raise RuntimeError( + f"Fixture creation failed: {response.status_code} {response.text}" + ) + cls.ids.append(response.json()["id"]) + print( + "Fixture setup: 3 actual CPU-embedded notes (not additional test cases)", + flush=True, + ) + + @classmethod + def tearDownClass(cls): + cls.client.close() + + def test_01_exact_cosine_related_fixture_and_independent_comparison(self): + query = "A daily journal helps me remember the books I read" + response = self.client.post("/api/search", json={"q": query, "k": 10}) + self.assertEqual(response.status_code, 200) + results = response.json()["results"] + vector = self.encoder.embed(query).astype(np.float64) + with connection() as conn: + rows = conn.execute( + "SELECT id,embedding FROM semantic_notes.notes" + ).fetchall() + expected = sorted( + [ + ( + str(i), + float( + 1 + - np.dot(v.to_numpy().astype(np.float64), vector) + / ( + np.linalg.norm(v.to_numpy().astype(np.float64)) + * np.linalg.norm(vector) + ) + ), + ) + for i, v in rows + ], + key=lambda pair: (pair[1], pair[0]), + ) + self.assertEqual([r["id"] for r in results], [i for i, d in expected[:10]]) + for observed, (_, distance) in zip(results, expected): + self.assertAlmostEqual(observed["distance"], distance, delta=1e-5) + distances = {r["id"]: r["distance"] for r in results} + self.assertLess(distances[self.ids[0]], distances[self.ids[2]]) + self.assertLess(distances[self.ids[1]], distances[self.ids[2]]) + print( + "Actual fixture ranks/distances:", + [(r["title"], round(r["distance"], 6)) for r in results], + flush=True, + ) + + def test_02_revisioned_edit_and_stale_rejection(self): + note = self.client.get(f"/api/notes/{self.ids[0]}").json() + fields = { + "title": "Reading habit updated", + "body": "Read each evening and keep a short summary in a notebook.", + "revision": note["revision"], + } + response = self.client.put(f"/api/notes/{self.ids[0]}", json=fields) + self.assertEqual(response.status_code, 200) + self.assertEqual(response.json()["revision"], note["revision"] + 1) + self.assertEqual( + self.client.put(f"/api/notes/{self.ids[0]}", json=fields).status_code, 409 + ) + saved = stored(self.ids[0]) + np.testing.assert_allclose( + saved[2].to_numpy(), + self.encoder.embed(Encoder.document_text(fields["title"], fields["body"])), + atol=1e-6, + ) + + def test_03_competing_http_updates_observe_database_waits(self): + note_id = self.ids[1] + note = self.client.get(f"/api/notes/{note_id}").json() + lock = connection(owner=True) + lock.execute( + "SELECT id FROM semantic_notes.notes WHERE id=%s FOR UPDATE", (note_id,) + ) + pool = concurrent.futures.ThreadPoolExecutor(max_workers=2) + + def update(label): + with httpx.Client( + base_url=ORIGIN, headers={"Origin": ORIGIN}, timeout=40 + ) as client: + return client.put( + f"/api/notes/{note_id}", + json={ + "title": label, + "body": "A short reading journal records ideas worth revisiting.", + "revision": note["revision"], + }, + ) + + futures = [ + pool.submit(update, label) + for label in ["Competing reading A", "Competing reading B"] + ] + try: + waited = False + with connection(owner=True) as observer: + for _ in range(100): + blocked = observer.execute( + "SELECT count(*) FROM pg_stat_activity WHERE usename='semantic_app' AND cardinality(pg_blocking_pids(pid)) > 0" + ).fetchone()[0] + if blocked >= 2: + waited = True + break + time.sleep(0.1) + self.assertTrue( + waited, + "two independent HTTP transactions must wait behind the held parent row", + ) + lock.commit() + responses = [f.result(timeout=40) for f in futures] + self.assertEqual(sorted(r.status_code for r in responses), [200, 409]) + self.assertEqual(stored(note_id)[3], note["revision"] + 1) + print( + "Observed two Cloud row-lock waits; competing HTTP statuses 200/409", + flush=True, + ) + finally: + lock.rollback() + lock.close() + pool.shutdown(wait=True) + + def test_04_slow_old_actual_embedding_cannot_overwrite_new_edit(self): + note_id = self.ids[0] + note = self.client.get(f"/api/notes/{note_id}").json() + old = UpdateInput( + title="Slow older reading edit", + body="Record an older summary of a book.", + revision=note["revision"], + ) + engine = make_engine() + entered, proceed = threading.Event(), threading.Event() + + def delayed(): + entered.set() + if not proceed.wait(10): + raise RuntimeError("test gate timed out") + vector = self.encoder.embed(Encoder.document_text(old.title, old.body)) + return update_note(engine, note_id, old, vector) + + pool = concurrent.futures.ThreadPoolExecutor(max_workers=1) + future = pool.submit(delayed) + try: + self.assertTrue(entered.wait(5)) + self.assertEqual(engine.pool.checkedout(), 0) + newer = { + "title": "Current reading summary", + "body": "The newest reading summary is the one to keep.", + "revision": note["revision"], + } + response = self.client.put(f"/api/notes/{note_id}", json=newer) + self.assertEqual(response.status_code, 200) + proceed.set() + with self.assertRaises(DomainError) as conflict: + future.result(timeout=20) + self.assertEqual(conflict.exception.status, 409) + saved = stored(note_id) + self.assertEqual(saved[:2], (newer["title"], newer["body"])) + np.testing.assert_allclose( + saved[2].to_numpy(), + self.encoder.embed( + Encoder.document_text(newer["title"], newer["body"]) + ), + atol=1e-6, + ) + print( + "Coordinated older real CPU job had zero checked-out DB connections; newer HTTP commit survived 409", + flush=True, + ) + finally: + proceed.set() + pool.shutdown(wait=True) + engine.dispose() + + def test_05_commit_time_failure_rolls_back_text_vector_and_revision(self): + note_id = self.ids[2] + before = stored(note_id) + with connection(owner=True) as owner: + owner.execute( + """CREATE FUNCTION semantic_notes.test_reject_commit() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN IF NEW.title='Rollback marker' THEN RAISE EXCEPTION 'Acceptance commit failure' USING ERRCODE='23514'; END IF; RETURN NEW; END $$""" + ) + owner.execute( + "CREATE CONSTRAINT TRIGGER test_reject_commit AFTER UPDATE ON semantic_notes.notes DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION semantic_notes.test_reject_commit()" + ) + try: + response = self.client.put( + f"/api/notes/{note_id}", + json={ + "title": "Rollback marker", + "body": "A changed text creates a different real vector before this commit fails.", + "revision": before[3], + }, + ) + self.assertEqual(response.status_code, 503) + after = stored(note_id) + self.assertEqual(after[:2], before[:2]) + self.assertEqual(after[3], before[3]) + np.testing.assert_array_equal(after[2].to_numpy(), before[2].to_numpy()) + print( + "Owner-only deferred trigger failed after UPDATE; complete text/vector/revision rolled back", + flush=True, + ) + finally: + with connection(owner=True) as owner: + owner.execute( + "DROP TRIGGER test_reject_commit ON semantic_notes.notes; DROP FUNCTION semantic_notes.test_reject_commit()" + ) + + def test_06_origin_token_and_request_bounds(self): + fields = {"title": "Cross site", "body": "This must not be created."} + with httpx.Client(base_url=ORIGIN, timeout=40) as foreign: + self.assertEqual(foreign.post("/api/notes", json=fields).status_code, 403) + self.assertEqual( + foreign.post( + "/api/notes", + json=fields, + headers={"Origin": "https://foreign.invalid"}, + ).status_code, + 403, + ) + self.assertEqual( + self.client.post( + "/api/notes", json={"title": "Long", "body": "hello " * 255} + ).status_code, + 422, + ) + self.assertEqual( + self.client.post( + "/api/notes", json={"title": "Bad", "body": "x", "embedding": [1]} + ).status_code, + 422, + ) + self.assertEqual( + self.client.post("/api/search", json={"q": "read", "k": 11}).status_code, + 422, + ) + self.assertEqual( + self.client.post("/api/search", json={"q": "x" * 513}).status_code, 422 + ) + self.assertEqual( + self.client.post("/api/notes", content=b"x" * 16385).status_code, 413 + ) + self.assertEqual(self.client.get("/api/notes/not-a-uuid").status_code, 422) + self.assertEqual(self.client.get("/api/notes?page=11").status_code, 422) + + def test_06b_numeric_form_and_unicode_shapes(self): + fields = {"title": "Valid", "body": "Short body", "revision": "9" * 5000} + response = self.client.post(f"/notes/{self.ids[0]}", data=fields) + self.assertEqual(response.status_code, 422) + self.assertIn("9" * 20, response.text) + self.assertEqual( + self.client.post( + "/search", data={"q": "read", "k": "9" * 5000} + ).status_code, + 422, + ) + self.assertEqual( + self.client.post("/search", data={"q": "read", "k": "two"}).status_code, 422 + ) + self.assertEqual( + self.client.post( + "/search", + content="q=read&k=2&k=3", + headers={"Content-Type": "application/x-www-form-urlencoded"}, + ).status_code, + 422, + ) + self.assertEqual( + self.client.post( + "/notes", + content="title=A&title=B&body=Short", + headers={"Content-Type": "application/x-www-form-urlencoded"}, + ).status_code, + 422, + ) + for payload in ['{"title":"Bad\\ud800","body":"Short"}', '{"q":"Bad\\ud800"}']: + path = "/api/notes" if "title" in payload else "/api/search" + self.assertEqual( + self.client.post( + path, content=payload, headers={"Content-Type": "application/json"} + ).status_code, + 422, + ) + + def test_07_roles_dimensions_and_fixed_full_collection_spec(self): + with connection() as runtime: + runtime.autocommit = True + for sql in [ + "CREATE SCHEMA forbidden_semantic", + "ALTER TABLE semantic_notes.notes ADD COLUMN forbidden integer", + "UPDATE semantic_notes.collection SET spec='{}'::jsonb", + "DELETE FROM semantic_notes.notes", + ]: + with self.assertRaises(psycopg.errors.InsufficientPrivilege): + runtime.execute(sql) + for vector in ["[1,2]", "[" + ",".join(["0"] * 384) + "]"]: + with self.assertRaises( + (psycopg.DataError, psycopg.errors.CheckViolation) + ): + runtime.execute( + "INSERT INTO semantic_notes.notes(id,title,body,embedding) VALUES(%s,'Invalid vector','Rejected',%s::vector)", + (uuid4(), vector), + ) + with connection(owner=True) as owner: + changed = {**SPEC, "document_format": "body only"} + from psycopg.types.json import Jsonb + + with self.assertRaises(psycopg.errors.CheckViolation): + owner.execute( + "UPDATE semantic_notes.collection SET spec=%s", (Jsonb(changed),) + ) + old = os.environ.get("MODEL_REVISION") + os.environ["MODEL_REVISION"] = "wrong" + try: + with self.assertRaisesRegex(ValueError, "fixed model"): + Encoder() + finally: + if old is None: + os.environ.pop("MODEL_REVISION", None) + else: + os.environ["MODEL_REVISION"] = old + + def test_08_database_quota_blocks_the_101st_note(self): + vector = stored(self.ids[0])[2] + with connection(owner=True) as owner: + total = owner.execute( + "SELECT count(*) FROM semantic_notes.notes" + ).fetchone()[0] + for i in range(100 - total): + owner.execute( + "INSERT INTO semantic_notes.notes(id,title,body,embedding) VALUES(%s,%s,'Quota fixture reuses an actual model vector',%s)", + (uuid4(), f"Quota {i}", vector), + ) + with self.assertRaises(psycopg.errors.CheckViolation): + owner.execute( + "INSERT INTO semantic_notes.notes(id,title,body,embedding) VALUES(%s,'Over quota','Must fail',%s)", + (uuid4(), vector), + ) + owner.rollback() + self.assertEqual(self.client.get("/api/notes").json()["total"], len(self.ids)) + + def test_09_tls_certificate_and_actual_hostname_negative_controls(self): + with self.assertRaises(psycopg.OperationalError) as ca: + connection(sslrootcert="/etc/ssl/certs/ca-certificates.crt") + self.assertRegex(str(ca.exception).lower(), r"certificate|root cert") + address = socket.getaddrinfo( + os.environ["PGHOST"], 5432, type=socket.SOCK_STREAM + )[0][4][0] + with self.assertRaises(psycopg.OperationalError) as hostname: + connection(host="wrong-hostname.invalid", hostaddr=address) + self.assertRegex( + str(hostname.exception).lower(), + r"does not match host name|hostname|host name.*certificate", + ) + print( + "verify-full wrong CA and wrong hostname reached certificate-specific failures", + flush=True, + ) + + def test_10_stable_ties_literal_filter_and_bounded_pages(self): + vector = stored(self.ids[0])[2] + fixture_ids = [uuid4() for _ in range(13)] + with connection(owner=True) as owner: + for i, note_id in enumerate(fixture_ids): + owner.execute( + "INSERT INTO semantic_notes.notes(id,title,body,embedding) VALUES(%s,%s,'Paging fixture reuses a real model vector',%s)", + (note_id, "Literal %_" if i < 2 else f"Paging {i}", vector), + ) + try: + first = self.client.get("/api/notes").json() + second = self.client.get("/api/notes?page=2").json() + self.assertEqual(len(first["notes"]), 12) + self.assertEqual(len(second["notes"]), 4) + self.assertFalse( + set(n["id"] for n in first["notes"]) + & set(n["id"] for n in second["notes"]) + ) + result = self.client.post( + "/api/search", json={"q": "reading", "k": 10, "title_filter": "%_"} + ).json()["results"] + self.assertEqual( + [n["id"] for n in result], sorted(str(i) for i in fixture_ids[:2]) + ) + self.assertEqual(result[0]["distance"], result[1]["distance"]) + self.assertLessEqual( + len( + self.client.post( + "/api/search", json={"q": "reading", "k": 3} + ).json()["results"] + ), + 3, + ) + finally: + with connection(owner=True) as owner: + owner.execute( + "DELETE FROM semantic_notes.notes WHERE id = ANY(%s)", + (fixture_ids,), + ) + + +if __name__ == "__main__": + unittest.main(verbosity=2) diff --git a/applications/semantic-notes/tests/test_boundaries.py b/applications/semantic-notes/tests/test_boundaries.py new file mode 100644 index 00000000..83c9f969 --- /dev/null +++ b/applications/semantic-notes/tests/test_boundaries.py @@ -0,0 +1,57 @@ +import json +import unittest +from unittest.mock import patch + +import numpy as np +from pydantic import ValidationError + +from semantic_notes.embeddings import checked_vector +from semantic_notes.spec import SPEC, check_source_spec +from semantic_notes.validation import NoteInput, SearchInput, UpdateInput + + +class Boundaries(unittest.TestCase): + def test_vector_rejects_wrong_shape_zero_and_nonfinite_values(self): + for value in [ + np.ones(383), + np.zeros(384), + np.full(384, np.nan), + np.full(384, np.inf), + ]: + with self.assertRaises(ValueError): + checked_vector(value) + self.assertEqual(checked_vector(np.ones(384)).dtype, np.float32) + + def test_revision_requires_integer_not_bool_or_coerced_string(self): + for value in [True, "1", 0, 2147483647]: + with self.assertRaises(ValidationError): + UpdateInput(title="Title", body="Body", revision=value) + + def test_unicode_controls_and_unknown_fields_are_rejected(self): + for title in ["Bad\x7f", "Bad\u0085", "Bad\ud800"]: + with self.assertRaises(ValidationError): + NoteInput(title=title, body="Body") + with self.assertRaises(ValidationError): + NoteInput(title="Title", body="Body", model="other") + with self.assertRaises(ValidationError): + SearchInput(q="Bad\ud800") + self.assertEqual(NoteInput(title=" T ", body="Line one\nLine two").title, "T") + + def test_search_bounds_and_collection_format_are_fixed(self): + for values in [ + {"q": ""}, + {"q": "x", "k": 11}, + {"q": "x", "title_filter": "x" * 81}, + ]: + with self.assertRaises(ValidationError): + SearchInput(**values) + for key, value in [ + ("document_format", "body only"), + ("dtype", "float64"), + ("normalize_embeddings", False), + ]: + with patch( + "pathlib.Path.read_text", return_value=json.dumps({**SPEC, key: value}) + ): + with self.assertRaises(ValueError): + check_source_spec() diff --git a/applications/semantic-notes/tests/test_worker.py b/applications/semantic-notes/tests/test_worker.py new file mode 100644 index 00000000..eeda2263 --- /dev/null +++ b/applications/semantic-notes/tests/test_worker.py @@ -0,0 +1,50 @@ +import asyncio +import threading +import unittest + +from semantic_notes.service import DomainError +from semantic_notes.worker import EmbeddingWorker + + +class WorkerCancellation(unittest.IsolatedAsyncioTestCase): + async def test_cancelled_request_keeps_capacity_until_thread_completion(self): + gate = threading.Event() + started = threading.Event() + lock = threading.Lock() + count = 0 + + class BlockingEncoder: + def embed(self, text): + nonlocal count + with lock: + count += 1 + if count == 2: + started.set() + if not gate.wait(5): + raise RuntimeError("test gate timed out") + return text + + worker = EmbeddingWorker(BlockingEncoder()) + first = asyncio.create_task(worker.embed("one")) + second = asyncio.create_task(worker.embed("two")) + try: + for _ in range(100): + if started.is_set(): + break + await asyncio.sleep(0.01) + self.assertTrue(started.is_set()) + first.cancel() + with self.assertRaises(asyncio.CancelledError): + await first + with self.assertRaises(DomainError) as busy: + await worker.embed("three") + self.assertEqual(busy.exception.status, 429) + self.assertEqual(count, 2) + gate.set() + self.assertEqual(await second, "two") + while worker.pending: + await asyncio.sleep(0.01) + self.assertEqual(await worker.embed("four"), "four") + finally: + gate.set() + await asyncio.gather(first, second, return_exceptions=True) From 37ed017f64f079641c97e6569c49be12da7494d9 Mon Sep 17 00:00:00 2001 From: sdairs Date: Fri, 2 Oct 2026 21:31:28 +0100 Subject: [PATCH 2/2] docs: remove public beta qualifier --- applications/semantic-notes/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/applications/semantic-notes/README.md b/applications/semantic-notes/README.md index ffb0a8a6..f08a2325 100644 --- a/applications/semantic-notes/README.md +++ b/applications/semantic-notes/README.md @@ -1,6 +1,6 @@ # Semantic notes -A trusted local notebook using FastAPI, SQLModel, Sentence Transformers and pgvector on **ClickHouse Managed Postgres (public beta)**. Add short titled notes, edit them with an observed revision, and retrieve excerpts by meaning. The application returns stored notes and cosine distances; it does not generate answers. +A trusted local notebook using FastAPI, SQLModel, Sentence Transformers and pgvector on **ClickHouse Managed Postgres**. Add short titled notes, edit them with an observed revision, and retrieve excerpts by meaning. The application returns stored notes and cosine distances; it does not generate answers. The CPU model is `sentence-transformers/all-MiniLM-L6-v2`, fixed at public revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`. Only allowlisted model files and safetensors weights are downloaded; remote custom code is disabled. After the explicit download, application startup is offline. A full collection specification fixes model ID, revision, 384 dimensions, float32 values, normalized embeddings, 256 tokens including special tokens, and document formatting (`title + two newlines + body`). Source metadata and database metadata must match at startup.