diff --git a/.github/workflows/e2e_tests.yaml b/.github/workflows/e2e_tests.yaml index 5e14d19cf..b5c0fe76e 100644 --- a/.github/workflows/e2e_tests.yaml +++ b/.github/workflows/e2e_tests.yaml @@ -36,7 +36,7 @@ jobs: - name: shields tags: "not @skip and @cfg_shields" - name: other - tags: "not @skip and (@cfg_rh_identity or @cfg_negative or @cfg_byok_pdf or @cfg_degraded or @cfg_unified)" + tags: "not @skip and (@cfg_rh_identity or @cfg_negative or @cfg_byok_pdf or @cfg_degraded or @cfg_unified or @cfg_compaction)" # Server-only; listed in shard (not matrix.include) so it expands with # mode=server before any library jobs. include would append after library. - name: tls diff --git a/Makefile b/Makefile index 2fa421ada..44beb3a48 100644 --- a/Makefile +++ b/Makefile @@ -164,7 +164,7 @@ test-e2e-local: ## Run end to end tests for the service (no script wrapper) # Tag-based subsets (@cfg_* on features/scenarios). Default runs all config groups; override for one shard, e.g. # E2E_BEHAVE_TAG_EXPR='not @skip and @cfg_authorized' make test-e2e-tagged-local -E2E_BEHAVE_TAG_EXPR ?= not @skip and (@cfg_default or @cfg_authorized or @cfg_mcp or @cfg_mcp_invalid or @cfg_mcp_api_auth or @cfg_rbac or @cfg_rh_identity or @cfg_negative or @cfg_skills or @cfg_skills_directory or @cfg_shields or @cfg_byok_pdf or @cfg_tls or @cfg_degraded or @cfg_unified) +E2E_BEHAVE_TAG_EXPR ?= not @skip and (@cfg_default or @cfg_authorized or @cfg_mcp or @cfg_mcp_invalid or @cfg_mcp_api_auth or @cfg_rbac or @cfg_rh_identity or @cfg_negative or @cfg_skills or @cfg_skills_directory or @cfg_shields or @cfg_byok_pdf or @cfg_tls or @cfg_degraded or @cfg_unified or @cfg_compaction) test-e2e-tagged: ## Run e2e tests with E2E_BEHAVE_TAG_EXPR (default: all @cfg_*) script -q -e -c "uv run behave --color --format pretty --tags=\"$(E2E_BEHAVE_TAG_EXPR)\" -D dump_errors=true @tests/e2e/test_list.txt" diff --git a/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml new file mode 100644 index 000000000..227792a2f --- /dev/null +++ b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml @@ -0,0 +1,61 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Library mode - embeds OGX as library + use_as_library_client: true + # Unified mode: run.yaml (materialized per provider by CI/the harness) + # is consumed as the synthesis profile instead of the legacy two-file path. + config: + profile: run.yaml +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Same small window and threshold as lightspeed-stack-compaction.yaml, but +# compaction switched off: context_status must stay "full" past the +# threshold (enabled is a full off-switch). +compaction: + enabled: false + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml new file mode 100644 index 000000000..94675e16c --- /dev/null +++ b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml @@ -0,0 +1,60 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Library mode - embeds OGX as library + use_as_library_client: true + # Unified mode: run.yaml (materialized per provider by CI/the harness) + # is consumed as the synthesis profile instead of the legacy two-file path. + config: + profile: run.yaml +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Compaction on with a low threshold: 10% of the 2000-token window, +# above a 100-token floor, keeping one recent turn verbatim. +compaction: + enabled: true + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml new file mode 100644 index 000000000..ff1e3510c --- /dev/null +++ b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml @@ -0,0 +1,59 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Server mode - connects to separate OGX service + use_as_library_client: false + url: http://${env.E2E_OGX_HOSTNAME}:8321 + api_key: xyzzy +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Same small window and threshold as lightspeed-stack-compaction.yaml, but +# compaction switched off: context_status must stay "full" past the +# threshold (enabled is a full off-switch). +compaction: + enabled: false + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml new file mode 100644 index 000000000..82dd0c32b --- /dev/null +++ b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml @@ -0,0 +1,58 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Server mode - connects to separate OGX service + use_as_library_client: false + url: http://${env.E2E_OGX_HOSTNAME}:8321 + api_key: xyzzy +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Compaction on with a low threshold: 10% of the 2000-token window, +# above a 100-token floor, keeping one recent turn verbatim. +compaction: + enabled: true + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/features/conversation-compaction.feature b/tests/e2e/features/conversation-compaction.feature new file mode 100644 index 000000000..7e0fafc4a --- /dev/null +++ b/tests/e2e/features/conversation-compaction.feature @@ -0,0 +1,102 @@ +@cfg_compaction +Feature: Conversation compaction + + Once the estimated input crosses the configured share of the model's + context window, older turns are summarized before the request reaches + the model. The compaction fixtures register a 2000-token window with a + 10% threshold and keep one recent turn verbatim, so a long third query + is what crosses it: turn one ends up in the summary, turn two stays in + the verbatim buffer, and the third query asks for a fact from each. + + Background: + Given The service is started locally + And The system is in default state + And REST API service prefix is /v1 + And the Lightspeed stack configuration directory is "tests/e2e/configuration" + + + Scenario: the third query crosses the threshold, older turns are summarized, recall and history survive + Given The service uses the lightspeed-stack-compaction.yaml configuration + And the active model has a registered context window + And The service is restarted + When I use "query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" + And I store conversation details + When I use "query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" + When I use "query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "summarized" + And The response contains following fragments + | Fragments in LLM response | + | aurora-prod-7 | + | blue-lagoon | + When I use REST API conversation endpoint with conversation_id from above using HTTP GET method + Then The status code of the response is 200 + And The conversation history includes the following user queries + | User query | + | My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only. | + | My application namespace is called blue-lagoon. Remember that name too and reply with OK only. | + + + Scenario: the native stream announces compaction on the query that crosses the threshold + Given The service uses the lightspeed-stack-compaction.yaml configuration + And the active model has a registered context window + And The service is restarted + When I use "streaming_query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response end event has context_status "full" + And I store conversation details + When I use "streaming_query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response end event has context_status "full" + When I use "streaming_query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response contains a compaction event before the first token + And The streamed response end event has context_status "summarized" + + + Scenario: compaction stays off when disabled, even past the threshold + Given The service uses the lightspeed-stack-compaction-disabled.yaml configuration + And the active model has a registered context window + And The service is restarted + When I use "query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I store conversation details + When I use "query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + When I use "query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" diff --git a/tests/e2e/features/steps/README.md b/tests/e2e/features/steps/README.md index db2233cfe..885cfd273 100644 --- a/tests/e2e/features/steps/README.md +++ b/tests/e2e/features/steps/README.md @@ -20,6 +20,10 @@ Common steps for HTTP-related operations. Implementation of common test steps. +## [conversation_compaction.py](conversation_compaction.py) + +Steps observing conversation compaction from outside: `context_status`, the stream's `compaction` event, and the history the Conversations API keeps. + ## [feedback.py](feedback.py) Implementation of common test steps for the feedback API. diff --git a/tests/e2e/features/steps/conversation_compaction.py b/tests/e2e/features/steps/conversation_compaction.py new file mode 100644 index 000000000..691b2b01b --- /dev/null +++ b/tests/e2e/features/steps/conversation_compaction.py @@ -0,0 +1,157 @@ +"""Step definitions for the conversation-compaction e2e feature (LCORE-2230). + +Everything here observes compaction from outside the deployed stack — the +``context_status`` field on responses, the ``compaction`` event on the native +stream, and the conversation history the Conversations API serves. Steps +never import from or execute anything under ``src/`` +(``docs/testing/e2e_testing.md``, "Choosing the Test Layer"); the internals +(buffer, additive summaries, blocking) are integration tests (LCORE-1574). +""" + +import json +import os +from typing import Any, Optional + +import yaml +from behave import given, then # pyright: ignore +from behave.runner import Context + +from tests.e2e.utils.utils import absolute_repo_path, is_prow_environment + + +def _active_fixture_path(context: Context) -> str: + """Resolve the fixture file the last ``The service uses ...`` step applied. + + Mirrors the lookup in ``common.configure_service``: the configuration + directory from the Background, the deployment-mode subdirectory when it + exists, and the basename recorded on the context. The repo-root + ``lightspeed-stack.yaml`` copy is not used because on Prow the config is + pushed into a ConfigMap and that file is never written. + """ + config_name = context.active_lightspeed_stack_config_basename + mode_dir = "library-mode" if context.is_library_mode else "server-mode" + raw_base = getattr(context, "lightspeed_stack_config_directory", None) + base = str(raw_base).strip().rstrip("/") if raw_base else "tests/e2e/configuration" + mode_base = os.path.join(base, mode_dir) + if is_prow_environment(): + mode_base = absolute_repo_path(mode_base) + base = absolute_repo_path(base) + if os.path.isdir(mode_base): + return os.path.join(mode_base, config_name) + return os.path.join(base, config_name) + + +def _sse_events(response_text: str) -> list[dict[str, Any]]: + """Return the decoded SSE ``data:`` payloads of a streamed response, in order.""" + events: list[dict[str, Any]] = [] + for line in response_text.strip().split("\n"): + if not line.startswith("data: "): + continue + try: + events.append(json.loads(line[6:])) + except json.JSONDecodeError: + continue # Skip malformed lines + return events + + +def _first_index(events: list[dict[str, Any]], name: str) -> Optional[int]: + """Return the position of the first event called ``name``, or None.""" + return next((i for i, e in enumerate(events) if e.get("event") == name), None) + + +@given("the active model has a registered context window") +def require_context_window_for_active_model(context: Context) -> None: + """Skip the scenario when the active model has no ``context_windows`` entry. + + The compaction trigger only runs for models listed under + ``inference.context_windows`` in the active lightspeed-stack.yaml. The + fixtures list every fixed provider/model pair the e2e workflows use, but + the vLLM-backed runs (rhaiis, rhelai) take their model id from an env var + and mapping keys are not env-substituted, so those runs have no entry and + ``context_status`` can never become ``summarized``. Rather than fail there, + the scenario is skipped, the same way shield scenarios skip in library mode. + + Reads the fixture source file, never anything under ``src/``. The + provider/model pair comes from the context, which already reflects the + ``E2E_DEFAULT_*_OVERRIDE`` values on every path. + """ + fixture_path = _active_fixture_path(context) + with open(fixture_path, encoding="utf-8") as config_file: + config = yaml.safe_load(config_file) or {} + windows = (config.get("inference") or {}).get("context_windows") or {} + model_key = f"{context.default_provider}/{context.default_model}" + if model_key not in windows: + context.scenario.skip( + f"no context window registered for {model_key} in {fixture_path}; " + f"compaction cannot trigger (registered: {sorted(windows)})" + ) + + +@then('The response context_status is "{status}"') +def check_context_status(context: Context, status: str) -> None: + """Assert the non-streaming response reports the expected ``context_status`` (R7).""" + assert context.response is not None, "Request needs to be performed first" + response_json = context.response.json() + assert ( + "context_status" in response_json + ), f"context_status missing from response; keys: {list(response_json)}" + actual = response_json["context_status"] + assert actual == status, f"context_status is {actual!r}, expected {status!r}" + + +@then("The conversation history includes the following user queries") +def check_history_includes_user_queries(context: Context) -> None: + """Assert every listed query is still a user message in the conversation (R6). + + Reads ``chat_history`` from the GET conversation response and collects the + content of every ``user``-typed message across all turns; each row of the + scenario's "User query" table must appear among them verbatim. + """ + assert context.response is not None, "Request needs to be performed first" + assert context.table is not None, "Table with column 'User query' is required" + response_json = context.response.json() + assert "chat_history" in response_json, "chat_history not found in response" + user_queries = [ + message["content"].strip() + for turn in response_json["chat_history"] + for message in turn.get("messages", []) + if message.get("type") == "user" + ] + for row in context.table: + expected = row["User query"].strip() + assert expected in user_queries, ( + f"user query {expected!r} not found in conversation history; " + f"user queries present: {user_queries!r}" + ) + + +@then("The streamed response contains a compaction event before the first token") +def check_compaction_event_precedes_tokens(context: Context) -> None: + """Assert the stream announced compaction before any answer token (R12).""" + assert context.response is not None, "Request needs to be performed first" + events = _sse_events(context.response.text) + names = [e.get("event") for e in events] + compaction_at = _first_index(events, "compaction") + assert compaction_at is not None, f"no compaction event in stream; events: {names}" + token_at = _first_index(events, "token") + assert token_at is None or compaction_at < token_at, ( + f"compaction event at position {compaction_at} came after the first " + f"token at {token_at}; events: {names}" + ) + + +@then('The streamed response end event has context_status "{status}"') +def check_end_event_context_status(context: Context, status: str) -> None: + """Assert the stream's ``end`` event payload carries the expected ``context_status`` (R7).""" + assert context.response is not None, "Request needs to be performed first" + events = _sse_events(context.response.text) + end_at = _first_index(events, "end") + assert ( + end_at is not None + ), f"no end event in stream; events: {[e.get('event') for e in events]}" + data = events[end_at].get("data") or {} + assert "context_status" in data, f"end event carries no context_status: {data!r}" + actual = data["context_status"] + assert ( + actual == status + ), f"end event context_status is {actual!r}, expected {status!r}" diff --git a/tests/e2e/test_list.txt b/tests/e2e/test_list.txt index 79f34fcb5..c37e04e98 100644 --- a/tests/e2e/test_list.txt +++ b/tests/e2e/test_list.txt @@ -23,6 +23,7 @@ features/rlsapi_v1.feature features/streaming_query.feature features/vector_stores.feature features/conversation_cache_v2.feature +features/conversation-compaction.feature features/feedback.feature features/http_401_unauthorized.feature features/rbac.feature