diff --git a/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml new file mode 100644 index 000000000..227792a2f --- /dev/null +++ b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction-disabled.yaml @@ -0,0 +1,61 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Library mode - embeds OGX as library + use_as_library_client: true + # Unified mode: run.yaml (materialized per provider by CI/the harness) + # is consumed as the synthesis profile instead of the legacy two-file path. + config: + profile: run.yaml +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Same small window and threshold as lightspeed-stack-compaction.yaml, but +# compaction switched off: context_status must stay "full" past the +# threshold (enabled is a full off-switch). +compaction: + enabled: false + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml new file mode 100644 index 000000000..94675e16c --- /dev/null +++ b/tests/e2e/configuration/library-mode/lightspeed-stack-compaction.yaml @@ -0,0 +1,60 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Library mode - embeds OGX as library + use_as_library_client: true + # Unified mode: run.yaml (materialized per provider by CI/the harness) + # is consumed as the synthesis profile instead of the legacy two-file path. + config: + profile: run.yaml +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Compaction on with a low threshold: 10% of the 2000-token window, +# above a 100-token floor, keeping one recent turn verbatim. +compaction: + enabled: true + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml new file mode 100644 index 000000000..ff1e3510c --- /dev/null +++ b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction-disabled.yaml @@ -0,0 +1,59 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Server mode - connects to separate OGX service + use_as_library_client: false + url: http://${env.E2E_OGX_HOSTNAME}:8321 + api_key: xyzzy +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Same small window and threshold as lightspeed-stack-compaction.yaml, but +# compaction switched off: context_status must stay "full" past the +# threshold (enabled is a full off-switch). +compaction: + enabled: false + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml new file mode 100644 index 000000000..82dd0c32b --- /dev/null +++ b/tests/e2e/configuration/server-mode/lightspeed-stack-compaction.yaml @@ -0,0 +1,58 @@ +name: Lightspeed Core Service (LCS) +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: false + workers: 1 + color_log: true + access_log: true +ogx: + # Server mode - connects to separate OGX service + use_as_library_client: false + url: http://${env.E2E_OGX_HOSTNAME}:8321 + api_key: xyzzy +user_data_collection: + feedback_enabled: true + feedback_storage: "/tmp/data/feedback" + transcripts_enabled: true + transcripts_storage: "/tmp/data/transcripts" +authentication: + module: "noop" +inference: + default_provider: openai + default_model: gpt-4o-mini + # Compaction e2e (LCORE-1673): a deliberately small window for every + # model the e2e workflows run against, so the third query of the + # compaction scenarios crosses the trigger threshold. The real windows + # are far larger; this only drives the local estimate, never the provider. + # The vLLM-backed runs (rhaiis, rhelai) take the model id from an env + # var, and context_windows keys are not env-substituted, so they are + # not listed and skip the token-based trigger. + context_windows: + openai/gpt-4o-mini: 2000 + azure/gpt-4o-mini: 2000 + google-vertex/publishers/google/models/gemini-2.5-flash: 2000 + watsonx/meta-llama/llama-3-3-70b-instruct: 2000 + aws-bedrock/deepseek.v3-v1:0: 2000 +rag: + byok: + stores: + - rag_id: e2e-test-docs + backend: faiss + embedding_model: sentence-transformers/all-mpnet-base-v2 + embedding_dimension: 768 + vector_db_id: ${env.FAISS_VECTOR_STORE_ID} + db_path: ${env.KV_RAG_PATH:=~/.llama/storage/rag/kv_store.db} + score_multiplier: 1.0 + retrieval: + tool: + sources: + - e2e-test-docs + +# Compaction on with a low threshold: 10% of the 2000-token window, +# above a 100-token floor, keeping one recent turn verbatim. +compaction: + enabled: true + threshold_ratio: 0.1 + token_floor: 100 + buffer_turns: 1 diff --git a/tests/e2e/features/conversation-compaction.feature b/tests/e2e/features/conversation-compaction.feature new file mode 100644 index 000000000..89d794788 --- /dev/null +++ b/tests/e2e/features/conversation-compaction.feature @@ -0,0 +1,102 @@ +# @skip until LCORE-2230 lands the step definitions; Konflux runs the whole +# test list and would fail on the undefined steps. @cfg_compaction is not in +# any GitHub CI shard yet, LCORE-2230 adds it. +@cfg_compaction @skip +Feature: Conversation compaction + + Once the estimated input crosses the configured share of the model's + context window, older turns are summarized before the request reaches + the model. The compaction fixtures register a 2000-token window with a + 10% threshold and keep one recent turn verbatim, so a long third query + is what crosses it: turn one ends up in the summary, turn two stays in + the verbatim buffer, and the third query asks for a fact from each. + + Background: + Given The service is started locally + And The system is in default state + And REST API service prefix is /v1 + And the Lightspeed stack configuration directory is "tests/e2e/configuration" + + + Scenario: the third query crosses the threshold, older turns are summarized, recall and history survive + Given The service uses the lightspeed-stack-compaction.yaml configuration + And The service is restarted + When I use "query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" + And I store conversation details + When I use "query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" + When I use "query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "summarized" + And The response contains following fragments + | Fragments in LLM response | + | aurora-prod-7 | + | blue-lagoon | + When I use REST API conversation endpoint with conversation_id from above using HTTP GET method + Then The status code of the response is 200 + And The conversation history includes the following user queries + | User query | + | My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only. | + | My application namespace is called blue-lagoon. Remember that name too and reply with OK only. | + + + Scenario: the native stream announces compaction on the query that crosses the threshold + Given The service uses the lightspeed-stack-compaction.yaml configuration + And The service is restarted + When I use "streaming_query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response end event has context_status "full" + And I store conversation details + When I use "streaming_query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response end event has context_status "full" + When I use "streaming_query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I wait for the response to be completed + And The streamed response contains a compaction event before the first token + And The streamed response end event has context_status "summarized" + + + Scenario: compaction stays off when disabled, even past the threshold + Given The service uses the lightspeed-stack-compaction-disabled.yaml configuration + And The service is restarted + When I use "query" to ask question + """ + {"query": "My OpenShift cluster is named aurora-prod-7. Remember that name and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And I store conversation details + When I use "query" to ask question with same conversation_id + """ + {"query": "My application namespace is called blue-lagoon. Remember that name too and reply with OK only.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + When I use "query" to ask question with same conversation_id + """ + {"query": "Some background on my environment first, no need to comment on it. The cluster runs on bare metal in two racks with three control plane nodes and nine worker nodes, all on the same subnet behind a pair of hardware load balancers. Storage is provided by an external Ceph cluster exposed through the CSI driver, with three storage classes for block, file and object access. Ingress is handled by the default router with two replicas pinned to the infra nodes, and TLS certificates are issued by an internal certificate authority and rotated every ninety days. Monitoring uses the built-in Prometheus stack with a remote write to a central Thanos instance, and alerts are routed to an on-call rotation through a webhook receiver. The image registry is the internal one, backed by an object storage bucket, and images are mirrored from an upstream registry once a day by a scheduled job. Upgrades follow the stable channel, one minor version at a time, and are rehearsed on a staging cluster of the same shape a week before production. Backups of etcd are taken hourly and copied off-site nightly. Now the question: what is the name of my cluster and what is the name of my application namespace? Reply with the two names only, separated by a comma.", "model": "{MODEL}", "provider": "{PROVIDER}"} + """ + Then The status code of the response is 200 + And The response context_status is "full" diff --git a/tests/e2e/test_list.txt b/tests/e2e/test_list.txt index 79f34fcb5..c37e04e98 100644 --- a/tests/e2e/test_list.txt +++ b/tests/e2e/test_list.txt @@ -23,6 +23,7 @@ features/rlsapi_v1.feature features/streaming_query.feature features/vector_stores.feature features/conversation_cache_v2.feature +features/conversation-compaction.feature features/feedback.feature features/http_401_unauthorized.feature features/rbac.feature