From 286e1be614fbdee82c6a3edb53ed9c8c9bc3d19a Mon Sep 17 00:00:00 2001 From: callumreid <75899979+callumreid@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:09:24 +0000 Subject: [PATCH 1/3] chore(cli): refresh API parity report --- api-coverage-report.md | 184 +++++++++++++++++++++++++++++++++++++---- 1 file changed, 167 insertions(+), 17 deletions(-) diff --git a/api-coverage-report.md b/api-coverage-report.md index d69f221..82c1bec 100644 --- a/api-coverage-report.md +++ b/api-coverage-report.md @@ -12,20 +12,21 @@ only when coverage actually changes. | Metric | Value | | --- | ---: | -| Reconciliation status | PASS | -| Published operations | 181 | +| Reconciliation status | ACTION REQUIRED | +| Published operations | 183 | | First-class CLI operations | 119 | | Reviewed gaps | 62 | | Client operations | 143 | -| Published request fields on covered operations | 365 | -| Request fields modeled by the CLI | 359 | -| Reviewed request-field gaps | 6 | +| Published request fields on covered operations | 437 | +| Request fields modeled by the CLI | 370 | +| Reviewed request-field gaps | 0 | Catalog: https://api.coval.dev/v1/openapi ## New published operations without CLI commands -- None. +- `POST /review-projects/{project_id}/complete-conversation` +- `POST /sofia/delegation-token` ## Reviewed gaps no longer present @@ -49,7 +50,9 @@ Catalog: https://api.coval.dev/v1/openapi ## Coverage snapshot mismatches -- None. +- `published_operations: recorded 181, current 183` +- `published_request_fields: recorded 365, current 437` +- `cli_modeled_request_fields: recorded 359, current 370` ## Client-only operations @@ -112,6 +115,8 @@ Catalog: https://api.coval.dev/v1/openapi - `POST /personas/{persona_id}/versions/{version_id}/revert` - `POST /review-annotations:withMetricOutputs` - `POST /review-projects/disagreement-state` +- `POST /review-projects/{project_id}/complete-conversation` +- `POST /sofia/delegation-token` - `POST /test-sets/{test_set_id}/agents:add` - `POST /test-sets/{test_set_id}/duplicate` - `POST /test-sets/{test_set_id}/versions/{version_id}/revert` @@ -122,25 +127,170 @@ Catalog: https://api.coval.dev/v1/openapi ## New published request fields the CLI drops -- None. +- `PATCH /metrics/{id} aggregation_method` +- `PATCH /metrics/{id} detection_preset` +- `PATCH /metrics/{id} enabled_tools` +- `PATCH /metrics/{id} harmonics_to_noise_ratio_threshold_offset_db` +- `PATCH /metrics/{id} jitter_threshold_multiplier` +- `PATCH /metrics/{id} judge_mode` +- `PATCH /metrics/{id} loud_threshold_db` +- `PATCH /metrics/{id} low_pitch_threshold_multiplier` +- `PATCH /metrics/{id} mad_z_score_threshold` +- `PATCH /metrics/{id} metric_attribute` +- `PATCH /metrics/{id} metric_metadata` +- `PATCH /metrics/{id} min_fry_segment_seconds` +- `PATCH /metrics/{id} pause_detection_preset` +- `PATCH /metrics/{id} pitch_change_threshold_hz` +- `PATCH /metrics/{id} significant_changes_threshold_hz` +- `PATCH /metrics/{id} soft_threshold_db` +- `PATCH /metrics/{id} span_name` +- `PATCH /metrics/{id} threshold_preset` +- `PATCH /metrics/{id} unit` +- `PATCH /metrics/{id} value_source` +- `PATCH /personas/{id} custom_persona_data` +- `PATCH /personas/{id} custom_voice_id` +- `PATCH /personas/{id} initialization_parameters` +- `PATCH /personas/{id} multi_phone_config` +- `PATCH /personas/{id} silent_mode` +- `PATCH /personas/{id} voice` +- `PATCH /review-annotations/{id} annotations` +- `PATCH /review-annotations/{id} ground_truth_json` +- `PATCH /review-annotations/{id} ground_truth_set_value` +- `PATCH /review-projects/{id} project_type` +- `PATCH /review-projects/{id} review_label_input_mode` +- `PATCH /review-projects/{id} review_label_options` +- `PATCH /review-projects/{id} review_label_selection_mode` +- `PATCH /run-templates/{id} test_case_ids` +- `POST /metrics aggregation_method` +- `POST /metrics detection_preset` +- `POST /metrics enabled_tools` +- `POST /metrics harmonics_to_noise_ratio_threshold_offset_db` +- `POST /metrics jitter_threshold_multiplier` +- `POST /metrics judge_mode` +- `POST /metrics loud_threshold_db` +- `POST /metrics low_pitch_threshold_multiplier` +- `POST /metrics mad_z_score_threshold` +- `POST /metrics metric_attribute` +- `POST /metrics metric_metadata` +- `POST /metrics min_fry_segment_seconds` +- `POST /metrics pause_detection_preset` +- `POST /metrics pitch_change_threshold_hz` +- `POST /metrics significant_changes_threshold_hz` +- `POST /metrics soft_threshold_db` +- `POST /metrics span_name` +- `POST /metrics threshold_preset` +- `POST /metrics unit` +- `POST /metrics value_source` +- `POST /personas custom_persona_data` +- `POST /personas custom_voice_id` +- `POST /personas initialization_parameters` +- `POST /personas multi_phone_config` +- `POST /personas silent_mode` +- `POST /personas voice` +- `POST /review-annotations annotations` +- `POST /review-annotations ground_truth_json` +- `POST /review-annotations ground_truth_set_value` +- `POST /review-projects review_label_input_mode` +- `POST /review-projects review_label_options` +- `POST /review-projects review_label_selection_mode` +- `POST /run-templates test_case_ids` ## Reviewed request-field gaps no longer present -- None. +- `PATCH /run-templates/{id} agent_id` +- `PATCH /run-templates/{id} persona_id` +- `PATCH /run-templates/{id} test_set_id` +- `POST /run-templates agent_id` +- `POST /run-templates persona_id` +- `POST /run-templates test_set_id` ## CLI request fields absent from published OpenAPI -- None. +- `POST /review-annotations priority` +- `POST /runs persona_metrics` ## Allowed extra request fields no longer present -- None. +- `PATCH /metrics/{id} case_insensitive` +- `PATCH /metrics/{id} match_mode` +- `PATCH /metrics/{id} position` +- `PATCH /run-templates/{id} agent_ids` +- `PATCH /run-templates/{id} persona_ids` +- `PATCH /run-templates/{id} test_set_ids` +- `POST /api-keys environment` +- `POST /metrics case_insensitive` +- `POST /metrics match_mode` +- `POST /metrics position` +- `POST /run-templates agent_ids` +- `POST /run-templates persona_ids` +- `POST /run-templates test_set_ids` ## All current request-field gaps -- `PATCH /run-templates/{id} agent_id` -- `PATCH /run-templates/{id} persona_id` -- `PATCH /run-templates/{id} test_set_id` -- `POST /run-templates agent_id` -- `POST /run-templates persona_id` -- `POST /run-templates test_set_id` +- `PATCH /metrics/{id} aggregation_method` +- `PATCH /metrics/{id} detection_preset` +- `PATCH /metrics/{id} enabled_tools` +- `PATCH /metrics/{id} harmonics_to_noise_ratio_threshold_offset_db` +- `PATCH /metrics/{id} jitter_threshold_multiplier` +- `PATCH /metrics/{id} judge_mode` +- `PATCH /metrics/{id} loud_threshold_db` +- `PATCH /metrics/{id} low_pitch_threshold_multiplier` +- `PATCH /metrics/{id} mad_z_score_threshold` +- `PATCH /metrics/{id} metric_attribute` +- `PATCH /metrics/{id} metric_metadata` +- `PATCH /metrics/{id} min_fry_segment_seconds` +- `PATCH /metrics/{id} pause_detection_preset` +- `PATCH /metrics/{id} pitch_change_threshold_hz` +- `PATCH /metrics/{id} significant_changes_threshold_hz` +- `PATCH /metrics/{id} soft_threshold_db` +- `PATCH /metrics/{id} span_name` +- `PATCH /metrics/{id} threshold_preset` +- `PATCH /metrics/{id} unit` +- `PATCH /metrics/{id} value_source` +- `PATCH /personas/{id} custom_persona_data` +- `PATCH /personas/{id} custom_voice_id` +- `PATCH /personas/{id} initialization_parameters` +- `PATCH /personas/{id} multi_phone_config` +- `PATCH /personas/{id} silent_mode` +- `PATCH /personas/{id} voice` +- `PATCH /review-annotations/{id} annotations` +- `PATCH /review-annotations/{id} ground_truth_json` +- `PATCH /review-annotations/{id} ground_truth_set_value` +- `PATCH /review-projects/{id} project_type` +- `PATCH /review-projects/{id} review_label_input_mode` +- `PATCH /review-projects/{id} review_label_options` +- `PATCH /review-projects/{id} review_label_selection_mode` +- `PATCH /run-templates/{id} test_case_ids` +- `POST /metrics aggregation_method` +- `POST /metrics detection_preset` +- `POST /metrics enabled_tools` +- `POST /metrics harmonics_to_noise_ratio_threshold_offset_db` +- `POST /metrics jitter_threshold_multiplier` +- `POST /metrics judge_mode` +- `POST /metrics loud_threshold_db` +- `POST /metrics low_pitch_threshold_multiplier` +- `POST /metrics mad_z_score_threshold` +- `POST /metrics metric_attribute` +- `POST /metrics metric_metadata` +- `POST /metrics min_fry_segment_seconds` +- `POST /metrics pause_detection_preset` +- `POST /metrics pitch_change_threshold_hz` +- `POST /metrics significant_changes_threshold_hz` +- `POST /metrics soft_threshold_db` +- `POST /metrics span_name` +- `POST /metrics threshold_preset` +- `POST /metrics unit` +- `POST /metrics value_source` +- `POST /personas custom_persona_data` +- `POST /personas custom_voice_id` +- `POST /personas initialization_parameters` +- `POST /personas multi_phone_config` +- `POST /personas silent_mode` +- `POST /personas voice` +- `POST /review-annotations annotations` +- `POST /review-annotations ground_truth_json` +- `POST /review-annotations ground_truth_set_value` +- `POST /review-projects review_label_input_mode` +- `POST /review-projects review_label_options` +- `POST /review-projects review_label_selection_mode` +- `POST /run-templates test_case_ids` From 9c0afcaa93a549a49eec4effab507bfdbf8816e4 Mon Sep 17 00:00:00 2001 From: Callum Reid Date: Mon, 5 Oct 2026 08:50:26 -0700 Subject: [PATCH 2/3] [COVAL-2079] Reconcile CLI API parity --- api-coverage-report.md | 173 ++------------------- api-coverage.toml | 111 ++------------ src/client/models/metric.rs | 85 +++++++++++ src/client/models/persona.rs | 50 +++++++ src/client/models/review_annotation.rs | 12 +- src/client/models/review_project.rs | 14 ++ src/client/models/run.rs | 2 - src/client/models/run_template.rs | 4 + src/commands/review_annotations.rs | 4 - src/commands/runs.rs | 1 - tests/cli_tests.rs | 200 ++++++++++++++++++++++++- 11 files changed, 385 insertions(+), 271 deletions(-) diff --git a/api-coverage-report.md b/api-coverage-report.md index 82c1bec..fba412d 100644 --- a/api-coverage-report.md +++ b/api-coverage-report.md @@ -12,21 +12,20 @@ only when coverage actually changes. | Metric | Value | | --- | ---: | -| Reconciliation status | ACTION REQUIRED | +| Reconciliation status | PASS | | Published operations | 183 | | First-class CLI operations | 119 | -| Reviewed gaps | 62 | +| Reviewed gaps | 64 | | Client operations | 143 | | Published request fields on covered operations | 437 | -| Request fields modeled by the CLI | 370 | +| Request fields modeled by the CLI | 437 | | Reviewed request-field gaps | 0 | Catalog: https://api.coval.dev/v1/openapi ## New published operations without CLI commands -- `POST /review-projects/{project_id}/complete-conversation` -- `POST /sofia/delegation-token` +- None. ## Reviewed gaps no longer present @@ -50,9 +49,7 @@ Catalog: https://api.coval.dev/v1/openapi ## Coverage snapshot mismatches -- `published_operations: recorded 181, current 183` -- `published_request_fields: recorded 365, current 437` -- `cli_modeled_request_fields: recorded 359, current 370` +- None. ## Client-only operations @@ -127,170 +124,20 @@ Catalog: https://api.coval.dev/v1/openapi ## New published request fields the CLI drops -- `PATCH /metrics/{id} aggregation_method` -- `PATCH /metrics/{id} detection_preset` -- `PATCH /metrics/{id} enabled_tools` -- `PATCH /metrics/{id} harmonics_to_noise_ratio_threshold_offset_db` -- `PATCH /metrics/{id} jitter_threshold_multiplier` -- `PATCH /metrics/{id} judge_mode` -- `PATCH /metrics/{id} loud_threshold_db` -- `PATCH /metrics/{id} low_pitch_threshold_multiplier` -- `PATCH /metrics/{id} mad_z_score_threshold` -- `PATCH /metrics/{id} metric_attribute` -- `PATCH /metrics/{id} metric_metadata` -- `PATCH /metrics/{id} min_fry_segment_seconds` -- `PATCH /metrics/{id} pause_detection_preset` -- `PATCH /metrics/{id} pitch_change_threshold_hz` -- `PATCH /metrics/{id} significant_changes_threshold_hz` -- `PATCH /metrics/{id} soft_threshold_db` -- `PATCH /metrics/{id} span_name` -- `PATCH /metrics/{id} threshold_preset` -- `PATCH /metrics/{id} unit` -- `PATCH /metrics/{id} value_source` -- `PATCH /personas/{id} custom_persona_data` -- `PATCH /personas/{id} custom_voice_id` -- `PATCH /personas/{id} initialization_parameters` -- `PATCH /personas/{id} multi_phone_config` -- `PATCH /personas/{id} silent_mode` -- `PATCH /personas/{id} voice` -- `PATCH /review-annotations/{id} annotations` -- `PATCH /review-annotations/{id} ground_truth_json` -- `PATCH /review-annotations/{id} ground_truth_set_value` -- `PATCH /review-projects/{id} project_type` -- `PATCH /review-projects/{id} review_label_input_mode` -- `PATCH /review-projects/{id} review_label_options` -- `PATCH /review-projects/{id} review_label_selection_mode` -- `PATCH /run-templates/{id} test_case_ids` -- `POST /metrics aggregation_method` -- `POST /metrics detection_preset` -- `POST /metrics enabled_tools` -- `POST /metrics harmonics_to_noise_ratio_threshold_offset_db` -- `POST /metrics jitter_threshold_multiplier` -- `POST /metrics judge_mode` -- `POST /metrics loud_threshold_db` -- `POST /metrics low_pitch_threshold_multiplier` -- `POST /metrics mad_z_score_threshold` -- `POST /metrics metric_attribute` -- `POST /metrics metric_metadata` -- `POST /metrics min_fry_segment_seconds` -- `POST /metrics pause_detection_preset` -- `POST /metrics pitch_change_threshold_hz` -- `POST /metrics significant_changes_threshold_hz` -- `POST /metrics soft_threshold_db` -- `POST /metrics span_name` -- `POST /metrics threshold_preset` -- `POST /metrics unit` -- `POST /metrics value_source` -- `POST /personas custom_persona_data` -- `POST /personas custom_voice_id` -- `POST /personas initialization_parameters` -- `POST /personas multi_phone_config` -- `POST /personas silent_mode` -- `POST /personas voice` -- `POST /review-annotations annotations` -- `POST /review-annotations ground_truth_json` -- `POST /review-annotations ground_truth_set_value` -- `POST /review-projects review_label_input_mode` -- `POST /review-projects review_label_options` -- `POST /review-projects review_label_selection_mode` -- `POST /run-templates test_case_ids` +- None. ## Reviewed request-field gaps no longer present -- `PATCH /run-templates/{id} agent_id` -- `PATCH /run-templates/{id} persona_id` -- `PATCH /run-templates/{id} test_set_id` -- `POST /run-templates agent_id` -- `POST /run-templates persona_id` -- `POST /run-templates test_set_id` +- None. ## CLI request fields absent from published OpenAPI -- `POST /review-annotations priority` -- `POST /runs persona_metrics` +- None. ## Allowed extra request fields no longer present -- `PATCH /metrics/{id} case_insensitive` -- `PATCH /metrics/{id} match_mode` -- `PATCH /metrics/{id} position` -- `PATCH /run-templates/{id} agent_ids` -- `PATCH /run-templates/{id} persona_ids` -- `PATCH /run-templates/{id} test_set_ids` -- `POST /api-keys environment` -- `POST /metrics case_insensitive` -- `POST /metrics match_mode` -- `POST /metrics position` -- `POST /run-templates agent_ids` -- `POST /run-templates persona_ids` -- `POST /run-templates test_set_ids` +- None. ## All current request-field gaps -- `PATCH /metrics/{id} aggregation_method` -- `PATCH /metrics/{id} detection_preset` -- `PATCH /metrics/{id} enabled_tools` -- `PATCH /metrics/{id} harmonics_to_noise_ratio_threshold_offset_db` -- `PATCH /metrics/{id} jitter_threshold_multiplier` -- `PATCH /metrics/{id} judge_mode` -- `PATCH /metrics/{id} loud_threshold_db` -- `PATCH /metrics/{id} low_pitch_threshold_multiplier` -- `PATCH /metrics/{id} mad_z_score_threshold` -- `PATCH /metrics/{id} metric_attribute` -- `PATCH /metrics/{id} metric_metadata` -- `PATCH /metrics/{id} min_fry_segment_seconds` -- `PATCH /metrics/{id} pause_detection_preset` -- `PATCH /metrics/{id} pitch_change_threshold_hz` -- `PATCH /metrics/{id} significant_changes_threshold_hz` -- `PATCH /metrics/{id} soft_threshold_db` -- `PATCH /metrics/{id} span_name` -- `PATCH /metrics/{id} threshold_preset` -- `PATCH /metrics/{id} unit` -- `PATCH /metrics/{id} value_source` -- `PATCH /personas/{id} custom_persona_data` -- `PATCH /personas/{id} custom_voice_id` -- `PATCH /personas/{id} initialization_parameters` -- `PATCH /personas/{id} multi_phone_config` -- `PATCH /personas/{id} silent_mode` -- `PATCH /personas/{id} voice` -- `PATCH /review-annotations/{id} annotations` -- `PATCH /review-annotations/{id} ground_truth_json` -- `PATCH /review-annotations/{id} ground_truth_set_value` -- `PATCH /review-projects/{id} project_type` -- `PATCH /review-projects/{id} review_label_input_mode` -- `PATCH /review-projects/{id} review_label_options` -- `PATCH /review-projects/{id} review_label_selection_mode` -- `PATCH /run-templates/{id} test_case_ids` -- `POST /metrics aggregation_method` -- `POST /metrics detection_preset` -- `POST /metrics enabled_tools` -- `POST /metrics harmonics_to_noise_ratio_threshold_offset_db` -- `POST /metrics jitter_threshold_multiplier` -- `POST /metrics judge_mode` -- `POST /metrics loud_threshold_db` -- `POST /metrics low_pitch_threshold_multiplier` -- `POST /metrics mad_z_score_threshold` -- `POST /metrics metric_attribute` -- `POST /metrics metric_metadata` -- `POST /metrics min_fry_segment_seconds` -- `POST /metrics pause_detection_preset` -- `POST /metrics pitch_change_threshold_hz` -- `POST /metrics significant_changes_threshold_hz` -- `POST /metrics soft_threshold_db` -- `POST /metrics span_name` -- `POST /metrics threshold_preset` -- `POST /metrics unit` -- `POST /metrics value_source` -- `POST /personas custom_persona_data` -- `POST /personas custom_voice_id` -- `POST /personas initialization_parameters` -- `POST /personas multi_phone_config` -- `POST /personas silent_mode` -- `POST /personas voice` -- `POST /review-annotations annotations` -- `POST /review-annotations ground_truth_json` -- `POST /review-annotations ground_truth_set_value` -- `POST /review-projects review_label_input_mode` -- `POST /review-projects review_label_options` -- `POST /review-projects review_label_selection_mode` -- `POST /run-templates test_case_ids` +- None. diff --git a/api-coverage.toml b/api-coverage.toml index 9a47429..93415ad 100644 --- a/api-coverage.toml +++ b/api-coverage.toml @@ -5,11 +5,11 @@ [snapshot] catalog_url = "https://api.coval.dev/v1/openapi" -reviewed_at = "2026-09-14" -published_operations = 181 +reviewed_at = "2026-10-05" +published_operations = 183 cli_supported_operations = 119 -published_request_fields = 365 -cli_modeled_request_fields = 359 +published_request_fields = 437 +cli_modeled_request_fields = 437 [[allowed_extra]] operation = "POST /test-cases/{test_case_id}/media:upload-url" @@ -376,6 +376,14 @@ reason = "Workspace management remains to be modeled under COVAL-2079." operation = "POST /workspaces/{workspace_id}/archive" reason = "Workspace management remains to be modeled under COVAL-2079." +[[known_gap]] +operation = "POST /review-projects/{project_id}/complete-conversation" +reason = "Project-scoped completion remains to be modeled under COVAL-2079; this explicit write needs a deliberate command workflow rather than being exposed by the parity refresh." + +[[known_gap]] +operation = "POST /sofia/delegation-token" +reason = "Customer API-key delegation token minting remains to be modeled under COVAL-2079 with an explicit security-sensitive CLI workflow." + # ============================================================================= # REQUEST FIELDS # ============================================================================= @@ -391,36 +399,6 @@ reason = "Workspace management remains to be modeled under COVAL-2079." # Pydantic model; coval-ai/backend records the known divergence in # src/services/api/tests/v1/openapi_parity_baseline.txt. -[[known_field_gap]] -operation = "PATCH /run-templates/{run_template_id}" -field = "agent_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `agent_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - -[[known_field_gap]] -operation = "PATCH /run-templates/{run_template_id}" -field = "persona_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `persona_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - -[[known_field_gap]] -operation = "PATCH /run-templates/{run_template_id}" -field = "test_set_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `test_set_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - -[[known_field_gap]] -operation = "POST /run-templates" -field = "agent_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `agent_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - -[[known_field_gap]] -operation = "POST /run-templates" -field = "persona_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `persona_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - -[[known_field_gap]] -operation = "POST /run-templates" -field = "test_set_id" -reason = "Documented but not served: the transport forbids extra fields and takes the plural `test_set_ids` array the CLI already sends, so modeling this name would break the call. Spec drift tracked under COVAL-5825." - # Fields the CLI sends that the published schema does not declare. Most are served # but undocumented, so removing them from the CLI would lose working behavior. @@ -429,21 +407,6 @@ operation = "PATCH /conversations/uploaded/{conversation_id}" field = "audio_reference" reason = "Served but deliberately hidden from the published schema as an internal field; the API still accepts it." -[[allowed_extra_field]] -operation = "PATCH /metrics/{metric_id}" -field = "case_insensitive" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - -[[allowed_extra_field]] -operation = "PATCH /metrics/{metric_id}" -field = "match_mode" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - -[[allowed_extra_field]] -operation = "PATCH /metrics/{metric_id}" -field = "position" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - [[allowed_extra_field]] operation = "PATCH /review-annotations/{annotation_id}" field = "completion_status" @@ -454,26 +417,6 @@ operation = "PATCH /review-annotations/{annotation_id}" field = "status" reason = "Advertised by the CLI but absent from the served contract, which ignores unknown fields, so the flag silently does nothing. Tracked under COVAL-5824." -[[allowed_extra_field]] -operation = "PATCH /run-templates/{run_template_id}" -field = "agent_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." - -[[allowed_extra_field]] -operation = "PATCH /run-templates/{run_template_id}" -field = "persona_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." - -[[allowed_extra_field]] -operation = "PATCH /run-templates/{run_template_id}" -field = "test_set_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." - -[[allowed_extra_field]] -operation = "POST /api-keys" -field = "environment" -reason = "Served but undocumented; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - [[allowed_extra_field]] operation = "POST /conversations/simulated/{simulation_id}/resimulate" field = "dev_id" @@ -483,33 +426,3 @@ reason = "Served but deliberately hidden from the published schema as an interna operation = "POST /conversations/uploaded:submit" field = "audio" reason = "Served but deliberately hidden from the published schema as an internal field; the API still accepts it." - -[[allowed_extra_field]] -operation = "POST /metrics" -field = "case_insensitive" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - -[[allowed_extra_field]] -operation = "POST /metrics" -field = "match_mode" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - -[[allowed_extra_field]] -operation = "POST /metrics" -field = "position" -reason = "Served but undocumented regex-matching field; recorded as spec drift in the backend's own openapi_parity_baseline.txt." - -[[allowed_extra_field]] -operation = "POST /run-templates" -field = "agent_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." - -[[allowed_extra_field]] -operation = "POST /run-templates" -field = "persona_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." - -[[allowed_extra_field]] -operation = "POST /run-templates" -field = "test_set_ids" -reason = "Served but undocumented: the run-template models take these plural arrays. The published singular form is stale spec, tracked under COVAL-5825." diff --git a/src/client/models/metric.rs b/src/client/models/metric.rs index c6c86f8..2a7318f 100644 --- a/src/client/models/metric.rs +++ b/src/client/models/metric.rs @@ -13,6 +13,9 @@ pub struct Metric { pub metric_type: MetricType, #[serde(skip_serializing_if = "Option::is_none")] pub prompt: Option, + /// Agent Judge evidence tools. An empty list disables all tools. + #[serde(skip_serializing_if = "Option::is_none")] + pub enabled_tools: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub categories: Option>, #[serde(skip_serializing_if = "Option::is_none")] @@ -116,6 +119,47 @@ pub struct CreateMetricRequest { pub metric_type: MetricType, #[serde(skip_serializing_if = "Option::is_none")] pub prompt: Option, + /// Agent Judge evidence tools. An empty list disables all tools. + #[serde(skip_serializing_if = "Option::is_none")] + pub enabled_tools: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub judge_mode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub aggregation_method: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub unit: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub detection_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub harmonics_to_noise_ratio_threshold_offset_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub jitter_threshold_multiplier: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub loud_threshold_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub low_pitch_threshold_multiplier: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub mad_z_score_threshold: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub metric_attribute: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub metric_metadata: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub min_fry_segment_seconds: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub pause_detection_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub pitch_change_threshold_hz: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub significant_changes_threshold_hz: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub soft_threshold_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub span_name: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub threshold_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub value_source: Option, #[serde(skip_serializing_if = "Option::is_none")] pub categories: Option>, #[serde(skip_serializing_if = "Option::is_none")] @@ -202,6 +246,47 @@ pub struct UpdateMetricRequest { pub metric_type: Option, #[serde(skip_serializing_if = "Option::is_none")] pub prompt: Option, + /// Agent Judge evidence tools. An empty list disables all tools. + #[serde(skip_serializing_if = "Option::is_none")] + pub enabled_tools: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub judge_mode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub aggregation_method: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub unit: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub detection_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub harmonics_to_noise_ratio_threshold_offset_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub jitter_threshold_multiplier: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub loud_threshold_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub low_pitch_threshold_multiplier: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub mad_z_score_threshold: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub metric_attribute: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub metric_metadata: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub min_fry_segment_seconds: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub pause_detection_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub pitch_change_threshold_hz: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub significant_changes_threshold_hz: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub soft_threshold_db: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub span_name: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub threshold_preset: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub value_source: Option, #[serde(skip_serializing_if = "Option::is_none")] pub categories: Option>, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/src/client/models/persona.rs b/src/client/models/persona.rs index db82bf2..3e87ed1 100644 --- a/src/client/models/persona.rs +++ b/src/client/models/persona.rs @@ -97,6 +97,18 @@ pub struct CreatePersonaRequest { /// `situate_speaker`. #[serde(skip_serializing_if = "Option::is_none")] pub audio_degradation: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub silent_mode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub multi_phone_config: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub initialization_parameters: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub custom_persona_data: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub voice: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub custom_voice_id: Option, /// Tag names. Omitted leaves tags unchanged; an empty list clears them. #[serde(skip_serializing_if = "Option::is_none")] pub tags: Option>, @@ -158,6 +170,44 @@ pub struct UpdatePersonaRequest { skip_serializing_if = "Option::is_none" )] pub audio_degradation: Option>, + // Advanced persona values are stored in metadata. The API uses field + // presence to distinguish an omitted update from an explicit null clear. + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub silent_mode: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub multi_phone_config: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub initialization_parameters: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub custom_persona_data: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub voice: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub custom_voice_id: Option>, /// Tag names. Omitted leaves tags unchanged; an empty list clears them. #[serde(skip_serializing_if = "Option::is_none")] pub tags: Option>, diff --git a/src/client/models/review_annotation.rs b/src/client/models/review_annotation.rs index 4ff2344..9ba6fe5 100644 --- a/src/client/models/review_annotation.rs +++ b/src/client/models/review_annotation.rs @@ -98,7 +98,11 @@ pub struct CreateReviewAnnotationRequest { #[serde(skip_serializing_if = "Option::is_none")] pub reviewer_notes: Option, #[serde(skip_serializing_if = "Option::is_none")] - pub priority: Option, + pub annotations: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ground_truth_json: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ground_truth_set_value: Option>, } #[derive(Debug, Default, Serialize, Deserialize)] @@ -112,6 +116,12 @@ pub struct UpdateReviewAnnotationRequest { #[serde(skip_serializing_if = "Option::is_none")] pub reviewer_notes: Option, #[serde(skip_serializing_if = "Option::is_none")] + pub annotations: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ground_truth_json: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ground_truth_set_value: Option>, + #[serde(skip_serializing_if = "Option::is_none")] pub priority: Option, #[serde(skip_serializing_if = "Option::is_none")] pub assignee: Option, diff --git a/src/client/models/review_project.rs b/src/client/models/review_project.rs index 5fcb701..288ef49 100644 --- a/src/client/models/review_project.rs +++ b/src/client/models/review_project.rs @@ -65,6 +65,12 @@ pub struct CreateReviewProjectRequest { /// Only takes effect on a collaborative project. #[serde(skip_serializing_if = "Option::is_none")] pub enforced_collaboration: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_input_mode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_options: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_selection_mode: Option, } #[derive(Debug, Default, Serialize, Deserialize)] @@ -80,6 +86,8 @@ pub struct UpdateReviewProjectRequest { #[serde(skip_serializing_if = "Option::is_none")] pub linked_metric_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] + pub project_type: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub notifications: Option, #[serde(skip_serializing_if = "Option::is_none")] pub opted_out_assignees: Option>, @@ -102,6 +110,12 @@ pub struct UpdateReviewProjectRequest { /// Only takes effect on a collaborative project. #[serde(skip_serializing_if = "Option::is_none")] pub enforced_collaboration: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_input_mode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_options: Option>, + #[serde(skip_serializing_if = "Option::is_none")] + pub review_label_selection_mode: Option, } #[derive(Debug, Deserialize)] diff --git a/src/client/models/run.rs b/src/client/models/run.rs index d289e04..9d67d4c 100644 --- a/src/client/models/run.rs +++ b/src/client/models/run.rs @@ -89,8 +89,6 @@ pub struct LaunchRunRequest { #[serde(skip_serializing_if = "Option::is_none")] pub mutation_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] - pub persona_metrics: Option>, - #[serde(skip_serializing_if = "Option::is_none")] pub options: Option, #[serde(skip_serializing_if = "Option::is_none")] pub metadata: Option, diff --git a/src/client/models/run_template.rs b/src/client/models/run_template.rs index d47cfe7..e266d0a 100644 --- a/src/client/models/run_template.rs +++ b/src/client/models/run_template.rs @@ -62,6 +62,8 @@ pub struct CreateRunTemplateRequest { pub persona_ids: Vec, pub test_set_ids: Vec, #[serde(skip_serializing_if = "Option::is_none")] + pub test_case_ids: Option>, + #[serde(skip_serializing_if = "Option::is_none")] pub metric_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub mutation_ids: Option>, @@ -93,6 +95,8 @@ pub struct UpdateRunTemplateRequest { #[serde(skip_serializing_if = "Option::is_none")] pub test_set_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] + pub test_case_ids: Option>, + #[serde(skip_serializing_if = "Option::is_none")] pub metric_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub mutation_ids: Option>, diff --git a/src/commands/review_annotations.rs b/src/commands/review_annotations.rs index 2f865c4..a8db08e 100644 --- a/src/commands/review_annotations.rs +++ b/src/commands/review_annotations.rs @@ -75,9 +75,6 @@ pub struct CreateArgs { /// Reviewer notes #[arg(long)] notes: Option, - /// Annotation priority - #[arg(long, value_enum)] - priority: Option, } #[derive(Args)] @@ -183,7 +180,6 @@ pub async fn execute( )?; input_json::insert(&mut input, "ground_truth_subvalues_by_timestamp", subvalues)?; input_json::insert(&mut input, "reviewer_notes", args.notes)?; - input_json::insert(&mut input, "priority", args.priority)?; let req: CreateReviewAnnotationRequest = input_json::finish(input)?; let annotation = client.review_annotations().create(req).await?; emit_one_with_actions( diff --git a/src/commands/runs.rs b/src/commands/runs.rs index be8b49c..d4b54bb 100644 --- a/src/commands/runs.rs +++ b/src/commands/runs.rs @@ -204,7 +204,6 @@ pub async fn execute(cmd: RunCommands, client: &CovalClient, ctx: &OutputContext metric_ids: args.metric_ids, mutation_id: args.mutation_id, mutation_ids: args.mutation_ids, - persona_metrics: None, options, metadata, config_overrides, diff --git a/tests/cli_tests.rs b/tests/cli_tests.rs index 5c9a2e4..b626442 100644 --- a/tests/cli_tests.rs +++ b/tests/cli_tests.rs @@ -1882,6 +1882,12 @@ fn newly_modeled_persona_fields() -> Value { "hold_music_timeout_seconds": 45.0, "situate_speaker": "speakerphone-easy", "interruption_rate": "HIGH", + "silent_mode": false, + "multi_phone_config": {"phone_number_index": 1, "phone_number_name": "Primary"}, + "initialization_parameters": {"account_tier": "premium"}, + "custom_persona_data": "{\"customer\":\"example\"}", + "voice": "marin", + "custom_voice_id": "voice_custom_123", "tags": ["support", "noisy"] }) } @@ -1985,7 +1991,13 @@ async fn test_personas_update_forwards_an_explicit_null_to_clear() { "audio_degradation": null, "voice_volume": null, "voice_speed": null, - "hold_music_timeout_seconds": null + "hold_music_timeout_seconds": null, + "silent_mode": null, + "multi_phone_config": null, + "initialization_parameters": null, + "custom_persona_data": null, + "voice": null, + "custom_voice_id": null }) .to_string(), ) @@ -2001,6 +2013,12 @@ async fn test_personas_update_forwards_an_explicit_null_to_clear() { "voice_volume", "voice_speed", "hold_music_timeout_seconds", + "silent_mode", + "multi_phone_config", + "initialization_parameters", + "custom_persona_data", + "voice", + "custom_voice_id", ] { assert!(body.get(key).is_some(), "{key} must be sent"); assert!(body[key].is_null(), "{key} must be sent as null"); @@ -3244,6 +3262,48 @@ async fn test_run_templates_update_forwards_tags() { assert_eq!(capture.take()["tags"], json!(["nightly"])); } +#[tokio::test] +async fn test_run_templates_forward_test_case_ids() { + for (method_name, command, path_value, required) in [ + ("POST", "create", "/v1/run-templates", true), + ("PATCH", "update", "/v1/run-templates/rt123", false), + ] { + let mock_server = MockServer::start().await; + let capture = BodyCapture::default(); + Mock::given(method(method_name)) + .and(path(path_value)) + .and(capture.clone()) + .respond_with(ResponseTemplate::new(200).set_body_json(run_template_response())) + .mount(&mock_server) + .await; + let mut input = json!({"test_case_ids": ["tc1", "tc2"]}); + if required { + let object = input.as_object_mut().unwrap(); + object.insert("display_name".into(), json!("My Template")); + object.insert("agent_ids".into(), json!(["agent1"])); + object.insert("persona_ids".into(), json!(["persona1"])); + object.insert("test_set_ids".into(), json!(["ts123"])); + } + let mut command_line = coval(); + command_line + .arg("--api-key") + .arg("test_key") + .arg("--api-url") + .arg(mock_server.uri()) + .arg("run-templates") + .arg(command); + if !required { + command_line.arg("rt123"); + } + command_line + .arg("--input-json") + .arg(input.to_string()) + .assert() + .success(); + assert_eq!(capture.take()["test_case_ids"], json!(["tc1", "tc2"])); + } +} + #[tokio::test] async fn test_uploaded_conversations_submit_forwards_tags() { let mock_server = MockServer::start().await; @@ -4408,6 +4468,66 @@ async fn test_review_projects_get() { .stdout(predicate::str::contains("proj123")); } +#[tokio::test] +async fn test_review_annotations_forward_structured_ground_truth_fields() { + for (method_name, command, path_value, required) in [ + ("POST", "create", "/v1/review-annotations", true), + ("PATCH", "update", "/v1/review-annotations/ann123", false), + ] { + let mock_server = MockServer::start().await; + let capture = BodyCapture::default(); + Mock::given(method(method_name)) + .and(path(path_value)) + .and(capture.clone()) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "review_annotation": { + "id": "ann123", "simulation_output_id": "so123", "metric_id": "met123", + "assignee": "reviewer@example.com", "status": "ACTIVE", + "completion_status": "COMPLETED", "priority": "PRIORITY_STANDARD", + "create_time": "2025-01-15T10:30:00Z", "update_time": "2025-01-15T11:00:00Z" + } + }))) + .mount(&mock_server) + .await; + + let mut input = json!({ + "annotations": {"label": "accurate"}, + "ground_truth_json": {"score": 1}, + "ground_truth_set_value": ["accurate", "complete"] + }); + if required { + let object = input.as_object_mut().unwrap(); + object.insert("simulation_output_id".into(), json!("so123")); + object.insert("metric_id".into(), json!("met123")); + object.insert("assignee".into(), json!("reviewer@example.com")); + } + let mut command_line = coval(); + command_line + .arg("--api-key") + .arg("test_key") + .arg("--api-url") + .arg(mock_server.uri()) + .arg("review-annotations") + .arg(command); + if !required { + command_line.arg("ann123"); + } + command_line + .arg("--input-json") + .arg(input.to_string()) + .assert() + .success(); + + let body = capture.take(); + assert_eq!(body["annotations"], json!({"label": "accurate"})); + assert_eq!(body["ground_truth_json"], json!({"score": 1})); + assert_eq!( + body["ground_truth_set_value"], + json!(["accurate", "complete"]) + ); + } +} + fn review_project_response() -> Value { json!({ "review_project": { @@ -4478,6 +4598,64 @@ async fn test_review_projects_create_forwards_every_modeled_field() { assert_eq!(body["enforced_collaboration"], true); } +#[tokio::test] +async fn test_review_projects_forward_label_configuration() { + let fields = json!({ + "review_label_input_mode": "OPTION_OR_CUSTOM", + "review_label_options": ["pass", "fail"], + "review_label_selection_mode": "MULTIPLE" + }); + for (method_name, command, path_value, required) in [ + ("POST", "create", "/v1/review-projects", true), + ("PATCH", "update", "/v1/review-projects/proj456", false), + ] { + let mock_server = MockServer::start().await; + let capture = BodyCapture::default(); + Mock::given(method(method_name)) + .and(path(path_value)) + .and(capture.clone()) + .respond_with(ResponseTemplate::new(200).set_body_json(review_project_response())) + .mount(&mock_server) + .await; + let mut input = fields.clone(); + if required { + let object = input.as_object_mut().unwrap(); + object.insert("display_name".into(), json!("New Project")); + object.insert("assignees".into(), json!(["alice@example.com"])); + object.insert("linked_simulation_ids".into(), json!(["sim1"])); + object.insert("linked_metric_ids".into(), json!(["met1"])); + } else { + input + .as_object_mut() + .unwrap() + .insert("project_type".into(), json!("PROJECT_COLLABORATIVE")); + } + let mut command_line = coval(); + command_line + .arg("--api-key") + .arg("test_key") + .arg("--api-url") + .arg(mock_server.uri()) + .arg("review-projects") + .arg(command); + if !required { + command_line.arg("proj456"); + } + command_line + .arg("--input-json") + .arg(input.to_string()) + .assert() + .success(); + let body = capture.take(); + for (key, expected) in fields.as_object().unwrap() { + assert_eq!(&body[key], expected, "field {key} must reach the API"); + } + if !required { + assert_eq!(body["project_type"], "PROJECT_COLLABORATIVE"); + } + } +} + #[tokio::test] async fn test_review_projects_update_sends_membership_deltas() { let mock_server = MockServer::start().await; @@ -4871,6 +5049,26 @@ fn metric_response_body() -> Value { /// struct that stops declaring one fails here rather than dropping it in silence. fn newly_modeled_metric_fields() -> Value { json!({ + "enabled_tools": ["get_transcript", "get_trace_spans"], + "judge_mode": "AGENTIC", + "aggregation_method": "AVERAGE", + "unit": "s", + "detection_preset": "normal", + "harmonics_to_noise_ratio_threshold_offset_db": -10.0, + "jitter_threshold_multiplier": 2.0, + "loud_threshold_db": -8.0, + "low_pitch_threshold_multiplier": 0.7, + "mad_z_score_threshold": 3.0, + "metric_attribute": "http.status_code", + "metric_metadata": {"derived_type": "DERIVED_AGGREGATE"}, + "min_fry_segment_seconds": 0.2, + "pause_detection_preset": "strict", + "pitch_change_threshold_hz": 25.0, + "significant_changes_threshold_hz": 30.0, + "soft_threshold_db": -35.0, + "span_name": "http.request", + "threshold_preset": "lenient", + "value_source": "attribute", "max_silence_duration_seconds": 4.5, "min_silence_gap_seconds": 0.75, "frequency_threshold": 2.0, From 89a66a6cdc9d300139cf82510d3383fa4337d995 Mon Sep 17 00:00:00 2001 From: Callum Reid Date: Mon, 5 Oct 2026 10:32:10 -0700 Subject: [PATCH 3/3] [COVAL-7144] Forward explicit nulls on new PATCH fields and release 0.9.0 - Run-template test_case_ids, metric aggregation_method and unit, and the review-annotation structured ground-truth fields now forward an explicit null, which the API uses to clear them. Plain Option dropped the null, so a test-case subset could never be removed (the API rejects an empty list). - Use caller data for the initialization_parameters test value; it is a persona setting, not an account attribute. - Bump to 0.9.0: review-annotations create --priority is removed. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 2 +- Cargo.toml | 2 +- src/client/models/metric.rs | 18 +++-- src/client/models/review_annotation.rs | 25 +++++-- src/client/models/run_template.rs | 9 ++- tests/cli_tests.rs | 97 +++++++++++++++++++++++++- 6 files changed, 138 insertions(+), 15 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index e566dcd..6e24a24 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -278,7 +278,7 @@ checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" [[package]] name = "coval" -version = "0.8.3" +version = "0.9.0" dependencies = [ "anyhow", "assert_cmd", diff --git a/Cargo.toml b/Cargo.toml index c59499b..b461b0f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "coval" -version = "0.8.3" +version = "0.9.0" edition = "2021" description = "CLI for Coval AI agent evaluation platform" license = "MIT" diff --git a/src/client/models/metric.rs b/src/client/models/metric.rs index 2a7318f..be169a6 100644 --- a/src/client/models/metric.rs +++ b/src/client/models/metric.rs @@ -251,10 +251,20 @@ pub struct UpdateMetricRequest { pub enabled_tools: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub judge_mode: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub aggregation_method: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub unit: Option, + /// SQL and custom-trace aggregation; an explicit null resets it to the default. + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub aggregation_method: Option>, + /// Display unit; an explicit null clears it. + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub unit: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub detection_preset: Option, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/src/client/models/review_annotation.rs b/src/client/models/review_annotation.rs index 9ba6fe5..bbacd2e 100644 --- a/src/client/models/review_annotation.rs +++ b/src/client/models/review_annotation.rs @@ -115,12 +115,25 @@ pub struct UpdateReviewAnnotationRequest { pub ground_truth_subvalues_by_timestamp: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub reviewer_notes: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub annotations: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub ground_truth_json: Option, - #[serde(skip_serializing_if = "Option::is_none")] - pub ground_truth_set_value: Option>, + /// Structured ground truth. An explicit null clears each of these fields. + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub annotations: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub ground_truth_json: Option>, + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub ground_truth_set_value: Option>>, #[serde(skip_serializing_if = "Option::is_none")] pub priority: Option, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/src/client/models/run_template.rs b/src/client/models/run_template.rs index e266d0a..70f37e3 100644 --- a/src/client/models/run_template.rs +++ b/src/client/models/run_template.rs @@ -94,8 +94,13 @@ pub struct UpdateRunTemplateRequest { pub persona_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] pub test_set_ids: Option>, - #[serde(skip_serializing_if = "Option::is_none")] - pub test_case_ids: Option>, + /// Test-case subset. An explicit null clears it; the API rejects an empty list. + #[serde( + default, + deserialize_with = "super::explicit_option", + skip_serializing_if = "Option::is_none" + )] + pub test_case_ids: Option>>, #[serde(skip_serializing_if = "Option::is_none")] pub metric_ids: Option>, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/tests/cli_tests.rs b/tests/cli_tests.rs index b626442..867dd54 100644 --- a/tests/cli_tests.rs +++ b/tests/cli_tests.rs @@ -1884,7 +1884,7 @@ fn newly_modeled_persona_fields() -> Value { "interruption_rate": "HIGH", "silent_mode": false, "multi_phone_config": {"phone_number_index": 1, "phone_number_name": "Primary"}, - "initialization_parameters": {"account_tier": "premium"}, + "initialization_parameters": {"caller_name": "Jordan", "loyalty_status": "gold"}, "custom_persona_data": "{\"customer\":\"example\"}", "voice": "marin", "custom_voice_id": "voice_custom_123", @@ -7672,3 +7672,98 @@ async fn test_simulated_conversation_commands_use_canonical_routes() { .assert() .success(); } + +async fn patch_body_for(resource_path: &str, response: Value, args: &[&str]) -> Value { + let mock_server = MockServer::start().await; + let capture = BodyCapture::default(); + + Mock::given(method("PATCH")) + .and(path(resource_path)) + .and(header("X-API-Key", "test_key")) + .and(capture.clone()) + .respond_with(ResponseTemplate::new(200).set_body_json(response)) + .mount(&mock_server) + .await; + + coval_with_api(&mock_server).args(args).assert().success(); + + capture.take() +} + +fn assert_sent_as_null(body: &Value, keys: &[&str]) { + // The API clears a stored value only when the key is present and null, so an + // omitted key would silently turn "clear this" into "leave it alone". + for key in keys { + assert!(body.get(key).is_some(), "{key} must be sent"); + assert!(body[key].is_null(), "{key} must be sent as null"); + } +} + +#[tokio::test] +async fn test_run_templates_update_clears_test_case_ids_with_an_explicit_null() { + let body = patch_body_for( + "/v1/run-templates/rt123", + run_template_response(), + &[ + "run-templates", + "update", + "rt123", + "--input-json", + r#"{"test_case_ids":null}"#, + ], + ) + .await; + + assert_sent_as_null(&body, &["test_case_ids"]); +} + +#[tokio::test] +async fn test_metrics_update_clears_aggregation_and_unit_with_an_explicit_null() { + let body = patch_body_for( + "/v1/metrics/met1", + metric_response_body(), + &[ + "metrics", + "update", + "met1", + "--input-json", + r#"{"aggregation_method":null,"unit":null}"#, + ], + ) + .await; + + assert_sent_as_null(&body, &["aggregation_method", "unit"]); +} + +#[tokio::test] +async fn test_review_annotations_update_clears_structured_ground_truth_with_an_explicit_null() { + let body = patch_body_for( + "/v1/review-annotations/ann123", + json!({ + "review_annotation": { + "id": "ann123", + "simulation_output_id": "so123", + "metric_id": "met123", + "assignee": "reviewer@example.com", + "status": "ACTIVE", + "completion_status": "COMPLETED", + "priority": "PRIORITY_PRIMARY", + "create_time": "2025-01-15T10:30:00Z", + "update_time": "2025-01-15T11:00:00Z" + } + }), + &[ + "review-annotations", + "update", + "ann123", + "--input-json", + r#"{"annotations":null,"ground_truth_json":null,"ground_truth_set_value":null}"#, + ], + ) + .await; + + assert_sent_as_null( + &body, + &["annotations", "ground_truth_json", "ground_truth_set_value"], + ); +}