From b4ce62cd44e426a0bb161e2739a4d366df926d4e Mon Sep 17 00:00:00 2001 From: "xiaotong.wang" <18648483389@163.com> Date: Wed, 2 Sep 2026 14:57:23 +0800 Subject: [PATCH 01/12] =?UTF-8?q?docs:=20=E6=9B=B4=E6=96=B0=E6=89=A7?= =?UTF-8?q?=E8=A1=8C=E9=85=8D=E7=BD=AE=E4=BA=92=E6=96=A5=E7=BA=A6=E6=9D=9F?= =?UTF-8?q?=E5=8F=8A=E6=9E=B6=E6=9E=84=E6=96=87=E6=A1=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 移除 1.4 迁移说明,明确 execution 单命令与 steps 形态互斥 --- docs/design.md | 17 --- docs/design_en.md | 21 +--- docs/user_manual.md | 7 ++ docs/user_manual_en.md | 7 ++ examples/skill/SKILL.md | 2 + examples/skill/references/user_manual.md | 7 ++ src/symtest/__init__.py | 2 +- src/symtest/cli.py | 10 ++ src/symtest/config/config_schema.py | 46 +++++++- src/symtest/config/normalize.py | 43 +++++++ src/symtest/core/base_runner.py | 19 --- src/symtest/core/config_loader.py | 54 ++++----- src/symtest/core/orchestration/sequence.py | 34 ++---- src/symtest/core/orchestration/single.py | 15 +-- src/symtest/core/parallel_runner.py | 2 - src/symtest/core/process_worker.py | 8 +- src/symtest/core/test_case.py | 108 ++++++++++++++---- src/symtest/reporting/__init__.py | 11 +- src/symtest/reporting/diagnosis.py | 67 ++++++++++- .../tui/controllers/case_controller.py | 20 +++- src/symtest/tui/widgets/steps_editor.py | 10 +- tests/integration/test_sequence.py | 16 +-- tests/unit/core/test_architecture_guard.py | 10 +- tests/unit/core/test_case_env.py | 8 +- tests/unit/core/test_config_loader.py | 6 +- .../core/test_config_loader_new_features.py | 23 ++-- .../unit/core/test_execution_new_features.py | 58 +++++----- tests/unit/core/test_model_v2.py | 18 +-- tests/unit/core/test_orchestration_accept.py | 14 ++- tests/unit/core/test_process_worker.py | 12 +- tests/unit/core/test_sequence_state.py | 6 +- tests/unit/test_migrate.py | 10 +- tests/unit/tui/conftest.py | 6 +- tests/unit/tui/test_case_controller.py | 30 +++-- tests/unit/tui/test_tui_run_test.py | 6 +- tests/unit/tui/test_widgets_data.py | 4 +- 36 files changed, 474 insertions(+), 263 deletions(-) create mode 100644 src/symtest/config/normalize.py diff --git a/docs/design.md b/docs/design.md index 2bf9b0e..0d9d636 100644 --- a/docs/design.md +++ b/docs/design.md @@ -155,9 +155,6 @@ PathResolver 解析(系统命令直通、shell builtin 平台包装、复合 → 超时 kill 整个进程组 → 返回附带 `next_action_hint` 的结构化结果。 -当前实现的上述行为仍聚合在一处;1.4 将按 §10 宪法重排为 -Executor / Validator / Orchestration 三段(迁移明细见 docs/design_1_4.md)。 - ### 4.6 断言与文件比较集成 - `compare_files` 是一等断言,经 ComparatorFactory 按类型分发(详见 §6) @@ -252,7 +249,6 @@ TUI 与 runner 共用同一解析器;宽松的展示形态在 TUI 侧自行处 > **地位与效力**:本节自 Symtest 1.4 Phase 0 评审定稿,是本项目的核心架构契约, > 对所有后续 feature 具有最高约束力——任何功能需求先对照本宪法确定归属,再写代码。 > 修订宪法必须在变更说明中显式指出所放宽/违反的条款及理由。 -> 定稿前的演进推导见开发阶段文档 docs/design_1_4.md。 ### 10.1 数据流主线 @@ -373,16 +369,3 @@ Reporter 的合法输入只能是 Result 类型;`next_action_hint` 属于 resu `reporter`; - guard 测试纳入常规回归套件(`python tests\run_all.py`),违规即失败; - 冲突裁决次序:guard 测试 > 本节文字 > 个人偏好。 - -### 10.6 现状差距与收敛路径 - -定稿时刻(1.4 Phase 0)现行实现与本宪法的已知偏差如下,将在 1.4 对应 Phase 内 -逐项收敛(迁移明细见 docs/design_1_4.md): - -| 现状 | 违反 | 收敛 | -|---|---|---| -| `execution.py::validate_result` 住在执行层 | 原则 2 | Phase 2 迁入 `validation/validator.py` | -| `next_action_hint` 在 execution 层构造 | 原则 5 | Phase 2 迁入 `reporting/diagnosis.py` | -| retry 循环在 executor 内部 | 原则 2 / 4 | Phase 2 上移编排层 | -| Validator 侧 update_baseline 写文件 | 原则 3 | Phase 2 改为 runner 独立 accept 步骤 | -| `parse_test_cases` TUI mode 后门 | 原则 6 | Phase 2 拆除,单一解析器 | diff --git a/docs/design_en.md b/docs/design_en.md index 71acdeb..eeb6b02 100644 --- a/docs/design_en.md +++ b/docs/design_en.md @@ -164,10 +164,6 @@ stdout / stderr / returncode / duration) → assertions (return_code / contains matches / compare_files) → retries on failure up to `retry_count` → kills the whole process group on timeout → structured result with `next_action_hint`. -Today these behaviors are aggregated in one place; 1.4 rearranges them into -Executor / Validator / Orchestration per the §10 Constitution (migration details -in docs/design_1_4.md). - ### 4.6 Assertions and File Comparison Integration - `compare_files` is a first-class assertion dispatched via ComparatorFactory (see §6) @@ -266,8 +262,7 @@ TUI side itself (§10 Principle 6). > project's core architectural contract with supreme authority over all future > features — resolve ownership against this constitution before writing any code. > Amending it requires explicitly naming the clause being relaxed/waived and the -> rationale in the change description. Pre-ratification evolution history lives in -> the development-phase document docs/design_1_4.md. +> rationale in the change description. ### 10.1 Data Flow Spine @@ -398,17 +393,3 @@ architecture guard tests enforced by CI: - Guard tests are part of the regular regression suite (`python tests\run_all.py`); violations fail the build; - Conflict resolution order: guard tests > this text > personal preference. - -### 10.6 Current Gap to This Constitution - -Known deviations of the current implementation from this constitution as of -ratification (1.4 Phase 0). Each item converges during the corresponding 1.4 -phase (migration details in docs/design_1_4.md): - -| Current state | Violated | Convergence | -|---|---|---| -| `execution.py::validate_result` lives in the execution layer | Principle 2 | Phase 2: move to `validation/validator.py` | -| `next_action_hint` built in the execution layer | Principle 5 | Phase 2: move to `reporting/diagnosis.py` | -| Retry loop inside the executor | Principles 2 / 4 | Phase 2: lift to orchestration | -| update_baseline writes files inside validation | Principle 3 | Phase 2: runner-side independent accept step | -| `parse_test_cases` TUI-mode back door | Principle 6 | Phase 2: remove; single parser | diff --git a/docs/user_manual.md b/docs/user_manual.md index 429bd8c..065515d 100644 --- a/docs/user_manual.md +++ b/docs/user_manual.md @@ -275,6 +275,13 @@ test_cases: | `expected.output_matches` | 否 | 输出需匹配的正则表达式(单个字符串) | | `expected.compare_files` | 否 | 文件比较断言列表,见下文 | +### execution 二选一(互斥) + +`execution` 有两种形态,**二选一**且互斥,同时声明两种形态属于配置错误: + +- **单命令形态**:`command` + `args`(`timeout` / `retry_count` / `env` 可选) +- **序列形态**:`steps`(列表,每个 step 为 `command + args + expected`);`command` / `args` 不得与 `steps` 并存 + ## Case 级环境变量(env) 通过 `execution` 内与 `command`、`steps` 同级的 `env` 字段,可为单个用例注入环境变量,仅在该用例(序列模式为所有 step)的子进程内生效,不影响其他用例。 diff --git a/docs/user_manual_en.md b/docs/user_manual_en.md index 9a86f88..6328e55 100644 --- a/docs/user_manual_en.md +++ b/docs/user_manual_en.md @@ -275,6 +275,13 @@ test_cases: | `expected.output_matches` | No | Regex pattern the output must match (single string) | | `expected.compare_files` | No | File comparison assertions list, see below | +### execution — pick one form (mutually exclusive) + +`execution` has two forms, **choose exactly one**; declaring both is a configuration error: + +- **Single-command form**: `command` + `args` (with optional `timeout` / `retry_count` / `env`) +- **Steps form**: `steps` (a list; each step is `command + args + expected`); `command` / `args` must not be declared alongside `steps` + ### File Comparison Assertions (compare_files) Declare one or more file comparison rules in `expected.compare_files`. The framework automatically uses the corresponding comparator to diff the actual output file against the baseline after command execution. The test passes only when all comparisons pass; these coexist with `return_code`, `output_contains`, and other assertions. diff --git a/examples/skill/SKILL.md b/examples/skill/SKILL.md index 5b19783..5faa73a 100644 --- a/examples/skill/SKILL.md +++ b/examples/skill/SKILL.md @@ -118,6 +118,8 @@ starting point. (`execution.command` form). - **Step sequence** (`execution.steps`): multiple ordered commands, fail-fast. Use when output of step N is input to step N+1. +- The two forms are mutually exclusive: declaring both `command`/`args` and + `steps` in one `execution` block is a configuration error. ### When to use `import` (config splitting) - When the config grows large (>30 cases) or cases naturally group by module. diff --git a/examples/skill/references/user_manual.md b/examples/skill/references/user_manual.md index 429bd8c..065515d 100644 --- a/examples/skill/references/user_manual.md +++ b/examples/skill/references/user_manual.md @@ -275,6 +275,13 @@ test_cases: | `expected.output_matches` | 否 | 输出需匹配的正则表达式(单个字符串) | | `expected.compare_files` | 否 | 文件比较断言列表,见下文 | +### execution 二选一(互斥) + +`execution` 有两种形态,**二选一**且互斥,同时声明两种形态属于配置错误: + +- **单命令形态**:`command` + `args`(`timeout` / `retry_count` / `env` 可选) +- **序列形态**:`steps`(列表,每个 step 为 `command + args + expected`);`command` / `args` 不得与 `steps` 并存 + ## Case 级环境变量(env) 通过 `execution` 内与 `command`、`steps` 同级的 `env` 字段,可为单个用例注入环境变量,仅在该用例(序列模式为所有 step)的子进程内生效,不影响其他用例。 diff --git a/src/symtest/__init__.py b/src/symtest/__init__.py index 49f83a1..869adc9 100644 --- a/src/symtest/__init__.py +++ b/src/symtest/__init__.py @@ -16,7 +16,7 @@ setup_console_logging(level=logging.DEBUG) """ -__version__ = "1.4.0" +__version__ = "1.4.1" __author__ = "Xiaotong Wang" __email__ = "xiaotongwang98@gmail.com" diff --git a/src/symtest/cli.py b/src/symtest/cli.py index b8c7ee0..8eb5803 100644 --- a/src/symtest/cli.py +++ b/src/symtest/cli.py @@ -15,6 +15,7 @@ from pathlib import Path from .logging_config import setup_console_logging +from .reporting.diagnosis import attach_next_action_hints from .runners import JSONRunner, ParallelJSONRunner, ParallelYAMLRunner, YAMLRunner from .utils.report_generator import ReportGenerator from .utils.junit_xml_writer import write_junit_xml @@ -396,6 +397,15 @@ def run_tests(args): # Output results using ReportGenerator and honor --output-format if hasattr(runner, 'results'): results = runner.results + + # ── Reporting 装配点(原则 5):next_action_hint 在报告输出前 + # 按 failure_kind 填充,orchestration 只产出失败结论 ── + attach_next_action_hints( + results, + update_baseline=bool(getattr(args, 'update_baseline', False)), + config_path=str(config_file), + ) + output_format = getattr(args, 'output_format', 'text') if output_format == 'json': diff --git a/src/symtest/config/config_schema.py b/src/symtest/config/config_schema.py index 08c1d5b..c2d5f7f 100644 --- a/src/symtest/config/config_schema.py +++ b/src/symtest/config/config_schema.py @@ -164,11 +164,24 @@ }, "executionSpec": { "type": "object", - "additionalProperties": False, "description": ( "Execution semantics (v2): single-command shorthand " "(command/args/timeout/retry_count/env) OR the full steps form " - "(steps[]) — 二选一. Case-level env applies to all steps." + "(steps[]) — 二选一,由 oneOf 强制互斥. Case-level env applies " + "to all steps. Declaring both forms is a schema violation." + ), + "oneOf": [ + {"$ref": "#/$defs/commandExecution"}, + {"$ref": "#/$defs/stepsExecution"}, + ], + }, + "commandExecution": { + "type": "object", + "required": ["command", "args"], + "additionalProperties": False, + "description": ( + "Single-command shorthand form: requires command + args; " + "the steps form must not be declared alongside." ), "properties": { "command": { @@ -201,12 +214,41 @@ "and scheduler-injected environment variables." ), }, + }, + }, + "stepsExecution": { + "type": "object", + "required": ["steps"], + "additionalProperties": False, + "description": ( + "Full steps form: requires steps; command/args must not be " + "declared alongside." + ), + "properties": { "steps": { "type": "array", "minItems": 1, "items": {"$ref": "#/$defs/step"}, "description": "Steps run in order with fail-fast semantics.", }, + "timeout": { + "type": ["number", "null"], + "description": "Timeout in seconds (default 3600); null = no limit.", + }, + "retry_count": { + "type": "integer", + "minimum": 0, + "description": "Retries after the first failure; passing after retry marks the result flaky.", + }, + "env": { + "type": "object", + "additionalProperties": {"type": ["string", "number", "boolean"]}, + "description": ( + "Case-level environment variables injected into every " + "step (subprocess). Overrides setup-level and " + "scheduler-injected environment variables." + ), + }, }, }, "schedulingSpec": { diff --git a/src/symtest/config/normalize.py b/src/symtest/config/normalize.py new file mode 100644 index 0000000..ad7bc08 --- /dev/null +++ b/src/symtest/config/normalize.py @@ -0,0 +1,43 @@ +"""Draft 配置规范化(宽松形态归一,1.4 原则 6 的 DSL 层配套)。 + +core parser 全系统唯一且只接受 canonical TestCase;半成品/草稿配置 +(TUI 编辑中间态、v1 迁移输出等)先经 ``normalize_draft_config`` 补齐 +必填字段缺省值,再调用严格 ``core.config_loader.parse_test_cases``。 + +本模块属于 ``config`` 层(DSL 规范化),不在 ``core`` 内 —— core 不提供 +任何宽松解析形态。 +""" +from __future__ import annotations + +import copy +from typing import Any, Dict + + +def normalize_draft_config(config: Dict[str, Any]) -> Dict[str, Any]: + """Normalize half-finished draft cases into canonical config dicts. + + 为缺字段的 case 补齐 canonical 必填项的缺省值(深拷贝,不改动入参): + + - case 级:``name`` / ``execution`` / ``expected``; + - 单命令形态:``execution.command`` / ``execution.args``; + - steps 形态:每个 step 的 ``command`` / ``args`` / ``expected``。 + + ``import`` 引用项原样跳过(由 import 展开管线处理)。 + """ + normalized = copy.deepcopy(config) + for case in normalized.get("test_cases", []): + if not isinstance(case, dict) or "import" in case: + continue + case.setdefault("name", "") + execution = case.setdefault("execution", {}) + if "steps" in execution: + for step in execution.get("steps", []): + if isinstance(step, dict): + step.setdefault("command", "") + step.setdefault("args", []) + step.setdefault("expected", {}) + else: + execution.setdefault("command", "") + execution.setdefault("args", []) + case.setdefault("expected", {}) + return normalized diff --git a/src/symtest/core/base_runner.py b/src/symtest/core/base_runner.py index 9d38c36..9ef654c 100644 --- a/src/symtest/core/base_runner.py +++ b/src/symtest/core/base_runner.py @@ -176,7 +176,6 @@ def run_tests(self) -> bool: "description": case.description or None, "tags": case.tags or [], } - self._fill_hint_command(skip_result, case.name) self.results["details"].append(skip_result) logger.warning("⊘ Test %d skipped: %s", i, case.name) logger.warning(" Reason: %s", skip_result["message"]) @@ -191,7 +190,6 @@ def run_tests(self) -> bool: result["expected"] = case.expected if case.expected else None result["description"] = case.description or None result["tags"] = case.tags or [] - self._fill_hint_command(result, case.name) self.results["details"].append(result) duration = result.get("duration", 0) @@ -298,23 +296,6 @@ def _topological_order(self) -> List[TestCase]: return result - def _fill_hint_command(self, result: Dict[str, Any], case_name: str) -> None: - """Fill in the concrete CLI command inside ``next_action_hint``. - - The execution layer attaches the hint with ``command=None`` because it - does not know the config file path; the runner does. - """ - hint = result.get("next_action_hint") - if not hint or hint.get("command"): - return - config = str(self.config_path) - if hint.get("action") == "update_baseline": - hint["command"] = ( - f'symtest run "{config}" --update-baseline -t "{case_name}"' - ) - else: - hint["command"] = f'symtest run "{config}" -t "{case_name}"' - def _apply_xfail_status(self, result: Dict[str, Any], case: "TestCase") -> None: """Apply xfail (expected failure) status mapping to a test result. diff --git a/src/symtest/core/config_loader.py b/src/symtest/core/config_loader.py index fa3411c..a5cdfd7 100644 --- a/src/symtest/core/config_loader.py +++ b/src/symtest/core/config_loader.py @@ -17,7 +17,7 @@ from pathlib import Path from typing import Any, Dict, List, Optional, Tuple -from .test_case import TestCase, TestCaseStep +from .test_case import TestCase, TestStep from ..utils.path_resolver import resolve_paths logger = logging.getLogger("symtest.core.config_loader") @@ -94,8 +94,6 @@ def parse_test_cases( config: Dict[str, Any], workspace: Optional[Path] = None, path_resolver: Any = None, - *, - strict: bool = True, ) -> List[TestCase]: """Parse ``config['test_cases']`` (Schema v2) into ``TestCase`` objects. @@ -106,11 +104,9 @@ def parse_test_cases( When *workspace* and *path_resolver* are provided (Runner mode), command/args paths are resolved. - Required-field validation is controlled by the explicit *strict* flag - (1.4 原则 6:拆除隐式 TUI 后门). Runners use the default ``strict=True``; - the TUI passes ``strict=False`` explicitly and owns the relaxed form - (missing fields get sensible defaults and raw values are kept as-is for - display purposes). + 解析器全系统唯一且只接受 canonical TestCase:必填字段缺失即抛 + ``ValueError``(1.4 原则 6:无 TUI 宽松模式后门)。编辑半成品的 + 宽松形态由 TUI 侧自行 normalize 后再调用本函数。 """ cases: List[TestCase] = [] resolve = workspace is not None and path_resolver is not None @@ -122,8 +118,16 @@ def parse_test_cases( # Normalize both modes to a single ``steps`` list. A single-command # shorthand becomes a single-element list; a sequence case uses its - # ``execution.steps`` directly. + # ``execution.steps`` directly. Declaring both forms is ambiguous + # (Schema v2 oneOf 互斥) and is rejected outright instead of being + # silently resolved. is_sequence = "steps" in execution + if is_sequence and ("command" in execution or "args" in execution): + raise ValueError( + f"Test case '{case.get('name', 'unnamed')}': 'execution' " + f"declares both 'steps' and 'command'/'args' — the two forms " + f"are mutually exclusive (choose one)" + ) if is_sequence: step_configs: List[Dict[str, Any]] = list(execution.get("steps", [])) else: @@ -135,15 +139,14 @@ def parse_test_cases( "retry_count": execution.get("retry_count", 0), }] - steps: List[TestCaseStep] = [] + steps: List[TestStep] = [] for step in step_configs: - if strict: - step_required = ["command", "args", "expected"] - if not all(field in step for field in step_required): - raise ValueError( - f"Step in test case '{case.get('name', 'unnamed')}' " - f"is missing required fields" - ) + step_required = ["command", "args", "expected"] + if not all(field in step for field in step_required): + raise ValueError( + f"Step in test case '{case.get('name', 'unnamed')}' " + f"is missing required fields" + ) if resolve: executable, resolved_args = _split_and_resolve( step["command"], step["args"], workspace, path_resolver @@ -151,7 +154,7 @@ def parse_test_cases( else: executable = step.get("command", "") resolved_args = step.get("args", []) - steps.append(TestCaseStep( + steps.append(TestStep.from_flat( command=executable, args=resolved_args, expected=step["expected"] if "expected" in step else step.get("expected", {}), @@ -175,14 +178,13 @@ def parse_test_cases( )) else: # ── Single-command mode: execution shorthand fields → steps[0] ── - if strict: - missing = [f for f in ("name", "execution", "expected") if f not in case] - missing += [f for f in ("command", "args") if f not in execution] - if missing: - raise ValueError( - f"Test case {case.get('name', 'unnamed')} " - f"is missing required fields" - ) + missing = [f for f in ("name", "execution", "expected") if f not in case] + missing += [f for f in ("command", "args") if f not in execution] + if missing: + raise ValueError( + f"Test case {case.get('name', 'unnamed')} " + f"is missing required fields" + ) cases.append(TestCase( name=case.get("name", ""), steps=None, diff --git a/src/symtest/core/orchestration/sequence.py b/src/symtest/core/orchestration/sequence.py index fad6e62..a848ec7 100644 --- a/src/symtest/core/orchestration/sequence.py +++ b/src/symtest/core/orchestration/sequence.py @@ -7,7 +7,8 @@ (executor 参数仍可注入,保持 monkeypatch 支持); - case 级判定经由 ``validation.validator.validate_result``(只读), ``--update-baseline`` 经由 ``orchestration.accept`` accept 步骤; -- ``next_action_hint`` 由 ``reporting.diagnosis`` 生成。 +- 结果只携带 ``failure_kind``;``next_action_hint`` 由表现层装配点 + (``reporting.diagnosis.attach_next_action_hint``)填充(原则 5)。 """ from __future__ import annotations @@ -20,22 +21,21 @@ from ..validation.assertions import ValidationError # noqa: F401 (legacy except 兼容) from .accept import apply_baseline_accept from .single import execute_single_test_case -from ...reporting.diagnosis import build_next_action_hint logger = logging.getLogger("symtest.core.orchestration.sequence") # --------------------------------------------------------------------------- -# Step helper (duck-typed access for TestCaseStep / dict) +# Step helper (duck-typed access for TestStep / dict) # --------------------------------------------------------------------------- def _step_attr(step: Any, key: str, default: Any = None) -> Any: - """Get attribute from a ``TestCaseStep``(dict 支持已在 Phase 3 移除)。""" + """Get attribute from a ``TestStep``(dict 支持已在 Phase 3 移除)。""" return getattr(step, key, default) # --------------------------------------------------------------------------- -# Shared sequence execution (TestCaseStep list → result dict) +# Shared sequence execution (TestStep list → result dict) # --------------------------------------------------------------------------- def execute_sequence( @@ -54,7 +54,7 @@ def execute_sequence( ) -> Dict[str, Any]: """Execute a sequence test case (fail-fast). - ``steps`` must be a list of ``TestCaseStep`` objects(v2:dict 形态已在 + ``steps`` must be a list of ``TestStep`` objects(v2:dict 形态已在 Schema v2 落地后移除,跨进程路径由 process_worker 重建为对象)。 Parameters @@ -101,7 +101,6 @@ def execute_sequence( failed_step = None step_results: List[Dict[str, Any]] = [] case_assertion_results: List[Dict[str, Any]] = [] - case_hint: Optional[Dict[str, Any]] = None prefix = f"{print_prefix} " if print_prefix else "" @@ -289,9 +288,6 @@ def execute_sequence( all_passed = False failed_step = len(steps) + 1 # synthetic step number case_assertion_results = case_vr.assertion_results - case_hint = build_next_action_hint( - case_vr.failure_kind, update_baseline=update_baseline, - ) last_result = { "name": case_name, "status": "failed", @@ -317,9 +313,6 @@ def execute_sequence( all_passed = False failed_step = len(steps) + 1 # synthetic step number case_assertion_results = getattr(exc, "assertion_results", []) - case_hint = build_next_action_hint( - getattr(exc, "failure_kind", None), update_baseline=update_baseline, - ) last_result = { "name": case_name, "status": "failed", @@ -374,10 +367,12 @@ def execute_sequence( else: slim_output = "" - # ── assertion_results / next_action_hint resolution ── + # ── assertion_results resolution ── # Case-level assertion data takes precedence; otherwise propagate the # failed step's data so sequence results honor the same contract as - # single-command results. + # single-command results. ``next_action_hint`` is left None here: the + # presentation layer (reporting attach point) builds it from + # ``failure_kind``(原则 5 单向流). if case_assertion_results: assertion_results = case_assertion_results elif not all_passed and last_result is not None: @@ -385,13 +380,6 @@ def execute_sequence( else: assertion_results = [] - if case_hint is not None: - next_action_hint = case_hint - elif not all_passed and last_result is not None: - next_action_hint = last_result.get("next_action_hint") - else: - next_action_hint = None - command_summary = " -> ".join( f"{_step_attr(s, 'command')} {' '.join(_step_attr(s, 'args'))}".strip() for s in steps @@ -412,5 +400,5 @@ def execute_sequence( "attempts": last_result.get("attempts", 1) if last_result else 1, "flaky": last_result.get("flaky", False) if last_result else False, "assertion_results": assertion_results, - "next_action_hint": next_action_hint, + "next_action_hint": None, } diff --git a/src/symtest/core/orchestration/single.py b/src/symtest/core/orchestration/single.py index 6950514..797fcac 100644 --- a/src/symtest/core/orchestration/single.py +++ b/src/symtest/core/orchestration/single.py @@ -8,7 +8,8 @@ - 判定交给 ``validation.validator.validate_result``(只读); - ``--update-baseline`` 写盘改为本层独立 accept 步骤 (``accept.apply_baseline_accept``); -- ``next_action_hint`` 由 ``reporting.diagnosis`` 生成(原则 5)。 +- 结果只携带 ``failure_kind``;``next_action_hint`` 由表现层装配点 + (``reporting.diagnosis.attach_next_action_hint``)填充(原则 5)。 Phase 3 Schema v2:``case`` 只接受 ``ExecutionSpec``,期望规格由调用方 显式传入(``expectation``),不再回落读取旧 dict 形态。 @@ -24,7 +25,6 @@ from ..validation.validator import _trim_compare_failures, validate_result from ..validation.assertions import ValidationError from .accept import apply_baseline_accept -from ...reporting.diagnosis import build_next_action_hint logger = logging.getLogger("symtest.core.orchestration.single") @@ -58,14 +58,12 @@ def _execute_command_once( if exec_result.error is not None: result.message = f"Execution error: {exec_result.error}" result.failure_kind = "execution_error" - result.next_action_hint = build_next_action_hint("execution_error") elif exec_result.timed_out: result.status = "timeout" result.failure_kind = "timeout" result.message = ( f"Timeout reached! Killed after {exec_result.timeout_limit} seconds." ) - result.next_action_hint = build_next_action_hint("timeout") else: try: vr = validate_result( @@ -78,18 +76,13 @@ def _execute_command_once( result.compare_failures = _trim_compare_failures(exc.compare_failures) result.baseline_updated = list(exc.baseline_updated) result.assertion_results = list(exc.assertion_results) - result.next_action_hint = build_next_action_hint( - exc.failure_kind, update_baseline=update_baseline, - ) except AssertionError as exc: # Legacy AssertionError catch for backward compatibility result.message = str(exc) result.failure_kind = result.failure_kind or "unknown" - result.next_action_hint = build_next_action_hint(result.failure_kind) except Exception as exc: result.message = f"Execution error: {str(exc)}" result.failure_kind = "execution_error" - result.next_action_hint = build_next_action_hint("execution_error") else: if vr.passed: result.status = "passed" @@ -99,9 +92,6 @@ def _execute_command_once( result.failure_kind = vr.failure_kind result.compare_failures = _trim_compare_failures(vr.compare_failures) result.assertion_results = vr.assertion_results - result.next_action_hint = build_next_action_hint( - vr.failure_kind, update_baseline=update_baseline, - ) # ── accept 步骤(原则 3:写盘在编排层,不在 Validator) ── if update_baseline and vr.failure_kind == "file_compare": accepted = apply_baseline_accept(vr, workspace) @@ -111,7 +101,6 @@ def _execute_command_once( result.failure_kind = None result.compare_failures = [] result.assertion_results = accepted - result.next_action_hint = None result.duration = time.perf_counter() - start_time return result.to_dict() diff --git a/src/symtest/core/parallel_runner.py b/src/symtest/core/parallel_runner.py index 53457ee..b8b1a07 100644 --- a/src/symtest/core/parallel_runner.py +++ b/src/symtest/core/parallel_runner.py @@ -382,7 +382,6 @@ def _case_command_str(case: Optional[TestCase]) -> str: def _update_results_skipped(self, result: Dict[str, Any], test_index: int, case_name: str) -> None: """Thread-safe result update for skipped cases (no xfail processing).""" with self.lock: - self._fill_hint_command(result, case_name) self.results["details"].append(result) logger.warning("⊘ Test %d skipped: %s", test_index, case_name) if result.get("message"): @@ -400,7 +399,6 @@ def _update_results(self, result: Dict[str, Any], test_index: int, case: TestCas # Apply xfail status mapping before counting self._apply_xfail_status(result, case) - self._fill_hint_command(result, case.name) self.results["details"].append(result) duration = result.get("duration", 0) status = result["status"] diff --git a/src/symtest/core/process_worker.py b/src/symtest/core/process_worker.py index fa142dc..40ed6b6 100644 --- a/src/symtest/core/process_worker.py +++ b/src/symtest/core/process_worker.py @@ -3,7 +3,7 @@ 用于多进程并行测试执行,避免序列化问题 Phase 3 Schema v2:跨进程传递 v2 wire dict(``TestCase.to_dict()`` 输出), -本模块负责将其重建为 ``ExecutionSpec`` / ``TestCaseStep`` 对象后进入编排层。 +本模块负责将其重建为 ``ExecutionSpec`` / ``TestStep`` 对象后进入编排层。 """ import logging @@ -11,17 +11,17 @@ from .orchestration.sequence import execute_sequence from .orchestration.single import execute_single_test_case -from .test_case import ExecutionSpec, TestCaseStep +from .test_case import ExecutionSpec, TestStep logger = logging.getLogger("symtest.core.process_worker") def _spec_from_v2(case_data: Dict[str, Any]) -> ExecutionSpec: - """v2 wire dict(case.to_dict())→ ExecutionSpec(含 TestCaseStep 重建)。""" + """v2 wire dict(case.to_dict())→ ExecutionSpec(含 TestStep 重建)。""" execution = case_data.get("execution") or {} if "steps" in execution: steps = [ - TestCaseStep( + TestStep.from_flat( command=s["command"], args=s["args"], expected=s.get("expected") or {}, diff --git a/src/symtest/core/test_case.py b/src/symtest/core/test_case.py index 45bde15..86b8bc9 100644 --- a/src/symtest/core/test_case.py +++ b/src/symtest/core/test_case.py @@ -12,30 +12,22 @@ - 构造函数保留平铺关键字参数作为 legacy 语义归一入口(TUI 编辑路径、 迁移等价性测试的 legacy 侧复用它); - ``case.command`` / ``case.expected`` / ``case.env`` ... 等属性直通访问器 - 映射到子 Spec,使现有 ``case.xxx`` 访问点零改动。 + 映射到子 Spec,使现有 ``case.xxx`` 访问点零改动; +- 序列步骤为 :class:`TestStep`(execution + expectation 分层); + ``TestStep.from_flat`` 是 DSL 平铺字段的归一入口。 """ from dataclasses import dataclass, field from typing import Any, Dict, List, Optional -@dataclass -class TestCaseStep: - """A single step within a sequence test case.""" - __test__ = False - command: str - args: List[str] - expected: Dict[str, Any] - timeout: Optional[float] = None - retry_count: int = 0 - - @dataclass class ExecutionSpec: """执行语义:一个执行单元(case 或 step)要执行什么。 纯数据,不包含任何判定语义(expected 属于 ExpectationSpec)。 - ``steps`` 非 None 表示 sequence 模式(steps 为原子"执行+判定"对列表); - 为 None 表示单命令模式,由 ``command/args/timeout/retry_count`` 描述。 + ``steps`` 非 None 表示 sequence 模式(steps 为 ``TestStep`` 列表, + 每项是原子"执行+判定"对);为 None 表示单命令模式,由 + ``command/args/timeout/retry_count`` 描述。 """ __test__ = False name: str = "" @@ -44,7 +36,7 @@ class ExecutionSpec: timeout: Optional[float] = None retry_count: int = 0 env: Dict[str, str] = field(default_factory=dict) - steps: Optional[List[TestCaseStep]] = None + steps: Optional[List["TestStep"]] = None @dataclass @@ -59,6 +51,82 @@ class ExpectationSpec: assertions: Dict[str, Any] = field(default_factory=dict) +@dataclass +class TestStep: + """序列中的一个原子步骤:execution + expectation 分层(1.4 v2 模型)。 + + DSL 形态不变(step dict:``command/args/expected/timeout/retry_count``), + 由 parser / ``from_flat`` 负责归一;本类型提供平铺直通访问器,使 + duck-typing 消费点(``_step_attr``、``compute_config_hash``、TUI steps + 编辑器)零改动。 + """ + __test__ = False + execution: ExecutionSpec + expectation: ExpectationSpec + + @classmethod + def from_flat( + cls, + command: str, + args: List[str], + expected: Optional[Dict[str, Any]] = None, + timeout: Optional[float] = None, + retry_count: int = 0, + ) -> "TestStep": + """DSL 平铺字段 → 分层 TestStep(parser / wire dict 重建入口)。""" + return cls( + execution=ExecutionSpec( + command=command, args=args, + timeout=timeout, retry_count=retry_count, + ), + expectation=ExpectationSpec(assertions=expected if expected else {}), + ) + + # ── execution 直通访问器 ───────────────────────────────────────────── + + @property + def command(self) -> str: + return self.execution.command + + @command.setter + def command(self, value: str) -> None: + self.execution.command = value + + @property + def args(self) -> List[str]: + return self.execution.args + + @args.setter + def args(self, value: List[str]) -> None: + self.execution.args = value + + @property + def timeout(self) -> Optional[float]: + return self.execution.timeout + + @timeout.setter + def timeout(self, value: Optional[float]) -> None: + self.execution.timeout = value + + @property + def retry_count(self) -> int: + return self.execution.retry_count + + @retry_count.setter + def retry_count(self, value: int) -> None: + self.execution.retry_count = value + + # ── expectation 直通访问器 ─────────────────────────────────────────── + + @property + def expected(self) -> Dict[str, Any]: + return self.expectation.assertions + + @expected.setter + def expected(self, value: Dict[str, Any]) -> None: + self.expectation.assertions = value if value else {} + + @dataclass class SchedulingSpec: """调度语义:什么时候、以什么资源执行(原则 4:编排层消费)。""" @@ -92,7 +160,7 @@ def __init__( description: str = "", timeout: Optional[float] = None, resources: Optional[Dict[str, Any]] = None, - steps: Optional[List[TestCaseStep]] = None, + steps: Optional[List["TestStep"]] = None, tags: Optional[List[str]] = None, retry_count: int = 0, expected_failure: bool = False, @@ -176,11 +244,11 @@ def env(self, value: Dict[str, str]) -> None: self.execution.env = value @property - def steps(self) -> Optional[List[TestCaseStep]]: + def steps(self) -> Optional[List["TestStep"]]: return self.execution.steps @steps.setter - def steps(self, value: Optional[List[TestCaseStep]]) -> None: + def steps(self, value: Optional[List["TestStep"]]) -> None: self.execution.steps = value # ── expectation / scheduling 直通访问器 ───────────────────────────── @@ -213,7 +281,7 @@ def resources(self, value: Optional[Dict[str, Any]]) -> None: # ── 统一步骤访问(单命令模式返回单元素列表) ───────────────────────── @property - def all_steps(self) -> List[TestCaseStep]: + def all_steps(self) -> List["TestStep"]: """Return the unified list of steps regardless of mode. Single-command cases yield a single-element list; sequence cases yield @@ -222,7 +290,7 @@ def all_steps(self) -> List[TestCaseStep]: """ if self.execution.steps is not None: return self.execution.steps - return [TestCaseStep( + return [TestStep.from_flat( command=self.execution.command, args=self.execution.args, expected=self.expectation.assertions, diff --git a/src/symtest/reporting/__init__.py b/src/symtest/reporting/__init__.py index 6baec8e..15b1150 100644 --- a/src/symtest/reporting/__init__.py +++ b/src/symtest/reporting/__init__.py @@ -1,10 +1,17 @@ """Reporting / result-consumer 层(原则 5)。 ``diagnosis.build_next_action_hint`` 是 result consumer:只消费失败结论, -不参与执行与判定。 +不参与执行与判定。``attach_next_action_hint(s)`` 是表现层装配点:orchestration +只产出 ``failure_kind``,hint 在 CLI 报告装配阶段 / TUI run_case 填充。 """ -from .diagnosis import build_next_action_hint +from .diagnosis import ( + attach_next_action_hint, + attach_next_action_hints, + build_next_action_hint, +) __all__ = [ "build_next_action_hint", + "attach_next_action_hint", + "attach_next_action_hints", ] diff --git a/src/symtest/reporting/diagnosis.py b/src/symtest/reporting/diagnosis.py index 4d44892..e3d7106 100644 --- a/src/symtest/reporting/diagnosis.py +++ b/src/symtest/reporting/diagnosis.py @@ -1,8 +1,8 @@ """失败诊断(next_action_hint,原则 5:result consumer)。 自 1.3 ``core/execution.py::_build_next_action_hint`` 原样搬入(公开化命名)。 -``command`` 字段留 ``None``:本层不知道 config 文件路径,由 runner 填充 -(见 ``BaseRunner._fill_hint_command``)。 +``command`` 字段由装配点填充(``attach_next_action_hint(s)``,CLI 报告 +装配阶段 / TUI 表现层调用),orchestration 只产出 ``failure_kind``。 """ from typing import Any, Dict, Optional @@ -82,3 +82,66 @@ def build_next_action_hint( "'expected' in this result." ), } + + +def _hint_command( + case_name: str, + config_path: Optional[str], + *, + is_update_baseline: bool, +) -> str: + """Build the concrete re-run CLI command inside a hint.""" + base = f'symtest run "{config_path}"' + if is_update_baseline: + base += " --update-baseline" + return f'{base} -t "{case_name}"' + + +def attach_next_action_hint( + result: Dict[str, Any], + *, + update_baseline: bool = False, + config_path: Optional[str] = None, +) -> Optional[Dict[str, Any]]: + """Reporting 装配点:按 ``failure_kind`` 为单个失败结果填充 hint。 + + Orchestration 只产出 ``failure_kind``(原则 5 单向流);本函数在结果 + 装配/输出阶段(CLI 报告输出前 / TUI run_case)调用,为 ``failure_kind`` + 非空的结果构建结构化建议并填充具体 CLI 命令。已带 hint 的结果只补 + command 字段(幂等)。 + + Returns: + 填充后的 hint;结果无失败结论时为 ``None``。 + """ + hint = result.get("next_action_hint") + if hint is None: + failure_kind = result.get("failure_kind") + if not failure_kind: + return None + hint = build_next_action_hint(failure_kind, update_baseline=update_baseline) + result["next_action_hint"] = hint + if hint and not hint.get("command") and config_path is not None: + # 与 1.3 runner 行为一致:--update-baseline 标志只由 hint 的 action + # 决定(update_baseline 形态仅在未开启 --update-baseline 时出现)。 + hint["command"] = _hint_command( + result.get("name", ""), config_path, + is_update_baseline=(hint.get("action") == "update_baseline"), + ) + return hint + + +def attach_next_action_hints( + results: Dict[str, Any], + *, + update_baseline: bool = False, + config_path: Optional[str] = None, +) -> None: + """对整次运行的 ``results``(含 ``details`` 列表)批量装配 hint。 + + CLI 报告输出前的唯一装配点,覆盖顺序 / 并行 / skip 全部路径; + passed / skipped 结果无 ``failure_kind``,天然不产生 hint。 + """ + for detail in results.get("details", []): + attach_next_action_hint( + detail, update_baseline=update_baseline, config_path=config_path, + ) diff --git a/src/symtest/tui/controllers/case_controller.py b/src/symtest/tui/controllers/case_controller.py index 9e5f75b..b49d360 100644 --- a/src/symtest/tui/controllers/case_controller.py +++ b/src/symtest/tui/controllers/case_controller.py @@ -8,14 +8,17 @@ from pathlib import Path from typing import Any, Dict, List, Optional, Tuple -from ...core.test_case import TestCase, TestCaseStep +from ...core.test_case import TestCase, TestStep from ...core.orchestration.single import execute_single_test_case from ...core.orchestration.sequence import execute_sequence from ...core.config_loader import parse_test_cases from ...config.config_io import load_config, save_config +from ...config.normalize import normalize_draft_config +from ...reporting.diagnosis import attach_next_action_hint logger = logging.getLogger("symtest.tui.controller") + # --------------------------------------------------------------------------- # Search helpers # --------------------------------------------------------------------------- @@ -169,10 +172,10 @@ def load(self, file_path: str, workspace: Optional[str] = None) -> int: self._file_path = path self._workspace = workspace - # Parse test cases via unified entry point. TUI owns the relaxed - # form explicitly (1.4 原则 6):no path resolution for display and - # strict=False so incomplete cases get sensible defaults. - self._cases = parse_test_cases(config, strict=False) + # Parse test cases via the unified strict entry point. TUI owns the + # relaxed form (1.4 原则 6):draft configs are normalized here before + # parsing; no path resolution for display purposes. + self._cases = parse_test_cases(normalize_draft_config(config)) self._dirty = False return len(self._cases) @@ -297,6 +300,13 @@ def run_case(self, index: int) -> Dict[str, Any]: update_baseline=self._update_baseline, ) self._record_history(case.name, result) + # ── Reporting 装配点(原则 5):TUI 属表现层,在此按 failure_kind + # 填充 next_action_hint(orchestration 只产出失败结论) ── + attach_next_action_hint( + result, + update_baseline=self._update_baseline, + config_path=str(self._file_path) if self._file_path else None, + ) return result def _record_history(self, case_name: str, result: Dict[str, Any]) -> None: diff --git a/src/symtest/tui/widgets/steps_editor.py b/src/symtest/tui/widgets/steps_editor.py index 5dfc812..1e796dd 100644 --- a/src/symtest/tui/widgets/steps_editor.py +++ b/src/symtest/tui/widgets/steps_editor.py @@ -15,7 +15,7 @@ from textual.message import Message from textual.app import ComposeResult -from ...core.test_case import TestCaseStep +from ...core.test_case import TestStep class StepsEditor(Vertical): @@ -59,7 +59,7 @@ class Changed(Message): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) - self._steps: List[TestCaseStep] = [] + self._steps: List[TestStep] = [] self._editing_idx: int = -1 # -1 = new step def compose(self) -> ComposeResult: @@ -91,7 +91,7 @@ def on_mount(self) -> None: # -- public API ---------------------------------------------------------- - def load(self, steps: List[TestCaseStep]) -> None: + def load(self, steps: List[TestStep]) -> None: """Populate the editor with *steps*.""" self._steps = copy.deepcopy(steps) self._editing_idx = -1 @@ -99,7 +99,7 @@ def load(self, steps: List[TestCaseStep]) -> None: self._refresh_list() self._clear_edit_form() - def to_steps(self) -> List[TestCaseStep]: + def to_steps(self) -> List[TestStep]: if self.is_mounted: # Pull any in-progress edit from UI into steps before returning pass # edits are applied immediately via buttons @@ -177,7 +177,7 @@ def _save_current_step(self) -> None: retry_text = self.query_one("#step-retry-count", Input).value.strip() retry_count = int(retry_text) if retry_text else 0 - new_step = TestCaseStep( + new_step = TestStep.from_flat( command=cmd, args=args, expected=expected, timeout=timeout, retry_count=retry_count, ) diff --git a/tests/integration/test_sequence.py b/tests/integration/test_sequence.py index c3ffd82..5f103af 100644 --- a/tests/integration/test_sequence.py +++ b/tests/integration/test_sequence.py @@ -8,7 +8,7 @@ from symtest.runners.json_runner import JSONRunner from symtest.runners.yaml_runner import YAMLRunner -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep class TestSequenceLoadJSON(unittest.TestCase): @@ -279,18 +279,18 @@ def tearDown(self): self.temp_dir.cleanup() -class TestTestCaseStepDataclass(unittest.TestCase): - """Test TestCaseStep and TestCase with steps field.""" +class TestTestStepDataclass(unittest.TestCase): + """Test TestStep and TestCase with steps field.""" def test_test_case_step_creation(self): - step = TestCaseStep(command="echo", args=["hello"], expected={"return_code": 0}) + step = TestStep.from_flat(command="echo", args=["hello"], expected={"return_code": 0}) self.assertEqual(step.command, "echo") self.assertEqual(step.timeout, None) def test_test_case_with_steps(self): steps = [ - TestCaseStep(command="echo", args=["a"], expected={"return_code": 0}), - TestCaseStep(command="echo", args=["b"], expected={"return_code": 0}, timeout=5), + TestStep.from_flat(command="echo", args=["a"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["b"], expected={"return_code": 0}, timeout=5), ] case = TestCase(name="seq", steps=steps) self.assertEqual(case.command, "") # default for sequence mode @@ -299,8 +299,8 @@ def test_test_case_with_steps(self): def test_test_case_to_dict_with_steps(self): steps = [ - TestCaseStep(command="echo", args=["a"], expected={"return_code": 0}), - TestCaseStep(command="echo", args=["b"], expected={"return_code": 0}, retry_count=2), + TestStep.from_flat(command="echo", args=["a"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["b"], expected={"return_code": 0}, retry_count=2), ] case = TestCase(name="seq", steps=steps, retry_count=3) d = case.to_dict() diff --git a/tests/unit/core/test_architecture_guard.py b/tests/unit/core/test_architecture_guard.py index 54e4e16..7117d87 100644 --- a/tests/unit/core/test_architecture_guard.py +++ b/tests/unit/core/test_architecture_guard.py @@ -5,9 +5,8 @@ expected 的存在 —— Phase 2 唯一验收标准); - 原则 3:validation 只允许 import execution 的 result 类型(读取执行事实), 不得 import 编排层; -- 原则 6:core 不得 import cli / tui / commands / runners(表现层)。 - 注:``symtest.reporting.diagnosis`` 是唯一例外 —— 它是 result consumer - 工具,被 orchestration 调用以填充 wire format 的 next_action_hint 字段。 +- 原则 6:core 不得 import cli / tui / commands / runners / reporting + (表现层与 result consumer)。 """ import re from pathlib import Path @@ -78,8 +77,9 @@ class TestCoreDoesNotImportPresentation: """原则 6:核心模型不依赖表现层。""" def test_core_has_no_presentation_imports(self): - forbidden_abs = ("symtest.cli", "symtest.tui", "symtest.commands", "symtest.runners") - forbidden_rel = ("cli", "tui", "commands", "runners") + forbidden_abs = ("symtest.cli", "symtest.tui", "symtest.commands", + "symtest.runners", "symtest.reporting") + forbidden_rel = ("cli", "tui", "commands", "runners", "reporting") offenders = [] for f in _py_files(SRC / "core"): for mod in _imports(f): diff --git a/tests/unit/core/test_case_env.py b/tests/unit/core/test_case_env.py index 778ae3c..672fc35 100644 --- a/tests/unit/core/test_case_env.py +++ b/tests/unit/core/test_case_env.py @@ -14,7 +14,7 @@ from symtest.core.config_loader import parse_test_cases from symtest.core.orchestration.sequence import execute_sequence from symtest.core.orchestration.single import execute_single_test_case -from symtest.core.test_case import ExecutionSpec, TestCaseStep +from symtest.core.test_case import ExecutionSpec, TestStep # --------------------------------------------------------------------------- @@ -178,12 +178,12 @@ class TestEnvSequence: def test_env_applied_to_all_steps(self): steps = [ - TestCaseStep( + TestStep.from_flat( command=sys.executable, args=["-c", "import os;print('S='+os.environ.get('SCALE','?'))"], expected={"return_code": 0, "output_contains": ["S=1.0"]}, ), - TestCaseStep( + TestStep.from_flat( command=sys.executable, args=["-c", "import os;print('S='+os.environ.get('SCALE','?'))"], expected={"return_code": 0, "output_contains": ["S=1.0"]}, @@ -200,7 +200,7 @@ def test_hash_differs_when_env_differs(self): from symtest.core.sequence_state import compute_config_hash steps = [ - TestCaseStep(command="echo", args=["a"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["a"], expected={"return_code": 0}), ] h1 = compute_config_hash(steps, None, {"SCALE": "1.0"}) h2 = compute_config_hash(steps, None, {"SCALE": "2.0"}) diff --git a/tests/unit/core/test_config_loader.py b/tests/unit/core/test_config_loader.py index 75f0ba0..f49e57a 100644 --- a/tests/unit/core/test_config_loader.py +++ b/tests/unit/core/test_config_loader.py @@ -8,7 +8,7 @@ substitute_placeholders, parse_test_cases, ) -from symtest.core.test_case import TestCaseStep +from symtest.core.test_case import TestStep from symtest.core.orchestration.sequence import execute_sequence from symtest.core.validation.result import ValidationResult @@ -369,9 +369,9 @@ def _failed_result(name="step"): def _steps(*specs): - """Build TestCaseStep objects from (command, args, expected) tuples.""" + """Build TestStep objects from (command, args, expected) tuples.""" return [ - TestCaseStep(command=cmd, args=args, expected=expected) + TestStep.from_flat(command=cmd, args=args, expected=expected) for cmd, args, expected in specs ] diff --git a/tests/unit/core/test_config_loader_new_features.py b/tests/unit/core/test_config_loader_new_features.py index 406161a..aad292a 100644 --- a/tests/unit/core/test_config_loader_new_features.py +++ b/tests/unit/core/test_config_loader_new_features.py @@ -4,15 +4,16 @@ import pytest from symtest.core.orchestration.sequence import execute_sequence -from symtest.core.test_case import TestCaseStep +from symtest.core.test_case import TestStep from symtest.core.validation.assertions import ValidationError from symtest.core.validation.result import ValidationResult +from symtest.reporting.diagnosis import attach_next_action_hint def _steps(*specs): - """Build TestCaseStep objects from (command, args, expected) tuples.""" + """Build TestStep objects from (command, args, expected) tuples.""" return [ - TestCaseStep(command=cmd, args=args, expected=expected) + TestStep.from_flat(command=cmd, args=args, expected=expected) for cmd, args, expected in specs ] @@ -124,16 +125,13 @@ def test_step_results_truncates_on_failure(self): class TestSequenceStructuredDiagnostics: - """Sequence results propagate assertion_results / next_action_hint.""" + """Sequence results propagate assertion_results; hint 由装配层按 failure_kind 填充.""" - def test_failed_step_propagates_hint_and_assertions(self): + def test_failed_step_propagates_assertions_and_failure_kind(self): failed = _failed_result("s2") failed["assertion_results"] = [ {"assertion": "return_code", "passed": False, "message": "rc mismatch"} ] - failed["next_action_hint"] = { - "action": "update_expected", "command": None, "reason": "r", - } steps = _steps( ("e1", ["a"], {"return_code": 0}), ("e2", ["b"], {"return_code": 0}), @@ -145,7 +143,10 @@ def test_failed_step_propagates_hint_and_assertions(self): result = execute_sequence("case", steps) assert result["status"] == "failed" assert result["assertion_results"] == failed["assertion_results"] - assert result["next_action_hint"] == failed["next_action_hint"] + # Orchestration 只产 failure_kind;hint 由 reporting 装配点填充 + assert result["failure_kind"] == "return_code" + assert result["next_action_hint"] is None + assert attach_next_action_hint(result)["action"] == "update_expected" def test_case_level_failure_builds_hint(self): steps = _steps(("echo", ["x"], {"return_code": 0})) @@ -171,7 +172,9 @@ def test_case_level_failure_builds_hint(self): assert result["assertion_results"] == [ {"assertion": "compare_files", "passed": False} ] - assert result["next_action_hint"]["action"] == "update_baseline" + assert result["failure_kind"] == "file_compare" + assert result["next_action_hint"] is None + assert attach_next_action_hint(result)["action"] == "update_baseline" def test_passed_sequence_uses_case_level_assertion_results(self): steps = _steps(("echo", ["x"], {"return_code": 0})) diff --git a/tests/unit/core/test_execution_new_features.py b/tests/unit/core/test_execution_new_features.py index bd0197e..4bd212b 100644 --- a/tests/unit/core/test_execution_new_features.py +++ b/tests/unit/core/test_execution_new_features.py @@ -9,7 +9,10 @@ from symtest.core.validation.validator import validate_result from symtest.core.execution.result import ExecutionResult from symtest.core.execution.executor import _trim_output, DEFAULT_OUTPUT_MAX_CHARS -from symtest.reporting.diagnosis import build_next_action_hint +from symtest.reporting.diagnosis import ( + attach_next_action_hint, + build_next_action_hint, +) class TestRetryFeatures: @@ -218,9 +221,13 @@ def test_assertion_failure_attaches_hint(self, case, expectation): ) result = execute_single_test_case(case, expectation=expectation) assert result["status"] == "failed" - hint = result["next_action_hint"] + # Orchestration only produces failure_kind; the hint is filled by + # the reporting attach point (principle 5). + assert result["next_action_hint"] is None + assert result["failure_kind"] == "return_code" + hint = attach_next_action_hint(result) assert hint["action"] == "update_expected" - assert hint["command"] is None # filled by the runner layer + assert hint["command"] is None # filled when config_path is given assert hint["reason"] def test_execution_error_attaches_hint(self, case, expectation): @@ -228,7 +235,8 @@ def test_execution_error_attaches_hint(self, case, expectation): result = execute_single_test_case(case, expectation=expectation) assert result["status"] == "failed" assert result["failure_kind"] == "execution_error" - assert result["next_action_hint"]["action"] == "investigate" + hint = attach_next_action_hint(result) + assert hint["action"] == "investigate" def test_timeout_attaches_hint(self, case, expectation): case.timeout = 1 @@ -242,7 +250,9 @@ def test_timeout_attaches_hint(self, case, expectation): with patch("subprocess.Popen", return_value=mock_proc): result = execute_single_test_case(case, expectation=expectation) assert result["status"] == "timeout" - assert result["next_action_hint"]["action"] == "increase_timeout" + assert result["failure_kind"] == "timeout" + hint = attach_next_action_hint(result) + assert hint["action"] == "increase_timeout" # PID 99999 is harmless even if killpg() is called; # mock_proc.kill is also safe. Assertions above verify # the result format remains correct. @@ -259,8 +269,8 @@ def test_passed_case_has_no_hint(self, case, expectation): assert result["next_action_hint"] is None -class TestRunnerHintCommandFilling: - """Runners fill the concrete symtest command into next_action_hint.""" +class TestAttachHintCommandFilling: + """The reporting attach point fills the concrete symtest command.""" def _make_failed_result(self, action): return { @@ -270,38 +280,32 @@ def _make_failed_result(self, action): } def test_update_baseline_command(self): - from symtest.runners.json_runner import JSONRunner - - runner = JSONRunner(config_file="cfg.json") result = self._make_failed_result("update_baseline") - runner._fill_hint_command(result, "case_a") + attach_next_action_hint(result, config_path="cfg.json") cmd = result["next_action_hint"]["command"] - assert cmd == ( - f'symtest run "{runner.config_path}" --update-baseline -t "case_a"' - ) + assert cmd == 'symtest run "cfg.json" --update-baseline -t "case_a"' def test_rerun_command_for_other_actions(self): - from symtest.runners.json_runner import JSONRunner - - runner = JSONRunner(config_file="cfg.json") result = self._make_failed_result("update_expected") - runner._fill_hint_command(result, "case_a") + attach_next_action_hint(result, config_path="cfg.json") cmd = result["next_action_hint"]["command"] - assert cmd == f'symtest run "{runner.config_path}" -t "case_a"' + assert cmd == 'symtest run "cfg.json" -t "case_a"' def test_existing_command_not_overwritten(self): - from symtest.runners.json_runner import JSONRunner - - runner = JSONRunner(config_file="cfg.json") result = self._make_failed_result("update_baseline") result["next_action_hint"]["command"] = "custom" - runner._fill_hint_command(result, "case_a") + attach_next_action_hint(result, config_path="cfg.json") assert result["next_action_hint"]["command"] == "custom" def test_no_hint_is_noop(self): - from symtest.runners.json_runner import JSONRunner - - runner = JSONRunner(config_file="cfg.json") result = {"name": "ok", "status": "passed", "next_action_hint": None} - runner._fill_hint_command(result, "ok") # should not raise + attach_next_action_hint(result, config_path="cfg.json") # should not raise assert result["next_action_hint"] is None + + def test_builds_hint_from_failure_kind(self): + """Orchestration only sets failure_kind; the attach point builds the hint.""" + result = {"name": "case_b", "status": "failed", "failure_kind": "file_compare"} + attach_next_action_hint(result, config_path="cfg.json") + hint = result["next_action_hint"] + assert hint["action"] == "update_baseline" + assert hint["command"] == 'symtest run "cfg.json" --update-baseline -t "case_b"' diff --git a/tests/unit/core/test_model_v2.py b/tests/unit/core/test_model_v2.py index 053fa8f..2dfa486 100644 --- a/tests/unit/core/test_model_v2.py +++ b/tests/unit/core/test_model_v2.py @@ -13,7 +13,7 @@ ExpectationSpec, SchedulingSpec, TestCase, - TestCaseStep, + TestStep, ) @@ -73,8 +73,8 @@ def test_explicit_specs_take_precedence(self): def test_sequence_mode(self): steps = [ - TestCaseStep(command="a", args=[], expected={"return_code": 0}), - TestCaseStep(command="b", args=[], expected={}), + TestStep.from_flat(command="a", args=[], expected={"return_code": 0}), + TestStep.from_flat(command="b", args=[], expected={}), ] tc = TestCase(name="seq", steps=steps) assert tc.execution.steps is steps @@ -115,7 +115,7 @@ def test_execution_passthrough_write(self): tc.timeout = 5 tc.retry_count = 3 tc.env = {"A": "B"} - tc.steps = [TestCaseStep(command="s", args=[], expected={})] + tc.steps = [TestStep.from_flat(command="s", args=[], expected={})] assert tc.execution.command == "ls" assert tc.execution.args == ["-l"] assert tc.execution.timeout == 5 @@ -180,7 +180,7 @@ def test_is_single_command_true_for_flat_case(self): assert TestCase(name="t", command="echo").is_single_command is True def test_sequence_all_steps_returns_steps_list(self): - steps = [TestCaseStep(command="s", args=[], expected={})] + steps = [TestStep.from_flat(command="s", args=[], expected={})] tc = TestCase(name="t", steps=steps) assert tc.all_steps == steps @@ -197,7 +197,7 @@ def test_execution_spec_has_no_expected_field(self): def test_step_carries_its_own_expected(self): """steps 是"执行+判定"原子对,step 级 expected 属于 step。""" - step = TestCaseStep(command="a", args=[], expected={"return_code": 0}) + step = TestStep.from_flat(command="a", args=[], expected={"return_code": 0}) spec = ExecutionSpec(steps=[step]) assert spec.steps[0].expected == {"return_code": 0} @@ -233,7 +233,7 @@ def test_to_dict_sequence_mode_keeps_steps(self): tc = TestCase( name="seq", steps=[ - TestCaseStep(command="a", args=["1"], expected={}, timeout=5.0), + TestStep.from_flat(command="a", args=["1"], expected={}, timeout=5.0), ], ) assert tc.to_dict()["execution"]["steps"] == [ @@ -243,7 +243,7 @@ def test_to_dict_sequence_mode_keeps_steps(self): def test_to_dict_sequence_mode_omits_command_args(self): """steps 模式下 execution 省略 command/args(二选一)。""" - tc = TestCase(name="seq", steps=[TestCaseStep(command="a", args=[], expected={})]) + tc = TestCase(name="seq", steps=[TestStep.from_flat(command="a", args=[], expected={})]) execution = tc.to_dict()["execution"] assert "command" not in execution assert "args" not in execution @@ -288,7 +288,7 @@ def test_deepcopy_works(self): """TUI duplicate_case 依赖 deepcopy。""" tc = TestCase( name="t", - steps=[TestCaseStep(command="a", args=[], expected={})], + steps=[TestStep.from_flat(command="a", args=[], expected={})], ) clone = copy.deepcopy(tc) clone.name = "clone" diff --git a/tests/unit/core/test_orchestration_accept.py b/tests/unit/core/test_orchestration_accept.py index 3ff7624..843e954 100644 --- a/tests/unit/core/test_orchestration_accept.py +++ b/tests/unit/core/test_orchestration_accept.py @@ -13,7 +13,10 @@ from symtest.core.orchestration.single import execute_single_test_case from symtest.core.test_case import ExecutionSpec from symtest.core.validation.validator import validate_result -from symtest.reporting.diagnosis import build_next_action_hint +from symtest.reporting.diagnosis import ( + attach_next_action_hint, + build_next_action_hint, +) def _exec(output="", return_code=0, name="t"): @@ -162,7 +165,9 @@ def test_without_update_baseline_fails(self): assert result["status"] == "failed" assert result["failure_kind"] == "file_compare" - assert result["next_action_hint"]["action"] == "update_baseline" + # Orchestration 只产 failure_kind;hint 由 reporting 装配点填充 + assert result["next_action_hint"] is None + assert attach_next_action_hint(result)["action"] == "update_baseline" with open(os.path.join(d, "baseline.txt")) as f: assert f.read() == "old\n" @@ -178,6 +183,9 @@ def test_comparator_error_stays_failed_even_with_update_baseline(self): ) assert result["status"] == "failed" - assert result["next_action_hint"]["action"] == "investigate" + assert result["next_action_hint"] is None + assert attach_next_action_hint( + result, update_baseline=True, + )["action"] == "investigate" with open(os.path.join(d, "baseline.txt")) as f: assert f.read() == "old\n" diff --git a/tests/unit/core/test_process_worker.py b/tests/unit/core/test_process_worker.py index 0113e41..5058976 100644 --- a/tests/unit/core/test_process_worker.py +++ b/tests/unit/core/test_process_worker.py @@ -1,7 +1,7 @@ from unittest.mock import patch from symtest.core import process_worker -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep def passed_result(name="case", output="ok\n"): @@ -65,8 +65,8 @@ def test_run_sequence_in_process_aggregates_successful_steps(): case = TestCase( name="sequence", steps=[ - TestCaseStep(command="echo", args=["one"], expected={"return_code": 0}), - TestCaseStep(command="echo", args=["two"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["one"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["two"], expected={"return_code": 0}), ], ).to_dict() @@ -87,9 +87,9 @@ def test_run_sequence_in_process_stops_on_first_failure(): case = TestCase( name="sequence", steps=[ - TestCaseStep(command="echo", args=["one"], expected={"return_code": 0}), - TestCaseStep(command="tool", args=["fail"], expected={"return_code": 0}), - TestCaseStep(command="echo", args=["three"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["one"], expected={"return_code": 0}), + TestStep.from_flat(command="tool", args=["fail"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["three"], expected={"return_code": 0}), ], ).to_dict() diff --git a/tests/unit/core/test_sequence_state.py b/tests/unit/core/test_sequence_state.py index 721bed2..5855583 100644 --- a/tests/unit/core/test_sequence_state.py +++ b/tests/unit/core/test_sequence_state.py @@ -6,7 +6,7 @@ save_sequence_state, save_step_output, ) -from symtest.core.test_case import TestCaseStep +from symtest.core.test_case import TestStep def test_config_hash_is_stable_for_mapping_key_order(): @@ -36,14 +36,14 @@ def test_config_hash_preserves_argument_order(): def test_config_hash_covers_step_controls_and_case_expected(): - step = TestCaseStep( + step = TestStep.from_flat( command="solver", args=["model.dat"], expected={"return_code": 0}, timeout=10, retry_count=1, ) - changed_timeout = TestCaseStep( + changed_timeout = TestStep.from_flat( command="solver", args=["model.dat"], expected={"return_code": 0}, diff --git a/tests/unit/test_migrate.py b/tests/unit/test_migrate.py index dfd59a2..2b1421a 100644 --- a/tests/unit/test_migrate.py +++ b/tests/unit/test_migrate.py @@ -19,8 +19,9 @@ from symtest.commands.migrate import run_migrate from symtest.config.migrate import migrate_case, migrate_config +from symtest.config.normalize import normalize_draft_config from symtest.core.config_loader import parse_test_cases -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep V1_FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" / "migration" / "v1" @@ -166,7 +167,7 @@ def _legacy_case_to_testcase(case): steps = None if "steps" in case: steps = [ - TestCaseStep( + TestStep.from_flat( command=s.get("command", ""), args=s.get("args", []), expected=s.get("expected", {}), @@ -227,11 +228,12 @@ def test_migration_equivalence_invariant(path): # A 侧:legacy dict 经 TestCase 平铺 kwargs 构造(legacy 语义)→ to_dict() side_a = [_legacy_case_to_testcase(tc).to_dict() for tc in v1_cases] - # B 侧:migrate(legacy) 经生产 parse_test_cases → to_dict() + # B 侧:migrate(legacy) → draft normalize(迁移不补必填,宽松形态由 + # DSL 层 normalize 承担)→ 生产 parse_test_cases → to_dict() migrated = migrate_config({"test_cases": v1_cases}) side_b = [ tc.to_dict() - for tc in parse_test_cases(migrated, strict=False) + for tc in parse_test_cases(normalize_draft_config(migrated)) ] assert side_a == side_b diff --git a/tests/unit/tui/conftest.py b/tests/unit/tui/conftest.py index b2a4083..93f977b 100644 --- a/tests/unit/tui/conftest.py +++ b/tests/unit/tui/conftest.py @@ -4,7 +4,7 @@ from pathlib import Path from unittest.mock import patch -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep from symtest.tui.controllers.case_controller import CaseController @@ -19,8 +19,8 @@ def sample_cases(): TestCase( name="multi_step", steps=[ - TestCaseStep(command="echo", args=["step1"], expected={"return_code": 0}), - TestCaseStep(command="echo", args=["step2"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["step1"], expected={"return_code": 0}), + TestStep.from_flat(command="echo", args=["step2"], expected={"return_code": 0}), ], tags=["seq"], description="A sequence test case", ), diff --git a/tests/unit/tui/test_case_controller.py b/tests/unit/tui/test_case_controller.py index 3393405..83f85ef 100644 --- a/tests/unit/tui/test_case_controller.py +++ b/tests/unit/tui/test_case_controller.py @@ -6,9 +6,10 @@ import pytest -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep from symtest.core.config_loader import parse_test_cases from symtest.tui.controllers.case_controller import CaseController +from symtest.config.normalize import normalize_draft_config # --------------------------------------------------------------------------- @@ -30,8 +31,8 @@ def _seq_tc(steps=None, **kwargs) -> TestCase: return TestCase(**defaults) -def _step(cmd="echo", args=None, expected=None) -> TestCaseStep: - return TestCaseStep(command=cmd, +def _step(cmd="echo", args=None, expected=None) -> TestStep: + return TestStep.from_flat(command=cmd, args=args or [], expected=expected or {}) @@ -375,13 +376,13 @@ def test_save_resets_dirty(self): class TestParseFromDict: - """TUI 侧显式宽松形态(1.4 原则 6:strict=False,非隐式 workspace 嗅探)。""" + """TUI 宽松形态自处理(原则 6):draft 先 normalize 再走严格 parser。""" def test_single_cmd_case(self): - result = parse_test_cases({"test_cases": [ + result = parse_test_cases(normalize_draft_config({"test_cases": [ {"name": "simple", "execution": {"command": "echo", "args": ["hello"]}, "expected": {"return_code": 0}, "tags": ["demo"]}, - ]}, strict=False) + ]})) assert len(result) == 1 tc = result[0] assert tc.name == "simple" @@ -390,7 +391,7 @@ def test_single_cmd_case(self): assert tc.tags == ["demo"] def test_sequence_case(self): - result = parse_test_cases({"test_cases": [ + result = parse_test_cases(normalize_draft_config({"test_cases": [ {"name": "seq", "execution": { "steps": [ @@ -398,7 +399,7 @@ def test_sequence_case(self): {"command": "step2", "args": ["b"], "expected": {"return_code": 1}}, ], }}, - ]}, strict=False) + ]})) assert len(result) == 1 tc = result[0] assert tc.name == "seq" @@ -408,9 +409,9 @@ def test_sequence_case(self): assert tc.steps[1].expected == {"return_code": 1} def test_missing_fields_get_defaults(self): - result = parse_test_cases({"test_cases": [ + result = parse_test_cases(normalize_draft_config({"test_cases": [ {"name": "minimal"}, - ]}, strict=False) + ]})) tc = result[0] assert tc.command == "" assert tc.args == [] @@ -422,12 +423,17 @@ def test_empty_list(self): assert parse_test_cases({"test_cases": []}) == [] def test_step_timeout(self): - result = parse_test_cases({"test_cases": [ + result = parse_test_cases(normalize_draft_config({"test_cases": [ {"name": "with_timeout", "execution": { "steps": [{"command": "sleep", "args": ["10"], "expected": {}, "timeout": 30.0}], }}, - ]}, strict=False) + ]})) tc = result[0] assert tc.steps[0].timeout == 30.0 + + def test_strict_parser_rejects_incomplete_case(self): + """core parser 无宽松模式:半成品未经 normalize 直接解析即报错。""" + with pytest.raises(ValueError): + parse_test_cases({"test_cases": [{"name": "minimal"}]}) diff --git a/tests/unit/tui/test_tui_run_test.py b/tests/unit/tui/test_tui_run_test.py index 80b5207..ca09b56 100644 --- a/tests/unit/tui/test_tui_run_test.py +++ b/tests/unit/tui/test_tui_run_test.py @@ -13,7 +13,7 @@ from symtest.tui.widgets.case_table import CaseTable from symtest.tui.widgets.steps_editor import StepsEditor from symtest.tui.widgets.expected_editor import ExpectedEditor -from symtest.core.test_case import TestCase, TestCaseStep +from symtest.core.test_case import TestCase, TestStep # ============================================================================= @@ -280,8 +280,8 @@ async def test_load_refreshes_list(self, sample_config): # Load steps via the widget API steps = [ - TestCaseStep(command="step1", args=["a"], expected={}), - TestCaseStep(command="step2", args=["b"], expected={"return_code": 0}), + TestStep.from_flat(command="step1", args=["a"], expected={}), + TestStep.from_flat(command="step2", args=["b"], expected={"return_code": 0}), ] steps_editor.load(steps) diff --git a/tests/unit/tui/test_widgets_data.py b/tests/unit/tui/test_widgets_data.py index f9f1853..61a9ba4 100644 --- a/tests/unit/tui/test_widgets_data.py +++ b/tests/unit/tui/test_widgets_data.py @@ -8,13 +8,13 @@ import copy import pytest -from symtest.core.test_case import TestCaseStep +from symtest.core.test_case import TestStep from symtest.tui.widgets.steps_editor import StepsEditor from symtest.tui.widgets.expected_editor import ExpectedEditor def _step(cmd="echo", args=None, expected=None, timeout=None, retry_count=0): - return TestCaseStep( + return TestStep.from_flat( command=cmd, args=args or [], expected=expected or {}, From a47cad2ff26cc43e82e16acd6fc0be0b368791f1 Mon Sep 17 00:00:00 2001 From: "xiaotong.wang" <18648483389@163.com> Date: Wed, 2 Sep 2026 15:57:43 +0800 Subject: [PATCH 02/12] =?UTF-8?q?feat(migrate):=20=E6=94=AF=E6=8C=81?= =?UTF-8?q?=E9=80=92=E5=BD=92=E8=BF=81=E7=A7=BB=E6=95=B4=E6=A3=B5=20import?= =?UTF-8?q?=20=E6=A0=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/user_manual.md | 2 +- docs/user_manual_en.md | 2 +- examples/skill-migration/SKILL.md | 17 +- .../references/migration_checklist.md | 6 +- src/symtest/cli.py | 6 + src/symtest/commands/migrate.py | 123 +++++++++-- tests/unit/test_migrate.py | 193 ++++++++++++++++++ 7 files changed, 328 insertions(+), 21 deletions(-) diff --git a/docs/user_manual.md b/docs/user_manual.md index 065515d..708c153 100644 --- a/docs/user_manual.md +++ b/docs/user_manual.md @@ -48,7 +48,7 @@ HDF5 文件比较依赖 `h5py`(已随框架安装)。如需在无 HDF5 环 ## 测试用例定义 -> **Schema v2(1.4)**:配置采用分层 DSL——执行相关字段(`command`、`args`、`timeout`、`retry_count`、`env`、`steps`)位于 `execution` 块,调度相关字段(`depends_on`、`resources`)位于 `scheduling` 块,`expected` 保持在顶层。旧的扁平布局已移除,可使用 `symtest migrate` 迁移。 +> **Schema v2(1.4)**:配置采用分层 DSL——执行相关字段(`command`、`args`、`timeout`、`retry_count`、`env`、`steps`)位于 `execution` 块,调度相关字段(`depends_on`、`resources`)位于 `scheduling` 块,`expected` 保持在顶层。旧的扁平布局已移除,可使用 `symtest migrate` 迁移。迁移会递归处理整棵 `import` 树:默认为每个文件生成 `.v2` 副本并自动重写父文件中的 import 路径;`--in-place` 则原地覆盖所有文件(与 `--output` 互斥)。 ### JSON 格式 diff --git a/docs/user_manual_en.md b/docs/user_manual_en.md index 6328e55..6c99569 100644 --- a/docs/user_manual_en.md +++ b/docs/user_manual_en.md @@ -48,7 +48,7 @@ HDF5 file comparison depends on `h5py` (installed with the framework). If you ne ## Test Case Definition -> **Schema v2 (1.4)**: The config uses a layered DSL — execution-related fields (`command`, `args`, `timeout`, `retry_count`, `env`, `steps`) live in the `execution` block, scheduling-related fields (`depends_on`, `resources`) live in the `scheduling` block, and `expected` stays at the top level. The old flat layout was removed; use `symtest migrate` to migrate. +> **Schema v2 (1.4)**: The config uses a layered DSL — execution-related fields (`command`, `args`, `timeout`, `retry_count`, `env`, `steps`) live in the `execution` block, scheduling-related fields (`depends_on`, `resources`) live in the `scheduling` block, and `expected` stays at the top level. The old flat layout was removed; use `symtest migrate` to migrate. Migration recursively covers the whole `import` tree: by default each file gets a `.v2` copy with import paths rewritten automatically; `--in-place` overwrites all files in place (mutually exclusive with `--output`). ### JSON Format diff --git a/examples/skill-migration/SKILL.md b/examples/skill-migration/SKILL.md index a945a25..6d80bc1 100644 --- a/examples/skill-migration/SKILL.md +++ b/examples/skill-migration/SKILL.md @@ -42,8 +42,15 @@ old config → symtest migrate → new schema → symtest validate → manual re symtest migrate old.json --output new.json ``` Default output is `.v2` (e.g. `old.json` → `old.v2.json`). - `migrate` is a mechanical field move; it does not expand imports, resolve - paths, or validate required fields. + `migrate` recursively follows the whole `import` tree as a mechanical + field move; it never expands (inlines) imports, resolves paths, or + validates required fields: + - Default: every file in the tree gets a `.v2` copy and import + paths in migrated parents are rewritten to the `.v2` names, so the + migrated tree is self-consistent for `symtest validate`. + - `--in-place`: every file in the tree is overwritten in place (file + names, formats and import paths unchanged; mutually exclusive with + `--output`). 2. **Validate the result**: ```bash @@ -77,8 +84,10 @@ old config → symtest migrate → new schema → symtest validate → manual re - Idempotent: input already containing `execution` is passed through unchanged (deep copy). - JSON/YAML in → same format out (chosen by output file extension). -- Import entries are preserved as-is (not expanded) so the migrated project - keeps its file split; each imported file must be migrated separately. +- Imports are followed recursively (never expanded inline) so the migrated + project keeps its file split: default mode writes `.v2` copies + with import paths rewritten; `--in-place` overwrites every file in place + (originals are not kept). - Unknown top-level fields are preserved verbatim — `symtest validate` and this skill's checklist decide whether they are still meaningful. diff --git a/examples/skill-migration/references/migration_checklist.md b/examples/skill-migration/references/migration_checklist.md index 85c43a1..589c38f 100644 --- a/examples/skill-migration/references/migration_checklist.md +++ b/examples/skill-migration/references/migration_checklist.md @@ -69,8 +69,10 @@ will run before their dependencies). ## 6. Import structures -- `migrate` does NOT follow imports: every imported sub-file must be migrated - on its own. List all files reachable via `import` and migrate each. +- `migrate` follows imports recursively: default mode writes `.v2` + copies and rewrites import paths; `--in-place` overwrites each file in + place. Verify every file reachable via `import` was migrated (the command + prints one output path per migrated file). - Import-level `tags` injection still works in v2; confirm split files no longer rely on v1-only top-level case fields. - Circular imports remain an error (validation stage), unchanged. diff --git a/src/symtest/cli.py b/src/symtest/cli.py index 8eb5803..9756e1e 100644 --- a/src/symtest/cli.py +++ b/src/symtest/cli.py @@ -208,6 +208,12 @@ def create_parser(): '--output', '-o', help='Output path (default: .v2, e.g. old.json -> old.v2.json)', ) + migrate_parser.add_argument( + '--in-place', action='store_true', + help='Migrate the config and every file it imports recursively, ' + 'overwriting each file in place (originals are not kept; ' + 'mutually exclusive with --output)', + ) # ---- Compare command ---- compare_parser = subparsers.add_parser('compare', help='Compare two files') diff --git a/src/symtest/commands/migrate.py b/src/symtest/commands/migrate.py index e053f80..7827c02 100644 --- a/src/symtest/commands/migrate.py +++ b/src/symtest/commands/migrate.py @@ -1,7 +1,17 @@ """``symtest migrate`` 子命令实现(1.4 Phase 4 迁移设计第一层)。 确定性转换:加载 v1 配置 → ``config.migrate.migrate_config`` 机械映射 → -按输出扩展名写出 JSON/YAML。不做 import 展开(``expand=False``)、 +按输出扩展名写出 JSON/YAML。 + +支持递归迁移整棵 ``import`` 树(两阶段:先把全部文件加载并迁移到内存, +任一失败则不写任何文件,全部成功后才统一写盘): + +- 默认模式:每个文件写出 ``.v2`` 兄弟副本(原文件不动), + 并把已迁移配置中的 import 路径重写为对应的 ``.v2`` 文件名, + 迁移产物整棵树自洽、可直接 ``symtest validate``; +- ``--in-place``:整树原地覆盖(文件名/格式/import 路径全不变), + 默认关闭,开启即接受覆盖风险;与 ``--output`` 互斥。 + 不做路径解析、不校验必填 —— 校验归 ``symtest validate``,人工判断项归 迁移复查 Skill(见 examples/skill-migration)。 """ @@ -10,8 +20,10 @@ import logging from pathlib import Path +from typing import Any, Dict, List, Tuple -from ..config.config_io import load_config, save_config +from ..config.config_io import save_config +from ..config.import_expander import _load_raw_config from ..config.migrate import migrate_config logger = logging.getLogger("symtest.commands.migrate") @@ -24,12 +36,74 @@ def _resolve(path_str: str, workspace: Path) -> Path: return path.resolve() +def _collect_import_tree(entry_path: Path) -> List[Path]: + """DFS 后序收集入口文件及全部 import 子文件。 + + - processed 集合保证菱形 import(同一文件被多个父文件引用)只收集一次; + - 递归栈检测真循环 import 并抛 RuntimeError; + - 子文件缺失抛 FileNotFoundError; + - 返回顺序保证子文件先于引用它的父文件。 + """ + ordered: List[Path] = [] + processed: set = set() + + def _walk(path: Path, stack: Tuple[str, ...]) -> None: + canonical = str(path.resolve()) + if canonical in processed: + return + if canonical in stack: + raise RuntimeError( + f"Circular import detected: {canonical} " + f"(chain: {' -> '.join(stack + (canonical,))})" + ) + config = _load_raw_config(path) + for item in config.get("test_cases", []) or []: + if isinstance(item, dict) and "import" in item: + sub_path = (path.parent / str(item["import"])).resolve() + if not sub_path.exists(): + raise FileNotFoundError( + f"Imported config file not found: {sub_path} " + f"(referenced from {path})" + ) + _walk(sub_path, stack + (canonical,)) + processed.add(canonical) + ordered.append(path) + + _walk(entry_path, ()) + return ordered + + +def _v2_sibling(import_path: str) -> str: + """'cases/sub.json' -> 'cases/sub.v2.json'(纯字符串变换,保留原分隔符)。""" + sep_idx = max(import_path.rfind("/"), import_path.rfind("\\")) + if sep_idx >= 0: + directory, filename = import_path[: sep_idx + 1], import_path[sep_idx + 1:] + else: + directory, filename = "", import_path + stem, dot, ext = filename.rpartition(".") + if not dot or not stem: + # 无扩展名(或 .gitignore 类隐藏名):直接追加后缀 + return f"{directory}{filename}.v2" + return f"{directory}{stem}.v2.{ext}" + + +def _rewrite_import_paths(config: Dict[str, Any]) -> Dict[str, Any]: + """把 config 中 import 条目的路径改写为 .v2 兄弟名(就地修改并返回)。 + + import 条目中的其他字段(如 tags)原样保留。 + """ + for item in config.get("test_cases", []) or []: + if isinstance(item, dict) and "import" in item: + item["import"] = _v2_sibling(str(item["import"])) + return config + + def run_migrate(args) -> bool: - """CLI 壳:load_config(expand=False) → migrate_config → save_config。 + """CLI 壳:收集 import 树 → 整树 migrate_config → 按模式统一写盘。 Returns ------- - True 成功;输入不存在/加载失败/输出格式不支持返回 False。 + True 成功;输入不存在/加载失败/循环 import/输出格式不支持返回 False。 """ workspace = Path(getattr(args, "workspace", None) or Path.cwd()).resolve() @@ -38,26 +112,49 @@ def run_migrate(args) -> bool: logger.error("Configuration file not found: %s", input_path) return False + in_place = getattr(args, "in_place", False) + output_arg = getattr(args, "output", None) + if in_place and output_arg: + logger.error("--output cannot be combined with --in-place") + return False + try: - config = load_config(input_path, expand=False) - migrated = migrate_config(config) + files = _collect_import_tree(input_path) + # 第一阶段:整树加载并迁移到内存,任一失败则不写任何文件 + pending: List[Tuple[Path, Dict[str, Any]]] = [ + (f, migrate_config(_load_raw_config(f))) for f in files + ] except Exception as exc: logger.error("Migration failed: %s", exc) return False - output_arg = getattr(args, "output", None) - if output_arg: - output_path = _resolve(output_arg, workspace) + if output_arg and not in_place: + entry_output = _resolve(output_arg, workspace) else: # 默认输出 .v2<原扩展名>,如 old.json → old.v2.json - output_path = input_path.parent / f"{input_path.stem}.v2{input_path.suffix}" + entry_output = input_path.parent / f"{input_path.stem}.v2{input_path.suffix}" + written: List[Path] = [] try: - save_config(migrated, output_path) + # 第二阶段:统一写盘 + for src, migrated in pending: + if in_place: + target = src + else: + # import 路径指向 .v2 子副本(--output 仅移动入口文件位置, + # 隐含假设输出与入口同目录,否则需手工核对相对路径) + _rewrite_import_paths(migrated) + if src == input_path: + target = entry_output + else: + target = src.parent / f"{src.stem}.v2{src.suffix}" + save_config(migrated, target) + written.append(target) except ValueError as exc: logger.error("%s", exc) return False - logger.info("Migrated config written to: %s", output_path) - print(str(output_path)) + for target in written: + logger.info("Migrated config written to: %s", target) + print(str(target)) return True diff --git a/tests/unit/test_migrate.py b/tests/unit/test_migrate.py index 2b1421a..145c7ee 100644 --- a/tests/unit/test_migrate.py +++ b/tests/unit/test_migrate.py @@ -8,6 +8,8 @@ legacy config ──migrate──▶ new config ──new parser──▶ Normalized B 断言 A == B +3. 递归 import 树迁移:默认 .v2 副本 + 路径重写,--in-place 原地覆盖。 + 语料:tests/fixtures/migration/v1/(tests/ 与 examples/ 存量 v1 配置的原件拷贝)。 """ @@ -295,3 +297,194 @@ def test_migrate_unsupported_output_format_fails(self, tmp_path): assert run_migrate( self._args(src, output=tmp_path / "out.txt", workspace=str(tmp_path)) ) is False + + +# --------------------------------------------------------------------------- +# 递归 import 树迁移(默认 .v2 副本 + 路径重写 / --in-place 原地覆盖) +# --------------------------------------------------------------------------- + +def _v1_case(name): + return { + "name": name, + "command": "echo", + "args": [name], + "expected": {"return_code": 0}, + } + + +class TestRunMigrateTree: + def _args(self, config, output=None, workspace=None, in_place=False): + return argparse.Namespace( + config_file=str(config), + output=str(output) if output else None, + workspace=workspace, + in_place=in_place, + ) + + @staticmethod + def _read(path): + if path.suffix.lower() in (".yaml", ".yml"): + import yaml + return yaml.safe_load(path.read_text(encoding="utf-8")) + return json.loads(path.read_text(encoding="utf-8")) + + def _three_level_tree(self, tmp_path): + """main.json -> sub.json -> subsub.json 三层 v1 import 树。""" + (tmp_path / "subsub.json").write_text( + json.dumps({"test_cases": [_v1_case("leaf")]}), encoding="utf-8") + (tmp_path / "sub.json").write_text(json.dumps({ + "test_cases": [{"import": "subsub.json"}, _v1_case("mid")], + }), encoding="utf-8") + (tmp_path / "main.json").write_text(json.dumps({ + "test_cases": [ + {"import": "sub.json", "tags": ["api"]}, + _v1_case("root"), + ], + }), encoding="utf-8") + + def test_default_mode_migrates_whole_tree(self, tmp_path): + self._three_level_tree(tmp_path) + assert run_migrate( + self._args(tmp_path / "main.json", workspace=str(tmp_path)) + ) is True + + main_v2 = tmp_path / "main.v2.json" + sub_v2 = tmp_path / "sub.v2.json" + subsub_v2 = tmp_path / "subsub.v2.json" + for p in (main_v2, sub_v2, subsub_v2): + assert p.exists() + + # 父文件 import 路径已重写为 .v2 名,tags 等其他字段保留 + migrated_main = self._read(main_v2) + assert migrated_main["test_cases"][0]["import"] == "sub.v2.json" + assert migrated_main["test_cases"][0]["tags"] == ["api"] + assert migrated_main["test_cases"][1]["execution"]["command"] == "echo" + migrated_sub = self._read(sub_v2) + assert migrated_sub["test_cases"][0]["import"] == "subsub.v2.json" + assert self._read(subsub_v2)["test_cases"][0]["execution"]["args"] == ["leaf"] + + # 原文件保持 v1 未动 + for name in ("main.json", "sub.json", "subsub.json"): + for tc in self._read(tmp_path / name)["test_cases"]: + if "import" not in tc: + assert "execution" not in tc + + def test_in_place_overwrites_tree(self, tmp_path): + self._three_level_tree(tmp_path) + assert run_migrate( + self._args(tmp_path / "main.json", + workspace=str(tmp_path), in_place=True) + ) is True + + # 原文件原地变为 v2,import 路径不变,不产生 .v2 副本 + migrated_main = self._read(tmp_path / "main.json") + assert migrated_main["test_cases"][0]["import"] == "sub.json" + assert migrated_main["test_cases"][1]["execution"]["command"] == "echo" + assert self._read(tmp_path / "sub.json")["test_cases"][0]["import"] \ + == "subsub.json" + assert self._read(tmp_path / "subsub.json")["test_cases"][0] \ + ["execution"]["args"] == ["leaf"] + assert not (tmp_path / "main.v2.json").exists() + + def test_diamond_import_migrated_once(self, tmp_path): + (tmp_path / "d.json").write_text( + json.dumps({"test_cases": [_v1_case("shared")]}), encoding="utf-8") + for name in ("b.json", "c.json"): + (tmp_path / name).write_text(json.dumps( + {"test_cases": [{"import": "d.json"}]}), encoding="utf-8") + (tmp_path / "main.json").write_text(json.dumps({ + "test_cases": [{"import": "b.json"}, {"import": "c.json"}], + }), encoding="utf-8") + + assert run_migrate( + self._args(tmp_path / "main.json", + workspace=str(tmp_path), in_place=True) + ) is True + # 菱形引用不误报循环,d.json 已迁移 + assert self._read(tmp_path / "d.json")["test_cases"][0] \ + ["execution"]["command"] == "echo" + + def test_circular_import_writes_nothing(self, tmp_path): + (tmp_path / "a.json").write_text( + json.dumps({"test_cases": [{"import": "b.json"}]}), encoding="utf-8") + (tmp_path / "b.json").write_text( + json.dumps({"test_cases": [{"import": "a.json"}]}), encoding="utf-8") + snapshot = {p.name: p.read_text(encoding="utf-8") for p in tmp_path.iterdir()} + + assert run_migrate( + self._args(tmp_path / "a.json", workspace=str(tmp_path)) + ) is False + # 两阶段保证:任一失败不写任何文件 + assert {p.name: p.read_text(encoding="utf-8") for p in tmp_path.iterdir()} \ + == snapshot + + def test_missing_import_writes_nothing(self, tmp_path): + (tmp_path / "main.json").write_text( + json.dumps({"test_cases": [{"import": "ghost.json"}]}), + encoding="utf-8") + before = (tmp_path / "main.json").read_text(encoding="utf-8") + + assert run_migrate( + self._args(tmp_path / "main.json", workspace=str(tmp_path)) + ) is False + assert (tmp_path / "main.json").read_text(encoding="utf-8") == before + assert not (tmp_path / "main.v2.json").exists() + + def test_sub_without_test_cases_writes_nothing(self, tmp_path): + (tmp_path / "sub.json").write_text( + json.dumps({"setup": {}}), encoding="utf-8") + (tmp_path / "main.json").write_text( + json.dumps({"test_cases": [{"import": "sub.json"}]}), + encoding="utf-8") + + assert run_migrate( + self._args(tmp_path / "main.json", workspace=str(tmp_path)) + ) is False + assert not (tmp_path / "main.v2.json").exists() + assert not (tmp_path / "sub.v2.json").exists() + + def test_output_and_in_place_conflict(self, tmp_path): + src = tmp_path / "old.json" + src.write_text(json.dumps({"test_cases": []}), encoding="utf-8") + assert run_migrate(self._args( + src, output=tmp_path / "new.json", + workspace=str(tmp_path), in_place=True, + )) is False + assert not (tmp_path / "new.json").exists() + assert json.loads(src.read_text(encoding="utf-8")) == {"test_cases": []} + + def test_in_place_idempotent_on_v2_tree(self, tmp_path): + self._three_level_tree(tmp_path) + assert run_migrate( + self._args(tmp_path / "main.json", + workspace=str(tmp_path), in_place=True) + ) is True + first_pass = {p.name: p.read_text(encoding="utf-8") for p in tmp_path.iterdir()} + assert run_migrate( + self._args(tmp_path / "main.json", + workspace=str(tmp_path), in_place=True) + ) is True + assert {p.name: p.read_text(encoding="utf-8") for p in tmp_path.iterdir()} \ + == first_pass + + def test_mixed_json_yaml_tree(self, tmp_path): + (tmp_path / "sub.yaml").write_text( + "test_cases:\n" + " - name: y\n" + " command: echo\n" + " args: []\n" + " expected: {}\n", + encoding="utf-8", + ) + (tmp_path / "main.json").write_text( + json.dumps({"test_cases": [{"import": "sub.yaml"}]}), + encoding="utf-8") + + assert run_migrate( + self._args(tmp_path / "main.json", workspace=str(tmp_path)) + ) is True + assert (tmp_path / "sub.v2.yaml").exists() + assert self._read(tmp_path / "sub.v2.yaml")["test_cases"][0] \ + ["execution"]["command"] == "echo" + assert self._read(tmp_path / "main.v2.json")["test_cases"][0]["import"] \ + == "sub.v2.yaml" From 0d19bf978a1ef62cf9b302ffbf5f6649cd22cfc7 Mon Sep 17 00:00:00 2001 From: "xiaotong.wang" <18648483389@163.com> Date: Thu, 3 Sep 2026 14:16:04 +0800 Subject: [PATCH 03/12] =?UTF-8?q?fix(file=5Fcomparator):=20=E4=BF=AE?= =?UTF-8?q?=E6=AD=A3=E8=AF=AF=E5=B7=AE=E7=BB=9F=E8=AE=A1=E5=8F=A3=E5=BE=84?= =?UTF-8?q?=E4=B8=BA=E5=85=A8=E4=BD=93=E6=95=B0=E5=80=BC=E5=8D=95=E5=85=83?= =?UTF-8?q?=E6=A0=BC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/design.md | 2 +- docs/user_manual.md | 10 +-- src/symtest/file_comparator/csv_comparator.py | 4 +- src/symtest/file_comparator/h5_comparator.py | 6 +- .../file_comparator/numeric_compare.py | 83 ++++++++++--------- .../file_comparator/test_error_analysis.py | 26 +++++- .../file_comparator/test_numeric_compare.py | 8 +- 7 files changed, 85 insertions(+), 54 deletions(-) diff --git a/docs/design.md b/docs/design.md index 0d9d636..95317b2 100644 --- a/docs/design.md +++ b/docs/design.md @@ -159,7 +159,7 @@ PathResolver 解析(系统命令直通、shell builtin 平台包装、复合 - `compare_files` 是一等断言,经 ComparatorFactory 按类型分发(详见 §6) - 所有断言可选;未声明的字段不做校验 -- `--error-analysis` 为 CSV/H5 数值比较提供流式误差统计:`total_numeric_cells` / `mismatched_cells` / `max_abs_error` / `max_rel_error` / `mean_abs_error` / `rms_abs_error` +- `--error-analysis` 为 CSV/H5 数值比较提供流式误差统计:`total_numeric_cells` / `mismatched_cells` / `max_abs_error` / `max_rel_error` / `mean_abs_error` / `rms_abs_error`;幅值统计(max/mean/rms)覆盖全体参与比较的数值单元格(含通过格),mean/rms 以 `total_numeric_cells` 为分母。`--error-analysis-all` 额外对通过用例输出统计,除此之外两者行为一致 ## 5. 运行时状态持久化(`.symtest/`) diff --git a/docs/user_manual.md b/docs/user_manual.md index 708c153..70c4746 100644 --- a/docs/user_manual.md +++ b/docs/user_manual.md @@ -1795,12 +1795,12 @@ CSV 比较按行列结构逐单元格比对;数值单元格在容差范围内 |---|---| | `total_numeric_cells` | 参与数值比较的单元格总数 | | `mismatched_cells` | 超出容差的单元格数 | -| `max_abs_error` | 最大绝对误差及其位置 | -| `max_rel_error` | 最大相对误差及其位置 | -| `mean_abs_error` | 平均绝对误差 | -| `rms_abs_error` | 均方根绝对误差(RMSE) | +| `max_abs_error` | 全体数值单元格中的最大绝对误差及其位置 | +| `max_rel_error` | 全体数值单元格中的最大相对误差及其位置 | +| `mean_abs_error` | 全体数值单元格的平均绝对误差(分母为 `total_numeric_cells`) | +| `rms_abs_error` | 全体数值单元格的均方根绝对误差(RMSE,分母为 `total_numeric_cells`) | -统计是**流式**计算的,不依赖差异截断,覆盖全体数值单元格。默认仅失败的比较输出统计,通过的比较不输出。 +统计是**流式**计算的,不依赖差异截断,幅值统计(max/mean/rms)覆盖**全体**参与比较的数值单元格(含在容差内通过的单元格),可用于观察通过格离容差的余量;非有限值(NaN/inf)单元格不参与幅值统计。两个参数的统计口径完全一致,区别仅在于:默认仅失败的比较输出统计,通过的比较不输出。 如需让**通过**的用例也输出统计信息(例如用于监控容差余量),可改用 `--error-analysis-all`(它会隐含启用 `--error-analysis`): diff --git a/src/symtest/file_comparator/csv_comparator.py b/src/symtest/file_comparator/csv_comparator.py index b597ecc..62e4e04 100644 --- a/src/symtest/file_comparator/csv_comparator.py +++ b/src/symtest/file_comparator/csv_comparator.py @@ -102,7 +102,9 @@ def compare_content(self, content1, content2): """ self._error_stats = None # Reset per comparison - if content1 == content2: + # Fast path only when stats are not requested; with error_analysis we + # must still traverse all numeric cells to build the statistics. + if content1 == content2 and not self.error_analysis: return True, [], False differences = [] diff --git a/src/symtest/file_comparator/h5_comparator.py b/src/symtest/file_comparator/h5_comparator.py index 82a9bb3..02cd87e 100644 --- a/src/symtest/file_comparator/h5_comparator.py +++ b/src/symtest/file_comparator/h5_comparator.py @@ -470,6 +470,8 @@ def compare_content(self, content1, content2): identical = False # ── Store error stats ── + # Mean / RMS use total numeric cells as the denominator (stats cover + # ALL compared cells, matching ones included). if self.error_analysis and _ea_total > 0: self._error_stats = { "total_numeric_cells": _ea_total, @@ -478,8 +480,8 @@ def compare_content(self, content1, content2): "max_abs_error_at": _ea_max_abs_at, "max_rel_error": _ea_max_rel, "max_rel_error_at": _ea_max_rel_at, - "mean_abs_error": _ea_sum_abs / _ea_mismatched if _ea_mismatched > 0 else 0.0, - "rms_abs_error": (np.sqrt(_ea_sum_sq / _ea_mismatched) if _ea_mismatched > 0 else 0.0), + "mean_abs_error": _ea_sum_abs / _ea_total, + "rms_abs_error": np.sqrt(_ea_sum_sq / _ea_total), } return identical, differences, False diff --git a/src/symtest/file_comparator/numeric_compare.py b/src/symtest/file_comparator/numeric_compare.py index 1659703..6c37299 100644 --- a/src/symtest/file_comparator/numeric_compare.py +++ b/src/symtest/file_comparator/numeric_compare.py @@ -143,45 +143,48 @@ def compare_numeric(expected, actual, rtol=1e-5, atol=1e-8, # Flat indices (within the original array) of kept positions. keep_idx = np.flatnonzero(keep) - if mismatched == 0: - return NumericComparisonStats( - total=total, - mismatched=0, - mismatch_mask=np.zeros(expected.size, dtype=bool), - max_abs_error=None, - max_abs_error_index=None, - max_rel_error=None, - max_rel_error_index=None, - mean_abs_error=0.0, - rms_abs_error=0.0, - sum_abs_error=0.0, - sum_sq_abs_error=0.0, - ) - - # ── Error statistics (over mismatched cells only) ── - mism_abs_err = abs_err[mismatch] - mism_expected = exp_keep[mismatch] - - sum_abs = float(np.sum(mism_abs_err)) - sum_sq = float(np.sum(mism_abs_err ** 2)) - - max_abs_error = float(np.max(mism_abs_err)) - max_abs_keep_pos = int(np.argmax(mism_abs_err)) - max_abs_error_index = int(keep_idx[mismatch][max_abs_keep_pos]) - - # Relative error: inf when reference value is zero but error is non-zero (CSV semantics). - ref_abs = np.abs(mism_expected) - rel_err = np.full(mism_abs_err.shape, np.inf) - nonzero = ref_abs > 0 - with np.errstate(invalid="ignore"): - rel_err[nonzero] = mism_abs_err[nonzero] / ref_abs[nonzero] - - max_rel_error = float(np.max(rel_err)) - max_rel_keep_pos = int(np.argmax(rel_err)) - max_rel_error_index = int(keep_idx[mismatch][max_rel_keep_pos]) - - mean_abs_error = sum_abs / mismatched if mismatched > 0 else 0.0 - rms_abs_error = np.sqrt(sum_sq / mismatched) if mismatched > 0 else 0.0 + # ── Error statistics (over ALL kept numeric cells, matching ones included) ── + # Non-finite cells (NaN / inf) are excluded from magnitude statistics; + # identical NaN / inf pairs contribute zero error by definition. + finite = np.isfinite(exp_keep) & np.isfinite(act_keep) + stat_idx = keep_idx[finite] + stat_abs_err = abs_err[finite] + stat_expected = exp_keep[finite] + n_stat = int(np.sum(finite)) + + if n_stat > 0: + sum_abs = float(np.sum(stat_abs_err)) + sum_sq = float(np.sum(stat_abs_err ** 2)) + + max_abs_error = float(np.max(stat_abs_err)) + max_abs_stat_pos = int(np.argmax(stat_abs_err)) + max_abs_error_index = int(stat_idx[max_abs_stat_pos]) + + # Relative error: inf when reference value is zero but error is non-zero; + # zero when the reference is zero and the values agree exactly. + ref_abs = np.abs(stat_expected) + rel_err = np.full(stat_abs_err.shape, np.inf) + nonzero = ref_abs > 0 + with np.errstate(invalid="ignore"): + rel_err[nonzero] = stat_abs_err[nonzero] / ref_abs[nonzero] + rel_err[~nonzero & (stat_abs_err == 0)] = 0.0 + + max_rel_error = float(np.max(rel_err)) + max_rel_stat_pos = int(np.argmax(rel_err)) + max_rel_error_index = int(stat_idx[max_rel_stat_pos]) + + mean_abs_error = sum_abs / total + rms_abs_error = np.sqrt(sum_sq / total) + else: + # No finite cells: magnitudes are undefined. + sum_abs = 0.0 + sum_sq = 0.0 + max_abs_error = None + max_abs_error_index = None + max_rel_error = None + max_rel_error_index = None + mean_abs_error = 0.0 + rms_abs_error = 0.0 # Build full-length mismatch mask (True where values differ beyond tolerance). mismatch_mask = np.zeros(expected.size, dtype=bool) @@ -195,7 +198,7 @@ def compare_numeric(expected, actual, rtol=1e-5, atol=1e-8, max_abs_error_index=max_abs_error_index, max_rel_error=max_rel_error, max_rel_error_index=max_rel_error_index, - mean_abs_error=mean_abs_error, + mean_abs_error=float(mean_abs_error), rms_abs_error=float(rms_abs_error), sum_abs_error=sum_abs, sum_sq_abs_error=sum_sq, diff --git a/tests/unit/file_comparator/test_error_analysis.py b/tests/unit/file_comparator/test_error_analysis.py index 3912a34..5ed96f0 100644 --- a/tests/unit/file_comparator/test_error_analysis.py +++ b/tests/unit/file_comparator/test_error_analysis.py @@ -1,4 +1,5 @@ """Unit tests for error_analysis streaming statistics in CSV/H5 comparators.""" +import math import os import tempfile @@ -43,13 +44,34 @@ def test_error_analysis_accumulates_stats(self): assert stats["mean_abs_error"] >= 0.0 def test_error_analysis_all_matching(self): - """When all numeric cells match, stats show zero mismatches.""" + """When all numeric cells match, stats cover all cells with zero errors.""" cmp = CsvComparator(error_analysis=True) content1 = [["1.0", "2.0"], ["3.0", "4.0"]] content2 = [["1.0", "2.0"], ["3.0", "4.0"]] identical, diffs, truncated = cmp.compare_content(content1, content2) assert identical - assert cmp._error_stats is None # no stats when identical (compare_content returns early) + assert cmp._error_stats is not None + stats = cmp._error_stats + assert stats["total_numeric_cells"] == 4 + assert stats["mismatched_cells"] == 0 + assert stats["max_abs_error"] == pytest.approx(0.0, abs=1e-12) + assert stats["mean_abs_error"] == pytest.approx(0.0, abs=1e-12) + assert stats["rms_abs_error"] == pytest.approx(0.0, abs=1e-12) + + def test_error_analysis_stats_over_all_cells(self): + """Magnitude stats cover ALL cells; mean/rms divide by total, not mismatches.""" + cmp = CsvComparator(error_analysis=True) + # Cell 1 matches exactly (0.0 error), cell 2 differs by 3.0. + content1 = [["1.0", "2.0"]] + content2 = [["1.0", "5.0"]] + identical, diffs, truncated = cmp.compare_content(content1, content2) + stats = cmp._error_stats + assert stats["total_numeric_cells"] == 2 + assert stats["mismatched_cells"] == 1 + assert stats["max_abs_error"] == pytest.approx(3.0, abs=1e-9) + # Old semantics: mean over mismatched only = 3.0; now over all cells = 1.5. + assert stats["mean_abs_error"] == pytest.approx(1.5, abs=1e-9) + assert stats["rms_abs_error"] == pytest.approx(3.0 / math.sqrt(2), abs=1e-9) def test_error_analysis_tracks_max_abs_error(self): """Verify max_abs_error tracks the largest absolute difference.""" diff --git a/tests/unit/file_comparator/test_numeric_compare.py b/tests/unit/file_comparator/test_numeric_compare.py index d741895..ca7af18 100644 --- a/tests/unit/file_comparator/test_numeric_compare.py +++ b/tests/unit/file_comparator/test_numeric_compare.py @@ -12,8 +12,9 @@ def test_exact_equality(): assert res.total == 3 assert res.mismatched == 0 assert np.all(~res.mismatch_mask) - assert res.max_abs_error is None - assert res.max_rel_error is None + # Stats cover ALL cells: zero errors everywhere. + assert res.max_abs_error == pytest.approx(0.0, abs=1e-12) + assert res.max_rel_error == pytest.approx(0.0, abs=1e-12) def test_within_rtol(): @@ -171,7 +172,8 @@ def test_collect_stats_false_all_filtered_out(): def test_no_mismatch_collect_stats_true(): res = compare_numeric([1.0, 2.0], [1.0, 2.0], collect_stats=True) assert res.mismatched == 0 - assert res.max_abs_error is None + # Stats cover ALL cells: zero errors everywhere, denominators = total. + assert res.max_abs_error == pytest.approx(0.0, abs=1e-12) assert res.mean_abs_error == 0.0 assert res.rms_abs_error == 0.0 From e66b021ee45345050697bd9bec3d7100ff15fea0 Mon Sep 17 00:00:00 2001 From: "xiaotong.wang" <18648483389@163.com> Date: Fri, 4 Sep 2026 13:41:47 +0800 Subject: [PATCH 04/12] =?UTF-8?q?docs:=20=E6=9B=B4=E6=96=B0=E6=AF=94?= =?UTF-8?q?=E8=BE=83=E5=99=A8=E4=B8=89=E6=B3=B3=E9=81=93=E6=9E=B6=E6=9E=84?= =?UTF-8?q?=E6=96=87=E6=A1=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/design.md | 51 +++- docs/user_manual.md | 175 ++++++++++++-- docs/user_manual_en.md | 183 +++++++++++++-- examples/plugins/README.md | 126 ++++++---- .../plugins/hourglass_tangent_comparator.py | 61 +++-- examples/skill/SKILL.md | 69 +++++- .../templates/extract_channels_script.py | 77 ++++++ .../templates/my_analysis_comparator.py | 78 ++++--- .../assets/templates/my_channel_extractor.py | 80 +++++++ examples/skill/references/user_manual.md | 175 ++++++++++++-- src/symtest/__init__.py | 2 +- src/symtest/commands/compare.py | 19 +- src/symtest/config/config_schema.py | 36 ++- src/symtest/core/validation/assertions.py | 43 +++- src/symtest/core/validation/validator.py | 40 +++- src/symtest/file_comparator/__init__.py | 42 +++- .../file_comparator/base_comparator.py | 169 ++++++-------- .../file_comparator/binary_comparator.py | 7 +- .../file_comparator/extractor_comparator.py | 197 ++++++++++++++++ src/symtest/file_comparator/factory.py | 29 ++- .../file_comparator/file_comparator_base.py | 131 +++++++++++ src/symtest/file_comparator/h5_comparator.py | 4 +- src/symtest/file_comparator/result.py | 61 +++++ .../file_comparator/script_comparator.py | 49 ++-- .../script_extract_comparator.py | 160 +++++++++++++ .../file_comparator/text_comparator.py | 4 +- src/symtest/utils/report_generator.py | 78 ++++++- tests/unit/core/test_assertions.py | 1 + .../test_extractor_channels.py | 220 ++++++++++++++++++ .../file_comparator/test_script_comparator.py | 19 +- .../file_comparator/test_script_extract.py | 184 +++++++++++++++ .../file_comparator/test_workspace_plugins.py | 77 +++--- tests/unit/utils/test_report_generator.py | 117 ++++++++++ 33 files changed, 2344 insertions(+), 420 deletions(-) create mode 100644 examples/skill/assets/templates/extract_channels_script.py create mode 100644 examples/skill/assets/templates/my_channel_extractor.py create mode 100644 src/symtest/file_comparator/extractor_comparator.py create mode 100644 src/symtest/file_comparator/file_comparator_base.py create mode 100644 src/symtest/file_comparator/script_extract_comparator.py create mode 100644 tests/unit/file_comparator/test_extractor_channels.py create mode 100644 tests/unit/file_comparator/test_script_extract.py diff --git a/docs/design.md b/docs/design.md index 95317b2..37f53e5 100644 --- a/docs/design.md +++ b/docs/design.md @@ -175,24 +175,55 @@ PathResolver 解析(系统命令直通、shell builtin 平台包装、复合 ## 6. 文件比较子系统 -### 6.1 比较器分层 +### 6.1 三泳道比较器架构 -- 文本系比较器共享 difflib 行级基底,json / csv / xml 为其结构化特化(键对齐 / 列结构 / DOM 对齐) -- h5 面向科学数据集;binary 流式分块 + LCS 相似度;script 委托外部脚本 -- 统一返回 ComparisonResult(identical / differences / error / script command_output),支持 text / json / html 渲染;支持行列窗口范围参数截取后比较 +所有比较器共享根契约 `ComparatorBase.compare(ctx) -> ComparisonResult`: +框架构建 `CompareContext`(workspace / actual / baseline / params / error_analysis) +并统一做路径解析——`actual`/`baseline` 以及插件 `path_params` 声明的构造参数 +均按 workspace 解析,插件不得自行对 CWD resolve。三条泳道职责互斥: -### 6.2 工厂与插件发现 +| 泳道 | 基类 | 契约 | verdict 归属 | 内置实例 | +|---|---|---|---|---| +| 文件泳道 | `FileComparator` | `read_content` + `compare_content`(两文件模型,行列窗口、line N 偏移、chunk_size、similarity 均属本泳道) | 插件 | text / json / csv / xml / h5 / binary | +| 数据泳道 | `ExtractorComparator` | `extract(ctx) -> {channel: ChannelData}`;框架逐通道跑 `compare_numeric`,per-channel 容差(`channels` / `default_channel`),聚合 `ChannelResult` | **框架**(容差语义对 AI 消费端可预期) | script_extract、extractor 插件 | +| 自主泳道 | 直接继承根 | `compare(ctx)` 全权判定 | 插件 | script、hourglass 式分析插件 | -- `file_type` 取值:`text` / `json` / `csv` / `xml` / `h5` / `binary` / `script`;工厂按类型分发,支持动态注册与全局 reset(测试用) +- 文本系共享 difflib 行级基底,json / csv / xml 为其结构化特化;h5 面向科学数据集;binary 流式分块 + LCS 相似度 +- 统一返回 `ComparisonResult`(identical / differences / error / error_stats / command_output / channels),支持 text / json / html 渲染 +- 数据泳道插件可通过 `ChannelData.extra_stats` 附带自定义误差指标(自由 dict 并入通道 stats);但 verdict 仍由框架容差持有——需要自定 verdict 的走自主泳道 + +### 6.2 通道协议(数据泳道) + +- 通道判定:`identical = all(channels.passed)`;差异 position 带通道前缀(`channel S33`);`error_stats` 按通道名嵌套 +- 内置 `script_extract` 类型:子进程执行用户脚本(零改动接入),脚本 stdout 输出约定 JSON: + +```json +{"channels": {"S11": {"expected": [...], "actual": [...], "extra_stats": {...}}}} +``` + + 非零退出、超时、畸形 JSON 一律判为比较 error(绝不静默通过);`actual`/`baseline` 按惯例以 baseline 在前的顺序追加为尾部 argv 槽位(可缺省) +- 结果粒度:一条 compareSpec = 一条断言;`ComparisonResult.channels` 携带通道级子结果,报告逐通道展示 pass/fail + stats;JSON 输出中通道 differences 按 per-channel 配额截断 + +### 6.3 工厂与插件发现 + +- `file_type` 取值:`text` / `json` / `csv` / `xml` / `h5` / `binary` / `script` / `script_extract`;工厂按类型分发,支持动态注册与全局 reset(测试用) - 插件发现四处来源:内置 `*_comparator.py` 自动发现、`workspace/comparators/` 自动扫描、`--plugin-dir` CLI 参数、`CLITEST_PLUGIN_DIRS` 环境变量(供进程模式 worker 使用) -- 命名约定:`*_comparator.py` + `*Comparator` 类名 +- 命名约定:`*_comparator.py` + `*Comparator` 类名;可用类属性 `comparator_type` 显式指定类型名(如 `script_extract`);抽象基类自动跳过注册 +- compareSpec 透传机制:`actual`/`baseline`/`type` 之外的键全部转发给比较器构造函数(kwargs),`inputs`/`channels`/`default_channel` 亦走此通道 -### 6.3 script 比较协议(对外契约) +### 6.4 script 比较协议(自主泳道对外契约) -- 子进程方式执行 `