From 418bd9f190174da9f82ac55583c65468b832f496 Mon Sep 17 00:00:00 2001 From: sida Date: Sun, 26 Jul 2026 17:19:44 +0800 Subject: [PATCH] feat: verify Playwright locator repairs by rerun --- CHANGELOG.md | 9 + README.md | 93 +++---- README.zh-CN.md | 29 +- docs/VERIFIED_HEAL.md | 51 ++++ failure_doctor/cli.py | 78 +++++- failure_doctor/heal.py | 340 +++++++++++++++++++++++ failure_doctor/lite.py | 1 + tests/test_batch_diagnosis_fleet_mode.py | 2 +- tests/test_lite_experience.py | 2 +- tests/test_open_source_entry.py | 10 +- tests/test_public_release_cleanup.py | 8 +- tests/test_release_alignment_pack.py | 14 +- tests/test_release_trust_pack.py | 18 +- tests/test_verified_heal_loop.py | 236 ++++++++++++++++ 14 files changed, 805 insertions(+), 86 deletions(-) create mode 100644 docs/VERIFIED_HEAL.md create mode 100644 failure_doctor/heal.py create mode 100644 tests/test_verified_heal_loop.py diff --git a/CHANGELOG.md b/CHANGELOG.md index b3d8e98..2d01f7e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,15 @@ > Current package stable line: v6.3.0. > v4.2.0 remains the previous plugin SDK stable release, v4.1.0 remains the previous enterprise governance stable release, and v2.4.1 remains the previous P95 stable release. Ecosystem maturity is tracked separately from P98 controlled maturity. +## Unreleased + +- Added `failure-doctor heal`, a conservative Playwright locator repair loop that joins diagnosis, one exact source edit, a real rerun, and before/after evidence. +- Added automatic source backup and rollback when the verification command fails. +- Added explicit `verified`, `failed_rolled_back`, `failed_patch_retained`, and `manual_restore_required` states instead of treating an edited file as a successful repair. +- Blocked unsupported failure types, ambiguous matches, path escapes, symlink targets, and dirty Git targets unless the caller explicitly opts in. +- Refocused the default CLI and README first screen on diagnosis, verified repair, the offline benchmark, and the local console; specialist tracks remain available under `advanced`. +- Added unit and CLI coverage for successful reruns, failed reruns, automatic rollback, ambiguity blocking, and diagnosis eligibility. + ## v6.3.0 - Added a simplified default CLI with `diagnose`, `bench`, `console`, `doctor`, and an explicit `advanced` entry for the full legacy command set. diff --git a/README.md b/README.md index f0b829c..5f5ee9d 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# Agent Failure Doctor +# Agent Failure Doctor [中文文档](README.zh-CN.md) @@ -6,30 +6,43 @@ ![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg) ![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg) -Local-first failure diagnosis lifecycle tool for AI browser automation, -Playwright, crawler, RPA, and business automation failures. +Diagnose Playwright failures locally, apply one scoped locator repair, and keep +the change only when the same real test command passes. -**Input:** trace.zip / error.log / console.txt / network.json / probe_report.json / screenshot metadata / user_description.txt / visual_run / OCR or document evidence. +The public core deliberately does two things: -**Output:** diagnosis, evidence, next action, repair suggestions, GitHub issue draft, Codex fix prompt. +1. Turn a trace, log, screenshot, or failed-run directory into an evidence report. +2. Repair an explicitly supplied stale locator, rerun the real test, and emit a + truthful `verified`, `failed_rolled_back`, or `manual_restore_required` result. ```powershell git clone https://github.com/tobybgy-lsd/web-agent-runtime-bench.git cd web-agent-runtime-bench python -m pip install agent-failure-doctor failure-doctor diagnose .\examples\failed_runs\proxy_network_error --out .\report -failure-doctor plan .\report --out .\fix_plan +failure-doctor bench ``` -**Lifecycle commands:** `diagnose` / `plan` / `verify` / `run`. -**Classic lifecycle:** diagnose -> plan -> AI handoff / patch proposal -> verify -> sanitize/share. -**Share/adapt commands:** `sanitize` / `adapt`. -**Bootstrap command:** `failure-doctor agent-bootstrap`. -**Patch command:** `failure-doctor propose-patch`. -**Fleet command:** `failure-doctor batch`. -Compatibility track: Agent Failure Doctor v4.2.0 Plugin SDK & Adapter Ecosystem Pack. +## Verified locator repair + +```powershell +failure-doctor heal .\test-results\failed-case ` + --project . ` + --target tests\checkout.spec.ts ` + --old-locator "button.old-submit" ` + --new-locator "button[data-testid=submit]" ` + --out .\heal-report ` + --test-command npx playwright test tests\checkout.spec.ts +``` + +The command writes the diagnosis, exact diff, source backup, sanitized command +logs, exit code, and verification report. It blocks ambiguous or unsupported +repairs. A non-zero rerun restores the original file by default. It never marks +a proposal as repaired merely because a file was edited. See +[Verified repair contract](docs/VERIFIED_HEAL.md). Current milestone: Agent Failure Doctor v6.3 Lite UX & Bench Core Release. +Unreleased focus: verified Playwright locator repair with automatic rollback. Current stable line: v6.3.0. Previous stable line: Agent Failure Doctor v6.2 Local RPA Ops & Browser Backend Release. Earlier stable line: Agent Failure Doctor v6.1 Composite Diagnosis Runtime Triage Release. @@ -47,13 +60,14 @@ Install only the local diagnosis and benchmark core: python -m pip install agent-failure-doctor failure-doctor doctor failure-doctor diagnose .\examples\failed_runs\proxy_network_error --out .\report +failure-doctor heal --help failure-doctor bench failure-doctor console ``` -The default CLI promotes only five entries: `diagnose`, `bench`, `console`, -`doctor`, and `advanced`. The existing full command set remains available under -`failure-doctor advanced` and through its original command names. +The default CLI promotes only six entries: `diagnose`, `heal`, `bench`, +`console`, `doctor`, and `advanced`. The existing full command set remains +available under `failure-doctor advanced` and through its original command names. Install optional capabilities only where they run: @@ -64,49 +78,18 @@ python -m pip install "agent-failure-doctor[ocr-paddle]" python -m pip install "agent-failure-doctor[enterprise]" ``` -Focused console modes are available with `--mode lite`, `diagnose`, `bench`, -or `full`. The built-in `lite-core` benchmark is offline and uses packaged, -sanitized failure artifacts; it does not access real commerce or ERP systems. -Challenge/CAPTCHA cases are detection and manual-handoff tests, not bypasses. - -Classic quickstart: `failure-doctor diagnose .\examples\failed_runs\proxy_network_error --out .\report`; `failure-doctor plan .\report --out .\fix_plan`; `failure-doctor propose-patch --repo . --report .\report --out .\patch_plan`; `failure-doctor agent-bootstrap --target all --project .`. - -Earlier stable line: Agent Failure Doctor v4.2.0 Plugin SDK & Adapter Ecosystem Pack. -Previous P95 stable: v2.4.1. +The built-in `lite-core` benchmark is offline and uses packaged, sanitized +failure artifacts. Challenge/CAPTCHA cases are detection and manual-handoff +tests, not bypasses. **Lifecycle commands:** `diagnose` / `plan` / `verify` / `run`. - -**Classic lifecycle:** diagnose -> plan -> AI handoff / patch proposal -> verify -> sanitize/share. - -**Key commands:** `failure-doctor propose-patch`; `failure-doctor batch`; `sanitize` / `adapt`. - -Optional v6.0 output: sanitized public cases, benchmark reports, adapter reports, Android APK UI evidence reports, Android Pro hardening reports, Android Ops reports, Android authoring reports, Android pilot reports, Android deep diagnostic bundles, Android playbooks, Android device-lab reports, mobile-stability reports, deployment health reports, stability reports, plugin validation reports, evidence-bound reasoning, local web console, and CI/CD gate. - -- Current milestone: Agent Failure Doctor v6.3 Lite UX & Bench Core Release -- Current stable line: v6.3.0 -- Previous stable line: Agent Failure Doctor v6.2.0 Local RPA Ops & Browser Backend Release -- Previous stable line: Agent Failure Doctor v6.1.0 Composite Diagnosis Runtime Triage Release -- Previous stable line: Agent Failure Doctor v6.0.0 Mobile Automation Stable Standardization Release -- Previous stable line: Agent Failure Doctor v5.3.0 Android Real Device Farm & Business Workflow Operations Pack -- Previous stable line: Agent Failure Doctor v5.2.0 Android APK Production Hardening & Workflow Template Pack -- Previous stable line: Agent Failure Doctor v5.1.0 Android APK UI Automation Adapter Pack -- Previous stable line: Agent Failure Doctor v5.0.0 Stable API / Schema / Plugin ABI Standardization Release -- Earlier stable line: Agent Failure Doctor v4.3.0 Real User Case Program & Public Benchmark Pack -- Previous stable line: Agent Failure Doctor v4.2.0 Plugin SDK & Adapter Ecosystem Pack -- Earlier stable line: Agent Failure Doctor v4.1.0 Enterprise Governance & Role-Based Console Pack -- Earlier stable line: Agent Failure Doctor v4.0.0 Hybrid Evidence Reasoning Pack -- Earlier stable line: Agent Failure Doctor v3.9.0 Local Failure Knowledge Base Pack (v3.9 Local Failure Knowledge Base Pack) -- Previous P95 stable line: Agent Failure Doctor v2.4.1 P95 Alignment & Missing Tracks Pack - **Classic lifecycle:** diagnose -> plan -> AI handoff / patch proposal -> verify -> sanitize/share. +Compatibility track: Agent Failure Doctor v4.2.0 Plugin SDK & Adapter Ecosystem Pack. +Previous P95 stable: v2.4.1. -**Patch command:** `failure-doctor propose-patch`. - -**Share/adapt commands:** `sanitize` / `adapt`. - -**Fleet command:** `failure-doctor batch`. - -**Core commands:** `diagnose` / `plan` / `verify` / `run`; `android`; `android-pro`; `android-ops`; `android-author`; `android-pilot`; `android-dx`; `android-playbook`; `android-real-pilot`; `android-lab`; `mobile-stability`; `case`; `issue-pack`; `benchmark`; `adapter`; `deploy`; `stability`; `plugin`; `reason` / `root-cause` / `causal-chain`; `agent-bootstrap`; `sanitize` / `adapt`; `ocr-evidence`; `visual-runtime`; `regulated-eval`; `full-chain-eval`; `console`; `ci`; `kb`; `failure-doctor propose-patch`; `failure-doctor batch`. +Specialist Android, OCR, enterprise, adapter, and LAN Worker capabilities are +optional compatibility tracks. They are intentionally absent from the default +help; use `failure-doctor advanced` when you actually operate those components. ### Unreleased LAN Worker v1 diff --git a/README.zh-CN.md b/README.zh-CN.md index fa9c774..375552f 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -1,7 +1,28 @@ -# Agent Failure Doctor 中文文档 +# Agent Failure Doctor 中文文档 [English documentation](README.md) +## 这个项目现在只突出两件事 + +1. 把 Playwright 的 trace、日志、截图或失败目录整理成有证据的诊断报告。 +2. 对一个明确的失效定位器做最小修改,真实重跑同一条测试命令;只有重跑通过才保留修改。 + +```powershell +failure-doctor diagnose .\test-results\failed-case --out .\report + +failure-doctor heal .\test-results\failed-case ` + --project . ` + --target tests\checkout.spec.ts ` + --old-locator "button.old-submit" ` + --new-locator "button[data-testid=submit]" ` + --out .\heal-report ` + --test-command npx playwright test tests\checkout.spec.ts +``` + +`heal` 会保存原文件、补丁、命令日志和退出码。重跑失败默认自动回滚;定位器 +出现在多个文件、目标文件已有未提交修改、错误类型不属于定位器问题时会直接 +阻断,不会显示假成功。详见 [修复验证契约](docs/VERIFIED_HEAL.md)。 + ## v6.3 快速入口 只安装本地诊断与基准核心: @@ -9,12 +30,14 @@ ```powershell python -m pip install agent-failure-doctor failure-doctor doctor +failure-doctor diagnose .\examples\failed_runs\proxy_network_error --out .\report +failure-doctor heal --help failure-doctor bench failure-doctor console ``` -默认 CLI 只突出 `diagnose`、`bench`、`console`、`doctor` 和 `advanced` -五个入口。原有完整命令没有删除,可通过 `failure-doctor advanced` 查看, +默认 CLI 只突出 `diagnose`、`heal`、`bench`、`console`、`doctor` 和 `advanced` +六个入口。原有完整命令没有删除,可通过 `failure-doctor advanced` 查看, 也仍可直接调用原命令。 Web、Android、PaddleOCR 和企业集成按运行机器选装: diff --git a/docs/VERIFIED_HEAL.md b/docs/VERIFIED_HEAL.md new file mode 100644 index 0000000..9481bd4 --- /dev/null +++ b/docs/VERIFIED_HEAL.md @@ -0,0 +1,51 @@ +# Verified Playwright Locator Repair + +`failure-doctor heal` joins diagnosis, one exact source edit, a real rerun, and +before/after evidence into one conservative command. + +## Contract + +The command will modify source only when all of these are true: + +- The failure report is `selector_drift`, `playwright_strict_mode_violation`, or + `selector_syntax_error`. +- The target is a supported Python or JavaScript/TypeScript source file inside + the authorized project. +- The old locator occurs exactly once in the selected file. +- The target is not a symlink. +- A Git target has no pre-existing uncommitted change unless the caller passes + `--allow-dirty-target` explicitly. +- A real verification command is supplied with `--test-command`. + +## Result states + +| Status | Meaning | +| --- | --- | +| `planned` | Dry run only; no source was changed. | +| `verified` | The exact patch was applied, the real command exited `0`, and the command did not alter the patched source. | +| `failed_rolled_back` | The real command failed and the original source was restored. | +| `failed_patch_retained` | The real command failed and `--keep-failed-patch` explicitly retained the edit. | +| `manual_restore_required` | The test command changed the target source, so automatic rollback was blocked to avoid overwriting concurrent work. | + +Only `verified` means the repair passed its execution contract. A generated +patch, a diagnosis, or an edited file is never reported as a successful repair. + +## Example + +```powershell +failure-doctor heal .\test-results\checkout-failure ` + --project . ` + --target tests\checkout.spec.ts ` + --old-locator "button.old-submit" ` + --new-locator "button[data-testid=submit]" ` + --out .\outputs\checkout-heal ` + --test-command npx playwright test tests\checkout.spec.ts +``` + +Review `heal_report.json`, `heal_report.md`, `locator.patch`, `backup/`, and +`runs/after-rerun/`. Run with `--dry-run` first when the replacement needs human +review. + +This command does not solve challenges, bypass access controls, change browser +fingerprints, extract credentials, or infer a replacement selector from private +production pages. diff --git a/failure_doctor/cli.py b/failure_doctor/cli.py index 35e0c72..b392075 100644 --- a/failure_doctor/cli.py +++ b/failure_doctor/cli.py @@ -49,6 +49,7 @@ from failure_doctor.deploy.cli import add_deploy_parser, handle_deploy from failure_doctor.stability.cli import add_stability_parser, handle_stability from failure_doctor.lite import capability_report, default_benchmark_output, print_capability_report +from failure_doctor.heal import HealBlocked, execute_verified_locator_repair from failure_doctor.visual_runtime.adapter import adapt_visual_artifacts from failure_doctor.visual_runtime.compare import compare_visual_runs from failure_doctor.visual_runtime.loader import load_visual_run, validate_visual_run @@ -90,6 +91,8 @@ def main(argv: list[str] | None = None) -> int: args = parser.parse_args(raw_args) if args.command == "diagnose": return diagnose_inputs(args) + if args.command == "heal": + return heal_inputs(args) if args.command == "plan": return plan_from_report(args) if args.command == "verify": @@ -216,6 +219,33 @@ def build_parser() -> argparse.ArgumentParser: diagnose.add_argument("--plugin", default=None, help="Optional enabled diagnosis-rule plugin id") diagnose.add_argument("--plugins", default=".failure-doctor-plugins", help="Plugin workspace") diagnose.add_argument("--adapter", default=None, choices=["android-apk"], help="Optional specialized evidence adapter") + heal = sub.add_parser( + "heal", + help="Apply one scoped Playwright locator repair and prove it by rerunning a real command", + ) + heal.add_argument("input", help="Failure artifact or existing diagnosis report directory") + heal.add_argument("--project", required=True, help="Authorized Playwright project directory") + heal.add_argument("--old-locator", required=True, help="Exact stale locator text in source") + heal.add_argument("--new-locator", required=True, help="Exact replacement locator text") + heal.add_argument("--target", default=None, help="Optional source file relative to project") + heal.add_argument("--out", required=True, help="Output directory for backup and verification evidence") + heal.add_argument("--dry-run", action="store_true", help="Write the proposed patch without changing source") + heal.add_argument( + "--allow-dirty-target", + action="store_true", + help="Allow modifying a source file that already has uncommitted changes", + ) + heal.add_argument( + "--keep-failed-patch", + action="store_true", + help="Keep the patch when verification fails instead of restoring the backup", + ) + heal.add_argument( + "--test-command", + nargs=argparse.REMAINDER, + required=True, + help="Real command and its arguments, for example --test-command npx playwright test tests/cart.spec.ts", + ) plan = sub.add_parser("plan", help="Generate a fix plan from a diagnosis report directory") plan.add_argument("report", help="Path to a report directory containing diagnosis.json") plan.add_argument("--out", required=True, help="Output fix plan directory") @@ -477,12 +507,13 @@ def build_parser() -> argparse.ArgumentParser: def print_lite_help() -> None: print( - """usage: failure-doctor {diagnose,bench,console,doctor,advanced} ... + """usage: failure-doctor {diagnose,heal,bench,console,doctor,advanced} ... Local-first Web Agent benchmark and failure diagnosis. quick commands: diagnose Diagnose a trace, log, screenshot, or failure directory + heal Patch one Playwright locator, rerun, and keep it only when verified bench Run the built-in evidence-based Web Agent benchmark console Open the local Web console doctor Check optional capability profiles without network access @@ -493,6 +524,51 @@ def print_lite_help() -> None: ) +def heal_inputs(args: argparse.Namespace) -> int: + out_dir = Path(args.out) + input_path = Path(args.input) + before_report = out_dir / "before_report" + try: + has_existing_diagnosis = input_path.is_dir() and (input_path / "diagnosis.json").is_file() + if has_existing_diagnosis: + diagnosis = _load_report_diagnosis(input_path) + else: + diagnose_args = argparse.Namespace( + input=str(input_path), + out=str(before_report), + run_id=None, + kb=None, + hybrid_reasoning=False, + reasoner="mock_reasoner", + plugin=None, + plugins=".failure-doctor-plugins", + adapter=None, + ) + if diagnose_inputs(diagnose_args) != 0: + return 2 + diagnosis = _load_report_diagnosis(before_report) + report = execute_verified_locator_repair( + project=Path(args.project), + diagnosis=diagnosis, + old_locator=str(args.old_locator), + new_locator=str(args.new_locator), + target=Path(args.target) if args.target else None, + test_command=list(args.test_command), + out_dir=out_dir, + dry_run=bool(args.dry_run), + allow_dirty_target=bool(args.allow_dirty_target), + keep_failed_patch=bool(args.keep_failed_patch), + ) + except (HealBlocked, FileNotFoundError, OSError, ValueError) as exc: + print(f"Heal blocked: {exc}") + return 2 + print("Failure Doctor Verified Repair") + print(f"Status: {report.get('status')}") + print(f"Verified: {str(bool(report.get('verified'))).lower()}") + print(f"Output: {out_dir}") + return 0 if report.get("verified") or report.get("status") == "planned" else 1 + + def diagnose_inputs(args: argparse.Namespace) -> int: _configure_stdio() input_path = Path(args.input) diff --git a/failure_doctor/heal.py b/failure_doctor/heal.py new file mode 100644 index 0000000..0cf77b7 --- /dev/null +++ b/failure_doctor/heal.py @@ -0,0 +1,340 @@ +from __future__ import annotations + +import difflib +import hashlib +import json +import os +import subprocess +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Mapping, Sequence + +from failure_doctor.run_capture import capture_run + + +ALLOWED_FAILURE_TYPES = { + "selector_drift", + "playwright_strict_mode_violation", + "selector_syntax_error", +} +SOURCE_SUFFIXES = {".py", ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs"} +SKIP_DIRS = { + ".git", + ".hg", + ".svn", + ".venv", + "venv", + "node_modules", + "dist", + "build", + "coverage", + "test-results", + "playwright-report", +} + + +class HealBlocked(ValueError): + """Raised when the repair contract is unsafe or ambiguous.""" + + +def execute_verified_locator_repair( + *, + project: Path, + diagnosis: Mapping[str, Any], + old_locator: str, + new_locator: str, + test_command: Sequence[str], + out_dir: Path, + target: Path | None = None, + dry_run: bool = False, + allow_dirty_target: bool = False, + keep_failed_patch: bool = False, +) -> dict[str, Any]: + project = project.expanduser().resolve() + out_dir = out_dir.expanduser().resolve() + command = _normalized_command(test_command) + failure_type = str( + diagnosis.get("failure_type") + or diagnosis.get("technical_category") + or "unknown" + ) + if not project.is_dir(): + raise HealBlocked(f"project directory not found: {project}") + if failure_type not in ALLOWED_FAILURE_TYPES: + raise HealBlocked( + f"diagnosis {failure_type!r} is not eligible for automatic locator repair" + ) + _validate_locator_pair(old_locator, new_locator) + source = _resolve_single_source(project, old_locator, target) + if not allow_dirty_target and _git_target_is_dirty(project, source): + raise HealBlocked( + "target has uncommitted changes; commit/stash it or pass --allow-dirty-target" + ) + + original = source.read_text(encoding="utf-8") + occurrences = original.count(old_locator) + if occurrences != 1: + raise HealBlocked( + f"expected exactly one locator occurrence in {source}, found {occurrences}" + ) + patched = original.replace(old_locator, new_locator, 1) + relative_source = source.relative_to(project) + backup = out_dir / "backup" / relative_source + diff_path = out_dir / "locator.patch" + report_path = out_dir / "heal_report.json" + out_dir.mkdir(parents=True, exist_ok=True) + backup.parent.mkdir(parents=True, exist_ok=True) + backup.write_text(original, encoding="utf-8") + diff_path.write_text( + "".join( + difflib.unified_diff( + original.splitlines(keepends=True), + patched.splitlines(keepends=True), + fromfile=f"a/{relative_source.as_posix()}", + tofile=f"b/{relative_source.as_posix()}", + ) + ), + encoding="utf-8", + ) + + report: dict[str, Any] = { + "schema_version": "failure-doctor-heal/v1", + "status": "planned" if dry_run else "running", + "verified": False, + "failure_type": failure_type, + "project": str(project), + "target": relative_source.as_posix(), + "change": { + "kind": "exact_locator_replacement", + "old_locator": old_locator, + "new_locator": new_locator, + "replacement_count": 1, + "before_sha256": _sha256_text(original), + "proposed_sha256": _sha256_text(patched), + }, + "safety": { + "single_file": True, + "single_occurrence": True, + "backup": str(backup), + "automatic_rollback_on_failure": not keep_failed_patch, + "dirty_target_allowed": allow_dirty_target, + }, + "test": { + "configured": True, + "exit_code": None, + "evidence_dir": None, + }, + "created_at": datetime.now(timezone.utc).isoformat(), + } + _write_json(report_path, report) + if dry_run: + return report + + _atomic_write(source, patched) + try: + run_result = capture_run( + command, + workspace=out_dir, + run_id="after-rerun", + cwd=project, + ) + except OSError as exc: + current = source.read_text(encoding="utf-8") + source_unchanged_by_test = _sha256_text(current) == _sha256_text(patched) + report["test"] = { + "configured": True, + "exit_code": None, + "evidence_dir": None, + "source_unchanged_by_test": source_unchanged_by_test, + "launch_error": type(exc).__name__, + } + if not source_unchanged_by_test: + report["status"] = "manual_restore_required" + report["message"] = ( + "The target changed while the verification command was starting. " + "Restore from the recorded backup manually." + ) + elif keep_failed_patch: + report["status"] = "failed_patch_retained" + report["message"] = ( + "The verification command could not start and the patch was retained by request." + ) + else: + _atomic_write(source, original) + report["status"] = "failed_rolled_back" + report["message"] = ( + "The verification command could not start; the original source was restored." + ) + report["completed_at"] = datetime.now(timezone.utc).isoformat() + _write_json(report_path, report) + _write_markdown(out_dir / "heal_report.md", report) + return report + current = source.read_text(encoding="utf-8") + source_unchanged_by_test = _sha256_text(current) == _sha256_text(patched) + exit_code = int(run_result["exit_code"]) + report["test"] = { + "configured": True, + "exit_code": exit_code, + "evidence_dir": run_result["run_dir"], + "source_unchanged_by_test": source_unchanged_by_test, + } + + if exit_code == 0 and source_unchanged_by_test: + report["status"] = "verified" + report["verified"] = True + report["verification_contract"] = { + "diagnosis_supported": True, + "patch_applied": True, + "test_command_passed": True, + "source_unchanged_by_test": True, + } + elif not source_unchanged_by_test: + report["status"] = "manual_restore_required" + report["verification_contract"] = { + "diagnosis_supported": True, + "patch_applied": True, + "test_command_passed": exit_code == 0, + "source_unchanged_by_test": False, + } + report["message"] = ( + "The test command changed the target source. Automatic rollback was skipped " + "to avoid overwriting concurrent work; restore from the recorded backup manually." + ) + elif keep_failed_patch: + report["status"] = "failed_patch_retained" + report["message"] = "The verification command failed and the patch was retained by request." + else: + _atomic_write(source, original) + report["status"] = "failed_rolled_back" + report["message"] = "The verification command failed; the original source was restored." + + report["completed_at"] = datetime.now(timezone.utc).isoformat() + _write_json(report_path, report) + _write_markdown(out_dir / "heal_report.md", report) + return report + + +def _normalized_command(command: Sequence[str]) -> list[str]: + normalized = [str(part) for part in command] + if normalized and normalized[0] == "--": + normalized = normalized[1:] + if not normalized: + raise HealBlocked("a real verification command is required after --test-command") + return normalized + + +def _validate_locator_pair(old_locator: str, new_locator: str) -> None: + if not old_locator or not new_locator: + raise HealBlocked("old and new locators must be non-empty") + if old_locator == new_locator: + raise HealBlocked("old and new locators are identical") + if "\x00" in old_locator or "\x00" in new_locator: + raise HealBlocked("locators may not contain NUL bytes") + if len(old_locator) > 500 or len(new_locator) > 500: + raise HealBlocked("locator values are unexpectedly long") + + +def _resolve_single_source(project: Path, old_locator: str, target: Path | None) -> Path: + if target is not None: + requested = target if target.is_absolute() else project / target + _reject_symlink_path(requested, project) + candidate = requested.resolve() + _require_within(candidate, project) + if not candidate.is_file(): + raise HealBlocked(f"target source not found: {candidate}") + if candidate.suffix.lower() not in SOURCE_SUFFIXES: + raise HealBlocked(f"unsupported source type: {candidate.suffix}") + return candidate + + matches: list[Path] = [] + for path in project.rglob("*"): + if not path.is_file() or path.suffix.lower() not in SOURCE_SUFFIXES: + continue + try: + _reject_symlink_path(path, project) + except HealBlocked: + continue + relative = path.relative_to(project) + if any(part in SKIP_DIRS for part in relative.parts[:-1]): + continue + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeError): + continue + if old_locator in text: + matches.append(path) + if len(matches) > 1: + break + if not matches: + raise HealBlocked("old locator was not found in supported project source files") + if len(matches) > 1: + raise HealBlocked("old locator is ambiguous across multiple source files; pass --target") + return matches[0] + + +def _reject_symlink_path(path: Path, root: Path) -> None: + absolute = path.absolute() + _require_within(absolute, root) + relative = absolute.relative_to(root) + current = root + for part in relative.parts: + current = current / part + if current.is_symlink(): + raise HealBlocked("symbolic-link targets are not modified") + + +def _require_within(path: Path, root: Path) -> None: + try: + path.relative_to(root) + except ValueError as exc: + raise HealBlocked("target must stay inside the authorized project directory") from exc + + +def _git_target_is_dirty(project: Path, source: Path) -> bool: + try: + result = subprocess.run( + ["git", "status", "--porcelain", "--", str(source)], + cwd=project, + text=True, + encoding="utf-8", + errors="replace", + capture_output=True, + check=False, + ) + except OSError: + return False + return result.returncode == 0 and bool(result.stdout.strip()) + + +def _atomic_write(path: Path, text: str) -> None: + temporary = path.with_name(f".{path.name}.failure-doctor-{os.getpid()}.tmp") + temporary.write_text(text, encoding="utf-8") + temporary.replace(path) + + +def _sha256_text(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _write_json(path: Path, payload: Mapping[str, Any]) -> None: + path.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") + + +def _write_markdown(path: Path, report: Mapping[str, Any]) -> None: + test = report.get("test", {}) + lines = [ + "# Verified Playwright Repair", + "", + f"- Status: `{report.get('status')}`", + f"- Verified: `{str(bool(report.get('verified'))).lower()}`", + f"- Failure type: `{report.get('failure_type')}`", + f"- Target: `{report.get('target')}`", + f"- Verification exit code: `{test.get('exit_code')}`", + f"- Evidence: `{test.get('evidence_dir')}`", + "", + "A successful command exit is necessary but not sufficient: the target must also remain " + "equal to the proposed patch after the test command finishes.", + ] + if report.get("message"): + lines.extend(["", f"> {report.get('message')}"]) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") diff --git a/failure_doctor/lite.py b/failure_doctor/lite.py index 057c54c..f145b15 100644 --- a/failure_doctor/lite.py +++ b/failure_doctor/lite.py @@ -11,6 +11,7 @@ LITE_COMMANDS = ( ("diagnose", "Diagnose a trace, log, screenshot, or failure directory"), + ("heal", "Patch one Playwright locator and keep it only after a passing rerun"), ("bench", "Run the built-in evidence-based Web Agent benchmark"), ("console", "Open the local Web console"), ("doctor", "Check which optional capability profiles are ready"), diff --git a/tests/test_batch_diagnosis_fleet_mode.py b/tests/test_batch_diagnosis_fleet_mode.py index d062d43..b2a7928 100644 --- a/tests/test_batch_diagnosis_fleet_mode.py +++ b/tests/test_batch_diagnosis_fleet_mode.py @@ -77,7 +77,7 @@ def test_docs_and_version_expose_batch_mode_under_current_release(self): changelog = (ROOT / "CHANGELOG.md").read_text(encoding="utf-8") pyproject = (ROOT / "pyproject.toml").read_text(encoding="utf-8") self.assertIn('version = "6.3.0"', pyproject) - self.assertIn("v3.9 Local Failure Knowledge Base Pack", readme) + self.assertIn("v3.9.0 Local Failure Knowledge Base Pack", readme) self.assertIn("v2.4.1 P95 Alignment & Missing Tracks Pack", readme) self.assertIn("Batch Diagnosis / Fleet Mode", readme) self.assertIn("failure-doctor batch", readme) diff --git a/tests/test_lite_experience.py b/tests/test_lite_experience.py index 98ce0c0..d826539 100644 --- a/tests/test_lite_experience.py +++ b/tests/test_lite_experience.py @@ -27,7 +27,7 @@ def test_default_help_only_promotes_quick_commands(self) -> None: ) self.assertEqual(result.returncode, 0, result.stderr) - self.assertIn("{diagnose,bench,console,doctor,advanced}", result.stdout) + self.assertIn("{diagnose,heal,bench,console,doctor,advanced}", result.stdout) self.assertNotIn("android-real-pilot", result.stdout) def test_advanced_help_keeps_legacy_commands(self) -> None: diff --git a/tests/test_open_source_entry.py b/tests/test_open_source_entry.py index cb65d96..09950d1 100644 --- a/tests/test_open_source_entry.py +++ b/tests/test_open_source_entry.py @@ -13,14 +13,15 @@ def test_readme_top_is_plain_product_entry(self): for phrase in ( "# Agent Failure Doctor", - "Local-first failure diagnosis lifecycle tool for AI browser automation, Playwright, crawler, RPA, and business automation failures.", - "trace.zip / error.log / console.txt / network.json / probe_report.json / screenshot metadata / user_description.txt", - "diagnosis, evidence, next action, repair suggestions, GitHub issue draft, Codex fix prompt.", + "Diagnose Playwright failures locally, apply one scoped locator repair", + "Turn a trace, log, screenshot, or failed-run directory into an evidence report.", + "truthful `verified`, `failed_rolled_back`, or `manual_restore_required` result.", "git clone https://github.com/tobybgy-lsd/web-agent-runtime-bench.git", "cd web-agent-runtime-bench", "python -m pip install agent-failure-doctor", "failure-doctor diagnose .\\examples\\failed_runs\\proxy_network_error --out .\\report", - "failure-doctor plan .\\report --out .\\fix_plan", + "failure-doctor heal .\\test-results\\failed-case", + "--test-command npx playwright test tests\\checkout.spec.ts", ): self.assertIn(" ".join(phrase.split()), normalized_opening) @@ -77,4 +78,3 @@ def test_community_post_drafts_request_failure_samples_not_stars(self): if __name__ == "__main__": unittest.main() - diff --git a/tests/test_public_release_cleanup.py b/tests/test_public_release_cleanup.py index dda4bfd..025fcb5 100644 --- a/tests/test_public_release_cleanup.py +++ b/tests/test_public_release_cleanup.py @@ -101,17 +101,17 @@ def test_root_directory_has_no_internal_report_files(self): def test_readme_is_english_first_with_badges_and_links(self): readme = (ROOT / "README.md").read_text(encoding="utf-8") - top = readme[:2200] + top = readme[:2400] for phrase in ( "[中文文档](README.zh-CN.md)", "![CI]", "![License: MIT]", "![Python 3.10+]", - "Local-first failure diagnosis", + "Diagnose Playwright failures locally", "python -m pip install agent-failure-doctor", "failure-doctor diagnose .\\examples\\failed_runs\\proxy_network_error --out .\\report", - "Plugin SDK & Adapter Ecosystem Pack", - "GitHub issue draft", + "verified Playwright locator repair", + "never marks", "See [validation/dashboard.md](validation/dashboard.md)", ): self.assertIn(phrase, top) diff --git a/tests/test_release_alignment_pack.py b/tests/test_release_alignment_pack.py index 733fa9a..eb7ba63 100644 --- a/tests/test_release_alignment_pack.py +++ b/tests/test_release_alignment_pack.py @@ -12,18 +12,18 @@ def test_readme_first_screen_shows_current_lifecycle_commands(self): opening = readme[:2200] self.assertIn("Current milestone: Agent Failure Doctor v6.3 Lite UX & Bench Core Release", opening) + self.assertIn("Unreleased focus: verified Playwright locator repair", opening) self.assertIn("Earlier stable line: Agent Failure Doctor v3.9.0", opening) self.assertIn("P98 gate:", opening) self.assertNotIn("Current milestone: v0.8", opening) for phrase in ( "failure-doctor diagnose", - "`diagnose` / `plan` / `verify` / `run`", - "`sanitize` / `adapt`", - "failure-doctor agent-bootstrap", - "`failure-doctor propose-patch`", - "`failure-doctor batch`", - "diagnose -> plan -> AI handoff / patch proposal", - "-> verify -> sanitize/share", + "failure-doctor heal", + "--old-locator", + "--new-locator", + "--test-command", + "failed_rolled_back", + "never marks", ): self.assertIn(phrase, opening) diff --git a/tests/test_release_trust_pack.py b/tests/test_release_trust_pack.py index d15abc1..6788a9c 100644 --- a/tests/test_release_trust_pack.py +++ b/tests/test_release_trust_pack.py @@ -17,8 +17,8 @@ def test_versions_and_readme_validation_milestone_are_aligned(self): self.assertIn('version = "6.3.0"', pyproject) self.assertIn("## v2.1.0", changelog) self.assertIn("## v2.0.0", changelog) - self.assertIn("v3.9 Local Failure Knowledge Base Pack", readme) - self.assertIn("v2.4.1 P95 Alignment & Missing Tracks Pack", readme) + self.assertIn("v3.9.0 Local Failure Knowledge Base Pack", readme) + self.assertIn("Previous P95 stable: v2.4.1", readme) self.assertIn("Current package stable line: v6.3.0", changelog) self.assertIn("P98 master gate passed", readme) self.assertIn("v2.0 Auto Capture", readme) @@ -27,13 +27,13 @@ def test_versions_and_readme_validation_milestone_are_aligned(self): def test_readme_top_is_not_redundant_or_mojibake(self): readme = (ROOT / "README.md").read_text(encoding="utf-8") - top = readme[:1200] - self.assertIn("Local-first failure diagnosis", top) - self.assertIn("trace.zip / error.log / console.txt / network.json", top) - self.assertIn("screenshot metadata / user_description.txt", top) - self.assertIn("diagnosis, evidence, next action, repair suggestions", top) - self.assertIn("GitHub issue draft, Codex fix prompt.", top) - self.assertEqual(top.count("diagnosis, evidence, next action"), 1) + top = readme[:1700] + self.assertIn("Diagnose Playwright failures locally", top) + self.assertIn("failed-run directory into an evidence report", top) + self.assertIn("failure-doctor heal", top) + self.assertIn("--test-command", top) + self.assertIn("never marks", top) + self.assertEqual(top.count("## Verified locator repair"), 1) self.assertNotIn("mojibake marker", top) self.assertNotIn("another mojibake marker", top) diff --git a/tests/test_verified_heal_loop.py b/tests/test_verified_heal_loop.py new file mode 100644 index 0000000..df0a406 --- /dev/null +++ b/tests/test_verified_heal_loop.py @@ -0,0 +1,236 @@ +from __future__ import annotations + +import json +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +from failure_doctor.heal import HealBlocked, execute_verified_locator_repair + + +ROOT = Path(__file__).resolve().parents[1] +SELECTOR_DIAGNOSIS = { + "technical_category": "selector_drift", + "failure_type": "selector_drift", + "subtype": "missing_selector", +} + + +class VerifiedHealLoopTests(unittest.TestCase): + def test_verified_repair_keeps_patch_and_writes_real_rerun_evidence(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + out = root / "heal" + project.mkdir() + source = project / "checkout.spec.ts" + source.write_text('await page.locator("button.old").click();\n', encoding="utf-8") + check = project / "verify.py" + check.write_text( + "from pathlib import Path\n" + "text = Path('checkout.spec.ts').read_text(encoding='utf-8')\n" + "raise SystemExit(0 if 'button.new' in text else 9)\n", + encoding="utf-8", + ) + + report = execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + target=Path("checkout.spec.ts"), + test_command=[sys.executable, "verify.py"], + out_dir=out, + ) + + self.assertEqual(report["status"], "verified") + self.assertTrue(report["verified"]) + self.assertIn("button.new", source.read_text(encoding="utf-8")) + self.assertEqual((out / "runs" / "after-rerun" / "exit_code.txt").read_text().strip(), "0") + self.assertTrue((out / "backup" / "checkout.spec.ts").exists()) + self.assertTrue((out / "locator.patch").exists()) + + def test_failed_rerun_restores_original_source(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + project.mkdir() + source = project / "checkout.py" + original = 'page.locator("button.old").click()\n' + source.write_text(original, encoding="utf-8") + + report = execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + target=Path("checkout.py"), + test_command=[sys.executable, "-c", "raise SystemExit(7)"], + out_dir=root / "heal", + ) + + self.assertEqual(report["status"], "failed_rolled_back") + self.assertFalse(report["verified"]) + self.assertEqual(source.read_text(encoding="utf-8"), original) + self.assertEqual(report["test"]["exit_code"], 7) + + def test_ambiguous_source_is_blocked_before_any_edit(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + project = Path(tmp) + first = project / "one.ts" + second = project / "two.ts" + first.write_text("button.old\n", encoding="utf-8") + second.write_text("button.old\n", encoding="utf-8") + + with self.assertRaisesRegex(HealBlocked, "ambiguous"): + execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + test_command=[sys.executable, "-c", "raise SystemExit(0)"], + out_dir=project / "out", + ) + + self.assertEqual(first.read_text(encoding="utf-8"), "button.old\n") + self.assertEqual(second.read_text(encoding="utf-8"), "button.old\n") + + def test_missing_verification_executable_restores_original_source(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + project.mkdir() + source = project / "flow.ts" + original = 'const selector = "button.old";\n' + source.write_text(original, encoding="utf-8") + + report = execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + target=Path("flow.ts"), + test_command=["definitely-not-an-installed-command-4e2f"], + out_dir=root / "heal", + ) + + self.assertEqual(report["status"], "failed_rolled_back") + self.assertEqual(report["test"]["launch_error"], "FileNotFoundError") + self.assertEqual(source.read_text(encoding="utf-8"), original) + + def test_non_locator_diagnosis_is_not_auto_modified(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + project = Path(tmp) + (project / "test.ts").write_text("button.old\n", encoding="utf-8") + + with self.assertRaisesRegex(HealBlocked, "not eligible"): + execute_verified_locator_repair( + project=project, + diagnosis={"technical_category": "network_http_error"}, + old_locator="button.old", + new_locator="button.new", + test_command=[sys.executable, "-c", "raise SystemExit(0)"], + out_dir=project / "out", + ) + + def test_target_outside_project_is_blocked(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + project.mkdir() + outside = root / "outside.ts" + outside.write_text("button.old\n", encoding="utf-8") + + with self.assertRaisesRegex(HealBlocked, "inside the authorized project"): + execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + target=outside, + test_command=[sys.executable, "-c", "raise SystemExit(0)"], + out_dir=root / "heal", + ) + + @unittest.skipUnless(hasattr(os, "symlink"), "symbolic links are unsupported") + def test_symbolic_link_target_is_blocked(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + project.mkdir() + real_source = project / "real.ts" + link_source = project / "linked.ts" + real_source.write_text("button.old\n", encoding="utf-8") + try: + link_source.symlink_to(real_source) + except OSError as exc: + self.skipTest(f"symbolic-link creation is unavailable: {exc}") + + with self.assertRaisesRegex(HealBlocked, "symbolic-link"): + execute_verified_locator_repair( + project=project, + diagnosis=SELECTOR_DIAGNOSIS, + old_locator="button.old", + new_locator="button.new", + target=Path("linked.ts"), + test_command=[sys.executable, "-c", "raise SystemExit(0)"], + out_dir=root / "heal", + ) + + def test_cli_runs_diagnosis_patch_and_verification_as_one_command(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + project = root / "project" + failure = root / "failure" + out = root / "heal" + project.mkdir() + failure.mkdir() + (failure / "error.log").write_text( + "TimeoutError: locator.click: Timeout waiting for selector button.old\n", + encoding="utf-8", + ) + (project / "flow.py").write_text('LOCATOR = "button.old"\n', encoding="utf-8") + (project / "verify.py").write_text( + "from pathlib import Path\n" + "raise SystemExit(0 if 'button.new' in Path('flow.py').read_text() else 8)\n", + encoding="utf-8", + ) + + result = subprocess.run( + [ + sys.executable, + "-m", + "failure_doctor", + "heal", + str(failure), + "--project", + str(project), + "--target", + "flow.py", + "--old-locator", + "button.old", + "--new-locator", + "button.new", + "--out", + str(out), + "--test-command", + sys.executable, + "verify.py", + ], + cwd=ROOT, + text=True, + encoding="utf-8", + capture_output=True, + ) + + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + report = json.loads((out / "heal_report.json").read_text(encoding="utf-8")) + self.assertEqual(report["status"], "verified") + self.assertTrue((out / "before_report" / "diagnosis.json").exists()) + + +if __name__ == "__main__": + unittest.main()