diff --git a/docs/README.agents.md b/docs/README.agents.md index eac43b3ef..608834918 100644 --- a/docs/README.agents.md +++ b/docs/README.agents.md @@ -90,7 +90,7 @@ See [CONTRIBUTING.md](../CONTRIBUTING.md#adding-agents) for guidelines on how to | [Doublecheck](../agents/doublecheck.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdoublecheck.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdoublecheck.agent.md) | Interactive verification agent for AI-generated output. Runs a three-layer pipeline (self-audit, source verification, adversarial review) and produces structured reports with source links for human review. | | | [Droid](../agents/droid.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdroid.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdroid.agent.md) | Provides installation guidance, usage examples, and automation patterns for the Droid CLI, with emphasis on droid exec for CI/CD and non-interactive automation | | | [Drupal Expert](../agents/drupal-expert.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdrupal-expert.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdrupal-expert.agent.md) | Expert assistant for Drupal development, architecture, and best practices using PHP 8.3+ and modern Drupal patterns | | -| [Dynatrace Expert](../agents/dynatrace-expert.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdynatrace-expert.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdynatrace-expert.agent.md) | The Dynatrace Expert Agent integrates observability and security capabilities directly into GitHub workflows, enabling development teams to investigate incidents, validate deployments, triage errors, detect performance regressions, validate releases, and manage security vulnerabilities by autonomously analysing traces, logs, and Dynatrace findings. This enables targeted and precise remediation of identified issues directly within the repository. | [dynatrace](https://github.com/mcp/io.github.dynatrace-oss/Dynatrace-mcp)
[![Install MCP](https://img.shields.io/badge/Install-VS_Code-0098FF?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscode?name=dynatrace&config=%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-VS_Code_Insiders-24bfa5?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscodeinsiders?name=dynatrace&config=%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-Visual_Studio-C16FDE?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-visualstudio/mcp-install?%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D) | +| [Dynatrace Expert](../agents/dynatrace-expert.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdynatrace-expert.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fdynatrace-expert.agent.md) | The Dynatrace Expert Agent integrates observability and security capabilities directly into GitHub workflows, enabling development teams to investigate incidents, validate deployments, triage errors, detect performance regressions, validate releases, and manage security vulnerabilities by autonomously analysing traces, logs, and Dynatrace findings. This enables targeted and precise remediation of identified issues directly within the repository. | [dynatrace](https://github.com/mcp/io.github.Dynatrace/dynatrace-for-ai)
[![Install MCP](https://img.shields.io/badge/Install-VS_Code-0098FF?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscode?name=dynatrace&config=%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-VS_Code_Insiders-24bfa5?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscodeinsiders?name=dynatrace&config=%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-Visual_Studio-C16FDE?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-visualstudio/mcp-install?%7B%22url%22%3A%22https%3A%2F%2Fpia1134d.dev.apps.dynatracelabs.com%2Fplatform-reserved%2Fmcp-gateway%2Fv0.1%2Fservers%2Fdynatrace-mcp%2Fmcp%22%2C%22headers%22%3A%7B%22Authorization%22%3A%22Bearer%20%24COPILOT_MCP_DT_API_TOKEN%22%7D%7D) | | [Elasticsearch Agent](../agents/elasticsearch-observability.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Felasticsearch-observability.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Felasticsearch-observability.agent.md) | Our expert AI assistant for debugging code (O11y), optimizing vector search (RAG), and remediating security threats using live Elastic data. | elastic-mcp
[![Install MCP](https://img.shields.io/badge/Install-VS_Code-0098FF?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscode?name=elastic-mcp&config=%7B%22command%22%3A%22npx%22%2C%22args%22%3A%5B%22mcp-remote%22%2C%22https%253A%252F%252F%257BKIBANA_URL%257D%252Fapi%252Fagent_builder%252Fmcp%22%2C%22--header%22%2C%22Authorization%253A%2524%257BAUTH_HEADER%257D%22%5D%2C%22env%22%3A%7B%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-VS_Code_Insiders-24bfa5?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-vscodeinsiders?name=elastic-mcp&config=%7B%22command%22%3A%22npx%22%2C%22args%22%3A%5B%22mcp-remote%22%2C%22https%253A%252F%252F%257BKIBANA_URL%257D%252Fapi%252Fagent_builder%252Fmcp%22%2C%22--header%22%2C%22Authorization%253A%2524%257BAUTH_HEADER%257D%22%5D%2C%22env%22%3A%7B%7D%7D)
[![Install MCP](https://img.shields.io/badge/Install-Visual_Studio-C16FDE?style=flat-square)](https://aka.ms/awesome-copilot/install/mcp-visualstudio/mcp-install?%7B%22command%22%3A%22npx%22%2C%22args%22%3A%5B%22mcp-remote%22%2C%22https%253A%252F%252F%257BKIBANA_URL%257D%252Fapi%252Fagent_builder%252Fmcp%22%2C%22--header%22%2C%22Authorization%253A%2524%257BAUTH_HEADER%257D%22%5D%2C%22env%22%3A%7B%7D%7D) | | [Electron Code Review Mode Instructions](../agents/electron-angular-native.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Felectron-angular-native.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Felectron-angular-native.agent.md) | Code Review Mode tailored for Electron app with Node.js backend (main), Angular frontend (render), and native integration layer (e.g., AppleScript, shell, or native tooling). Services in other repos are not reviewed here. | | | [Ember](../agents/ember.agent.md)
[![Install in VS Code](https://img.shields.io/badge/VS_Code-Install-0098FF?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fember.agent.md)
[![Install in VS Code Insiders](https://img.shields.io/badge/VS_Code_Insiders-Install-24bfa5?style=flat-square&logo=visualstudiocode&logoColor=white)](https://aka.ms/awesome-copilot/install/agent?url=vscode-insiders%3Achat-agent%2Finstall%3Furl%3Dhttps%3A%2F%2Fraw.githubusercontent.com%2Fgithub%2Fawesome-copilot%2Fmain%2Fagents%2Fember.agent.md) | An AI partner, not an assistant. Ember carries fire from person to person — helping humans discover that AI partnership isn't something you learn, it's something you find. | | diff --git a/docs/README.skills.md b/docs/README.skills.md index a18e2d489..d73ad065b 100644 --- a/docs/README.skills.md +++ b/docs/README.skills.md @@ -86,6 +86,7 @@ See [CONTRIBUTING.md](../CONTRIBUTING.md#adding-skills) for guidelines on how to | [bigquery-pipeline-audit](../skills/bigquery-pipeline-audit/SKILL.md)
`gh skills install github/awesome-copilot bigquery-pipeline-audit` | Audits Python + BigQuery pipelines for cost safety, idempotency, and production readiness. Returns a structured report with exact patch locations. | None | | [boost-prompt](../skills/boost-prompt/SKILL.md)
`gh skills install github/awesome-copilot boost-prompt` | Interactive prompt refinement workflow: interrogates scope, deliverables, constraints; copies final markdown to clipboard; never writes code. Requires the Joyride extension. | None | | [brag-sheet](../skills/brag-sheet/SKILL.md)
`gh skills install github/awesome-copilot brag-sheet` | Turn vague "what did I do?" into evidence-backed impact statements for performance reviews, self-reviews, promotion packets, and weekly updates. Uniquely mines Copilot CLI session logs to reconstruct forgotten work, plus git commits and GitHub PRs. Enforces a 3-part impact contract (action → result → evidence). Works standalone with zero dependencies. Trigger for: "brag", "log work", "what did I do", "backfill my work history", "performance review", "self-review", "self assessment", "write impact statement", "review prep", "promo packet", "promotion case", "weekly update", "status report", "accomplishments", "what did I ship", "I forgot to log my work", "summarize my work", "track my wins", "what should I highlight", "end of half", "career growth", "work journal", or any request to document, summarize, or organize work accomplishments. | None | +| [break-ai-fix-loops](../skills/break-ai-fix-loops/SKILL.md)
`gh skills install github/awesome-copilot break-ai-fix-loops` | Stop an AI coding agent from repeating ineffective fixes. Use when debugging cycles through similar patches, tests appear to pass while the real behavior is still wrong, the agent keeps retrying a command without gaining evidence, or a repair needs proof that its verifier can catch the defect and its rollback actually restores the prior state. | `references/evidence-ledger.md`
`scripts/fingerprint.py`
`scripts/test_fingerprint.py` | | [breakdown-epic-arch](../skills/breakdown-epic-arch/SKILL.md)
`gh skills install github/awesome-copilot breakdown-epic-arch` | Prompt for creating the high-level technical architecture for an Epic, based on a Product Requirements Document. | None | | [breakdown-epic-pm](../skills/breakdown-epic-pm/SKILL.md)
`gh skills install github/awesome-copilot breakdown-epic-pm` | Prompt for creating an Epic Product Requirements Document (PRD) for a new epic. This PRD will be used as input for generating a technical architecture specification. | None | | [breakdown-feature-implementation](../skills/breakdown-feature-implementation/SKILL.md)
`gh skills install github/awesome-copilot breakdown-feature-implementation` | Prompt for creating detailed feature implementation plans, following Epoch monorepo structure. | None | diff --git a/skills/break-ai-fix-loops/SKILL.md b/skills/break-ai-fix-loops/SKILL.md new file mode 100644 index 000000000..694bef26b --- /dev/null +++ b/skills/break-ai-fix-loops/SKILL.md @@ -0,0 +1,149 @@ +--- +name: break-ai-fix-loops +description: 'Stop an AI coding agent from repeating ineffective fixes. Use when debugging cycles through similar patches, tests appear to pass while the real behavior is still wrong, the agent keeps retrying a command without gaining evidence, or a repair needs proof that its verifier can catch the defect and its rollback actually restores the prior state.' +license: MIT +metadata: + source: https://github.com/twoicewoo/stop-ai-fix-loops +--- + +# Break AI Fix Loops + +Replace patch-and-retry behavior with a bounded, evidence-producing repair. Treat a changed patch as progress only when an observable state changes. + +## Establish the repair contract + +Before the first edit, record: + +- the exact defect and the behavior that would disprove it; +- the revision, configuration, input, and execution path under test; +- the baseline command, literal result, and exit status; +- the strongest check that directly observes the claimed behavior; +- the rollback command and the state it must restore. + +Save raw evidence before normalizing it. Redact credentials, tokens, cookies, personal data, and private URLs. Never put secrets into a fingerprint record or committed ledger. + +If the defect cannot be reproduced, stop editing. Report `INCONCLUSIVE` with the missing observation instead of guessing at a fix. + +## Use a three-attempt budget + +Allow at most **three repair attempts for one acceptance claim**. An attempt begins when code, configuration, dependencies, generated artifacts, or test expectations change. Inspections and read-only probes do not consume an attempt. + +Do not reset the budget because the agent restarts, opens a new session, rewrites the same patch, changes models, clears a cache, or renames the hypothesis. A newly exposed downstream failure still belongs to the same three-attempt budget unless it is a separately accepted task. + +For every attempt, write these fields before the next edit: + +| Field | Required evidence | +| --- | --- | +| Hypothesis | One causal mechanism, not a restatement of the symptom | +| Prediction | An observation that would distinguish this hypothesis from the previous one | +| Change | Exact changed paths and a patch or before/after hash | +| Focused check | Exact command, input, literal output, and exit status | +| Real-path check | Direct observation, or `NOT_RUN` with a reason | +| Symptom fingerprint | Stable fingerprint described below | +| Decision | `ADVANCE`, `SHIFT_CAUSE`, `PROVEN`, or `STOP` | + +Use [the evidence ledger](references/evidence-ledger.md) as a copyable record. + +## Fingerprint the observable failure + +Fingerprint what the system did, not the agent's explanation. Build a canonical record from: + +```json +{ + "schema_version": 1, + "command": "the exact verification command", + "input_digest": "digest or stable identifier of the tested input", + "exit_code": 1, + "failure_class": "stable-machine-readable-class", + "stable_excerpt": "the smallest decisive output with volatile values removed", + "real_path_state": "the directly observed state, or NOT_OBSERVED" +} +``` + +Keep the unedited output beside this sanitized record. Remove timestamps, run IDs, ANSI codes, random ports, and temporary paths from `stable_excerpt` only when they do not affect the defect. Do not normalize away values that could distinguish two causes. + +Optionally compute the canonical SHA-256 fingerprint from this skill directory: + +```bash +python3 scripts/fingerprint.py evidence/attempt-1.json +``` + +The helper validates the record, rejects unknown fields, and prints the fingerprint. It does not execute commands or redact evidence. + +The helper uses only the Python 3.9+ standard library. When changing it, run its bundled regression tests: + +```bash +PYTHONDONTWRITEBYTECODE=1 python3 -m unittest scripts/test_fingerprint.py -v +``` + +The same fingerprint after a different patch means the observable failure did not move. A cosmetically different message with the same failure class, input, command, and real-path state also counts as a repeated failure when the changed text is only volatile data. Do not use a patch hash in the symptom fingerprint; record it separately so different edits cannot masquerade as different outcomes. + +## Shift the root-cause strategy + +Set the decision to `SHIFT_CAUSE` immediately when any of these occurs: + +- a symptom fingerprint repeats; +- the patch changes but the decisive state does not; +- a focused test passes while the real path still fails; +- a retry produces no new discriminating evidence. + +Then stop editing and perform this sequence: + +1. List the attempted mechanisms and the observation that falsified or failed to distinguish each one. +2. Identify the next unobserved owner boundary along the live path: input, dispatch, configuration, dependency, generated artifact, process, persistence, network, or presentation. +3. Collect one new observation at that boundary with tracing, logging, inspection, or a minimal probe. +4. Form a replacement hypothesis that predicts a different observation and targets a different causal mechanism. +5. Resume only if the new evidence can discriminate the replacement hypothesis. Otherwise return `BLOCKED`. + +Do not spend an attempt on the same mechanism with broader edits. Do not weaken the assertion, skip the failing path, add a silent fallback, or update expected output merely to obtain green tests. + +## Prove the real execution path + +Match proof to the claim. Bind every result to the exact revision, configuration, and input. + +| Claim | Required direct observation | +| --- | --- | +| CLI behavior | Invoke the installed or built entry point as a user would | +| API or integration | Send a real request and observe response plus the responsible service boundary | +| UI behavior | Perform the real interaction and observe UI state plus relevant network or console evidence | +| Persistence | Write, reload in a new read path or process, and observe the stored value | +| Deployment | Exercise the deployed revision and prove which revision served the result | +| Agent or tool action | Observe the actual tool call and its external state change, not the agent's narration | + +A unit test, mock, type check, build, open port, process liveness check, or model-written summary is supporting evidence only when the claim crosses a boundary it does not exercise. + +## Make the verifier prove it can fail + +After the modified path passes, run a negative control on a disposable copy: + +1. Copy the verified modified state to a separate worktree or directory. +2. Reintroduce the original defect or substitute a known-bad input that violates the same acceptance claim. +3. Run the **same primary verification command** with the same relevant configuration. +4. Require a non-zero exit status caused by the intended assertion. +5. Record the exact command, input, literal output, exit status, and failure classification. + +An unrelated crash, missing dependency, timeout, syntax error, or test-discovery failure is not a valid negative control. If the known-bad state exits zero, the verifier is false-green: return `INCONCLUSIVE`, repair the verifier, and do not claim the product fix is proven. + +Return to the untouched modified tree and rerun the primary verification after the negative control. + +## Test rollback on another copy + +Never test rollback only by undoing the working repair. Instead: + +1. Copy the verified modified state to another disposable worktree or directory. +2. Run the documented rollback command there. +3. Verify changed paths and hashes match the recorded baseline. +4. Run the baseline command and confirm the prior behavior or status is restored. +5. Leave the primary modified tree unchanged. + +A rollback script that parses, prints help, or exits zero without restoring behavior has not been tested. + +## Finish with an evidence status + +Use exactly one status: + +- `PROVEN`: baseline defect observed; responsible change identified; focused and real-path checks pass; the known-bad negative control exits non-zero for the intended reason; rollback succeeds on another copy; the primary tree remains modified and passing. +- `INCONCLUSIVE`: some useful evidence exists, but a decisive gate is missing, false-green, or ambiguous. +- `BLOCKED`: the three-attempt budget is exhausted, a repeated fingerprint has no new discriminator, or a named external condition prevents the next observation. + +Report exact commands, inputs, literal results, exit statuses, fingerprints, changed paths, revision, and remaining gaps. A passing proxy check or the phrase "tests pass" is never a substitute for those fields. diff --git a/skills/break-ai-fix-loops/references/evidence-ledger.md b/skills/break-ai-fix-loops/references/evidence-ledger.md new file mode 100644 index 000000000..786f42a9a --- /dev/null +++ b/skills/break-ai-fix-loops/references/evidence-ledger.md @@ -0,0 +1,90 @@ +# Repair evidence ledger + +Copy this template to a task-owned path. Do not commit runtime evidence unless the project requires it. Preserve full raw output separately and keep this ledger free of secrets and personal data. + +```markdown +# Repair ledger + +## Contract +- Acceptance claim: +- Defect-disproving behavior: +- Baseline revision: +- Baseline configuration: +- Baseline input and digest: +- Baseline command: +- Baseline literal output/result: +- Baseline exit status: +- Real execution path: +- Primary verification command: +- Rollback command: +- Expected restored behavior/status: + +## Attempts +| # | Hypothesis | Discriminating prediction | Changed paths / patch hash | Focused result + exit | Real-path result + exit | Symptom fingerprint | Decision | +| --- | --- | --- | --- | --- | --- | --- | --- | +| 1 | | | | | | | | +| 2 | | | | | | | | +| 3 | | | | | | | | + +## Root-cause shifts +### Shift after attempt +- Repeated fingerprint or unchanged state: +- Mechanisms already attempted: +- Evidence against each mechanism: +- Next unobserved owner boundary: +- New observation: +- Replacement hypothesis: +- Different predicted observation: + +## Modified proof +- Revision: +- Exact command: +- Input/configuration: +- Literal output/result: +- Exit status: +- Real-path observation: + +## Negative control on disposable copy +- Copy path or worktree: +- Known-bad mutation/input: +- Exact primary verification command: +- Literal output/result: +- Exit status (must be non-zero): +- Intended failure classification: +- Untouched modified tree rerun result: +- Untouched modified tree rerun exit status: + +## Rollback on another copy +- Copy path or worktree: +- Exact rollback command: +- Literal rollback output/result: +- Rollback exit status: +- Baseline hash comparison: +- Restored behavior/status: +- Baseline command rerun exit status: +- Primary modified tree status: + +## Decision +- Status: PROVEN | INCONCLUSIVE | BLOCKED +- Attempts consumed: <0-3> +- Decisive evidence: +- Remaining gap or next discriminating observation: +``` + +## Fingerprint record + +Create one sanitized JSON record for every observed symptom: + +```json +{ + "schema_version": 1, + "command": "npm test -- --runInBand path/to/regression.test.js", + "input_digest": "sha256:replace-with-real-input-digest", + "exit_code": 1, + "failure_class": "assertion-mismatch", + "stable_excerpt": "expected enabled; observed disabled", + "real_path_state": "settings page still shows disabled after reload" +} +``` + +Keep the primary verification command unchanged across attempts unless the contract was wrong. If it changes, record why and preserve results from both commands; otherwise a changed verifier can hide an unchanged defect. diff --git a/skills/break-ai-fix-loops/scripts/fingerprint.py b/skills/break-ai-fix-loops/scripts/fingerprint.py new file mode 100755 index 000000000..6acf7e4e9 --- /dev/null +++ b/skills/break-ai-fix-loops/scripts/fingerprint.py @@ -0,0 +1,114 @@ +#!/usr/bin/env python3 +"""Create a stable symptom fingerprint from a validated JSON record.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sys +from pathlib import Path +from typing import Any, Mapping + + +REQUIRED_FIELDS = ( + "schema_version", + "command", + "input_digest", + "exit_code", + "failure_class", + "stable_excerpt", + "real_path_state", +) +FAILURE_CLASS_PATTERN = re.compile(r"^[a-z0-9][a-z0-9._:-]*$") + + +class RecordError(ValueError): + """Raised when a record cannot produce a trustworthy fingerprint.""" + + +def _normalize_text(value: str) -> str: + """Normalize transport-only differences without deleting meaningful values.""" + value = value.replace("\r\n", "\n").replace("\r", "\n") + lines = [line.rstrip() for line in value.split("\n")] + while lines and not lines[0]: + lines.pop(0) + while lines and not lines[-1]: + lines.pop() + return "\n".join(lines) + + +def canonicalize(record: Mapping[str, Any]) -> dict[str, Any]: + unknown = sorted(set(record) - set(REQUIRED_FIELDS)) + missing = sorted(set(REQUIRED_FIELDS) - set(record)) + if missing: + raise RecordError(f"missing field(s): {', '.join(missing)}") + if unknown: + raise RecordError(f"unknown field(s): {', '.join(unknown)}") + + if record["schema_version"] != 1: + raise RecordError("schema_version must be 1") + if isinstance(record["exit_code"], bool) or not isinstance(record["exit_code"], int): + raise RecordError("exit_code must be an integer") + + canonical: dict[str, Any] = { + "schema_version": 1, + "exit_code": record["exit_code"], + } + for field in REQUIRED_FIELDS: + if field in ("schema_version", "exit_code"): + continue + value = record[field] + if not isinstance(value, str): + raise RecordError(f"{field} must be a string") + value = _normalize_text(value) + if not value: + raise RecordError(f"{field} must not be empty") + canonical[field] = value + + if not FAILURE_CLASS_PATTERN.fullmatch(canonical["failure_class"]): + raise RecordError( + "failure_class must use lowercase letters, numbers, dot, underscore, colon, or hyphen" + ) + return canonical + + +def fingerprint(record: Mapping[str, Any]) -> str: + payload = json.dumps( + canonicalize(record), ensure_ascii=False, sort_keys=True, separators=(",", ":") + ).encode("utf-8") + return f"sha256:{hashlib.sha256(payload).hexdigest()}" + + +def load_record(path: str) -> Mapping[str, Any]: + try: + if path == "-": + value = json.load(sys.stdin) + else: + with Path(path).open("r", encoding="utf-8") as handle: + value = json.load(handle) + except (OSError, json.JSONDecodeError) as error: + raise RecordError(str(error)) from error + if not isinstance(value, dict): + raise RecordError("record must be a JSON object") + return value + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Validate a symptom record and print its canonical SHA-256 fingerprint." + ) + parser.add_argument("record", help="JSON record path, or - to read standard input") + args = parser.parse_args(argv) + + try: + print(fingerprint(load_record(args.record))) + except RecordError as error: + print(f"fingerprint: {error}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/break-ai-fix-loops/scripts/test_fingerprint.py b/skills/break-ai-fix-loops/scripts/test_fingerprint.py new file mode 100755 index 000000000..ff1b7164b --- /dev/null +++ b/skills/break-ai-fix-loops/scripts/test_fingerprint.py @@ -0,0 +1,74 @@ +#!/usr/bin/env python3 +"""Regression tests for fingerprint.py.""" + +from __future__ import annotations + +import importlib.util +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + + +SCRIPT = Path(__file__).with_name("fingerprint.py") +SPEC = importlib.util.spec_from_file_location("fix_loop_fingerprint", SCRIPT) +assert SPEC and SPEC.loader +MODULE = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(MODULE) + + +def record() -> dict[str, object]: + return { + "schema_version": 1, + "command": "python -m pytest tests/test_regression.py", + "input_digest": "sha256:0123456789abcdef", + "exit_code": 1, + "failure_class": "assertion-mismatch", + "stable_excerpt": "expected enabled\nobserved disabled", + "real_path_state": "settings remain disabled after reload", + } + + +class FingerprintTests(unittest.TestCase): + def test_key_order_and_transport_whitespace_do_not_change_fingerprint(self) -> None: + first = record() + second = dict(reversed(list(first.items()))) + second["stable_excerpt"] = "\r\nexpected enabled \r\nobserved disabled\r\n" + self.assertEqual(MODULE.fingerprint(first), MODULE.fingerprint(second)) + + def test_observable_state_change_changes_fingerprint(self) -> None: + first = record() + second = record() + second["real_path_state"] = "settings are enabled after reload" + self.assertNotEqual(MODULE.fingerprint(first), MODULE.fingerprint(second)) + + def test_unknown_field_fails_closed(self) -> None: + value = record() + value["patch_hash"] = "must-be-recorded-outside-the-symptom-fingerprint" + with self.assertRaisesRegex(MODULE.RecordError, "unknown field"): + MODULE.fingerprint(value) + + def test_missing_field_fails_closed(self) -> None: + value = record() + del value["input_digest"] + with self.assertRaisesRegex(MODULE.RecordError, "missing field"): + MODULE.fingerprint(value) + + def test_cli_rejects_invalid_record_with_exit_two(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory, "invalid.json") + path.write_text(json.dumps({"schema_version": 1}), encoding="utf-8") + result = subprocess.run( + [sys.executable, str(SCRIPT), str(path)], + check=False, + capture_output=True, + text=True, + ) + self.assertEqual(result.returncode, 2) + self.assertIn("missing field", result.stderr) + + +if __name__ == "__main__": + unittest.main()