diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.md index c0046e5783..e1b5412c8f 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.md @@ -78,11 +78,11 @@ inventory-only row for the same provider id. ## Product surfaces -- CLI: `external-evidence discover|plan|receipt|admit|retire`. -- Managed Turn: the same five effect-runtime methods. -- Frontend/Lark: not changed in this Core slice. A companion slice should render - the same typed plan/admission projection and readback; it must not invent a - second registry or lifecycle. +- CLI: `external-evidence discover|plan|execute|receipt|admit|readback|retire`. +- Managed Turn: the same five typed effect-runtime methods; explicit provider + execution and ledger projection use the capability's CLI owner. +- Frontend/Lark: existing conversation answer/report and Markdown transports + render the shared validated readback. No independent registry or lifecycle. ## Acceptance @@ -102,6 +102,29 @@ inventory-only row for the same provider id. - retirement waits for downstream coverage of every admitted source; - CLI and effect-runtime TypeScript tests pass from the source checkout. +## Delivery checkpoint (2026-10-02) + +The public GitHub method now completes a bounded real journey: anonymous pinned +file reads, exact-plan receipt validation, a separate parent decision, projection +into the existing deepresearch source ledger, actual lineage readback and retirement. +Optional source refs and literal search terms are bound into the request/plan digest; +legacy requests retain their existing identity. The provider is bundled in extensions +under `method:public-github`; capability and ledger owners remain unchanged. + +Passed: real public-provider/source CLI journey; negative cases for private or stale +readiness, malformed/unpinned sources, plan/admission mutation, partial/empty/failed +reads, independent admission and coverage, wrong-question projection, budget failure +and idempotent replay; packaged desktop/mobile conversation readback and reload; +existing Lark Markdown presentation. Source bodies are not persisted. The shared +Markdown readback uses existing answer/report and Lark transports; no frontend +configuration or parallel evidence authority is needed. + +Commands are in the [versioned capability guide](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法). +Live Lark delivery, authenticated connector execution and broader semantic research +quality remain untested; this checkpoint does not promote those providers or close +S6/S8. Failed or partial results preserve original-source fallback, and neither a +successful read nor a parent admission certifies evidence completeness. + ## Non-goals - a universal browser/search engine; diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md index 898efaee2f..d939db48c7 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md @@ -60,10 +60,11 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 ## 产品入口 -- CLI:`external-evidence discover|plan|receipt|admit|retire`; -- Managed Turn:复用同五个 effect-runtime 方法; -- Frontend/Lark:本 Core 切片不修改。后续 companion slice 只渲染同源 plan/admission - 投影与读回,不建立第二个 registry 或生命周期。 +- CLI:`external-evidence discover|plan|execute|receipt|admit|readback|retire`; +- Managed Turn:复用五个 typed effect-runtime 方法;显式 provider 执行与账本投影 + 使用能力的 CLI owner; +- Frontend/Lark:现有会话答复/报告和 Markdown 运输渲染同源校验回读, + 不建立独立 registry 或生命周期。 ## 验收 @@ -80,6 +81,24 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 - 全部被采纳来源完成下游覆盖前不得退休; - CLI 与 effect-runtime TypeScript 测试在源码 checkout 中通过。 +## 交付检查点(2026-10-02) + +公开 GitHub method 已完成有界真实链路:匿名读取固定提交文件、精确 plan 回执校验、 +独立父 Agent 决定、投影到现有 deepresearch 来源账本、实际 lineage 回读与退休。 +可选 source refs 和字面检索词进入 request/plan digest;旧请求身份保持兼容。 +provider 以 `method:public-github` 内置在 extensions,能力和账本 owner 不变。 + +通过:真实公开 provider/源码 CLI 链路;私有或过期 readiness、无效/未固定来源、 +plan/admission 篡改、部分/空/失败读取、独立采纳与覆盖、问题不匹配、预算耗尽及 +幂等重放等负向用例;打包桌面/移动会话回读与重载;现有 Lark Markdown 展示。 +不持久化来源正文。同源 Markdown 沿用现有答复/报告和 Lark 运输路径, +无需新增前端配置或并行证据权威。 + +命令参见[版本化能力指南](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法)。 +真实 Lark 送达、带凭据 connector 执行和更广泛语义研究质量尚未验证; +该检查点不晋升这些 provider,也不关闭 S6/S8。失败或部分结果保留原始来源退路; +读取成功和父 Agent 采纳均不证明证据完整性。 + ## 非目标 - 通用浏览器或搜索引擎; diff --git a/examples/personal-workspace-browser-smoke.mjs b/examples/personal-workspace-browser-smoke.mjs index ee9cf4830d..1f6df697af 100644 --- a/examples/personal-workspace-browser-smoke.mjs +++ b/examples/personal-workspace-browser-smoke.mjs @@ -1,6 +1,7 @@ #!/usr/bin/env node import {nativeChildActivityScenario} from "./personal-workspace-browser/native-child-activity.mjs"; import {conversationImageRequestScenario} from "./personal-workspace-browser/conversation-image-request.mjs"; +import {externalEvidenceReadbackScenario} from "./personal-workspace-browser/external-evidence-readback.mjs"; // Isolated browser acceptance scenarios for the personal Agent workspace. import { mkdir, writeFile } from "node:fs/promises"; @@ -74,6 +75,7 @@ scenarioCatalog.push(goalWorkMapScenario); scenarioCatalog.push(performanceDiagnosisScenario); scenarioCatalog.push(blockedNoticeSettingsScenario); scenarioCatalog.push(nativeChildActivityScenario); +scenarioCatalog.push(externalEvidenceReadbackScenario); const requestedScenario = process.env.LOOPX_PERSONAL_WORKSPACE_SCENARIO; const scenarios = requestedScenario ? scenarioCatalog.filter((scenario) => scenario.id === requestedScenario) diff --git a/examples/personal-workspace-browser/external-evidence-readback.mjs b/examples/personal-workspace-browser/external-evidence-readback.mjs new file mode 100644 index 0000000000..ed2ec67cdf --- /dev/null +++ b/examples/personal-workspace-browser/external-evidence-readback.mjs @@ -0,0 +1,57 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { resolve } from "node:path"; +import { resolveTestPython } from "../../scripts/test-python.mjs"; +import { outputDir, repoRoot } from "./fixture.mjs"; +import { openWorkspacePage } from "./scenario-context.mjs"; + +export const externalEvidenceReadbackScenario = { + id: "external-evidence-readback", + async run({browser, collectCoverage, url}) { + // Exercise the product's typed plan/admission and actual downstream ledger, + // then hand the resulting shared readback to the existing answer surface. + const result = spawnSync(resolveTestPython(), ["-X", "utf8", "-c", ` +import tempfile +from pathlib import Path +from loopx.control_plane.effect_runtime import effect_runtime_result as effect +from loopx.capabilities.deep_research.runtime import start_research +from loopx.capabilities.external_research.projection import readback, render_readback +provider = {"provider_id":"method:public-github", "provider_kind":"method", "protocol":"external_evidence_research_v0", "declared":True, "installed":True, "enabled":True, "ready":True, "unavailable_reason":None} +ref = "https://github.com/example/public/blob/" + "a"*40 + "/README.md" +plan = effect("external_evidence.plan", {"request":{"objective":"Inspect public fixture", "user_activity":"Choose a source", "decision":"Whether to use the fixture", "evidence_kinds":["literal_match"], "source_refs":[ref]}, "providers":[provider]}) +receipt = {"schema_version":"loopx_external_evidence_receipt_v0", "plan_id":plan["plan_id"], "request_id":plan["request"]["request_id"], "provider_id":provider["provider_id"], "provider_kind":"method", "status":"succeeded", "summary":"Read pinned public fixture", "completed_at":"2026-10-02T00:00:00Z", "sources":[{"source_ref":ref, "source_family":"github_repository_file", "basis":"observed", "finding":"Literal fixture marker observed at line 1", "content_digest":"sha256:"+"b"*64, "accessed_at":"2026-10-02T00:00:00Z", "limitation":"Retrieval only; completeness unverified"}]} +admission = effect("external_evidence.admit", {"plan":plan, "receipt":receipt, "decision":{"disposition":"admit", "reason":"Synthetic parent checked the direct source", "admitted_source_refs":[ref]}}) +with tempfile.TemporaryDirectory(prefix="lxe-ui-") as folder: + project=Path(folder) + start_research(project, question=plan["request"]["objective"], max_sources=8, max_subquestions=4) + print(render_readback(readback(plan, receipt, admission, project=project, execute=True))) +`], {cwd:repoRoot, encoding:"utf8", env:{...process.env, PYTHONPATH:repoRoot}, timeout:45000}); + assert.equal(result.status, 0, result.stderr); + const markdown = result.stdout; + const context = await openWorkspacePage(browser, url, {collectCoverage}); + const {api, page} = context; + try { + api.answerForMessage = () => markdown; + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await page.getByLabel("向 LoopX 发送消息").fill("Show the external evidence readback"); + await page.getByRole("button", {name:"发送",exact:true}).click(); + const answer = page.locator(".personal-channel-timeline .personal-message.is-assistant", {hasText:"External evidence"}); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.match(await answer.innerText(), /Downstream coverage.*1 admitted/u); + assert.match(await answer.innerText(), /Evidence completeness is unverified/u); + await answer.scrollIntoViewIfNeeded(); + await page.screenshot({path:resolve(outputDir,"external-evidence-desktop.png"),fullPage:false,animations:"disabled"}); + await page.setViewportSize({width:390,height:844}); + await answer.scrollIntoViewIfNeeded(); + assert.equal(await answer.evaluate(el => el.scrollWidth <= el.clientWidth + 1), true); + await page.screenshot({path:resolve(outputDir,"external-evidence-mobile.png"),fullPage:false,animations:"disabled"}); + await page.reload({waitUntil:"networkidle"}); + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.equal(context.errors.length, 0); + return {coverageEntries:context.coverageEntries, note:"Shared typed evidence readback survives conversation reload and fits desktop/mobile."}; + } finally { await context.close(); } + }, +}; diff --git a/examples/public-github-evidence-live-smoke.py b/examples/public-github-evidence-live-smoke.py new file mode 100644 index 0000000000..f3587892e5 --- /dev/null +++ b/examples/public-github-evidence-live-smoke.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Opt-in public GitHub exact-plan journey through the shipped source CLI. + +Only anonymous public GETs and a disposable synthetic research ledger are used. +The explicit synthetic parent decision is separate from provider execution. +""" +from __future__ import annotations + +import argparse +import json +import subprocess +import sys +import tempfile +from pathlib import Path + + +def qualify(source: str, root: Path) -> dict: + def run(*args, json_output=True): + result = subprocess.run([sys.executable, "-X", "utf8", "-m", "loopx.entrypoint", + "--runtime-root", str(root / "runtime"), "--registry", str(root / "registry.json"), + *args, "--format", "json" if json_output else "markdown"], capture_output=True, + text=True, encoding="utf-8", timeout=90, check=True) + return json.loads(result.stdout) if json_output else result.stdout + + def save(name, value): + path = root / name + path.write_text(json.dumps(value), encoding="utf-8") + return str(path) + + objective = "Inspect the public pinned README for the LoopX literal" + plan = run("external-evidence", "plan", "--objective", objective, + "--user-activity", "Choose a public source", "--decision", "Whether the source contains LoopX", + "--evidence-kind", "literal_match", "--public-github", "--source", source, "--search-term", "LoopX") + assert plan["status"] == "ready" + plan_path = save("plan.json", plan) + execution = run("external-evidence", "execute", "--plan-json", plan_path, "--execute") + receipt = execution["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + assert "observed at lines" in receipt["sources"][0]["finding"] + receipt_path = save("receipt.json", execution) + before = run("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path) + assert before["parent_admission"] is None and before["retirement"] is None + admitted = run("external-evidence", "admit", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--decision", "admit", "--reason", "Synthetic parent checked the pinned file and literal-match finding", + "--admit-source", source) + admission_path = save("admission.json", admitted) + project = root / "research" + run("deepresearch", "start", "--project", str(project), "--question", objective) + args = ("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--admission-json", admission_path, "--project", str(project)) + assert run(*args)["retirement"]["status"] == "retained" + result = run(*args, "--execute") + assert result["downstream_source_refs"] == [source] + assert result["retirement"]["status"] == "retire_ready" + assert run(*args, "--execute") == result + markdown = run(*args, json_output=False) + assert source in markdown and "retire_ready" in markdown and "Evidence completeness is unverified" in markdown + return {"ok": True, "provider": "method:public-github", "sources_observed": 1, + "explicit_parent_admission": True, "actual_ledger_readback": True, + "retained_before_projection": True, "idempotent_projection": True, + "raw_content_persisted": False, "source_ref": source} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--execute-public-provider", action="store_true") + parser.add_argument("--source", required=True) + args = parser.parse_args() + if not args.execute_public_provider: + parser.error("--execute-public-provider is required for anonymous public GETs") + with tempfile.TemporaryDirectory(prefix="lxe-") as folder: + print(json.dumps(qualify(args.source, Path(folder)), sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/loopx/capabilities/deep_research/runtime.py b/loopx/capabilities/deep_research/runtime.py index bbabbaed81..0abf998ea2 100644 --- a/loopx/capabilities/deep_research/runtime.py +++ b/loopx/capabilities/deep_research/runtime.py @@ -13,6 +13,7 @@ from pathlib import Path from typing import Any +from ...control_plane.content_digest import ENVELOPED_SHA256_PATTERN from ...file_lock import exclusive_file_lock COMMAND = "/loopx-deepresearch" @@ -238,6 +239,8 @@ def add_source( tool: str, title: str | None, claims: list[dict[str, Any]], + external_evidence: dict[str, str] | None = None, + expected_question: str | None = None, ) -> dict[str, Any]: # Load, validate the whole batch, allocate ids, mutate, and save all under # one project-level lock: concurrent deepresearch commands are a normal @@ -245,13 +248,31 @@ def add_source( # when the atomic rename itself succeeds. with exclusive_file_lock(state_path(project), operation="deepresearch_add_source"): state = _require_active_state(project) + if expected_question is not None and state["question"] != expected_question: + raise ValueError("external evidence objective does not match the active research question") + if external_evidence is not None: + fields = {"plan_id", "admission_id", "receipt_digest", "content_digest"} + if set(external_evidence) != fields or any( + not isinstance(value, str) or not ENVELOPED_SHA256_PATTERN.fullmatch(value) + for value in external_evidence.values() + ): + raise ValueError("external evidence lineage requires exact content-addressed identities") url_or_path = url_or_path.strip() tool = tool.strip() or "unspecified" if not url_or_path: raise ValueError("--url-or-path must be non-empty") normalized = _normalize_source_ref(url_or_path) for source in state["sources"]: - if _normalize_source_ref(str(source["url_or_path"])) == normalized: + same_source = ( + str(source["url_or_path"]).strip().rstrip("/") == url_or_path.rstrip("/") + if external_evidence is not None else + _normalize_source_ref(str(source["url_or_path"])) == normalized + ) + if same_source: + if (external_evidence is not None and source.get("external_evidence") == external_evidence + and [claim["text"] for claim in state["claims"] if claim["id"] in source["claims"]] + == [str(claim.get("text", "")).strip() for claim in claims]): + return {"source_id": source["id"], "claim_ids": source["claims"], "state": state} raise ValueError( f"source already recorded as {source['id']} " f"({source['url_or_path']}); reuse its claims instead of re-reading" @@ -325,6 +346,7 @@ def add_source( "title": (title or "").strip() or None, "accessed_at": _now_iso(), "claims": claim_ids, + **({"external_evidence": dict(external_evidence)} if external_evidence is not None else {}), } ) _save_state(project, state) diff --git a/loopx/capabilities/external_research/README.md b/loopx/capabilities/external_research/README.md index ef3874156e..3535d58a41 100644 --- a/loopx/capabilities/external_research/README.md +++ b/loopx/capabilities/external_research/README.md @@ -106,10 +106,73 @@ credentials, and private notes remain provider-private. `external_evidence.discover`, `external_evidence.plan`, `external_evidence.receipt`, `external_evidence.admit`, and `external_evidence.retire`. -- Frontend and Lark are companion slices. They should render the same plan and - admission projection; neither gets an independent provider registry or - evidence state machine. +- CLI `readback` renders the same validated plan, source, parent decision, + actual deepresearch-ledger coverage and retirement as Markdown. Existing + conversation answer/report and Lark Markdown transports consume that output; + no new UI configuration or evidence state machine is introduced. TypeScript 是 discovery 真值边界、请求身份、provider 准入、provenance 校验、父 Agent 采纳、紧凑投影与 退休条件的唯一语义 owner。Python 仅适配 CLI 与 effect-runtime transport。Managed -Turn 复用同一方法;frontend/Lark 后续只渲染同源投影,不新建 registry 或状态机。 +Turn 复用同一方法;CLI `readback` 的同源 Markdown 可由现有会话答复/报告和 Lark Markdown 运输路径展示,不新建 registry 或状态机。 + +## Public GitHub method / 公开 GitHub 方法 + +The bundled `method:public-github` provider performs anonymous, bounded HTTPS +GETs only. It reads UTF-8 files from explicitly selected full commit SHA URLs. +No token, cookie, private repository, branch-head URL, redirect, proxy credential, +raw page persistence or automatic admission is used. Repository public visibility +is checked at planning and again for each execution read. A saved ready row alone +cannot authorize or prove a successful read. + +该内置 method 只做匿名、有界 HTTPS GET,只接受显式选择的完整 commit SHA 文件 URL。 +规划和执行均检查仓库当前公开性;不使用 token、cookie、私有仓库、分支 head、重定向、 +代理凭据、原文持久化或自动采纳。保留的 ready 行本身不证明执行成功。 + +```bash +loopx external-evidence plan --public-github \ + --objective "Inspect public source" --user-activity "Choose a source" \ + --decision "Whether a literal is present" --evidence-kind literal_match \ + --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --search-term LoopX --format json > plan.json +loopx external-evidence execute --plan-json plan.json --execute --format json > execution.json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json +# The parent separately inspects findings and chooses admit/reject. +loopx external-evidence admit --plan-json plan.json --receipt-json execution.json \ + --decision admit --reason "Direct source answers this bounded decision" \ + --admit-source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --format json > admission.json +loopx deepresearch start --project research --question "Inspect public source" +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research --execute --format json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research +``` + +`execute --execute` authorizes source reads only. `readback --execute` authorizes +writing already explicitly admitted compact sources to the **existing** research +ledger. It requires the active research question to match the plan objective. +Retries for the same admission are idempotent. Wrong questions, unrelated existing +sources and exhausted source budgets remain blockers; partial projection stays +retained until every admitted source is actually read back with matching lineage. +Omit `--execute` for read-only projection; omit `--public-github` to avoid the +provider readiness probe. There is no persistent provider enablement to uninstall. +Original sources remain available on partial, empty and failed results. Literal +matches prove neither semantic conclusions nor evidence completeness. + +`execute --execute` 只授权读取来源;`readback --execute` 只把已明确采纳的紧凑证据 +写入现有研究账本,且要求研究问题与 plan objective 相同。同一 admission 可幂等重试; +问题不匹配、已有不相关来源或预算耗尽仍为 blocker。部分下游投影保持 retained, +直至全部已采纳来源以匹配 lineage 实际回读。省略 `--execute` 可只读回读;省略 +`--public-github` 不探测该 provider。没有持久开关需要卸载。部分、空或失败证据 +保留原始来源退路;字面匹配不证明语义结论或证据完整性。 + +Real qualification (anonymous network reads, disposable synthetic ledger): +`uv run --extra test python examples/public-github-evidence-live-smoke.py --execute-public-provider --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md`. +Packaged conversation readback: set `LOOPX_PERSONAL_WORKSPACE_PACKAGED=1` and +`LOOPX_PERSONAL_WORKSPACE_SCENARIO=external-evidence-readback`, then run +`node examples/personal-workspace-browser-smoke.mjs`. +Live authenticated connector and Lark delivery qualification remain separate; +this method grants no connector credentials or outbound-message authority. + +真实验收会进行匿名网络读取并使用一次性合成账本;打包会话验收沿用上述环境变量和命令。 +真实带凭据 connector 与 Lark 送达资格仍是独立边界;本方法不授予 connector 凭据或外发消息权限。 diff --git a/loopx/capabilities/external_research/catalog_entry.py b/loopx/capabilities/external_research/catalog_entry.py index 047db8bc36..97d874b3ea 100644 --- a/loopx/capabilities/external_research/catalog_entry.py +++ b/loopx/capabilities/external_research/catalog_entry.py @@ -21,12 +21,12 @@ ), "user_value": ( "Discover method and connector inventory, select only a currently ready provider, " - "bind a caller-presented provider receipt to its exact plan, and admit or reject compact " + "execute explicitly selected public GitHub sources, bind a provider receipt to its exact plan, and admit or reject compact " "provenance without copying raw provider content." ), "next_real_step": ( - "run `loopx external-evidence discover --connector-registry`, then provide a current " - "provider inventory to plan and admit the returned provider receipt" + "run `loopx external-evidence plan --public-github --help`, then inspect the returned " + "receipt before explicit admission and downstream ledger readback" ), "entry_command": "loopx external-evidence discover --help", "commands": [ @@ -40,6 +40,16 @@ "purpose": "Bind object, user activity, decision, and evidence kinds to one ready provider.", "write_boundary": "read-only", }, + { + "command": "loopx external-evidence execute --plan-json plan.json --execute", + "purpose": "Read explicit pinned public GitHub sources; parent admission remains separate.", + "write_boundary": "anonymous public HTTPS reads; no raw persistence", + }, + { + "command": "loopx external-evidence readback --plan-json plan.json --receipt-json execution.json ...", + "purpose": "Show the same source lineage, parent decision and actual downstream coverage.", + "write_boundary": "read-only by default; --execute projects admitted sources to the existing research ledger", + }, { "command": "loopx external-evidence receipt --plan-json plan.json --receipt-json receipt.json", "purpose": ( diff --git a/loopx/capabilities/external_research/cli.py b/loopx/capabilities/external_research/cli.py index fd3c4da9db..697741e8bf 100644 --- a/loopx/capabilities/external_research/cli.py +++ b/loopx/capabilities/external_research/cli.py @@ -8,6 +8,8 @@ from ...control_plane.effect_runtime import effect_runtime_result from ..connector_registry.core import load_connector_registry +from ...extensions.public_github_research import execute_public_github, inspect_provider +from .projection import readback, render_readback, validate_receipt PrintPayload = Callable[ @@ -32,6 +34,14 @@ def _load_object(path_text: str, *, label: str) -> dict[str, Any]: return value +def _load_receipt(path_text: str) -> dict: + value = _load_object(path_text, label="external evidence receipt") + receipt = value.get("receipt", value) + if not isinstance(receipt, dict): + raise ValueError("external evidence receipt must be an object") + return receipt + + def _provider_inventory(value: Mapping[str, Any]) -> list[dict[str, object]]: providers = value.get("providers") if not isinstance(providers, list): @@ -80,6 +90,8 @@ def _merge_providers( def _render(payload: dict[str, object]) -> str: + if payload.get("schema_version") == "loopx_external_evidence_readback_v0": + return render_readback(payload) lines = ["# LoopX External Evidence", ""] for field in ( "status", @@ -137,11 +149,27 @@ def register_external_evidence_commands( plan.add_argument("--decision", required=True) plan.add_argument("--evidence-kind", action="append", required=True) plan.add_argument("--constraint", action="append", default=[]) - plan.add_argument("--provider-inventory-json", required=True) + plan.add_argument("--provider-inventory-json") + plan.add_argument("--public-github", action="store_true", help="Opt in to a fresh anonymous public GitHub readiness probe.") + plan.add_argument("--source", action="append", default=[]) + plan.add_argument("--search-term", action="append", default=[]) plan.add_argument("--connector-registry", nargs="?", const="") plan.add_argument("--preferred-provider-id") add_subcommand_format(plan) + execute = actions.add_parser("execute", help="Execute the exact public GitHub plan without admitting evidence.") + execute.add_argument("--plan-json", required=True) + execute.add_argument("--execute", action="store_true", help="Authorize bounded anonymous public-source reads.") + add_subcommand_format(execute) + + project = actions.add_parser("readback", help="Show receipt, parent decision and actual research-ledger coverage.") + project.add_argument("--plan-json", required=True) + project.add_argument("--receipt-json", required=True) + project.add_argument("--admission-json") + project.add_argument("--project", help="Existing deepresearch project; no implicit run is created.") + project.add_argument("--execute", action="store_true", help="Write explicitly admitted sources to the existing ledger.") + add_subcommand_format(project) + receipt = actions.add_parser( "receipt", help="Validate and bind a caller-presented provider receipt to its exact plan.", @@ -194,10 +222,14 @@ def handle_external_evidence_command( {"providers": providers}, ) elif args.external_evidence_action == "plan": - inventory = _load_object( - args.provider_inventory_json, - label="external evidence provider inventory", - ) + if not args.provider_inventory_json and not getattr(args, "public_github", False): + raise ValueError("plan requires provider inventory or explicit --public-github") + inventory = _load_object(args.provider_inventory_json, + label="external evidence provider inventory") if args.provider_inventory_json else {"providers": []} + if getattr(args, "public_github", False): + if len(getattr(args, "source", [])) > 8: + raise ValueError("public GitHub plan allows at most eight sources") + inventory["providers"] = _merge_providers(_provider_inventory(inventory), [inspect_provider(getattr(args, "source", []))]) providers = _merge_providers( _connector_inventory(args.connector_registry), _provider_inventory(inventory), @@ -211,11 +243,31 @@ def handle_external_evidence_command( "decision": args.decision, "evidence_kinds": args.evidence_kind, "constraints": args.constraint, + **({"source_refs": getattr(args, "source", [])} if getattr(args, "source", []) else {}), + **({"search_terms": getattr(args, "search_term", [])} if getattr(args, "search_term", []) else {}), }, "providers": providers, "preferred_provider_id": args.preferred_provider_id, }, ) + elif args.external_evidence_action == "execute": + if not args.execute: + raise ValueError("--execute is required for real public-source reads") + plan = _load_object(args.plan_json, label="external evidence plan") + # Canonical identity must pass the typed owner before any HTTP call. + probe_receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan.get("plan_id"), "request_id": plan.get("request", {}).get("request_id"), + "provider_id": plan.get("selected_provider", {}).get("provider_id"), + "provider_kind": plan.get("selected_provider", {}).get("provider_kind"), + "status": "failed", "sources": [], "summary": "Validation only", "completed_at": "not-executed"} + validate_receipt(plan, probe_receipt) + payload = execute_public_github(plan) + payload["observation"] = validate_receipt(plan, payload["receipt"]) + elif args.external_evidence_action == "readback": + payload = readback(_load_object(args.plan_json, label="external evidence plan"), + _load_receipt(args.receipt_json), + _load_object(args.admission_json, label="external evidence admission") if args.admission_json else None, + project=Path(args.project).expanduser() if args.project else None, execute=args.execute) elif args.external_evidence_action == "receipt": payload = effect_runtime_result( "external_evidence.receipt", @@ -223,9 +275,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), }, ) elif args.external_evidence_action == "admit": @@ -235,9 +285,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), "decision": { "disposition": args.decision, "reason": args.reason, diff --git a/loopx/capabilities/external_research/projection.py b/loopx/capabilities/external_research/projection.py new file mode 100644 index 0000000000..3d01582468 --- /dev/null +++ b/loopx/capabilities/external_research/projection.py @@ -0,0 +1,111 @@ +"""Shared external evidence readback for CLI and existing conversation surfaces. + +Typed effect-runtime reductions remain the sole admission/retirement authority. +Deep-research owns its existing durable source ledger; no parallel evidence store. +""" +from __future__ import annotations + +import re +from pathlib import Path + +from ...control_plane.effect_runtime import effect_runtime_result +from ..deep_research.runtime import add_source, load_state + + +def validate_receipt(plan: dict, receipt: dict) -> dict: + return effect_runtime_result("external_evidence.receipt", {"plan": plan, "receipt": receipt}) + + +def _validate_admission(plan: dict, receipt: dict, admission: dict) -> dict: + normalized = effect_runtime_result("external_evidence.admit", {"plan": plan, "receipt": receipt, + "decision": {"disposition": admission.get("disposition"), "reason": admission.get("reason"), + "admitted_source_refs": admission.get("admitted_source_refs", [])}}) + if normalized != admission: + raise ValueError("admission does not match its exact plan and receipt") + return normalized + + +def _lineage(admission: dict, source: dict) -> dict: + return {"plan_id": admission["plan_id"], "admission_id": admission["admission_id"], + "receipt_digest": admission["receipt_digest"], "content_digest": source["content_digest"]} + + +def readback(plan: dict, receipt: dict, admission: dict | None = None, + *, project: Path | None = None, execute: bool = False) -> dict: + observation = validate_receipt(plan, receipt) + receipt = observation["receipt"] + if execute and (admission is None or project is None): + raise ValueError("downstream write requires an explicit parent admission and research project") + normalized = _validate_admission(plan, receipt, admission) if admission is not None else None + write_blockers = [] + if execute and normalized["disposition"] == "admit": + for source in normalized["downstream_projection"]["sources"]: + try: + add_source(project, url_or_path=source["source_ref"], tool="external-evidence", + title="Public evidence", claims=[{"text": source["finding"], "stance": "neutral"}], + external_evidence=_lineage(normalized, source), expected_question=plan["request"]["objective"]) + except ValueError as error: + # A bounded partial projection is retained. Retrying is idempotent + # for the same identity; failed sources never count as covered. + write_blockers.append(str(error)) + break + covered = [] + if project is not None and normalized is not None: + state = load_state(project) + if state is not None and state["question"] == plan["request"]["objective"]: + for source in normalized["downstream_projection"]["sources"]: + for row in state["sources"]: + if (row["url_or_path"] == source["source_ref"] and + row.get("external_evidence") == _lineage(normalized, source) and + any(claim["id"] in row["claims"] and claim["text"] == source["finding"] + for claim in state["claims"])): + covered.append(source["source_ref"]) + break + retirement = effect_runtime_result("external_evidence.retire", {"admission": normalized, + "downstream_source_refs": covered}) if normalized is not None else None + return {"schema_version": "loopx_external_evidence_readback_v0", "plan_id": plan["plan_id"], + "request": plan["request"], "provider_id": receipt["provider_id"], "receipt_status": receipt["status"], + "sources": receipt["sources"], "summary": receipt["summary"], "limitations": receipt["limitations"], + "parent_admission": None if normalized is None else {"disposition": normalized["disposition"], + "reason": normalized["reason"], "admission_id": normalized["admission_id"], + "admitted_source_refs": normalized["admitted_source_refs"]}, + "downstream_source_refs": covered, "retirement": retirement, "write_blockers": write_blockers, + "original_source_fallback_allowed": True, "automatic_admission": False, + "evidence_coverage_observed": False} + + +def _text(value: object) -> str: + return re.sub(r"([\\`*_{}\[\]()<>#!|])", r"\\\1", str(value)).replace("\n", " ") + + +def render_readback(payload: dict) -> str: + admission = payload["parent_admission"] + retirement = payload["retirement"] + lines = ["## External evidence / 外部证据", "", _text(payload["request"]["objective"]), "", + "- Provider / 来源:" + _text(payload["provider_id"]), + "- Receipt / 读取结果:" + payload["receipt_status"], + "- Parent decision / 父 Agent 决定:" + ("pending / 待采纳" if admission is None else _text(admission["disposition"])), + f"- Downstream coverage / 下游覆盖:{len(payload['downstream_source_refs'])} admitted sources / 已采纳来源", + "- Retirement / 退休:" + ("pending / 待决定" if retirement is None else retirement["status"]), + "- Original-source fallback remains available / 可继续使用原始来源。", + "- Evidence completeness is unverified / 证据完整性尚未验证。", ""] + if admission is not None: + lines.extend(["Parent reason / 采纳理由:" + _text(admission["reason"]), ""]) + for ref in payload["request"].get("source_refs", []): + lines.extend(["Original source / 原始来源:" + _text(ref), ""]) + admitted = set(admission["admitted_source_refs"]) if admission is not None else set() + covered = set(payload["downstream_source_refs"]) + for index, source in enumerate(payload["sources"], 1): + ref = source["source_ref"] + lines.extend([f"### Source {index} / 来源 {index}", "", _text(ref), "", + _text(source["finding"]), "", + "- Evidence basis / 依据:" + _text(source["basis"]), + "- Accessed / 读取时间:" + _text(source["accessed_at"]), + "- Digest / 摘要:" + _text(source["content_digest"]), + "- Admission / 采纳:" + ("admitted" if ref in admitted else "not admitted"), + "- Downstream / 下游:" + ("observed" if ref in covered else "not observed"), + "- Limitation / 限制:" + _text(source.get("limitation") or "unspecified"), ""]) + for limitation in [*payload["limitations"], *payload["write_blockers"]]: + lines.append("- " + _text(limitation)) + lines.extend(["", "Plan identity / 计划身份:" + _text(payload["plan_id"])]) + return "\n".join(lines) + "\n" diff --git a/loopx/control_plane/capabilities/external_evidence.ts b/loopx/control_plane/capabilities/external_evidence.ts index 4dcf8ed30b..260f0e8246 100644 --- a/loopx/control_plane/capabilities/external_evidence.ts +++ b/loopx/control_plane/capabilities/external_evidence.ts @@ -92,6 +92,19 @@ function normalizeRequest(value: unknown): JsonObject { (normalized.evidence_kinds as string[]).length > 0, "request.evidence_kinds must not be empty", ); + // Optional source selection is part of the exact request identity. Preserve + // legacy request digests when no source-bound provider is requested. + if (request.source_refs !== undefined) { + const refs = boundedStrings(request.source_refs, "request.source_refs", 8, 2048); + requireThat(refs.length > 0 && new Set(refs).size === refs.length, + "request.source_refs must be nonempty and unique"); + requireThat(refs.every((ref) => SOURCE_REF_RE.test(ref) && !ref.startsWith("file://")), + "request.source_refs must be non-file provenance URIs"); + normalized.source_refs = refs; + } + if (request.search_terms !== undefined) { + normalized.search_terms = boundedStrings(request.search_terms, "request.search_terms", 8, 256); + } normalized.request_id = digest(normalized); return normalized; } diff --git a/loopx/extensions/public_github_research.py b/loopx/extensions/public_github_research.py new file mode 100644 index 0000000000..6903c9d32e --- /dev/null +++ b/loopx/extensions/public_github_research.py @@ -0,0 +1,131 @@ +"""Explicit public GitHub source-inspection method; no credentials or raw persistence. + +This bundled provider owns HTTP access. External evidence's TypeScript contract +owns exact-plan validation, parent admission and retirement. +""" +from __future__ import annotations + +import hashlib +import json +import re +from datetime import datetime, timezone +from urllib.error import HTTPError, URLError +from urllib.parse import quote, unquote, urlsplit +from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener + +PROVIDER_ID = "method:public-github" +MAX_SOURCE_BYTES = 1_000_000 +SOURCE = re.compile(r"/([A-Za-z0-9_.-]+)/([A-Za-z0-9_.-]+)/blob/([0-9a-f]{40})/(.+)") + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _read(url: str) -> bytes: + # No provider credentials, cookies, user-configured proxy credentials or + # caller-selected origins enter the request. Redirects fail closed. + request = Request(url, headers={"User-Agent": "LoopX-public-evidence", + "Accept": "application/vnd.github+json" if url.startswith("https://api.github.com/") else "text/plain"}) + with build_opener(ProxyHandler({}), _NoRedirect()).open(request, timeout=15) as response: + result = response.read(MAX_SOURCE_BYTES + 1) + if len(result) > MAX_SOURCE_BYTES: + raise ValueError("public source exceeds the bounded read limit") + return result + + +def _source(ref: str) -> tuple[str, str, str, str]: + parsed = urlsplit(ref) + match = SOURCE.fullmatch(parsed.path) + if (parsed.scheme != "https" or parsed.netloc != "github.com" or parsed.query + or parsed.fragment or match is None): + raise ValueError("public GitHub sources require https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/PATH") + owner, repo, revision, path = match.groups() + decoded = unquote(path) + if (any(part in {"", ".", ".."} for part in decoded.split("/")) or "\\" in decoded + or any(ord(char) < 32 for char in decoded) or "%" in decoded + or owner in {".", ".."} or repo in {".", ".."}): + raise ValueError("public GitHub source path is invalid") + return owner, repo, revision, decoded + + +def _public_repository(owner: str, repo: str) -> None: + value = json.loads(_read(f"https://api.github.com/repos/{owner}/{repo}")) + if not isinstance(value, dict) or value.get("private") is not False: + raise ValueError("provider only reads currently public GitHub repositories") + + +def inspect_provider(source_refs: list[str]) -> dict: + """Current opt-in readiness, not registry inventory or execution proof.""" + sources = [_source(ref) for ref in source_refs] + if not sources: + raise ValueError("public GitHub inspection requires at least one pinned source") + reason = None + try: + for owner, repo in sorted({(source[0], source[1]) for source in sources}): + _public_repository(owner, repo) + except (OSError, ValueError, URLError): + reason = "public_github_readiness_unavailable" + return {"provider_id": PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, + "installed": True, "enabled": True, "ready": reason is None, + "unavailable_reason": reason} + + +def execute_public_github(plan: dict) -> dict: + """Read only exact-plan sources, with fresh public-visibility checks. + + Caller validates the canonical plan through the typed owner before entry. + Findings are retrieval and literal-match facts, not autonomous conclusions. + """ + selected = plan["selected_provider"] + if selected["provider_id"] != PROVIDER_ID or selected["provider_kind"] != "method": + raise ValueError("this executor requires the selected public GitHub method") + request = plan["request"] + refs = request.get("source_refs", []) + sources = [_source(ref) for ref in refs] + if not sources or len(sources) > 8: + raise ValueError("public GitHub execution requires one to eight pinned sources") + terms = request.get("search_terms", []) + records, failures = [], [] + now = datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + for index, (ref, (owner, repo, revision, path)) in enumerate(zip(refs, sources), 1): + try: + _public_repository(owner, repo) + raw = _read(f"https://raw.githubusercontent.com/{owner}/{repo}/{revision}/{quote(path, safe='/')}") + text = raw.decode("utf-8") + if "\x00" in text: + raise ValueError("source is not UTF-8 text") + if not text.strip(): + failures.append(f"Source {index}: empty source") + continue + matches = [] + for term in terms: + lines = [str(index) for index, line in enumerate(text.splitlines(), 1) if term in line] + matches.append(f"Literal term {json.dumps(term)}: " + + ("observed at lines " + ", ".join(lines[:16]) if lines else "not observed") + + (" (additional matches omitted)" if len(lines) > 16 else "")) + finding = f"Read pinned file {path}; {len(raw)} UTF-8 bytes. " + " ".join(matches) + if len(finding) > 4096: + finding = finding[:4040] + " (additional match metadata omitted)" + records.append({"source_ref": ref, "source_family": "github_repository_file", + "basis": "observed", "finding": finding, + "limitation": "Retrieval/literal matches only; no semantic conclusion, execution test or completeness claim.", + "publication_date": None, "accessed_at": now, + "content_digest": "sha256:" + hashlib.sha256(raw).hexdigest()}) + except HTTPError as error: + failures.append(f"Source {index}: HTTP {error.code}") + except (OSError, ValueError, UnicodeError, URLError): + failures.append(f"Source {index}: source read unavailable") + status = "succeeded" if records else "no_evidence" if all("empty source" in item for item in failures) else "failed" + receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan["plan_id"], "request_id": request["request_id"], + "provider_id": PROVIDER_ID, "provider_kind": "method", "status": status, + "sources": records, "summary": f"Read {len(records)} of {len(refs)} requested public sources.", + "limitations": ["Original-source fallback remains available; parent admission is required.", *failures], + "completed_at": now} + return {"receipt": receipt, "execution": {"provider_id": PROVIDER_ID, + "source_reads_observed": len(records), "requested_source_count": len(refs), + "raw_content_persisted": False, "credentials_used": False, + "automatic_admission": False, "evidence_coverage_observed": False}} diff --git a/tests/architecture/test_content_digest_single_owner.py b/tests/architecture/test_content_digest_single_owner.py index d3c40bf38f..c454d42233 100644 --- a/tests/architecture/test_content_digest_single_owner.py +++ b/tests/architecture/test_content_digest_single_owner.py @@ -161,6 +161,7 @@ "loopx.capabilities.benchmark_toolkit.runtime_continuity", "loopx.capabilities.benchmark_toolkit.study_projection", "loopx.capabilities.content_ops.item_lifecycle", + "loopx.capabilities.deep_research.runtime", "loopx.capabilities.issue_fix.outcome_projection", "loopx.capabilities.issue_fix.reviewer_notification", "loopx.capabilities.machine_configuration.store", diff --git a/tests/capabilities/test_public_github_evidence.py b/tests/capabilities/test_public_github_evidence.py new file mode 100644 index 0000000000..b490902269 --- /dev/null +++ b/tests/capabilities/test_public_github_evidence.py @@ -0,0 +1,210 @@ +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from urllib.error import HTTPError + +import pytest + +from loopx.capabilities.external_research import cli +from loopx.capabilities.external_research.projection import readback, render_readback +from loopx.capabilities.deep_research.runtime import add_source, load_state, start_research +from loopx.control_plane.effect_runtime import effect_runtime_result +from loopx.extensions import public_github_research as provider + +REF = "https://github.com/example/public/blob/" + "a" * 40 + "/README.md" +SECOND = REF.replace("README.md", "missing.md") + + +def plan(refs=None): + return effect_runtime_result("external_evidence.plan", {"request": { + "objective": "Inspect public fixture", "user_activity": "Choose a source", + "decision": "Whether the pinned source contains the fixture marker", + "evidence_kinds": ["literal_match"], "source_refs": refs or [REF], "search_terms": ["fixture"]}, + "providers": [{"provider_id": provider.PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, "installed": True, + "enabled": True, "ready": True, "unavailable_reason": None}]}) + + +def reader(url): + if url.startswith("https://api.github.com/"): + return b'{"private": false}' + if url.endswith("missing.md"): + raise HTTPError(url, 404, "not found", {}, None) + return b"fixture public data\n" + + +def admission(p, receipt, disposition="admit"): + return effect_runtime_result("external_evidence.admit", {"plan": p, "receipt": receipt, + "decision": {"disposition": disposition, "reason": "Fixture parent decision", + "admitted_source_refs": [s["source_ref"] for s in receipt["sources"]] if disposition == "admit" else []}}) + + +@pytest.mark.parametrize("ref", ["file:///private", REF.replace("https://", "http://"), + REF.replace("github.com", "github.com.evil"), REF.replace("/" + "a"*40 + "/", "/main/"), + REF + "?query=fixture", REF + "#L1", REF.replace("README.md", "../private"), + REF.replace("README.md", "%2e%2e/private"), REF.replace("README.md", "%252e%252e/private")]) +def test_provider_rejects_unpinned_or_out_of_scope_sources(ref, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: pytest.fail("invalid input made a HTTP call")) + with pytest.raises(ValueError): + provider.inspect_provider([ref]) + + +def test_fresh_visibility_probe_does_not_trust_old_ready_plan(monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: b'{"private": true}') + assert provider.inspect_provider([REF])["ready"] is False + output = provider.execute_public_github(plan()) + assert output["receipt"]["status"] == "failed" + assert output["receipt"]["sources"] == [] + assert output["execution"]["automatic_admission"] is False + + +@pytest.mark.parametrize("content,status", [(b"", "no_evidence"), (b"\x00binary", "failed")]) +def test_empty_or_binary_source_preserves_fallback(monkeypatch, content, status): + monkeypatch.setattr(provider, "_read", lambda url: b'{"private":false}' if "api.github.com" in url else content) + p = plan() + output = provider.execute_public_github(p) + assert output["receipt"]["status"] == status + projected = readback(p, output["receipt"]) + assert projected["original_source_fallback_allowed"] is True + assert projected["parent_admission"] is None + assert REF in render_readback(projected) + + +def test_partial_reads_are_observed_not_admitted_or_complete(monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan([REF, SECOND]) + output = provider.execute_public_github(p) + receipt = output["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + source = receipt["sources"][0] + assert source["basis"] == "observed" + assert source["content_digest"] == "sha256:" + hashlib.sha256(b"fixture public data\n").hexdigest() + assert "lines 1" in source["finding"] + assert "fixture public data" not in json.dumps(output) + before = readback(p, receipt) + assert before["parent_admission"] is None and before["downstream_source_refs"] == [] + assert before["evidence_coverage_observed"] is False + assert any("HTTP 404" in item for item in before["limitations"]) + rejected = readback(p, receipt, admission(p, receipt, "reject")) + assert rejected["retirement"]["reason"] == "parent_rejected" + + +@pytest.mark.parametrize("mutate", ["source_refs", "search_terms", "objective"]) +def test_execute_rejects_mutated_plan_before_provider_calls(tmp_path, monkeypatch, mutate): + p = plan() + p["request"][mutate] = [SECOND] if mutate == "source_refs" else ["different"] if mutate == "search_terms" else "different" + path = tmp_path / "plan.json" + path.write_text(json.dumps(p), encoding="utf-8") + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("mutated plan reached provider")) + payloads = [] + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(path), execute=True) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda payload, *_: payloads.append(payload)) == 1 + assert payloads[0]["status"] == "invalid_request" + + +def test_execution_requires_explicit_opt_in(tmp_path, monkeypatch): + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("default-off execution called provider")) + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(tmp_path / "absent.json"), execute=False) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda *_: None) == 1 + + +def test_parent_admission_and_real_ledger_coverage_are_independent(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + retained = readback(p, receipt, accepted, project=tmp_path) + assert retained["retirement"]["status"] == "retained" + projected = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert projected["downstream_source_refs"] == [REF] + assert projected["retirement"]["status"] == "retire_ready" + replay = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert replay == projected + assert len(load_state(tmp_path)["sources"]) == 1 + assert REF in render_readback(replay) and "admitted" in render_readback(replay) + corrupt = copy.deepcopy(accepted) + corrupt["downstream_projection"]["sources"][0]["finding"] = "mutated" + with pytest.raises(ValueError, match="exact plan"): + readback(p, receipt, corrupt, project=tmp_path, execute=True) + + +def test_wrong_question_or_unrelated_source_never_proves_coverage(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question="Unrelated question", max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["write_blockers"] and result["retirement"]["status"] == "retained" + assert not load_state(tmp_path)["sources"] + other = tmp_path / "other" + start_research(other, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + add_source(other, url_or_path=REF, tool="manual", title=None, claims=[{"text":"Unrelated observation"}]) + result = readback(p, receipt, accepted, project=other, execute=True) + assert result["write_blockers"] and result["downstream_source_refs"] == [] + + +def test_partial_downstream_projection_retains_until_all_sources_read_back(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda url: reader(url.replace("second.md", "README.md"))) + p = plan([REF, REF.replace("README.md", "second.md")]) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=1, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == [REF] + assert result["retirement"]["status"] == "retained" + assert len(result["retirement"]["missing_downstream_source_refs"]) == 1 + assert result["write_blockers"] and result["original_source_fallback_allowed"] + + +def test_existing_lark_sink_preserves_shared_readback_facts(tmp_path, monkeypatch): + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + card = build_lark_markdown_reply_card(render_readback(result)) + content = card["elements"][0]["text"]["content"] + assert "retire_ready" in content and REF in content + assert "Evidence completeness is unverified" in content + assert result["plan_id"] in content + assert "truncated" not in content + + +def test_long_readback_uses_existing_lossless_lark_transport(monkeypatch): + from loopx.extensions.lark.outbound import split_lark_outbound_text + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + receipt["sources"][0]["finding"] = "Bounded public finding. " * 150 + result = readback(p, receipt, admission(p, receipt)) + markdown = render_readback(result) + parts = split_lark_outbound_text(markdown, limit=3000, preserve_format=True) + cards = [build_lark_markdown_reply_card(part) for part in parts] + assert len(cards) > 1 + body = "\n".join(card["elements"][0]["text"]["content"] for card in cards) + assert result["plan_id"] in body and REF in body + assert "Evidence completeness is unverified" in body and "truncated" not in body + + +def test_pinned_paths_keep_case_sensitive_identity(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + refs = [REF, REF.replace("README.md", "readme.md")] + p = plan(refs) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == refs + assert result["retirement"]["retire_ready"] is True diff --git a/tests/control_plane_ts/external_evidence_research.test.ts b/tests/control_plane_ts/external_evidence_research.test.ts index f1a0993893..380672b780 100644 --- a/tests/control_plane_ts/external_evidence_research.test.ts +++ b/tests/control_plane_ts/external_evidence_research.test.ts @@ -392,3 +392,19 @@ test("retirement fails closed on mutated admission semantics", () => { /admission_id does not match/, ); }); + + +test("optional source selection preserves legacy identity and binds source/query mutations", () => { + const legacy = plan(); + assert.equal((legacy.request as Record).source_refs, undefined); + const bound = planExternalEvidenceRequest({request: {...request, + source_refs: ["https://example.com/pinned"], search_terms: ["literal"]}, providers:[methodProvider]}); + assert.notEqual(bound.plan_id, legacy.plan_id); + for (const [field, value] of [["source_refs", ["https://example.com/other"]], ["search_terms", ["different"]]]) { + const changed = structuredClone(bound); + (changed.request as Record)[field as string] = value; + assert.throws(() => recordExternalEvidenceReceiptObservation({plan:changed, receipt:receipt(bound)}), /request_id|plan_id/); + } + assert.throws(() => planExternalEvidenceRequest({request:{...request, source_refs:["file:///private"]}, + providers:[methodProvider]}), /non-file/); +});