From 196986b5b46fb75a7f1fc414a42cd74918b23f7e Mon Sep 17 00:00:00 2001 From: wefio <48851810+wefio@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:02:15 +0800 Subject: [PATCH 1/2] fix(verification): make tests-surface type checking blocking Repair 79 errors in the existing test dependency graph without relaxing compiler flags or excluding fixtures. Keep JavaScript hook exports typed and presentation fixtures complete. Run check:tests once in the shared static contract and ci-and-tests route; remove the duplicate advisory CI step. Pin the contract with a red-green test and update the owning bilingual decision and CI design. Validation: targeted 56/56; agent:verify passed all 15 blocking checks. The research embedding timeout remains an unverified external limitation. --- .github/workflows/ci.yml | 20 ----- agent-context.yaml | 1 + .../2026-09-16-ci-static-coverage.md | 43 +++++------ .../2026-09-16-ci-static-coverage.zh-CN.md | 30 ++++---- docs/design/ci-cd-and-quality.md | 4 +- evals/omnimemeval/bridge.ts | 2 +- evals/topology/namesakes-agent.ts | 6 +- kimi-plugin/nmg-hook.d.mts | 35 +++++++++ package.json | 2 +- src/core/store/row-parse.ts | 2 +- tests/cli/daemon-client.test.ts | 5 +- tests/cli/service.test.ts | 16 ++-- tests/cli/task-run-surface.test.ts | 1 + tests/core/multi-hop-path.test.ts | 11 ++- tests/core/router-session.test.ts | 1 + tests/core/shadow-evaluation.test.ts | 2 + tests/core/task-board-retention.test.ts | 1 - .../evals/controller-shadow/calibrate.test.ts | 3 +- tests/evals/controller-shadow/dataset.test.ts | 1 + tests/evals/controller-shadow/report.test.ts | 1 + tests/evals/omnimemeval-runner.test.ts | 7 +- tests/evals/skillopt/dataset.test.ts | 1 + tests/extensions/nmg/index.test.ts | 4 +- tests/helpers/search-result.ts | 71 +++++++++++++++++ tests/integration/agent-surface.test.ts | 34 +++++---- tests/integration/chain-projection.test.ts | 16 +--- .../ooo-acceptance-one-predicate.test.ts | 4 +- tests/integration/ooo-advisers.test.ts | 1 - .../integration/ooo-ordinary-failure.test.ts | 4 +- .../integration/ooo-ordinary-handoff.test.ts | 2 +- .../ooo-publication-invariants.test.ts | 1 + .../ooo-session-chain-contract.test.ts | 2 +- .../integration/task-semantics-cases.test.ts | 2 +- tests/support/cordis-adapter.test.ts | 4 +- tests/support/cordis-adapter.ts | 2 +- tests/tools/agent-verify.test.ts | 2 + tests/tools/mutation-anchor.test.ts | 5 +- tests/tools/test-groups.test.ts | 23 +++++- tools/complexity-gate.ts | 2 +- tools/mutation-anchor.ts | 8 +- workbuddy-plugin/nmg-hook.ts | 76 ++++++++++++++----- 41 files changed, 305 insertions(+), 153 deletions(-) create mode 100644 kimi-plugin/nmg-hook.d.mts create mode 100644 tests/helpers/search-result.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7d65216f..316b3d24 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -36,26 +36,6 @@ jobs: run: npm run ci:uncovered-tests - name: Shared static verification contract run: npm run verify:static - # Advisory, and deliberately not part of the blocking set. `verify:static` runs the - # widened `lint`, which holds tests/, evals/, scripts/ and tools/ to the same rules - # as src/ except the two liveness rules (unused code, dead stores): a static "this - # value is never read" judgement is the one most often wrong about live code, so on - # the developer surfaces those two report as warnings instead of failing the build. - # What no blocking route does at all is type-check the tests surface, so - # `check:tests` runs it with unused locals and parameters enabled and is expected to - # fail while its pre-existing type errors are paid down. - # The failure is reported, not enforced: the step runs and fails, tsc emits one - # error annotation per finding, and its exit code is in this job's log. The checks - # list still says "All checks passed", because the remaining debt is deliberately - # not a merge blocker. A track that cannot be told apart from "never ran" is the - # failure mode rejected in - # docs/decisions/implemented/2026-09-12-visible-non-blocking-research-track.md, and - # `continue-on-error` on a step (not on the job) keeps the aggregate green while - # the annotations carry the detail. - # See docs/decisions/implemented/2026-09-16-ci-static-coverage.md. - - name: Advisory tests-surface type check (non-blocking) - continue-on-error: true - run: npm run check:tests - name: Dependency audit run: npm audit --production --audit-level=high diff --git a/agent-context.yaml b/agent-context.yaml index d0eaa0cd..b730e723 100644 --- a/agent-context.yaml +++ b/agent-context.yaml @@ -143,6 +143,7 @@ routes: - build - package:check - check + - check:tests - mutation:anchors - check:lock - lint diff --git a/docs/decisions/implemented/2026-09-16-ci-static-coverage.md b/docs/decisions/implemented/2026-09-16-ci-static-coverage.md index aeb1fcb7..72431cea 100644 --- a/docs/decisions/implemented/2026-09-16-ci-static-coverage.md +++ b/docs/decisions/implemented/2026-09-16-ci-static-coverage.md @@ -58,18 +58,20 @@ matches a file that the script scans. It imports `eslint.config.js` and asks ESL Its first check is red on the pre-change config, where `evals/**/*.ts` and `scripts/**/*.ts` pointed outside the scanned surface. -**Static checking and the unused-code scan run in CI, and the debt does not block.** The -required `static` job runs the widened `lint`: liveness findings appear as warnings and -as annotations on the run, and never fail the build, while everything else is held to the -same standard as `src/`. One advisory step, `npm run check:tests`, runs -`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters` — the first reference -to `tsconfig.tests.json` — because no blocking route type-checks `tests/`. It is expected -to fail while its pre-existing type errors are paid down; `continue-on-error` sits on the -step, not on the job, so its findings appear as annotations and its exit code is in the -job log while the required aggregate stays green. A job-level `continue-on-error` remains -rejected by [2026-09-12](2026-09-12-visible-non-blocking-research-track.md): there the -objection was a job whose real result could not be told apart from one that never ran, so -the detail is emitted as annotations rather than swallowed. +**Tests-surface type checking blocks, while ESLint liveness findings remain advisory.** +`check:tests` runs `tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters` +once in `verify:static`; `ci-and-tests` lists the same atomic checks. CI invokes the +shared contract without a duplicate advisory type-check step. The checked surface is +the configuration's explicit includes plus their imported dependencies, not all of +`evals/`, `scripts/` or `tools/`. The required `static` job still reports the two ESLint +liveness rules as warnings on developer surfaces. This does not weaken TypeScript's +strict type or unused checks. + +The 2026-09-25 follow-up measured 79 type errors and repaired them without exclusions, +`@ts-ignore`, or relaxed compiler flags. Fixtures supply current contract fields and +narrow discriminated results; the Kimi hook's JavaScript exports have a declaration +boundary, and presentation fixtures use complete retrieval records. The blocking +contract test pins one `check:tests` invocation and rejects a duplicate advisory step. ### What the first widened scan reported @@ -132,6 +134,10 @@ evidence behind the advisory severity above: - **Guard the config by reading it as text.** Rejected: comments and formatting would decide the result, which is what the ticket means by "a comment must not be able to fail". +- **Leave tests-surface type checking permanently advisory.** Rejected once the measured + errors were repaired: it would permit the same fixture and API drift to accumulate again. + Casting malformed fixtures through `unknown` or excluding failing files is also rejected; + neither establishes that the test obeys the contract it exercises. - **Widen `format:check` in the same change.** Deferred, not rejected: it is the same class of hole, but it needs a Prettier pass over the newly covered directories, and a formatting rewrite of research code would bury the lint change it travels with. @@ -144,9 +150,8 @@ evidence behind the advisory severity above: - Unused imports, unused variables and dead stores are reported on every developer surface. Nine of them are gone with this change, and the next one is visible in the run as a warning annotation rather than as a merge blocker. -- The `tests/` surface is type-checked for the first time, by an advisory step whose - failure count is its debt. `tsconfig.tests.json` is now referenced by a script instead - of being an unread file. +- Type errors in the `tests/` dependency graph fail local static verification and the + required CI static job. The compiler flags and inclusion surface remain unchanged. - Cost: an advisory finding that nobody reads is not a gate. The counter-pressure is that the findings are annotated on the diff and counted in the run, and the guard test keeps the severities themselves from drifting silently. @@ -161,10 +166,4 @@ evidence behind the advisory severity above: names in `tsconfig.json`. `check:tests` is the first slice; the product surface with the same flags reports one error, so the same treatment is available for a later change. -- `check:tests` reports 69 pre-existing type errors (37 under `tests/`, 26 in - `workbuddy-plugin/nmg-hook.ts`, 5 under `evals/`, 1 under `tools/`). None of them is a - liveness finding; paying them down and moving the step into `verify:static` is its own - change. The count is a property of the tree, not of the step: the tree that merges - `feat/ooo-run-namespace` measures 74 (41 under `tests/`, everything else unchanged), the - four extra findings being that branch's own test files. Whoever merges it re-measures this - line rather than carrying 69 forward. + diff --git a/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md b/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md index 9bf99c56..7f502316 100644 --- a/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md +++ b/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md @@ -49,15 +49,16 @@ severity。 `calculateConfigForFile` 询问生效的 severity,所以注释无法让它通过。它的第一条断言在改动前的 配置上就是红的 —— 当时 `evals/**/*.ts` 与 `scripts/**/*.ts` 指向扫描面之外。 -**静态检查与未使用代码扫描在 CI 里运行,而债不阻塞合并。** 必跑的 `static` job 运行拓宽后的 -`lint`:活性发现以警告形式出现、并作为运行里的 annotation,不失败构建;其余规则与 `src/` 同标准。 -另有一个 advisory 步骤 `npm run check:tests`,运行 -`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters`(`tsconfig.tests.json` 第一次 -被脚本引用),因为没有任何阻塞轨道对 `tests/` 做类型检查。在既有类型错误还清之前它预期失败; -`continue-on-error` 加在**步骤**而非 job 上,所以它的发现以 annotation 出现、退出码留在 job 日志里, -而必跑聚合仍然为绿。job 级 `continue-on-error` 仍被 -[2026-09-12](2026-09-12-visible-non-blocking-research-track.md) 否决:那里的反对理由是 job 的真实 -结果与"从未运行过"无法区分,因此这里把细节以 annotation 形式发出,而不是把步骤吞掉。 +**测试面的类型检查阻塞,ESLint 活性发现仍只报告。** `check:tests` 运行 +`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters`,在 `verify:static` +里执行一次;`ci-and-tests` 声明相同的原子检查。CI 调用共享契约,不重复运行 advisory 类型检查步骤。 +覆盖面是配置显式包含的文件及其导入依赖,不等于全部 `evals/`、`scripts/`、`tools/`。 +必跑的 `static` job 仍把开发面上的两条 ESLint 活性规则报为警告;这不放松 TypeScript 的 strict +类型检查或 unused 检查。 + +2026-09-25 的后续清理测得 79 条类型错误,并在不排除文件、不加 `@ts-ignore`、不放松编译器 flag +的前提下修复。fixture 补齐当前契约字段并收窄判别结果;Kimi hook 的 JavaScript 导出有声明边界, +渲染 fixture 使用完整检索记录。阻塞契约测试锁定一次 `check:tests` 调用,并拒绝重复 advisory 步骤。 ### 首次拓宽扫描报出的 11 条 @@ -112,6 +113,9 @@ severity。 漏掉一个全局变量,就会让每个使用处长期被报成未定义。 - **把配置当文本读取来做守卫。** 否决:那样注释与排版就能决定结果,而工单要求的正是"注释不能让它 失败"。 +- **永久保留 advisory 类型检查。** 实测错误修复后否决:它会允许同样的 fixture 与 API 漂移 + 再次积累。把错误 fixture 经 `unknown` 强转或排除报错文件也被否决;它们不能证明测试遵守了 + 自己要检验的契约。 - **在同一次改动里拓宽 `format:check`。** 记为 Deferred 而非否决:它是同一类漏洞,但需要对新增 目录跑一次 Prettier,而研究代码的格式化重写会淹没与它同行的 lint 改动。 @@ -121,8 +125,8 @@ severity。 落在扫描面之外或匹配不到任何文件。 - 未使用的导入、未使用的变量与死存储会在每一个开发面上被报告。本次改动删掉了其中 9 条,下一条会 以警告 annotation 的形式出现在运行里,而不是变成一个合并阻塞项。 -- `tests/` 这条面第一次获得类型检查,形式是一个 advisory 步骤,其失败计数就是它的债。 - `tsconfig.tests.json` 从一个没人读的文件变成了被脚本引用的文件。 +- `tests/` 依赖图中的类型错误会使本地静态验证与必跑 CI static job 失败;编译器 flag 与 + include 面不变。 - 代价:一条没人读的 advisory 发现不是门禁。反作用力是这些发现会作为 annotation 挂在 diff 上、 在运行里被计数,而守卫测试让 severity 自身不会悄悄漂移。 - 回滚:revert 一个 commit。这里没有数据迁移,也没有运行时契约变更。 @@ -134,6 +138,4 @@ severity。 - 没有任何轨道对 `evals/`、`scripts/` 以及 `tsconfig.json` 里那三个文件名之外的 `tools/` 做类型 检查。`check:tests` 只是第一刀;产品面加同样 flag 只报 1 个错误,因此后续改动可以对它使用同样的 处理。 -- `check:tests` 报出 69 个既有类型错误(`tests/` 37 个、`workbuddy-plugin/nmg-hook.ts` 26 个、 - `evals/` 5 个、`tools/` 1 个)。其中没有一条是活性发现;把它们还清、并把该步骤移入 - `verify:static` 是另一次改动。 + diff --git a/docs/design/ci-cd-and-quality.md b/docs/design/ci-cd-and-quality.md index fad73648..73aa97ef 100644 --- a/docs/design/ci-cd-and-quality.md +++ b/docs/design/ci-cd-and-quality.md @@ -57,7 +57,7 @@ exit_criteria: Replace with a stable contract test or remove after the redesign `npm run lint` 的扫描面是 `src/ .pi/extensions/ claude-plugins/ workbuddy-plugin/ tests/ evals/ scripts/ tools/`。两条**活性规则**(`@typescript-eslint/no-unused-vars`、`no-useless-assignment`,都在断言"这个值从未被读取")在 `tests/`、`evals/`、`scripts/`、`tools/` 上只报告为 `warn`,在 `src/` 上仍为 error:静态工具最容易在这里把活代码判成死的(`tests/core/graph-cycles.test.ts` 的三个"未使用绑定"实际是漏掉的断言),阻塞门禁会把修法推向删掉线索。其余规则对开发面与 `src/` 同标准(决策:[静态覆盖面](../decisions/implemented/2026-09-16-ci-static-coverage.md))。测试、研究 harness、脚本与仓库工具按设计打印,因此 `no-console` 对这些面关闭。扫描面自身由 `tests/tools/eslint-config-coverage.test.ts` 守住:`lint` 与 `lint:fix` 同面,config 里每个以目录锚定的 `files:` 块都必须落在扫描面内并仍能匹配到文件,且生效 severity 由 ESLint 自己回答(提升某条规则是一次刻意改动)。 -`npm run check:tests`(`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters`,`tests/` 面第一次被类型检查)刻意不进入任何阻塞契约,只在 CI 以 advisory 步骤运行。 +`npm run check:tests`(`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters`)进入 `verify:static` 与 `ci-and-tests` 的阻塞集合。它检查 `tests/`、配置显式包含的源文件以及它们导入的依赖;不声明覆盖全部 `evals/`、`scripts/`、`tools/`。 `verify:static` 中的 `mutation:anchors`(`tools/mutation-teeth.ts --anchors-only`)只做一件事:把 110 颗具名 mutant 的位置全部解析一遍,不跑任何用例、不写任何字节,在 1 秒内回答“每一颗牙是否还瞄着东西”。它进入静态契约是因为**一颗锚点失效时没有别的检查会注意到**:全量 sweep 不跑(`mutation:teeth` 不在任何 CI 作业里),而一颗匹配不到位置的牙在 sweep 报告里只是“不可应用”并被排除出分母——本轮修掉的两颗牙就是这样悄无声息地停摆的。全量 sweep 仍然不进闸门:它是分钟级、要跑用例,属于推送前的常设规则([决策](../decisions/implemented/2026-09-24-mutants-are-derived-not-anchored.md))。 @@ -76,7 +76,7 @@ exit_criteria: Replace with a stable contract test or remove after the redesign `verify:chaos` 这些命名 package contract,使本地可复现入口和远程 CI 保持同源: - `static`:build、package、type、lint、format、文档、术语索引、需求追溯、Agent context 与生产依赖审计; -- `static` 内的 advisory 步骤:`check:tests`(`tests/` 面的类型检查 + 未使用局部量/参数)。`continue-on-error` 加在步骤上而非 job 上,`static` 仍是必跑且绿的 job;它的发现以 `error` annotation 与 job 日志里的退出码出现在运行里,刻意不是合并阻塞项(决策:[静态覆盖面](../decisions/implemented/2026-09-16-ci-static-coverage.md))。同一 job 里拓宽后的 `lint` 把开发面上的活性发现报为 `warn`,同样以 annotation 出现在运行里而不失败构建; +- `static` 通过 `verify:static` 阻塞执行 `check:tests`,不重复运行 advisory 类型检查步骤。ESLint 在开发面上的两条活性规则仍是 `warn`,与 TypeScript 的类型及 unused 检查分别执行(决策:[静态覆盖面](../decisions/implemented/2026-09-16-ci-static-coverage.md)); - `tests`:Node 24 产品测试和覆盖率; - `research-tests`:研究/benchmark adapter 表征,**非阻塞但可见**:它不进入 `all-checks-passed`,因此失败不阻塞合并;但它不再带 `continue-on-error`,所以失败会以红色 check 出现在 PR 上,而不是永远显示绿色(决策:`docs/decisions/implemented/2026-09-12-visible-non-blocking-research-track.md`)。 - `node-compat`:最低支持版本 Node 22.19 的 build/package; diff --git a/evals/omnimemeval/bridge.ts b/evals/omnimemeval/bridge.ts index 6f12901f..d63b7018 100644 --- a/evals/omnimemeval/bridge.ts +++ b/evals/omnimemeval/bridge.ts @@ -515,7 +515,7 @@ export class OmniMemEvalBridge { }, }, ); - const rankedMemoryIds = new Set(context.activeGraph.memoryIds); + const rankedMemoryIds = new Set(context.activeGraph?.memoryIds ?? []); const memories = context.results.map((result) => ({ memoryId: result.memory.id, nodeId: result.node.id, diff --git a/evals/topology/namesakes-agent.ts b/evals/topology/namesakes-agent.ts index 351da072..19495d98 100644 --- a/evals/topology/namesakes-agent.ts +++ b/evals/topology/namesakes-agent.ts @@ -5,7 +5,7 @@ import { pathToFileURL } from "node:url"; import { performance } from "node:perf_hooks"; import { RpcClient } from "@earendil-works/pi-coding-agent"; -import type { AgentSessionEvent } from "@earendil-works/pi-coding-agent"; +import type { JsonAgentSessionEvent } from "@earendil-works/pi-coding-agent"; import { cosineSimilarity, HashingVectorEmbedder } from "../../src/core/vector.ts"; import { benchmarkIsolationArgs, counterbalancedOrder } from "../benchmarks/matched.ts"; @@ -395,7 +395,7 @@ async function runArm( const started = performance.now(); let rawResponse = ""; let selectedIds: string[] = []; - let events: AgentSessionEvent[] = []; + let events: JsonAgentSessionEvent[] = []; let error: string | undefined; try { events = await client.promptAndWait(prompt, undefined, 300_000); @@ -497,7 +497,7 @@ function normalizeName(value: string): string { .replace(/[^a-z0-9]+/gu, ""); } -function collectTokenUsage(events: readonly AgentSessionEvent[]): TokenUsage | undefined { +function collectTokenUsage(events: readonly JsonAgentSessionEvent[]): TokenUsage | undefined { const total: TokenUsage = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }; let found = false; for (const event of events) { diff --git a/kimi-plugin/nmg-hook.d.mts b/kimi-plugin/nmg-hook.d.mts new file mode 100644 index 00000000..acf2ff88 --- /dev/null +++ b/kimi-plugin/nmg-hook.d.mts @@ -0,0 +1,35 @@ +interface HookPayload { + session_id?: string; + sessionId?: string; +} + +interface WakeEntry { + sourceSessionId?: string | null; + agentId?: string | null; + claimExpiresAt?: string | null; + to?: string | null; + serialState?: string | null; + status?: string; + kind: string; + content?: string; +} + +export function isBoardWakeCandidate( + entry: WakeEntry, + identity: { sessionId: string; agentId: string; now?: number }, +): boolean; + +export function kimiAgentIdentity( + payload: HookPayload, + environment?: NodeJS.ProcessEnv, +): { sessionId: string; agentId: string; capabilities: string | undefined }; + +export function reportAgentPresence( + payload: HookPayload, + options?: { + dataDir?: string; + lease?: { host: string; port: number; token: string; pid?: number } | null; + environment?: NodeJS.ProcessEnv; + rpc?: (method: string, params: Record) => Promise; + }, +): Promise; diff --git a/package.json b/package.json index 7b1f3f25..5481c1e2 100644 --- a/package.json +++ b/package.json @@ -130,7 +130,7 @@ "mutation:anchors": "node --experimental-strip-types tools/mutation-teeth.ts --anchors-only", "verify:packages": "node --experimental-strip-types tools/verify-packages.ts", "check:lock": "node --experimental-strip-types tools/check-lock.ts", - "verify:static": "npm run build && npm run package:check && npm run check && npm run mutation:anchors && npm run check:lock && npm run lint && npm run format:check && npm run docs:check && npm run agent:context:check && npm run complexity:gate && npm run verify:packages && npm run glossary:check && npm run rtm:check", + "verify:static": "npm run build && npm run package:check && npm run check && npm run check:tests && npm run mutation:anchors && npm run check:lock && npm run lint && npm run format:check && npm run docs:check && npm run agent:context:check && npm run complexity:gate && npm run verify:packages && npm run glossary:check && npm run rtm:check", "verify:product-ci": "npm run build && npm run test:coverage", "verify:research": "npm run prompts:generate && npm run test:research", "verify:node-compat": "npm run build && npm run check && npm run package:check", diff --git a/src/core/store/row-parse.ts b/src/core/store/row-parse.ts index 2fe54b06..c79743ce 100644 --- a/src/core/store/row-parse.ts +++ b/src/core/store/row-parse.ts @@ -7,7 +7,7 @@ * every query touching that table. */ -export function parseStringArray(value: string | number | Uint8Array | null): string[] { +export function parseStringArray(value: string | number | bigint | Uint8Array | null): string[] { if (typeof value !== "string") return []; try { const parsed = JSON.parse(value) as unknown; diff --git a/tests/cli/daemon-client.test.ts b/tests/cli/daemon-client.test.ts index e1b20077..a0518746 100644 --- a/tests/cli/daemon-client.test.ts +++ b/tests/cli/daemon-client.test.ts @@ -68,9 +68,8 @@ test("daemon protocol guard accepts the current compatibility epoch", () => { }); test("same-epoch capability additions do not affect protocol compatibility", () => { - assert.doesNotThrow(() => - assertDaemonProtocol({ protocol: NMG_PROTOCOL_VERSION, capabilities: ["future-feature"] }), - ); + const hello = { protocol: NMG_PROTOCOL_VERSION, capabilities: ["future-feature"] }; + assert.doesNotThrow(() => assertDaemonProtocol(hello)); }); test("daemon protocol guard fails closed with restart guidance", () => { diff --git a/tests/cli/service.test.ts b/tests/cli/service.test.ts index 84f0c14d..a2d3fbf4 100644 --- a/tests/cli/service.test.ts +++ b/tests/cli/service.test.ts @@ -5,7 +5,7 @@ import { tmpdir } from "node:os"; import { join, resolve } from "node:path"; import test from "node:test"; -import { NMG_METHODS, NMG_PROTOCOL_VERSION } from "../../src/cli/protocol.ts"; +import { NMG_METHODS, NMG_PROTOCOL_VERSION, type NmgMethodResult } from "../../src/cli/protocol.ts"; import { NmgService } from "../../src/cli/service.ts"; import { columnsForBlocks } from "../../src/core/relevance-features.ts"; import { NmgStore } from "../../src/core/store.ts"; @@ -481,7 +481,7 @@ test("task board RPC shares temporary coordination without creating semantic mem resolution: "Parser review complete.", }); assert.equal(resolved.action, "resolve"); - if (resolved.action === "read") throw new Error("expected task board resolve result"); + if (resolved.action !== "resolve") throw new Error("expected task board resolve result"); assert.equal(resolved.entry.resolvedBy, "agent-b"); const store = new NmgStore(databasePath); @@ -514,7 +514,7 @@ test("task board acknowledge records a no-reply confirmation visible on read and // Ack from two collaborators. for (const agentId of ["agent-a", "agent-b"]) { - const acked = await service.invoke("taskBoard", { + const acked: NmgMethodResult["taskBoard"] = await service.invoke("taskBoard", { action: "acknowledge", taskId: "acks", agentId, @@ -687,6 +687,8 @@ test("remember relation resolution creates a reversible proposal without merging assert.deepEqual(resolved.proposal.evidenceMemoryIds, [specific.memory.id, general.memory.id]); const listed = await service.invoke("topologyProposal", { action: "list" }); + assert.equal(listed.action, "list"); + if (listed.action !== "list") throw new Error("expected topology list"); assert.deepEqual( listed.proposals.map((proposal) => proposal.id), [resolved.proposal.id], @@ -695,6 +697,8 @@ test("remember relation resolution creates a reversible proposal without merging action: "assess", proposalId: resolved.proposal.id, }); + assert.equal(assessment.action, "assess"); + if (assessment.action !== "assess") throw new Error("expected topology assessment"); assert.equal(assessment.assessment.proposalId, resolved.proposal.id); assert.equal(assessment.assessment.eligible, false); const reviewed = await service.invoke("topologyProposal", { @@ -702,6 +706,8 @@ test("remember relation resolution creates a reversible proposal without merging proposalId: resolved.proposal.id, decision: "accept", }); + assert.equal(reviewed.action, "review"); + if (reviewed.action !== "review") throw new Error("expected topology review"); assert.equal(reviewed.proposal.status, "accepted"); await assert.rejects( service.invoke("topologyProposal", { @@ -1924,7 +1930,7 @@ test("opt-in embedding auto-sync makes remembered records available to hybrid se }); for (let attempt = 0; attempt < 100; attempt += 1) { const status = await service.invoke("status"); - if (status.embedding.health?.lastSucceededAt) break; + if (status.embedding.health && typeof status.embedding.health === "object" && "lastSucceededAt" in status.embedding.health && status.embedding.health.lastSucceededAt) break; await new Promise((resolve) => setTimeout(resolve, 10)); } const searched = await service.invoke("search", { @@ -1978,7 +1984,7 @@ test("provider presence alone (no AUTO_SYNC env) auto-syncs remembered records t }); for (let attempt = 0; attempt < 100; attempt += 1) { const status = await service.invoke("status"); - if (status.embedding.health?.lastSucceededAt) break; + if (status.embedding.health && typeof status.embedding.health === "object" && "lastSucceededAt" in status.embedding.health && status.embedding.health.lastSucceededAt) break; await new Promise((resolve) => setTimeout(resolve, 10)); } const searched = await service.invoke("search", { diff --git a/tests/cli/task-run-surface.test.ts b/tests/cli/task-run-surface.test.ts index 4c525782..70b5b1ae 100644 --- a/tests/cli/task-run-surface.test.ts +++ b/tests/cli/task-run-surface.test.ts @@ -97,6 +97,7 @@ test( // that wants to adopt an entry gates that field on the method being there (an earlier daemon in // the same epoch would ignore `adopt` and create an unmanaged entry). const hello = await service.invoke("hello"); + assert.ok(hello.methods); assert.ok(hello.methods.includes("taskRun")); await registerAndFreeze(service); const written = await service.invoke("taskBoard", { diff --git a/tests/core/multi-hop-path.test.ts b/tests/core/multi-hop-path.test.ts index 87398b65..3eab8bfb 100644 --- a/tests/core/multi-hop-path.test.ts +++ b/tests/core/multi-hop-path.test.ts @@ -4,6 +4,7 @@ import { mkdtempSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { NmgStore } from "../../src/core/store.ts"; +import type { NodeRelation, NodeRelationType } from "../../src/core/types.ts"; import { propagateEdgeActivation } from "../../src/core/edge-activation.ts"; function withStore(run: (store: NmgStore) => void): void { @@ -18,7 +19,7 @@ function withStore(run: (store: NmgStore) => void): void { } test("propagateEdgeActivation traces the best-activation path per node", () => { - const rel = (id: string, s: string, t: string, type: string) => ({ + const rel = (id: string, s: string, t: string, type: NodeRelationType): NodeRelation => ({ id, sourceNodeId: s, targetNodeId: t, @@ -28,7 +29,13 @@ test("propagateEdgeActivation traces the best-activation path per node", () => { direction: "source->target", fanBudget: false, status: "consolidated", - } as const); + evidenceIds: [], + residence: "ltg", + stability: 1, + consolidationSource: "explicit", + consolidatedAt: "2026-08-01T00:00:00.000Z", + createdAt: "2026-08-01T00:00:00.000Z", + }); const result = propagateEdgeActivation( new Map([["A", 1.0]]), [rel("r1", "A", "B", "causes"), rel("r2", "B", "C", "causes")], diff --git a/tests/core/router-session.test.ts b/tests/core/router-session.test.ts index 731178c4..2bb52aea 100644 --- a/tests/core/router-session.test.ts +++ b/tests/core/router-session.test.ts @@ -6,6 +6,7 @@ import { Router } from "../../src/core/router.ts"; test("hierarchical activation keeps temporal state isolated per session", () => { const router = new Router({ dimensions: 2, + model: "test-embedder", embed: () => [1, 0], }); const first = router.ensureHA(2, "session-a"); diff --git a/tests/core/shadow-evaluation.test.ts b/tests/core/shadow-evaluation.test.ts index 213e61af..d1b98ca3 100644 --- a/tests/core/shadow-evaluation.test.ts +++ b/tests/core/shadow-evaluation.test.ts @@ -48,6 +48,8 @@ test("shadow evaluation separates retrieval, disclosure, attribution, outcome, a topGap: 1, intentCoverage: 1, reasonHealth: 1, + expansionDependence: 0, + expansionRisk: 0, directCount: 1, totalCount: 1, }, diff --git a/tests/core/task-board-retention.test.ts b/tests/core/task-board-retention.test.ts index 4ca74531..1dbcf727 100644 --- a/tests/core/task-board-retention.test.ts +++ b/tests/core/task-board-retention.test.ts @@ -52,7 +52,6 @@ function expiredEntry(store: NmgStore, content = "an entry that is past its TTL" now: WHEN_LIVE, }); store.acknowledgeTaskBoardEntry({ - taskId: "retention-channel", entryId: entry.id, agentId: "scout-b", now: WHEN_LIVE, diff --git a/tests/evals/controller-shadow/calibrate.test.ts b/tests/evals/controller-shadow/calibrate.test.ts index e40d08d3..9a2ac4ef 100644 --- a/tests/evals/controller-shadow/calibrate.test.ts +++ b/tests/evals/controller-shadow/calibrate.test.ts @@ -172,6 +172,7 @@ function eventsFor( expansionUseful: false, excessiveNoise: true, noMemoryNeeded: false, + memoryMisleading: null, }, ]; } @@ -181,7 +182,7 @@ function selection(memoryId: string, nodeId: string, rank: number, usefulness: n memoryId, nodeId, source: "direct" as const, - reason: "lexical_match", + reason: "lexical_match" as const, rank, tier: 0 as const, estimatedTokens: 20, diff --git a/tests/evals/controller-shadow/dataset.test.ts b/tests/evals/controller-shadow/dataset.test.ts index eecf04cb..65526a4f 100644 --- a/tests/evals/controller-shadow/dataset.test.ts +++ b/tests/evals/controller-shadow/dataset.test.ts @@ -188,6 +188,7 @@ function taskEvents( expansionUseful: false, excessiveNoise: false, noMemoryNeeded: false, + memoryMisleading: null, }, ]; } diff --git a/tests/evals/controller-shadow/report.test.ts b/tests/evals/controller-shadow/report.test.ts index 6a131beb..248ded8c 100644 --- a/tests/evals/controller-shadow/report.test.ts +++ b/tests/evals/controller-shadow/report.test.ts @@ -54,6 +54,7 @@ test("shadow coverage keeps missing labels unknown and reports calibration block expansionUseful: true, excessiveNoise: false, noMemoryNeeded: false, + memoryMisleading: null, }, { ...base, diff --git a/tests/evals/omnimemeval-runner.test.ts b/tests/evals/omnimemeval-runner.test.ts index be492867..23c6709a 100644 --- a/tests/evals/omnimemeval-runner.test.ts +++ b/tests/evals/omnimemeval-runner.test.ts @@ -48,10 +48,9 @@ test("CLI exposes only suite, config, resume, and dry-run", () => { }); test("one config supplies common and suite-specific official arguments", () => { - const suites = Object.fromEntries(Object.keys(SUITES).map((name) => [name, []])) as Record< - BenchmarkSuite, - string[] - >; + const suites: Record = { + beam: [], locomo: [], longmemeval: [], "personamem-v2": [], halumem: [], + }; suites.beam = ["--scale", "100k", "--judge-batch-size", "4"]; const repoRoot = fixtureRepo("beam", { suites }); const plan = createRunPlan(parseRunOptions(["beam", "--config", "benchmark.json"]), { diff --git a/tests/evals/skillopt/dataset.test.ts b/tests/evals/skillopt/dataset.test.ts index 47b6d101..7ef047cb 100644 --- a/tests/evals/skillopt/dataset.test.ts +++ b/tests/evals/skillopt/dataset.test.ts @@ -235,6 +235,7 @@ function eventsFor( expansionUseful, excessiveNoise, noMemoryNeeded, + memoryMisleading: null, }; return [retrieval, feedback]; } diff --git a/tests/extensions/nmg/index.test.ts b/tests/extensions/nmg/index.test.ts index ef647a8f..47ae8ac1 100644 --- a/tests/extensions/nmg/index.test.ts +++ b/tests/extensions/nmg/index.test.ts @@ -2254,9 +2254,9 @@ test("/nmg with no arguments opens the interactive select menu", async () => { ctx: { hasUI: boolean; ui: { - select: (...args: unknown[]) => Promise; + select: (title: string, options: string[]) => Promise; input: (...args: unknown[]) => Promise; - notify: (...args: unknown[]) => void; + notify: (message: string) => void; }; }, ) => Promise; diff --git a/tests/helpers/search-result.ts b/tests/helpers/search-result.ts new file mode 100644 index 00000000..8b4e2b3c --- /dev/null +++ b/tests/helpers/search-result.ts @@ -0,0 +1,71 @@ +import type { MemorySearchResult } from "../../src/core/types.ts"; + +/** Complete retrieval record for presentation tests; no database or unsafe fixture cast. */ +export function searchResultFixture(id: string, statement: string): MemorySearchResult { + const stamp = "2026-08-20T12:00:00.000Z"; + const evidence = { + id: `evidence-${id}`, + sessionId: null, + sourceMessageId: null, + role: "user" as const, + content: statement, + sourceRef: null, + createdAt: stamp, + }; + return { + memory: { + id, + nodeId: `node-${id}`, + evidenceId: evidence.id, + evidenceIds: [evidence.id], + statement, + memoryType: "fact", + stateKey: null, + eventTime: null, + sourceActor: "user", + truthStatus: "asserted", + confidence: null, + polarity: null, + predicateKey: null, + extractMethod: null, + claims: null, + markers: [], + scope: { project: "atlas" }, + validFrom: null, + validUntil: null, + status: "active", + resolution: "resolved", + openedAt: null, + relatedMemoryIds: [], + residence: "ltg", + promotedAt: null, + expiresAt: null, + evidenceRole: "support", + supersedesId: null, + sessionId: null, + tier: 1, + importance: 0.5, + accessCount: 0, + lastAccessedAt: null, + writeReason: "presentation fixture", + writeSource: "user", + createdAt: stamp, + }, + node: { + id: `node-${id}`, + canonicalName: `Atlas ${id}`, + kind: "topic", + summary: "", + createdAt: stamp, + updatedAt: stamp, + status: "active", + residence: "ltg", + }, + evidence, + evidenceRecords: [evidence], + lexicalScore: 1, + vectorScore: 0, + routeScore: 0, + combinedScore: 1, + }; +} diff --git a/tests/integration/agent-surface.test.ts b/tests/integration/agent-surface.test.ts index 59fd786e..ed7d7575 100644 --- a/tests/integration/agent-surface.test.ts +++ b/tests/integration/agent-surface.test.ts @@ -1,6 +1,7 @@ import assert from "node:assert/strict"; import test from "node:test"; +import { searchResultFixture } from "../helpers/search-result.ts"; import type { MemoryContext } from "../../src/core/types.ts"; import { renderEvidenceSurface, @@ -18,6 +19,9 @@ test("session Active Graph surface renders only projected temporary items", () = projectionSequence: 1, latestProjectionId: "projection-a", temporaryProjectionActive: true, + disclosedProjectionIds: [], + disclosureTurn: 0, + disclosures: [], items: [ { id: "memory:a", @@ -52,22 +56,20 @@ test("session Active Graph surface renders only projected temporary items", () = function context(): MemoryContext { const chainId = "chain-atlas"; - const result = (id: string, statement: string, position: number) => - ({ - memory: { - id, - statement, - memoryType: "fact", - tier: 1, - truthStatus: "asserted", - scope: { project: "atlas" }, - eventTime: "2026-08-20T12:00:00.000Z", - }, - node: { canonicalName: "Atlas" }, - evidence: { content: `Exact source detail for ${id}.` }, + const result = (id: string, statement: string, position: number): MemoryContext["results"][number] => { + const fixture = searchResultFixture(id, statement); + return { + ...fixture, + memory: { ...fixture.memory, eventTime: "2026-08-20T12:00:00.000Z" }, + node: { ...fixture.node, canonicalName: "Atlas" }, + evidence: { ...fixture.evidence, content: `Exact source detail for ${id}.` }, + evidenceRecords: [{ ...fixture.evidence, content: `Exact source detail for ${id}.` }], recallReason: "vector_match", + lexicalScore: 0, + vectorScore: 1, chainMemberships: [{ chainId, chainType: "logical", topic: "Atlas flow", position }], - }) as MemoryContext["results"][number]; + }; + }; return { results: [ result("memory-a", "Atlas receives input.", 0), @@ -166,7 +168,7 @@ test("shared task-board surface renders coordination without promoting it to mem entries: [ { id: "entry-1", - sequence: 7, + kind: "question", status: "open", agentId: "agent-a", @@ -174,7 +176,7 @@ test("shared task-board surface renders coordination without promoting it to mem ackedBy: ["agent-b"], }, ], - nextCursor: 7, + nextCursor: "7", }, { taskId: "adapter-migration" }, ); diff --git a/tests/integration/chain-projection.test.ts b/tests/integration/chain-projection.test.ts index 3b21fc0c..b9ee6f40 100644 --- a/tests/integration/chain-projection.test.ts +++ b/tests/integration/chain-projection.test.ts @@ -2,26 +2,18 @@ import assert from "node:assert/strict"; import test from "node:test"; import type { MemoryContext } from "../../src/core/types.ts"; +import { searchResultFixture } from "../helpers/search-result.ts"; import { projectLogicalChains } from "../../src/integration/chain-projection.ts"; function logicalChainContext(): MemoryContext { const chainId = "logical-merge"; const result = (id: string, statement: string, position: number) => ({ - memory: { - id, - statement, - memoryType: "fact", - tier: 1, - truthStatus: "asserted", - scope: { project: "atlas" }, - }, - node: { canonicalName: `Atlas ${id}` }, - evidence: { content: statement }, + ...searchResultFixture(id, statement), chainMemberships: [ - { chainId, position, chainType: "logical", topic: "Atlas merge evidence" }, + { chainId, position, chainType: "logical" as const, topic: "Atlas merge evidence" }, ], - }) as MemoryContext["results"][number]; + }); return { results: [ diff --git a/tests/integration/ooo-acceptance-one-predicate.test.ts b/tests/integration/ooo-acceptance-one-predicate.test.ts index 2d39d83a..517faa0a 100644 --- a/tests/integration/ooo-acceptance-one-predicate.test.ts +++ b/tests/integration/ooo-acceptance-one-predicate.test.ts @@ -2,7 +2,7 @@ import assert from "node:assert/strict"; import { readFileSync, readdirSync, statSync } from "node:fs"; import { join } from "node:path"; import { test } from "node:test"; -import { acceptedFact, isAccepted, type TaskUnit } from "../../src/integration/task-semantics.ts"; +import { acceptedFact, isAccepted, type TaskUnit, type RecordedFacts } from "../../src/integration/task-semantics.ts"; const root = new URL("../../", import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1"); const read = (relative: string) => readFileSync(join(root, relative), "utf8"); @@ -70,7 +70,7 @@ test("the two readers agree over the same recorded facts", () => { ] as const; for (const item of cases) { - const facts = { + const facts: RecordedFacts = { artifacts: { T: "rev-1" }, revisions: { T: "rev-1" }, verdicts: item.verdict ? { T: item.verdict } : {}, diff --git a/tests/integration/ooo-advisers.test.ts b/tests/integration/ooo-advisers.test.ts index 16c8ab7e..25e8dfae 100644 --- a/tests/integration/ooo-advisers.test.ts +++ b/tests/integration/ooo-advisers.test.ts @@ -52,7 +52,6 @@ const SCOPE = { /** What the projection says when the set holds two selectable tasks. */ const projection = (legal: readonly string[]): Omit => ({ ...SCOPE, - legal: [], ready: [...legal], accepted: [], blocked: {}, diff --git a/tests/integration/ooo-ordinary-failure.test.ts b/tests/integration/ooo-ordinary-failure.test.ts index 94a29188..e9c50558 100644 --- a/tests/integration/ooo-ordinary-failure.test.ts +++ b/tests/integration/ooo-ordinary-failure.test.ts @@ -102,7 +102,7 @@ test("a deliverable the host refuses neither accepts nor releases, and the coord ); // The coordinator's recovery, then the same task with an artifact the host accepts. - gate.reopen("B"); + gate.reopen("B", "retry the refused attempt"); verdictB = "accept"; assert.equal(gate.next(), "B", "the reopened task is selectable again"); const retry = gate.claim("B", "worker-one") as BoardTicket; @@ -203,7 +203,7 @@ test("a later refusal does not withdraw the prefix that was already accepted", a assert.equal(gate.next(), null, "and the refused task is not re-selected on its own"); // Reopening it and accepting it completes the plan; the prefix was never lost along the way. - gate.reopen("A"); + gate.reopen("A", "retry the refused attempt"); verdictA = "accept"; const retry = gate.claim("A", "worker-two") as BoardTicket; assert.equal( diff --git a/tests/integration/ooo-ordinary-handoff.test.ts b/tests/integration/ooo-ordinary-handoff.test.ts index a7582ecc..d9cbfeb2 100644 --- a/tests/integration/ooo-ordinary-handoff.test.ts +++ b/tests/integration/ooo-ordinary-handoff.test.ts @@ -76,7 +76,7 @@ test("an ordinary handoff reaches the shared semantics, runs, is accepted, and r // The worker is the caller's own code. Here it delivers a patch whose digest is the one the // claim froze, which is the only envelope the board accepts for a patch task. - const delivery = (taskId: string, ticket: BoardTicket, path: string, content: string) => + const delivery = (_taskId: string, ticket: BoardTicket, path: string, content: string) => JSON.stringify({ digest: ticket.inputDigest, files: [{ path, content }] }); const ticketB = gate.claim("B", "worker-one") as BoardTicket; diff --git a/tests/integration/ooo-publication-invariants.test.ts b/tests/integration/ooo-publication-invariants.test.ts index 698da413..0be62bbd 100644 --- a/tests/integration/ooo-publication-invariants.test.ts +++ b/tests/integration/ooo-publication-invariants.test.ts @@ -29,6 +29,7 @@ import { compileTaskUnits, deriveStatus, type RecordedFacts, + type CompileInput, } from "../../src/integration/task-semantics.ts"; import { nextTask, type DispatchTask } from "../../src/integration/ooo-execution.ts"; import type { PatchTaskSpec } from "../../src/integration/ooo-board.ts"; diff --git a/tests/integration/ooo-session-chain-contract.test.ts b/tests/integration/ooo-session-chain-contract.test.ts index af999442..f0e0c133 100644 --- a/tests/integration/ooo-session-chain-contract.test.ts +++ b/tests/integration/ooo-session-chain-contract.test.ts @@ -65,7 +65,7 @@ test("a chain input names the check when the unit has one", () => { const frozen = patchWork(); const input = patchSessionInput(frozen, { looseConclusion: true, - check: { label: "the unit check", maxRuns: 1, run: async () => ({ ok: true, output: "" }) }, + check: { label: "the unit check", maxRuns: 1, run: async () => ({ verdict: "accept", log: "" }) }, }); assert.match(input.prompt, /the check the unit check/); assert.match(input.prompt, /call only read_snapshot, run_check and submit_artifact/); diff --git a/tests/integration/task-semantics-cases.test.ts b/tests/integration/task-semantics-cases.test.ts index 30f01469..1d477977 100644 --- a/tests/integration/task-semantics-cases.test.ts +++ b/tests/integration/task-semantics-cases.test.ts @@ -67,7 +67,7 @@ test("case 1: a join waits for every dependency, and concurrent candidates never artifacts, verdicts: facts({ P: "accepted", T: "accepted" }), }); - assert.deepEqual(both.accepted.sort(), ["P", "T"]); + assert.deepEqual([...both.accepted].sort(), ["P", "T"]); assert.equal(both.ready[0], "J", "only acceptance of both makes the join next"); // Preparation and mergeability are the two halves of this case the finite model does not diff --git a/tests/support/cordis-adapter.test.ts b/tests/support/cordis-adapter.test.ts index 550c0dd5..86c3862e 100644 --- a/tests/support/cordis-adapter.test.ts +++ b/tests/support/cordis-adapter.test.ts @@ -22,10 +22,10 @@ test("Cordis adapter disposes registered effects in reverse order", async () => const runtime = adapter.createTestRuntime(); await runtime.use((scope) => { - scope.effect(() => () => events.push("first"), "first"); + scope.effect(() => () => { events.push("first"); }, "first"); }); await runtime.use((scope) => { - scope.effect(() => () => events.push("second"), "second"); + scope.effect(() => () => { events.push("second"); }, "second"); }); await runtime.dispose(); diff --git a/tests/support/cordis-adapter.ts b/tests/support/cordis-adapter.ts index bae93b84..06b62750 100644 --- a/tests/support/cordis-adapter.ts +++ b/tests/support/cordis-adapter.ts @@ -25,7 +25,7 @@ export function createTestRuntime(): CordisTestRuntime { const fiber = await root.plugin((context) => effect({ effect(register, name) { - context.effect(register, name); + context.effect(() => register() ?? (() => {}), name); }, }), ); diff --git a/tests/tools/agent-verify.test.ts b/tests/tools/agent-verify.test.ts index 52fe93d1..bc4a7fd4 100644 --- a/tests/tools/agent-verify.test.ts +++ b/tests/tools/agent-verify.test.ts @@ -47,6 +47,8 @@ function report(): AgentContextReport { }, ], availableRoutes: ["store", "docs"], + capabilities: [], + availableCapabilities: [], guardrails: [], canonical: { design: "design.md", completion: "audit.md", todo: "todo.md" }, state: { desiredRevision: "desired", observedRevision: "observed" }, diff --git a/tests/tools/mutation-anchor.test.ts b/tests/tools/mutation-anchor.test.ts index 71bd197e..908d117e 100644 --- a/tests/tools/mutation-anchor.test.ts +++ b/tests/tools/mutation-anchor.test.ts @@ -33,10 +33,7 @@ function selected(source: string, mutant: Mutant): string { function refusal(source: string, mutant: Mutant): string { const site = locate(source, mutant); - assert.ok( - "reason" in site, - `expected a refusal, got a site: ${source.slice(site.start, site.end)}`, - ); + assert.ok("reason" in site, "expected a refusal, got a site"); return site.reason; } diff --git a/tests/tools/test-groups.test.ts b/tests/tools/test-groups.test.ts index 35f3c69a..b9086bc9 100644 --- a/tests/tools/test-groups.test.ts +++ b/tests/tools/test-groups.test.ts @@ -63,15 +63,32 @@ test("local and CI verification groups share named package contracts", () => { assert.match(packageJson.scripts["verify:product-ci"], /test:coverage/); }); +test("tests-surface type checking is blocking in the shared static contract", () => { + const checks = [...packageJson.scripts["verify:static"]!.matchAll(/npm run ([\w:-]+)/gu)].map( + (match) => match[1]!, + ); + assert.equal(checks.filter((name) => name === "check:tests").length, 1); + const ci = parseYaml(workflow) as { + jobs: { static: { steps: { run?: string; "continue-on-error"?: boolean }[] } }; + }; + const staticStep = ci.jobs.static.steps.find((step) => step.run === "npm run verify:static"); + assert.ok(staticStep); + assert.notEqual(staticStep["continue-on-error"], true); + assert.ok( + !ci.jobs.static.steps.some((step) => step.run === "npm run check:tests"), + "CI uses the shared contract, not a duplicate advisory type-check step", + ); +}); + test("agent static checks match the CI contract without nested duplicate execution", () => { const context = parseYaml( readFileSync(new URL("../../agent-context.yaml", import.meta.url), "utf8"), ) as { routes: { id: string; verify: { blocking: string[] } }[] }; const route = context.routes.find(({ id }) => id === "ci-and-tests"); assert.ok(route); - const staticChecks = [...packageJson.scripts["verify:static"]!.matchAll(/npm run ([\w:-]+)/gu)].map( - (match) => match[1]!, - ); + const staticChecks = [ + ...packageJson.scripts["verify:static"]!.matchAll(/npm run ([\w:-]+)/gu), + ].map((match) => match[1]!); assert.deepEqual(route.verify.blocking, [...staticChecks, "test:product"]); assert.equal(new Set(route.verify.blocking).size, route.verify.blocking.length); }); diff --git a/tools/complexity-gate.ts b/tools/complexity-gate.ts index 319f907c..250de4f5 100644 --- a/tools/complexity-gate.ts +++ b/tools/complexity-gate.ts @@ -313,7 +313,7 @@ export function functionIdentityAtLine(filePath: string, source: string, line: n ); let best: ts.FunctionLikeDeclaration | undefined; const visit = (node: ts.Node): void => { - if (ts.isFunctionLike(node) && containsLine(sourceFile, node, line)) { + if (ts.isFunctionLike(node) && "body" in node && containsLine(sourceFile, node, line)) { if (!best || node.getWidth(sourceFile) < best.getWidth(sourceFile)) best = node; } ts.forEachChild(node, visit); diff --git a/tools/mutation-anchor.ts b/tools/mutation-anchor.ts index bb0a2d24..00bedac1 100644 --- a/tools/mutation-anchor.ts +++ b/tools/mutation-anchor.ts @@ -285,16 +285,16 @@ function collect(root: ts.Node, isWanted: (node: ts.Node) => * => ...` or `claim: (store, parsed) => ...` names one function, and whether the formatter wrote it as * a declaration is not a fact about which rules live inside it. */ function uniqueMember(source: ts.SourceFile, name: string): ts.Node | { reason: string } { - const bound = (node: ts.Node, initializer: ts.Expression | undefined): boolean => + const bound = (initializer: ts.Expression | undefined): boolean => initializer !== undefined && (ts.isArrowFunction(initializer) || ts.isFunctionExpression(initializer)); const boundFunction = (node: ts.Node): boolean => (ts.isVariableDeclaration(node) && node.name.getText(source) === name && - bound(node, node.initializer)) || + bound(node.initializer)) || (ts.isPropertyAssignment(node) && node.name.getText(source) === name && - bound(node, node.initializer)); + bound(node.initializer)); const named = (node: ts.Node): boolean => ((ts.isMethodDeclaration(node) || ts.isFunctionDeclaration(node)) && node.name?.getText(source) === name) || @@ -944,7 +944,7 @@ const RESOLVERS: Readonly> = { */ function deriveSite( text: string, - derive: Scoped, + derive: Derive, to: string | undefined, ): Site | { reason: string } { const source = ts.createSourceFile("mutant.ts", text, ts.ScriptTarget.Latest, true); diff --git a/workbuddy-plugin/nmg-hook.ts b/workbuddy-plugin/nmg-hook.ts index 2d1e1569..a82cea3b 100644 --- a/workbuddy-plugin/nmg-hook.ts +++ b/workbuddy-plugin/nmg-hook.ts @@ -32,6 +32,39 @@ import type { MemoryContext } from "../src/core/types.ts"; import { assertDaemonProtocol } from "../src/cli/daemon-client.ts"; import type { NmgHelloResult } from "../src/cli/protocol.ts"; +interface HookPayload { + prompt?: unknown; + hook_event_name?: string; + event?: string; + tool_name?: string; + tool_input?: { command?: unknown }; + session_id?: string; + sessionId?: string; + cwd?: string; +} + +interface HttpLease { + host: string; + port: number; + token: string; + pid: number; +} + +interface WakeEntry { + id: string; + taskId: string; + kind: string; + content: string; + sequence?: number; + status?: string; + sourceSessionId?: string | null; + agentId?: string | null; + claimExpiresAt?: string | null; + to?: string | null; + serialState?: string | null; + createdAt?: string | null; +} + const NUDGE = [ "", "A code commit or task completion was just detected. NMG long-term memory is", @@ -63,8 +96,8 @@ const WORLD_BOARD_ID = "default"; const FALSE_LIKE = new Set(["0", "false", "off", "no"]); const BROADCAST_PREFIX = "[NMG board 协作广播]"; const WAKE_KINDS = new Set(["question", "blocker", "handoff"]); -const KIND_RANK = { question: 0, blocker: 1, handoff: 2 }; -const KIND_LABEL = { +const KIND_RANK: Readonly> = { question: 0, blocker: 1, handoff: 2 }; +const KIND_LABEL: Readonly> = { question: "问题", blocker: "阻塞", handoff: "交接", @@ -78,7 +111,7 @@ function dataDir(): string { return process.env.NMG_DATA_DIR?.trim() || join(homedir(), ".nmg"); } -function promptText(payload): string { +function promptText(payload: HookPayload): string { const prompt = payload?.prompt; if (typeof prompt === "string") return prompt; if (Array.isArray(prompt)) { @@ -90,12 +123,12 @@ function promptText(payload): string { return ""; } -function isGitCommit(payload): boolean { +function isGitCommit(payload: HookPayload): boolean { const command = payload?.tool_input?.command; return typeof command === "string" && /\bgit\s+commit\b/u.test(command); } -function shouldNudge(payload): boolean { +function shouldNudge(payload: HookPayload): boolean { const event = payload?.hook_event_name ?? payload?.event; if (event === "UserPromptSubmit") return COMPLETION_PATTERN.test(promptText(payload)); if (event === "PreToolUse") { @@ -105,7 +138,7 @@ function shouldNudge(payload): boolean { return false; } -function isPromptSubmit(payload): boolean { +function isPromptSubmit(payload: HookPayload): boolean { const event = payload?.hook_event_name ?? payload?.event; return event === "UserPromptSubmit"; } @@ -118,7 +151,7 @@ function readJson(path: string) { } } -function writeJson(path: string, value) { +function writeJson(path: string, value: unknown) { try { writeFileSync(path, JSON.stringify(value), "utf8"); } catch { @@ -126,16 +159,16 @@ function writeJson(path: string, value) { } } -function pidAlive(pid): boolean { +function pidAlive(pid: unknown): boolean { try { process.kill(Number(pid), 0); return true; } catch (error) { - return error?.code === "EPERM"; + return (error as NodeJS.ErrnoException | null)?.code === "EPERM"; } } -export function liveLease(dir: string) { +export function liveLease(dir: string): HttpLease | null { const lease = readJson(join(dir, "nmg.sqlite.server.json")); if ( !lease || @@ -151,7 +184,7 @@ export function liveLease(dir: string) { return pidAlive(lease.pid) ? lease : null; } -async function rpcCall(lease, method: string, params) { +async function rpcCall(lease: HttpLease, method: string, params: unknown) { const response = await fetch(`http://${lease.host}:${lease.port}/`, { method: "POST", headers: { @@ -190,7 +223,7 @@ async function compatibleLease(dir: string) { return pending; } -function agentIdentity(payload, environment = process.env) { +function agentIdentity(payload: HookPayload, environment = process.env) { const sessionId = payload?.session_id ?? payload?.sessionId ?? "workbuddy-hook"; return { sessionId, @@ -203,7 +236,7 @@ function agentIdentity(payload, environment = process.env) { * Discovery belongs to the system layer; it never injects context or wakes * another model. */ export async function reportAgentPresence( - payload, + payload: HookPayload, dir = dataDir(), environment = process.env, ): Promise { @@ -223,7 +256,7 @@ export async function reportAgentPresence( /** Small-budget automatic recall: one daemon search per user turn, compact * header projection from the shared Agent Surface, per-session id dedup. * Returns "" when there is nothing to inject. */ -export async function recallHeaders(payload, dir = dataDir()): Promise { +export async function recallHeaders(payload: HookPayload, dir = dataDir()): Promise { const lease = await compatibleLease(dir); if (!lease) return ""; const query = promptText(payload).trim().slice(0, RECALL_QUERY_CHARS); @@ -272,7 +305,10 @@ export async function recallHeaders(payload, dir = dataDir()): Promise { /** Pure board-wake gate (same semantics as the Kimi hook). A claim suppresses * a notice only while its lease is live; notify-only kinds never wake. */ -export function isBoardWakeCandidate(entry, { sessionId, agentId, now = Date.now() }): boolean { +export function isBoardWakeCandidate( + entry: WakeEntry, + { sessionId, agentId, now = Date.now() }: { sessionId: string; agentId: string; now?: number }, +): boolean { const ownEcho = entry.sourceSessionId === sessionId || (entry.sourceSessionId == null && entry.agentId === agentId); @@ -293,7 +329,7 @@ export function isBoardWakeCandidate(entry, { sessionId, agentId, now = Date.now /** Poll the task board for one undelivered open entry and format its wake * notice. Returns "" when there is nothing to say (or wake is off). */ export async function pollBoardWake( - payload, + payload: HookPayload, dir = dataDir(), environment = process.env, ): Promise { @@ -317,10 +353,10 @@ export async function pollBoardWake( const lease = await compatibleLease(dir); if (!lease) return ""; const { sessionId, agentId } = agentIdentity(payload, environment); - const rpc = (method: string, params) => rpcCall(lease, method, params); + const rpc = (method: string, params: unknown) => rpcCall(lease, method, params); - const candidates = []; - const collect = (taskId: string, entries) => { + const candidates: WakeEntry[] = []; + const collect = (taskId: string, entries: readonly WakeEntry[] | undefined) => { for (const entry of entries ?? []) { if (isBoardWakeCandidate(entry, { sessionId, agentId, now })) { candidates.push({ ...entry, taskId }); @@ -381,7 +417,7 @@ export async function pollBoardWake( } export async function runHook( - payload, + payload: HookPayload, options: { dir?: string; environment?: NodeJS.ProcessEnv } = {}, ): Promise { const output: string[] = []; From 613076d84a845f50d3e6669271a50f64cabc13bf Mon Sep 17 00:00:00 2001 From: wefio <48851810+wefio@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:16:46 +0800 Subject: [PATCH 2/2] chore(format): cover the developer TypeScript surfaces Widen format and format:check to tests/, evals/, scripts/ and tools/, and bring the staged-file hook to the same directory set. Subpackages keep their own pipelines and non-TypeScript file types are out of scope. Run one controlled Prettier pass. Structure and literals are unchanged: a syntax-tree comparison over the 382 pre-existing files reported no difference beyond equivalent parenthesis and quoted-property spellings. Deliberately broken evaluation fixtures are formatted, not repaired. Validation: agent:verify passed all 16 blocking checks, including the full static contract and test:product. --- .githooks/pre-commit | 2 +- .../2026-09-16-ci-static-coverage.md | 24 +- .../2026-09-16-ci-static-coverage.zh-CN.md | 17 +- docs/design/ci-cd-and-quality.md | 2 + evals/agent-telemetry.ts | 4 +- evals/benchmarks/loaders.ts | 80 +++---- evals/consolidation/run.ts | 12 +- evals/controller-shadow/independence.ts | 4 +- evals/controller-shadow/tau-worker.ts | 24 +- evals/gate/run.ts | 7 +- evals/halumem/agent-extract.ts | 35 ++- evals/halumem/prepare.ts | 18 +- evals/halumem/promotion-audit.ts | 41 ++-- evals/local-env.ts | 5 +- evals/longmemeval/official.ts | 12 +- evals/longmemeval/retrieval-evidence.ts | 14 +- evals/memory-quality/run.ts | 17 +- evals/natural-maintenance/audit.ts | 126 ++++++++--- evals/natural-readiness/report.ts | 4 +- evals/official/bootstrap.ts | 4 +- evals/official/protocol.ts | 35 ++- evals/official/python.ts | 4 +- evals/official/unified-score.ts | 7 +- evals/omnimemeval/bridge.ts | 55 +++-- evals/omnimemeval/judge-provider.ts | 27 ++- evals/omnimemeval/merge-longmemeval-shards.ts | 8 +- .../research/audits/audit-elbow.ts | 50 ++++- .../research/audits/audit-fibonacci-recall.ts | 124 ++++++++--- .../research/audits/audit-qpp-signal.ts | 84 +++++-- .../research/probes/fake-cache-api.ts | 6 +- evals/omnimemeval/run.ts | 70 +++--- evals/ooo-execution/families.test.ts | 17 +- .../fixtures/pipeline/normalize.canned.ts | 5 +- .../fixtures/report/alpha.canned.ts | 5 +- .../fixtures/report/summary.canned.ts | 6 +- evals/ooo-execution/pilot.ts | 77 ++++--- evals/ooo-execution/plan-driver.test.ts | 12 +- evals/recall-compression.ts | 52 +++-- evals/retrieval/ingest-ablation.ts | 7 +- evals/retrieval/profile-ingest.ts | 5 +- evals/retrieval/profile-progressive.ts | 35 ++- evals/retrieval/profile-size.ts | 10 +- evals/skillopt/export.ts | 10 +- evals/topology/namesakes.ts | 8 +- evals/topology/run.ts | 59 +++-- package.json | 4 +- scripts/sync-nmg-skill.ts | 14 +- tests/cli/data-path.test.ts | 5 +- tests/cli/service.test.ts | 16 +- tests/core/advanced-query.test.ts | 24 +- tests/core/analogy.test.ts | 99 +++++++-- tests/core/autodiff.test.ts | 12 +- tests/core/community.test.ts | 39 +++- tests/core/context-reward.test.ts | 8 +- tests/core/fork-merge.test.ts | 5 +- tests/core/graph-cycles.test.ts | 207 +++++++++++++++--- tests/core/hierarchical-activation.test.ts | 71 +++--- tests/core/memory-chains.test.ts | 97 +++++--- tests/core/openai-embedding.test.ts | 5 +- tests/core/rank-fusion.test.ts | 13 +- tests/core/reasoning-workspace.test.ts | 7 +- tests/core/semantic-domain.test.ts | 13 +- tests/core/stg-v2.test.ts | 50 +++-- tests/core/store/duplicates.test.ts | 61 ++++-- tests/core/store/leaf-summaries.test.ts | 19 +- tests/core/store/node-summaries.test.ts | 48 ++-- tests/core/store/retention.test.ts | 6 +- tests/core/store/retrieval.test.ts | 5 +- tests/core/store/search-ranking.test.ts | 23 +- tests/core/store/vector-codec.test.ts | 8 +- tests/core/store/writes.test.ts | 4 +- tests/core/task-board-deliverable.test.ts | 6 +- tests/docker/container-definition.test.ts | 5 +- tests/evals/benchmarks/loaders.test.ts | 133 ++++++----- tests/evals/bge-batcher.test.ts | 2 +- tests/evals/bge-server-contract.test.ts | 2 +- .../evals/cache-environment-simulator.test.ts | 17 +- tests/evals/consolidation/run.test.ts | 6 +- tests/evals/context-live-canary.test.ts | 41 ++-- .../controller-shadow/tau-worker.test.ts | 8 +- tests/evals/halumem/agent-extract.test.ts | 9 +- tests/evals/longmemeval/official.test.ts | 62 ++++-- tests/evals/longmemeval/report.test.ts | 36 ++- .../longmemeval/retrieval-evidence.test.ts | 3 +- tests/evals/natural-maintenance-audit.test.ts | 4 +- tests/evals/official/protocol.test.ts | 58 +++-- tests/evals/official/python.test.ts | 5 +- tests/evals/omnimemeval-install.test.ts | 11 +- .../evals/omnimemeval-judge-provider.test.ts | 65 ++++-- tests/evals/omnimemeval-merge-shards.test.ts | 15 +- tests/evals/omnimemeval-runner.test.ts | 6 +- tests/evals/retrieval-score.test.ts | 5 +- tests/evals/retrieval-smoke.test.ts | 2 +- tests/evals/skillopt/dataset.test.ts | 4 +- tests/evals/topology/namesakes.test.ts | 2 +- tests/evals/topology/run.test.ts | 14 +- .../extensions/pi-dependency-boundary.test.ts | 12 +- tests/integration/agent-surface.test.ts | 6 +- tests/integration/chain-projection.test.ts | 13 +- tests/integration/controller-channel.test.ts | 8 +- tests/integration/lab-capabilities.test.ts | 5 +- tests/integration/leaf-summarizer.test.ts | 10 +- .../ooo-acceptance-one-predicate.test.ts | 7 +- tests/integration/ooo-fusion-plan.test.ts | 5 +- .../ooo-post-commit-notification.test.ts | 6 +- .../ooo-publication-invariants.test.ts | 17 +- .../ooo-session-chain-contract.test.ts | 6 +- tests/skills/nmg-memory.test.ts | 13 +- tests/skills/nmg-skill-sync.test.ts | 8 +- tests/support/cordis-adapter.test.ts | 14 +- tests/tools/agent-verify.test.ts | 5 +- tests/tools/ci-status-snapshot.test.ts | 12 +- tests/tools/recall-instance.test.ts | 24 +- tests/tools/recall-probe.test.ts | 7 +- tests/tools/test-groups.test.ts | 10 + tools/check-lock.ts | 23 +- tools/ci-status-snapshot.ts | 13 +- tools/recall-instance-judge.ts | 10 +- tools/recall-probe.ts | 10 +- tools/relevance-gate-calibration.ts | 14 +- tools/rtm-check.ts | 13 +- 121 files changed, 1901 insertions(+), 984 deletions(-) diff --git a/.githooks/pre-commit b/.githooks/pre-commit index f15ff044..9a2f4df4 100755 --- a/.githooks/pre-commit +++ b/.githooks/pre-commit @@ -19,7 +19,7 @@ set -eu # files are excluded: prettier has nothing to format on a path that is gone. # Keep only paths under the root Prettier surface. files="$(git diff --cached --name-only --diff-filter=ACM -- '*.ts' \ - | grep -E '^(src|\\.pi|workbuddy-plugin)/' || true)" + | grep -E '^(src|\\.pi|workbuddy-plugin|tests|evals|scripts|tools)/' || true)" [ -z "$files" ] && exit 0 # Format only files still present on disk, then re-stage what prettier rewrote. diff --git a/docs/decisions/implemented/2026-09-16-ci-static-coverage.md b/docs/decisions/implemented/2026-09-16-ci-static-coverage.md index 72431cea..17366c44 100644 --- a/docs/decisions/implemented/2026-09-16-ci-static-coverage.md +++ b/docs/decisions/implemented/2026-09-16-ci-static-coverage.md @@ -73,6 +73,14 @@ narrow discriminated results; the Kimi hook's JavaScript exports have a declarat boundary, and presentation fixtures use complete retrieval records. The blocking contract test pins one `check:tests` invocation and rejects a duplicate advisory step. +**Formatting covers the declared TypeScript surfaces.** `format` and `format:check` +scan `src/`, `.pi/`, `workbuddy-plugin/`, `tests/`, `evals/`, `scripts/` and `tools/` +with `**/*.ts`. The staged-file hook uses the same directory set; subpackages retain +their own formatting pipelines. This does not extend to `.mjs`, `.mts`, Python or +other file types. `test-groups.test.ts` pins both commands' surface and hook parity. +Deliberately broken evaluation fixtures are formatted, not repaired: their wrong +behavior is an input to the evaluator, not a product defect. + ### What the first widened scan reported The scan's first pass over the newly covered surface produced 11 findings. Nine were @@ -138,9 +146,9 @@ evidence behind the advisory severity above: errors were repaired: it would permit the same fixture and API drift to accumulate again. Casting malformed fixtures through `unknown` or excluding failing files is also rejected; neither establishes that the test obeys the contract it exercises. -- **Widen `format:check` in the same change.** Deferred, not rejected: it is the same - class of hole, but it needs a Prettier pass over the newly covered directories, and a - formatting rewrite of research code would bury the lint change it travels with. +- **Mix the formatting pass into lint or type repairs.** Rejected: broad research + formatting obscures semantic fixes. The coverage expansion and controlled Prettier + pass are a separate commit, with fixture semantics and mutation evidence checked. ## Consequences @@ -159,11 +167,7 @@ evidence behind the advisory severity above: ## Deferred -- `format:check` scans `src/`, `.pi/` and `workbuddy-plugin/` only. `tests/`, `evals/`, - `scripts/` and `tools/` are still unformatted, and closing that hole needs its own - Prettier pass. -- No route type-checks `evals/`, `scripts/`, or the `tools/` files outside the three - names in `tsconfig.json`. `check:tests` is the first slice; the product surface with - the same flags reports one error, so the same treatment is available for a later - change. +- Files under `evals/`, `scripts/` or `tools/` that are neither explicitly included + nor imported by `tsconfig.json` or `tsconfig.tests.json` remain outside their type + checks. Full-directory type-check coverage is a separate scope. diff --git a/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md b/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md index 7f502316..55fa3f55 100644 --- a/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md +++ b/docs/decisions/implemented/2026-09-16-ci-static-coverage.zh-CN.md @@ -60,6 +60,12 @@ severity。 的前提下修复。fixture 补齐当前契约字段并收窄判别结果;Kimi hook 的 JavaScript 导出有声明边界, 渲染 fixture 使用完整检索记录。阻塞契约测试锁定一次 `check:tests` 调用,并拒绝重复 advisory 步骤。 +**格式检查覆盖声明的 TypeScript 面。** `format` 与 `format:check` 以 `**/*.ts` +扫描 `src/`、`.pi/`、`workbuddy-plugin/`、`tests/`、`evals/`、`scripts/`、`tools/`。 +暂存文件 hook 使用相同目录集合,子包保留自己的格式流程;本次不拓宽至 `.mjs`、`.mts`、Python +等类型。`test-groups.test.ts` 锁定两条命令的扫描面与 hook 一致性。故意错误的评估 fixture +只格式化、不修正确性:它们的错误行为是 evaluator 的输入,不是产品缺陷。 + ### 首次拓宽扫描报出的 11 条 第一次扫过新覆盖面时共 11 条。其中 9 条确实是死代码,已删除;另 2 条根本不是死代码: @@ -116,8 +122,8 @@ severity。 - **永久保留 advisory 类型检查。** 实测错误修复后否决:它会允许同样的 fixture 与 API 漂移 再次积累。把错误 fixture 经 `unknown` 强转或排除报错文件也被否决;它们不能证明测试遵守了 自己要检验的契约。 -- **在同一次改动里拓宽 `format:check`。** 记为 Deferred 而非否决:它是同一类漏洞,但需要对新增 - 目录跑一次 Prettier,而研究代码的格式化重写会淹没与它同行的 lint 改动。 +- **把格式化 pass 混进 lint 或类型修复提交。** 否决:大范围研究代码格式化会淹没语义修复。 + 覆盖扩展与受控 Prettier pass 单独提交,并验证 fixture 语义与 mutation 证据。 ## Consequences @@ -133,9 +139,6 @@ severity。 ## Deferred -- `format:check` 仍只扫 `src/`、`.pi/`、`workbuddy-plugin/`。`tests/`、`evals/`、`scripts/`、 - `tools/` 仍未纳入格式检查,补上这个洞需要单独一次 Prettier pass。 -- 没有任何轨道对 `evals/`、`scripts/` 以及 `tsconfig.json` 里那三个文件名之外的 `tools/` 做类型 - 检查。`check:tests` 只是第一刀;产品面加同样 flag 只报 1 个错误,因此后续改动可以对它使用同样的 - 处理。 +- `evals/`、`scripts/`、`tools/` 中既未被 `tsconfig.json` / `tsconfig.tests.json` 显式 + 包含、也未被其依赖图导入的文件仍在类型检查之外;全目录类型覆盖是独立范围。 diff --git a/docs/design/ci-cd-and-quality.md b/docs/design/ci-cd-and-quality.md index 73aa97ef..13e86ef2 100644 --- a/docs/design/ci-cd-and-quality.md +++ b/docs/design/ci-cd-and-quality.md @@ -57,6 +57,8 @@ exit_criteria: Replace with a stable contract test or remove after the redesign `npm run lint` 的扫描面是 `src/ .pi/extensions/ claude-plugins/ workbuddy-plugin/ tests/ evals/ scripts/ tools/`。两条**活性规则**(`@typescript-eslint/no-unused-vars`、`no-useless-assignment`,都在断言"这个值从未被读取")在 `tests/`、`evals/`、`scripts/`、`tools/` 上只报告为 `warn`,在 `src/` 上仍为 error:静态工具最容易在这里把活代码判成死的(`tests/core/graph-cycles.test.ts` 的三个"未使用绑定"实际是漏掉的断言),阻塞门禁会把修法推向删掉线索。其余规则对开发面与 `src/` 同标准(决策:[静态覆盖面](../decisions/implemented/2026-09-16-ci-static-coverage.md))。测试、研究 harness、脚本与仓库工具按设计打印,因此 `no-console` 对这些面关闭。扫描面自身由 `tests/tools/eslint-config-coverage.test.ts` 守住:`lint` 与 `lint:fix` 同面,config 里每个以目录锚定的 `files:` 块都必须落在扫描面内并仍能匹配到文件,且生效 severity 由 ESLint 自己回答(提升某条规则是一次刻意改动)。 +`format` 与 `format:check` 的 `**/*.ts` 面为 `src/`、`.pi/`、`workbuddy-plugin/`、`tests/`、`evals/`、`scripts/`、`tools/`;暂存格式 hook 与它们同面,子包保留独立流程。守卫测试锁定两条命令及 hook 一致,评估 fixture 的故意错误行为不因格式化而修正([静态覆盖面决策](../decisions/implemented/2026-09-16-ci-static-coverage.md))。 + `npm run check:tests`(`tsc -p tsconfig.tests.json --noUnusedLocals --noUnusedParameters`)进入 `verify:static` 与 `ci-and-tests` 的阻塞集合。它检查 `tests/`、配置显式包含的源文件以及它们导入的依赖;不声明覆盖全部 `evals/`、`scripts/`、`tools/`。 `verify:static` 中的 `mutation:anchors`(`tools/mutation-teeth.ts --anchors-only`)只做一件事:把 110 颗具名 mutant 的位置全部解析一遍,不跑任何用例、不写任何字节,在 1 秒内回答“每一颗牙是否还瞄着东西”。它进入静态契约是因为**一颗锚点失效时没有别的检查会注意到**:全量 sweep 不跑(`mutation:teeth` 不在任何 CI 作业里),而一颗匹配不到位置的牙在 sweep 报告里只是“不可应用”并被排除出分母——本轮修掉的两颗牙就是这样悄无声息地停摆的。全量 sweep 仍然不进闸门:它是分钟级、要跑用例,属于推送前的常设规则([决策](../decisions/implemented/2026-09-24-mutants-are-derived-not-anchored.md))。 diff --git a/evals/agent-telemetry.ts b/evals/agent-telemetry.ts index 7de6853d..96848629 100644 --- a/evals/agent-telemetry.ts +++ b/evals/agent-telemetry.ts @@ -22,9 +22,7 @@ export interface AgentRunTelemetry { * turn. Token usage is the sum of every assistant message in the prompt so * retries and multi-turn tool loops are not silently dropped. */ -export function collectAgentRunTelemetry( - events: readonly AgentSessionEvent[], -): AgentRunTelemetry { +export function collectAgentRunTelemetry(events: readonly AgentSessionEvent[]): AgentRunTelemetry { const tokenUsage: AgentTokenUsage = { input: 0, output: 0, diff --git a/evals/benchmarks/loaders.ts b/evals/benchmarks/loaders.ts index ffb5e14b..bce8134b 100644 --- a/evals/benchmarks/loaders.ts +++ b/evals/benchmarks/loaders.ts @@ -1,12 +1,7 @@ import { readdirSync, readFileSync, statSync } from "node:fs"; import { basename, join, resolve } from "node:path"; -import type { - BenchmarkCase, - BenchmarkRole, - BenchmarkSession, - BenchmarkTurn, -} from "./types.ts"; +import type { BenchmarkCase, BenchmarkRole, BenchmarkSession, BenchmarkTurn } from "./types.ts"; type JsonObject = Record; @@ -54,10 +49,7 @@ export function loadLocomo(path: string): BenchmarkCase[] { }); } -export function loadPersonaMem( - questionsPath: string, - contextsPath: string, -): BenchmarkCase[] { +export function loadPersonaMem(questionsPath: string, contextsPath: string): BenchmarkCase[] { const contexts = loadPersonaContexts(contextsPath); return parseCsv(readFileSync(resolve(questionsPath), "utf8")).map((row) => { const contextId = requiredString(row.shared_context_id, "shared_context_id"); @@ -80,10 +72,12 @@ export function loadPersonaMem( topic: row.topic, correctAnswer: row.correct_answer, }, - sessions: [{ - id: contextId, - turns: selected.map((message, index) => toTurn(message, `${contextId}:${index}`)), - }], + sessions: [ + { + id: contextId, + turns: selected.map((message, index) => toTurn(message, `${contextId}:${index}`)), + }, + ], }; }); } @@ -103,7 +97,8 @@ export function loadBeam(chatsDirectory: string): BenchmarkCase[] { turns: asArray(value.turns).flatMap((group, groupIndex) => asArray(group).map((message, messageIndex) => toTurn(message, `${chatId}:${batchNumber}:${groupIndex}:${messageIndex}`), - )), + ), + ), } satisfies BenchmarkSession; }); const probing = asObject(readJson(probingPath)); @@ -126,31 +121,25 @@ export function loadBeam(chatsDirectory: string): BenchmarkCase[] { }); } -export function stratifiedSample( - cases: BenchmarkCase[], - perCategory: number, -): BenchmarkCase[] { +export function stratifiedSample(cases: BenchmarkCase[], perCategory: number): BenchmarkCase[] { const grouped = new Map(); for (const item of cases) { const group = grouped.get(item.category) ?? []; group.push(item); grouped.set(item.category, group); } - return [...grouped.keys()].sort().flatMap( - (category) => grouped.get(category)!.slice(0, perCategory), - ); + return [...grouped.keys()] + .sort() + .flatMap((category) => grouped.get(category)!.slice(0, perCategory)); } function loadPersonaContexts(path: string): Map { const result = new Map(); - for (const [lineIndex, line] of readFileSync(resolve(path), "utf8") - .split(/\r?\n/u).entries()) { + for (const [lineIndex, line] of readFileSync(resolve(path), "utf8").split(/\r?\n/u).entries()) { if (!line.trim()) continue; const parsed: unknown = JSON.parse(line); if (Array.isArray(parsed)) { - const first = parsed[0] && typeof parsed[0] === "object" - ? asObject(parsed[0]) - : {}; + const first = parsed[0] && typeof parsed[0] === "object" ? asObject(parsed[0]) : {}; const id = stringValue(first.shared_context_id ?? first.context_id ?? first.id); if (!id) throw new Error(`PersonaMem context line ${lineIndex + 1} has no id`); result.set(id, parsed); @@ -176,13 +165,16 @@ function toTurn(value: unknown, sourceId: string): BenchmarkTurn { return { role: normalizeRole(object.role ?? object.speaker), ...(stringValue(object.speaker) ? { speaker: stringValue(object.speaker)! } : {}), - content: typeof content === "string" - ? content - : asArray(content).map((part) => { - if (typeof part === "string") return part; - const item = asObject(part); - return stringValue(item.text) ?? ""; - }).join("\n"), + content: + typeof content === "string" + ? content + : asArray(content) + .map((part) => { + if (typeof part === "string") return part; + const item = asObject(part); + return stringValue(item.text) ?? ""; + }) + .join("\n"), sourceId: stringValue(object.id ?? object.index) ?? sourceId, officialMetadata: { ...object }, }; @@ -224,18 +216,18 @@ function parseCsv(input: string): Record[] { rows.push(row); } const headers = rows.shift() ?? []; - return rows.map((values) => Object.fromEntries( - headers.map((header, index) => [header.replace(/^\uFEFF/u, ""), values[index] ?? ""]), - )); + return rows.map((values) => + Object.fromEntries( + headers.map((header, index) => [header.replace(/^\uFEFF/u, ""), values[index] ?? ""]), + ), + ); } function parseOptions(value: string): string[] | undefined { const trimmed = value.trim(); if (!trimmed) return undefined; try { - const normalized = trimmed.startsWith("[") - ? trimmed.replaceAll("'", '"') - : trimmed; + const normalized = trimmed.startsWith("[") ? trimmed.replaceAll("'", '"') : trimmed; const parsed: unknown = JSON.parse(normalized); return Array.isArray(parsed) ? parsed.map(String) : undefined; } catch { @@ -261,9 +253,7 @@ function findFiles(directory: string, name: string): string[] { if (!statSync(directory, { throwIfNoEntry: false })?.isDirectory()) return []; return readdirSync(directory).flatMap((entry) => { const path = join(directory, entry); - return statSync(path).isDirectory() - ? findFiles(path, name) - : entry === name ? [path] : []; + return statSync(path).isDirectory() ? findFiles(path, name) : entry === name ? [path] : []; }); } @@ -280,9 +270,7 @@ function asArray(value: unknown): unknown[] { } function asObject(value: unknown): JsonObject { - return value && typeof value === "object" && !Array.isArray(value) - ? value as JsonObject - : {}; + return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonObject) : {}; } function stringValue(value: unknown): string | undefined { diff --git a/evals/consolidation/run.ts b/evals/consolidation/run.ts index 6823682a..3b0cb682 100644 --- a/evals/consolidation/run.ts +++ b/evals/consolidation/run.ts @@ -95,14 +95,16 @@ export function evaluateLocomoConsolidation( } const counts = [...uses.values()]; const eligible = counts.filter((supported) => - consolidationEligible(posteriorAfterOutcomes(priorConfidence, supported, 0), policy) + consolidationEligible(posteriorAfterOutcomes(priorConfidence, supported, 0), policy), ); const repeated = counts.filter((count) => count >= policy.minimumIndependentVotes); const reversal = eligible.map((supported) => - contradictionsToRetract(priorConfidence, supported, policy) + contradictionsToRetract(priorConfidence, supported, policy), ); const reversalHistogram = Object.fromEntries( - [...new Set(reversal)].sort(numberOrNull).map((count) => [String(count), reversal.filter((x) => x === count).length]), + [...new Set(reversal)] + .sort(numberOrNull) + .map((count) => [String(count), reversal.filter((x) => x === count).length]), ); return { protocol: "nmg.stg-consolidation-locomo.v1", @@ -160,5 +162,7 @@ function dataPath(): string { if (import.meta.url === `file:///${process.argv[1]?.replaceAll("\\", "/")}`) { const policy = configuredStgConsolidationPolicy(process.env); - process.stdout.write(`${JSON.stringify(evaluateLocomoConsolidation(dataPath(), 0.5, policy), null, 2)}\n`); + process.stdout.write( + `${JSON.stringify(evaluateLocomoConsolidation(dataPath(), 0.5, policy), null, 2)}\n`, + ); } diff --git a/evals/controller-shadow/independence.ts b/evals/controller-shadow/independence.ts index 327954ca..b7db132f 100644 --- a/evals/controller-shadow/independence.ts +++ b/evals/controller-shadow/independence.ts @@ -15,9 +15,7 @@ export interface IndependentGroup { * A split may never separate rows sharing either identity, including transitive * links (session A -> task X -> session B). */ -export function independentGroups( - rows: readonly IndependentRowIdentity[], -): IndependentGroup[] { +export function independentGroups(rows: readonly IndependentRowIdentity[]): IndependentGroup[] { const parent = rows.map((_, index) => index); const find = (index: number): number => { while (parent[index] !== index) { diff --git a/evals/controller-shadow/tau-worker.ts b/evals/controller-shadow/tau-worker.ts index c1e527b6..a5e65a6e 100644 --- a/evals/controller-shadow/tau-worker.ts +++ b/evals/controller-shadow/tau-worker.ts @@ -50,11 +50,13 @@ export function calibrateRollingTau( const baseline = evaluate(validation, previousThreshold); const candidate = evaluate(validation, threshold); const blockers: string[] = []; - if (usable.length < MIN_TOTAL_ROWS) blockers.push(`requires at least ${MIN_TOTAL_ROWS} labelled rows`); + if (usable.length < MIN_TOTAL_ROWS) + blockers.push(`requires at least ${MIN_TOTAL_ROWS} labelled rows`); if (validation.length < MIN_VALIDATION_ROWS) { blockers.push(`requires at least ${MIN_VALIDATION_ROWS} held-out rows`); } - if (!hasBothLabels(train)) blockers.push("training window requires positive and negative expansion labels"); + if (!hasBothLabels(train)) + blockers.push("training window requires positive and negative expansion labels"); if (!hasBothLabels(validation)) { blockers.push("held-out window requires positive and negative expansion labels"); } @@ -86,7 +88,11 @@ export function calibrateRollingTau( function bestThreshold(rows: readonly ShadowDatasetRow[], fallback: number): number { if (!rows.length || !hasBothLabels(rows)) return fallback; const scores = [...new Set(rows.map(qpp))].sort((left, right) => left - right); - const candidates = [0, ...scores.map((score, index) => (score + (scores[index + 1] ?? 1)) / 2), 1]; + const candidates = [ + 0, + ...scores.map((score, index) => (score + (scores[index + 1] ?? 1)) / 2), + 1, + ]; return candidates.reduce((best, candidate) => { const next = evaluate(rows, candidate); const current = evaluate(rows, best); @@ -126,8 +132,10 @@ function evaluate(rows: readonly ShadowDatasetRow[], threshold: number): TauMetr } function hasBothLabels(rows: readonly ShadowDatasetRow[]): boolean { - return rows.some((row) => row.feedback.expansionUseful === true) && - rows.some((row) => row.feedback.expansionUseful === false); + return ( + rows.some((row) => row.feedback.expansionUseful === true) && + rows.some((row) => row.feedback.expansionUseful === false) + ); } function qpp(row: ShadowDatasetRow): number { @@ -138,7 +146,11 @@ function bounded(value: number, minimum: number, maximum: number): number { return Math.min(maximum, Math.max(minimum, value)); } -function fingerprint(rows: readonly ShadowDatasetRow[], previous: number, candidate: number): string { +function fingerprint( + rows: readonly ShadowDatasetRow[], + previous: number, + candidate: number, +): string { return createHash("sha256") .update( JSON.stringify({ diff --git a/evals/gate/run.ts b/evals/gate/run.ts index f39a263a..036e8d39 100644 --- a/evals/gate/run.ts +++ b/evals/gate/run.ts @@ -36,7 +36,12 @@ const cases: GateCase[] = [ const rows = cases.map((item) => { const decision = decideMemoryLoad(item.prompt); const predictedRecall = decision.mode === "retrieve"; - return { ...item, mode: decision.mode, predictedRecall, correct: predictedRecall === item.needsRecall }; + return { + ...item, + mode: decision.mode, + predictedRecall, + correct: predictedRecall === item.needsRecall, + }; }); const languages = [...new Set(rows.map((row) => row.language))]; diff --git a/evals/halumem/agent-extract.ts b/evals/halumem/agent-extract.ts index c3538824..4865cf8f 100644 --- a/evals/halumem/agent-extract.ts +++ b/evals/halumem/agent-extract.ts @@ -54,12 +54,19 @@ export function parseAgentExtraction(raw: string): AgentExtractedMemory[] { const row = value as Record; const statement = String(row.statement ?? "").trim(); const evidence = String(row.evidence ?? "").trim(); - const rawMemoryType = String(row.memoryType ?? "").trim().toLowerCase(); + const rawMemoryType = String(row.memoryType ?? "") + .trim() + .toLowerCase(); const memoryType = - ({ goal: "fact", decision: "fact", persona: "fact", relationship: "fact", procedure: "strategy" } as Record< - string, - string - >)[rawMemoryType] ?? rawMemoryType; + ( + { + goal: "fact", + decision: "fact", + persona: "fact", + relationship: "fact", + procedure: "strategy", + } as Record + )[rawMemoryType] ?? rawMemoryType; if (!statement || !evidence || !allowed.has(memoryType)) { throw new Error(`invalid memory at index ${index}`); } @@ -79,9 +86,7 @@ async function main(): Promise { const input = resolve( args.input ?? ".benchmarks/official/OmniMemEval/data/halumem/HaluMem-Medium.jsonl", ); - const output = resolve( - args.output ?? ".benchmarks/halumem-nmg/results/agent-extractions.jsonl", - ); + const output = resolve(args.output ?? ".benchmarks/halumem-nmg/results/agent-extractions.jsonl"); const cacheDir = resolve(args.cacheDir ?? ".benchmarks/halumem-nmg/extraction-cache"); const maxUsers = positive(args.users, 1); const throughSession = positive(args.throughSession, 1); @@ -113,7 +118,10 @@ async function main(): Promise { for (let index = 0; index < Math.min(throughSession, user.sessions.length); index += 1) { const dialogue = user.sessions[index]!.dialogue; const dialogueHash = digest(JSON.stringify(dialogue)); - const cachePath = resolve(cacheDir, `${digest(`${model}\0${policyHash}\0${dialogueHash}`)}.json`); + const cachePath = resolve( + cacheDir, + `${digest(`${model}\0${policyHash}\0${dialogueHash}`)}.json`, + ); let memories: AgentExtractedMemory[]; if (existsSync(cachePath)) { memories = JSON.parse(readFileSync(cachePath, "utf8")) as AgentExtractedMemory[]; @@ -149,7 +157,14 @@ async function main(): Promise { } writeFileSync(cachePath, JSON.stringify(memories, null, 2), "utf8"); } - rows.push({ uuid: user.uuid, sessionIndex: index + 1, dialogueHash, policyHash, model, memories }); + rows.push({ + uuid: user.uuid, + sessionIndex: index + 1, + dialogueHash, + policyHash, + model, + memories, + }); } users += 1; } diff --git a/evals/halumem/prepare.ts b/evals/halumem/prepare.ts index 85a32fb2..ccce093c 100644 --- a/evals/halumem/prepare.ts +++ b/evals/halumem/prepare.ts @@ -96,9 +96,9 @@ export async function prepareHaluMem(options: PrepareOptions): Promise { sessionIndex: number; memories: Array<{ statement: string }>; }; - result.set(`${row.uuid}:${row.sessionIndex}`, row.memories.map((memory) => memory.statement)); + result.set( + `${row.uuid}:${row.sessionIndex}`, + row.memories.map((memory) => memory.statement), + ); } return result; } diff --git a/evals/halumem/promotion-audit.ts b/evals/halumem/promotion-audit.ts index c5dfacb6..1484f819 100644 --- a/evals/halumem/promotion-audit.ts +++ b/evals/halumem/promotion-audit.ts @@ -61,7 +61,8 @@ export function parsePromotionVotes( const fenced = /```(?:json)?\s*(\{[\s\S]*\})\s*```/iu.exec(raw.trim()); const source = fenced?.[1] ?? raw.slice(raw.indexOf("{"), raw.lastIndexOf("}") + 1); const parsed = JSON.parse(source) as { votes?: unknown }; - if (!Array.isArray(parsed.votes)) throw new Error("promotion audit response must contain votes[]"); + if (!Array.isArray(parsed.votes)) + throw new Error("promotion audit response must contain votes[]"); const seen = new Set(); return parsed.votes.map((value, index) => { const row = value as Record; @@ -79,7 +80,8 @@ export function parsePromotionVotes( const attributable = dialogue.some( (turn) => turn.role === "user" && turn.content.includes(evidence), ); - if (!attributable) throw new Error(`vote evidence is not an exact user excerpt: ${candidateId}`); + if (!attributable) + throw new Error(`vote evidence is not an exact user excerpt: ${candidateId}`); seen.add(candidateId); return { candidateId, outcome, evidence }; }); @@ -107,9 +109,7 @@ async function main(): Promise { const output = resolve( args.output ?? ".benchmarks/halumem-nmg/results/promotion-qualified-extractions.jsonl", ); - const reportPath = resolve( - args.report ?? ".benchmarks/halumem-nmg/results/promotion-audit.json", - ); + const reportPath = resolve(args.report ?? ".benchmarks/halumem-nmg/results/promotion-audit.json"); const cacheDir = resolve(args.cacheDir ?? ".benchmarks/halumem-nmg/promotion-cache"); const dataDir = resolve(args.dataDir ?? ".benchmarks/halumem-nmg/promotion-store"); const originStart = positive(args.originStart, 1); @@ -129,7 +129,9 @@ async function main(): Promise { const users = readFileSync(input, "utf8") .split(/\r?\n/u) .filter(Boolean) - .map((line) => JSON.parse(line) as { uuid: string; sessions: Array<{ dialogue: DialogueTurn[] }> }); + .map( + (line) => JSON.parse(line) as { uuid: string; sessions: Array<{ dialogue: DialogueTurn[] }> }, + ); const user = users[positive(args.user, 1) - 1]; if (!user) throw new Error("requested user does not exist"); const extractionRows = readFileSync(extractionsPath, "utf8") @@ -208,7 +210,9 @@ async function main(): Promise { session, }); for (const vote of votes) { - candidates.find((candidate) => candidate.candidateId === vote.candidateId)!.votes.push(vote); + candidates + .find((candidate) => candidate.candidateId === vote.candidateId)! + .votes.push(vote); } const byOrigin = new Map(); for (const vote of votes) { @@ -279,13 +283,17 @@ async function main(): Promise { .map((row) => ({ ...row, memories: - row.sessionIndex >= originStart - ? (qualifiedBySession.get(row.sessionIndex) ?? []) - : [], + row.sessionIndex >= originStart ? (qualifiedBySession.get(row.sessionIndex) ?? []) : [], })); - writeFileSync(output, `${qualifiedRows.map((row) => JSON.stringify(row)).join("\n")}\n`, "utf8"); + writeFileSync( + output, + `${qualifiedRows.map((row) => JSON.stringify(row)).join("\n")}\n`, + "utf8", + ); writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`, "utf8"); - process.stdout.write(`${JSON.stringify({ ...report, candidates: undefined, output, report: reportPath }, null, 2)}\n`); + process.stdout.write( + `${JSON.stringify({ ...report, candidates: undefined, output, report: reportPath }, null, 2)}\n`, + ); } finally { service.close(); } @@ -306,7 +314,10 @@ async function auditSession(input: { statement: candidate.memory.statement, })); const payload = JSON.stringify({ candidates: candidatePayload, userMessages: userDialogue }); - const cachePath = resolve(input.cacheDir, `${digest(`${input.model}\0${SYSTEM_PREFIX}\0${payload}`)}.json`); + const cachePath = resolve( + input.cacheDir, + `${digest(`${input.model}\0${SYSTEM_PREFIX}\0${payload}`)}.json`, + ); if (existsSync(cachePath)) return JSON.parse(readFileSync(cachePath, "utf8")) as CandidateVote[]; const response = await fetch(`${input.baseUrl}/chat/completions`, { method: "POST", @@ -323,7 +334,9 @@ async function auditSession(input: { }), }); if (!response.ok) throw new Error(`promotion audit failed with HTTP ${response.status}`); - const body = (await response.json()) as { choices?: Array<{ message?: { content?: string | null } }> }; + const body = (await response.json()) as { + choices?: Array<{ message?: { content?: string | null } }>; + }; const votes = parsePromotionVotes( body.choices?.[0]?.message?.content ?? "", new Set(input.candidates.map((candidate) => candidate.candidateId)), diff --git a/evals/local-env.ts b/evals/local-env.ts index 6c9cde71..e7c8b07a 100644 --- a/evals/local-env.ts +++ b/evals/local-env.ts @@ -1,10 +1,7 @@ import { existsSync, readFileSync } from "node:fs"; import { resolve } from "node:path"; -const BENCHMARK_SECRET_KEYS = new Set([ - "DEEPSEEK_API_KEY", - "OPENCODE_API_KEY", -]); +const BENCHMARK_SECRET_KEYS = new Set(["DEEPSEEK_API_KEY", "OPENCODE_API_KEY"]); /** * Load only benchmark model credentials from the repository-local ignored diff --git a/evals/longmemeval/official.ts b/evals/longmemeval/official.ts index aa4513eb..20c32290 100644 --- a/evals/longmemeval/official.ts +++ b/evals/longmemeval/official.ts @@ -46,10 +46,7 @@ export function scoreLongMemRetrieval( (sum, id, index) => sum + (relevant.has(id) ? 1 / Math.log2(index + 2) : 0), 0, ); - const ideal = [...relevant].reduce( - (sum, _id, index) => sum + 1 / Math.log2(index + 2), - 0, - ); + const ideal = [...relevant].reduce((sum, _id, index) => sum + 1 / Math.log2(index + 2), 0); return { recallAny, recallAll, recall, ndcg: ideal === 0 ? 0 : dcg / ideal }; } @@ -66,9 +63,10 @@ export function longMemEvalJudgePrompt( if (task === "single-session-preference") { return `I will give you a question, a rubric for desired personalized response, and a response from a model. Please answer yes if the response satisfies the desired response. Otherwise, answer no. The model does not need to reflect all the points in the rubric. The response is correct as long as it recalls and utilizes the user's personal information correctly.\n\nQuestion: ${question}\n\nRubric: ${answer}\n\nModel Response: ${response}\n\nIs the model response correct? Answer yes or no only.`; } - const updateRule = task === "knowledge-update" - ? " If the response contains some previous information along with an updated answer, the response should be considered as correct as long as the updated answer is the required answer." - : " If the response is equivalent to the correct answer or contains all the intermediate steps to get the correct answer, you should also answer yes. If the response only contains a subset of the information required by the answer, answer no."; + const updateRule = + task === "knowledge-update" + ? " If the response contains some previous information along with an updated answer, the response should be considered as correct as long as the updated answer is the required answer." + : " If the response is equivalent to the correct answer or contains all the intermediate steps to get the correct answer, you should also answer yes. If the response only contains a subset of the information required by the answer, answer no."; return `I will give you a question, a correct answer, and a response from a model. Please answer yes if the response contains the correct answer. Otherwise, answer no.${updateRule}\n\nQuestion: ${question}\n\nCorrect Answer: ${answer}\n\nModel Response: ${response}\n\nIs the model response correct? Answer yes or no only.`; } diff --git a/evals/longmemeval/retrieval-evidence.ts b/evals/longmemeval/retrieval-evidence.ts index 52c7fa7d..486d5f35 100644 --- a/evals/longmemeval/retrieval-evidence.ts +++ b/evals/longmemeval/retrieval-evidence.ts @@ -23,11 +23,8 @@ export function officialRetrievalForMemoryIds( ): OfficialRetrievalMetrics | null { const store = new NmgStore(resolve(nmgDirectory, "nmg.sqlite")); try { - return officialMetricsForContext( - store.getContext(memoryIds, 0), - questionId, - answerSessionIds, - ).officialMetrics; + return officialMetricsForContext(store.getContext(memoryIds, 0), questionId, answerSessionIds) + .officialMetrics; } finally { store.close(); } @@ -50,9 +47,10 @@ export function latestAutomaticRecallEvidence( ) .get() as { id?: unknown; session_id?: unknown } | undefined; traceId = row?.id === undefined ? null : String(row.id); - sessionId = row?.session_id === null || row?.session_id === undefined - ? undefined - : String(row.session_id); + sessionId = + row?.session_id === null || row?.session_id === undefined + ? undefined + : String(row.session_id); } finally { database.close(); } diff --git a/evals/memory-quality/run.ts b/evals/memory-quality/run.ts index e7cab009..5162100a 100644 --- a/evals/memory-quality/run.ts +++ b/evals/memory-quality/run.ts @@ -37,7 +37,8 @@ try { }); const current = store.search("Atlas current Python", { maxTier: 3, limit: 5 }); return { - passed: current.some((item) => item.memory.id === currentState.memory.id) && + passed: + current.some((item) => item.memory.id === currentState.memory.id) && !current.some((item) => item.memory.id === oldState.memory.id), detail: "new state supersedes the old state in ordinary retrieval", }; @@ -61,9 +62,9 @@ try { derivation: "Aggregation of two independently recorded return events", }); return { - passed: aggregate.memory.evidenceIds.length >= 2 && - store.search("two returned items", { maxTier: 3 })[0]?.memory.id === - aggregate.memory.id, + passed: + aggregate.memory.evidenceIds.length >= 2 && + store.search("two returned items", { maxTier: 3 })[0]?.memory.id === aggregate.memory.id, detail: "derived memory preserves both source evidence chains", }; }); @@ -90,7 +91,8 @@ try { graphHops: 1, }); return { - passed: context.results.some((item) => item.memory.id === left.memory.id) && + passed: + context.results.some((item) => item.memory.id === left.memory.id) && context.results.some((item) => item.memory.id === right.memory.id) && context.relations.some((relation) => relation.type === "contradicts"), detail: "both claims and their typed contradiction remain visible", @@ -184,10 +186,7 @@ const report = { process.stdout.write(`${JSON.stringify(report, null, 2)}\n`); if (report.passed !== report.cases) process.exitCode = 1; -function measure( - category: string, - run: () => { passed: boolean; detail: string }, -): void { +function measure(category: string, run: () => { passed: boolean; detail: string }): void { const started = performance.now(); const result = run(); results.push({ diff --git a/evals/natural-maintenance/audit.ts b/evals/natural-maintenance/audit.ts index 60c47b76..ddbd0651 100644 --- a/evals/natural-maintenance/audit.ts +++ b/evals/natural-maintenance/audit.ts @@ -113,7 +113,9 @@ const EMPTY_STORE_COUNTS = { * Inspect maintenance evidence without opening a writable NMG store. Missing * databases are reported rather than created, and no maintenance actuator runs. */ -export function auditNaturalMaintenance(options: NaturalMaintenanceAuditOptions): NaturalMaintenanceAudit { +export function auditNaturalMaintenance( + options: NaturalMaintenanceAuditOptions, +): NaturalMaintenanceAudit { const environment = options.environment ?? process.env; const stgPolicy = configuredStgConsolidationPolicy(environment); const maintenancePolicy = configuredMaintenancePolicy(environment); @@ -121,14 +123,18 @@ export function auditNaturalMaintenance(options: NaturalMaintenanceAuditOptions) const ltgDetails = ltg.exists ? withReadOnlyDatabase(ltg.path, (db) => auditLtgDetails(db, maintenancePolicy)) : emptyLtgDetails(); - const stg = [...new Set(options.stgPaths ?? [])].map((path) => auditStore(resolve(path), stgPolicy)); + const stg = [...new Set(options.stgPaths ?? [])].map((path) => + auditStore(resolve(path), stgPolicy), + ); const evidenceGaps: string[] = []; const naturalClaimEvents = stg.reduce((sum, store) => sum + store.claims.naturalOutcomeEvents, 0); const candidates = stg.reduce((sum, store) => sum + store.claims.promotionCandidates.length, 0); - if (stg.length === 0 || stg.every((store) => !store.exists)) evidenceGaps.push("no_stg_store_observed"); + if (stg.length === 0 || stg.every((store) => !store.exists)) + evidenceGaps.push("no_stg_store_observed"); if (naturalClaimEvents === 0) evidenceGaps.push("no_stg_claim_outcomes"); if (candidates === 0) evidenceGaps.push("no_stg_consolidation_candidates"); - if (ltgDetails.consolidatedFromStg.length === 0) evidenceGaps.push("no_materialized_stg_to_ltg_examples"); + if (ltgDetails.consolidatedFromStg.length === 0) + evidenceGaps.push("no_materialized_stg_to_ltg_examples"); if (ltgDetails.stgConsolidation.retracted === 0) { evidenceGaps.push("no_stg_consolidation_retractions"); } @@ -157,7 +163,12 @@ export function auditNaturalMaintenance(options: NaturalMaintenanceAuditOptions) function auditStore(path: string, policy: StgConsolidationPolicyConfig): StoreAudit { if (!existsSync(path)) { - return { path, exists: false, ...structuredClone(EMPTY_STORE_COUNTS), warnings: ["database_missing"] }; + return { + path, + exists: false, + ...structuredClone(EMPTY_STORE_COUNTS), + warnings: ["database_missing"], + }; } return withReadOnlyDatabase(path, (db) => { const warnings: string[] = []; @@ -187,19 +198,21 @@ function auditStore(path: string, policy: StgConsolidationPolicyConfig): StoreAu ) : {}; const hasCollectionOrigin = columnExists(db, "claim_outcome_events", "collection_origin"); - const outcomeOrigins = tableExists(db, "claim_outcome_events") && hasCollectionOrigin - ? allRows( - db, - `SELECT collection_origin AS origin, COUNT(*) AS events, + const outcomeOrigins = + tableExists(db, "claim_outcome_events") && hasCollectionOrigin + ? allRows( + db, + `SELECT collection_origin AS origin, COUNT(*) AS events, COUNT(DISTINCT semantic_task_id) AS tasks FROM claim_outcome_events GROUP BY collection_origin`, - ) - : []; + ) + : []; const outcomeEventsByOrigin = Object.fromEntries( outcomeOrigins.map((row) => [String(row.origin), numberValue(row.events)]), ); const naturalOutcome = outcomeOrigins.find((row) => String(row.origin) === "natural"); - if (!tableExists(db, "claim_outcome_events")) warnings.push("claim_outcome_events_table_missing"); + if (!tableExists(db, "claim_outcome_events")) + warnings.push("claim_outcome_events_table_missing"); const grouped = groupPosteriors(posteriorRows); return { path, @@ -218,10 +231,16 @@ function auditStore(path: string, policy: StgConsolidationPolicyConfig): StoreAu posteriors: posteriorRows.length, memoriesWithPosteriors: grouped.size, promotionCandidates: [...grouped.entries()] - .filter(([, claims]) => claims.length > 0 && claims.every((claim) => qualifiesForPromotion(claim, policy))) + .filter( + ([, claims]) => + claims.length > 0 && claims.every((claim) => qualifiesForPromotion(claim, policy)), + ) .map(([memoryId]) => memoryId), belowRetention: [...grouped.entries()] - .filter(([, claims]) => claims.length > 0 && claims.some((claim) => !qualifiesForRetention(claim, policy))) + .filter( + ([, claims]) => + claims.length > 0 && claims.some((claim) => !qualifiesForRetention(claim, policy)), + ) .map(([memoryId]) => memoryId), }, warnings, @@ -264,17 +283,23 @@ function auditLtgDetails( accessDueNodes, largestNodeWrites: Math.max(0, ...backlogRows.map((row) => numberValue(row.writes))), largestNodeAccesses: Math.max(0, ...backlogRows.map((row) => numberValue(row.accesses))), - distributedWritePressure: indexDeltas >= maintenancePolicy.writeThreshold && writeDueNodes === 0, - distributedAccessPressure: pendingAccesses >= maintenancePolicy.accessThreshold && accessDueNodes === 0, + distributedWritePressure: + indexDeltas >= maintenancePolicy.writeThreshold && writeDueNodes === 0, + distributedAccessPressure: + pendingAccesses >= maintenancePolicy.accessThreshold && accessDueNodes === 0, }; const proposals = tableExists(db, "topology_proposals") ? allRows(db, "SELECT * FROM topology_proposals ORDER BY created_at, id") : []; - const transforms = tableExists(db, "node_transforms") ? allRows(db, "SELECT * FROM node_transforms") : []; + const transforms = tableExists(db, "node_transforms") + ? allRows(db, "SELECT * FROM node_transforms") + : []; const journals = tableExists(db, "node_transform_journals") ? allRows(db, "SELECT * FROM node_transform_journals") : []; - const relations = tableExists(db, "node_relations") ? allRows(db, "SELECT * FROM node_relations") : []; + const relations = tableExists(db, "node_relations") + ? allRows(db, "SELECT * FROM node_relations") + : []; const stgMaterializations = tableExists(db, "memory_records") ? allRows(db, "SELECT id, status, markers_json FROM memory_records").flatMap((row) => { const sourceMemoryId = consolidatedSource(row.markers_json); @@ -311,7 +336,9 @@ function auditLtgDetails( proposalsByStatus: countBy(proposals, "status"), proposalsByRelation: countBy(proposals, "relation_type", "none"), pendingAutomaticMergeAssessments: proposals - .filter((row) => String(row.status) === "pending" && String(row.relation_type) === "same_as") + .filter( + (row) => String(row.status) === "pending" && String(row.relation_type) === "same_as", + ) .map((row) => assessAutomaticMerge(db, row)), relationsByType: countBy(relations, "relation_type"), transformsByType: countBy(transforms, "transform_type"), @@ -365,17 +392,22 @@ function assessAutomaticMerge(db: DatabaseSync, proposal: Row): AutomaticMergeAu if (numberValue(proposal.observations) < 5) reasons.push("insufficient_observations"); if (numberValue(proposal.estimated_gain) < 0.98) reasons.push("insufficient_confidence"); if (evidenceMemoryIds.length < 4) reasons.push("insufficient_evidence_memories"); - const evidenceRows = evidenceMemoryIds.map((id) => - db - .prepare( - "SELECT node_id, scope_json, status, source_actor FROM memory_records WHERE id = ?", - ) - .get(id) as Row | undefined, + const evidenceRows = evidenceMemoryIds.map( + (id) => + db + .prepare( + "SELECT node_id, scope_json, status, source_actor FROM memory_records WHERE id = ?", + ) + .get(id) as Row | undefined, ); if (evidenceRows.some((row) => !row || String(row.status) !== "active")) { reasons.push("missing_or_inactive_evidence"); } - if (sourceNodeIds.some((nodeId) => !evidenceRows.some((row) => String(row?.node_id ?? "") === nodeId))) { + if ( + sourceNodeIds.some( + (nodeId) => !evidenceRows.some((row) => String(row?.node_id ?? "") === nodeId), + ) + ) { reasons.push("evidence_not_balanced_across_nodes"); } const evidenceByNode = new Map>(); @@ -399,7 +431,9 @@ function assessAutomaticMerge(db: DatabaseSync, proposal: Row): AutomaticMergeAu ) { reasons.push("source_actor_mismatch_across_nodes"); } - const scopes = new Set(evidenceRows.filter(Boolean).map((row) => String(row?.scope_json ?? "{}"))); + const scopes = new Set( + evidenceRows.filter(Boolean).map((row) => String(row?.scope_json ?? "{}")), + ); if (scopes.size > 1) reasons.push("scope_mismatch"); let targetName: string | null = null; if (scopes.size === 1) { @@ -416,9 +450,10 @@ function assessAutomaticMerge(db: DatabaseSync, proposal: Row): AutomaticMergeAu } if (targetName && tableExists(db, "memory_nodes")) { const identity = canonicalNodeIdentity(targetName); - const duplicate = allRows(db, "SELECT canonical_name FROM memory_nodes WHERE status = 'active'").some( - (row) => canonicalNodeIdentity(String(row.canonical_name)) === identity, - ); + const duplicate = allRows( + db, + "SELECT canonical_name FROM memory_nodes WHERE status = 'active'", + ).some((row) => canonicalNodeIdentity(String(row.canonical_name)) === identity); if (duplicate) reasons.push("target_name_already_active"); } const nodeKey = [...sourceNodeIds].sort().join("\0"); @@ -427,7 +462,12 @@ function assessAutomaticMerge(db: DatabaseSync, proposal: Row): AutomaticMergeAu `SELECT source_node_ids_json FROM topology_proposals WHERE id <> ? AND status = 'pending' AND relation_type IN ('distinct_from', 'contradicts')`, String(proposal.id), - ).some((row) => parseStringArray(row.source_node_ids_json ?? null).sort().join("\0") === nodeKey); + ).some( + (row) => + parseStringArray(row.source_node_ids_json ?? null) + .sort() + .join("\0") === nodeKey, + ); if (competing) reasons.push("competing_conflict_proposal"); return { proposalId: String(proposal.id), @@ -462,7 +502,10 @@ function groupPosteriors(rows: readonly Row[]): Map { return grouped; } -function qualifiesForPromotion(claim: PosteriorAudit, policy: StgConsolidationPolicyConfig): boolean { +function qualifiesForPromotion( + claim: PosteriorAudit, + policy: StgConsolidationPolicyConfig, +): boolean { return ( claim.independentVoteCount >= policy.minimumIndependentVotes && claim.mean >= policy.minimumPosteriorMean && @@ -470,7 +513,10 @@ function qualifiesForPromotion(claim: PosteriorAudit, policy: StgConsolidationPo ); } -function qualifiesForRetention(claim: PosteriorAudit, policy: StgConsolidationPolicyConfig): boolean { +function qualifiesForRetention( + claim: PosteriorAudit, + policy: StgConsolidationPolicyConfig, +): boolean { return ( claim.mean >= policy.minimumRetainedPosteriorMean && claim.lowerBound >= policy.minimumRetainedConservativeLowerBound @@ -485,7 +531,10 @@ function consolidatedSource(value: SQLOutputValue | undefined): string | null { for (const marker of markers) { if (!marker || typeof marker !== "object") continue; const candidate = marker as { kind?: unknown; attributes?: { sourceMemoryId?: unknown } }; - if (candidate.kind === "consolidated_from_stg" && typeof candidate.attributes?.sourceMemoryId === "string") { + if ( + candidate.kind === "consolidated_from_stg" && + typeof candidate.attributes?.sourceMemoryId === "string" + ) { return candidate.attributes.sourceMemoryId; } } @@ -505,7 +554,9 @@ function withReadOnlyDatabase(path: string, run: (db: DatabaseSync) => T): T } function tableExists(db: DatabaseSync, table: string): boolean { - return Boolean(db.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?").get(table)); + return Boolean( + db.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?").get(table), + ); } function columnExists(db: DatabaseSync, table: string, column: string): boolean { @@ -536,7 +587,10 @@ function countBy( const counts: Record = {}; for (const row of rows) { const value = row[key]; - const label = value === null || value === undefined || String(value).length === 0 ? fallback : String(value); + const label = + value === null || value === undefined || String(value).length === 0 + ? fallback + : String(value); counts[label] = (counts[label] ?? 0) + 1; } return counts; diff --git a/evals/natural-readiness/report.ts b/evals/natural-readiness/report.ts index 0a060fd8..e5f56c20 100644 --- a/evals/natural-readiness/report.ts +++ b/evals/natural-readiness/report.ts @@ -183,5 +183,7 @@ if (process.argv[1] && import.meta.url === pathToFileURL(resolve(process.argv[1] }), }); if (options.outputPath) writeJsonAtomic(options.outputPath, packet); - process.stdout.write(`${JSON.stringify({ eventPath, outputPath: options.outputPath ?? null, ...packet }, null, 2)}\n`); + process.stdout.write( + `${JSON.stringify({ eventPath, outputPath: options.outputPath ?? null, ...packet }, null, 2)}\n`, + ); } diff --git a/evals/official/bootstrap.ts b/evals/official/bootstrap.ts index 2569132d..3a68c092 100644 --- a/evals/official/bootstrap.ts +++ b/evals/official/bootstrap.ts @@ -25,7 +25,9 @@ const python = resolve(root, ".benchmarks", "python"); run("uv", ["venv", "--python", "3.11", python]); const executable = officialPythonExecutable(root, {}, process.platform); run("uv", ["pip", "install", "--python", executable, "regex", "numpy", "nltk"]); -process.stdout.write(`Official benchmark sources and Python are ready under ${resolve(root, ".benchmarks")}\n`); +process.stdout.write( + `Official benchmark sources and Python are ready under ${resolve(root, ".benchmarks")}\n`, +); function run(command: string, args: string[], probe = false): boolean { const result = spawnSync(command, args, { cwd: root, stdio: probe ? "ignore" : "inherit" }); diff --git a/evals/official/protocol.ts b/evals/official/protocol.ts index 7a465815..738db962 100644 --- a/evals/official/protocol.ts +++ b/evals/official/protocol.ts @@ -1,18 +1,26 @@ export function personaMemCorrect(hypothesis: string, reference: string): boolean { const expected = reference.toLocaleLowerCase().replace(/[()\s]/gu, ""); const final = hypothesis.includes("") - ? hypothesis.split("").at(-1)!.replace(/<\/final_answer>\s*$/u, "").trim() + ? hypothesis + .split("") + .at(-1)! + .replace(/<\/final_answer>\s*$/u, "") + .trim() : hypothesis.trim(); - return optionLetters(final).size === 1 && optionLetters(final).has(expected) || - optionLetters(hypothesis).size === 1 && optionLetters(hypothesis).has(expected); + return ( + (optionLetters(final).size === 1 && optionLetters(final).has(expected)) || + (optionLetters(hypothesis).size === 1 && optionLetters(hypothesis).has(expected)) + ); } function optionLetters(value: string): Set { const lower = value.toLocaleLowerCase(); const parenthesized = [...lower.matchAll(/\(([a-d])\)/gu)].map((match) => match[1]!); - return new Set(parenthesized.length > 0 - ? parenthesized - : [...lower.matchAll(/\b([a-d])\b/gu)].map((match) => match[1]!)); + return new Set( + parenthesized.length > 0 + ? parenthesized + : [...lower.matchAll(/\b([a-d])\b/gu)].map((match) => match[1]!), + ); } export function beamJudgePrompt(question: string, rubric: string, response: string): string { @@ -70,10 +78,18 @@ QUESTION: ${question} REFERENCE EVENTS (their array indices are stable identifiers): -${JSON.stringify(rubric.map((event, index) => ({ index, event })), null, 2)} +${JSON.stringify( + rubric.map((event, index) => ({ index, event })), + null, + 2, +)} SYSTEM ITEMS (one item per non-empty response line): -${JSON.stringify(systemItems.map((item, index) => ({ index, item })), null, 2)} +${JSON.stringify( + systemItems.map((item, index) => ({ index, item })), + null, + 2, +)} For every system item, in the original order, return {"referenceIndex": , "item": }. Match by semantic equivalence, not exact wording. A reference index may be used at most once. Use null when no unused reference event is equivalent. Return exactly one output object per system item and only the JSON array, for example [{"referenceIndex":0,"item":"first line"},{"referenceIndex":null,"item":"extra line"}].`; } @@ -110,8 +126,7 @@ export function normalizedKendallTauB(reference: number[], candidate: number[]): } const denominator = Math.sqrt( - (concordant + discordant + referenceOnlyTies) * - (concordant + discordant + candidateOnlyTies), + (concordant + discordant + referenceOnlyTies) * (concordant + discordant + candidateOnlyTies), ); if (denominator === 0) return 0; return ((concordant - discordant) / denominator + 1) / 2; diff --git a/evals/official/python.ts b/evals/official/python.ts index 6fc4acf5..99a3751e 100644 --- a/evals/official/python.ts +++ b/evals/official/python.ts @@ -19,6 +19,8 @@ export function probePython(executable: string): { available: boolean; error: st return { available: false, error: - result.error?.message || result.stderr.trim() || `process exited with status ${result.status}`, + result.error?.message || + result.stderr.trim() || + `process exited with status ${result.status}`, }; } diff --git a/evals/official/unified-score.ts b/evals/official/unified-score.ts index 9d24500d..92e30472 100644 --- a/evals/official/unified-score.ts +++ b/evals/official/unified-score.ts @@ -36,7 +36,12 @@ export function scoreEvidenceIds( retrievedIds: readonly string[] | null | undefined, expectedIds: readonly string[] | null | undefined, ): UnifiedEvidenceScore | null { - if (!expectedIds || expectedIds.length === 0 || retrievedIds === null || retrievedIds === undefined) { + if ( + !expectedIds || + expectedIds.length === 0 || + retrievedIds === null || + retrievedIds === undefined + ) { return null; } const expected = new Set(expectedIds); diff --git a/evals/omnimemeval/bridge.ts b/evals/omnimemeval/bridge.ts index d63b7018..1f260a9f 100644 --- a/evals/omnimemeval/bridge.ts +++ b/evals/omnimemeval/bridge.ts @@ -484,37 +484,32 @@ export class OmniMemEvalBridge { if (this.#embeddingClient) { await this.#syncSemanticIndex(store); } - const context = await searchMemoryContext( - store, - this.#embeddingClient, - query, - { - limit, - maxTier: 3, - graphHops: 1, - vectorGranularity: this.#embeddingClient ? "records" : undefined, - secondPass: this.#secondPass, - progressiveWarmDisclosure: false, - tieredDisclosure: true, - initialEvidenceTarget: this.#qppInitialEvidenceTarget, - qppThreshold: this.#qppThreshold, - strongHitTopGap: this.#strongHitTopGap, - strongHitInitialTarget: this.#strongHitInitialTarget, - expandChains: true, - leafBlockRouting: this.#leafBlockRouting, - appendedMaxChars: this.#appendedMaxChars, - chainExpansionMaxChains: this.#chainExpansionMaxChains, - chainExpansionMaxHops: this.#chainExpansionMaxHops, - chainExpansionMaxMemoryHops: this.#chainExpansionMaxMemoryHops, - appendedMaxRatio: this.#appendedMaxRatio, - activeGraphBudget: { - maxNodes: limit, - maxEvidence: limit, - maxTokens: Math.max(1_000, limit * 300), - maxTierBudget: limit, - }, + const context = await searchMemoryContext(store, this.#embeddingClient, query, { + limit, + maxTier: 3, + graphHops: 1, + vectorGranularity: this.#embeddingClient ? "records" : undefined, + secondPass: this.#secondPass, + progressiveWarmDisclosure: false, + tieredDisclosure: true, + initialEvidenceTarget: this.#qppInitialEvidenceTarget, + qppThreshold: this.#qppThreshold, + strongHitTopGap: this.#strongHitTopGap, + strongHitInitialTarget: this.#strongHitInitialTarget, + expandChains: true, + leafBlockRouting: this.#leafBlockRouting, + appendedMaxChars: this.#appendedMaxChars, + chainExpansionMaxChains: this.#chainExpansionMaxChains, + chainExpansionMaxHops: this.#chainExpansionMaxHops, + chainExpansionMaxMemoryHops: this.#chainExpansionMaxMemoryHops, + appendedMaxRatio: this.#appendedMaxRatio, + activeGraphBudget: { + maxNodes: limit, + maxEvidence: limit, + maxTokens: Math.max(1_000, limit * 300), + maxTierBudget: limit, }, - ); + }); const rankedMemoryIds = new Set(context.activeGraph?.memoryIds ?? []); const memories = context.results.map((result) => ({ memoryId: result.memory.id, diff --git a/evals/omnimemeval/judge-provider.ts b/evals/omnimemeval/judge-provider.ts index 2367db27..dfec6115 100644 --- a/evals/omnimemeval/judge-provider.ts +++ b/evals/omnimemeval/judge-provider.ts @@ -58,14 +58,11 @@ Return ONLY JSON with no prose: "supersededMemoryId": "", "reason": ""}`; function buildUserMessage(input: JudgeInput): string { - const lines = [ - `New statement:`, - input.statement, - ``, - `Candidates:`, - ]; + const lines = [`New statement:`, input.statement, ``, `Candidates:`]; for (const c of input.supersedeCandidates ?? input.candidates) { - lines.push(`- id=${c.memoryId} event_time=${c.eventTime ?? "unknown"} similarity=${c.similarity.toFixed(2)}`); + lines.push( + `- id=${c.memoryId} event_time=${c.eventTime ?? "unknown"} similarity=${c.similarity.toFixed(2)}`, + ); lines.push(` ${c.statement.slice(0, 400)}`); } return lines.join("\n"); @@ -73,7 +70,10 @@ function buildUserMessage(input: JudgeInput): string { /** Parse the model's JSON answer defensively; anything unexpected -> keep. */ function parseJudgement(raw: string): DuplicateJudgement { - const trimmed = raw.trim().replace(/^```(?:json)?\s*/i, "").replace(/\s*```$/, ""); + const trimmed = raw + .trim() + .replace(/^```(?:json)?\s*/i, "") + .replace(/\s*```$/, ""); try { const parsed = JSON.parse(trimmed) as { action?: string; @@ -87,7 +87,12 @@ function parseJudgement(raw: string): DuplicateJudgement { return { merge: true, reason }; } if (action === "supersede" && parsed.supersededMemoryId) { - return { merge: false, supersede: true, supersededMemoryId: String(parsed.supersededMemoryId), reason }; + return { + merge: false, + supersede: true, + supersededMemoryId: String(parsed.supersededMemoryId), + reason, + }; } return { merge: false, reason }; } catch { @@ -214,8 +219,6 @@ export function createJudgeClientFromEnv( timeoutMs: Number.isFinite(timeoutMs) ? timeoutMs : 30_000, thinking, reasoningEffort: effort === "low" || effort === "medium" ? effort : "high", - ...(Number.isInteger(maxTokensRaw) && maxTokensRaw > 0 - ? { maxTokens: maxTokensRaw } - : {}), + ...(Number.isInteger(maxTokensRaw) && maxTokensRaw > 0 ? { maxTokens: maxTokensRaw } : {}), }); } diff --git a/evals/omnimemeval/merge-longmemeval-shards.ts b/evals/omnimemeval/merge-longmemeval-shards.ts index 5c037d38..2f239c91 100644 --- a/evals/omnimemeval/merge-longmemeval-shards.ts +++ b/evals/omnimemeval/merge-longmemeval-shards.ts @@ -35,12 +35,10 @@ export function mergeLongMemEvalShards( if (process.argv[1]?.endsWith("merge-longmemeval-shards.ts")) { const [output, ...inputs] = process.argv.slice(2); if (!output || inputs.length === 0) { - throw new Error( - "usage: merge-longmemeval-shards.ts ...", - ); + throw new Error("usage: merge-longmemeval-shards.ts ..."); } - const shards = inputs.map((path) => - JSON.parse(readFileSync(resolve(path), "utf8")) as SearchResults + const shards = inputs.map( + (path) => JSON.parse(readFileSync(resolve(path), "utf8")) as SearchResults, ); const merged = mergeLongMemEvalShards(shards, 500); writeFileSync(resolve(output), `${JSON.stringify(merged, null, 2)}\n`); diff --git a/evals/omnimemeval/research/audits/audit-elbow.ts b/evals/omnimemeval/research/audits/audit-elbow.ts index 6aff9b44..ee53a6d1 100644 --- a/evals/omnimemeval/research/audits/audit-elbow.ts +++ b/evals/omnimemeval/research/audits/audit-elbow.ts @@ -38,16 +38,27 @@ interface Row { async function run(): Promise { const client = createEmbeddingClientFromEnv(); if (!client) { - console.error("no embedding client: set NMG_EMBED_BASE_URL / NMG_EMBED_MODEL / NMG_EMBED_API_KEY"); + console.error( + "no embedding client: set NMG_EMBED_BASE_URL / NMG_EMBED_MODEL / NMG_EMBED_API_KEY", + ); process.exit(1); } const cached = new CachedOmniEmbeddingClient(EMBED_CACHE, client); const cases = loadLocomo(DATA).slice(0, MAX_CASES); - const conversations = new Map(); + const conversations = new Map< + string, + { + sessions: (typeof cases)[0]["sessions"]; + questions: { question: string; evidenceIds: string[] }[]; + } + >(); for (const benchmarkCase of cases) { const key = benchmarkCase.officialMetadata?.sampleId ?? benchmarkCase.sessions[0]?.id ?? "x"; const group = conversations.get(key) ?? { sessions: benchmarkCase.sessions, questions: [] }; - group.questions.push({ question: benchmarkCase.question, evidenceIds: benchmarkCase.evidenceIds }); + group.questions.push({ + question: benchmarkCase.question, + evidenceIds: benchmarkCase.evidenceIds, + }); conversations.set(key, group); } @@ -93,12 +104,21 @@ async function run(): Promise { questions += 1; const diaIds = new Set(caseQa.evidenceIds); const queryVector = (await cached.embedQueries([caseQa.question]))[0]!; - const ctx = store.searchContext(caseQa.question, { - limit: TOPK, - maxTier: 3, - activeGraphBudget: { maxNodes: TOPK, maxEvidence: TOPK, maxTokens: 10_000, maxTierBudget: TOPK }, - vectorGranularity: "records", - }, { queryVector, model: cached.indexId }); + const ctx = store.searchContext( + caseQa.question, + { + limit: TOPK, + maxTier: 3, + activeGraphBudget: { + maxNodes: TOPK, + maxEvidence: TOPK, + maxTokens: 10_000, + maxTierBudget: TOPK, + }, + vectorGranularity: "records", + }, + { queryVector, model: cached.indexId }, + ); const hitAt: number[] = []; ctx.results.forEach((result, index) => { if (diaIds.has(result.memory.scope?.diaId as string)) hitAt.push(index + 1); @@ -106,7 +126,8 @@ async function run(): Promise { const sortedRanks = [...hitAt].sort((a, b) => a - b); const kneed100 = sortedRanks.length > 0 ? sortedRanks[sortedRanks.length - 1]! : 0; const target80 = Math.ceil(diaIds.size * 0.8); - const kneed80 = sortedRanks.length > 0 ? sortedRanks[Math.min(target80, sortedRanks.length) - 1]! : 0; + const kneed80 = + sortedRanks.length > 0 ? sortedRanks[Math.min(target80, sortedRanks.length) - 1]! : 0; rows.push({ scores: ctx.results.map((r) => r.combinedScore), vectorScores: ctx.results.map((r) => r.vectorScore), @@ -117,11 +138,16 @@ async function run(): Promise { }); } store.close(); - if (built % 2 === 0) process.stderr.write(`built ${built}/${conversations.size} | ${questions} questions | ${((Date.now() - start) / 1000).toFixed(0)}s\r`); + if (built % 2 === 0) + process.stderr.write( + `built ${built}/${conversations.size} | ${questions} questions | ${((Date.now() - start) / 1000).toFixed(0)}s\r`, + ); } rmSync(tmp, { recursive: true, force: true }); writeFileSync(JSON_OUT, JSON.stringify(rows)); - console.log(`\nrows: ${rows.length} | ${((Date.now() - start) / 1000).toFixed(0)}s | out: ${JSON_OUT}`); + console.log( + `\nrows: ${rows.length} | ${((Date.now() - start) / 1000).toFixed(0)}s | out: ${JSON_OUT}`, + ); } run().catch((error) => { diff --git a/evals/omnimemeval/research/audits/audit-fibonacci-recall.ts b/evals/omnimemeval/research/audits/audit-fibonacci-recall.ts index 162d28b6..b0236379 100644 --- a/evals/omnimemeval/research/audits/audit-fibonacci-recall.ts +++ b/evals/omnimemeval/research/audits/audit-fibonacci-recall.ts @@ -43,30 +43,57 @@ function hitsInPrefix(results: MemoryContext["results"], k: number, diaIds: Set< async function run(): Promise { const client = createEmbeddingClientFromEnv(); if (!client) { - console.error("no embedding client: set NMG_EMBED_BASE_URL / NMG_EMBED_MODEL / NMG_EMBED_API_KEY"); + console.error( + "no embedding client: set NMG_EMBED_BASE_URL / NMG_EMBED_MODEL / NMG_EMBED_API_KEY", + ); process.exit(1); } const cached = new CachedOmniEmbeddingClient(EMBED_CACHE, client); const cases = loadLocomo(DATA).slice(0, MAX_CASES); - const conversations = new Map(); + const conversations = new Map< + string, + { + sessions: (typeof cases)[0]["sessions"]; + questions: { question: string; evidenceIds: string[] }[]; + } + >(); for (const benchmarkCase of cases) { const key = benchmarkCase.officialMetadata?.sampleId ?? benchmarkCase.sessions[0]?.id ?? "x"; const group = conversations.get(key) ?? { sessions: benchmarkCase.sessions, questions: [] }; - group.questions.push({ question: benchmarkCase.question, evidenceIds: benchmarkCase.evidenceIds }); + group.questions.push({ + question: benchmarkCase.question, + evidenceIds: benchmarkCase.evidenceIds, + }); conversations.set(key, group); } // fixed: per-K aggregates; adaptive: per limit:threshold aggregates const fixedHits = new Map(); // K -> recall values for (const k of KS) fixedHits.set(k, []); - interface AdaptStat { results: number[]; recalls: number[]; triggers: number; stages: number[]; } + interface AdaptStat { + results: number[]; + recalls: number[]; + triggers: number; + stages: number[]; + } const adapt = new Map(); - for (const limit of ADAPTIVE_LIMITS) for (const thr of THRESHOLDS) adapt.set(`${limit}:${thr}`, { results: [], recalls: [], triggers: 0, stages: [] }); + for (const limit of ADAPTIVE_LIMITS) + for (const thr of THRESHOLDS) + adapt.set(`${limit}:${thr}`, { results: [], recalls: [], triggers: 0, stages: [] }); let built = 0; let questions = 0; - const root = mkdtempSync(join(tmpdir(), "nmg-fib-audit-")); const start = Date.now(); - const rows: Array<{ top1: number; nqc: number; c: number; kneed100: number; kneed80: number; numEvidence: number; stageHit: number }> = []; + const root = mkdtempSync(join(tmpdir(), "nmg-fib-audit-")); + const start = Date.now(); + const rows: Array<{ + top1: number; + nqc: number; + c: number; + kneed100: number; + kneed80: number; + numEvidence: number; + stageHit: number; + }> = []; for (const [, group] of conversations) { const storePath = join(STORE_DIR, `case-${built}.sqlite`); @@ -106,12 +133,21 @@ async function run(): Promise { const diaIds = new Set(caseQa.evidenceIds); const queryVector = (await cached.embedQueries([caseQa.question]))[0]!; // ONE retrieval at limit 34; truncate for all K. - const pool = store.searchContext(caseQa.question, { - limit: 34, - maxTier: 3, - activeGraphBudget: { maxNodes: 34, maxEvidence: 34, maxTokens: 10_000, maxTierBudget: 34 }, - vectorGranularity: "records", - }, { queryVector, model: client.indexId }); + const pool = store.searchContext( + caseQa.question, + { + limit: 34, + maxTier: 3, + activeGraphBudget: { + maxNodes: 34, + maxEvidence: 34, + maxTokens: 10_000, + maxTierBudget: 34, + }, + vectorGranularity: "records", + }, + { queryVector, model: client.indexId }, + ); for (const k of KS) { fixedHits.get(k)!.push(hitsInPrefix(pool.results, k, diaIds) / Math.max(diaIds.size, 1)); } @@ -124,7 +160,8 @@ async function run(): Promise { const sortedRanks = [...ranks].sort((a, b) => a - b); const kneed100 = sortedRanks.length > 0 ? sortedRanks[sortedRanks.length - 1]! : 0; const target80 = Math.ceil(diaIds.size * 0.8); - const kneed80 = sortedRanks.length > 0 ? sortedRanks[Math.min(target80, sortedRanks.length) - 1]! : 0; + const kneed80 = + sortedRanks.length > 0 ? sortedRanks[Math.min(target80, sortedRanks.length) - 1]! : 0; // QPP components from the same ranked pool (selection = top 34). const selections = pool.results.map((result, index) => ({ memoryId: result.memory.id, @@ -134,24 +171,46 @@ async function run(): Promise { rank: index + 1, tier: result.memory.tier, estimatedTokens: 1, - scores: { lexical: result.lexicalScore, vector: result.vectorScore, route: result.routeScore, combined: result.combinedScore }, + scores: { + lexical: result.lexicalScore, + vector: result.vectorScore, + route: result.routeScore, + combined: result.combinedScore, + }, })); const comps = computeQppComponents(caseQa.question, qppCandidates(pool.results, selections)); const c = comps.top1 + 0.5 * comps.nqc; const stageHit = KS.find((k) => kneed100 > 0 && k >= kneed100) ?? 0; - rows.push({ top1: comps.top1, nqc: comps.nqc, c, kneed100, kneed80, numEvidence: diaIds.size, stageHit }); + rows.push({ + top1: comps.top1, + nqc: comps.nqc, + c, + kneed100, + kneed80, + numEvidence: diaIds.size, + stageHit, + }); // adaptive walks for (const limit of ADAPTIVE_LIMITS) { - const budget = { maxNodes: limit, maxEvidence: limit, maxTokens: Math.max(1_000, limit * 300), maxTierBudget: limit }; + const budget = { + maxNodes: limit, + maxEvidence: limit, + maxTokens: Math.max(1_000, limit * 300), + maxTierBudget: limit, + }; for (const thr of THRESHOLDS) { const stat = adapt.get(`${limit}:${thr}`)!; - const ctx = store.searchContextWithSecondPass(caseQa.question, { - limit, - maxTier: 3, - qppThreshold: thr, - activeGraphBudget: budget, - vectorGranularity: "records", - }, { queryVector, model: client.indexId }); + const ctx = store.searchContextWithSecondPass( + caseQa.question, + { + limit, + maxTier: 3, + qppThreshold: thr, + activeGraphBudget: budget, + vectorGranularity: "records", + }, + { queryVector, model: client.indexId }, + ); stat.results.push(ctx.results.length); stat.recalls.push(evidenceRecall(ctx.results, diaIds)); if (ctx.activeGraph?.qpp?.trigger === true) stat.triggers += 1; @@ -160,13 +219,18 @@ async function run(): Promise { } } store.close(); - if (built % 2 === 0) process.stderr.write(`built ${built}/${conversations.size} | ${questions} questions | ${((Date.now() - start) / 1000).toFixed(0)}s\r`); + if (built % 2 === 0) + process.stderr.write( + `built ${built}/${conversations.size} | ${questions} questions | ${((Date.now() - start) / 1000).toFixed(0)}s\r`, + ); } if (JSON_OUT) writeFileSync(JSON_OUT, JSON.stringify(rows)); rmSync(root, { recursive: true, force: true }); const mean = (v: number[]): number => (v.length ? v.reduce((a, b) => a + b, 0) / v.length : 0); - console.log(`\nconversations: ${built} | questions: ${questions} | ${((Date.now() - start) / 1000).toFixed(0)}s | model: ${client.indexId}`); + console.log( + `\nconversations: ${built} | questions: ${questions} | ${((Date.now() - start) / 1000).toFixed(0)}s | model: ${client.indexId}`, + ); console.log(`\n=== fixed top-K recall@returned (single retrieval, truncated) ===`); for (const k of KS) { const r = mean(fixedHits.get(k)!); @@ -177,10 +241,14 @@ async function run(): Promise { const n = mean(stat.results); const r = mean(stat.recalls); console.log(`\n=== adaptive limit=${limit} thr=${thr} ===`); - console.log(` results: ${n.toFixed(2)} (min ${Math.min(...stat.results)} max ${Math.max(...stat.results)}) | recall: ${r.toFixed(4)} | trigger: ${stat.triggers}/${stat.questions ?? questions} | stages: ${(stat.stages.reduce((a, b) => a + b, 0) / stat.stages.length).toFixed(2)}`); + console.log( + ` results: ${n.toFixed(2)} (min ${Math.min(...stat.results)} max ${Math.max(...stat.results)}) | recall: ${r.toFixed(4)} | trigger: ${stat.triggers}/${stat.questions ?? questions} | stages: ${(stat.stages.reduce((a, b) => a + b, 0) / stat.stages.length).toFixed(2)}`, + ); // per-record efficiency vs the K=limit fixed point const fixedR = mean(fixedHits.get(Number(limit)) ?? []); - console.log(` vs fixed K=${limit} (recall ${fixedR.toFixed(4)}): delta ${(r - fixedR).toFixed(4)} | recall/record ${(r / Math.max(n, 1)).toFixed(4)}`); + console.log( + ` vs fixed K=${limit} (recall ${fixedR.toFixed(4)}): delta ${(r - fixedR).toFixed(4)} | recall/record ${(r / Math.max(n, 1)).toFixed(4)}`, + ); } } diff --git a/evals/omnimemeval/research/audits/audit-qpp-signal.ts b/evals/omnimemeval/research/audits/audit-qpp-signal.ts index 75ea5b49..c4ebe8ed 100644 --- a/evals/omnimemeval/research/audits/audit-qpp-signal.ts +++ b/evals/omnimemeval/research/audits/audit-qpp-signal.ts @@ -49,7 +49,13 @@ if (!resultsDir) { } const report = JSON.parse(readFileSync(resolve(resultsDir, "report.json"), "utf8")) as { - results: Array<{ questionId: string; mode: string; questionType: string; passed: boolean; retrievalPassed: boolean | null }>; + results: Array<{ + questionId: string; + mode: string; + questionType: string; + passed: boolean; + retrievalPassed: boolean | null; + }>; }; const outcomeByQuestion = new Map(); for (const r of report.results) { @@ -73,7 +79,10 @@ for (const questionId of readdirSorted(armsDir)) { try { for (const repeat of readdirSorted(detDir)) { const candidate = resolve(detDir, repeat, "nmg.sqlite"); - if (exists(candidate)) { dbPath = candidate; break; } + if (exists(candidate)) { + dbPath = candidate; + break; + } } } catch { // no nmg-deterministic arm for this question @@ -82,8 +91,10 @@ for (const questionId of readdirSorted(armsDir)) { const db = new DatabaseSync(dbPath, { readOnly: true }); try { - const typeRows = db.prepare("SELECT id, memory_type FROM memory_records").all() as - Array<{ id: string; memory_type: string }>; + const typeRows = db.prepare("SELECT id, memory_type FROM memory_records").all() as Array<{ + id: string; + memory_type: string; + }>; const memoryType = new Map(typeRows.map((r) => [r.id, r.memory_type])); const traceRows = db .prepare("SELECT query, selections_json AS sels FROM retrieval_traces") @@ -136,21 +147,34 @@ if (rows.length === 0) { // Per-question table. console.log(`\n=== QPP vs outcome (n=${rows.length}, threshold=${DEFAULT_QPP_THRESHOLD}) ===`); -console.log("qid type qpp trig reason nSel top1 var ic rh pass ret"); +console.log( + "qid type qpp trig reason nSel top1 var ic rh pass ret", +); for (const r of rows.sort((a, b) => a.qpp - b.qpp)) { console.log( - pad(r.questionId, 12) + " " + - pad(r.questionType, 20) + " " + - r.qpp.toFixed(2) + " " + - (r.trigger ? "TRIG" : "ok ") + " " + - pad(r.reason, 17) + " " + - pad(String(r.nSelections), 4) + " " + - r.top1.toFixed(2) + " " + - r.variance.toFixed(2) + " " + - r.intentCoverage.toFixed(2) + " " + - r.reasonHealth.toFixed(2) + " " + - (r.passed ? "P" : "F") + " " + - (r.retrievalPassed === null ? "-" : r.retrievalPassed ? "P" : "F"), + pad(r.questionId, 12) + + " " + + pad(r.questionType, 20) + + " " + + r.qpp.toFixed(2) + + " " + + (r.trigger ? "TRIG" : "ok ") + + " " + + pad(r.reason, 17) + + " " + + pad(String(r.nSelections), 4) + + " " + + r.top1.toFixed(2) + + " " + + r.variance.toFixed(2) + + " " + + r.intentCoverage.toFixed(2) + + " " + + r.reasonHealth.toFixed(2) + + " " + + (r.passed ? "P" : "F") + + " " + + (r.retrievalPassed === null ? "-" : r.retrievalPassed ? "P" : "F"), ); } @@ -163,8 +187,12 @@ const ret = (rs: Row[]) => { return m.length === 0 ? NaN : m.filter((r) => r.retrievalPassed).length / m.length; }; console.log("\n=== bucket: Stage-0 trigger decision ==="); -console.log(`would-trigger (qpp<τ or guardrail): n=${trig.length}, answer-acc=${fmt(acc(trig))}, retrieval-acc=${fmt(ret(trig))}`); -console.log(`would-skip (qpp≥τ): n=${ok.length}, answer-acc=${fmt(acc(ok))}, retrieval-acc=${fmt(ret(ok))}`); +console.log( + `would-trigger (qpp<τ or guardrail): n=${trig.length}, answer-acc=${fmt(acc(trig))}, retrieval-acc=${fmt(ret(trig))}`, +); +console.log( + `would-skip (qpp≥τ): n=${ok.length}, answer-acc=${fmt(acc(ok))}, retrieval-acc=${fmt(ret(ok))}`, +); // Bucket: tertiles by qpp. const sorted = [...rows].sort((a, b) => a.qpp - b.qpp); @@ -172,13 +200,19 @@ const t = Math.ceil(sorted.length / 3); const low = sorted.slice(0, t); const high = sorted.slice(sorted.length - t); console.log("\n=== bucket: qpp tertiles ==="); -console.log(`low-qpp: n=${low.length}, qpp∈[${low[0]?.qpp.toFixed(2)},${low.at(-1)?.qpp.toFixed(2)}], answer-acc=${fmt(acc(low))}, retrieval-acc=${fmt(ret(low))}`); -console.log(`high-qpp: n=${high.length}, qpp∈[${high[0]?.qpp.toFixed(2)},${high.at(-1)?.qpp.toFixed(2)}], answer-acc=${fmt(acc(high))}, retrieval-acc=${fmt(ret(high))}`); +console.log( + `low-qpp: n=${low.length}, qpp∈[${low[0]?.qpp.toFixed(2)},${low.at(-1)?.qpp.toFixed(2)}], answer-acc=${fmt(acc(low))}, retrieval-acc=${fmt(ret(low))}`, +); +console.log( + `high-qpp: n=${high.length}, qpp∈[${high[0]?.qpp.toFixed(2)},${high.at(-1)?.qpp.toFixed(2)}], answer-acc=${fmt(acc(high))}, retrieval-acc=${fmt(ret(high))}`, +); // Component health. const allConv = rows.filter((r) => r.intentCoverage === 0.5).length; console.log(`\nintentCoverage neutral (no intent family matched): ${allConv}/${rows.length}`); -console.log(`reasonHealth=0 (all hybrid_match) rows: ${rows.filter((r) => r.reasonHealth === 0).length}/${rows.length}`); +console.log( + `reasonHealth=0 (all hybrid_match) rows: ${rows.filter((r) => r.reasonHealth === 0).length}/${rows.length}`, +); function fmt(x: number): string { return Number.isNaN(x) ? "n/a" : x.toFixed(2); @@ -187,7 +221,11 @@ function readdirSorted(p: string): string[] { return readdirSync(p).sort(); } function exists(p: string): boolean { - try { return statSync(p).isFile(); } catch { return false; } + try { + return statSync(p).isFile(); + } catch { + return false; + } } function pad(s: string, n: number): string { return s.length >= n ? s.slice(0, n) : s + " ".repeat(n - s.length); diff --git a/evals/omnimemeval/research/probes/fake-cache-api.ts b/evals/omnimemeval/research/probes/fake-cache-api.ts index f606521e..fd1b4ad6 100644 --- a/evals/omnimemeval/research/probes/fake-cache-api.ts +++ b/evals/omnimemeval/research/probes/fake-cache-api.ts @@ -26,7 +26,8 @@ function sendJson(response: ServerResponse, status: number, value: unknown): voi async function readBody(request: IncomingMessage): Promise { const chunks: Buffer[] = []; - for await (const chunk of request) chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk)); + for await (const chunk of request) + chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk)); return Buffer.concat(chunks); } @@ -101,7 +102,8 @@ function runCli(): void { process.env.CACHE_PREFIX_BLOCK_BYTES ?? String(DEFAULT_PREFIX_BLOCK_BYTES), 10, ); - if (!Number.isInteger(port) || port < 1 || port > 65_535) throw new Error("PORT must be 1..65535"); + if (!Number.isInteger(port) || port < 1 || port > 65_535) + throw new Error("PORT must be 1..65535"); const { server } = createFakeCacheApi({ blockBytes }); server.listen(port, "127.0.0.1", () => { console.log(`fake cache API listening on http://127.0.0.1:${port}/v1`); diff --git a/evals/omnimemeval/run.ts b/evals/omnimemeval/run.ts index d75c1330..75e84058 100644 --- a/evals/omnimemeval/run.ts +++ b/evals/omnimemeval/run.ts @@ -9,12 +9,7 @@ import { } from "../../src/core/embedding-provider.ts"; import { loadEnvironmentFile } from "../local-env.ts"; -export type BenchmarkSuite = - | "longmemeval" - | "locomo" - | "beam" - | "personamem-v2" - | "halumem"; +export type BenchmarkSuite = "longmemeval" | "locomo" | "beam" | "personamem-v2" | "halumem"; type SuiteDefinition = { runner: string; @@ -143,7 +138,10 @@ export function loadBenchmarkConfig(path: string): BenchmarkConfig { } function generatedVersion(suite: BenchmarkSuite, now: Date): string { - const stamp = now.toISOString().replace(/[-:]/g, "").replace(/\.\d{3}Z$/, "Z"); + const stamp = now + .toISOString() + .replace(/[-:]/g, "") + .replace(/\.\d{3}Z$/, "Z"); return `${suite}_${stamp}`; } @@ -397,9 +395,7 @@ export function createRunPlan( export async function runPlan(plan: RunPlan): Promise { await preflightEmbeddingProvider(loadBenchmarkEnvironment(plan)); const { spawn } = await import("node:child_process"); - const { ResourceSampler, writeResourceReport } = await import( - "./resource-observability.ts" - ); + const { ResourceSampler, writeResourceReport } = await import("./resource-observability.ts"); const child = spawn(plan.bash, plan.args, { cwd: plan.omniRoot, @@ -426,12 +422,7 @@ export async function runPlan(plan: RunPlan): Promise { // Write one bounded report beside the run's result directory. try { - const resultDir = join( - plan.omniRoot, - "results", - SUITES[plan.suite].resultFolder, - plan.version, - ); + const resultDir = join(plan.omniRoot, "results", SUITES[plan.suite].resultFolder, plan.version); writeResourceReport(sampler.report, resultDir); } catch { // A missing/unwritable result dir must not fail the run itself. @@ -440,31 +431,34 @@ export async function runPlan(plan: RunPlan): Promise { } function isMainModule(): boolean { - return process.argv[1] !== undefined && resolve(process.argv[1]) === fileURLToPath(import.meta.url); + return ( + process.argv[1] !== undefined && resolve(process.argv[1]) === fileURLToPath(import.meta.url) + ); } if (isMainModule()) { if (process.argv.slice(2).some((argument) => argument === "--help" || argument === "-h")) { console.log(usage()); - } else try { - const options = parseRunOptions(process.argv.slice(2)); - const plan = createRunPlan(options); - console.log( - JSON.stringify( - { - suite: plan.suite, - version: plan.version, - config: plan.configPath, - envFile: plan.envFile, - command: [plan.bash, ...plan.args], - }, - null, - 2, - ), - ); - if (!options.dryRun) process.exitCode = await runPlan(plan); - } catch (error) { - console.error(error instanceof Error ? error.message : String(error)); - process.exitCode = 1; - } + } else + try { + const options = parseRunOptions(process.argv.slice(2)); + const plan = createRunPlan(options); + console.log( + JSON.stringify( + { + suite: plan.suite, + version: plan.version, + config: plan.configPath, + envFile: plan.envFile, + command: [plan.bash, ...plan.args], + }, + null, + 2, + ), + ); + if (!options.dryRun) process.exitCode = await runPlan(plan); + } catch (error) { + console.error(error instanceof Error ? error.message : String(error)); + process.exitCode = 1; + } } diff --git a/evals/ooo-execution/families.test.ts b/evals/ooo-execution/families.test.ts index 42a9b7b7..9cb7f35a 100644 --- a/evals/ooo-execution/families.test.ts +++ b/evals/ooo-execution/families.test.ts @@ -59,9 +59,7 @@ for (const family of FAMILIES) { Object.values(fine.units) .flatMap((unit) => unit.editable) .sort(), - coarse.units[family.coarseUnit]! - .editable.slice() - .sort(), + coarse.units[family.coarseUnit]!.editable.slice().sort(), "the fine plan's units cover exactly the files the coarse unit may edit", ); }); @@ -75,7 +73,11 @@ for (const family of FAMILIES) { const seen: { name: string; slots: number; units: number; host: number }[] = []; for (const arm of arms) { const run = await runPlan(specFrom(arm.file, cannedWorker(arm.file), arm.slots)); - assert.deepEqual(run.incomplete, [], `${arm.name}/${arm.slots}: ${run.incomplete.join("; ")}`); + assert.deepEqual( + run.incomplete, + [], + `${arm.name}/${arm.slots}: ${run.incomplete.join("; ")}`, + ); assert.deepEqual( run.units.map((unit) => unit.verdict), run.units.map(() => "accepted"), @@ -88,7 +90,12 @@ for (const family of FAMILIES) { "frozen copies of its siblings", ); assert.equal(run.units.length, arm.units); - seen.push({ name: arm.name, slots: arm.slots, units: run.units.length, host: run.hostChecks }); + seen.push({ + name: arm.name, + slots: arm.slots, + units: run.units.length, + host: run.hostChecks, + }); } // The composed acceptance is the same one in both plans - the same frozen checks over the same // frozen files - so the granularity is the only thing the arms differ in. diff --git a/evals/ooo-execution/fixtures/pipeline/normalize.canned.ts b/evals/ooo-execution/fixtures/pipeline/normalize.canned.ts index 74c26b2d..8e93e183 100644 --- a/evals/ooo-execution/fixtures/pipeline/normalize.canned.ts +++ b/evals/ooo-execution/fixtures/pipeline/normalize.canned.ts @@ -5,5 +5,8 @@ import type { Step } from "./frozen.ts"; export function normalize(steps: readonly Step[]): Step[] { - return steps.filter((step) => step.ms > 0).slice().sort((a, b) => a.name.localeCompare(b.name)); + return steps + .filter((step) => step.ms > 0) + .slice() + .sort((a, b) => a.name.localeCompare(b.name)); } diff --git a/evals/ooo-execution/fixtures/report/alpha.canned.ts b/evals/ooo-execution/fixtures/report/alpha.canned.ts index 32c3664d..50d9a3af 100644 --- a/evals/ooo-execution/fixtures/report/alpha.canned.ts +++ b/evals/ooo-execution/fixtures/report/alpha.canned.ts @@ -8,6 +8,9 @@ export function alphaSection(rows: readonly string[]): Section { return { id: "alpha", title: "Alpha", - lines: rows.filter((row) => row !== "").slice().sort(), + lines: rows + .filter((row) => row !== "") + .slice() + .sort(), }; } diff --git a/evals/ooo-execution/fixtures/report/summary.canned.ts b/evals/ooo-execution/fixtures/report/summary.canned.ts index 4c1a4b66..fbf8606a 100644 --- a/evals/ooo-execution/fixtures/report/summary.canned.ts +++ b/evals/ooo-execution/fixtures/report/summary.canned.ts @@ -1,5 +1,9 @@ import type { Section } from "./interface.ts"; export function summarySection(sections: readonly Section[]): Section { - return { id: "summary", title: "Summary", lines: sections.map((section) => `- ${section.title}`) }; + return { + id: "summary", + title: "Summary", + lines: sections.map((section) => `- ${section.title}`), + }; } diff --git a/evals/ooo-execution/pilot.ts b/evals/ooo-execution/pilot.ts index d05bcf97..449008f7 100644 --- a/evals/ooo-execution/pilot.ts +++ b/evals/ooo-execution/pilot.ts @@ -82,9 +82,7 @@ if (!reportOnly && (!provider || !model)) throw new Error( "Set PI_PROVIDER and PI_MODEL: the pilot's model is an input, not a default\n" + USAGE, ); -const reps = (values["reps"] ?? "3,3,2") - .split(",") - .map((value) => Number(value.trim())); +const reps = (values["reps"] ?? "3,3,2").split(",").map((value) => Number(value.trim())); if (reps.length !== 3 || reps.some((count) => !Number.isSafeInteger(count) || count < 1)) throw new Error(`--reps wants three positive integers, one per arm\n${USAGE}`); const seed = Number(values["seed"] ?? 1); @@ -227,44 +225,43 @@ process.stdout.write(`results: ${out}${reportOnly ? "" : `\nper-run: ${runsDir}` async function runArms(): Promise { for (const [index, step] of schedule.entries()) { - process.stdout.write( - `[${index + 1}/${schedule.length}] arm ${step.arm} rep ${step.rep} (${step.plan}, ` + - `${step.slots} slot${step.slots === 1 ? "" : "s"})\n`, - ); - const startedAt = Date.now(); - const run = await runPlan( - specFrom(specs.get(step.spec)!, piWorker({ provider, model }, true), step.slots), - ); - const verdicts: Record = {}; - for (const unit of run.units) verdicts[unit.verdict] = (verdicts[unit.verdict] ?? 0) + 1; - const one: Recorded = { - arm: step.arm, - plan: step.plan, - rep: step.rep, - slotsRequested: run.slotsRequested, - slotsUsed: run.slotsUsed, - ...(run.slotRefusal ? { slotRefusal: run.slotRefusal } : {}), - wallMs: run.wallMs, - hostMs: run.hostMs, - tokens: run.tokens, - units: run.units.length, - accepted: Object.keys(run.accepted).length, - ...(run.parent ? { parent: run.parent.verdict } : {}), - verdicts, - incomplete: run.incomplete, - }; - recorded.push(one); - writeFileSync( - resolve(runsDir, `${step.arm}-${step.rep}-${startedAt}.json`), - `${JSON.stringify(one, null, 2)}\n`, - ); - process.stdout.write( - ` wall ${one.wallMs}ms, host ${one.hostMs}ms, tokens ${one.tokens}, slots ${one.slotsUsed}/` + - `${one.slotsRequested}, verdicts ${JSON.stringify(verdicts)}, parent ${String(one.parent)}` + - `${one.incomplete.length ? `, incomplete ${JSON.stringify(one.incomplete)}` : ""}\n`, - ); + process.stdout.write( + `[${index + 1}/${schedule.length}] arm ${step.arm} rep ${step.rep} (${step.plan}, ` + + `${step.slots} slot${step.slots === 1 ? "" : "s"})\n`, + ); + const startedAt = Date.now(); + const run = await runPlan( + specFrom(specs.get(step.spec)!, piWorker({ provider, model }, true), step.slots), + ); + const verdicts: Record = {}; + for (const unit of run.units) verdicts[unit.verdict] = (verdicts[unit.verdict] ?? 0) + 1; + const one: Recorded = { + arm: step.arm, + plan: step.plan, + rep: step.rep, + slotsRequested: run.slotsRequested, + slotsUsed: run.slotsUsed, + ...(run.slotRefusal ? { slotRefusal: run.slotRefusal } : {}), + wallMs: run.wallMs, + hostMs: run.hostMs, + tokens: run.tokens, + units: run.units.length, + accepted: Object.keys(run.accepted).length, + ...(run.parent ? { parent: run.parent.verdict } : {}), + verdicts, + incomplete: run.incomplete, + }; + recorded.push(one); + writeFileSync( + resolve(runsDir, `${step.arm}-${step.rep}-${startedAt}.json`), + `${JSON.stringify(one, null, 2)}\n`, + ); + process.stdout.write( + ` wall ${one.wallMs}ms, host ${one.hostMs}ms, tokens ${one.tokens}, slots ${one.slotsUsed}/` + + `${one.slotsRequested}, verdicts ${JSON.stringify(verdicts)}, parent ${String(one.parent)}` + + `${one.incomplete.length ? `, incomplete ${JSON.stringify(one.incomplete)}` : ""}\n`, + ); } if (recorded.length !== reps.reduce((sum, count) => sum + count, 0)) throw new Error(`planned ${reps.join("+")} runs and recorded ${recorded.length}`); } - diff --git a/evals/ooo-execution/plan-driver.test.ts b/evals/ooo-execution/plan-driver.test.ts index 4cf2c829..0a1fe07f 100644 --- a/evals/ooo-execution/plan-driver.test.ts +++ b/evals/ooo-execution/plan-driver.test.ts @@ -521,14 +521,14 @@ test("a spec cannot declare a bound it does not enable the constraint for", () = }; // Without the declaration the run would be the control arm while the file says fusion, so the spec // is refused by name rather than believed. A bound of one asks for nothing and stays legal. - assert.throws( - () => validateSpecFile(file), - /declare fusion\.constraints = \["repair-first"\]/u, + assert.throws(() => validateSpecFile(file), /declare fusion\.constraints = \["repair-first"\]/u); + assert.equal( + validateSpecFile({ ...file, fusion: { unitsPerSession: 1 } }).fusion?.unitsPerSession, + 1, ); - assert.equal(validateSpecFile({ ...file, fusion: { unitsPerSession: 1 } }).fusion?.unitsPerSession, 1); assert.equal( - validateSpecFile({ ...file, fusion: { unitsPerSession: 2, constraints: ["repair-first"] } }).fusion - ?.unitsPerSession, + validateSpecFile({ ...file, fusion: { unitsPerSession: 2, constraints: ["repair-first"] } }) + .fusion?.unitsPerSession, 2, ); }); diff --git a/evals/recall-compression.ts b/evals/recall-compression.ts index ef799c0b..0725509f 100644 --- a/evals/recall-compression.ts +++ b/evals/recall-compression.ts @@ -13,14 +13,38 @@ const directory = mkdtempSync(join(tmpdir(), "nmg-recall-compression-")); const store = new NmgStore(join(directory, "nmg.sqlite")); try { const memories = [ - ["Project Atlas runtime", "Project Atlas currently uses Python 3.12.4 for its data-processing services."], - ["Project Atlas database", "Project Atlas stores local development state in SQLite and production state in PostgreSQL."], - ["Project Atlas deployment", "Project Atlas deploys through a staged Windows-to-Linux container workflow."], - ["Project Atlas testing", "Project Atlas requires unit, integration, and cross-session memory tests before release."], - ["Project Atlas preference", "The user prefers concise Chinese explanations for Project Atlas architecture decisions."], - ["Project Atlas history", "Project Atlas previously evaluated several agent harnesses before selecting Pi."], - ["Project Atlas indexing", "Project Atlas combines lexical, vector, graph, and learned-route retrieval scores."], - ["Project Atlas maintenance", "Project Atlas batches memory-tier maintenance instead of rebuilding after every access."], + [ + "Project Atlas runtime", + "Project Atlas currently uses Python 3.12.4 for its data-processing services.", + ], + [ + "Project Atlas database", + "Project Atlas stores local development state in SQLite and production state in PostgreSQL.", + ], + [ + "Project Atlas deployment", + "Project Atlas deploys through a staged Windows-to-Linux container workflow.", + ], + [ + "Project Atlas testing", + "Project Atlas requires unit, integration, and cross-session memory tests before release.", + ], + [ + "Project Atlas preference", + "The user prefers concise Chinese explanations for Project Atlas architecture decisions.", + ], + [ + "Project Atlas history", + "Project Atlas previously evaluated several agent harnesses before selecting Pi.", + ], + [ + "Project Atlas indexing", + "Project Atlas combines lexical, vector, graph, and learned-route retrieval scores.", + ], + [ + "Project Atlas maintenance", + "Project Atlas batches memory-tier maintenance instead of rebuilding after every access.", + ], ] as const; for (const [nodeName, statement] of memories) { store.remember({ nodeName, statement, tier: 1 }); @@ -36,11 +60,13 @@ try { const explicitQuery = "What did we decide before for Project Atlas?"; const recommendationQuery = "How should we plan the next Project Atlas release?"; const ordinaryQuery = "Explain how a B-tree works."; - const full = formatMemoryContext(store.searchContext(explicitQuery, { - maxTier: 1, - limit: 8, - graphHops: 1, - })); + const full = formatMemoryContext( + store.searchContext(explicitQuery, { + maxTier: 1, + limit: 8, + graphHops: 1, + }), + ); const compressed = formatRecallIndex(store.recallCues(recommendationQuery, { limit: 5 })); const kernel = formatResidentKernel(store.residentKernel()); const report = { diff --git a/evals/retrieval/ingest-ablation.ts b/evals/retrieval/ingest-ablation.ts index ae271a27..70ba7038 100644 --- a/evals/retrieval/ingest-ablation.ts +++ b/evals/retrieval/ingest-ablation.ts @@ -86,7 +86,8 @@ function runCoordinator(): void { } const signatures = new Set([...samples.values()].flat().map((sample) => sample.signature)); - if (signatures.size !== 1) throw new Error("ablation arms produced different persisted semantics"); + if (signatures.size !== 1) + throw new Error("ablation arms produced different persisted semantics"); const baseline = median(samples.get("baseline")!.map((sample) => sample.wallMs)); const report = { generatedAt: new Date().toISOString(), @@ -239,9 +240,7 @@ function rotate(values: readonly T[], offset: number): T[] { function median(values: readonly number[]): number { const sorted = [...values].sort((left, right) => left - right); const middle = Math.floor(sorted.length / 2); - return sorted.length % 2 === 0 - ? (sorted[middle - 1]! + sorted[middle]!) / 2 - : sorted[middle]!; + return sorted.length % 2 === 0 ? (sorted[middle - 1]! + sorted[middle]!) / 2 : sorted[middle]!; } function validateConfig(value: Config): void { diff --git a/evals/retrieval/profile-ingest.ts b/evals/retrieval/profile-ingest.ts index 54513a9c..1d7e6fcd 100644 --- a/evals/retrieval/profile-ingest.ts +++ b/evals/retrieval/profile-ingest.ts @@ -50,7 +50,10 @@ async function profileBridge(): Promise { console.log( `[bridge] ${spec.conversations.length} conversations, ${messages} messages, ` + `total ${(total / 1000).toFixed(1)}s, mean ${(total / durations.length).toFixed(1)}ms/conv, ` + - `p50 ${sorted[Math.floor(sorted.length / 2)]!.toFixed(1)}ms, slowest ${sorted.slice(0, 5).map((v) => v.toFixed(0)).join("/")}ms`, + `p50 ${sorted[Math.floor(sorted.length / 2)]!.toFixed(1)}ms, slowest ${sorted + .slice(0, 5) + .map((v) => v.toFixed(0)) + .join("/")}ms`, ); // Time growth within a user: split per-user durations into halves. const perUser = new Map(); diff --git a/evals/retrieval/profile-progressive.ts b/evals/retrieval/profile-progressive.ts index 695770a2..182ce232 100644 --- a/evals/retrieval/profile-progressive.ts +++ b/evals/retrieval/profile-progressive.ts @@ -63,7 +63,9 @@ async function main() { const f = storeFileFor(item.userId); try { const db = new DatabaseSync(resolve(dir, f), { readOnly: true }); - const blocks = (db.prepare("SELECT COUNT(*) c FROM memory_leaf_blocks").get() as { c: number }).c; + const blocks = ( + db.prepare("SELECT COUNT(*) c FROM memory_leaf_blocks").get() as { c: number } + ).c; db.close(); if (blocks > largest.blocks) largest = { file: f, blocks }; } catch { @@ -83,7 +85,9 @@ async function main() { if (pending > 0) { console.log(`draining ${pending} pending node summaries …`); const r = await drainNodeSummaries(store, provider, { maxCalls: 1000 }); - console.log(` node summaries: +${r.summarized} (failed ${r.failed}, truncated ${r.truncated})`); + console.log( + ` node summaries: +${r.summarized} (failed ${r.failed}, truncated ${r.truncated})`, + ); } else { console.log(" no pending node summaries (already summarized)"); } @@ -173,17 +177,28 @@ async function main() { }); } - const agg = (k: keyof (typeof results)[0]) => - median(results.map((r) => r[k] as number)); + const agg = (k: keyof (typeof results)[0]) => median(results.map((r) => r[k] as number)); console.log("\n== per-query medians =="); - console.log(`full-block candidates : ${agg("fullMs").toFixed(3)} ms, ${agg("fullHits").toFixed(1)} hits`); - console.log(`node routing : ${agg("nodeMs").toFixed(3)} ms, ${agg("nodeHits").toFixed(1)} nodes`); - console.log(`node->blocks : ${agg("nodeBlocksMs").toFixed(3)} ms, ${agg("nodeBlocks").toFixed(1)} blocks`); + console.log( + `full-block candidates : ${agg("fullMs").toFixed(3)} ms, ${agg("fullHits").toFixed(1)} hits`, + ); + console.log( + `node routing : ${agg("nodeMs").toFixed(3)} ms, ${agg("nodeHits").toFixed(1)} nodes`, + ); + console.log( + `node->blocks : ${agg("nodeBlocksMs").toFixed(3)} ms, ${agg("nodeBlocks").toFixed(1)} blocks`, + ); console.log(`round-1 size : ${agg("round1Size").toFixed(1)} blocks`); - console.log(`overlap top-3 : ${agg("overlapTop3").toFixed(2)} / 3 (${((agg("overlapTop3") / 3) * 100).toFixed(0)}%)`); + console.log( + `overlap top-3 : ${agg("overlapTop3").toFixed(2)} / 3 (${((agg("overlapTop3") / 3) * 100).toFixed(0)}%)`, + ); console.log(`overlap top-10 : ${agg("overlapTop10").toFixed(2)} / 10`); - console.log(`overlap all : ${agg("overlapAll").toFixed(2)} / ${agg("fullHits").toFixed(1)}`); - console.log(`\nnode-routed empty (no node FTS hits): ${results.filter((r) => r.nodeHits === 0).length}/${results.length}`); + console.log( + `overlap all : ${agg("overlapAll").toFixed(2)} / ${agg("fullHits").toFixed(1)}`, + ); + console.log( + `\nnode-routed empty (no node FTS hits): ${results.filter((r) => r.nodeHits === 0).length}/${results.length}`, + ); store.close(); } diff --git a/evals/retrieval/profile-size.ts b/evals/retrieval/profile-size.ts index d66b02aa..2332a32f 100644 --- a/evals/retrieval/profile-size.ts +++ b/evals/retrieval/profile-size.ts @@ -23,7 +23,8 @@ function main() { const dataset = process.argv[2] ?? "longmemeval"; const dir = resolve(STORES_ROOT, dataset); const files = readdirSync(dir).filter((f) => f.endsWith(".sqlite")); - const rows: Array<{ file: string; mem: number; blocks: number; nodes: number; leafFts: number }> = []; + const rows: Array<{ file: string; mem: number; blocks: number; nodes: number; leafFts: number }> = + []; for (const file of files) { const db = open(resolve(dir, file)); @@ -41,10 +42,13 @@ function main() { rows.sort((a, b) => b.blocks - a.blocks); const total = rows.length; - const sum = (k: "mem" | "blocks" | "nodes") => rows.reduce((s, r) => s + (r[k] > 0 ? r[k] : 0), 0); + const sum = (k: "mem" | "blocks" | "nodes") => + rows.reduce((s, r) => s + (r[k] > 0 ? r[k] : 0), 0); const avg = (k: "mem" | "blocks" | "nodes") => (sum(k) / Math.max(1, total)).toFixed(1); console.log(`\n== ${dataset}: ${total} stores ==`); - console.log(`avg per store — mem: ${avg("mem")} blocks: ${avg("blocks")} nodes: ${avg("nodes")}`); + console.log( + `avg per store — mem: ${avg("mem")} blocks: ${avg("blocks")} nodes: ${avg("nodes")}`, + ); console.log("\ntop-8 by blocks:"); for (const r of rows.slice(0, 8)) { console.log( diff --git a/evals/skillopt/export.ts b/evals/skillopt/export.ts index ec63de37..989852cc 100644 --- a/evals/skillopt/export.ts +++ b/evals/skillopt/export.ts @@ -32,7 +32,11 @@ export function exportSkillOptDataset(options: ExportSkillOptOptions): { mkdirSync(directory, { recursive: true }); writeFileSync( resolve(directory, "items.json"), - `${JSON.stringify(dataset.items.filter((item) => item.split === split), null, 2)}\n`, + `${JSON.stringify( + dataset.items.filter((item) => item.split === split), + null, + 2, + )}\n`, "utf8", ); } @@ -95,7 +99,9 @@ function parseArguments(args: readonly string[]): ExportSkillOptOptions { if (process.argv[1] && import.meta.url === pathToFileURL(resolve(process.argv[1])).href) { try { - process.stdout.write(`${JSON.stringify(exportSkillOptDataset(parseArguments(process.argv.slice(2))), null, 2)}\n`); + process.stdout.write( + `${JSON.stringify(exportSkillOptDataset(parseArguments(process.argv.slice(2))), null, 2)}\n`, + ); } catch (error) { process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); process.exitCode = 2; diff --git a/evals/topology/namesakes.ts b/evals/topology/namesakes.ts index 4af15bea..53bed05e 100644 --- a/evals/topology/namesakes.ts +++ b/evals/topology/namesakes.ts @@ -150,10 +150,7 @@ export function namesakesThresholdCurve( recall: ratio(truePositives, positives.length), precision: ratio(truePositives, selected.length), aliasRecall: ratio(selectedAliases, aliasPositives.length), - exactNameNegativeRejection: ratio( - rejectedExactNameNegatives, - exactNameNegatives.length, - ), + exactNameNegativeRejection: ratio(rejectedExactNameNegatives, exactNameNegatives.length), }; }); } @@ -266,8 +263,7 @@ function ratio(numerator: number, denominator: number): number { async function main(): Promise { const path = resolve( - process.env.NMG_NAMESAKES_DATA ?? - ".benchmarks/namesakes/data/Namesakes_entities.jsonl", + process.env.NMG_NAMESAKES_DATA ?? ".benchmarks/namesakes/data/Namesakes_entities.jsonl", ); const maxEntities = process.env.NMG_NAMESAKES_MAX_ENTITIES ? Number(process.env.NMG_NAMESAKES_MAX_ENTITIES) diff --git a/evals/topology/run.ts b/evals/topology/run.ts index 39a98a81..af333340 100644 --- a/evals/topology/run.ts +++ b/evals/topology/run.ts @@ -99,9 +99,7 @@ export function auditTopologyGate(cases: readonly BenchmarkCase[]): TopologyGate if (assessment.eligible) eligibleCrossPersonCandidates += 1; } - for (const [index, candidate] of discovered.filter( - (item) => !item.sameSpeaker, - ).entries()) { + for (const [index, candidate] of discovered.filter((item) => !item.sameSpeaker).entries()) { const early = people.find((person) => person.speaker === candidate.earlySpeaker)!; const late = people.find((person) => person.speaker === candidate.lateSpeaker)!; naturalFalseMergesAttempted += 1; @@ -219,9 +217,15 @@ function buildPersonFragments( sampleId: string, sessions: readonly BenchmarkSession[], ): PersonFragments[] { - const speakers = [...new Set(sessions.flatMap((session) => - session.turns.map((turn) => turn.speaker).filter((speaker): speaker is string => Boolean(speaker)) - ))]; + const speakers = [ + ...new Set( + sessions.flatMap((session) => + session.turns + .map((turn) => turn.speaker) + .filter((speaker): speaker is string => Boolean(speaker)), + ), + ), + ]; return speakers.flatMap((speaker) => { const midpoint = Math.max(1, Math.floor(sessions.length / 2)); const earlyTurns = turnsForSpeaker(sessions.slice(0, midpoint), speaker).slice(0, 3); @@ -230,21 +234,29 @@ function buildPersonFragments( // One scoped identity dimension is deliberate: the conversation prefix // prevents same-name speakers in different samples from sharing identity. const scope = { person: `${sampleId}:${speaker}` }; - return [{ - speaker, - early: earlyTurns.map((turn, index) => store.remember({ - statement: turn.content, - nodeName: `${sampleId}:${speaker}:early`, - scope, - sourceRef: turn.sourceId ?? `${sampleId}:${speaker}:early:${index}`, - }).memory), - late: lateTurns.map((turn, index) => store.remember({ - statement: turn.content, - nodeName: `${sampleId}:${speaker}:late`, - scope, - sourceRef: turn.sourceId ?? `${sampleId}:${speaker}:late:${index}`, - }).memory), - }]; + return [ + { + speaker, + early: earlyTurns.map( + (turn, index) => + store.remember({ + statement: turn.content, + nodeName: `${sampleId}:${speaker}:early`, + scope, + sourceRef: turn.sourceId ?? `${sampleId}:${speaker}:early:${index}`, + }).memory, + ), + late: lateTurns.map( + (turn, index) => + store.remember({ + statement: turn.content, + nodeName: `${sampleId}:${speaker}:late`, + scope, + sourceRef: turn.sourceId ?? `${sampleId}:${speaker}:late:${index}`, + }).memory, + ), + }, + ]; }); } @@ -273,7 +285,10 @@ function repeatedProposal( return proposalId; } -function countReasons(counts: Record, assessment: TopologyAutomationAssessment): void { +function countReasons( + counts: Record, + assessment: TopologyAutomationAssessment, +): void { for (const reason of assessment.reasons) counts[reason] = (counts[reason] ?? 0) + 1; } diff --git a/package.json b/package.json index 5481c1e2..cd73e4f6 100644 --- a/package.json +++ b/package.json @@ -57,8 +57,8 @@ "tutorial:first-recall": "node --experimental-strip-types scripts/tutorial-first-recall.ts", "lint": "eslint src/ .pi/extensions/ claude-plugins/ workbuddy-plugin/ tests/ evals/ scripts/ tools/", "lint:fix": "eslint --fix src/ .pi/extensions/ claude-plugins/ workbuddy-plugin/ tests/ evals/ scripts/ tools/", - "format": "prettier --write \"src/**/*.ts\" \".pi/**/*.ts\" \"workbuddy-plugin/**/*.ts\"", - "format:check": "prettier --check \"src/**/*.ts\" \".pi/**/*.ts\" \"workbuddy-plugin/**/*.ts\"", + "format": "prettier --write \"src/**/*.ts\" \".pi/**/*.ts\" \"workbuddy-plugin/**/*.ts\" \"tests/**/*.ts\" \"evals/**/*.ts\" \"scripts/**/*.ts\" \"tools/**/*.ts\"", + "format:check": "prettier --check \"src/**/*.ts\" \".pi/**/*.ts\" \"workbuddy-plugin/**/*.ts\" \"tests/**/*.ts\" \"evals/**/*.ts\" \"scripts/**/*.ts\" \"tools/**/*.ts\"", "test:coverage": "c8 --check-coverage --lines 0.01 node --experimental-strip-types --test --test-concurrency=2 \"tests/cli/**/*.test.ts\" \"tests/core/**/*.test.ts\" \"tests/docker/**/*.test.ts\" \"tests/docs/**/*.test.ts\" \"tests/extensions/**/*.test.ts\" \"tests/guardrails/**/*.test.ts\" \"tests/integration/**/*.test.ts\" \"tests/lab/**/*.test.ts\" \"tests/prompts/**/*.test.ts\" \"tests/rcp/**/*.test.ts\" \"tests/scripts/**/*.test.ts\" \"tests/skills/**/*.test.ts\" \"tests/support/**/*.test.ts\" \"tests/tools/**/*.test.ts\"", "eval:agents": "node --experimental-strip-types evals/run.ts", "hotspot:modules": "node --experimental-strip-types scripts/hotspot-files.ts", diff --git a/scripts/sync-nmg-skill.ts b/scripts/sync-nmg-skill.ts index 91fdf2b2..dd2b2376 100644 --- a/scripts/sync-nmg-skill.ts +++ b/scripts/sync-nmg-skill.ts @@ -30,7 +30,9 @@ export interface SkillSyncReport { export function inspectNmgSkill(target: string): SkillSyncReport { const resolvedTarget = safeTarget(target); const sourceFiles = inventory(sourceRoot); - const targetFiles = existsSync(resolvedTarget) ? inventory(resolvedTarget) : new Map(); + const targetFiles = existsSync(resolvedTarget) + ? inventory(resolvedTarget) + : new Map(); const missing: string[] = []; const changed: string[] = []; const extra: string[] = []; @@ -105,7 +107,9 @@ export function recoverInterruptedSync(target: string): void { if (!existsSync(resolvedTarget) && backups.length > 0) { const newestBackup = backups.reduce((newest, entry) => - statSync(join(parent, entry)).mtimeMs > statSync(join(parent, newest)).mtimeMs ? entry : newest, + statSync(join(parent, entry)).mtimeMs > statSync(join(parent, newest)).mtimeMs + ? entry + : newest, ); renameSync(join(parent, newestBackup), resolvedTarget); backups.splice(backups.indexOf(newestBackup), 1); @@ -196,7 +200,8 @@ function inventory(root: string): Map { for (const entry of readdirSync(directory, { withFileTypes: true })) { const absolute = join(directory, entry.name); if (entry.isDirectory()) visit(absolute); - else if (entry.isFile()) files.set(relative(root, absolute).replaceAll("\\", "/"), readFileSync(absolute)); + else if (entry.isFile()) + files.set(relative(root, absolute).replaceAll("\\", "/"), readFileSync(absolute)); else throw new Error(`unsupported skill entry: ${absolute}`); } }; @@ -225,7 +230,8 @@ function validateArgs(args: string[]): void { if (process.argv[1] && import.meta.url === pathToFileURL(resolve(process.argv[1])).href) { const args = process.argv.slice(2); validateArgs(args); - const target = optionValue(args, "--target") ?? join(homedir(), ".agents", "skills", "nmg-memory"); + const target = + optionValue(args, "--target") ?? join(homedir(), ".agents", "skills", "nmg-memory"); const check = args.includes("--check"); const report = check ? inspectNmgSkill(target) : syncNmgSkill(target); process.stdout.write(`${JSON.stringify(report, null, 2)}\n`); diff --git a/tests/cli/data-path.test.ts b/tests/cli/data-path.test.ts index de903d87..792b0baa 100644 --- a/tests/cli/data-path.test.ts +++ b/tests/cli/data-path.test.ts @@ -18,5 +18,8 @@ test("an explicit NMG_DATA_DIR overrides the fallback", () => { test("controlled clients can supply a project-local fallback", () => { assert.equal(resolveNmgDataDir({}, "C:/project/.nmg"), resolve("C:/project/.nmg")); - assert.equal(resolveNmgDataDir({ NMG_DATA_DIR: " " }, "C:/project/.nmg"), resolve("C:/project/.nmg")); + assert.equal( + resolveNmgDataDir({ NMG_DATA_DIR: " " }, "C:/project/.nmg"), + resolve("C:/project/.nmg"), + ); }); diff --git a/tests/cli/service.test.ts b/tests/cli/service.test.ts index a2d3fbf4..393c13d7 100644 --- a/tests/cli/service.test.ts +++ b/tests/cli/service.test.ts @@ -1930,7 +1930,13 @@ test("opt-in embedding auto-sync makes remembered records available to hybrid se }); for (let attempt = 0; attempt < 100; attempt += 1) { const status = await service.invoke("status"); - if (status.embedding.health && typeof status.embedding.health === "object" && "lastSucceededAt" in status.embedding.health && status.embedding.health.lastSucceededAt) break; + if ( + status.embedding.health && + typeof status.embedding.health === "object" && + "lastSucceededAt" in status.embedding.health && + status.embedding.health.lastSucceededAt + ) + break; await new Promise((resolve) => setTimeout(resolve, 10)); } const searched = await service.invoke("search", { @@ -1984,7 +1990,13 @@ test("provider presence alone (no AUTO_SYNC env) auto-syncs remembered records t }); for (let attempt = 0; attempt < 100; attempt += 1) { const status = await service.invoke("status"); - if (status.embedding.health && typeof status.embedding.health === "object" && "lastSucceededAt" in status.embedding.health && status.embedding.health.lastSucceededAt) break; + if ( + status.embedding.health && + typeof status.embedding.health === "object" && + "lastSucceededAt" in status.embedding.health && + status.embedding.health.lastSucceededAt + ) + break; await new Promise((resolve) => setTimeout(resolve, 10)); } const searched = await service.invoke("search", { diff --git a/tests/core/advanced-query.test.ts b/tests/core/advanced-query.test.ts index b28e7000..0e77970e 100644 --- a/tests/core/advanced-query.test.ts +++ b/tests/core/advanced-query.test.ts @@ -1,7 +1,11 @@ import test from "node:test"; import assert from "node:assert/strict"; -import { applyAdvancedFilters, extractEventWindow, parseAdvancedQuery } from "../../src/core/store/advanced-query.ts"; +import { + applyAdvancedFilters, + extractEventWindow, + parseAdvancedQuery, +} from "../../src/core/store/advanced-query.ts"; function fakeResult(overrides: { statement: string; @@ -66,7 +70,10 @@ test("applyAdvancedFilters filters by type and node", () => { const byType = applyAdvancedFilters(results, { types: ["preference"], excludeTerms: [] }); assert.equal(byType.length, 1); assert.equal(byType[0].memory.memoryType, "preference"); - const byNode = applyAdvancedFilters(results, { nodeNames: ["conversation abc"], excludeTerms: [] }); + const byNode = applyAdvancedFilters(results, { + nodeNames: ["conversation abc"], + excludeTerms: [], + }); assert.equal(byNode.length, 1); assert.equal(byNode[0].memory.statement, "去过东京"); }); @@ -82,14 +89,19 @@ test("applyAdvancedFilters filters by time range and exclusions", () => { eventTimeFrom: "2026-02-01", eventTimeTo: "2026-06-30", }); - assert.deepEqual(ranged.map((r) => r.memory.statement), ["五月去了京都", "订了快餐外卖"]); + assert.deepEqual( + ranged.map((r) => r.memory.statement), + ["五月去了京都", "订了快餐外卖"], + ); const excluded = applyAdvancedFilters(results, { excludeTerms: ["快餐"], types: undefined }); assert.equal(excluded.length, 2); assert.ok(!excluded.some((r) => r.memory.statement.includes("快餐"))); }); test("extractEventWindow: as of → inclusive through that day", () => { - const w = extractEventWindow("What is Martin Mark's current mental health status as of Mar 10, 2029?"); + const w = extractEventWindow( + "What is Martin Mark's current mental health status as of Mar 10, 2029?", + ); assert.equal(w.to, "2029-03-11"); assert.equal(w.from, undefined); }); @@ -101,7 +113,9 @@ test("extractEventWindow: on → that exact day", () => { }); test("extractEventWindow: from to → range", () => { - const w = extractEventWindow("How did Karen's motivation evolve from February 28, 2035, to February 28, 2036?"); + const w = extractEventWindow( + "How did Karen's motivation evolve from February 28, 2035, to February 28, 2036?", + ); assert.equal(w.from, "2035-02-28"); assert.equal(w.to, "2036-02-29"); }); diff --git a/tests/core/analogy.test.ts b/tests/core/analogy.test.ts index de4b8600..ce85665f 100644 --- a/tests/core/analogy.test.ts +++ b/tests/core/analogy.test.ts @@ -29,8 +29,24 @@ function node(store: NmgStore, name: string) { test("abstractSubgraph extracts EVOLUTION from a supersede chain", () => { withStore((store) => { - const v2022 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "2022预算5000万", sessionId: "s1", sourceActor: "user", eventTime: "2022-01-01" }); - const v2023 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "2023预算6500万", sessionId: "s1", sourceActor: "user", eventTime: "2023-01-01" }); + const v2022 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "2022预算5000万", + sessionId: "s1", + sourceActor: "user", + eventTime: "2022-01-01", + }); + const v2023 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "2023预算6500万", + sessionId: "s1", + sourceActor: "user", + eventTime: "2023-01-01", + }); store.applySupersession({ newMemoryId: v2023.memory.id, supersededMemoryId: v2022.memory.id }); const sig = store.abstractSubgraph([v2022.node.id, v2023.node.id]); assert.ok(sig.patternTypes.includes("EVOLUTION")); @@ -40,7 +56,9 @@ test("abstractSubgraph extracts EVOLUTION from a supersede chain", () => { test("abstractSubgraph extracts all other pattern shapes from a mixed cluster", () => { withStore((store) => { - const [t, a, b, c, d, p1, p2] = ["主题", "甲", "乙", "丙", "丁", "容器", "零件"].map((x) => node(store, x)); + const [t, a, b, c, d, p1, p2] = ["主题", "甲", "乙", "丙", "丁", "容器", "零件"].map((x) => + node(store, x), + ); store.linkNodes({ sourceNodeId: a.node.id, targetNodeId: b.node.id, type: "contradicts" }); store.linkNodes({ sourceNodeId: b.node.id, targetNodeId: a.node.id, type: "contradicts" }); store.linkNodes({ sourceNodeId: a.node.id, targetNodeId: b.node.id, type: "depends_on" }); @@ -48,7 +66,15 @@ test("abstractSubgraph extracts all other pattern shapes from a mixed cluster", store.linkNodes({ sourceNodeId: p1.node.id, targetNodeId: p2.node.id, type: "part_of" }); store.linkNodes({ sourceNodeId: c.node.id, targetNodeId: d.node.id, type: "causes" }); store.linkNodes({ sourceNodeId: d.node.id, targetNodeId: c.node.id, type: "causes" }); - const sig = store.abstractSubgraph([t.node.id, a.node.id, b.node.id, c.node.id, d.node.id, p1.node.id, p2.node.id]); + const sig = store.abstractSubgraph([ + t.node.id, + a.node.id, + b.node.id, + c.node.id, + d.node.id, + p1.node.id, + p2.node.id, + ]); // Bidirectional contradiction is one CONTRADICTION; the causes cycle is one FEEDBACK. assert.equal(sig.patternCounts.CONTRADICTION, 1); assert.equal(sig.patternCounts.DEPENDENCY, 4); @@ -60,17 +86,54 @@ test("abstractSubgraph extracts all other pattern shapes from a mixed cluster", test("findStructuralAnalogies matches same-structure, semantically different domains", () => { withStore((store) => { // Domain A: budget evolution. - const b2022 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "2022预算5000万", sessionId: "s1", sourceActor: "user" }); - const b2023 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "2023预算6500万", sessionId: "s1", sourceActor: "user" }); + const b2022 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "2022预算5000万", + sessionId: "s1", + sourceActor: "user", + }); + const b2023 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "2023预算6500万", + sessionId: "s1", + sourceActor: "user", + }); store.applySupersession({ newMemoryId: b2023.memory.id, supersededMemoryId: b2022.memory.id }); // Domain B: tech-stack evolution (unrelated semantics, same structure). - const s1 = store.remember({ nodeName: "技术选型", nodeKind: "topic", nodeSummary: "选型", statement: "选型v1用Postgres", sessionId: "s1", sourceActor: "user" }); - const s2 = store.remember({ nodeName: "技术选型", nodeKind: "topic", nodeSummary: "选型", statement: "选型v2迁TiDB", sessionId: "s1", sourceActor: "user" }); + const s1 = store.remember({ + nodeName: "技术选型", + nodeKind: "topic", + nodeSummary: "选型", + statement: "选型v1用Postgres", + sessionId: "s1", + sourceActor: "user", + }); + const s2 = store.remember({ + nodeName: "技术选型", + nodeKind: "topic", + nodeSummary: "选型", + statement: "选型v2迁TiDB", + sessionId: "s1", + sourceActor: "user", + }); store.applySupersession({ newMemoryId: s2.memory.id, supersededMemoryId: s1.memory.id }); // Domain C: plain memory, no structure — must not match. - store.remember({ nodeName: "产品", nodeKind: "topic", nodeSummary: "产品", statement: "产品有3个功能", sessionId: "s1", sourceActor: "user" }); + store.remember({ + nodeName: "产品", + nodeKind: "topic", + nodeSummary: "产品", + statement: "产品有3个功能", + sessionId: "s1", + sourceActor: "user", + }); - const analogies = store.findStructuralAnalogies([b2022.node.id, b2023.node.id], { maxCandidates: 5 }); + const analogies = store.findStructuralAnalogies([b2022.node.id, b2023.node.id], { + maxCandidates: 5, + }); const hitTech = analogies.find((a) => a.targetNodeName === "技术选型"); assert.ok(hitTech, "tech-stack matched as budget's analogy"); assert.equal(hitTech!.score, 1); @@ -79,7 +142,10 @@ test("findStructuralAnalogies matches same-structure, semantically different dom // Symmetric: tech-stack finds budget. const back = store.findStructuralAnalogies([s1.node.id, s2.node.id], { maxCandidates: 5 }); - assert.ok(back.some((a) => a.targetNodeName === "预算"), "reverse analogy works"); + assert.ok( + back.some((a) => a.targetNodeName === "预算"), + "reverse analogy works", + ); }); }); @@ -87,9 +153,16 @@ test("findStructuralAnalogies excludes nodes directly related to the query clust withStore((store) => { const budget = node(store, "预算"); const finance = node(store, "财务"); - store.linkNodes({ sourceNodeId: budget.node.id, targetNodeId: finance.node.id, type: "related_to" }); + store.linkNodes({ + sourceNodeId: budget.node.id, + targetNodeId: finance.node.id, + type: "related_to", + }); // Same-domain node directly related to budget must not surface as an analogy. const analogies = store.findStructuralAnalogies([budget.node.id], { maxCandidates: 5 }); - assert.ok(!analogies.some((a) => a.targetNodeId === finance.node.id), "adjacent same-domain node excluded"); + assert.ok( + !analogies.some((a) => a.targetNodeId === finance.node.id), + "adjacent same-domain node excluded", + ); }); }); diff --git a/tests/core/autodiff.test.ts b/tests/core/autodiff.test.ts index 5ccea652..3d908583 100644 --- a/tests/core/autodiff.test.ts +++ b/tests/core/autodiff.test.ts @@ -20,21 +20,13 @@ test("UOp autodiff evaluates lazily and differentiates matrix multiplication", ( test("matrix multiplication preserves rectangular tails and gradients", () => { const left = Tensor.matrix([1, 2, 3, 4, 5, 6], 2, 3, true); - const right = Tensor.matrix( - [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], - 3, - 5, - true, - ); + const right = Tensor.matrix([1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], 3, 5, true); const product = left.matmul(right); assert.deepEqual([...product.data], [46, 52, 58, 64, 70, 100, 115, 130, 145, 160]); product.sum().backward(); assert.deepEqual([...left.grad], [15, 40, 65, 15, 40, 65]); - assert.deepEqual( - [...right.grad], - [5, 5, 5, 5, 5, 7, 7, 7, 7, 7, 9, 9, 9, 9, 9], - ); + assert.deepEqual([...right.grad], [5, 5, 5, 5, 5, 7, 7, 7, 7, 7, 9, 9, 9, 9, 9]); }); test("softmax cross entropy produces the expected graph gradient", () => { diff --git a/tests/core/community.test.ts b/tests/core/community.test.ts index 8b4faa49..e0d469b0 100644 --- a/tests/core/community.test.ts +++ b/tests/core/community.test.ts @@ -51,8 +51,23 @@ test("detectCommunities finds weakly-connected components and drops isolates", ( test("analyzeCommunities profiles patterns and emits natural-supervision suggestions", () => { withStore((store) => { // Community 1: evolution + dependency. - const b1 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "预算v2023", sessionId: "s1", sourceActor: "user" }); - const b2 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "预算v2024", sessionId: "s1", sourceActor: "user", supersedesId: b1.memory.id }); + const b1 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "预算v2023", + sessionId: "s1", + sourceActor: "user", + }); + const b2 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "预算v2024", + sessionId: "s1", + sourceActor: "user", + supersedesId: b1.memory.id, + }); const tech = node(store, "技术选型"); const ops = node(store, "运维"); store.linkNodes({ sourceNodeId: b2.node.id, targetNodeId: tech.node.id, type: "depends_on" }); @@ -68,10 +83,19 @@ test("analyzeCommunities profiles patterns and emits natural-supervision suggest const c0 = analysis.find((c) => c.patternCounts.EVOLUTION > 0); const c1 = analysis.find((c) => c.patternCounts.CONTRADICTION > 0); assert.ok(c0, "evolution community found"); - assert.ok(c0!.suggestions.some((s) => s.kind === "EVOLUTION_CHAIN"), "evolution suggestion"); - assert.ok(c0!.suggestions.some((s) => s.kind === "DEPENDENCY_CHAIN"), "dependency suggestion"); + assert.ok( + c0!.suggestions.some((s) => s.kind === "EVOLUTION_CHAIN"), + "evolution suggestion", + ); + assert.ok( + c0!.suggestions.some((s) => s.kind === "DEPENDENCY_CHAIN"), + "dependency suggestion", + ); assert.ok(c1, "contradiction community found"); - assert.ok(c1!.suggestions.some((s) => s.kind === "CONTRADICTION_PAIR"), "contradiction suggestion"); + assert.ok( + c1!.suggestions.some((s) => s.kind === "CONTRADICTION_PAIR"), + "contradiction suggestion", + ); }); }); @@ -84,6 +108,9 @@ test("analyzeCommunities flags a feedback loop for manual review", () => { const analysis = store.analyzeCommunities(); assert.equal(analysis.length, 1); assert.equal(analysis[0]!.patternCounts.FEEDBACK, 1); - assert.ok(analysis[0]!.suggestions.some((s) => s.kind === "FEEDBACK_REVIEW"), "feedback review suggestion"); + assert.ok( + analysis[0]!.suggestions.some((s) => s.kind === "FEEDBACK_REVIEW"), + "feedback review suggestion", + ); }); }); diff --git a/tests/core/context-reward.test.ts b/tests/core/context-reward.test.ts index 9469a64e..eba61b96 100644 --- a/tests/core/context-reward.test.ts +++ b/tests/core/context-reward.test.ts @@ -29,8 +29,7 @@ test("RSCB invariant: false positive is the worst outcome", () => { const fp = contextUseReward("falsePositive", "retrieve"); for (const outcome of OUTCOMES) { if (outcome === "falsePositive") continue; - const action = - outcome === "correctAbstain" ? "none" : ("retrieve" as const); + const action = outcome === "correctAbstain" ? "none" : ("retrieve" as const); assert.ok( fp < contextUseReward(outcome, action), `falsePositive (${fp}) should beat ${outcome} (${contextUseReward(outcome, action)})`, @@ -99,10 +98,7 @@ test("feedback labels with no usable signal map to null, never a silent vote", ( assert.equal(contextOutcomeFromFeedback({}), null); assert.equal(contextOutcomeFromFeedback({ taskSuccess: false }), null); assert.equal(contextOutcomeFromFeedback({ expansionUseful: true }), null); - assert.equal( - contextOutcomeFromFeedback({ taskSuccess: false, expansionUseful: false }), - null, - ); + assert.equal(contextOutcomeFromFeedback({ taskSuccess: false, expansionUseful: false }), null); // Contradictory: sufficient evidence yet the task failed -> no vote. assert.equal(contextOutcomeFromFeedback({ evidenceSufficient: true, taskSuccess: false }), null); }); diff --git a/tests/core/fork-merge.test.ts b/tests/core/fork-merge.test.ts index f6bd087e..61cd221a 100644 --- a/tests/core/fork-merge.test.ts +++ b/tests/core/fork-merge.test.ts @@ -53,10 +53,7 @@ test("ForkMerge forward: divergence stays in range under float32 rounding", () = // rounding pushed cos slightly above 1.0 in ~20% of random forwards, leaking a // small negative divergence. Many trials because the flake is data-dependent. for (let trial = 0; trial < 300; trial++) { - const fm = new ForkMerge( - new HierarchicalActivation(D), - new HierarchicalActivation(D), - ); + const fm = new ForkMerge(new HierarchicalActivation(D), new HierarchicalActivation(D)); const { divergence } = fm.forward(rvec(D), cands(D, 10)); assert.ok( divergence >= 0 && divergence <= 2, diff --git a/tests/core/graph-cycles.test.ts b/tests/core/graph-cycles.test.ts index a913e18f..7550846b 100644 --- a/tests/core/graph-cycles.test.ts +++ b/tests/core/graph-cycles.test.ts @@ -18,9 +18,30 @@ function withStore(run: (store: NmgStore) => void): void { test("detectGraphCycles finds a directed relation cycle", () => { withStore((store) => { - const m1 = store.remember({ nodeName: "N1", nodeKind: "topic", nodeSummary: "n1", statement: "节点1因果", sessionId: "s1", sourceActor: "user" }); - const m2 = store.remember({ nodeName: "N2", nodeKind: "topic", nodeSummary: "n2", statement: "节点2因果", sessionId: "s1", sourceActor: "user" }); - const m3 = store.remember({ nodeName: "N3", nodeKind: "topic", nodeSummary: "n3", statement: "节点3因果", sessionId: "s1", sourceActor: "user" }); + const m1 = store.remember({ + nodeName: "N1", + nodeKind: "topic", + nodeSummary: "n1", + statement: "节点1因果", + sessionId: "s1", + sourceActor: "user", + }); + const m2 = store.remember({ + nodeName: "N2", + nodeKind: "topic", + nodeSummary: "n2", + statement: "节点2因果", + sessionId: "s1", + sourceActor: "user", + }); + const m3 = store.remember({ + nodeName: "N3", + nodeKind: "topic", + nodeSummary: "n3", + statement: "节点3因果", + sessionId: "s1", + sourceActor: "user", + }); store.linkNodes({ sourceNodeId: m1.node.id, targetNodeId: m2.node.id, type: "causes" }); store.linkNodes({ sourceNodeId: m2.node.id, targetNodeId: m3.node.id, type: "causes" }); store.linkNodes({ sourceNodeId: m3.node.id, targetNodeId: m1.node.id, type: "causes" }); @@ -39,14 +60,36 @@ test("detectGraphCycles finds a directed relation cycle", () => { test("detectGraphCycles finds a supersede cycle (data anomaly)", () => { withStore((store) => { - const a = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "A版", sessionId: "s1", sourceActor: "user" }); - const b = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "B版", sessionId: "s1", sourceActor: "user" }); + const a = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "A版", + sessionId: "s1", + sourceActor: "user", + }); + const b = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "B版", + sessionId: "s1", + sourceActor: "user", + }); // Normal supersede is a DAG — no cycle. assert.equal(store.detectGraphCycles().supersedeCycles.length, 0); // Inject a mutual-supersede anomaly directly (write path would reject it). - const db = (store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } }).db; - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(b.memory.id, a.memory.id); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(a.memory.id, b.memory.id); + const db = ( + store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } } + ).db; + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + b.memory.id, + a.memory.id, + ); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + a.memory.id, + b.memory.id, + ); const r = store.detectGraphCycles(); assert.ok( r.supersedeCycles.some((cycle) => cycle.includes(a.memory.id) && cycle.includes(b.memory.id)), @@ -57,9 +100,30 @@ test("detectGraphCycles finds a supersede cycle (data anomaly)", () => { test("detectGraphCycles is empty for an acyclic chain", () => { withStore((store) => { - const p1 = store.remember({ nodeName: "P1", nodeKind: "topic", nodeSummary: "p1", statement: "正常链1", sessionId: "s1", sourceActor: "user" }); - const p2 = store.remember({ nodeName: "P2", nodeKind: "topic", nodeSummary: "p2", statement: "正常链2", sessionId: "s1", sourceActor: "user" }); - const p3 = store.remember({ nodeName: "P3", nodeKind: "topic", nodeSummary: "p3", statement: "正常链3", sessionId: "s1", sourceActor: "user" }); + const p1 = store.remember({ + nodeName: "P1", + nodeKind: "topic", + nodeSummary: "p1", + statement: "正常链1", + sessionId: "s1", + sourceActor: "user", + }); + const p2 = store.remember({ + nodeName: "P2", + nodeKind: "topic", + nodeSummary: "p2", + statement: "正常链2", + sessionId: "s1", + sourceActor: "user", + }); + const p3 = store.remember({ + nodeName: "P3", + nodeKind: "topic", + nodeSummary: "p3", + statement: "正常链3", + sessionId: "s1", + sourceActor: "user", + }); store.linkNodes({ sourceNodeId: p1.node.id, targetNodeId: p2.node.id, type: "causes" }); store.linkNodes({ sourceNodeId: p2.node.id, targetNodeId: p3.node.id, type: "causes" }); const r = store.detectGraphCycles(); @@ -70,8 +134,22 @@ test("detectGraphCycles is empty for an acyclic chain", () => { test("symmetric relations (contradicts) never count as cycles", () => { withStore((store) => { - const m1 = store.remember({ nodeName: "N1", nodeKind: "topic", nodeSummary: "n1", statement: "观点A", sessionId: "s1", sourceActor: "user" }); - const m2 = store.remember({ nodeName: "N2", nodeKind: "topic", nodeSummary: "n2", statement: "观点B", sessionId: "s1", sourceActor: "user" }); + const m1 = store.remember({ + nodeName: "N1", + nodeKind: "topic", + nodeSummary: "n1", + statement: "观点A", + sessionId: "s1", + sourceActor: "user", + }); + const m2 = store.remember({ + nodeName: "N2", + nodeKind: "topic", + nodeSummary: "n2", + statement: "观点B", + sessionId: "s1", + sourceActor: "user", + }); store.linkNodes({ sourceNodeId: m1.node.id, targetNodeId: m2.node.id, type: "contradicts" }); store.linkNodes({ sourceNodeId: m2.node.id, targetNodeId: m1.node.id, type: "contradicts" }); // A mutual contradiction is normal symmetric semantics, not an anomaly. @@ -85,12 +163,25 @@ test("symmetric relations (contradicts) never count as cycles", () => { test("applySupersession rejects writes that would create a supersede cycle", () => { withStore((store) => { - const a = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "预算5000版1", sessionId: "s1", sourceActor: "user" }); - const b = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "预算5000版2", sessionId: "s1", sourceActor: "user" }); + const a = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "预算5000版1", + sessionId: "s1", + sourceActor: "user", + }); + const b = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "预算5000版2", + sessionId: "s1", + sourceActor: "user", + }); store.applySupersession({ newMemoryId: a.memory.id, supersededMemoryId: b.memory.id }); assert.throws( - () => - store.applySupersession({ newMemoryId: b.memory.id, supersededMemoryId: a.memory.id }), + () => store.applySupersession({ newMemoryId: b.memory.id, supersededMemoryId: a.memory.id }), /supersede cycle/, "mutual supersession is rejected at write time", ); @@ -128,7 +219,14 @@ test("supersedeReachableFrom walks a deep chain and defends at depth", () => { ); assert.ok(reach.has(m1!) && reach.has(m5!), "chain head and tail present"); // Adding a normal successor on top is fine. - const m6 = store.remember({ nodeName: "预算", nodeKind: "topic", nodeSummary: "预算", statement: "深链预算v6", sessionId: "s1", sourceActor: "user" }); + const m6 = store.remember({ + nodeName: "预算", + nodeKind: "topic", + nodeSummary: "预算", + statement: "深链预算v6", + sessionId: "s1", + sourceActor: "user", + }); store.applySupersession({ newMemoryId: m6.memory.id, supersededMemoryId: m5! }); assert.deepEqual(store.detectGraphCycles().supersedeCycles, []); // Closing the loop (m1 supersedes m6) is rejected: m6 reaches m1. @@ -143,19 +241,44 @@ test("supersedeReachableFrom walks a deep chain and defends at depth", () => { test("supersede cycle detection handles loop + inflowing chain + independent chain", () => { withStore((store) => { const mk = (stmt: string) => - store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: stmt, sessionId: "s1", sourceActor: "user" }); + store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: stmt, + sessionId: "s1", + sourceActor: "user", + }); const [a, b, c] = [mk("环A"), mk("环B"), mk("环C")]; - const db = (store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } }).db; + const db = ( + store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } } + ).db; // 3-cycle A→B→C→A (write path rejects cycles, so inject directly). - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(b.memory.id, a.memory.id); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(c.memory.id, b.memory.id); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(a.memory.id, c.memory.id); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + b.memory.id, + a.memory.id, + ); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + c.memory.id, + b.memory.id, + ); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + a.memory.id, + c.memory.id, + ); // Inflowing chain X→A is NOT a cycle member. const x = mk("链尾X"); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(a.memory.id, x.memory.id); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + a.memory.id, + x.memory.id, + ); // Independent acyclic chain Y→Z. - const y = mk("Y"), z = mk("Z"); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(z.memory.id, y.memory.id); + const y = mk("Y"), + z = mk("Z"); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + z.memory.id, + y.memory.id, + ); const r = store.detectGraphCycles(); assert.equal(r.supersedeCycles.length, 1, "exactly the 3-cycle, not inflow/independent chain"); @@ -167,12 +290,34 @@ test("supersede cycle detection handles loop + inflowing chain + independent cha test("breakSupersedeCycle clears intra-cycle supersedes_id edges", () => { withStore((store) => { - const c = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "预算5000版3", sessionId: "s1", sourceActor: "user" }); - const d = store.remember({ nodeName: "X", nodeKind: "topic", nodeSummary: "x", statement: "预算5000版4", sessionId: "s1", sourceActor: "user" }); + const c = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "预算5000版3", + sessionId: "s1", + sourceActor: "user", + }); + const d = store.remember({ + nodeName: "X", + nodeKind: "topic", + nodeSummary: "x", + statement: "预算5000版4", + sessionId: "s1", + sourceActor: "user", + }); // Inject a mutual-supersede anomaly directly (write path would reject it). - const db = (store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } }).db; - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(d.memory.id, c.memory.id); - db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run(c.memory.id, d.memory.id); + const db = ( + store as unknown as { db: { prepare: (sql: string) => { run: (...p: unknown[]) => void } } } + ).db; + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + d.memory.id, + c.memory.id, + ); + db.prepare("UPDATE memory_records SET supersedes_id = ? WHERE id = ?").run( + c.memory.id, + d.memory.id, + ); const r = store.detectGraphCycles(); const cycle = r.supersedeCycles.find((x) => x.includes(c.memory.id) && x.includes(d.memory.id)); assert.ok(cycle, "cycle detected before break"); diff --git a/tests/core/hierarchical-activation.test.ts b/tests/core/hierarchical-activation.test.ts index f8dd4750..71fe97f9 100644 --- a/tests/core/hierarchical-activation.test.ts +++ b/tests/core/hierarchical-activation.test.ts @@ -69,7 +69,10 @@ test("HA h1 state updates across propagate calls (EMA)", () => { // First call: h1 is null → starts from zeros const out1 = ha.propagate(q, cs); const h1a = out1.h1State; - assert.ok(h1a.some((v) => v !== 0), "h1 should be non-zero after first call"); + assert.ok( + h1a.some((v) => v !== 0), + "h1 should be non-zero after first call", + ); // Second call: h1 should change (EMA updates) const out2 = ha.propagate(q, buildCandidates(d, 5)); @@ -102,9 +105,7 @@ test("HA multi-step: prev g3Context changes output", () => { const out2 = ha.propagate(q, cs, [], gs, { g3Context: out1.g3Context }); // With prev, scores should differ - const same = out2.nodeScores.every( - (v, i) => Math.abs(v - out1.nodeScores[i]!) < 1e-7, - ); + const same = out2.nodeScores.every((v, i) => Math.abs(v - out1.nodeScores[i]!) < 1e-7); assert.ok(!same, "multi-step scores should differ from single-step"); }); @@ -117,26 +118,38 @@ test("HA train reduces loss on repeated samples", () => { const cs = buildCandidates(d, 10); const gs = buildGraphState(d); - const r1 = ha.train({ - queryVector: q, - candidates: cs, - graphState: gs, - usedNodeIds: new Set(["c0", "c1"]), - }, 0.05); + const r1 = ha.train( + { + queryVector: q, + candidates: cs, + graphState: gs, + usedNodeIds: new Set(["c0", "c1"]), + }, + 0.05, + ); assert.ok(!isNaN(r1.loss), "loss should be a number"); for (let i = 0; i < 5; i++) { - ha.train({ queryVector: q, candidates: cs, graphState: gs, usedNodeIds: new Set(["c0", "c1"]) }, 0.05); + ha.train( + { queryVector: q, candidates: cs, graphState: gs, usedNodeIds: new Set(["c0", "c1"]) }, + 0.05, + ); } - const r2 = ha.train({ - queryVector: q, - candidates: cs, - graphState: gs, - usedNodeIds: new Set(["c0", "c1"]), - }, 0.05); + const r2 = ha.train( + { + queryVector: q, + candidates: cs, + graphState: gs, + usedNodeIds: new Set(["c0", "c1"]), + }, + 0.05, + ); // Loss should generally decrease (not guaranteed every step, but after several) - assert.ok(r2.loss < r1.loss * 1.1, `loss ${r2.loss} should not be much larger than initial ${r1.loss}`); + assert.ok( + r2.loss < r1.loss * 1.1, + `loss ${r2.loss} should not be much larger than initial ${r1.loss}`, + ); }); test("HA train requires at least one candidate", () => { @@ -183,15 +196,21 @@ test("HA toJSON/fromJSON round-trip produces identical output", () => { test("HA state survives training", () => { const d = 64; const ha = new HierarchicalActivation(d); - ha.train({ - queryVector: rvec(d), - candidates: buildCandidates(d, 5), - usedNodeIds: new Set(["c0"]), - }, 0.05); + ha.train( + { + queryVector: rvec(d), + candidates: buildCandidates(d, 5), + usedNodeIds: new Set(["c0"]), + }, + 0.05, + ); const json = ha.toJSON(); assert.ok(json.trainingSteps > 0, "trainingSteps should be preserved"); - assert.ok(json.scoreWeights.some((w) => w !== 0), "score weights should be set"); + assert.ok( + json.scoreWeights.some((w) => w !== 0), + "score weights should be set", + ); }); // ── determinism ── @@ -227,8 +246,6 @@ test("HA trainSequence requires at least one step with usedNodeIds", () => { const d = 64; const ha = new HierarchicalActivation(d); assert.throws(() => - ha.trainSequence([ - { queryVector: rvec(d), candidates: buildCandidates(d, 3) }, - ]), + ha.trainSequence([{ queryVector: rvec(d), candidates: buildCandidates(d, 3) }]), ); }); diff --git a/tests/core/memory-chains.test.ts b/tests/core/memory-chains.test.ts index c38a58c0..3a82a671 100644 --- a/tests/core/memory-chains.test.ts +++ b/tests/core/memory-chains.test.ts @@ -74,7 +74,10 @@ test("addMemoryToChain keeps ordered membership and getMemoryChain pulls the who got.members.map((m) => m.note), ["阶段1", "阶段2", "阶段3"], ); - assert.deepEqual(got.members.map((m) => m.memoryId), mids); + assert.deepEqual( + got.members.map((m) => m.memoryId), + mids, + ); }); }); @@ -104,16 +107,17 @@ test("membership is idempotent (PK chain_id+memory_id) and remove works", () => store.addMemoryToChain({ chainId: chain.id, memoryId: mids[1]!, position: 2 }); store.removeMemoryFromChain({ chainId: chain.id, memoryId: mids[1]! }); - assert.deepEqual(store.getMemoryChain(chain.id)!.members.map((m) => m.memoryId), [mids[0]]); + assert.deepEqual( + store.getMemoryChain(chain.id)!.members.map((m) => m.memoryId), + [mids[0]], + ); }); }); test("getMemoryChain returns null for unknown chain; add to unknown chain throws", () => { withStore((store) => { assert.equal(store.getMemoryChain("nope"), null); - assert.throws(() => - store.addMemoryToChain({ chainId: "nope", memoryId: "m", position: 0 }), - ); + assert.throws(() => store.addMemoryToChain({ chainId: "nope", memoryId: "m", position: 0 })); }); }); @@ -125,12 +129,12 @@ test("listMemoryChains filters by type and owner", () => { assert.equal(store.listMemoryChains({ chainType: "temporal" }).length, 2); assert.equal(store.listMemoryChains({ ownerSessionId: "s1" }).length, 2); + assert.equal(store.listMemoryChains({ chainType: "temporal", ownerSessionId: "s1" }).length, 1); assert.equal( - store.listMemoryChains({ chainType: "temporal", ownerSessionId: "s1" }).length, - 1, - ); - assert.equal( - store.listMemoryChains({ ownerSessionId: "s2" }).map((c) => c.topic).join(), + store + .listMemoryChains({ ownerSessionId: "s2" }) + .map((c) => c.topic) + .join(), "b", ); }); @@ -139,12 +143,7 @@ test("listMemoryChains filters by type and owner", () => { test("expandChains appends chain members after a hit (recall supplement, no re-rank)", () => { withStore((store) => { const mids: string[] = []; - for (const stmt of [ - "预算跟踪器需求", - "预算收支记录实现", - "预算分类图表添加", - "露营天气讨论", - ]) { + for (const stmt of ["预算跟踪器需求", "预算收支记录实现", "预算分类图表添加", "露营天气讨论"]) { const r = store.remember({ nodeName: "预算", nodeKind: "topic", @@ -160,9 +159,9 @@ test("expandChains appends chain members after a hit (recall supplement, no re-r topic: "预算演进", ownerSessionId: "s1", }); - mids.slice(0, 3).forEach((m, i) => - store.addMemoryToChain({ chainId: chain.id, memoryId: m, position: i }), - ); + mids + .slice(0, 3) + .forEach((m, i) => store.addMemoryToChain({ chainId: chain.id, memoryId: m, position: i })); const chainMemberIds = store.getMemoryChain(chain.id)!.members.map((m) => m.memoryId); // Without expansion, limit=1 returns only the single ranked hit. @@ -233,7 +232,10 @@ test("expandChains follows one adjacent chain through a shared memory, but not a }); const ids = context.results.map((result) => result.memory.id); assert.ok(ids.includes(adjacentEvidence), "one adjacent chain is expanded"); - assert.ok(!ids.includes(secondHopEvidence), "a chain discovered at hop one is not traversed again"); + assert.ok( + !ids.includes(secondHopEvidence), + "a chain discovered at hop one is not traversed again", + ); assert.equal(new Set(ids).size, ids.length, "shared memories are emitted once"); }); }); @@ -265,7 +267,10 @@ test("chain expansion is bounded relative to primary evidence, not only by the a appendedMaxChars: 16_000, appendedMaxRatio: 0.75, }); - assert.deepEqual(context.results.map((result) => result.memory.id), [anchor]); + assert.deepEqual( + context.results.map((result) => result.memory.id), + [anchor], + ); }); }); @@ -304,7 +309,10 @@ test("chain-level MMR spends a bounded slot on diverse evidence instead of a red chainExpansionMaxMembers: 20, }); const ids = context.results.map((result) => result.memory.id); - assert.ok(ids.includes(diverseOnly), "the lower-overlap adjacent chain receives the remaining slot"); + assert.ok( + ids.includes(diverseOnly), + "the lower-overlap adjacent chain receives the remaining slot", + ); assert.ok(!ids.includes(redundantOnly), "the near-duplicate chain remains folded"); }); }); @@ -405,8 +413,16 @@ test("logical chain expansion obeys explicit edge distance independently of chai }).memory.id, ); const chain = store.createMemoryChain({ chainType: "logical", topic: "bounded logical walk" }); - store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: ids[0]!, targetMemoryId: ids[1]! }); - store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: ids[1]!, targetMemoryId: ids[2]! }); + store.addMemoryChainEdge({ + chainId: chain.id, + sourceMemoryId: ids[0]!, + targetMemoryId: ids[1]!, + }); + store.addMemoryChainEdge({ + chainId: chain.id, + sourceMemoryId: ids[1]!, + targetMemoryId: ids[2]!, + }); const oneHop = store.searchContext("graph-hop anchor", { limit: 1, @@ -498,7 +514,10 @@ test("expandChains admits higher-activation evidence before chronological render appendedMaxChars: statements[1]!.length, }); const returned = context.results.map((result) => result.memory.id); - assert.ok(returned.includes(ids[2]!), "higher-activation candidate consumes the shared budget first"); + assert.ok( + returned.includes(ids[2]!), + "higher-activation candidate consumes the shared budget first", + ); assert.ok(!returned.includes(ids[1]!), "lower-activation long candidate cannot crowd it out"); }); }); @@ -661,7 +680,10 @@ test("activation gate: a query-term match escapes position decay at any distance chainExpansionMaxMembers: 10, }); const got = res.results.map((r) => r.memory.statement); - assert.ok(got.some((s) => s.startsWith("beacon ")), "term-matching distant member admitted"); + assert.ok( + got.some((s) => s.startsWith("beacon ")), + "term-matching distant member admitted", + ); assert.ok(!got.includes("填充丁"), "distance-4 filler dropped"); }); }); @@ -767,7 +789,11 @@ test("recency decay is skipped for historical (eventTimeTo) queries", () => { test("addMemoryChainEdge writes a directed DAG edge and auto-joins endpoints as members", () => { withStore((store) => { - const chain = store.createMemoryChain({ chainType: "logical", topic: "事故因果", ownerSessionId: "s1" }); + const chain = store.createMemoryChain({ + chainType: "logical", + topic: "事故因果", + ownerSessionId: "s1", + }); const [a, b, c] = seedMemories(store, 3); store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: a, targetMemoryId: b }); store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: b, targetMemoryId: c }); @@ -775,7 +801,10 @@ test("addMemoryChainEdge writes a directed DAG edge and auto-joins endpoints as assert.equal(edges.length, 2); assert.deepEqual( edges.map((e) => [e.sourceMemoryId, e.targetMemoryId]), - [[a, b], [b, c]], + [ + [a, b], + [b, c], + ], ); assert.equal(edges[0]!.edgeType, "order"); // Endpoints were auto-joined as members (no separate addMemoryToChain). @@ -789,7 +818,11 @@ test("addMemoryChainEdge writes a directed DAG edge and auto-joins endpoints as test("addMemoryChainEdge rejects an edge that would close a directed cycle", () => { withStore((store) => { - const chain = store.createMemoryChain({ chainType: "logical", topic: "反馈回路", ownerSessionId: "s1" }); + const chain = store.createMemoryChain({ + chainType: "logical", + topic: "反馈回路", + ownerSessionId: "s1", + }); const [a, b, c] = seedMemories(store, 3); store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: a, targetMemoryId: b }); store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: b, targetMemoryId: c }); @@ -812,7 +845,11 @@ test("addMemoryChainEdge rejects an edge that would close a directed cycle", () test("topologicalChainOrder returns a deterministic DAG order (branching chain)", () => { withStore((store) => { - const chain = store.createMemoryChain({ chainType: "logical", topic: "分叉事故", ownerSessionId: "s1" }); + const chain = store.createMemoryChain({ + chainType: "logical", + topic: "分叉事故", + ownerSessionId: "s1", + }); const [a, b, c, d] = seedMemories(store, 4); // a --> b, a --> c (branch), c --> d store.addMemoryChainEdge({ chainId: chain.id, sourceMemoryId: a, targetMemoryId: b }); diff --git a/tests/core/openai-embedding.test.ts b/tests/core/openai-embedding.test.ts index a31f0904..81b5f5d3 100644 --- a/tests/core/openai-embedding.test.ts +++ b/tests/core/openai-embedding.test.ts @@ -144,7 +144,10 @@ test("embedding templates require a text placeholder", () => { }); test("BAAI-prefixed and bare BGE model names share one index identity", () => { - const prefixed = new OpenAIEmbeddingClient({ model: "BAAI/bge-small-en-v1.5", profile: "bge-en" }); + const prefixed = new OpenAIEmbeddingClient({ + model: "BAAI/bge-small-en-v1.5", + profile: "bge-en", + }); const bare = new OpenAIEmbeddingClient({ model: "bge-small-en-v1.5", profile: "bge-en" }); // Same model, two spellings → same normalized identity (one embedding index). assert.equal(prefixed.indexId, bare.indexId); diff --git a/tests/core/rank-fusion.test.ts b/tests/core/rank-fusion.test.ts index f7036e41..52dbb466 100644 --- a/tests/core/rank-fusion.test.ts +++ b/tests/core/rank-fusion.test.ts @@ -11,11 +11,14 @@ test("RRF preserves one route and removes duplicate ids", () => { }); test("RRF rewards agreement across retrieval routes", () => { - const fused = reciprocalRankFusion([ - { ids: ["a", "b", "c"], weight: 1.5 }, - { ids: ["c", "b", "d"] }, - ], 4); - assert.deepEqual(fused.map(({ id }) => id), ["b", "c", "a", "d"]); + const fused = reciprocalRankFusion( + [{ ids: ["a", "b", "c"], weight: 1.5 }, { ids: ["c", "b", "d"] }], + 4, + ); + assert.deepEqual( + fused.map(({ id }) => id), + ["b", "c", "a", "d"], + ); }); test("RRF is deterministic and obeys the hard output cap", () => { diff --git a/tests/core/reasoning-workspace.test.ts b/tests/core/reasoning-workspace.test.ts index 125f7f4c..1f60c2bf 100644 --- a/tests/core/reasoning-workspace.test.ts +++ b/tests/core/reasoning-workspace.test.ts @@ -195,7 +195,8 @@ test("removing a reference cannot orphan an already supported downstream conclus () => workspace.updateNode(observation.id, { evidenceRefs: [] }), /would remove support/u, ); - assert.deepEqual(workspace.toJSON().nodes.find((node) => node.id === observation.id)?.evidenceRefs, [ - "tool:result", - ]); + assert.deepEqual( + workspace.toJSON().nodes.find((node) => node.id === observation.id)?.evidenceRefs, + ["tool:result"], + ); }); diff --git a/tests/core/semantic-domain.test.ts b/tests/core/semantic-domain.test.ts index 457e6666..6ef40445 100644 --- a/tests/core/semantic-domain.test.ts +++ b/tests/core/semantic-domain.test.ts @@ -9,10 +9,10 @@ import { } from "../../src/core/semantic-domain.ts"; test("scope compatibility is conjunction intersection, not exact equality", () => { - assert.deepEqual( - intersectScopes({ project: "atlas" }, { project: "atlas", device: "laptop" }), - { project: "atlas", device: "laptop" }, - ); + assert.deepEqual(intersectScopes({ project: "atlas" }, { project: "atlas", device: "laptop" }), { + project: "atlas", + device: "laptop", + }); assert.equal(scopesOverlap({}, { project: "atlas" }), true); assert.equal(scopesOverlap({ project: "atlas" }, { project: "beacon" }), false); assert.equal(intersectScopes({ project: "atlas" }, { project: "beacon" }), null); @@ -30,10 +30,7 @@ test("validity intervals are half-open and missing ends are unbounded", () => { }), true, ); - assert.equal( - validityIntervalsOverlap(january, { validFrom: "2026-02-01T00:00:00.000Z" }), - false, - ); + assert.equal(validityIntervalsOverlap(january, { validFrom: "2026-02-01T00:00:00.000Z" }), false); assert.equal(validityIntervalsOverlap({}, january), true); }); diff --git a/tests/core/stg-v2.test.ts b/tests/core/stg-v2.test.ts index 6cc3b34f..42ce4126 100644 --- a/tests/core/stg-v2.test.ts +++ b/tests/core/stg-v2.test.ts @@ -4,11 +4,7 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import test from "node:test"; -import { - createStgStore, - purgeSessionFromStg, - stgStorePath, -} from "../../src/core/stg.ts"; +import { createStgStore, purgeSessionFromStg, stgStorePath } from "../../src/core/stg.ts"; import { NmgStore } from "../../src/core/store.ts"; import { HashingVectorEmbedder } from "../../src/core/vector.ts"; @@ -98,12 +94,18 @@ test("v2: session-scoped search isolates provisional rows", () => { const forA = stg.searchContext("private plan", { sessionId: "session-a", maxTier: 3 }); const aStatements = forA.results.map((r) => r.memory.statement); - assert.ok(aStatements.some((s) => s.includes("Atlas 2.0")), "A sees its own row"); + assert.ok( + aStatements.some((s) => s.includes("Atlas 2.0")), + "A sees its own row", + ); assert.ok(!aStatements.some((s) => s.includes("legacy pipeline")), "A does not see B's row"); const forB = stg.searchContext("private plan", { sessionId: "session-b", maxTier: 3 }); const bStatements = forB.results.map((r) => r.memory.statement); - assert.ok(bStatements.some((s) => s.includes("legacy pipeline")), "B sees its own row"); + assert.ok( + bStatements.some((s) => s.includes("legacy pipeline")), + "B sees its own row", + ); assert.ok(!bStatements.some((s) => s.includes("Atlas 2.0")), "B does not see A's row"); // anonymous read (no sessionId): provisional rows must NOT leak — only @@ -169,7 +171,10 @@ test("v2: derived retrieval paths cannot expose another session's provisional ro ); const exact = stg.getContext([anchor.memory.id], 1, "session-a"); - assert.deepEqual(exact.results.map((result) => result.memory.id), [anchor.memory.id]); + assert.deepEqual( + exact.results.map((result) => result.memory.id), + [anchor.memory.id], + ); assert.equal(exact.relations.length, 0, "exact expansion also hides session B's relation"); }); }); @@ -283,21 +288,30 @@ test("v2: purgeSession removes only that session's provisional rows", () => { assert.ok(stg.getMemory(cached.memory.id)); // B's row survives under its own session; A's row is gone from its own // session (physical purge, not just a visibility filter). - const forB = stg.searchContext("scratch", { sessionId: "session-b", maxTier: 3 }).results.map( - (r) => r.memory.statement, - ); - assert.ok(forB.some((s) => s.includes("B's fresh")), "B's row kept"); - const forA = stg.searchContext("scratch", { sessionId: "session-a", maxTier: 3 }).results.map( - (r) => r.memory.statement, + const forB = stg + .searchContext("scratch", { sessionId: "session-b", maxTier: 3 }) + .results.map((r) => r.memory.statement); + assert.ok( + forB.some((s) => s.includes("B's fresh")), + "B's row kept", ); + const forA = stg + .searchContext("scratch", { sessionId: "session-a", maxTier: 3 }) + .results.map((r) => r.memory.statement); assert.ok(!forA.some((s) => s.includes("A's stale")), "A's row purged from its own session"); // anonymous read: private rows stay invisible (shared visibility is // already covered by getMemory(cached.memory.id) above) - const remaining = stg.searchContext("scratch", { maxTier: 3 }).results.map( - (r) => r.memory.statement, + const remaining = stg + .searchContext("scratch", { maxTier: 3 }) + .results.map((r) => r.memory.statement); + assert.ok( + !remaining.some((s) => s.includes("A's stale")), + "A's row purged / invisible anonymously", + ); + assert.ok( + !remaining.some((s) => s.includes("B's fresh")), + "anonymous does not see B's private row", ); - assert.ok(!remaining.some((s) => s.includes("A's stale")), "A's row purged / invisible anonymously"); - assert.ok(!remaining.some((s) => s.includes("B's fresh")), "anonymous does not see B's private row"); }); }); diff --git a/tests/core/store/duplicates.test.ts b/tests/core/store/duplicates.test.ts index 8e0a5cd4..66776915 100644 --- a/tests/core/store/duplicates.test.ts +++ b/tests/core/store/duplicates.test.ts @@ -5,10 +5,7 @@ import { join } from "node:path"; import test from "node:test"; import { NmgStore } from "../../../src/core/store.ts"; -import { - normalizeStatement, - statementSimilarity, -} from "../../../src/core/store/search-ranking.ts"; +import { normalizeStatement, statementSimilarity } from "../../../src/core/store/search-ranking.ts"; function withStore(run: (store: NmgStore) => void): void { const directory = mkdtempSync(join(tmpdir(), "nmg-dup-")); @@ -24,7 +21,10 @@ function withStore(run: (store: NmgStore) => void): void { test("normalizeStatement: strips case, punctuation, whitespace", () => { assert.equal(normalizeStatement("I like dogs."), normalizeStatement("i like dogs")); assert.equal(normalizeStatement(" A, B! "), normalizeStatement("a b")); - assert.equal(normalizeStatement("Hello, world — really!"), normalizeStatement("hello world really")); + assert.equal( + normalizeStatement("Hello, world — really!"), + normalizeStatement("hello world really"), + ); }); test("statementSimilarity: exact normalized = 1, partial between, unrelated low", () => { @@ -90,7 +90,9 @@ test("scope write index preserves duplicate and supersession candidates", () => nodeName: "work", scope: { user: "a" }, }); - assert.ok(newer.supersedeCandidates?.some((candidate) => candidate.memoryId === first.memory.id)); + assert.ok( + newer.supersedeCandidates?.some((candidate) => candidate.memoryId === first.memory.id), + ); store.applySupersession({ newMemoryId: newer.memory.id, supersededMemoryId: first.memory.id, @@ -101,7 +103,11 @@ test("scope write index preserves duplicate and supersession candidates", () => scope: { user: "a" }, supersedeScan: false, }); - assert.notEqual(after.memory.id, first.memory.id, "superseded rows must leave the active index"); + assert.notEqual( + after.memory.id, + first.memory.id, + "superseded rows must leave the active index", + ); } finally { store.close(); rmSync(directory, { force: true, recursive: true }); @@ -143,7 +149,10 @@ test("remember: judge merge returns target without writing a new record", () => }); assert.equal(judged.length, 1); assert.equal(judged[0]!.candidates.length, 1); - assert.ok(judged[0]!.candidates[0] && (judged[0]!.candidates[0] as { similarity: number }).similarity >= 0.7); + assert.ok( + judged[0]!.candidates[0] && + (judged[0]!.candidates[0] as { similarity: number }).similarity >= 0.7, + ); // returned the existing record, not a new write assert.equal(result.memory.statement, "I like dogs and cats."); assert.ok(result.duplicates && result.duplicates.length >= 1); @@ -399,8 +408,16 @@ test("searchContext: as-of ranking lifts the record current at the asked date", const i2026 = idx("work-life balance"); assert.ok(i2033 >= 0, "2033 record must be retrieved"); assert.ok(i2026 >= 0, "2026 record must be retrieved"); - assert.equal(h.results[i2033]?.memory.id, new2033.memory.id, "the 2033 slot is the 2033 record"); - assert.equal(h.results[i2026]?.memory.id, old2026.memory.id, "the 2026 slot is the 2026 record"); + assert.equal( + h.results[i2033]?.memory.id, + new2033.memory.id, + "the 2033 slot is the 2033 record", + ); + assert.equal( + h.results[i2026]?.memory.id, + old2026.memory.id, + "the 2026 slot is the 2026 record", + ); assert.ok(i2033 < i2026, `as-of 2033 ranks the 2033 record (${i2033}) above 2026 (${i2026})`); // No window (current query): relevance order is untouched by the temporal boost. @@ -449,7 +466,10 @@ test("searchContext: historical query keeps a superseded value when its successo eventTimeTo: "2026-12-31T00:00:00Z", }); const st26 = h2026.results.map((r) => r.memory.statement); - assert.ok(st26.some((s) => s.includes("senior engineer")), "mid value must survive as-of 2026"); + assert.ok( + st26.some((s) => s.includes("senior engineer")), + "mid value must survive as-of 2026", + ); assert.ok( !st26.some((s) => s.includes("principal engineer")), "successor outside the window must not replace the historical value", @@ -461,7 +481,10 @@ test("searchContext: historical query keeps a superseded value when its successo eventTimeTo: "2031-12-31T00:00:00Z", }); const st31 = h2031.results.map((r) => r.memory.statement); - assert.ok(st31.some((s) => s.includes("principal engineer")), "successor inside window replaces"); + assert.ok( + st31.some((s) => s.includes("principal engineer")), + "successor inside window replaces", + ); assert.ok( !st31.some((s) => s.includes("senior engineer")), "superseded mid dropped when its successor is inside the window", @@ -470,7 +493,10 @@ test("searchContext: historical query keeps a superseded value when its successo // current query (no window): replace with the newest value. const cur = store.searchContext("current job title", { limit: 10 }); const stc = cur.results.map((r) => r.memory.statement); - assert.ok(stc.some((s) => s.includes("principal engineer")), "current query surfaces newest"); + assert.ok( + stc.some((s) => s.includes("principal engineer")), + "current query surfaces newest", + ); assert.ok( !stc.some((s) => s.includes("senior engineer")), "current query drops superseded mid", @@ -597,7 +623,8 @@ test("remember: transition from-side word recalls low-overlap predecessor", () = withStore((store) => { // 旧值:和新语句共享 "Employed"(大写) 但 normalize 后共享 employed + salary 无(措辞差异) store.remember({ - statement: "I am Employed at Huaxin Consulting and earn a monthly salary of twenty thousand yuan.", + statement: + "I am Employed at Huaxin Consulting and earn a monthly salary of twenty thousand yuan.", nodeName: "work", scope: { user: "a" }, }); @@ -620,13 +647,15 @@ test("transition phrase outranks higher-lexical-similarity chit-chat for superse withStore((store) => { // 真正的旧值(含 from 侧词 employed) store.remember({ - statement: "I am currently Employed, working in the healthcare industry at Huaxin Consulting.", + statement: + "I am currently Employed, working in the healthcare industry at Huaxin Consulting.", nodeName: "work", scope: { user: "a" }, }); // 高 sim 闲聊(共享 make/positive/impact/healthcare,但无关) store.remember({ - statement: "I am determined to make a meaningful positive impact in global healthcare through my work.", + statement: + "I am determined to make a meaningful positive impact in global healthcare through my work.", nodeName: "chat", scope: { user: "a" }, }); diff --git a/tests/core/store/leaf-summaries.test.ts b/tests/core/store/leaf-summaries.test.ts index 7e56d857..61f90338 100644 --- a/tests/core/store/leaf-summaries.test.ts +++ b/tests/core/store/leaf-summaries.test.ts @@ -182,14 +182,16 @@ test("leafBlockRouting: chain edges pull cross-block neighbors into the append", }).memory.id; store.rebuildLeafBlocks(); const taskA = store.pendingLeafSummaries().find((t) => t.nodeName === "alpha trips")!; - store.setLeafSummary(taskA.blockId, "zebra index terms for alpha", "test-model", taskA.membersKey); + store.setLeafSummary( + taskA.blockId, + "zebra index terms for alpha", + "test-model", + taskA.membersKey, + ); // Without a chain, the append covers block A only. const plain = store.searchContext("zebra", { leafBlockRouting: true }); - assert.deepEqual( - plain.results.map((r) => r.memory.id).sort(), - [a1, a2].sort(), - ); + assert.deepEqual(plain.results.map((r) => r.memory.id).sort(), [a1, a2].sort()); // A temporal chain a1 → b1 makes the cross-block continuation surface: // b1 lives in another block and is appended with chain markings. @@ -225,7 +227,12 @@ test("leafBlockRouting: member-only chains pull position-adjacent neighbors", () }).memory.id; store.rebuildLeafBlocks(); const taskA = store.pendingLeafSummaries().find((t) => t.nodeName === "alpha trips")!; - store.setLeafSummary(taskA.blockId, "zebra index terms for alpha", "test-model", taskA.membersKey); + store.setLeafSummary( + taskA.blockId, + "zebra index terms for alpha", + "test-model", + taskA.membersKey, + ); const chain = store.createMemoryChain({ chainType: "logical", topic: "march trip" }); store.addMemoryToChain({ chainId: chain.id, memoryId: a1, position: 0 }); diff --git a/tests/core/store/node-summaries.test.ts b/tests/core/store/node-summaries.test.ts index c423976a..79d4c012 100644 --- a/tests/core/store/node-summaries.test.ts +++ b/tests/core/store/node-summaries.test.ts @@ -68,7 +68,12 @@ test("setNodeSummary: persists, indexes node FTS, clears pending", () => { summarizeAllBlocks(store); const task = store.pendingNodeSummaries({ minBlocks: 1 })[0]!; assert.equal( - store.setNodeSummary(task.nodeId, "travel: thai lunch, tokyo flight, neovim editor", "test-model", task.memberCount), + store.setNodeSummary( + task.nodeId, + "travel: thai lunch, tokyo flight, neovim editor", + "test-model", + task.memberCount, + ), true, ); assert.equal(store.pendingNodeSummaries({ minBlocks: 1 }).length, 0, "summary clears pending"); @@ -259,14 +264,12 @@ test("summary routing signal: routed+recalled detail persists and aggregates", ( assert.equal(signal!.recalled, true, "node in base result set is recalled"); // Aggregate tier: the node accumulated a routed (and recalled) counter. - const aggregate = ( - (store as unknown as { db: import("node:sqlite").DatabaseSync }).db - .prepare( - `SELECT summary_routed_count, summary_recalled_count + const aggregate = (store as unknown as { db: import("node:sqlite").DatabaseSync }).db + .prepare( + `SELECT summary_routed_count, summary_recalled_count FROM node_retrieval_signals WHERE node_id = ?`, - ) - .get(node.id) as { summary_routed_count: number; summary_recalled_count: number } - ); + ) + .get(node.id) as { summary_routed_count: number; summary_recalled_count: number }; assert.ok(aggregate.summary_routed_count >= 1, "routed count aggregated"); assert.ok(aggregate.summary_recalled_count >= 1, "recalled count aggregated"); }); @@ -318,7 +321,11 @@ test("summary routing signal: routed but NOT recalled marks the IR gap", () => { test("trainRouter: triple-confirmed nodes learn at twice the base rate", () => { withStore((store) => { - store.remember({ statement: "user booked a flight to Tokyo", nodeName: "alpha", sourceActor: "user" }); + store.remember({ + statement: "user booked a flight to Tokyo", + nodeName: "alpha", + sourceActor: "user", + }); store.remember({ statement: "user likes green tea", nodeName: "beta", sourceActor: "user" }); const aNode = store.searchContext("tokyo", { limit: 1 }).results[0]!.node; const bNode = store.searchContext("green tea", { limit: 1 }).results[0]!.node; @@ -326,18 +333,15 @@ test("trainRouter: triple-confirmed nodes learn at twice the base rate", () => { // aNode is triple-confirmed (boosted lr), bNode is plain use. store.trainRouter(query, [aNode.id, bNode.id], 0.2, [aNode.id]); const readWeights = (nodeId: string): number[] => { - const row = ( - (store as unknown as { db: import("node:sqlite").DatabaseSync }).db - .prepare("SELECT weights_json FROM router_weights WHERE node_id = ?") - .get(nodeId) as { weights_json: string } - ); + const row = (store as unknown as { db: import("node:sqlite").DatabaseSync }).db + .prepare("SELECT weights_json FROM router_weights WHERE node_id = ?") + .get(nodeId) as { weights_json: string }; return JSON.parse(row.weights_json) as number[]; }; // Cosine is scale-invariant: from a zero init, both nodes sit on the query // direction after one update regardless of lr. The boosted lr instead shows // up in the vector NORM (each update moves confirmed nodes farther). - const norm = (v: readonly number[]): number => - Math.sqrt(v.reduce((s, x) => s + x * x, 0)); + const norm = (v: readonly number[]): number => Math.sqrt(v.reduce((s, x) => s + x * x, 0)); const normA = norm(readWeights(aNode.id)); const normB = norm(readWeights(bNode.id)); assert.ok( @@ -368,8 +372,16 @@ test("summaryRouteGapReport: routed∧!recalled nodes surface as the IR gap", () 1, ); // Two queries both gap the travel node (summary-routed, base misses it). - store.searchContext("plans", { limit: 8, leafBlockRouting: true, leafBlockRoutingMaxMembers: 12 }); - store.searchContext("plans", { limit: 8, leafBlockRouting: true, leafBlockRoutingMaxMembers: 12 }); + store.searchContext("plans", { + limit: 8, + leafBlockRouting: true, + leafBlockRoutingMaxMembers: 12, + }); + store.searchContext("plans", { + limit: 8, + leafBlockRouting: true, + leafBlockRoutingMaxMembers: 12, + }); const report = store.summaryRouteGapReport(10); const travel = report.find((r) => r.nodeId === travelNode.id); assert.ok(travel, "travel node appears in the gap report"); diff --git a/tests/core/store/retention.test.ts b/tests/core/store/retention.test.ts index f8a43e4d..974f0714 100644 --- a/tests/core/store/retention.test.ts +++ b/tests/core/store/retention.test.ts @@ -113,7 +113,7 @@ test("session-private STG memories do not enter the shared LTG retention lifecyc nodeName: "session hypothesis", memoryType: "derived", residence: "stg", - sessionId: "test-session", + sessionId: "test-session", }); assert.throws( () => store.setMemoryStorageState(provisional.memory.id, "dormant"), @@ -182,13 +182,13 @@ test("open STG memories do not expire before resolution", () => { statement: "The current session is testing Atlas storage", nodeName: "Atlas storage session", residence: "stg", - sessionId: "test-session", + sessionId: "test-session", }); const open = store.remember({ statement: "Check the Atlas storage benchmark result", nodeName: "Atlas storage benchmark", residence: "stg", - sessionId: "test-session", + sessionId: "test-session", resolution: "open", relatedMemoryIds: [anchor.memory.id], expiresAt: "2000-01-01T00:00:00.000Z", diff --git a/tests/core/store/retrieval.test.ts b/tests/core/store/retrieval.test.ts index ad9e7fc4..b60a3d95 100644 --- a/tests/core/store/retrieval.test.ts +++ b/tests/core/store/retrieval.test.ts @@ -78,7 +78,10 @@ test("lexical tie-break keeps Active Graph selection order aligned with returned ); const trace = store.retrievalTrace(context.activeGraph.id); assert.deepEqual(trace?.resultMemoryIds, context.activeGraph.memoryIds); - assert.deepEqual(trace?.selections?.map((selection) => selection.memoryId), context.activeGraph.memoryIds); + assert.deepEqual( + trace?.selections?.map((selection) => selection.memoryId), + context.activeGraph.memoryIds, + ); }); }); diff --git a/tests/core/store/search-ranking.test.ts b/tests/core/store/search-ranking.test.ts index b1f5b7c9..42114fe5 100644 --- a/tests/core/store/search-ranking.test.ts +++ b/tests/core/store/search-ranking.test.ts @@ -9,14 +9,21 @@ test("query coverage resolves lexical score ties without crossing score boundari { id: "complete", score: 4, text: "Atlas project uses SQLite" }, { id: "lower", score: 2, text: "Atlas project uses SQLite" }, ]; - const rank = (items: typeof candidates) => rerankEqualScoresByQueryCoverage( - "Atlas project SQLite", - items, - (item) => item.score, - (item) => item.text, - ); + const rank = (items: typeof candidates) => + rerankEqualScoresByQueryCoverage( + "Atlas project SQLite", + items, + (item) => item.score, + (item) => item.text, + ); const ranked = rank(candidates); - assert.deepEqual(ranked.map((item) => item.id), ["complete", "partial", "lower"]); + assert.deepEqual( + ranked.map((item) => item.id), + ["complete", "partial", "lower"], + ); assert.deepEqual(rank(ranked), ranked, "reranking is stable on repeated calls"); - assert.deepEqual(candidates.map((item) => item.id), ["partial", "complete", "lower"]); + assert.deepEqual( + candidates.map((item) => item.id), + ["partial", "complete", "lower"], + ); }); diff --git a/tests/core/store/vector-codec.test.ts b/tests/core/store/vector-codec.test.ts index d48ce993..8e8e4178 100644 --- a/tests/core/store/vector-codec.test.ts +++ b/tests/core/store/vector-codec.test.ts @@ -1,11 +1,7 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { - encodeVector, - parseVector, - storedVector, -} from "../../../src/core/store/vector-codec.ts"; +import { encodeVector, parseVector, storedVector } from "../../../src/core/store/vector-codec.ts"; test("encodeVector and parseVector round-trip through float32", () => { const original = [0.5, -0.25, 1, 0]; @@ -25,7 +21,7 @@ test("parseVector returns empty for malformed or absent values", () => { assert.deepEqual(parseVector("not json"), []); assert.deepEqual(parseVector(null), []); assert.deepEqual(parseVector(undefined), []); - assert.deepEqual(parseVector("[1,\"x\",null,2]"), [1, 2]); + assert.deepEqual(parseVector('[1,"x",null,2]'), [1, 2]); }); test("storedVector prefers the binary column over legacy JSON", () => { diff --git a/tests/core/store/writes.test.ts b/tests/core/store/writes.test.ts index f2120295..632911e8 100644 --- a/tests/core/store/writes.test.ts +++ b/tests/core/store/writes.test.ts @@ -71,7 +71,9 @@ test("remember validates exact harness evidence provenance", () => { test("remember bounds supersession prefilter terms for very long evidence", () => { withStore((store) => { store.remember({ statement: "baseline durable fact", nodeName: "Long evidence" }); - const statement = Array.from({ length: 1_500 }, (_, index) => `distincttoken${index}`).join(" "); + const statement = Array.from({ length: 1_500 }, (_, index) => `distincttoken${index}`).join( + " ", + ); const result = store.remember({ statement, nodeName: "Long evidence" }); assert.equal(result.memory.statement, statement); }); diff --git a/tests/core/task-board-deliverable.test.ts b/tests/core/task-board-deliverable.test.ts index 01c6ac7e..adc1d3e9 100644 --- a/tests/core/task-board-deliverable.test.ts +++ b/tests/core/task-board-deliverable.test.ts @@ -181,11 +181,7 @@ test("renewing your own live claim does not start a new attempt", () => { agentId: "worker-b", }); assert.equal(renewed.attempt, 1, "a heartbeat is not a new attempt"); - assert.equal( - renewed.deliverableDigest, - "d1", - "a heartbeat must not discard work in progress", - ); + assert.equal(renewed.deliverableDigest, "d1", "a heartbeat must not discard work in progress"); }); }); diff --git a/tests/docker/container-definition.test.ts b/tests/docker/container-definition.test.ts index a7c8c7b6..05f256b5 100644 --- a/tests/docker/container-definition.test.ts +++ b/tests/docker/container-definition.test.ts @@ -3,10 +3,7 @@ import { readFileSync } from "node:fs"; import test from "node:test"; const dockerfile = readFileSync(new URL("../../Dockerfile", import.meta.url), "utf8"); -const entrypoint = readFileSync( - new URL("../../docker/entrypoint.sh", import.meta.url), - "utf8", -); +const entrypoint = readFileSync(new URL("../../docker/entrypoint.sh", import.meta.url), "utf8"); test("external target stays free of the bundled embedding environment", () => { const baseStart = dockerfile.indexOf("AS nmg-runtime"); diff --git a/tests/evals/benchmarks/loaders.test.ts b/tests/evals/benchmarks/loaders.test.ts index 495f5979..92e8fcdf 100644 --- a/tests/evals/benchmarks/loaders.test.ts +++ b/tests/evals/benchmarks/loaders.test.ts @@ -4,24 +4,36 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import test from "node:test"; -import { loadBeam, loadLocomo, loadPersonaMem, stratifiedSample } from "../../../evals/benchmarks/loaders.ts"; +import { + loadBeam, + loadLocomo, + loadPersonaMem, + stratifiedSample, +} from "../../../evals/benchmarks/loaders.ts"; test("loads LoCoMo sessions, evidence ids, and QA categories", () => { withTempDirectory((directory) => { const path = join(directory, "locomo.json"); - writeFileSync(path, JSON.stringify([{ - sample_id: "conv-1", - conversation: { - speaker_a: "Alex", - speaker_b: "Sam", - session_1_date_time: "2026-01-01", - session_1: [ - { speaker: "Alex", dia_id: "d1", text: "I prefer tea." }, - { speaker: "Sam", dia_id: "d2", text: "Noted." }, - ], - }, - qa: [{ question: "What does Alex prefer?", answer: "tea", category: 1, evidence: ["d1"] }], - }])); + writeFileSync( + path, + JSON.stringify([ + { + sample_id: "conv-1", + conversation: { + speaker_a: "Alex", + speaker_b: "Sam", + session_1_date_time: "2026-01-01", + session_1: [ + { speaker: "Alex", dia_id: "d1", text: "I prefer tea." }, + { speaker: "Sam", dia_id: "d2", text: "Noted." }, + ], + }, + qa: [ + { question: "What does Alex prefer?", answer: "tea", category: 1, evidence: ["d1"] }, + ], + }, + ]), + ); const [item] = loadLocomo(path); assert.equal(item?.question, "What does Alex prefer?"); assert.equal(item?.sessions[0]?.turns[0]?.role, "user"); @@ -35,18 +47,24 @@ test("joins PersonaMem CSV questions to sliced JSONL contexts", () => { withTempDirectory((directory) => { const questions = join(directory, "questions.csv"); const contexts = join(directory, "contexts.jsonl"); - writeFileSync(questions, [ - "persona_id,question_id,question_type,user_question_or_message,correct_answer,all_options,shared_context_id,end_index_in_shared_context", - '1,q1,preference,"What should I drink?",(b),"[""(a) coffee"",""(b) tea""]",ctx,2', - ].join("\n")); - writeFileSync(contexts, `${JSON.stringify({ - shared_context_id: "ctx", - messages: [ - { role: "user", content: "I prefer tea." }, - { role: "assistant", content: "Okay." }, - { role: "user", content: "This turn is after the question." }, - ], - })}\n`); + writeFileSync( + questions, + [ + "persona_id,question_id,question_type,user_question_or_message,correct_answer,all_options,shared_context_id,end_index_in_shared_context", + '1,q1,preference,"What should I drink?",(b),"[""(a) coffee"",""(b) tea""]",ctx,2', + ].join("\n"), + ); + writeFileSync( + contexts, + `${JSON.stringify({ + shared_context_id: "ctx", + messages: [ + { role: "user", content: "I prefer tea." }, + { role: "assistant", content: "Okay." }, + { role: "user", content: "This turn is after the question." }, + ], + })}\n`, + ); const [item] = loadPersonaMem(questions, contexts); assert.equal(item?.sessions[0]?.turns.length, 2); assert.deepEqual(item?.options, ["(a) coffee", "(b) tea"]); @@ -60,21 +78,33 @@ test("loads BEAM directory chats and probing categories", () => { const caseDirectory = join(directory, "1"); const probingDirectory = join(caseDirectory, "probing_questions"); mkdirSync(probingDirectory, { recursive: true }); - writeFileSync(join(caseDirectory, "chat.json"), JSON.stringify([{ - batch_number: 1, - turns: [[ - { role: "user", id: 7, content: "My deadline is Friday." }, - { role: "assistant", id: 8, content: "Understood." }, - ]], - }])); - writeFileSync(join(probingDirectory, "probing_questions.json"), JSON.stringify({ - information_extraction: [{ - question: "When is the deadline?", - ideal_answer: "Friday", - source_chat_ids: [7], - }], - abstention: [{ question: "What is the budget?", ideal_response: "Unknown" }], - })); + writeFileSync( + join(caseDirectory, "chat.json"), + JSON.stringify([ + { + batch_number: 1, + turns: [ + [ + { role: "user", id: 7, content: "My deadline is Friday." }, + { role: "assistant", id: 8, content: "Understood." }, + ], + ], + }, + ]), + ); + writeFileSync( + join(probingDirectory, "probing_questions.json"), + JSON.stringify({ + information_extraction: [ + { + question: "When is the deadline?", + ideal_answer: "Friday", + source_chat_ids: [7], + }, + ], + abstention: [{ question: "What is the budget?", ideal_response: "Unknown" }], + }), + ); const cases = loadBeam(directory); assert.equal(cases.length, 2); assert.equal(cases[0]?.sessions[0]?.turns[0]?.sourceId, "7"); @@ -91,14 +121,19 @@ test("BEAM only treats official source_chat_ids as retrieval evidence", () => { const probingDirectory = join(caseDirectory, "probing_questions"); mkdirSync(probingDirectory, { recursive: true }); writeFileSync(join(caseDirectory, "chat.json"), JSON.stringify([])); - writeFileSync(join(probingDirectory, "probing_questions.json"), JSON.stringify({ - abstention: [{ - question: "Unknown?", - ideal_response: "Unknown", - conversation_references: ["Session 99"], - rubric: ["The response abstains"], - }], - })); + writeFileSync( + join(probingDirectory, "probing_questions.json"), + JSON.stringify({ + abstention: [ + { + question: "Unknown?", + ideal_response: "Unknown", + conversation_references: ["Session 99"], + rubric: ["The response abstains"], + }, + ], + }), + ); const [item] = loadBeam(directory); assert.equal(item?.evidenceIds, undefined); assert.deepEqual(item?.rubric, ["The response abstains"]); diff --git a/tests/evals/bge-batcher.test.ts b/tests/evals/bge-batcher.test.ts index 2d171adf..41692a1c 100644 --- a/tests/evals/bge-batcher.test.ts +++ b/tests/evals/bge-batcher.test.ts @@ -8,7 +8,7 @@ const repoRoot = resolve(import.meta.dirname, "..", ".."); const gpuPython = resolve(repoRoot, ".benchmarks", "bge-venv", "Scripts", "python.exe"); const python = existsSync(gpuPython) ? gpuPython - : process.env.PYTHON ?? (process.platform === "win32" ? "python" : "python3"); + : (process.env.PYTHON ?? (process.platform === "win32" ? "python" : "python3")); function runPython(source: string) { return spawnSync(python, ["-c", source], { diff --git a/tests/evals/bge-server-contract.test.ts b/tests/evals/bge-server-contract.test.ts index 604bbc36..d28ca7c3 100644 --- a/tests/evals/bge-server-contract.test.ts +++ b/tests/evals/bge-server-contract.test.ts @@ -8,7 +8,7 @@ const repoRoot = resolve(import.meta.dirname, "..", ".."); const gpuPython = resolve(repoRoot, ".benchmarks", "bge-venv", "Scripts", "python.exe"); const python = existsSync(gpuPython) ? gpuPython - : process.env.PYTHON ?? (process.platform === "win32" ? "python" : "python3"); + : (process.env.PYTHON ?? (process.platform === "win32" ? "python" : "python3")); test("BGE has one Python service entrypoint with explicit device policy", () => { const canonical = resolve(repoRoot, "evals", "omnimemeval", "bge_server.py"); diff --git a/tests/evals/cache-environment-simulator.test.ts b/tests/evals/cache-environment-simulator.test.ts index fd50fe9b..32f67ca0 100644 --- a/tests/evals/cache-environment-simulator.test.ts +++ b/tests/evals/cache-environment-simulator.test.ts @@ -31,7 +31,10 @@ test("fixed-byte chains expose only same-length prefix hashes", () => { assert.deepEqual(left.checkpoints.slice(0, 2), right.checkpoints.slice(0, 2)); assert.notEqual(left.requestHash, right.requestHash); - assert.deepEqual(left.checkpoints.map((checkpoint) => checkpoint.length), [8, 16, 20]); + assert.deepEqual( + left.checkpoints.map((checkpoint) => checkpoint.length), + [8, 16, 20], + ); assert.ok(!JSON.stringify(left).includes("0123456789abcdef")); }); @@ -49,10 +52,12 @@ test("requests arriving while a shared prefix is building miss, then a later req }); test("expired prefixes no longer count as reusable", () => { - const report = simulateCacheEnvironment( - [request("a", 0, "aaaa"), request("b", 51, "bbbb")], - { ...environment, concurrency: 1, cacheBuildMs: 0, cacheTtlMs: 50 }, - ); + const report = simulateCacheEnvironment([request("a", 0, "aaaa"), request("b", 51, "bbbb")], { + ...environment, + concurrency: 1, + cacheBuildMs: 0, + cacheTtlMs: 50, + }); assert.equal(report.hitRequests, 0); assert.equal(report.reusablePrefixBytes, 0); @@ -75,7 +80,7 @@ test("fake API accepts ordinary chat requests and reports hashes without prompt body, }); assert.equal(completion.status, 200); - const completionBody = await completion.json() as { + const completionBody = (await completion.json()) as { choices: Array<{ message: { content: string } }>; }; assert.equal(completionBody.choices[0]?.message.content, "CACHE_TRACE_ONLY"); diff --git a/tests/evals/consolidation/run.test.ts b/tests/evals/consolidation/run.test.ts index 51ee549e..a795e8e0 100644 --- a/tests/evals/consolidation/run.test.ts +++ b/tests/evals/consolidation/run.test.ts @@ -36,11 +36,7 @@ test("LoCoMo audit deduplicates evidence within a task and reports repeated cove answer: "answer", evidence: index === 0 ? ["d1", "d1"] : ["d1"], })); - writeFileSync( - path, - JSON.stringify([{ sample_id: "sample", conversation: {}, qa }]), - "utf8", - ); + writeFileSync(path, JSON.stringify([{ sample_id: "sample", conversation: {}, qa }]), "utf8"); try { const report = evaluateLocomoConsolidation(path); assert.equal(report.cases, 5); diff --git a/tests/evals/context-live-canary.test.ts b/tests/evals/context-live-canary.test.ts index ddcdfb3e..454ce051 100644 --- a/tests/evals/context-live-canary.test.ts +++ b/tests/evals/context-live-canary.test.ts @@ -1,14 +1,7 @@ import assert from "node:assert/strict"; import { test } from "node:test"; import { execFileSync } from "node:child_process"; -import { - mkdtempSync, - readFileSync, - rmSync, - writeFileSync, - readdirSync, - existsSync, -} from "node:fs"; +import { mkdtempSync, readFileSync, rmSync, writeFileSync, readdirSync, existsSync } from "node:fs"; import { tmpdir } from "node:os"; import { join, dirname } from "node:path"; import { fileURLToPath } from "node:url"; @@ -19,13 +12,7 @@ import { fileURLToPath } from "node:url"; // search/dataset artifact and asserts selection rule + run-dir uniqueness. const REPO_ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", ".."); -const CANARY = join( - REPO_ROOT, - "evals", - "omnimemeval", - "research", - "context-live-canary.py", -); +const CANARY = join(REPO_ROOT, "evals", "omnimemeval", "research", "context-live-canary.py"); const PYTHON = process.env.PYTHON || "python"; @@ -65,19 +52,25 @@ test("research canary: dry-run selection follows order-first rule and filters ca writeFileSync(searchFile, JSON.stringify(search)); const dataset = Array.from({ length: 6 }, () => ({ qa: [] as unknown[] })); - (dataset[0].qa as { question: string; answer: string; category: string }[]).push( - { question: "q0", answer: "a0", category: "2" }, - ); + (dataset[0].qa as { question: string; answer: string; category: string }[]).push({ + question: "q0", + answer: "a0", + category: "2", + }); (dataset[1].qa as { question: string; answer: string; category: string }[]).push( { question: "q1_cat5", answer: "a", category: "5" }, { question: "q1", answer: "a1", category: "3" }, ); - (dataset[2].qa as { question: string; answer: string; category: string }[]).push( - { question: "q2", answer: "a2", category: "4" }, - ); - (dataset[3].qa as { question: string; answer: string; category: string }[]).push( - { question: "q3", answer: "a3", category: "1" }, - ); + (dataset[2].qa as { question: string; answer: string; category: string }[]).push({ + question: "q2", + answer: "a2", + category: "4", + }); + (dataset[3].qa as { question: string; answer: string; category: string }[]).push({ + question: "q3", + answer: "a3", + category: "1", + }); const datasetFile = join(dir, "locomo10.json"); writeFileSync(datasetFile, JSON.stringify(dataset)); diff --git a/tests/evals/controller-shadow/tau-worker.test.ts b/tests/evals/controller-shadow/tau-worker.test.ts index cd142bfb..dbaaf246 100644 --- a/tests/evals/controller-shadow/tau-worker.test.ts +++ b/tests/evals/controller-shadow/tau-worker.test.ts @@ -17,10 +17,10 @@ function row(index: number, split: "train" | "validation", qpp: number, useful: } test("rolling tau remains shadow-only and fails closed on sparse data", () => { - const artifact = calibrateRollingTau([ - row(0, "train", 0.2, true), - row(1, "validation", 0.8, false), - ], { generatedAt: "2026-08-13T00:00:00.000Z" }); + const artifact = calibrateRollingTau( + [row(0, "train", 0.2, true), row(1, "validation", 0.8, false)], + { generatedAt: "2026-08-13T00:00:00.000Z" }, + ); assert.equal(artifact.eligibleForShadow, false); assert.equal(artifact.eligibleForActivation, false); assert.ok(artifact.blockers.some((blocker) => blocker.includes("50"))); diff --git a/tests/evals/halumem/agent-extract.test.ts b/tests/evals/halumem/agent-extract.test.ts index 3fc9383b..0e998593 100644 --- a/tests/evals/halumem/agent-extract.test.ts +++ b/tests/evals/halumem/agent-extract.test.ts @@ -14,7 +14,10 @@ test("agent extraction parser accepts bare and fenced durable memories", () => { ], }; assert.deepEqual(parseAgentExtraction(JSON.stringify(object)), object.memories); - assert.deepEqual(parseAgentExtraction(`\`\`\`json\n${JSON.stringify(object)}\n\`\`\``), object.memories); + assert.deepEqual( + parseAgentExtraction(`\`\`\`json\n${JSON.stringify(object)}\n\`\`\``), + object.memories, + ); }); test("agent extraction parser fails closed on unsupported or unattributed output", () => { @@ -31,7 +34,9 @@ test("agent extraction maps only documented semantic aliases", () => { assert.equal( parseAgentExtraction( JSON.stringify({ - memories: [{ statement: "The user aims to help.", memoryType: "goal", evidence: "I aim to help." }], + memories: [ + { statement: "The user aims to help.", memoryType: "goal", evidence: "I aim to help." }, + ], }), )[0]?.memoryType, "fact", diff --git a/tests/evals/longmemeval/official.test.ts b/tests/evals/longmemeval/official.test.ts index a6152f0b..a275c155 100644 --- a/tests/evals/longmemeval/official.test.ts +++ b/tests/evals/longmemeval/official.test.ts @@ -10,19 +10,31 @@ test("loads official LongMemEval evidence labels without deriving them", () => { const directory = mkdtempSync(join(tmpdir(), "nmg-longmem-official-")); try { const path = join(directory, "data.json"); - writeFileSync(path, JSON.stringify([{ - question_id: "q1", - question_type: "single-session-user", - question: "What changed?", - answer: "The deadline", - question_date: "2026-01-03", - haystack_session_ids: ["s1", "s2"], - haystack_dates: ["2026-01-01", "2026-01-02"], - haystack_sessions: [[{ role: "user", content: "Old" }], [{ - role: "user", content: "New", has_answer: true, - }]], - answer_session_ids: ["s2"], - }])); + writeFileSync( + path, + JSON.stringify([ + { + question_id: "q1", + question_type: "single-session-user", + question: "What changed?", + answer: "The deadline", + question_date: "2026-01-03", + haystack_session_ids: ["s1", "s2"], + haystack_dates: ["2026-01-01", "2026-01-02"], + haystack_sessions: [ + [{ role: "user", content: "Old" }], + [ + { + role: "user", + content: "New", + has_answer: true, + }, + ], + ], + answer_session_ids: ["s2"], + }, + ]), + ); const [item] = loadLongMemEval(path); assert.deepEqual(item?.answer_session_ids, ["s2"]); assert.equal(item?.haystack_sessions[1]?.[0]?.has_answer, true); @@ -36,8 +48,7 @@ test("scores LongMemEval retrieval from official session IDs", () => { recallAny: 1, recallAll: 1, recall: 1, - ndcg: (1 / Math.log2(3) + 1 / Math.log2(4)) / - (1 / Math.log2(2) + 1 / Math.log2(3)), + ndcg: (1 / Math.log2(3) + 1 / Math.log2(4)) / (1 / Math.log2(2) + 1 / Math.log2(3)), }); assert.equal(scoreLongMemRetrieval(["x"], []), null); }); @@ -46,11 +57,22 @@ test("normalizes official numeric LongMemEval answers", () => { const directory = mkdtempSync(join(tmpdir(), "nmg-longmem-number-")); try { const path = join(directory, "data.json"); - writeFileSync(path, JSON.stringify([{ - question_id: "q-number", question_type: "multi-session", question: "How many?", - answer: 3, question_date: "2026-01-01", haystack_session_ids: ["s1"], - haystack_dates: ["2026-01-01"], haystack_sessions: [[]], answer_session_ids: ["s1"], - }])); + writeFileSync( + path, + JSON.stringify([ + { + question_id: "q-number", + question_type: "multi-session", + question: "How many?", + answer: 3, + question_date: "2026-01-01", + haystack_session_ids: ["s1"], + haystack_dates: ["2026-01-01"], + haystack_sessions: [[]], + answer_session_ids: ["s1"], + }, + ]), + ); assert.equal(loadLongMemEval(path)[0]?.answer, "3"); } finally { rmSync(directory, { recursive: true, force: true }); diff --git a/tests/evals/longmemeval/report.test.ts b/tests/evals/longmemeval/report.test.ts index 1e60d45a..2c97986f 100644 --- a/tests/evals/longmemeval/report.test.ts +++ b/tests/evals/longmemeval/report.test.ts @@ -16,27 +16,51 @@ import { const rows = [ { - questionId: "q1", repeat: 0, mode: "flat", passed: true, durationMs: 10, + questionId: "q1", + repeat: 0, + mode: "flat", + passed: true, + durationMs: 10, retrievalPassed: true, }, { - questionId: "q2", repeat: 0, mode: "flat", passed: false, durationMs: 30, + questionId: "q2", + repeat: 0, + mode: "flat", + passed: false, + durationMs: 30, retrievalPassed: true, }, { - questionId: "q1", repeat: 0, mode: "nmg", passed: false, durationMs: 20, + questionId: "q1", + repeat: 0, + mode: "nmg", + passed: false, + durationMs: 20, retrievalPassed: false, }, { - questionId: "q2", repeat: 0, mode: "nmg", passed: true, durationMs: 40, + questionId: "q2", + repeat: 0, + mode: "nmg", + passed: true, + durationMs: 40, retrievalPassed: true, }, { - questionId: "q1", repeat: 1, mode: "flat", passed: false, durationMs: 50, + questionId: "q1", + repeat: 1, + mode: "flat", + passed: false, + durationMs: 50, retrievalPassed: false, }, { - questionId: "q1", repeat: 1, mode: "nmg", passed: true, durationMs: 60, + questionId: "q1", + repeat: 1, + mode: "nmg", + passed: true, + durationMs: 60, retrievalPassed: null, }, ]; diff --git a/tests/evals/longmemeval/retrieval-evidence.test.ts b/tests/evals/longmemeval/retrieval-evidence.test.ts index 3d99887f..92f22653 100644 --- a/tests/evals/longmemeval/retrieval-evidence.test.ts +++ b/tests/evals/longmemeval/retrieval-evidence.test.ts @@ -13,7 +13,8 @@ import { NmgStore } from "../../../src/core/store.ts"; const directories: string[] = []; afterEach(() => { - for (const directory of directories.splice(0)) rmSync(directory, { recursive: true, force: true }); + for (const directory of directories.splice(0)) + rmSync(directory, { recursive: true, force: true }); }); describe("LongMemEval automatic recall evidence", () => { diff --git a/tests/evals/natural-maintenance-audit.test.ts b/tests/evals/natural-maintenance-audit.test.ts index f0b72d52..03056bd8 100644 --- a/tests/evals/natural-maintenance-audit.test.ts +++ b/tests/evals/natural-maintenance-audit.test.ts @@ -240,7 +240,9 @@ test("natural maintenance audit reads claim, consolidation, and topology evidenc ), ); assert.equal( - report.ltg.stgConsolidation.materializations.some((item) => item.memoryId === manualLtgMemoryId), + report.ltg.stgConsolidation.materializations.some( + (item) => item.memoryId === manualLtgMemoryId, + ), false, "a manual LTG row without the source marker is not a materialization", ); diff --git a/tests/evals/official/protocol.test.ts b/tests/evals/official/protocol.test.ts index 4495a3e1..e968c2bb 100644 --- a/tests/evals/official/protocol.test.ts +++ b/tests/evals/official/protocol.test.ts @@ -11,10 +11,7 @@ import { normalizedKendallTauB, personaMemCorrect, } from "../../../evals/official/protocol.ts"; -import { - officialPythonExecutable, - probePython, -} from "../../../evals/official/python.ts"; +import { officialPythonExecutable, probePython } from "../../../evals/official/python.ts"; test("PersonaMem uses the official single-option extraction rule", () => { assert.equal(personaMemCorrect("(b)", "(b)"), true); @@ -23,12 +20,14 @@ test("PersonaMem uses the official single-option extraction rule", () => { }); test("LongMemEval protocol preserves update and abstention instructions", () => { - assert.match(longMemEvalJudgePrompt( - "knowledge-update", "Q", "A", "H", false, - ), /previous information.*updated answer/su); - assert.match(longMemEvalJudgePrompt( - "single-session-user", "Q", "A", "H", true, - ), /unanswerable question/u); + assert.match( + longMemEvalJudgePrompt("knowledge-update", "Q", "A", "H", false), + /previous information.*updated answer/su, + ); + assert.match( + longMemEvalJudgePrompt("single-session-user", "Q", "A", "H", true), + /unanswerable question/u, + ); }); test("BEAM protocol includes the official rubric inputs and score scale", () => { @@ -50,10 +49,7 @@ test("BEAM event ordering uses normalized Kendall tau-b", () => { assert.equal(normalizedKendallTauB([0, 1, 2], [2, 1, 0]), 0); const partial = normalizedKendallTauB([0, 1, 2], [0, 2]); assert.ok(partial > 0 && partial < 1); - assert.equal( - normalizedKendallTauB([0, 1, 2], [3]), - 0.1464466094067262, - ); + assert.equal(normalizedKendallTauB([0, 1, 2], [3]), 0.1464466094067262); }); test("LoCoMo bridge invokes the pinned official scorer when bootstrapped", (context) => { @@ -75,22 +71,24 @@ test("LoCoMo bridge invokes the pinned official scorer when bootstrapped", (cont } const result = spawnSync(python, [resolve(root, "evals/official/locomo_score.py")], { cwd: root, - input: JSON.stringify({ qas: [ - { - answer: "tea", - category: 2, - evidence: ["d1"], - prediction: "Tea", - prediction_context: ["d1"], - }, - { - answer: "coffee", - category: 2, - evidence: ["d2"], - prediction: "Coffee", - prediction_context: [], - }, - ] }), + input: JSON.stringify({ + qas: [ + { + answer: "tea", + category: 2, + evidence: ["d1"], + prediction: "Tea", + prediction_context: ["d1"], + }, + { + answer: "coffee", + category: 2, + evidence: ["d2"], + prediction: "Coffee", + prediction_context: [], + }, + ], + }), encoding: "utf8", }); assert.equal(result.status, 0, result.stderr); diff --git a/tests/evals/official/python.test.ts b/tests/evals/official/python.test.ts index 0ad00039..89413fc8 100644 --- a/tests/evals/official/python.test.ts +++ b/tests/evals/official/python.test.ts @@ -2,10 +2,7 @@ import assert from "node:assert/strict"; import { resolve } from "node:path"; import test from "node:test"; -import { - officialPythonExecutable, - probePython, -} from "../../../evals/official/python.ts"; +import { officialPythonExecutable, probePython } from "../../../evals/official/python.ts"; test("official Python uses one explicit override without a fallback chain", () => { assert.equal( diff --git a/tests/evals/omnimemeval-install.test.ts b/tests/evals/omnimemeval-install.test.ts index 7bc9feea..308eddb9 100644 --- a/tests/evals/omnimemeval-install.test.ts +++ b/tests/evals/omnimemeval-install.test.ts @@ -23,11 +23,7 @@ test("OmniMemEval adapter installer patches the registry idempotently", () => { "utf8", ); const ingestHelpers = join(utils, "ingest_helpers.py"); - writeFileSync( - ingestHelpers, - '_CONV_ID_LIBS = frozenset({"memos", "everos"})\n', - "utf8", - ); + writeFileSync(ingestHelpers, '_CONV_ID_LIBS = frozenset({"memos", "everos"})\n', "utf8"); const locomoSearch = join(locomo, "locomo_search.py"); writeFileSync( locomoSearch, @@ -41,10 +37,7 @@ test("OmniMemEval adapter installer patches the registry idempotently", () => { const source = readFileSync(registry, "utf8"); assert.equal(source.match(/"nmg": \("nmg_client", "NmgClient"\)/g)?.length, 1); assert.equal(existsSync(join(factory, "nmg_client.py")), true); - assert.match( - readFileSync(join(factory, "nmg_client.py"), "utf8"), - /ensure_ascii=True/, - ); + assert.match(readFileSync(join(factory, "nmg_client.py"), "utf8"), /ensure_ascii=True/); assert.equal( readFileSync(searchHelpers, "utf8").match(/"nmg": generic_text_search/g)?.length, 1, diff --git a/tests/evals/omnimemeval-judge-provider.test.ts b/tests/evals/omnimemeval-judge-provider.test.ts index 2efc3454..75f99ffd 100644 --- a/tests/evals/omnimemeval-judge-provider.test.ts +++ b/tests/evals/omnimemeval-judge-provider.test.ts @@ -1,10 +1,17 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { OpenAiCompatibleJudgeClient, createJudgeClientFromEnv } from "../../evals/omnimemeval/judge-provider.ts"; +import { + OpenAiCompatibleJudgeClient, + createJudgeClientFromEnv, +} from "../../evals/omnimemeval/judge-provider.ts"; import type { DuplicateCandidate } from "../../src/core/types.ts"; -function candidate(id: string, statement: string, eventTime = "2026-01-01T00:00:00Z"): DuplicateCandidate { +function candidate( + id: string, + statement: string, + eventTime = "2026-01-01T00:00:00Z", +): DuplicateCandidate { return { memoryId: id, nodeId: "n", statement, eventTime, similarity: 0.3 }; } @@ -36,13 +43,22 @@ test("judge: supersede decision parsed from model JSON", async () => { const body = JSON.parse(String(init?.body)); assert.equal(body.temperature, 0, "non-thinking mode uses temperature 0"); assert.ok(!("thinking" in body)); - return new Response(JSON.stringify({ - choices: [{ message: { content: JSON.stringify({ - action: "supersede", - supersededMemoryId: "stale-1", - reason: "newer value", - }) } }], - }), { status: 200, headers: { "content-type": "application/json" } }); + return new Response( + JSON.stringify({ + choices: [ + { + message: { + content: JSON.stringify({ + action: "supersede", + supersededMemoryId: "stale-1", + reason: "newer value", + }), + }, + }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); }) as typeof fetch, }); const r = await client.judge({ @@ -63,9 +79,14 @@ test("judge: thinking mode sends DeepSeek reasoning fields, no temperature", asy reasoningEffort: "high", fetch: (async (_url, init) => { sentBody = JSON.parse(String(init?.body)); - return new Response(JSON.stringify({ - choices: [{ message: { content: JSON.stringify({ action: "keep", reason: "distinct" }) } }], - }), { status: 200, headers: { "content-type": "application/json" } }); + return new Response( + JSON.stringify({ + choices: [ + { message: { content: JSON.stringify({ action: "keep", reason: "distinct" }) } }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); }) as typeof fetch, }); await client.judge({ @@ -83,9 +104,13 @@ test("judge: code-fenced / malformed model output degrades to keep", async () => const client = new OpenAiCompatibleJudgeClient({ baseUrl: "https://x.example", model: "m", - fetch: (async (_url, _init) => new Response(JSON.stringify({ - choices: [{ message: { content: "```json\n{\"action\":\"merge\",\"memoryId\":\"c1\"}\n```" } }], - }), { status: 200, headers: { "content-type": "application/json" } })) as typeof fetch, + fetch: (async (_url, _init) => + new Response( + JSON.stringify({ + choices: [{ message: { content: '```json\n{"action":"merge","memoryId":"c1"}\n```' } }], + }), + { status: 200, headers: { "content-type": "application/json" } }, + )) as typeof fetch, }); const r = await client.judge({ statement: "s", @@ -97,9 +122,13 @@ test("judge: code-fenced / malformed model output degrades to keep", async () => const bad = new OpenAiCompatibleJudgeClient({ baseUrl: "https://x.example", model: "m", - fetch: (async () => new Response(JSON.stringify({ - choices: [{ message: { content: "not json at all" } }], - }), { status: 200, headers: { "content-type": "application/json" } })) as typeof fetch, + fetch: (async () => + new Response( + JSON.stringify({ + choices: [{ message: { content: "not json at all" } }], + }), + { status: 200, headers: { "content-type": "application/json" } }, + )) as typeof fetch, }); const r2 = await bad.judge({ statement: "s", diff --git a/tests/evals/omnimemeval-merge-shards.test.ts b/tests/evals/omnimemeval-merge-shards.test.ts index 2c872a4d..afe60eb6 100644 --- a/tests/evals/omnimemeval-merge-shards.test.ts +++ b/tests/evals/omnimemeval-merge-shards.test.ts @@ -4,24 +4,21 @@ import test from "node:test"; import { mergeLongMemEvalShards } from "../../evals/omnimemeval/merge-longmemeval-shards.ts"; test("LongMemEval shard merge restores deterministic conversation order", () => { - const merged = mergeLongMemEvalShards([ - { "user_shard_b_1": [{ question: "one" }] }, - { "user_shard_a_0": [{ question: "zero" }] }, - ], 2); + const merged = mergeLongMemEvalShards( + [{ user_shard_b_1: [{ question: "one" }] }, { user_shard_a_0: [{ question: "zero" }] }], + 2, + ); assert.deepEqual(Object.keys(merged), ["user_shard_a_0", "user_shard_b_1"]); }); test("LongMemEval shard merge rejects duplicate and missing indices", () => { assert.throws( - () => mergeLongMemEvalShards([ - { "user_a_0": [] }, - { "user_b_0": [] }, - ], 1), + () => mergeLongMemEvalShards([{ user_a_0: [] }, { user_b_0: [] }], 1), /duplicate conversation index 0/, ); assert.throws( - () => mergeLongMemEvalShards([{ "user_a_1": [] }], 2), + () => mergeLongMemEvalShards([{ user_a_1: [] }], 2), /missing conversation index 0/, ); }); diff --git a/tests/evals/omnimemeval-runner.test.ts b/tests/evals/omnimemeval-runner.test.ts index 23c6709a..b98209f9 100644 --- a/tests/evals/omnimemeval-runner.test.ts +++ b/tests/evals/omnimemeval-runner.test.ts @@ -49,7 +49,11 @@ test("CLI exposes only suite, config, resume, and dry-run", () => { test("one config supplies common and suite-specific official arguments", () => { const suites: Record = { - beam: [], locomo: [], longmemeval: [], "personamem-v2": [], halumem: [], + beam: [], + locomo: [], + longmemeval: [], + "personamem-v2": [], + halumem: [], }; suites.beam = ["--scale", "100k", "--judge-batch-size", "4"]; const repoRoot = fixtureRepo("beam", { suites }); diff --git a/tests/evals/retrieval-score.test.ts b/tests/evals/retrieval-score.test.ts index 9609d110..6dc57e09 100644 --- a/tests/evals/retrieval-score.test.ts +++ b/tests/evals/retrieval-score.test.ts @@ -48,10 +48,7 @@ test("scoreQuestion matches candidate inside a gold session blob (candidate-in-g { category: "single-session-user", golds: ["user: I graduated with a Business Administration degree\nassistant: congrats"], - candidates: [ - ["unrelated memory"], - ["I graduated with a Business Administration degree"], - ], + candidates: [["unrelated memory"], ["I graduated with a Business Administration degree"]], }, "candidate-in-gold", ); diff --git a/tests/evals/retrieval-smoke.test.ts b/tests/evals/retrieval-smoke.test.ts index 3bbacfd5..d0d57f71 100644 --- a/tests/evals/retrieval-smoke.test.ts +++ b/tests/evals/retrieval-smoke.test.ts @@ -56,7 +56,7 @@ test("parsePythonLiteral parses BEAM-style probing dicts", () => { "{'fact_recall': [{'question': 'It\\'s about \"x\", cost 42', 'source_chat_ids': [3, 7], 'ok': True, 'missing': None}]}", ) as Record>>; const entry = parsed["fact_recall"]![0]!; - assert.equal(entry["question"], "It's about \"x\", cost 42"); + assert.equal(entry["question"], 'It\'s about "x", cost 42'); assert.deepEqual(entry["source_chat_ids"], [3, 7]); assert.equal(entry["ok"], true); assert.equal(entry["missing"], null); diff --git a/tests/evals/skillopt/dataset.test.ts b/tests/evals/skillopt/dataset.test.ts index 7ef047cb..3c71ec6e 100644 --- a/tests/evals/skillopt/dataset.test.ts +++ b/tests/evals/skillopt/dataset.test.ts @@ -33,7 +33,9 @@ test("SkillOpt policy dataset keeps whole tasks split and excludes memory conten noise_labels: 2, }); assert.deepEqual( - new Set(dataset.items.filter((item) => item.semantic_task_id === "task-a").map((item) => item.split)), + new Set( + dataset.items.filter((item) => item.semantic_task_id === "task-a").map((item) => item.split), + ), new Set(["train"]), ); assert.deepEqual( diff --git a/tests/evals/topology/namesakes.test.ts b/tests/evals/topology/namesakes.test.ts index 268caeee..e767b30e 100644 --- a/tests/evals/topology/namesakes.test.ts +++ b/tests/evals/topology/namesakes.test.ts @@ -70,7 +70,7 @@ function fixture(pageid: string): NamesakesEntity { text: mention, start, end: cursor, - tag: index === 3 ? "Other" as const : "Same" as const, + tag: index === 3 ? ("Other" as const) : ("Same" as const), }; }); return { diff --git a/tests/evals/topology/run.test.ts b/tests/evals/topology/run.test.ts index 53af9fc6..51a42807 100644 --- a/tests/evals/topology/run.test.ts +++ b/tests/evals/topology/run.test.ts @@ -31,12 +31,14 @@ test("topology audit accepts same-person fragments, rejects cross-person pairs, function fixture(): BenchmarkCase { const sessions = Array.from({ length: 4 }, (_, sessionIndex) => ({ id: `session_${sessionIndex + 1}`, - turns: ["Alex", "Blair"].flatMap((speaker) => [0, 1].map((turnIndex) => ({ - role: speaker === "Alex" ? "user" as const : "assistant" as const, - speaker, - content: `${speaker === "Alex" ? "amber hiking" : "cobalt cooking"} detail ${sessionIndex}-${turnIndex}`, - sourceId: `${speaker}-${sessionIndex}-${turnIndex}`, - }))), + turns: ["Alex", "Blair"].flatMap((speaker) => + [0, 1].map((turnIndex) => ({ + role: speaker === "Alex" ? ("user" as const) : ("assistant" as const), + speaker, + content: `${speaker === "Alex" ? "amber hiking" : "cobalt cooking"} detail ${sessionIndex}-${turnIndex}`, + sourceId: `${speaker}-${sessionIndex}-${turnIndex}`, + })), + ), })); return { id: "q-1", diff --git a/tests/extensions/pi-dependency-boundary.test.ts b/tests/extensions/pi-dependency-boundary.test.ts index f80ede05..013a9566 100644 --- a/tests/extensions/pi-dependency-boundary.test.ts +++ b/tests/extensions/pi-dependency-boundary.test.ts @@ -12,23 +12,17 @@ const packageJson = JSON.parse( }; test("Pi harness stays an optional adapter peer instead of a runtime dependency", () => { - assert.equal( - packageJson.dependencies?.["@earendil-works/pi-coding-agent"], - undefined, - ); + assert.equal(packageJson.dependencies?.["@earendil-works/pi-coding-agent"], undefined); assert.ok(packageJson.devDependencies?.["@earendil-works/pi-coding-agent"]); assert.ok(packageJson.peerDependencies?.["@earendil-works/pi-coding-agent"]); assert.equal( - packageJson.peerDependenciesMeta?.["@earendil-works/pi-coding-agent"] - ?.optional, + packageJson.peerDependenciesMeta?.["@earendil-works/pi-coding-agent"]?.optional, true, ); }); test("the standalone TUI uses the Pi harness-compatible pi-tui line", () => { - const harnessVersion = packageJson.devDependencies?.[ - "@earendil-works/pi-coding-agent" - ]; + const harnessVersion = packageJson.devDependencies?.["@earendil-works/pi-coding-agent"]; const tuiVersion = packageJson.dependencies?.["@earendil-works/pi-tui"]; assert.equal(harnessVersion, "^0.84.1"); diff --git a/tests/integration/agent-surface.test.ts b/tests/integration/agent-surface.test.ts index ed7d7575..f8b6e7c0 100644 --- a/tests/integration/agent-surface.test.ts +++ b/tests/integration/agent-surface.test.ts @@ -56,7 +56,11 @@ test("session Active Graph surface renders only projected temporary items", () = function context(): MemoryContext { const chainId = "chain-atlas"; - const result = (id: string, statement: string, position: number): MemoryContext["results"][number] => { + const result = ( + id: string, + statement: string, + position: number, + ): MemoryContext["results"][number] => { const fixture = searchResultFixture(id, statement); return { ...fixture, diff --git a/tests/integration/chain-projection.test.ts b/tests/integration/chain-projection.test.ts index b9ee6f40..335ed9ac 100644 --- a/tests/integration/chain-projection.test.ts +++ b/tests/integration/chain-projection.test.ts @@ -7,13 +7,12 @@ import { projectLogicalChains } from "../../src/integration/chain-projection.ts" function logicalChainContext(): MemoryContext { const chainId = "logical-merge"; - const result = (id: string, statement: string, position: number) => - ({ - ...searchResultFixture(id, statement), - chainMemberships: [ - { chainId, position, chainType: "logical" as const, topic: "Atlas merge evidence" }, - ], - }); + const result = (id: string, statement: string, position: number) => ({ + ...searchResultFixture(id, statement), + chainMemberships: [ + { chainId, position, chainType: "logical" as const, topic: "Atlas merge evidence" }, + ], + }); return { results: [ diff --git a/tests/integration/controller-channel.test.ts b/tests/integration/controller-channel.test.ts index f19a0996..71287fdb 100644 --- a/tests/integration/controller-channel.test.ts +++ b/tests/integration/controller-channel.test.ts @@ -76,10 +76,14 @@ test("active controller binds candidate, three gate artifacts, and rollback", () try { const receiptPath = join(fixture.directory, "activation.json"); const gatePaths = ["retrieval.json", "controller.json", "product.json"]; - for (const path of gatePaths) writeFileSync(join(fixture.directory, path), `{"gate":"${path}"}`); + for (const path of gatePaths) + writeFileSync(join(fixture.directory, path), `{"gate":"${path}"}`); const rollbackPath = join(fixture.directory, "rollback.json"); new ControllerRuntime(rollbackPath).save(); - const reference = (path: string) => ({ path, sha256: fingerprint(join(fixture.directory, path)) }); + const reference = (path: string) => ({ + path, + sha256: fingerprint(join(fixture.directory, path)), + }); const receipt: ControllerActivationReceipt = { version: 1, status: "approved", diff --git a/tests/integration/lab-capabilities.test.ts b/tests/integration/lab-capabilities.test.ts index bfa20422..8907193b 100644 --- a/tests/integration/lab-capabilities.test.ts +++ b/tests/integration/lab-capabilities.test.ts @@ -43,6 +43,9 @@ test("agent self-service cannot bypass controlled or active controller gates", ( test("Lab capability discovery distinguishes self-service and gated features", () => { const descriptors = new LabActivationAuthority().list(); - assert.equal(descriptors.find((item) => item.id === "memory_graph_reasoner")?.agentMayEnable, true); + assert.equal( + descriptors.find((item) => item.id === "memory_graph_reasoner")?.agentMayEnable, + true, + ); assert.equal(descriptors.find((item) => item.id === "controller_active")?.agentMayEnable, false); }); diff --git a/tests/integration/leaf-summarizer.test.ts b/tests/integration/leaf-summarizer.test.ts index cb35513b..4c3b9105 100644 --- a/tests/integration/leaf-summarizer.test.ts +++ b/tests/integration/leaf-summarizer.test.ts @@ -21,7 +21,10 @@ function captureFetch( if (typeof body === "object" && body !== null && !Array.isArray(body)) { Object.assign(body as Record, parsed); } - const outcome = impl?.(parsed) ?? { ok: true, payload: { choices: [{ message: { content: "summary text" } }] } }; + const outcome = impl?.(parsed) ?? { + ok: true, + payload: { choices: [{ message: { content: "summary text" } }] }, + }; return new Response(outcome.ok ? JSON.stringify(outcome.payload) : "boom", { status: outcome.ok ? 200 : 500, }); @@ -61,7 +64,10 @@ test("OpenAiLeafSummaryProvider: HTTP errors and empty content throw", async () const empty = new OpenAiLeafSummaryProvider({ baseUrl: "https://example.test", model: "m", - fetch: captureFetch(null, () => ({ ok: true, payload: { choices: [{ message: { content: "" } }] } })), + fetch: captureFetch(null, () => ({ + ok: true, + payload: { choices: [{ message: { content: "" } }] }, + })), }); await assert.rejects(() => empty.summarize({ nodeName: "n", statements: ["s"] })); }); diff --git a/tests/integration/ooo-acceptance-one-predicate.test.ts b/tests/integration/ooo-acceptance-one-predicate.test.ts index 517faa0a..910d583f 100644 --- a/tests/integration/ooo-acceptance-one-predicate.test.ts +++ b/tests/integration/ooo-acceptance-one-predicate.test.ts @@ -2,7 +2,12 @@ import assert from "node:assert/strict"; import { readFileSync, readdirSync, statSync } from "node:fs"; import { join } from "node:path"; import { test } from "node:test"; -import { acceptedFact, isAccepted, type TaskUnit, type RecordedFacts } from "../../src/integration/task-semantics.ts"; +import { + acceptedFact, + isAccepted, + type TaskUnit, + type RecordedFacts, +} from "../../src/integration/task-semantics.ts"; const root = new URL("../../", import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1"); const read = (relative: string) => readFileSync(join(root, relative), "utf8"); diff --git a/tests/integration/ooo-fusion-plan.test.ts b/tests/integration/ooo-fusion-plan.test.ts index 86589f87..7b702859 100644 --- a/tests/integration/ooo-fusion-plan.test.ts +++ b/tests/integration/ooo-fusion-plan.test.ts @@ -183,7 +183,10 @@ test("the continuation is a declared constraint, not the planner's default", () const silent: SessionPlan = { ...declared, constraints: [] }; // The same plan and the same facts, and the only difference is which constraints the plan enabled. - assert.deepEqual(nextSessionMove({ plan: declared, ...boundary }), { kind: "admit", unit: "two" }); + assert.deepEqual(nextSessionMove({ plan: declared, ...boundary }), { + kind: "admit", + unit: "two", + }); assert.deepEqual(nextSessionMove({ plan: silent, ...boundary }), { kind: "close", reason: "repair-first is not enabled, so this plan runs one unit per session", diff --git a/tests/integration/ooo-post-commit-notification.test.ts b/tests/integration/ooo-post-commit-notification.test.ts index 2a08ca25..2632504c 100644 --- a/tests/integration/ooo-post-commit-notification.test.ts +++ b/tests/integration/ooo-post-commit-notification.test.ts @@ -11,7 +11,11 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { BoardAdmission, type PatchTaskSpec, type ProbePlan } from "../../src/integration/ooo-board.ts"; +import { + BoardAdmission, + type PatchTaskSpec, + type ProbePlan, +} from "../../src/integration/ooo-board.ts"; import { preparePatchWork } from "../../src/integration/ooo-patch.ts"; const TASK = "A"; diff --git a/tests/integration/ooo-publication-invariants.test.ts b/tests/integration/ooo-publication-invariants.test.ts index 0be62bbd..b64af297 100644 --- a/tests/integration/ooo-publication-invariants.test.ts +++ b/tests/integration/ooo-publication-invariants.test.ts @@ -71,9 +71,7 @@ test("no publication over the design's interleavings is unsupported by its own f "every publication is supported by the recorded facts", ); assert.deepEqual( - report.budgetFindings.map( - (finding) => `${finding.property}:${finding.budget}:${finding.unit}`, - ), + report.budgetFindings.map((finding) => `${finding.property}:${finding.budget}:${finding.unit}`), [], "every declared budget published a view its own budget allows", ); @@ -93,7 +91,11 @@ test("a declared budget publishes more than one slot can, and never a claimed ta ]; const report = enumerateInterleavings({ plan: wide, specs: { P: spec() }, scripts }); assert.equal(report.refused, undefined); - assert.deepEqual(report.budgetFindings, [], "no budget offered a claimed task or dropped a candidate"); + assert.deepEqual( + report.budgetFindings, + [], + "no budget offered a claimed task or dropped a candidate", + ); assert.ok( report.widened > 0, "the two-slot view published a task the one-slot view did not, which is what the budget buys", @@ -128,7 +130,12 @@ test("the budget properties fire on a hand-built view, so deleting them cannot p "a bigger budget adds candidates, it does not replace them", ); assert.deepEqual(checkBudget({ budget: 2, ready: ["Q"], claimed: ["P"], smallerReady: [] }), []); - const refused = enumerateInterleavings({ plan: DESIGN_PLAN, specs: { P: spec() }, scripts: [], budgets: [0] }); + const refused = enumerateInterleavings({ + plan: DESIGN_PLAN, + specs: { P: spec() }, + scripts: [], + budgets: [0], + }); assert.match(refused.refused!, /budget 0 is not a positive integer/); }); diff --git a/tests/integration/ooo-session-chain-contract.test.ts b/tests/integration/ooo-session-chain-contract.test.ts index f0e0c133..bf09dcc1 100644 --- a/tests/integration/ooo-session-chain-contract.test.ts +++ b/tests/integration/ooo-session-chain-contract.test.ts @@ -65,7 +65,11 @@ test("a chain input names the check when the unit has one", () => { const frozen = patchWork(); const input = patchSessionInput(frozen, { looseConclusion: true, - check: { label: "the unit check", maxRuns: 1, run: async () => ({ verdict: "accept", log: "" }) }, + check: { + label: "the unit check", + maxRuns: 1, + run: async () => ({ verdict: "accept", log: "" }), + }, }); assert.match(input.prompt, /the check the unit check/); assert.match(input.prompt, /call only read_snapshot, run_check and submit_artifact/); diff --git a/tests/skills/nmg-memory.test.ts b/tests/skills/nmg-memory.test.ts index d872e9e0..b93b85ad 100644 --- a/tests/skills/nmg-memory.test.ts +++ b/tests/skills/nmg-memory.test.ts @@ -31,14 +31,19 @@ test("NMG Skill natural evidence loop separates observation, calibration, and ac assert.match(naturalEvidence, /NMG_SHADOW_COLLECTION_ORIGIN/u); assert.match(naturalEvidence, /eval:natural-readiness -- --project-dir /u); assert.match(naturalEvidence, /eval:controller-dataset -- --compact/u); - assert.match(naturalEvidence, /writes a rollbackable candidate artifact; it does not activate it/u); + assert.match( + naturalEvidence, + /writes a rollbackable candidate artifact; it does not activate it/u, + ); assert.match(naturalEvidence, /must keep the corresponding production actuator disabled/u); }); test("NMG Skill eval definitions have a stable executable-harness schema", () => { - const cases = JSON.parse( - readFileSync(resolve(skillRoot, "evals/evals.json"), "utf8"), - ) as Array<{ name?: unknown; prompt?: unknown; expected?: unknown }>; + const cases = JSON.parse(readFileSync(resolve(skillRoot, "evals/evals.json"), "utf8")) as Array<{ + name?: unknown; + prompt?: unknown; + expected?: unknown; + }>; assert.ok(cases.length > 0); assert.equal(new Set(cases.map((entry) => entry.name)).size, cases.length); diff --git a/tests/skills/nmg-skill-sync.test.ts b/tests/skills/nmg-skill-sync.test.ts index 593d783f..5091c1a2 100644 --- a/tests/skills/nmg-skill-sync.test.ts +++ b/tests/skills/nmg-skill-sync.test.ts @@ -101,11 +101,9 @@ test("NMG Skill check exits one for drift and CLI options fail closed", (context assert.notEqual(missingValue.status, 0); assert.match(missingValue.stderr, /--target requires a value/u); - const unknown = spawnSync( - process.execPath, - ["--experimental-strip-types", script, "--unknown"], - { encoding: "utf8" }, - ); + const unknown = spawnSync(process.execPath, ["--experimental-strip-types", script, "--unknown"], { + encoding: "utf8", + }); assert.notEqual(unknown.status, 0); assert.match(unknown.stderr, /unknown option/u); diff --git a/tests/support/cordis-adapter.test.ts b/tests/support/cordis-adapter.test.ts index 86c3862e..1ffce348 100644 --- a/tests/support/cordis-adapter.test.ts +++ b/tests/support/cordis-adapter.test.ts @@ -22,10 +22,20 @@ test("Cordis adapter disposes registered effects in reverse order", async () => const runtime = adapter.createTestRuntime(); await runtime.use((scope) => { - scope.effect(() => () => { events.push("first"); }, "first"); + scope.effect( + () => () => { + events.push("first"); + }, + "first", + ); }); await runtime.use((scope) => { - scope.effect(() => () => { events.push("second"); }, "second"); + scope.effect( + () => () => { + events.push("second"); + }, + "second", + ); }); await runtime.dispose(); diff --git a/tests/tools/agent-verify.test.ts b/tests/tools/agent-verify.test.ts index bc4a7fd4..31850efe 100644 --- a/tests/tools/agent-verify.test.ts +++ b/tests/tools/agent-verify.test.ts @@ -888,7 +888,10 @@ test("a failing route test fails the narrow gate instead of passing vacuously", const receipt = JSON.parse(readFileSync(payload.rcp!.receiptPath!, "utf8")) as { checks: Array<{ name: string; status: string }>; }; - assert.deepEqual(receipt.checks.map((check) => check.name), ["node-test:plugin"]); + assert.deepEqual( + receipt.checks.map((check) => check.name), + ["node-test:plugin"], + ); assert.equal(receipt.checks.find((check) => check.name === "node-test:plugin")?.status, "failed"); }); diff --git a/tests/tools/ci-status-snapshot.test.ts b/tests/tools/ci-status-snapshot.test.ts index 6b523aa2..0646fd14 100644 --- a/tests/tools/ci-status-snapshot.test.ts +++ b/tests/tools/ci-status-snapshot.test.ts @@ -2,10 +2,7 @@ import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import test from "node:test"; -import { - buildCiStatusSnapshot, - renderCiStatusSummary, -} from "../../tools/ci-status-snapshot.ts"; +import { buildCiStatusSnapshot, renderCiStatusSummary } from "../../tools/ci-status-snapshot.ts"; test("CI status snapshot preserves GitHub identity and reports failed steps", () => { const snapshot = buildCiStatusSnapshot( @@ -40,7 +37,12 @@ test("CI status snapshot preserves GitHub identity and reports failed steps", () completed_at: "2026-08-31T00:01:00Z", steps: [ { name: "npm ci", status: "completed", conclusion: "success", number: 1 }, - { name: "Shared static verification contract", status: "completed", conclusion: "failure", number: 2 }, + { + name: "Shared static verification contract", + status: "completed", + conclusion: "failure", + number: 2, + }, ], }, { diff --git a/tests/tools/recall-instance.test.ts b/tests/tools/recall-instance.test.ts index 9bb6b504..b262329b 100644 --- a/tests/tools/recall-instance.test.ts +++ b/tests/tools/recall-instance.test.ts @@ -150,13 +150,24 @@ test("instancesSurfacing finds recalls whose retrieval surfaced a memory", () => ], }; const matches = instancesSurfacing([withM2, instance("g2")], "m2"); - assert.deepEqual(matches.map((match) => match.activeGraphId), ["g1"]); + assert.deepEqual( + matches.map((match) => match.activeGraphId), + ["g1"], + ); }); test("summarizeLabels counts remember-settled labels like any label", () => { - const labeled = applyRecallLabels([instance("g1")], [ - { activeGraphId: "g1", label: "on_target", source: "remember", at: "2026-09-07T00:00:00.000Z" }, - ]); + const labeled = applyRecallLabels( + [instance("g1")], + [ + { + activeGraphId: "g1", + label: "on_target", + source: "remember", + at: "2026-09-07T00:00:00.000Z", + }, + ], + ); const summary = summarizeLabels(labeled); assert.equal(summary.labeled, 1); assert.equal(summary.precision, 1); @@ -168,7 +179,10 @@ test("recordLabel writes a valid agent judgement and rejects an invalid one", () assert.equal(recordLabel(dir, "g1", "noise"), true); assert.equal(recordLabel(dir, "g1", "bogus"), false); const entries = readRecallLabels(recallLabelsPath(dir)); - assert.deepEqual(entries.map((entry) => entry.label), ["noise"]); + assert.deepEqual( + entries.map((entry) => entry.label), + ["noise"], + ); assert.equal(entries[0]?.source, "judge"); } finally { rmSync(dir, { recursive: true, force: true }); diff --git a/tests/tools/recall-probe.test.ts b/tests/tools/recall-probe.test.ts index 35a777ee..584b3913 100644 --- a/tests/tools/recall-probe.test.ts +++ b/tests/tools/recall-probe.test.ts @@ -34,7 +34,12 @@ test("executeProbeRow labels the primary query and each variant", async () => { test("summarizeProbe buckets gaps, robust and fragile groups", async () => { const rows: RecallProbeRow[] = [ - { groupId: "weather", query: "how do i check weather", expectMemoryId: "m1", variants: ["wttr"] }, + { + groupId: "weather", + query: "how do i check weather", + expectMemoryId: "m1", + variants: ["wttr"], + }, { groupId: "absent", query: "unused thing", expectMemoryId: "m2", variants: [] }, { groupId: "fragile", query: "recall me", expectMemoryId: "m3", variants: ["near alias"] }, ]; diff --git a/tests/tools/test-groups.test.ts b/tests/tools/test-groups.test.ts index b9086bc9..3aef99f8 100644 --- a/tests/tools/test-groups.test.ts +++ b/tests/tools/test-groups.test.ts @@ -25,6 +25,16 @@ test("research and chaos suites remain explicit execution groups", () => { assert.match(packageJson.scripts["test:chaos"], /tests\/chaos/); }); +test("format and format:check cover the same declared developer TypeScript surfaces", () => { + const expected = ["src", ".pi", "workbuddy-plugin", "tests", "evals", "scripts", "tools"].sort(); + for (const name of ["format", "format:check"]) { + const surfaces = [...packageJson.scripts[name]!.matchAll(/"([^"]+)\/\*\*\/\*\.ts"/gu)] + .map((match) => match[1]!) + .sort(); + assert.deepEqual(surfaces, expected, `${name} must cover the declared surface`); + } +}); + test("the pre-commit formatter surface is the format:check surface", () => { // Two surfaces decide which TypeScript Prettier rewrites: the commit hook (staged // files, so drift cannot land) and `format:check` (the whole tree, so CI can report diff --git a/tools/check-lock.ts b/tools/check-lock.ts index 739e851e..f5fabb5b 100644 --- a/tools/check-lock.ts +++ b/tools/check-lock.ts @@ -19,7 +19,15 @@ const manifest = JSON.parse(readFileSync(resolve(root, "package.json"), "utf8")) peerDependencies?: Record; }; const lock = JSON.parse(readFileSync(resolve(root, "package-lock.json"), "utf8")) as { - packages?: Record; devDependencies?: Record; optionalDependencies?: Record; peerDependencies?: Record }>; + packages?: Record< + string, + { + dependencies?: Record; + devDependencies?: Record; + optionalDependencies?: Record; + peerDependencies?: Record; + } + >; }; const lockRoot = lock.packages?.[""]; @@ -41,15 +49,22 @@ if (!lockRoot) { }; const drift: string[] = []; for (const [name, spec] of Object.entries(expected)) { - if (locked[name] !== spec) drift.push(`${name}: package.json "${spec}" != lock "${locked[name] ?? "(missing)"}"`); + if (locked[name] !== spec) + drift.push(`${name}: package.json "${spec}" != lock "${locked[name] ?? "(missing)"}"`); } for (const name of Object.keys(locked)) { if (!(name in expected)) drift.push(`${name}: present in lock but not in package.json`); } if (drift.length > 0) { - process.stderr.write("check:lock — package-lock.json is stale; run `npm install --package-lock-only`:\n" + drift.map((line) => ` - ${line}`).join("\n") + "\n"); + process.stderr.write( + "check:lock — package-lock.json is stale; run `npm install --package-lock-only`:\n" + + drift.map((line) => ` - ${line}`).join("\n") + + "\n", + ); process.exitCode = 1; } else { - process.stdout.write(`check:lock ok: ${Object.keys(expected).length} root dependency specifiers match package-lock.json\n`); + process.stdout.write( + `check:lock ok: ${Object.keys(expected).length} root dependency specifiers match package-lock.json\n`, + ); } } diff --git a/tools/ci-status-snapshot.ts b/tools/ci-status-snapshot.ts index 38100428..919100b8 100644 --- a/tools/ci-status-snapshot.ts +++ b/tools/ci-status-snapshot.ts @@ -116,7 +116,9 @@ export function buildCiStatusSnapshot( })) .sort((left, right) => left.name.localeCompare(right.name) || left.id - right.id); - const pullRequests = [...new Set((run.pull_requests ?? []).map((pull) => number(pull.number)).filter(Boolean))] + const pullRequests = [ + ...new Set((run.pull_requests ?? []).map((pull) => number(pull.number)).filter(Boolean)), + ] .sort((left, right) => left - right) .map((pullNumber) => ({ number: pullNumber })); @@ -180,7 +182,11 @@ export function renderCiStatusSummary(snapshot: CiStatusSnapshot): string { return lines.join("\n"); } -async function fetchWorkflowJobs(repository: string, runId: number, token: string): Promise { +async function fetchWorkflowJobs( + repository: string, + runId: number, + token: string, +): Promise { const api = process.env.GITHUB_API_URL ?? "https://api.github.com"; const jobs: WorkflowJob[] = []; for (let page = 1; ; page += 1) { @@ -228,7 +234,8 @@ async function main(): Promise { if (summaryPath) await appendFile(summaryPath, renderCiStatusSummary(snapshot), "utf8"); } -const isEntrypoint = process.argv[1] && resolve(process.argv[1]) === resolve(fileURLToPath(import.meta.url)); +const isEntrypoint = + process.argv[1] && resolve(process.argv[1]) === resolve(fileURLToPath(import.meta.url)); if (isEntrypoint) { main().catch((error: unknown) => { process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); diff --git a/tools/recall-instance-judge.ts b/tools/recall-instance-judge.ts index 46543d5d..cbf4357b 100644 --- a/tools/recall-instance-judge.ts +++ b/tools/recall-instance-judge.ts @@ -190,9 +190,7 @@ async function labelBatch( function applyExplicitLabels(directory: string, specs: string[]): number { const known = new Set( - readRecallInstances(recallInstancesPath(directory)).map( - (instance) => instance.activeGraphId, - ), + readRecallInstances(recallInstancesPath(directory)).map((instance) => instance.activeGraphId), ); let written = 0; for (const spec of specs) { @@ -246,7 +244,9 @@ async function main(argv: string[]): Promise { } else { process.stdout.write(aggregate(final)); process.stdout.write( - hadModel ? `\nnewlyLabeled=${newlyLabeled}\n` : "\n(set NMG_JUDGE_BASE_URL + NMG_JUDGE_MODEL to semantically label)\n", + hadModel + ? `\nnewlyLabeled=${newlyLabeled}\n` + : "\n(set NMG_JUDGE_BASE_URL + NMG_JUDGE_MODEL to semantically label)\n", ); } return 0; @@ -254,7 +254,7 @@ async function main(argv: string[]): Promise { if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { main(process.argv.slice(2)).then( - (code) => process.exitCode = code, + (code) => (process.exitCode = code), (error) => { process.stderr.write(`${error}\n`); process.exitCode = 1; diff --git a/tools/recall-probe.ts b/tools/recall-probe.ts index 0a437fd8..8bde9737 100644 --- a/tools/recall-probe.ts +++ b/tools/recall-probe.ts @@ -4,11 +4,7 @@ import { resolve } from "node:path"; import { NmgStore } from "../src/core/store.ts"; import { searchMemoryContext } from "../src/integration/search.ts"; -import { - executeProbeRow, - summarizeProbe, - type RecallProbeRow, -} from "../src/lab/recall-probe.ts"; +import { executeProbeRow, summarizeProbe, type RecallProbeRow } from "../src/lab/recall-probe.ts"; /** * Deterministic controlled recall probe over a store snapshot. @@ -72,7 +68,9 @@ async function main(argv: string[]): Promise { `robust=[${summary.robust.join(",")}] fragile=${summary.fragile.length}\n`, ); for (const fragile of summary.fragile) { - process.stdout.write(` fragile ${fragile.groupId} under: ${fragile.failedVariants.join(" | ")}\n`); + process.stdout.write( + ` fragile ${fragile.groupId} under: ${fragile.failedVariants.join(" | ")}\n`, + ); } } return 0; diff --git a/tools/relevance-gate-calibration.ts b/tools/relevance-gate-calibration.ts index 45afaf26..80985f20 100644 --- a/tools/relevance-gate-calibration.ts +++ b/tools/relevance-gate-calibration.ts @@ -78,7 +78,9 @@ async function collect(options: Options): Promise { const spec = loadDataset(options.dataset, {}); const root = resolve(options.storeRoot, options.dataset); if (!existsSync(root)) { - throw new Error(`no ingested store at ${root}; run \`npm run eval:retrieval -- --dataset ${options.dataset}\``); + throw new Error( + `no ingested store at ${root}; run \`npm run eval:retrieval -- --dataset ${options.dataset}\``, + ); } const stores = new Map(); const getStore = (userId: string): NmgStore | null => { @@ -106,7 +108,9 @@ async function collect(options: Options): Promise { const candidates = context.results.map((result) => ({ scores: result, text: normalizeText( - [result.memory.statement, ...(result.evidenceRecords ?? []).map((e) => e.content)].join(" "), + [result.memory.statement, ...(result.evidenceRecords ?? []).map((e) => e.content)].join( + " ", + ), ), })); if (candidates.length === 0) continue; @@ -118,7 +122,11 @@ async function collect(options: Options): Promise { : gold.includes(candidate.text), ), ); - const stats = queryScoreStats(candidates.map((candidate) => candidate.scores), undefined, "raw"); + const stats = queryScoreStats( + candidates.map((candidate) => candidate.scores), + undefined, + "raw", + ); samples.push({ hit, cv: stats.cv, top1: stats.top1 }); } for (const store of stores.values()) store.close(); diff --git a/tools/rtm-check.ts b/tools/rtm-check.ts index 3105f796..2a81b3f4 100644 --- a/tools/rtm-check.ts +++ b/tools/rtm-check.ts @@ -421,7 +421,9 @@ interface ScanInput { /** Reads every contract file, records each assertion's standing, and returns what the final pass * needs to report the claims that rest on nothing. */ -function scanContracts(input: ScanInput): Pick { +function scanContracts( + input: ScanInput, +): Pick { const scan: AssertionScan = { report: input.report, resolvable: createResolver(input.root, input.routes, input.scripts), @@ -497,8 +499,7 @@ function openCounterexamples( for (const item of report.items) { if (item.standing === "uncovered") open.push(`uncovered: ${item.contract}:${item.id} -> ${item.check}`); - if (item.standing === "documented-only") - open.push(`declared gap: ${item.contract}:${item.id}`); + if (item.standing === "documented-only") open.push(`declared gap: ${item.contract}:${item.id}`); if (item.standing === "not-recorded") open.push(`no execution evidence: ${item.contract}:${item.id}`); } @@ -551,7 +552,11 @@ export function checkRtm(rootDirectory = process.cwd()): RtmReport { report.unresolvedAssumptions = [...scan.unresolved].sort(); report.orphans = [...scan.declared].filter((check) => !scan.referenced.has(check)).sort(); appendExecutionErrors(report); - report.counterexamples = openCounterexamples(report, report.orphans, report.unresolvedAssumptions); + report.counterexamples = openCounterexamples( + report, + report.orphans, + report.unresolvedAssumptions, + ); report.riskClasses = groupRiskClasses(report.items); return report; }