diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0375f23b..c3f94475 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -49,9 +49,9 @@ jobs: run: cargo test # `cargo build --all-targets` only compiles an example; AGENTS.md - # promises `cargo run -p tinymemory --example basic` works. + # promises `cargo run -p tinymemory-integrations --example basic` works. - name: Run the bundled example - run: cargo run -p tinymemory --example basic + run: cargo run -p tinymemory-integrations --example basic # The contract is what engines and hosts compile against. It must stay # free of storage engines, native libraries, HTTP clients and async @@ -129,24 +129,39 @@ jobs: fail-fast: false matrix: include: - - name: no default features + - name: integrations, no default features + package: tinymemory-integrations features: --no-default-features + - name: cortex + package: tinymemory-integrations + features: --no-default-features --features cortex - name: documents + package: tinymemory-integrations features: --no-default-features --features documents - name: documents-office + package: tinymemory-integrations features: --no-default-features --features documents-office - name: sources network implication + package: tinymemory-integrations features: --no-default-features --features sources-network - name: safety + package: tinymemory-integrations features: --no-default-features --features safety - - name: context - features: --no-default-features --features context - name: legacy import + package: tinymemory-integrations features: --no-default-features --features legacy-import - - name: conformance - features: --no-default-features --features conformance - name: full aggregate + package: tinymemory-integrations features: --no-default-features --features full + - name: api, no features + package: tinymemory-api + features: --no-default-features + - name: api conformance + package: tinymemory-api + features: --features conformance + - name: tools + package: tinymemory-tools + features: "" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -157,7 +172,7 @@ jobs: - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2 - name: Test - run: cargo test -p tinymemory ${{ matrix.features }} + run: cargo test -p ${{ matrix.package }} ${{ matrix.features }} docs: name: Docs @@ -189,7 +204,7 @@ jobs: run: | set -euo pipefail msrv="$(cargo metadata --format-version 1 --no-deps \ - | jq -r '.packages[] | select(.name == "tinymemory") | .rust_version')" + | jq -r '.packages[] | select(.name == "tinymemory-api") | .rust_version')" if [[ -z "$msrv" || "$msrv" == "null" ]]; then echo "package.rust-version is not set in Cargo.toml" >&2 exit 1 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 88e23b7a..822f6bb6 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -60,10 +60,10 @@ jobs: run: | set -euo pipefail - # The root manifest is a virtual workspace, so name the facade - # rather than taking whichever package cargo lists first. + # Every crate inherits `version` from `[workspace.package]`, so any + # one of them names the current version; read the contract crate's. metadata="$(cargo metadata --format-version 1 --no-deps)" - crate_name="tinymemory" + crate_name="tinymemory-api" current_version="$( jq -r --arg name "$crate_name" \ '.packages[] | select(.name == $name) | .version' <<< "$metadata" @@ -96,9 +96,10 @@ jobs: echo "tag=${tag}" } >> "$GITHUB_OUTPUT" - # Bumps the facade's `[package]` version only. That is sufficient because - # no intra-workspace path dependency carries a `version = "…"` - # requirement; the guard below keeps it that way. + # Bumps `[workspace.package] version` in the root manifest, which every + # crate inherits. That is sufficient because no intra-workspace path + # dependency carries a `version = "…"` requirement; the guard below + # keeps it that way. - name: Update crate version env: CRATE_NAME: ${{ steps.version.outputs.crate_name }} @@ -117,9 +118,15 @@ jobs: exit 1 fi - perl -0pi -e 's/(\[package\][\s\S]*?\nversion = ")[^"]+(")/$1$ENV{NEXT_VERSION}$2/' \ - crates/tinymemory/Cargo.toml - cargo update -p "$CRATE_NAME" --precise "$NEXT_VERSION" + perl -0pi -e 's/(\[workspace\.package\][\s\S]*?\nversion = ")[^"]+(")/$1$ENV{NEXT_VERSION}$2/' \ + Cargo.toml + cargo update --workspace + current="$(cargo metadata --format-version 1 --no-deps \ + | jq -r --arg name "$CRATE_NAME" '.packages[] | select(.name == $name) | .version')" + if [[ "$current" != "$NEXT_VERSION" ]]; then + echo "Version bump did not take: $CRATE_NAME is $current, expected $NEXT_VERSION" >&2 + exit 1 + fi - name: Commit version bump and tag env: @@ -128,7 +135,7 @@ jobs: set -euo pipefail git config user.name "github-actions[bot]" git config user.email "41898282+github-actions[bot]@users.noreply.github.com" - git add crates/tinymemory/Cargo.toml Cargo.lock + git add Cargo.toml Cargo.lock git commit -m "Release ${RELEASE_TAG}" git tag -a "${RELEASE_TAG}" -m "Release ${RELEASE_TAG}" diff --git a/AGENTS.md b/AGENTS.md index 2fcca3e9..21b15a1f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -11,18 +11,27 @@ no longer applies rather than leaving it to rot. This is a Cargo **workspace** with a virtual root: there is no root package, and every crate lives in its own directory under `crates/`, named for the -package it holds. `members` is the glob `crates/*`, so a new crate joins the -workspace by existing. `crates/tinymemory` is the facade a host depends on; -`crates/tinymemory-api` is the contract; `crates/tinymemory-cortex` is the -CortexDB engine; the rest are subsystems, each reachable from the facade by a -feature named after it. The accepted behaviour is -[`docs/specs/memory-v2.md`](docs/specs/memory-v2.md). +package it holds. There are exactly three, one per part of the memory layer: + +- `crates/tinymemory-api` — the **core contract**: `MemoryEngine`, items, + metadata, namespaces, errors, and (feature `conformance`) the suite every + engine must pass. No I/O. +- `crates/tinymemory-tools` — the **agent tool spec**: `MemoryTools` over any + engine, and the `context.md` compiler. +- `crates/tinymemory-integrations` — the **integrations**: the CortexDB engine + and its registry, documents, sources, safety and the legacy v1 import, each + a module behind a feature. + +Shared package metadata, the version and the lint table live in the root +`Cargo.toml` (`[workspace.package]`, `[workspace.lints]`). The accepted +behaviour is [`docs/specs/memory-v2.md`](docs/specs/memory-v2.md); how it is +built is in [`docs/architecture/`](docs/architecture/README.md). See [`README.md`](README.md) for the full layout and the feature table. ```text crates// -├── Cargo.toml # one package; `[lints]` opted into per crate +├── Cargo.toml # one package; `[lints] workspace = true` ├── README.md # required of complex crates: design, surface, caveats └── src/ ├── lib.rs # crate docs + the entire public re-export surface @@ -39,10 +48,11 @@ docs/ └── adr/ # immutable architecture decision records ``` -A new crate goes in `crates//`, and a package that is not an engine -or a subsystem of the memory layer probably does not belong here at all. Reach -it from the facade by adding an optional dependency and a feature of the same -name, so a host keeps taking one dependency and stating what it wants. +Do not add a fourth crate. New contract surface goes in `tinymemory-api`, a new +agent-facing tool in `tinymemory-tools`, and anything that talks to the outside +world (a new engine, reader or converter) becomes a module of +`tinymemory-integrations` behind a feature named after it, with its +dependencies optional and enabled only by that feature. Each feature area belongs in a focused module directory under the crate's `src/`. A module root explains the module, wires its pieces together, and @@ -83,7 +93,7 @@ Supporting commands: - `cargo fmt --all` — format before committing. - `cargo test ` — run a focused subset while iterating. -- `cargo run -p tinymemory --example basic` — run the bundled example. The +- `cargo run -p tinymemory-integrations --example basic` — run the bundled example. The `-p` is required: the workspace root is virtual, so cargo cannot infer which package an example belongs to. - `cargo doc --no-deps --all-features` — build the rustdoc CI also builds with @@ -107,8 +117,8 @@ Use standard `rustfmt` output and Rust 2024 idioms. Do not hand-format around `impl Into` at boundaries; return owned, concrete types. - Keep the public surface minimal: default to private, and export deliberately from the crate's `src/lib.rs`. -- `unsafe` is forbidden crate-wide by the `[lints]` table in each crate's own - `Cargo.toml` — the root is virtual and carries no lint configuration. If a +- `unsafe` is forbidden in every crate by `[workspace.lints]` in the root + `Cargo.toml`, which each crate inherits with `[lints] workspace = true`. If a crate genuinely needs it, relax the lint in its own commit and document every invariant with a `// SAFETY:` comment. @@ -222,7 +232,8 @@ explicitly declined with a reason. Releases run from `.github/workflows/release.yml` via a manual `workflow_dispatch` with a `patch` / `minor` / `major` bump. The workflow re-runs formatting, clippy, tests, and rustdoc, computes the next version, -updates `crates/tinymemory/Cargo.toml` and `Cargo.lock`, commits and tags +updates `[workspace.package] version` in the root `Cargo.toml` and +`Cargo.lock`, commits and tags `vX.Y.Z`, pushes both, and creates a GitHub release for the tag. There are no binary artifacts. @@ -231,7 +242,7 @@ take this repository by git or path, pinned to a tag. Consequently: -- Do not hand-edit the `version` field in `crates/tinymemory/Cargo.toml`; the +- Do not hand-edit `[workspace.package] version` in the root `Cargo.toml`; the release workflow owns it. - Follow semantic versioning. Any change to the public surface that is not purely additive is a breaking change and needs a major bump. diff --git a/Cargo.lock b/Cargo.lock index 54d037c9..66564ce3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1592,29 +1592,9 @@ dependencies = [ "syn 3.0.3", ] -[[package]] -name = "tinymemory" -version = "1.22.4" -dependencies = [ - "async-trait", - "serde", - "serde_json", - "tinymemory-api", - "tinymemory-conformance", - "tinymemory-context", - "tinymemory-cortex", - "tinymemory-documents", - "tinymemory-import", - "tinymemory-safety", - "tinymemory-sources", - "tokio", - "toml", - "zip", -] - [[package]] name = "tinymemory-api" -version = "2.0.0" +version = "1.22.4" dependencies = [ "async-trait", "chrono", @@ -1626,106 +1606,47 @@ dependencies = [ ] [[package]] -name = "tinymemory-conformance" -version = "2.0.0" -dependencies = [ - "async-trait", - "thiserror", - "tinymemory-api", - "tokio", -] - -[[package]] -name = "tinymemory-context" -version = "2.0.0" -dependencies = [ - "async-trait", - "chrono", - "log", - "serde", - "thiserror", - "tinymemory-api", - "tinymemory-conformance", - "tokio", -] - -[[package]] -name = "tinymemory-cortex" -version = "2.0.0" +name = "tinymemory-integrations" +version = "1.22.4" dependencies = [ "async-trait", "axum", - "futures", - "reqwest", - "serde", - "serde_json", - "sha2", - "tinymemory-api", - "tinymemory-conformance", - "tinymemory-context", - "tokio", -] - -[[package]] -name = "tinymemory-documents" -version = "0.1.0" -dependencies = [ - "async-trait", "calamine", + "chrono", + "futures", + "log", "pdf-extract", "quick-xml", - "serde", - "serde_json", - "thiserror", - "tinymemory-api", - "tokio", - "zip", -] - -[[package]] -name = "tinymemory-import" -version = "0.1.0" -dependencies = [ + "regex", + "reqwest", "rusqlite", + "schemars", "serde", "serde_json", + "sha2", "tempfile", "thiserror", "tinymemory-api", + "tinymemory-tools", + "tokio", + "toml", + "tracing", + "walkdir", + "zip", ] [[package]] -name = "tinymemory-safety" -version = "0.1.0" -dependencies = [ - "log", - "regex", - "serde_json", - "tinymemory-api", -] - -[[package]] -name = "tinymemory-sources" -version = "0.1.0" +name = "tinymemory-tools" +version = "1.22.4" dependencies = [ "async-trait", "chrono", - "futures", "log", - "regex", - "reqwest", - "schemars", "serde", "serde_json", - "tempfile", "thiserror", "tinymemory-api", - "tinymemory-documents", "tokio", - "toml", - "tracing", - "uuid", - "walkdir", ] [[package]] @@ -2000,17 +1921,6 @@ version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" -[[package]] -name = "uuid" -version = "1.24.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2cefc03fd367c0c6d4305de1b312cf00248c4114f4a0418ce6a6af769e3b0bd9" -dependencies = [ - "getrandom 0.4.3", - "js-sys", - "wasm-bindgen", -] - [[package]] name = "vcpkg" version = "0.2.15" diff --git a/Cargo.toml b/Cargo.toml index b994a744..8cccb10a 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,16 +1,54 @@ [workspace] resolver = "2" -# Every crate in this repository lives under `crates/`, one directory per -# package, each directory named for the package it holds. There is no root -# package: the facade a host depends on is `crates/tinymemory`, the same as any -# other member. Every member is also a default member, so the four contract -# commands build and test all of them. +# TinyMemory is three crates, one directory each under `crates/`: +# +# - `tinymemory-api`: the core contract (`MemoryEngine`, items, metadata, +# namespaces, errors) and, behind `conformance`, the suite every engine must +# pass. +# - `tinymemory-tools`: the agent-facing tool spec over any engine, and the +# `context.md` compiler. +# - `tinymemory-integrations`: everything that talks to the outside world — +# the CortexDB engine and its registry, document conversion, source readers, +# safety scrubbing and the legacy v1 import — each behind a feature. members = ["crates/*"] # `worktrees/` holds `git worktree` checkouts of this same repository. Each one # contains a full copy of this manifest and every crate under it, so without # this entry cargo walks into them and reports duplicate packages. exclude = ["worktrees"] +# Shared by every crate. The release workflow bumps `version` here, so the +# three crates always carry the same version and one tag names all of them. +[workspace.package] +version = "1.22.4" +edition = "2024" +rust-version = "1.96" +license = "GPL-3.0-only" +repository = "https://github.com/tinyhumansai/tinymemory" +publish = false + +# One lint table for the whole workspace; each crate opts in with +# `[lints] workspace = true`. +[workspace.lints.rust] +unsafe_code = "forbid" +missing_docs = "warn" +missing_debug_implementations = "warn" +unreachable_pub = "warn" +rust_2018_idioms = { level = "warn", priority = -1 } + +[workspace.lints.clippy] +all = { level = "warn", priority = -1 } +unwrap_used = "warn" +expect_used = "warn" +panic = "warn" +todo = "warn" +unimplemented = "warn" +missing_errors_doc = "warn" +missing_panics_doc = "warn" + +[workspace.lints.rustdoc] +broken_intra_doc_links = "warn" +private_intra_doc_links = "warn" + [profile.release] # Cross-crate optimization and smaller, faster binaries for release builds. lto = "thin" diff --git a/README.md b/README.md index 69173d12..f9aa59c1 100644 --- a/README.md +++ b/README.md @@ -1,8 +1,9 @@ # TinyMemory The memory layer for TinyHumans agents: **recall, fetch and store** over -pluggable engines, plus a token-budgeted `context.md` compiled from whatever is -stored. +pluggable engines, a token-budgeted `context.md` compiled from whatever is +stored, and a set of agent tools that let a model use memory without ever +choosing whose memory it touches. | Operation | Meaning | | --- | --- | @@ -10,95 +11,131 @@ stored. | **Fetch** | Raw keyword, vector or hybrid retrieval over stored items, filtered by metadata. No synthesis. | | **Store** | Ingest a document, a conversation or a learning, each with typed metadata. | -The behaviour is specified in [`docs/specs/memory-v2.md`](docs/specs/memory-v2.md), -which is the source of truth. +Around those three sit `list`, `forget`, `explore` and `get`, and the +namespace tree that keeps one tenant's or agent's memory apart from another's. + +- **Specified behaviour:** [`docs/specs/memory-v2.md`](docs/specs/memory-v2.md) + is the source of truth for what the system does and why. +- **How it is built:** [`docs/architecture/README.md`](docs/architecture/README.md) + has one page per concern: the contract, operations, namespaces, the CortexDB + engine, the agent tools, the integrations and the test strategy. ## Layout +The workspace is three crates, split by what each may depend on. + ```text crates/ -├── tinymemory/ the facade a host depends on: re-exports the -│ contract, the engine registry (`list_engines`, -│ `build_engine`), `MemoryConfig`, and every other -│ crate behind a feature named after it -├── tinymemory-api/ the contract: `MemoryEngine`, `StoreItem`, -│ `MemoryMeta`, `MetaFilter`, request/response -│ types, `EngineDescriptor`, `Error`. No I/O -├── tinymemory-cortex/ the CortexDB engine, registered twice: `cortexdb` -│ (direct `/v1/*`) and `tinyhumans` (CortexDB behind -│ the TinyHumans backend `/memory/*`) -├── tinymemory-documents/ format sniffing and conversion to markdown -│ (markdown, text, HTML, code; PDF/DOCX through a -│ host converter), emitting `StoreItem::Document` -├── tinymemory-sources/ readers turning a source into `StoreItem`s: folder, -│ file, link, GitHub, RSS, Composio payloads, local -│ conversations; includes the SSRF guard -├── tinymemory-safety/ secret and PII scrubbing applied before `store` -├── tinymemory-context/ `ContextCompiler`: builds `context.md` from an engine -├── tinymemory-import/ reads a legacy v1 (embedded TinyCortex) workspace -│ and yields resumable `StoreItem`s -└── tinymemory-conformance/ the suite every engine must pass, plus a reference - in-memory engine +├── tinymemory-api/ the contract: `MemoryEngine`, `StoreItem`, `MemoryMeta`, +│ `MetaFilter`, `Namespace`/`Reach`, request and response +│ types, `EngineDescriptor`, `Error`. No I/O. Feature +│ `conformance` adds the suite every engine must pass and +│ an in-memory reference engine +├── tinymemory-tools/ the agent surface over any engine: `MemoryTools` (seven +│ model-callable tools with JSON Schemas and host-fixed +│ scoping) and the `context.md` compiler +└── tinymemory-integrations/ everything that touches the outside world, one module + per feature: `cortex` (+ `registry`, `config`), + `documents`, `sources`, `safety`, `import` docs/ -├── specs/ behaviour and architecture specifications -├── plans/ test-first implementation plans -└── adr/ immutable architecture decision records +├── architecture/ how the code delivers the spec, one page per concern +├── specs/ behaviour and architecture specifications +├── plans/ test-first implementation plans +└── adr/ immutable architecture decision records ``` -## Features +`tinymemory-tools` and `tinymemory-integrations` each depend only on +`tinymemory-api`, never on each other, so the contract is the one coupling +point. A host takes the crates it needs. -The facade reaches every optional crate through a feature of the same name. -Nothing is on by default: naming no feature gets the contract, the registry and -the CortexDB engines. +## Integrations features -| Feature | Adds | -| --- | --- | -| `documents` | `tinymemory::documents` | -| `documents-office` | `tinymemory::documents::OfficeConverter` (PDF, DOCX, PPTX, XLSX) | -| `sources` | `tinymemory::sources` (local readers) | -| `sources-network` | the GitHub, RSS, web-page and URL-fetch readers (implies `sources`) | -| `safety` | `tinymemory::safety` | -| `context` | `tinymemory::context` | -| `import` / `legacy-import` | `tinymemory::import` | -| `conformance` | `tinymemory::conformance` | -| `full` | all of the above | +`tinymemory-integrations` enables the CortexDB engine by default and everything +else on request, so a host pays only for what it uses. + +| Feature | Module | Adds | +| --- | --- | --- | +| `cortex` (default) | `cortex`, `registry`, `config` | `CortexEngine` over both wires, `list_engines`, `build_engine`, `EngineCredential`, `MemoryConfig` | +| `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document` | +| `documents-office` | `documents::OfficeConverter` | PDF, DOCX, PPTX and XLSX to markdown (implies `documents`) | +| `sources` | `sources` | Folder, file and conversation readers, Composio normalisers (implies `documents`) | +| `sources-network` | `sources::fetch` and the network readers | GitHub, RSS and web-page readers and `fetch_url`, behind the SSRF guard (implies `sources`) | +| `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` | +| `legacy-import` | `import` | Migrating a v1 (embedded TinyCortex) workspace into any engine | +| `full` | all of the above | `cortex`, `documents-office`, `sources-network`, `safety`, `legacy-import` | + +Dependency weight per feature is tabulated in +[`crates/tinymemory-integrations/README.md`](crates/tinymemory-integrations/README.md). ## Using from your project -Nothing is published to crates.io; take the facade by git, pinned to a tag: +Nothing is published to crates.io. Take the crates you need by git, pinned to +a tag: ```toml [dependencies] -tinymemory = { git = "https://github.com/tinyhumansai/tinymemory", tag = "vX.Y.Z", features = ["context", "safety"] } +tinymemory-api = { git = "https://github.com/tinyhumansai/tinymemory", tag = "vX.Y.Z" } +tinymemory-tools = { git = "https://github.com/tinyhumansai/tinymemory", tag = "vX.Y.Z" } +tinymemory-integrations = { git = "https://github.com/tinyhumansai/tinymemory", tag = "vX.Y.Z", features = ["sources", "safety"] } ``` +All three crates carry the same version, so one tag names them all. A host that +only implements an engine needs just `tinymemory-api`; one that only offers +tools over an engine it already has needs `tinymemory-api` and +`tinymemory-tools`. + +## Quickstart + Choose an engine by configuration and hand it a credential from your own -secret store: +secret store, give a model memory tools scoped to one agent, and run what the +model asks for: ```rust,no_run use std::sync::Arc; -use tinymemory::{ - EngineCredential, FetchMode, FetchRequest, MemoryConfig, MemoryMeta, SourceKind, StaticBearer, - StoreItem, -}; - -# async fn demo() -> tinymemory::Result<()> { -let config: MemoryConfig = toml::from_str(r#"engine = "tinyhumans""#).unwrap(); +use serde_json::json; +use tinymemory_api::Namespace; +use tinymemory_integrations::{EngineCredential, MemoryConfig, StaticBearer}; +use tinymemory_tools::{MEMORY_STORE, MemoryTools}; + +# async fn demo() -> Result<(), Box> { +// The config names the engine; it never holds a credential. +let config: MemoryConfig = toml::from_str(r#"engine = "tinyhumans""#)?; let engine = config.build(EngineCredential::Dynamic(Arc::new(StaticBearer::new("tiny_live_..."))))?; -let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); -meta.file_path = Some("/notes/rust/ownership.md".into()); -engine.store(StoreItem::document("Ownership moves values.", meta)).await?; +// The host fixes where this agent writes and how far it reads. The model can +// name neither: a `namespace` or `reach` argument is refused at any depth. +let tools = MemoryTools::new(engine).placed_at(Namespace::agent("researcher")); -let page = engine.fetch(FetchRequest::new("ownership", FetchMode::Hybrid, 5)).await?; -# let _ = page; +// Hand these (name, description, JSON Schema) to your tool runtime. +for spec in tools.specs() { + println!("{}: {}", spec.name, spec.description); +} + +// Run a tool call the model produced. +let receipt = tools + .call(MEMORY_STORE, json!({ "learning": { "text": "The user prefers short answers" } })) + .await?; +println!("stored {}", receipt["id"]); # Ok(()) # } ``` +The tools can also be used without a model. Call the engine directly: + +```rust,ignore +use tinymemory_api::{FetchMode, FetchRequest, MemoryMeta, SourceKind, StoreItem}; + +let meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); +engine.store(StoreItem::document("Ownership moves values.", meta)).await?; +let page = engine.fetch(FetchRequest::new("ownership", FetchMode::Hybrid, 5)).await?; +``` + `build_engine` refuses an unknown engine id, a missing required endpoint or credential, and a credentialed cleartext endpoint that is not loopback. +To compile a `context.md` for the start of a session, call +`tinymemory_tools::context::compile(&*engine, &ContextSpec::default())`. + ## Engines | Id | What | Fetch modes | @@ -110,13 +147,35 @@ CortexDB is an append-only event log: writes wait until they are readable, listings are de-duplicated, forgets always name event ids, and an empty forget selector (which CortexDB reads as "the whole scope") is never sent. Its recall route has no keyword/vector switch, so both wires declare hybrid fetch only. See -`crates/tinymemory-cortex/README.md`. +[`docs/architecture/cortex.md`](docs/architecture/cortex.md) and +[`crates/tinymemory-integrations/src/cortex/README.md`](crates/tinymemory-integrations/src/cortex/README.md). ### Adding an engine -Implement `tinymemory_api::MemoryEngine` in its own crate, declare its fetch -modes honestly in its `EngineDescriptor`, pass `tinymemory_conformance::run` -against it, and register it in `crates/tinymemory/src/registry/`. +An engine is a module of `tinymemory-integrations` behind a feature named after +it. Implement `tinymemory_api::MemoryEngine`, declare its fetch modes honestly +in its `EngineDescriptor`, pass `tinymemory_api::conformance::run` against it +(enable the `conformance` feature of `tinymemory-api` in dev-dependencies), and +register it in `registry`. + +## Migrating from v1 + +A v1 (embedded TinyCortex) workspace is read in place and copied into any +engine, resumably. `migrate` takes the checkpoint of an earlier run (`None` to +start from the beginning) and returns where it got to: + +```rust,ignore +use tinymemory_integrations::import::{LegacyWorkspace, migrate}; + +let report = migrate(engine.as_ref(), LegacyWorkspace::open(path)?, None).await?; +println!("stored {}, replayed {}", report.stored, report.replayed); +``` + +Use `migrate_with` to receive each committed `Checkpoint` as it happens, so a +host can persist it. The +legacy workspace is opened read-only, and re-running never duplicates. Needs +the `legacy-import` feature. See +[`docs/architecture/integrations.md`](docs/architecture/integrations.md#legacy-v1-import). ## Development @@ -129,8 +188,19 @@ cargo build --all-targets --all-features cargo test --all-features ``` -`cargo run -p tinymemory --example basic` lists the engines and builds one -from configuration. Contribution rules are in [`AGENTS.md`](AGENTS.md). +Also useful: + +```bash +RUSTDOCFLAGS="-D warnings" cargo doc --no-deps --all-features +cargo test --doc --all-features +cargo run -p tinymemory-integrations --example basic +``` + +The example lists the registered engines and builds one from configuration, +with no network access. The `-p` is required because the workspace root is +virtual. Live tests against a real CortexDB are described in +[`docs/architecture/testing.md`](docs/architecture/testing.md). Contribution +rules are in [`AGENTS.md`](AGENTS.md). ## License diff --git a/crates/tinymemory-api/Cargo.toml b/crates/tinymemory-api/Cargo.toml index d2e11c7e..16ae2546 100644 --- a/crates/tinymemory-api/Cargo.toml +++ b/crates/tinymemory-api/Cargo.toml @@ -1,16 +1,19 @@ [package] name = "tinymemory-api" -publish = false -version = "2.0.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "The TinyMemory v2 contract: recall, fetch and store over typed memory items" +description = "The TinyMemory core contract: recall, fetch and store over typed memory items" +readme = "README.md" +keywords = ["memory", "agent", "llm", "retrieval"] +categories = ["database"] +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +publish.workspace = true # The contract an engine compiles against. It performs no I/O and links no -# runtime, storage engine or HTTP stack; anything heavier belongs in an engine -# crate. +# runtime, storage engine or HTTP stack; anything heavier belongs in +# `tinymemory-integrations`. [dependencies] # `MemoryEngine` is an object-safe trait of `async fn`s. async-trait = "0.1" @@ -28,23 +31,16 @@ thiserror = "2" [dev-dependencies] tokio = { version = "1", features = ["macros", "rt"] } -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } +[features] +default = [] +# `conformance::run`, the behavioural suite every engine must pass, and +# `conformance::ReferenceEngine`, an in-memory engine that passes it. Adds no +# dependency: both are written against the contract alone. +conformance = [] -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" +[[test]] +name = "conformance_reference" +required-features = ["conformance"] -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" +[lints] +workspace = true diff --git a/crates/tinymemory-api/README.md b/crates/tinymemory-api/README.md new file mode 100644 index 00000000..48e02af6 --- /dev/null +++ b/crates/tinymemory-api/README.md @@ -0,0 +1,87 @@ +# tinymemory-api + +The TinyMemory core contract: the operations a host needs from memory, and the +types they speak. The crate performs **no I/O** and links no runtime, HTTP +stack or storage engine, so an engine, a tool layer or a host can depend on it +alone. Engines live in +[`tinymemory-integrations`](../tinymemory-integrations); the agent-facing tools +and `context.md` compiler live in [`tinymemory-tools`](../tinymemory-tools). + +## Surface + +| Item | Purpose | +| --- | --- | +| `MemoryEngine` | The object-safe async trait: `recall`, `fetch`, `store`, `store_many`, `forget`, `list`, `explore`, `get`, plus `descriptor` and `health` | +| `EngineDescriptor`, `EngineHealth` | What an engine is and offers (including its `fetch_modes`), and whether it can serve | +| `StoreItem` | A `Document`, `Conversation` or `Learning`, each with a `MemoryMeta`; `validate` and `fingerprint` | +| `MemoryMeta`, `MetaFilter` | Typed metadata on every item, and the query that selects by it | +| `Namespace`, `Segment`, `Reach` | The memory tree: where an item lives and how far a reader reaches | +| `RecallRequest`, `FetchRequest`, `ListRequest`, `ForgetTarget`, ... | Requests and responses, each with a `validate` engines call first | +| `Facet`, `ExploreRequest`, `GetRequest` | Browsing: counts per metadata facet, and reading items by id | +| `Error`, `Result` | The one error every engine returns | + +`store_many`, `explore` and `get` have default implementations (one `store` at +a time; paging through `list`), so a minimal engine implements seven methods +and overrides the defaults only when it can do better. + +## Example + +```rust +use tinymemory_api::{ItemKind, MemoryMeta, MetaFilter, SourceKind, StoreItem}; + +let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); +meta.file_path = Some("/notes/rust/ownership.md".into()); +let item = StoreItem::document("Ownership moves values.", meta); +item.validate()?; + +let filter = MetaFilter { + file_path: Some("/notes/rust".into()), + ..MetaFilter::kinds([ItemKind::Document]) +}; +assert!(filter.matches(item.kind(), item.meta())); +# Ok::<(), tinymemory_api::Error>(()) +``` + +## Limits + +| Constant | Value | Bounds | +| --- | --- | --- | +| `MAX_STORE_MANY` | 100 | items per `store_many` | +| `MAX_GET_IDS` | 200 | ids per `GetRequest` | +| `MAX_BUCKETS` | 500 | `ExploreRequest::limit` | +| `MAX_SCAN_LIMIT` | 50 000 | `ExploreRequest::scan_limit` (default 5 000) | + +A namespace nests at most 8 deep and a segment id is 1 to 128 characters of +`A-Za-z0-9_-`. `recall`, `fetch` and `list` limits must be positive. + +## Things worth knowing + +- **Idempotency.** `StoreItem::fingerprint` hashes the whole item except + `meta.observed_at`; an engine turns an identical store into a replay + (`StoreReceipt::replayed`). +- **Empty forget is refused.** `ForgetTarget` with no ids or an empty filter is + `Error::InvalidRequest`; it would mean "everything". +- **Namespaces.** An item lives at one node. `MetaFilter::reach` confines reads + (`None` reads every node); `forget` by ids is not scoped. +- **Fetch modes.** A mode missing from `EngineDescriptor::fetch_modes` fails + with `Error::Unsupported`. + +## The `conformance` feature + +`features = ["conformance"]` adds `tinymemory_api::conformance`: +`run(&dyn MemoryEngine)`, the behavioural suite every engine must pass, and +`ReferenceEngine`, an in-memory engine that passes it. It adds no dependency. + +```toml +[dev-dependencies] +tinymemory-api = { path = "../tinymemory-api", features = ["conformance"] } +``` + +## Further reading + +Architecture documents live in +[`docs/architecture`](../../docs/architecture/README.md): the +[contract reference](../../docs/architecture/api.md), +[operation semantics](../../docs/architecture/operations.md) and +[namespaces](../../docs/architecture/namespaces.md). The accepted behaviour is +[`docs/specs/memory-v2.md`](../../docs/specs/memory-v2.md). diff --git a/crates/tinymemory-conformance/src/error/mod.rs b/crates/tinymemory-api/src/conformance/error/mod.rs similarity index 95% rename from crates/tinymemory-conformance/src/error/mod.rs rename to crates/tinymemory-api/src/conformance/error/mod.rs index 972f73bd..f51252f8 100644 --- a/crates/tinymemory-conformance/src/error/mod.rs +++ b/crates/tinymemory-api/src/conformance/error/mod.rs @@ -17,7 +17,7 @@ pub enum Error { /// The check's name. check: &'static str, /// The engine's error. - source: tinymemory_api::Error, + source: crate::Error, }, } diff --git a/crates/tinymemory-conformance/src/lib.rs b/crates/tinymemory-api/src/conformance/mod.rs similarity index 86% rename from crates/tinymemory-conformance/src/lib.rs rename to crates/tinymemory-api/src/conformance/mod.rs index 9cbb042f..f562e353 100644 --- a/crates/tinymemory-conformance/src/lib.rs +++ b/crates/tinymemory-api/src/conformance/mod.rs @@ -2,7 +2,7 @@ //! in-memory engine to calibrate it. //! //! [`run`] stores, lists, fetches, recalls and forgets through any -//! [`tinymemory_api::MemoryEngine`] and reports the first behaviour that breaks +//! [`MemoryEngine`](crate::MemoryEngine) and reports the first behaviour that breaks //! the contract. It covers store/list round trips for each item kind, replay //! idempotency, fetch filtering by every metadata field in every declared //! mode, forget by id and by filter, refusal of an empty forget, @@ -16,14 +16,14 @@ //! # Example //! //! ``` -//! use tinymemory_conformance::{ReferenceEngine, run}; +//! use tinymemory_api::conformance::{ReferenceEngine, run}; //! //! # let runtime = tokio::runtime::Builder::new_current_thread().build()?; //! # runtime.block_on(async { //! let engine = ReferenceEngine::new(); //! run(&engine).await?; //! assert!(engine.is_empty(), "the suite cleans up after itself"); -//! # Ok::<(), tinymemory_conformance::Error>(()) +//! # Ok::<(), tinymemory_api::conformance::Error>(()) //! # })?; //! # Ok::<(), Box>(()) //! ``` diff --git a/crates/tinymemory-conformance/src/reference/mod.rs b/crates/tinymemory-api/src/conformance/reference/mod.rs similarity index 99% rename from crates/tinymemory-conformance/src/reference/mod.rs rename to crates/tinymemory-api/src/conformance/reference/mod.rs index 3319597d..68957e6f 100644 --- a/crates/tinymemory-conformance/src/reference/mod.rs +++ b/crates/tinymemory-api/src/conformance/reference/mod.rs @@ -10,12 +10,12 @@ mod score; use std::sync::Mutex; -use async_trait::async_trait; -use tinymemory_api::{ +use crate::{ Citation, EngineDescriptor, EngineHealth, Error, FetchMode, FetchPage, FetchRequest, ForgetReport, ForgetTarget, Hit, ItemId, ListPage, ListRequest, MemoryEngine, MetaFilter, RecallAnswer, RecallRequest, Result, StoreItem, StoreReceipt, }; +use async_trait::async_trait; /// The reference engine's id. pub const REFERENCE_ENGINE_ID: &str = "reference"; diff --git a/crates/tinymemory-conformance/src/reference/mod_tests.rs b/crates/tinymemory-api/src/conformance/reference/mod_tests.rs similarity index 97% rename from crates/tinymemory-conformance/src/reference/mod_tests.rs rename to crates/tinymemory-api/src/conformance/reference/mod_tests.rs index 95b55435..5ab52582 100644 --- a/crates/tinymemory-conformance/src/reference/mod_tests.rs +++ b/crates/tinymemory-api/src/conformance/reference/mod_tests.rs @@ -1,6 +1,6 @@ //! The reference engine's own behaviour, independent of the suite. -use tinymemory_api::{ItemKind, LearningKind, MemoryMeta}; +use crate::{ItemKind, LearningKind, MemoryMeta}; use super::*; diff --git a/crates/tinymemory-conformance/src/reference/score.rs b/crates/tinymemory-api/src/conformance/reference/score.rs similarity index 98% rename from crates/tinymemory-conformance/src/reference/score.rs rename to crates/tinymemory-api/src/conformance/reference/score.rs index 73440c24..64ac7a00 100644 --- a/crates/tinymemory-conformance/src/reference/score.rs +++ b/crates/tinymemory-api/src/conformance/reference/score.rs @@ -4,7 +4,7 @@ //! Neither is meant to rank well. They exist so the reference engine can serve //! all three fetch modes with behaviour that is obvious by inspection. -use tinymemory_api::FetchMode; +use crate::FetchMode; /// Dimensions of the toy vector. const DIMENSIONS: usize = 64; diff --git a/crates/tinymemory-conformance/src/reference/score_tests.rs b/crates/tinymemory-api/src/conformance/reference/score_tests.rs similarity index 100% rename from crates/tinymemory-conformance/src/reference/score_tests.rs rename to crates/tinymemory-api/src/conformance/reference/score_tests.rs diff --git a/crates/tinymemory-conformance/src/suite/bulk.rs b/crates/tinymemory-api/src/conformance/suite/bulk.rs similarity index 96% rename from crates/tinymemory-conformance/src/suite/bulk.rs rename to crates/tinymemory-api/src/conformance/suite/bulk.rs index ca56ae01..2917328b 100644 --- a/crates/tinymemory-conformance/src/suite/bulk.rs +++ b/crates/tinymemory-api/src/conformance/suite/bulk.rs @@ -1,10 +1,10 @@ //! The bulk check: `store_many` stores in order, every item is listed on //! return, a repeat is all replays, and an empty batch is refused. -use tinymemory_api::Error as ApiError; +use crate::Error as ApiError; use super::{Ctx, ensure}; -use crate::error::Result; +use crate::conformance::error::Result; pub(super) async fn store_many(ctx: &Ctx<'_>) -> Result<()> { const CHECK: &str = "store_many"; diff --git a/crates/tinymemory-conformance/src/suite/checks.rs b/crates/tinymemory-api/src/conformance/suite/checks.rs similarity index 99% rename from crates/tinymemory-conformance/src/suite/checks.rs rename to crates/tinymemory-api/src/conformance/suite/checks.rs index 36154b68..b83667d5 100644 --- a/crates/tinymemory-conformance/src/suite/checks.rs +++ b/crates/tinymemory-api/src/conformance/suite/checks.rs @@ -2,13 +2,13 @@ use std::collections::BTreeSet; -use tinymemory_api::{ +use crate::{ Error as ApiError, FetchMode, FetchRequest, ForgetTarget, ItemId, MemoryMeta, MetaFilter, RecallRequest, }; use super::{Ctx, ensure}; -use crate::error::Result; +use crate::conformance::error::Result; /// Every check, stopping at the first failure. pub(super) async fn all(ctx: &Ctx<'_>) -> Result<()> { diff --git a/crates/tinymemory-conformance/src/suite/explore.rs b/crates/tinymemory-api/src/conformance/suite/explore.rs similarity index 96% rename from crates/tinymemory-conformance/src/suite/explore.rs rename to crates/tinymemory-api/src/conformance/suite/explore.rs index 941f27ac..5fbdc38d 100644 --- a/crates/tinymemory-conformance/src/suite/explore.rs +++ b/crates/tinymemory-api/src/conformance/suite/explore.rs @@ -3,10 +3,10 @@ use std::collections::BTreeMap; -use tinymemory_api::{Error as ApiError, ExploreRequest, Facet, GetRequest, ItemId}; +use crate::{Error as ApiError, ExploreRequest, Facet, GetRequest, ItemId}; use super::{Ctx, ensure}; -use crate::error::Result; +use crate::conformance::error::Result; /// Groups the run's items by kind and by workspace and checks every count /// against a listing, then narrows by each bucket and lists again. @@ -51,7 +51,7 @@ pub(super) async fn explore(ctx: &Ctx<'_>) -> Result<()> { let mut narrowed = filter.clone(); Facet::Kind .narrow(&mut narrowed, &bucket.value) - .map_err(|source| crate::Error::Engine { + .map_err(|source| crate::conformance::Error::Engine { check: CHECK, source, })?; diff --git a/crates/tinymemory-conformance/src/suite/fixtures.rs b/crates/tinymemory-api/src/conformance/suite/fixtures.rs similarity index 99% rename from crates/tinymemory-conformance/src/suite/fixtures.rs rename to crates/tinymemory-api/src/conformance/suite/fixtures.rs index 5667fb38..0a7bf304 100644 --- a/crates/tinymemory-conformance/src/suite/fixtures.rs +++ b/crates/tinymemory-api/src/conformance/suite/fixtures.rs @@ -4,8 +4,8 @@ //! can isolate its own items on an engine that already holds data and find //! them with any fetch mode. -use tinymemory_api::chrono::{TimeZone, Utc}; -use tinymemory_api::{ +use crate::chrono::{TimeZone, Utc}; +use crate::{ DocumentBody, ItemKind, LearningKind, MemoryMeta, MetaFilter, Role, SourceKind, SourceRef, StoreItem, ToolCallRef, Turn, TurnRange, }; diff --git a/crates/tinymemory-conformance/src/suite/mod.rs b/crates/tinymemory-api/src/conformance/suite/mod.rs similarity index 96% rename from crates/tinymemory-conformance/src/suite/mod.rs rename to crates/tinymemory-api/src/conformance/suite/mod.rs index 4ca7d2ce..a79a99d0 100644 --- a/crates/tinymemory-conformance/src/suite/mod.rs +++ b/crates/tinymemory-api/src/conformance/suite/mod.rs @@ -39,9 +39,9 @@ mod namespaces; use std::collections::HashSet; use std::future::Future; -use tinymemory_api::{Hit, ListRequest, MemoryEngine, MetaFilter}; +use crate::{Hit, ListRequest, MemoryEngine, MetaFilter}; -use crate::error::{Error, Result}; +use crate::conformance::error::{Error, Result}; use fixtures::Run; /// Most pages one listing may take before the suite calls the cursor endless. @@ -74,7 +74,7 @@ impl Ctx<'_> { pub(crate) async fn call( &self, check: &'static str, - call: impl Future>, + call: impl Future>, ) -> Result { call.await.map_err(|source| Error::Engine { check, source }) } diff --git a/crates/tinymemory-conformance/src/suite/namespaces.rs b/crates/tinymemory-api/src/conformance/suite/namespaces.rs similarity index 98% rename from crates/tinymemory-conformance/src/suite/namespaces.rs rename to crates/tinymemory-api/src/conformance/suite/namespaces.rs index e0240437..40365ea9 100644 --- a/crates/tinymemory-conformance/src/suite/namespaces.rs +++ b/crates/tinymemory-api/src/conformance/suite/namespaces.rs @@ -3,13 +3,13 @@ use std::collections::BTreeSet; -use tinymemory_api::{ +use crate::{ ExploreRequest, Facet, FetchRequest, ForgetTarget, GetRequest, ItemId, MetaFilter, Namespace, Reach, StoreItem, }; use super::{Ctx, ensure}; -use crate::error::{Error, Result}; +use crate::conformance::error::{Error, Result}; const CHECK: &str = "namespaces"; @@ -50,7 +50,7 @@ pub(super) async fn namespaces(ctx: &Ctx<'_>) -> Result<()> { meta.tags = vec![TAG.to_string()]; StoreItem::learning( format!("{} {label} shared fact", ctx.run.marker), - tinymemory_api::LearningKind::Fact, + crate::LearningKind::Fact, 0.9, meta, ) diff --git a/crates/tinymemory-api/src/explore/mod.rs b/crates/tinymemory-api/src/explore/mod.rs index 4765c1cc..7e9533f6 100644 --- a/crates/tinymemory-api/src/explore/mod.rs +++ b/crates/tinymemory-api/src/explore/mod.rs @@ -30,7 +30,7 @@ use crate::query::{Hit, ListRequest}; pub const MAX_BUCKETS: usize = 500; /// Default for [`ExploreRequest::scan_limit`]. -pub const DEFAULT_SCAN_LIMIT: usize = 5_000; +const DEFAULT_SCAN_LIMIT: usize = 5_000; /// Most items a listing-based explore reads. pub const MAX_SCAN_LIMIT: usize = 50_000; @@ -71,31 +71,12 @@ pub enum Facet { ToolCall, /// One of `meta.tags`; an item with several tags counts in each. Tag, - /// `meta.namespace`: the memory node, [`crate::namespace::ROOT_LABEL`] for - /// the root. - /// Narrowing reads exactly that node. + /// `meta.namespace`: the memory node, `root` for the root. Narrowing + /// reads exactly that node. Namespace, } impl Facet { - /// Every facet, in declaration order. - pub const ALL: [Self; 14] = [ - Self::Kind, - Self::Source, - Self::SourceId, - Self::Workspace, - Self::Folder, - Self::FilePath, - Self::Language, - Self::Repo, - Self::Url, - Self::Thread, - Self::Agent, - Self::ToolCall, - Self::Tag, - Self::Namespace, - ]; - /// The stable snake_case wire string. #[must_use] pub fn as_str(self) -> &'static str { @@ -203,8 +184,8 @@ pub struct ExploreRequest { /// Most buckets to return, largest first; `1..=`[`MAX_BUCKETS`]. pub limit: usize, /// Most items a listing-based engine reads before it stops and reports - /// [`ExplorePage::truncated`]; `1..=`[`MAX_SCAN_LIMIT`]. An engine that - /// aggregates server-side may ignore it. + /// [`ExplorePage::truncated`]; `1..=`[`MAX_SCAN_LIMIT`], 5,000 when + /// omitted. An engine that aggregates server-side may ignore it. #[serde(default = "default_scan_limit")] pub scan_limit: usize, } @@ -350,10 +331,8 @@ pub async fn explore_by_listing( } /// Builds an [`ExplorePage`] from per-value counts: largest first, ties by -/// value, cut to `limit`. Public so an engine aggregating server-side -/// shapes its answer identically. -#[must_use] -pub fn page_of( +/// value, cut to `limit`. +fn page_of( facet: Facet, counts: BTreeMap, limit: usize, diff --git a/crates/tinymemory-api/src/explore/mod_tests.rs b/crates/tinymemory-api/src/explore/mod_tests.rs index d72efa9e..b207bc3c 100644 --- a/crates/tinymemory-api/src/explore/mod_tests.rs +++ b/crates/tinymemory-api/src/explore/mod_tests.rs @@ -135,9 +135,48 @@ fn fixture() -> Vec { ] } +/// Every facet. [`exhaustive`] fails to compile when a facet is added and +/// not listed here. +const FACETS: [Facet; 14] = [ + Facet::Kind, + Facet::Source, + Facet::SourceId, + Facet::Workspace, + Facet::Folder, + Facet::FilePath, + Facet::Language, + Facet::Repo, + Facet::Url, + Facet::Thread, + Facet::Agent, + Facet::ToolCall, + Facet::Tag, + Facet::Namespace, +]; + +fn exhaustive(facet: Facet) { + match facet { + Facet::Kind + | Facet::Source + | Facet::SourceId + | Facet::Workspace + | Facet::Folder + | Facet::FilePath + | Facet::Language + | Facet::Repo + | Facet::Url + | Facet::Thread + | Facet::Agent + | Facet::ToolCall + | Facet::Tag + | Facet::Namespace => {} + } +} + #[test] fn every_facet_round_trips_its_wire_name() { - for facet in Facet::ALL { + for facet in FACETS { + exhaustive(facet); let json = serde_json::to_value(facet).unwrap(); assert_eq!(json, serde_json::json!(facet.as_str())); assert_eq!(serde_json::from_value::(json).unwrap(), facet); @@ -147,7 +186,7 @@ fn every_facet_round_trips_its_wire_name() { #[test] fn a_narrowed_filter_admits_exactly_the_items_with_that_value() { let items = fixture(); - for facet in Facet::ALL { + for facet in FACETS { for item in &items { for value in facet.values(item.kind, &item.meta) { let mut filter = MetaFilter::default(); diff --git a/crates/tinymemory-api/src/lib.rs b/crates/tinymemory-api/src/lib.rs index eb2da603..6f023e5b 100644 --- a/crates/tinymemory-api/src/lib.rs +++ b/crates/tinymemory-api/src/lib.rs @@ -21,9 +21,9 @@ //! [`EngineDescriptor`]; a fetch mode it does not list fails with //! [`Error::Unsupported`]. //! -//! This crate performs no I/O. Engines live in their own crates -//! (`tinymemory-cortex`), and the `tinymemory` facade builds one from -//! configuration. +//! This crate performs no I/O. Engines live in `tinymemory-integrations` +//! (the CortexDB engine, and the registry that builds one from +//! configuration). //! //! # Example //! @@ -43,6 +43,8 @@ //! # Ok::<(), tinymemory_api::Error>(()) //! ``` +#[cfg(feature = "conformance")] +pub mod conformance; pub mod engine; pub mod error; pub mod explore; @@ -53,7 +55,10 @@ pub mod query; pub use engine::{EngineDescriptor, EngineHealth, MAX_STORE_MANY, MemoryEngine, validate_many}; pub use error::{Error, Result}; -pub use explore::{ExplorePage, ExploreRequest, Facet, FacetBucket, GetRequest}; +pub use explore::{ + ExplorePage, ExploreRequest, Facet, FacetBucket, GetRequest, MAX_BUCKETS, MAX_GET_IDS, + MAX_SCAN_LIMIT, +}; pub use item::{DocumentBody, ItemId, ItemKind, LearningKind, Role, StoreItem, StoreReceipt, Turn}; pub use meta::{MemoryMeta, MetaFilter, SourceKind, SourceRef, ToolCallRef, TurnRange}; pub use namespace::{Namespace, Reach, Segment, SegmentKind}; diff --git a/crates/tinymemory-api/src/meta/filter.rs b/crates/tinymemory-api/src/meta/filter.rs index fee99abb..612859b4 100644 --- a/crates/tinymemory-api/src/meta/filter.rs +++ b/crates/tinymemory-api/src/meta/filter.rs @@ -102,8 +102,7 @@ impl MetaFilter { /// Whether the filter's reach admits `namespace` (ignoring every other /// field). - #[must_use] - pub fn admits_namespace(&self, namespace: &Namespace) -> bool { + fn admits_namespace(&self, namespace: &Namespace) -> bool { self.reach .as_ref() .is_none_or(|reach| reach.admits(namespace)) diff --git a/crates/tinymemory-api/src/namespace/mod.rs b/crates/tinymemory-api/src/namespace/mod.rs index 68fbfdba..1f66c7e1 100644 --- a/crates/tinymemory-api/src/namespace/mod.rs +++ b/crates/tinymemory-api/src/namespace/mod.rs @@ -25,13 +25,13 @@ use serde::{Deserialize, Deserializer, Serialize, Serializer}; use crate::error::{Error, Result}; /// Deepest a namespace may nest. -pub const MAX_DEPTH: usize = 8; +pub(crate) const MAX_DEPTH: usize = 8; /// Longest a segment id may be. -pub const MAX_SEGMENT_ID: usize = 128; +pub(crate) const MAX_SEGMENT_ID: usize = 128; /// How the root namespace is written. -pub const ROOT_LABEL: &str = "root"; +pub(crate) const ROOT_LABEL: &str = "root"; /// What a namespace segment names. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] @@ -50,15 +50,6 @@ pub enum SegmentKind { } impl SegmentKind { - /// Every kind, in declaration order. - pub const ALL: [Self; 5] = [ - Self::Agent, - Self::Team, - Self::User, - Self::Workspace, - Self::Project, - ]; - /// The stable wire prefix (`agent`, `team`, `user`, `ws`, `project`). #[must_use] pub fn as_str(self) -> &'static str { @@ -72,12 +63,16 @@ impl SegmentKind { } fn parse(value: &str) -> Result { - Self::ALL - .into_iter() - .find(|kind| kind.as_str() == value) - .ok_or_else(|| { - Error::InvalidRequest(format!("`{value}` is not a namespace segment kind")) - }) + match value { + "agent" => Ok(Self::Agent), + "team" => Ok(Self::Team), + "user" => Ok(Self::User), + "ws" => Ok(Self::Workspace), + "project" => Ok(Self::Project), + _ => Err(Error::InvalidRequest(format!( + "`{value}` is not a namespace segment kind" + ))), + } } } @@ -89,8 +84,7 @@ pub struct Segment { } impl Segment { - /// A segment, checking its id: `1..=`[`MAX_SEGMENT_ID`] characters of - /// `[A-Za-z0-9_-]`. + /// A segment, checking its id: 1 to 128 characters of `[A-Za-z0-9_-]`. /// /// # Errors /// @@ -167,7 +161,7 @@ fn fnv1a(value: &str) -> u32 { /// A node of the memory tree, as the path from the root. /// /// Written `team:acme/agent:writer`; the root is the empty path, written -/// [`ROOT_LABEL`]. On the wire a namespace is that string. +/// `root`. On the wire a namespace is that string. #[derive(Debug, Clone, Default, PartialEq, Eq, Hash, PartialOrd, Ord)] pub struct Namespace(Vec); @@ -179,7 +173,7 @@ impl Namespace { /// /// # Errors /// - /// [`Error::InvalidRequest`] when deeper than [`MAX_DEPTH`]. + /// [`Error::InvalidRequest`] when deeper than 8 segments. pub fn new(segments: Vec) -> Result { if segments.len() > MAX_DEPTH { return Err(Error::InvalidRequest(format!( @@ -196,18 +190,6 @@ impl Namespace { Self(vec![Segment::sanitized(SegmentKind::Agent, id)]) } - /// This node with `segment` appended: a child. - /// - /// # Errors - /// - /// [`Error::InvalidRequest`] when the child would be deeper than - /// [`MAX_DEPTH`]. - pub fn child(&self, segment: Segment) -> Result { - let mut segments = self.0.clone(); - segments.push(segment); - Self::new(segments) - } - /// Whether this is the root. #[must_use] pub fn is_root(&self) -> bool { @@ -226,38 +208,17 @@ impl Namespace { self.0.len() } - /// The parent node; `None` for the root. - #[must_use] - pub fn parent(&self) -> Option { - (!self.is_root()).then(|| Self(self.0[..self.0.len() - 1].to_vec())) - } - /// The root, every ancestor, then this node. - #[must_use] - pub fn ancestors_and_self(&self) -> Vec { + fn ancestors_and_self(&self) -> Vec { (0..=self.0.len()) .map(|depth| Self(self.0[..depth].to_vec())) .collect() } /// Whether this node is `other` or lies below it. - #[must_use] - pub fn is_within(&self, other: &Self) -> bool { + fn is_within(&self, other: &Self) -> bool { self.0.starts_with(&other.0) } - - /// The nearest node at or above this one whose last segment is not an - /// agent: where memory meant to be shared with an agent's peers goes (a - /// team, a workspace, or the root). - #[must_use] - pub fn shared_ancestor(&self) -> Self { - let keep = self - .0 - .iter() - .rposition(|segment| segment.kind != SegmentKind::Agent) - .map_or(0, |index| index + 1); - Self(self.0[..keep].to_vec()) - } } impl fmt::Display for Namespace { @@ -278,7 +239,7 @@ impl fmt::Display for Namespace { impl FromStr for Namespace { type Err = Error; - /// Parses `team:acme/agent:writer`; `""` and [`ROOT_LABEL`] are the root. + /// Parses `team:acme/agent:writer`; `""` and `root` are the root. fn from_str(value: &str) -> Result { let value = value.trim(); if value.is_empty() || value == ROOT_LABEL { diff --git a/crates/tinymemory-api/src/namespace/mod_tests.rs b/crates/tinymemory-api/src/namespace/mod_tests.rs index aa06fb9a..7f0dbd6c 100644 --- a/crates/tinymemory-api/src/namespace/mod_tests.rs +++ b/crates/tinymemory-api/src/namespace/mod_tests.rs @@ -14,6 +14,10 @@ fn parses_and_prints_paths() { assert_eq!(writer.segments()[0].kind(), SegmentKind::Team); assert_eq!(writer.segments()[1].id(), "writer"); assert_eq!(ns("ws:shared").to_string(), "ws:shared"); + for kind in ["agent", "team", "user", "ws", "project"] { + let segment = ns(&format!("{kind}:x")).segments()[0].clone(); + assert_eq!(segment.kind().as_str(), kind); + } assert!(ns("").is_root()); assert!(ns(ROOT_LABEL).is_root()); assert_eq!(Namespace::ROOT.to_string(), ROOT_LABEL); @@ -56,8 +60,6 @@ fn sanitizes_host_ids_without_collisions() { #[test] fn walks_the_tree() { let scout = ns("agent:researcher/agent:scout"); - assert_eq!(scout.parent(), Some(ns("agent:researcher"))); - assert_eq!(Namespace::ROOT.parent(), None); assert_eq!( scout.ancestors_and_self(), vec![Namespace::ROOT, ns("agent:researcher"), scout.clone()] @@ -66,21 +68,6 @@ fn walks_the_tree() { assert!(scout.is_within(&Namespace::ROOT)); assert!(!ns("agent:researcher").is_within(&scout)); assert!(!ns("agent:researchers").is_within(&ns("agent:researcher"))); - let child = ns("team:acme") - .child(Segment::new(SegmentKind::Agent, "writer").unwrap()) - .unwrap(); - assert_eq!(child, ns("team:acme/agent:writer")); -} - -#[test] -fn shared_ancestor_skips_agents() { - assert_eq!(ns("agent:a/agent:b").shared_ancestor(), Namespace::ROOT); - assert_eq!( - ns("team:acme/agent:writer").shared_ancestor(), - ns("team:acme") - ); - assert_eq!(ns("team:acme").shared_ancestor(), ns("team:acme")); - assert_eq!(Namespace::ROOT.shared_ancestor(), Namespace::ROOT); } #[test] diff --git a/crates/tinymemory-conformance/tests/reference.rs b/crates/tinymemory-api/tests/conformance_reference.rs similarity index 98% rename from crates/tinymemory-conformance/tests/reference.rs rename to crates/tinymemory-api/tests/conformance_reference.rs index daed7176..c0f454b2 100644 --- a/crates/tinymemory-conformance/tests/reference.rs +++ b/crates/tinymemory-api/tests/conformance_reference.rs @@ -4,12 +4,12 @@ //! nothing would also be green. use async_trait::async_trait; +use tinymemory_api::conformance::{Error, ReferenceEngine, run}; use tinymemory_api::{ EngineDescriptor, EngineHealth, ExplorePage, ExploreRequest, FetchMode, FetchPage, FetchRequest, ForgetReport, ForgetTarget, GetRequest, Hit, ListPage, ListRequest, MemoryEngine, MetaFilter, RecallAnswer, RecallRequest, Result, StoreItem, StoreReceipt, }; -use tinymemory_conformance::{Error, ReferenceEngine, run}; #[tokio::test] async fn the_reference_engine_passes_and_cleans_up() { diff --git a/crates/tinymemory-conformance/Cargo.toml b/crates/tinymemory-conformance/Cargo.toml deleted file mode 100644 index dc19720f..00000000 --- a/crates/tinymemory-conformance/Cargo.toml +++ /dev/null @@ -1,41 +0,0 @@ -[package] -name = "tinymemory-conformance" -publish = false -version = "2.0.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "The behavioural suite every TinyMemory engine must pass, plus a reference in-memory engine" - -[dependencies] -# The contract the suite exercises and the reference engine implements. -tinymemory-api = { path = "../tinymemory-api" } -# `ReferenceEngine` implements the object-safe async `MemoryEngine` trait. -async-trait = "0.1" -# The crate-wide `Error`. -thiserror = "2" - -[dev-dependencies] -tokio = { version = "1", features = ["macros", "rt"] } - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-context/Cargo.toml b/crates/tinymemory-context/Cargo.toml deleted file mode 100644 index b4b87eba..00000000 --- a/crates/tinymemory-context/Cargo.toml +++ /dev/null @@ -1,47 +0,0 @@ -[package] -name = "tinymemory-context" -publish = false -version = "2.0.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "Compiles a token-budgeted context.md brief from any TinyMemory engine" - -[dependencies] -# The engine the brief is compiled from, and the items it cites. -tinymemory-api = { path = "../tinymemory-api" } -# `ContextDoc::generated_at` is stamped from the system clock. -chrono = { version = "0.4", default-features = false, features = ["clock", "std", "serde"] } -# A brief whose recall fails is skipped and logged, not fatal. -log = "0.4" -# `ContextSpec` and `Brief` are host configuration. -serde = { version = "1", features = ["derive"] } -# The crate-wide `Error`. -thiserror = "2" - -[dev-dependencies] -tinymemory-conformance = { path = "../tinymemory-conformance" } -async-trait = "0.1" -tokio = { version = "1", features = ["macros", "rt"] } - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-cortex/Cargo.toml b/crates/tinymemory-cortex/Cargo.toml deleted file mode 100644 index 77a6881d..00000000 --- a/crates/tinymemory-cortex/Cargo.toml +++ /dev/null @@ -1,60 +0,0 @@ -[package] -name = "tinymemory-cortex" -publish = false -version = "2.0.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "The CortexDB memory engine, direct (`/v1/*`) and behind the TinyHumans backend (`/memory/*`)" - -[dependencies] -# The contract this engine implements: `MemoryEngine`, the item and query -# types, `EngineDescriptor` and the one `Error`. -tinymemory-api = { path = "../tinymemory-api" } -# `BearerSource` is an object-safe async trait, like `MemoryEngine`. -async-trait = "0.1" -# CortexDB speaks HTTP/JSON. `stream` is for `bytes_stream()`: response bodies -# are read against a byte cap rather than buffered whole, because the endpoint -# is operator-supplied and a broken or hostile one must not exhaust the host. -reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "stream"] } -# Only the timer: read-retry backoff and visibility polling. reqwest already -# requires a tokio runtime, so this adds no new runtime assumption. -tokio = { version = "1", default-features = false, features = ["time"] } -# The v2 event envelope and the opaque page cursors are JSON. -serde = { version = "1", features = ["derive"] } -serde_json = "1" -# `StreamExt` to read a capped body chunk by chunk. -futures = "0.3" -# Lookup labels are fixed-length SHA-256 digests of metadata values. -sha2 = "0.10" - -[dev-dependencies] -# The behavioural suite every engine must pass, run over both wires' doubles. -tinymemory-conformance = { path = "../tinymemory-conformance" } -# `tests/live_cortexdb.rs` compiles `context.md` from a live server. -tinymemory-context = { path = "../tinymemory-context" } -# The CortexDB and TinyHumans doubles are real HTTP servers on loopback. -axum = "0.8" -tokio = { version = "1", features = ["macros", "rt-multi-thread", "net", "time"] } - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-documents/Cargo.toml b/crates/tinymemory-documents/Cargo.toml deleted file mode 100644 index 37a8c857..00000000 --- a/crates/tinymemory-documents/Cargo.toml +++ /dev/null @@ -1,75 +0,0 @@ -[package] -name = "tinymemory-documents" -version = "0.1.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "Document intake for TinyMemory: sniff a format, convert it to markdown, emit a StoreItem::Document" -publish = false - -[dependencies] -# The contract. Intake emits `StoreItem::Document` with the caller's -# `MemoryMeta`, so the item it builds is exactly what an engine stores. -tinymemory-api = { path = "../tinymemory-api" } -# `DocumentConverter` is an object-safe async trait: a host swaps the converter -# (a PDF or DOCX extractor) without this crate knowing which one it got. -async-trait = "0.1" -# `ConvertedDocument::metadata` is an open `Value` a converter fills with -# whatever it learned (page count, author, its own name). -serde_json = "1" -# `DocumentFormat` and `ConvertedDocument` cross host boundaries as JSON. -serde = { version = "1", features = ["derive"] } -# The crate-wide `Error`. -thiserror = "2" -# The office converter (`office` feature): text out of the formats people -# actually drop into memory — a contract PDF, a spec `.docx`, a pricing -# `.xlsx`, a deck. All pure Rust with no system libraries. -# -# `pdf-extract` reads a PDF's text layer. `zip` + `quick-xml` are the whole of -# `.docx` and `.pptx`, which are zip archives of XML: walking them directly is -# smaller than a document library. `.xlsx` is not — shared-string tables and -# cell typing make hand-parsing it a liability — so `calamine` reads that one. -# `zip` and `quick-xml` are the versions `calamine` already links, so the -# graph carries one copy of each. -pdf-extract = { version = "0.12", optional = true } -calamine = { version = "0.36", optional = true } -quick-xml = { version = "0.41", optional = true } -zip = { version = "8", default-features = false, features = ["deflate"], optional = true } - -[dev-dependencies] -# The converter and item paths are async. -tokio = { version = "1", features = ["macros", "rt", "rt-multi-thread"] } -# The format and office tests build their OOXML fixtures in code, so each test -# says what it is asserting about instead of pointing at an opaque binary. -zip = { version = "8", default-features = false, features = ["deflate"] } - -[features] -# Nothing by default: text, markdown, HTML and code need no extractor. -default = [] -# `OfficeConverter`: PDF, DOCX, PPTX and XLSX to markdown, for a host to -# prepend to its `ConverterChain`. Off by default because a PDF parser and a -# spreadsheet reader are real weight a host that takes only text should not -# link. -office = ["dep:pdf-extract", "dep:calamine", "dep:quick-xml", "dep:zip"] - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-import/Cargo.toml b/crates/tinymemory-import/Cargo.toml deleted file mode 100644 index 305c3ec2..00000000 --- a/crates/tinymemory-import/Cargo.toml +++ /dev/null @@ -1,52 +0,0 @@ -[package] -name = "tinymemory-import" -publish = false -version = "0.1.0" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -description = "Reads a legacy (v1, embedded TinyCortex) workspace and yields TinyMemory v2 `StoreItem`s, resumably" -repository = "https://github.com/tinyhumansai/tinymemory" - -# The importer reads a v1 workspace straight off disk: the SQLite files with -# rusqlite and the chunk bodies with `std::fs`. It deliberately does not link -# the engine that wrote them, which no longer exists in this tree. -[dependencies] -# `StoreItem`, `MemoryMeta`, `SourceKind::Import` and the re-exported `chrono` -# the importer maps every legacy row into. -tinymemory-api = { path = "../tinymemory-api" } -# Read-only access to `memory/memory.db` and `memory_tree/chunks.db`. Bundled -# so the importer does not depend on a system SQLite. -rusqlite = { version = "0.40", features = ["bundled"] } -# `Checkpoint` is persisted by the host between runs. -serde = { version = "1", features = ["derive"] } -# Legacy rows carry JSON columns (tags, metadata, learning candidates, tool -# calls), and `Checkpoint` round-trips through JSON. -serde_json = "1" -# The crate-wide `Error`. -thiserror = "2" - -[dev-dependencies] -# Each test builds a throwaway v1 workspace in a temporary directory. -tempfile = "3" - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-import/src/error/mod.rs b/crates/tinymemory-import/src/error/mod.rs deleted file mode 100644 index dc9817de..00000000 --- a/crates/tinymemory-import/src/error/mod.rs +++ /dev/null @@ -1,41 +0,0 @@ -//! The crate-wide [`Error`] and [`Result`]. - -use std::path::PathBuf; - -/// Everything that can go wrong opening or reading a legacy workspace. -#[derive(Debug, thiserror::Error)] -#[non_exhaustive] -pub enum Error { - /// The path given to [`crate::LegacyWorkspace::open`] does not exist. - #[error("no legacy workspace at {}", path.display())] - NotFound { - /// The path that was looked up. - path: PathBuf, - }, - /// The path exists but is not a v1 TinyCortex workspace. - #[error("{} is not a v1 tinycortex workspace: {reason}", path.display())] - NotLegacy { - /// The path that was inspected. - path: PathBuf, - /// What was missing or wrong. - reason: String, - }, - /// A legacy SQLite database could not be read. - #[error("legacy sqlite read failed: {0}")] - Sqlite(#[from] rusqlite::Error), - /// A file referenced by the legacy store could not be read. - #[error("reading {} failed: {source}", path.display())] - Io { - /// The file that was read. - path: PathBuf, - /// The underlying error. - #[source] - source: std::io::Error, - }, - /// A [`crate::Checkpoint`] could not be encoded or decoded as JSON. - #[error("checkpoint json is invalid: {0}")] - Json(#[from] serde_json::Error), -} - -/// The crate-wide result. -pub type Result = std::result::Result; diff --git a/crates/tinymemory-integrations/Cargo.toml b/crates/tinymemory-integrations/Cargo.toml new file mode 100644 index 00000000..70327fd4 --- /dev/null +++ b/crates/tinymemory-integrations/Cargo.toml @@ -0,0 +1,145 @@ +[package] +name = "tinymemory-integrations" +description = "TinyMemory integrations: the CortexDB engine, document conversion, source readers, safety scrubbing and the legacy v1 import" +readme = "README.md" +keywords = ["memory", "agent", "llm", "retrieval"] +categories = ["database"] +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +publish.workspace = true + +[dependencies] +# The contract every integration produces or implements: `MemoryEngine`, +# `StoreItem`, `MemoryMeta` and the one `Error` each module's failures map to. +tinymemory-api = { path = "../tinymemory-api" } +# `MemoryEngine`, `BearerSource`, `DocumentConverter` and `SourceReader` are +# object-safe async traits. +async-trait = { version = "0.1", optional = true } +# Every wire body, envelope, cursor, Composio payload and checkpoint is JSON. +serde = { version = "1", features = ["derive"], optional = true } +serde_json = { version = "1", optional = true } +# The typed errors of `documents`, `sources` and `import`. +thiserror = { version = "2", optional = true } +# Diagnostics: a skipped source item, a dropped Slack message, a scrubbing +# decision. +log = { version = "0.4", optional = true } + +# --- cortex --- +# CortexDB speaks HTTP/JSON. `stream` is for `bytes_stream()`: response bodies +# are read against a byte cap rather than buffered whole, because the endpoint +# is operator-supplied and a broken or hostile one must not exhaust the host. +# The network source readers share the same client stack. +reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "stream"], optional = true } +# Timers for read-retry backoff and visibility polling (cortex); process +# spawning and DNS lookups for the GitHub reader and the SSRF resolver +# (sources-network). reqwest already requires a tokio runtime, so this adds no +# new runtime assumption. +tokio = { version = "1", default-features = false, features = ["time"], optional = true } +# `StreamExt` to read a capped body chunk by chunk. +futures = { version = "0.3", optional = true } +# Lookup labels are fixed-length SHA-256 digests of metadata values. +sha2 = { version = "0.10", optional = true } + +# --- documents-office --- +# Text out of the formats people actually drop into memory — a contract PDF, a +# spec `.docx`, a pricing `.xlsx`, a deck. All pure Rust with no system +# libraries. `pdf-extract` reads a PDF's text layer. `zip` + `quick-xml` are +# the whole of `.docx` and `.pptx`, which are zip archives of XML. `.xlsx` is +# not — shared-string tables and cell typing make hand-parsing it a liability — +# so `calamine` reads that one. `zip` and `quick-xml` are the versions +# `calamine` already links, so the graph carries one copy of each. +pdf-extract = { version = "0.12", optional = true } +calamine = { version = "0.36", optional = true } +quick-xml = { version = "0.41", optional = true } +zip = { version = "8", default-features = false, features = ["deflate"], optional = true } + +# --- sources --- +# `MemorySourceEntry` and friends appear in generated schemas, same as the +# contract crate's own types. +schemars = { version = "1.2", optional = true } +# The folder reader compiles a source's glob to a regex; safety's credential +# and PII patterns are regexes too. +regex = { version = "1.10", optional = true } +# The folder reader walks the directory tree. +walkdir = { version = "2", optional = true } +# Timestamps: `MemoryMeta::observed_at`, file mtimes, feed and issue dates, +# Gmail `Date:` headers. +chrono = { version = "0.4", features = ["clock", "serde"], optional = true } +# Diagnostics on the network readers and the Gmail normaliser. +tracing = { version = "0.1", optional = true } + +# --- legacy-import --- +# Read-only access to a v1 workspace's `memory/memory.db` and +# `memory_tree/chunks.db`. Bundled so the importer does not depend on a system +# SQLite. +rusqlite = { version = "0.40", features = ["bundled"], optional = true } + +[dev-dependencies] +# The behavioural suite every engine must pass, run over both CortexDB wires' +# doubles, and the reference engine the import driver is tested against. +tinymemory-api = { path = "../tinymemory-api", features = ["conformance"] } +# `tests/live_cortexdb.rs` compiles `context.md` from a live server. +tinymemory-tools = { path = "../tinymemory-tools" } +# The CortexDB and TinyHumans doubles are real HTTP servers on loopback. +axum = "0.8" +tokio = { version = "1", features = ["macros", "rt", "rt-multi-thread", "net", "io-util", "time"] } +# Reader and import tests build throwaway folders and v1 workspaces. +tempfile = "3" +# The format and office tests build their OOXML fixtures in code. +zip = { version = "8", default-features = false, features = ["deflate"] } +# `MemoryConfig` round-trips through the TOML a host stores it in. +toml = "1.1" + +[features] +# The CortexDB engine and the registry that builds it: what most hosts want. +default = ["cortex"] +# `cortex::CortexEngine` over both wires (`cortexdb` direct `/v1/*`, and +# `tinyhumans` behind the TinyHumans backend `/memory/*`), plus `registry` +# (`list_engines`, `build_engine`) and `config` (`MemoryConfig`). +cortex = ["dep:async-trait", "dep:serde", "dep:serde_json", "dep:reqwest", "dep:tokio", "dep:futures", "dep:sha2"] +# `documents`: format sniffing and conversion to markdown, emitting +# `StoreItem::Document`. Does no I/O. +documents = ["dep:async-trait", "dep:serde", "dep:serde_json", "dep:thiserror"] +# `documents::OfficeConverter`: PDF, DOCX, PPTX and XLSX to markdown, for a +# host to prepend to its `ConverterChain`. Off by default because a PDF parser +# and a spreadsheet reader are real weight. +documents-office = ["documents", "dep:pdf-extract", "dep:calamine", "dep:quick-xml", "dep:zip"] +# `sources`: readers that turn folders, files and conversations into +# `StoreItem`s, and the Composio payload normalisers. Links no HTTP stack. +sources = ["documents", "dep:schemars", "dep:regex", "dep:walkdir", "dep:chrono", "dep:log", "dep:tracing"] +# The readers that fetch over the network — GitHub, RSS, web pages — and +# `sources::fetch::fetch_url`, all behind the shared SSRF guard. +sources-network = ["sources", "dep:reqwest", "dep:futures", "dep:tokio", "tokio/process", "tokio/io-util", "tokio/net"] +# `safety`: secret and PII scrubbing for a `StoreItem` before it is stored. +safety = ["dep:regex", "dep:serde_json", "dep:log"] +# `import`: reads a legacy v1 (embedded TinyCortex) workspace and migrates it +# into any engine, resumably. +legacy-import = ["dep:rusqlite", "dep:serde", "dep:serde_json", "dep:thiserror"] +# Every integration. +full = ["cortex", "documents-office", "sources-network", "safety", "legacy-import"] + +[[example]] +name = "basic" +required-features = ["cortex"] + +[[test]] +name = "live_cortexdb" +required-features = ["cortex"] + +[[test]] +name = "office_live" +required-features = ["documents-office", "cortex"] + +[[test]] +name = "legacy_import" +required-features = ["legacy-import"] + +[[test]] +name = "reader_dispatch" +required-features = ["sources"] + +[lints] +workspace = true diff --git a/crates/tinymemory-integrations/README.md b/crates/tinymemory-integrations/README.md new file mode 100644 index 00000000..2f0f88ed --- /dev/null +++ b/crates/tinymemory-integrations/README.md @@ -0,0 +1,116 @@ +# tinymemory-integrations + +Everything in TinyMemory that touches the outside world, as one crate with a +Cargo feature per integration: the CortexDB engine and the registry that builds +it, document conversion, source readers, secret and PII scrubbing, and the +import of legacy v1 workspaces. The contract these integrations implement or +produce for lives in [`tinymemory-api`](../tinymemory-api/README.md); the +agent-facing tools are in [`tinymemory-tools`](../tinymemory-tools/README.md). + +The crate's only unconditional dependency is `tinymemory-api`. A host that +wants one integration enables one feature and links nothing else. + +## Modules and features + +| Module | Feature | What it does | Module README | +| --- | --- | --- | --- | +| `cortex`, `registry`, `config` | `cortex` (default) | `CortexEngine` over two wires (`cortexdb`, `tinyhumans`); `list_engines`, `build_engine`, `EngineCredential`; `MemoryConfig` | [`src/cortex/README.md`](src/cortex/README.md) | +| `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document`. No I/O. | [`src/documents/README.md`](src/documents/README.md) | +| `documents::OfficeConverter` | `documents-office` | PDF, DOCX, PPTX and XLSX to markdown, in process | (same) | +| `sources` | `sources` | Readers for folders, files and conversations; Composio payload normalisers; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | +| `sources::fetch`, GitHub, RSS and web-page readers | `sources-network` | The network readers and `fetch_url`, all behind the SSRF guard | (same) | +| `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` before it is stored | [`src/safety/README.md`](src/safety/README.md) | +| `import` | `legacy-import` | Reads a v1 (embedded TinyCortex) workspace and migrates it into any engine, resumably | [`src/import/README.md`](src/import/README.md) | + +`full` turns on `cortex`, `documents-office`, `sources-network`, `safety` and +`legacy-import`. Feature implications: `documents-office` implies `documents`; +`sources` implies `documents`; `sources-network` implies `sources`. + +## Dependency weight per feature + +What each feature adds to the dependency graph (on top of `tinymemory-api` and +the small `serde`, `serde_json`, `thiserror`, `async-trait` set the feature +already needs): + +| Feature | Adds | +| --- | --- | +| `cortex` | `reqwest` (rustls TLS, streaming bodies), `tokio` (`time` only), `futures`, `sha2` | +| `documents` | nothing beyond the small set above | +| `documents-office` | `pdf-extract`, `calamine`, `quick-xml`, `zip` (all pure Rust, no system libraries) | +| `sources` | `schemars`, `regex`, `walkdir`, `chrono`, `log`, `tracing` | +| `sources-network` | `reqwest`, `futures`, `tokio` with `process`, `io-util` and `net` (the GitHub reader runs `gh` and `git`; the SSRF resolver does DNS) | +| `safety` | `regex`, `serde_json`, `log` | +| `legacy-import` | `rusqlite` with bundled SQLite (compiles C; no system SQLite needed) | + +`documents-office` and `legacy-import` are the heavy ones, which is why neither +is on by default. + +## The write pipeline + +An item reaches an engine through up to four stages. Each stage is its own +module, none calls the next, and the host composes them; no engine scrubs or +converts on its own. + +```text +sources ──▶ documents ──▶ safety ──▶ engine.store +(read) (to markdown) (scrub) (cortex, or any MemoryEngine) +``` + +1. **sources** lists a configured source and reads each entry. Local readers + hand raw bytes to the converter; network readers fetch through the SSRF + guard. `sources::collect_items` drives one source and collects per-item + failures instead of aborting. +2. **documents** sniffs the format and converts bodies to markdown. A + `ConverterChain` decides which converter handles which format; a host can + put its own PDF or Office converter in front. +3. **safety** (`scrub_item`) redacts credentials and personal identifiers from + every free text the item carries. Metadata identifiers are left alone + because filters match on them. +4. **engine** is any `tinymemory_api::MemoryEngine`, normally the one + `build_engine` returns. + +Imports skip the first three stages: `import::migrate` produces items +directly from a v1 workspace and stores them in batches. + +## Example + +```rust,no_run +use std::sync::Arc; +use tinymemory_integrations::{EngineCredential, MemoryConfig, StaticBearer}; + +# async fn demo() -> tinymemory_integrations::Result<()> { +// Select the engine by configuration; the credential comes from the host's +// secret store, never from the config. +let config: MemoryConfig = toml::from_str(r#"engine = "tinyhumans""#).unwrap(); +let engine = config.build(EngineCredential::Dynamic(Arc::new(StaticBearer::new("tiny_live_..."))))?; +assert_eq!(engine.descriptor().id, "tinyhumans"); +# Ok(()) +# } +``` + +`cargo run -p tinymemory-integrations --example basic` lists the registered +engines and builds one without any network access. + +## Errors + +The crate-level `Error` is `tinymemory_api::Error`; `cortex` and the registry +return it directly. `documents`, `sources` and `import` each keep a typed +error (a path escaping its root, a non-v1 workspace are worth matching on), +and each converts into the contract error with `From`. + +## Tests + +Unit tests sit beside their modules in `mod_tests.rs` files. `tests/` holds +the integration tests: `reader_dispatch` (sources), `legacy_import`, +`documents_office`, `feature_surface` (every feature composing), and the live suites +`live_cortexdb` and `office_live`. The live suites skip themselves unless +`TINYMEMORY_LIVE_CORTEXDB_URL` names a CortexDB server; `scripts/cortexdb-live.sh` +boots the pinned harness in `integration/cortexdb/` and runs them against it. + +## Architecture + +[`docs/architecture/integrations.md`](../../docs/architecture/integrations.md) +describes documents, sources, safety and the legacy import in detail; +[`docs/architecture/cortex.md`](../../docs/architecture/cortex.md) covers the +engine and registry; [`docs/architecture/README.md`](../../docs/architecture/README.md) +indexes the rest. diff --git a/crates/tinymemory/examples/basic.rs b/crates/tinymemory-integrations/examples/basic.rs similarity index 87% rename from crates/tinymemory/examples/basic.rs rename to crates/tinymemory-integrations/examples/basic.rs index 4bdc23f0..de4e6813 100644 --- a/crates/tinymemory/examples/basic.rs +++ b/crates/tinymemory-integrations/examples/basic.rs @@ -3,7 +3,7 @@ //! Run with: //! //! ```sh -//! cargo run -p tinymemory --example basic +//! cargo run -p tinymemory-integrations --example basic //! ``` //! //! It needs no network: building an engine validates configuration and @@ -12,7 +12,7 @@ use std::sync::Arc; use async_trait::async_trait; -use tinymemory::{BearerSource, EngineCredential, MemoryConfig, list_engines}; +use tinymemory_integrations::{BearerSource, EngineCredential, MemoryConfig, list_engines}; /// A host's session store: the token is looked up on every request, so a /// refreshed session is picked up without rebuilding the engine. @@ -20,7 +20,7 @@ struct Session; #[async_trait] impl BearerSource for Session { - async fn bearer(&self) -> tinymemory::Result { + async fn bearer(&self) -> tinymemory_integrations::Result { Ok("session-jwt-from-the-host".to_string()) } } diff --git a/crates/tinymemory/src/config/mod.rs b/crates/tinymemory-integrations/src/config/mod.rs similarity index 73% rename from crates/tinymemory/src/config/mod.rs rename to crates/tinymemory-integrations/src/config/mod.rs index 4c8076a6..39620445 100644 --- a/crates/tinymemory/src/config/mod.rs +++ b/crates/tinymemory-integrations/src/config/mod.rs @@ -1,8 +1,22 @@ //! [`MemoryConfig`]: which engine a host uses and how each is reached. //! //! The config holds no credential. A host keeps its keys in its own secret -//! store and hands one to [`crate::build_engine`] as an -//! [`crate::EngineCredential`], so a config file can be shared or logged. +//! store and hands one to [`crate::registry::build_engine`] as an +//! [`crate::registry::EngineCredential`], so a config file can be shared or +//! logged. +//! +//! The same shape is read from TOML or JSON: +//! +//! ```toml +//! engine = "cortexdb" +//! +//! [engines.cortexdb] +//! endpoint = "https://cortex.example.com" +//! ``` +//! +//! `engines` is optional, an engine with no entry uses its defaults, and a +//! blank or absent `endpoint` means the engine's default endpoint. Unknown +//! fields are ignored when reading. use std::collections::BTreeMap; use std::sync::Arc; @@ -13,12 +27,12 @@ use tinymemory_api::{MemoryEngine, Result}; use crate::registry::{EngineCredential, build_engine}; /// The engine a fresh config selects. -pub const DEFAULT_ENGINE: &str = tinymemory_cortex::TINYHUMANS_ENGINE_ID; +pub const DEFAULT_ENGINE: &str = crate::cortex::TINYHUMANS_ENGINE_ID; /// Which engine a host uses, and per-engine settings. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct MemoryConfig { - /// The selected engine's id (see [`crate::list_engines`]). + /// The selected engine's id (see [`crate::registry::list_engines`]). pub engine: String, /// Settings per engine id. An engine with no entry uses its defaults. #[serde(default)] diff --git a/crates/tinymemory/src/config/mod_tests.rs b/crates/tinymemory-integrations/src/config/mod_tests.rs similarity index 100% rename from crates/tinymemory/src/config/mod_tests.rs rename to crates/tinymemory-integrations/src/config/mod_tests.rs diff --git a/crates/tinymemory-cortex/README.md b/crates/tinymemory-integrations/src/cortex/README.md similarity index 62% rename from crates/tinymemory-cortex/README.md rename to crates/tinymemory-integrations/src/cortex/README.md index 67ca5570..1dd95807 100644 --- a/crates/tinymemory-cortex/README.md +++ b/crates/tinymemory-integrations/src/cortex/README.md @@ -1,8 +1,9 @@ -# tinymemory-cortex +# cortex -The CortexDB memory engine for TinyMemory v2. One type, `CortexEngine`, -implements `tinymemory_api::MemoryEngine` over CortexDB's append-only event -log on two wires: +The CortexDB memory engine for TinyMemory v2, the `cortex` module of +`tinymemory-integrations` (feature `cortex`, on by default). One type, +`CortexEngine`, implements `tinymemory_api::MemoryEngine` over CortexDB's +append-only event log on two wires: | Engine id | Constructor | Wire | Auth | Default endpoint | | --- | --- | --- | --- | --- | @@ -14,15 +15,51 @@ accepts only `scope`, `query`, `budgets`, `view`, `include`, `temporal` and `filters`, with no keyword/vector switch. `Keyword` and `Vector` fail with `Error::Unsupported` before any request. +This README is the short in-tree summary. The full reference is under +[`docs/architecture/`](../../../../docs/architecture/): + +- [`cortex.md`](../../../../docs/architecture/cortex.md): surface, credentials, + transport, failure mapping, endpoint security, the registry and `MemoryConfig`; +- [`cortex-wire.md`](../../../../docs/architecture/cortex-wire.md): every endpoint + and its shapes, scope layout, the v2 envelope, lookup labels; +- [`cortex-flows.md`](../../../../docs/architecture/cortex-flows.md): step-by-step + store, list, fetch, recall, forget, get, discovery; +- [`testing.md`](../../../../docs/architecture/testing.md): the doubles, the + conformance suite and the live tests. + ## Public surface -- `CortexEngine::{new, direct, tinyhumans, with_request_timeout, wire}` +From `tinymemory_integrations::cortex`: + +- `CortexEngine::{new, direct, tinyhumans, wire}` (requests time out after 60s) - `CortexWire { Direct, TinyHumans }`, `CortexCredential { Static, Dynamic }` - `BearerSource` (async `bearer()`), `StaticBearer` (redacted `Debug`) - `CORTEXDB_ENGINE_ID`, `TINYHUMANS_ENGINE_ID`, `CORTEX_API_ENDPOINT`, `TINYHUMANS_API_ENDPOINT`, `cortexdb_descriptor()`, `tinyhumans_descriptor()` - `Error`/`Result` (the contract's own `tinymemory_api::Error`), - `error_code`, `is_insufficient_credits`, `INSUFFICIENT_CREDITS_CODE` + `error_code`, `is_insufficient_credits` + +A host usually goes through the registry instead of naming the engine: +`tinymemory_integrations::{MemoryConfig, EngineCredential, build_engine, +list_engines}` (modules `config` and `registry`). + +## Module layout + +```text +cortex/ +├── mod.rs crate-facing docs and the public re-exports +├── credential/ CortexCredential, BearerSource, StaticBearer +├── descriptor/ the two registrations, CortexWire and its route table +├── engine/ CortexEngine and one file per operation: +│ store, list, fetch, recall, forget, items (get), scopes, cursor +├── envelope/ the v2 event envelope, scope paths, lookup labels, rebuild +├── log/ the event log: write, read (list, scopes, recall, answer), +│ visibility waits, forget +├── transport/ HttpClient: timeouts, retries, byte caps, failure mapping, +│ the actor header +├── error/ the contract's Error, error_code, is_insufficient_credits +└── testing/ loopback doubles of both wires (cfg(test) only) +``` ## Storage layout @@ -39,8 +76,8 @@ The hosted backend also re-roots every scope under the caller's tenant. node and inherited ancestors are known; a subtree reach or an unscoped read discovers the nodes below from the registered scopes (`v1/scopes/list`, `memory/scopes`). Every read names its scopes exactly; server-side traversal -(`holistic`, `descend`) is used only for an unscoped multi-scope recall, so -one agent's read never reaches a sibling's scope. +(`view: "descend"`) is used only for an unscoped multi-scope recall, so one +agent's read never reaches a sibling's scope. Namespace segments use CortexDB's built-in `agent`, `team`, `user`, `ws` and `project` types, and the root and kind segments its `app` type. From v0.10 a @@ -85,40 +122,43 @@ as prefixes, so they cannot be labelled and are filtered only client-side. ## Operations -- **Store.** The item id is `StoreItem::fingerprint()`. The item's events are - looked up by its label first. If all of them are already there, the store is - a replay (`replayed: true`) and nothing is written. If only some turns of a - conversation are present (an earlier store failed part-way), only the - missing turns are written. Direct writes `v1/experience?wait=indexed`, or for - a conversation `v1/experience/bulk?wait=indexed` with `ordering: - strict_temporal`. Hosted writes one event at a time, in order. Every write - uses a fresh `idempotency_key`, never a content-derived one, because - CortexDB keeps a forgotten event's key and would swallow a re-store. The - write then waits for its last event to be readable (see below). +- **Store.** `store` is `store_items(vec![item])`, so a single store and + `store_many` share **one** path and one set of guarantees. The item id is + `StoreItem::fingerprint()`. Each scope's items are looked up by label first: + if all of an item's events are there, it is a replay (`replayed: true`) and + nothing is written; if only some turns of a conversation are present (an + earlier store failed part-way), only the missing turns are written. Direct + writes `v1/experience?wait=indexed`, or `v1/experience/bulk?wait=indexed` + with `ordering: strict_temporal` when an item has two or more events due. + Hosted writes one event at a time, in order. Every write uses a fresh + `idempotency_key`, never a content-derived one, because CortexDB keeps a + forgotten event's key and would swallow a re-store. Then one listing wait per + scope written (for its last event) and one ranked-recall wait (best-effort) + for the final event. - **List.** Pages the scopes read (kind order, then namespace), newest first. - The opaque cursor holds the scope's path (so a scope created between pages - cannot shift the listing), the engine cursor, the offset into that page and the - last event id, which is enough to drop the engine's duplicate copies across - page boundaries. A conversation is emitted once, on the page holding its - turn 0, with its text assembled from all its turns (one label lookup per - page). Scores are `0`. + The opaque cursor holds the scope's path, the engine cursor, the offset into + that page and the last event id, which is enough to drop the engine's + duplicate copies across page boundaries. A conversation is emitted once, on + the page holding its turn 0, with its text assembled from all its turns. + Scores are `0`. - **Fetch (hybrid).** One recall per scope read with `budgets.per_layer_limits.events`. Events are decoded to items and the full filter is applied. Each item is kept once, at its best rank, and scopes are interleaved rank by rank. The score is `1/(1+rank)`, because CortexDB - reports none. Conversation hits carry the whole conversation. The cursor is - an offset into the merged ranking; the next page asks again with a larger - budget, capped at 1000 events. + reports none. The cursor is an offset into the merged ranking; the next page + asks again with a larger budget, capped at 1000 events. - **Recall.** One scope read: one pack over it. An unscoped read over several scopes: one pack over `app:tinymemory` with `view: "descend"`. A reach over several scopes: one pack per scope (four at a time), exact, and the answer comes from the pack holding the most admitted events. The answer route is - called **once** with `use_pack_id`. Hosted omits a null `answer_instructions`, because its schema - is strict; Direct sends `null`. Citations come from the pack's - `layers.events`, decoded, filtered (reach included), one per item, the most - specific node's first, capped at `limit`, with - `score: None`. `model` is `diagnostics.answer_model`. A pack with no - decodable events still returns the answer, with no citations. + called **once** with `use_pack_id`. Hosted omits a null + `answer_instructions`, because its schema is strict; Direct sends `null`. + Citations come from the packs' decoded events, filtered (reach included), + one per item, the most specific node's first, capped at `limit`, with + `score: None`. `model` is `diagnostics.answer_model`. +- **Get.** Overridden: by the items' id labels, one lookup per scope read, + rather than a scan. +- **Explore.** Not overridden: the contract's default pages through `list`. - **Forget.** `Ids` looks the items' labels up in every scope the engine holds. `Filter` (which must be non-empty) walks the scopes it reads and matches the full filter. Either way the matched events are then removed with @@ -129,10 +169,10 @@ as prefixes, so they cannot be labelled and are filtered only client-side. and any other failure to `Down`. The reason keeps the message head and withholds the backend's own text. -## Engine behaviours this crate is shaped around +## Engine behaviours this module is shaped around These were measured against a live CortexDB by the v1 adapter. The doubles in -`src/testing/` reproduce all of them. +`testing/` reproduce all of them. - **Append-only.** There is no update route. Forget removes events but not their idempotency records. @@ -178,6 +218,8 @@ These were measured against a live CortexDB by the v1 adapter. The doubles in ## Tests -`cargo test -p tinymemory-cortex` runs the unit tests and the shared -`tinymemory-conformance` suite against both wires, through loopback doubles -with short test-only timeouts. +`cargo test -p tinymemory-integrations` runs the unit tests and the shared +`tinymemory_api::conformance` suite against both wires, through loopback +doubles with short test-only timeouts. `tests/live_cortexdb.rs` runs against +a real server when `TINYMEMORY_LIVE_CORTEXDB_URL` is set. See +[`testing.md`](../../../../docs/architecture/testing.md). diff --git a/crates/tinymemory-cortex/src/conformance_tests.rs b/crates/tinymemory-integrations/src/cortex/conformance_tests.rs similarity index 64% rename from crates/tinymemory-cortex/src/conformance_tests.rs rename to crates/tinymemory-integrations/src/cortex/conformance_tests.rs index fbb4fa2a..4b1870ef 100644 --- a/crates/tinymemory-cortex/src/conformance_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/conformance_tests.rs @@ -1,11 +1,11 @@ //! The shared conformance suite, run against both wires through the doubles. -use crate::testing::{direct_double, direct_engine, hosted_double, hosted_engine}; +use crate::cortex::testing::{direct_double, direct_engine, hosted_double, hosted_engine}; #[tokio::test] async fn the_direct_wire_upholds_the_contract() { let (endpoint, _state) = direct_double().await; - tinymemory_conformance::run(&direct_engine(&endpoint)) + tinymemory_api::conformance::run(&direct_engine(&endpoint)) .await .unwrap(); } @@ -13,7 +13,7 @@ async fn the_direct_wire_upholds_the_contract() { #[tokio::test] async fn the_tinyhumans_wire_upholds_the_contract() { let (endpoint, _state) = hosted_double().await; - tinymemory_conformance::run(&hosted_engine(&endpoint)) + tinymemory_api::conformance::run(&hosted_engine(&endpoint)) .await .unwrap(); } diff --git a/crates/tinymemory-cortex/src/credential/mod.rs b/crates/tinymemory-integrations/src/cortex/credential/mod.rs similarity index 96% rename from crates/tinymemory-cortex/src/credential/mod.rs rename to crates/tinymemory-integrations/src/cortex/credential/mod.rs index 2e2f3245..a528cd4a 100644 --- a/crates/tinymemory-cortex/src/credential/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/credential/mod.rs @@ -13,7 +13,7 @@ use std::sync::Arc; use async_trait::async_trait; -use crate::error::Result; +use crate::cortex::error::Result; /// A per-request source of bearer tokens. /// @@ -23,7 +23,7 @@ use crate::error::Result; /// /// Implementations must not log or otherwise print the token they return, and /// should return an error (not an empty string) when no credential is -/// available. The engine reports either as [`crate::Error::Unauthorized`] +/// available. The engine reports either as [`crate::cortex::Error::Unauthorized`] /// without sending a request, and never stores the value past the request. #[async_trait] pub trait BearerSource: Send + Sync { diff --git a/crates/tinymemory-cortex/src/credential/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/credential/mod_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/credential/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/credential/mod_tests.rs diff --git a/crates/tinymemory-cortex/src/descriptor/mod.rs b/crates/tinymemory-integrations/src/cortex/descriptor/mod.rs similarity index 100% rename from crates/tinymemory-cortex/src/descriptor/mod.rs rename to crates/tinymemory-integrations/src/cortex/descriptor/mod.rs diff --git a/crates/tinymemory-cortex/src/descriptor/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/descriptor/mod_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/descriptor/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/descriptor/mod_tests.rs diff --git a/crates/tinymemory-cortex/src/engine/cursor.rs b/crates/tinymemory-integrations/src/cortex/engine/cursor.rs similarity index 98% rename from crates/tinymemory-cortex/src/engine/cursor.rs rename to crates/tinymemory-integrations/src/cortex/engine/cursor.rs index 4ba5d94c..acba3de4 100644 --- a/crates/tinymemory-cortex/src/engine/cursor.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/cursor.rs @@ -7,7 +7,7 @@ use serde::{Deserialize, Serialize}; -use crate::error::{Error, Result}; +use crate::cortex::error::{Error, Result}; /// Where a listing stopped. #[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] diff --git a/crates/tinymemory-cortex/src/engine/cursor_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/cursor_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/engine/cursor_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/cursor_tests.rs diff --git a/crates/tinymemory-cortex/src/engine/engine_test_support.rs b/crates/tinymemory-integrations/src/cortex/engine/engine_test_support.rs similarity index 89% rename from crates/tinymemory-cortex/src/engine/engine_test_support.rs rename to crates/tinymemory-integrations/src/cortex/engine/engine_test_support.rs index a6dbc1ac..c9f57c6e 100644 --- a/crates/tinymemory-cortex/src/engine/engine_test_support.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/engine_test_support.rs @@ -7,7 +7,7 @@ use super::CortexEngine; impl CortexEngine { /// Shortens every wait and backoff, so a test reaches timeouts fast. pub(crate) fn with_test_timing(mut self, visibility: Duration) -> Self { - self.log.timing = crate::log::Timing { + self.log.timing = crate::cortex::log::Timing { visibility, settle: visibility, poll: Duration::from_millis(5), diff --git a/crates/tinymemory-cortex/src/engine/fetch.rs b/crates/tinymemory-integrations/src/cortex/engine/fetch.rs similarity index 98% rename from crates/tinymemory-cortex/src/engine/fetch.rs rename to crates/tinymemory-integrations/src/cortex/engine/fetch.rs index cc592c25..d24c27bc 100644 --- a/crates/tinymemory-cortex/src/engine/fetch.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/fetch.rs @@ -25,8 +25,8 @@ use tinymemory_api::{FetchPage, FetchRequest, Hit, ItemKind, MetaFilter}; use super::CortexEngine; use super::cursor::{self, FetchCursor}; use super::items::{hit, keeps}; -use crate::envelope::{Envelope, decode_event, labels, rebuild}; -use crate::error::Result; +use crate::cortex::envelope::{Envelope, decode_event, labels, rebuild}; +use crate::cortex::error::Result; /// The cursor tag of a fetch. const TAG: char = 'f'; diff --git a/crates/tinymemory-cortex/src/engine/fetch_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/engine/fetch_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs diff --git a/crates/tinymemory-cortex/src/engine/forget.rs b/crates/tinymemory-integrations/src/cortex/engine/forget.rs similarity index 97% rename from crates/tinymemory-cortex/src/engine/forget.rs rename to crates/tinymemory-integrations/src/cortex/engine/forget.rs index 6cc8e5f8..86b70339 100644 --- a/crates/tinymemory-cortex/src/engine/forget.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/forget.rs @@ -20,8 +20,8 @@ use tinymemory_api::{ForgetReport, ForgetTarget, MetaFilter}; use super::CortexEngine; use super::items::keeps; use super::scopes::KindScope; -use crate::envelope::{decode_event, labels}; -use crate::error::Result; +use crate::cortex::envelope::{decode_event, labels}; +use crate::cortex::error::Result; impl CortexEngine { /// See the module docs. diff --git a/crates/tinymemory-cortex/src/engine/items.rs b/crates/tinymemory-integrations/src/cortex/engine/items.rs similarity index 97% rename from crates/tinymemory-cortex/src/engine/items.rs rename to crates/tinymemory-integrations/src/cortex/engine/items.rs index 6ea6b268..b730ede7 100644 --- a/crates/tinymemory-cortex/src/engine/items.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/items.rs @@ -8,8 +8,8 @@ use tinymemory_api::{GetRequest, Hit, ItemId, ItemKind, MetaFilter, Namespace, S use super::CortexEngine; use super::scopes::KindScope; -use crate::envelope::{Decoded, Envelope, decode_event, labels, rebuild}; -use crate::error::Result; +use crate::cortex::envelope::{Decoded, Envelope, decode_event, labels, rebuild}; +use crate::cortex::error::Result; /// The kinds `filter` admits, in the fixed order /// [`ItemKind::ALL`] lists them. diff --git a/crates/tinymemory-cortex/src/engine/list.rs b/crates/tinymemory-integrations/src/cortex/engine/list.rs similarity index 97% rename from crates/tinymemory-cortex/src/engine/list.rs rename to crates/tinymemory-integrations/src/cortex/engine/list.rs index 467027a8..e7e6e25b 100644 --- a/crates/tinymemory-cortex/src/engine/list.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/list.rs @@ -29,9 +29,9 @@ use super::CortexEngine; use super::cursor::{self, ListCursor}; use super::items::{hit, keeps}; use super::scopes::KindScope; -use crate::envelope::{Envelope, decode_event, labels, parse_scope, rebuild}; -use crate::error::{Error, Result}; -use crate::log::{MAX_PAGES, PAGE_SIZE}; +use crate::cortex::envelope::{Envelope, decode_event, labels, parse_scope, rebuild}; +use crate::cortex::error::{Error, Result}; +use crate::cortex::log::{MAX_PAGES, PAGE_SIZE}; /// The cursor tag of a listing. const TAG: char = 'l'; diff --git a/crates/tinymemory-cortex/src/engine/mod.rs b/crates/tinymemory-integrations/src/cortex/engine/mod.rs similarity index 85% rename from crates/tinymemory-cortex/src/engine/mod.rs rename to crates/tinymemory-integrations/src/cortex/engine/mod.rs index 27a3cb17..dd66634e 100644 --- a/crates/tinymemory-cortex/src/engine/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/mod.rs @@ -3,8 +3,9 @@ //! //! Each operation lives in its own module: //! -//! - `store` — replay detection by item label, then one experience (or one -//! ordered batch of turns), then the readability wait; +//! - `store` — one path for `store` and `store_many`: replay detection by +//! item label, then each item's experience (or ordered batch of turns), +//! then the readability waits; //! - `list` — a cursor over the kind scopes' listings, each item once; //! - `fetch` — hybrid retrieval through recall packs, ranked by the engine; //! - `recall` — one pack, one answer, citations from the pack; @@ -20,7 +21,6 @@ mod scopes; mod store; use std::sync::Arc; -use std::time::Duration; use async_trait::async_trait; use tinymemory_api::{ @@ -29,11 +29,11 @@ use tinymemory_api::{ StoreReceipt, }; -use crate::credential::{BearerSource, CortexCredential}; -use crate::descriptor::{CortexWire, Route}; -use crate::error::{Error, Result}; -use crate::log::Log; -use crate::transport::{HttpClient, health_reason, urlencode}; +use crate::cortex::credential::{BearerSource, CortexCredential}; +use crate::cortex::descriptor::{CortexWire, Route}; +use crate::cortex::error::{Error, Result}; +use crate::cortex::log::Log; +use crate::cortex::transport::{HttpClient, health_reason, urlencode}; /// The scope prefix the hosted health probe lists under. The memory API /// refuses a prefix that is not `type:id` segments (a bare word is a 400, which @@ -77,7 +77,7 @@ impl CortexEngine { } /// CortexDB's own `/v1/*` API at `endpoint` (for example - /// [`crate::CORTEX_API_ENDPOINT`]), registered as `cortexdb`. + /// [`crate::cortex::CORTEX_API_ENDPOINT`]), registered as `cortexdb`. /// /// # Errors /// @@ -87,7 +87,7 @@ impl CortexEngine { } /// CortexDB behind the TinyHumans backend at `base_url` (for example - /// [`crate::TINYHUMANS_API_ENDPOINT`]), registered as `tinyhumans`. + /// [`crate::cortex::TINYHUMANS_API_ENDPOINT`]), registered as `tinyhumans`. /// `bearer` supplies the session JWT or `tiny_live_` API key and is /// consulted on every request, so a refreshed session is used at once. /// @@ -102,17 +102,6 @@ impl CortexEngine { ) } - /// Rebuilds the transport with a different per-request deadline (60s by - /// default). A retrying read can take about three times this. - /// - /// # Errors - /// - /// [`Error::Config`] if the HTTP client cannot be rebuilt. - pub fn with_request_timeout(mut self, timeout: Duration) -> Result { - self.log.client.set_timeout(timeout)?; - Ok(self) - } - /// Which HTTP surface this engine talks to. #[must_use] pub fn wire(&self) -> CortexWire { @@ -156,8 +145,12 @@ impl MemoryEngine for CortexEngine { self.fetch_page(req).await } + /// A batch of one (see `store`): listed and ranked on return. async fn store(&self, item: StoreItem) -> Result { - self.store_item(item).await + self.store_items(vec![item]) + .await? + .pop() + .ok_or_else(|| Error::Engine("a store of one item returned no receipt".to_string())) } /// Ranked recall is awaited for the last item only (see `store`). diff --git a/crates/tinymemory-cortex/src/engine/mod_direct_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/mod_direct_tests.rs similarity index 96% rename from crates/tinymemory-cortex/src/engine/mod_direct_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/mod_direct_tests.rs index cad733b6..34de9de2 100644 --- a/crates/tinymemory-cortex/src/engine/mod_direct_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/mod_direct_tests.rs @@ -2,7 +2,7 @@ //! waits, forget selectors, the retry split and status mapping. use super::*; -use crate::testing::{direct_double, direct_engine, sample_items, thread_meta}; +use crate::cortex::testing::{direct_double, direct_engine, sample_items, thread_meta}; use std::sync::atomic::Ordering; use tinymemory_api::{MetaFilter, Role, Turn}; @@ -188,7 +188,11 @@ async fn statuses_map_onto_the_contract() { .await .unwrap_err(); assert!(check(&error), "{code}: {error:?}"); - assert!(!error.to_string().contains(crate::testing::TEST_TOKEN)); + assert!( + !error + .to_string() + .contains(crate::cortex::testing::TEST_TOKEN) + ); } } diff --git a/crates/tinymemory-cortex/src/engine/mod_hosted_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/mod_hosted_tests.rs similarity index 97% rename from crates/tinymemory-cortex/src/engine/mod_hosted_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/mod_hosted_tests.rs index 2bd27265..8ac93e5c 100644 --- a/crates/tinymemory-cortex/src/engine/mod_hosted_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/mod_hosted_tests.rs @@ -3,9 +3,9 @@ //! recovery, and riding out rate limits. use super::*; -use crate::StaticBearer; -use crate::error::{error_code, is_insufficient_credits}; -use crate::testing::{hosted_double, hosted_engine, sample_items, serve}; +use crate::cortex::StaticBearer; +use crate::cortex::error::{error_code, is_insufficient_credits}; +use crate::cortex::testing::{hosted_double, hosted_engine, sample_items, serve}; use std::collections::HashSet; use std::sync::atomic::{AtomicUsize, Ordering}; use tinymemory_api::{FetchMode, MetaFilter}; @@ -161,7 +161,11 @@ async fn a_402_is_insufficient_credits_and_codes_survive() { *state.fail_all.lock().unwrap() = Some((402, "USER_INSUFFICIENT_CREDITS")); let error = engine.store(sample_items().remove(0)).await.unwrap_err(); assert!(is_insufficient_credits(&error), "{error:?}"); - assert!(!error.to_string().contains(crate::testing::TEST_TOKEN)); + assert!( + !error + .to_string() + .contains(crate::cortex::testing::TEST_TOKEN) + ); *state.fail_all.lock().unwrap() = Some((400, "VALIDATION_ERROR")); let error = engine diff --git a/crates/tinymemory-cortex/src/engine/mod_list_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/mod_list_tests.rs similarity index 95% rename from crates/tinymemory-cortex/src/engine/mod_list_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/mod_list_tests.rs index 9131db08..4ee1695f 100644 --- a/crates/tinymemory-cortex/src/engine/mod_list_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/mod_list_tests.rs @@ -2,7 +2,7 @@ //! conversation once, and the page ceiling. use super::*; -use crate::testing::{both, direct_engine, serve, thread_meta}; +use crate::cortex::testing::{both, direct_engine, serve, thread_meta}; use std::collections::HashSet; use tinymemory_api::{ItemKind, MetaFilter, Role, Turn}; @@ -42,7 +42,7 @@ async fn paging_returns_every_item_exactly_once_despite_duplicate_copies() { #[tokio::test] async fn a_cursor_crosses_from_one_kind_scope_to_the_next() { for (engine, _state) in both().await { - for item in crate::testing::sample_items() { + for item in crate::cortex::testing::sample_items() { engine.store(item).await.unwrap(); } let mut kinds = Vec::new(); @@ -92,7 +92,7 @@ async fn a_long_conversation_is_listed_once_with_every_turn() { #[tokio::test] async fn a_labelled_filter_narrows_server_side_and_is_rechecked() { for (engine, state) in both().await { - for item in crate::testing::sample_items() { + for item in crate::cortex::testing::sample_items() { engine.store(item).await.unwrap(); } let mut filter = MetaFilter { @@ -107,7 +107,7 @@ async fn a_labelled_filter_narrows_server_side_and_is_rechecked() { assert_eq!(page.items[0].kind, ItemKind::Learning); let thread_label = format!( "labels=tm%3At%3A{}", - crate::envelope::labels::digest("t-learn") + crate::cortex::envelope::labels::digest("t-learn") ); assert!( state.requests().iter().any(|r| r.contains(&thread_label)), @@ -124,7 +124,7 @@ async fn a_labelled_filter_narrows_server_side_and_is_rechecked() { #[tokio::test] async fn a_malformed_cursor_is_an_invalid_request() { - let (endpoint, _state) = crate::testing::direct_double().await; + let (endpoint, _state) = crate::cortex::testing::direct_double().await; let mut req = ListRequest::new(MetaFilter::default(), 3); req.cursor = Some("garbage".into()); let error = direct_engine(&endpoint).list(req).await.unwrap_err(); diff --git a/crates/tinymemory-cortex/src/engine/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/mod_tests.rs similarity index 96% rename from crates/tinymemory-cortex/src/engine/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/mod_tests.rs index 7264ad9e..9502c746 100644 --- a/crates/tinymemory-cortex/src/engine/mod_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/mod_tests.rs @@ -1,7 +1,7 @@ //! Round trips of every operation on both wires, through the doubles. use super::*; -use crate::testing::{both, direct_engine, sample_items as items}; +use crate::cortex::testing::{both, direct_engine, sample_items as items}; use std::sync::atomic::Ordering; use tinymemory_api::{FetchMode, ItemKind, MemoryMeta, MetaFilter}; @@ -225,7 +225,7 @@ async fn health_is_ok_degraded_or_down_with_a_redacted_reason() { }; assert!(reason.contains("withheld"), "{reason}"); assert!(!reason.contains("failed: UNAUTHORIZED"), "{reason}"); - assert!(!reason.contains(crate::testing::TEST_TOKEN)); + assert!(!reason.contains(crate::cortex::testing::TEST_TOKEN)); } } @@ -237,7 +237,7 @@ async fn a_pack_without_a_pack_id_or_answer_text_is_an_engine_error() { "/v1/recall", post(|| async { Json(serde_json::json!({ "layers": {} })) }), ); - let endpoint = crate::testing::serve(app).await; + let endpoint = crate::cortex::testing::serve(app).await; let error = direct_engine(&endpoint) .recall(RecallRequest::new("q", 1)) .await @@ -251,13 +251,11 @@ fn debug_names_the_engine_but_never_the_credential() { "https://db.example", CortexCredential::api_key("ctx_secret"), ) - .unwrap() - .with_request_timeout(Duration::from_secs(5)) .unwrap(); let rendered = format!("{engine:?}"); assert!(rendered.contains("cortexdb") && rendered.contains("db.example")); assert!(!rendered.contains("ctx_secret")); - assert_eq!(engine.descriptor().id, crate::CORTEXDB_ENGINE_ID); + assert_eq!(engine.descriptor().id, crate::cortex::CORTEXDB_ENGINE_ID); assert_eq!(engine.wire(), CortexWire::Direct); } diff --git a/crates/tinymemory-cortex/src/engine/recall.rs b/crates/tinymemory-integrations/src/cortex/engine/recall.rs similarity index 98% rename from crates/tinymemory-cortex/src/engine/recall.rs rename to crates/tinymemory-integrations/src/cortex/engine/recall.rs index b82a3bd5..75fa9b3c 100644 --- a/crates/tinymemory-cortex/src/engine/recall.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/recall.rs @@ -28,9 +28,9 @@ use tinymemory_api::{Citation, ItemId, RecallAnswer, RecallRequest}; use super::CortexEngine; use super::fetch::{ranked, recall_body}; use super::scopes::KindScope; -use crate::descriptor::CortexWire; -use crate::envelope::{Envelope, ROOT_SCOPE}; -use crate::error::{Error, Result}; +use crate::cortex::descriptor::CortexWire; +use crate::cortex::envelope::{Envelope, ROOT_SCOPE}; +use crate::cortex::error::{Error, Result}; /// Recall packs built at once when a reach spans several scopes. const PACKS_AT_ONCE: usize = 4; diff --git a/crates/tinymemory-cortex/src/engine/recall_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/recall_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/engine/recall_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/recall_tests.rs diff --git a/crates/tinymemory-cortex/src/engine/scopes.rs b/crates/tinymemory-integrations/src/cortex/engine/scopes.rs similarity index 97% rename from crates/tinymemory-cortex/src/engine/scopes.rs rename to crates/tinymemory-integrations/src/cortex/engine/scopes.rs index adf28e79..b9cba4ee 100644 --- a/crates/tinymemory-cortex/src/engine/scopes.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/scopes.rs @@ -23,8 +23,8 @@ use tinymemory_api::{ItemKind, MetaFilter, Namespace, Reach}; use super::CortexEngine; use super::items::admitted; -use crate::envelope::{ROOT_SCOPE, parse_scope, scope_path}; -use crate::error::Result; +use crate::cortex::envelope::{ROOT_SCOPE, parse_scope, scope_path}; +use crate::cortex::error::Result; /// One scope to read: a kind at a namespace node. #[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)] diff --git a/crates/tinymemory-cortex/src/engine/scopes_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/scopes_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/engine/scopes_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/scopes_tests.rs diff --git a/crates/tinymemory-cortex/src/engine/store.rs b/crates/tinymemory-integrations/src/cortex/engine/store.rs similarity index 72% rename from crates/tinymemory-cortex/src/engine/store.rs rename to crates/tinymemory-integrations/src/cortex/engine/store.rs index 2e18de12..3aa1f874 100644 --- a/crates/tinymemory-cortex/src/engine/store.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/store.rs @@ -1,6 +1,10 @@ -//! Store: replay detection, then the item's missing events, then the wait. +//! Store: replay detection, then the items' missing events, then the wait. //! -//! The item id is the item's fingerprint, so the engine first looks up the +//! `store` is `store_many` of one item: there is one path, so a single store +//! gets exactly the batch's guarantees (listed on return, and ranked recall +//! awaited for its final event). +//! +//! An item id is the item's fingerprint, so the engine first looks up the //! events already carrying that id's label in the item's scope (its kind at //! its namespace node): //! @@ -21,16 +25,11 @@ use tinymemory_api::{ItemId, StoreItem, StoreReceipt, validate_many}; use super::CortexEngine; use super::scopes::KindScope; -use crate::envelope::Envelope; -use crate::error::Result; -use crate::log::Written; +use crate::cortex::envelope::Envelope; +use crate::cortex::error::Result; +use crate::cortex::log::Written; impl CortexEngine { - /// See the module docs. - pub(super) async fn store_item(&self, item: StoreItem) -> Result { - self.store_one(item).await - } - /// `store_many`, paying per batch rather than per item: /// /// - one id lookup per scope (kind and namespace) finds what the batch @@ -95,37 +94,6 @@ impl CortexEngine { } Ok(receipts) } - - async fn store_one(&self, item: StoreItem) -> Result { - item.validate()?; - let id = item.fingerprint(); - let envelopes = Envelope::for_item(&item, &id)?; - let scope = KindScope::new(item.meta().namespace.clone(), item.kind()); - let held = self - .item_events(&scope, std::slice::from_ref(&id)) - .await? - .remove(&id) - .unwrap_or_default(); - let present: HashSet> = held - .iter() - .map(|decoded| decoded.envelope.turn.as_ref().map(|turn| turn.index)) - .collect(); - let mut requests = Vec::new(); - for envelope in &envelopes { - if present.contains(&envelope.turn.as_ref().map(|turn| turn.index)) { - continue; - } - requests.push(envelope.request(&envelope.encode()?)); - } - let replayed = requests.is_empty(); - if !replayed { - self.log.append(&requests).await?; - } - Ok(StoreReceipt { - id: ItemId::new(id), - replayed, - }) - } } #[cfg(test)] diff --git a/crates/tinymemory-cortex/src/engine/store_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/store_tests.rs similarity index 65% rename from crates/tinymemory-cortex/src/engine/store_tests.rs rename to crates/tinymemory-integrations/src/cortex/engine/store_tests.rs index 825f94b9..dbd23978 100644 --- a/crates/tinymemory-cortex/src/engine/store_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/store_tests.rs @@ -2,7 +2,7 @@ //! probed for the last item only. use super::*; -use crate::testing::both; +use crate::cortex::testing::both; use tinymemory_api::{ListRequest, MemoryEngine, MemoryMeta, MetaFilter}; fn doc(text: &str) -> StoreItem { @@ -56,7 +56,7 @@ async fn an_empty_or_invalid_batch_is_refused() { for (engine, _state) in both().await { assert!(matches!( engine.store_many(Vec::new()).await, - Err(crate::Error::InvalidRequest(_)) + Err(crate::cortex::Error::InvalidRequest(_)) )); assert!(engine.store_many(vec![doc(" ")]).await.is_err()); } @@ -91,3 +91,53 @@ async fn a_repeat_inside_a_batch_and_mixed_kinds_are_handled() { ); } } + +#[tokio::test] +async fn a_single_store_is_listed_and_settled_on_return_like_a_batch_of_one() { + for (engine, state) in both().await { + let wire = engine.wire(); + let receipt = engine.store(doc("single settled note")).await.unwrap(); + assert!(!receipt.replayed, "{wire:?}"); + assert_eq!( + receipt.id.as_str(), + doc("single settled note").fingerprint(), + "{wire:?}" + ); + let listed = engine + .list(ListRequest::new(MetaFilter::default(), 10)) + .await + .unwrap(); + assert_eq!(listed.items.len(), 1, "{wire:?}: listed on return"); + let probed = state + .seen + .lock() + .unwrap() + .recalls + .iter() + .filter(|body| { + body["query"] + .as_str() + .is_some_and(|q| q.contains("single settled note")) + }) + .count(); + assert!( + probed >= 1, + "{wire:?}: a single store waits for ranked recall" + ); + assert!( + engine + .store(doc("single settled note")) + .await + .unwrap() + .replayed, + "{wire:?}" + ); + assert!( + matches!( + engine.store(doc(" ")).await, + Err(crate::cortex::Error::InvalidRequest(_)) + ), + "{wire:?}" + ); + } +} diff --git a/crates/tinymemory-cortex/src/envelope/labels.rs b/crates/tinymemory-integrations/src/cortex/envelope/labels.rs similarity index 100% rename from crates/tinymemory-cortex/src/envelope/labels.rs rename to crates/tinymemory-integrations/src/cortex/envelope/labels.rs diff --git a/crates/tinymemory-cortex/src/envelope/labels_tests.rs b/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/envelope/labels_tests.rs rename to crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs diff --git a/crates/tinymemory-cortex/src/envelope/mod.rs b/crates/tinymemory-integrations/src/cortex/envelope/mod.rs similarity index 98% rename from crates/tinymemory-cortex/src/envelope/mod.rs rename to crates/tinymemory-integrations/src/cortex/envelope/mod.rs index 756ffce3..01aad4cc 100644 --- a/crates/tinymemory-cortex/src/envelope/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/envelope/mod.rs @@ -50,7 +50,7 @@ use tinymemory_api::{ DocumentBody, ItemKind, LearningKind, MemoryMeta, Namespace, Role, StoreItem, ToolCallRef, }; -use crate::error::{Error, Result}; +use crate::cortex::error::{Error, Result}; pub(crate) use rebuild::{Decoded, decode_event, rebuild}; @@ -274,7 +274,7 @@ impl Envelope { json!({ "scope": scope_path(&self.meta.namespace, self.kind), "modality": modality, - "idempotency_key": crate::transport::fresh_idempotency_key(), + "idempotency_key": crate::cortex::transport::fresh_idempotency_key(), "content": { "kind": "message", "role": role, "text": text }, "context": Value::Object(context), }) diff --git a/crates/tinymemory-cortex/src/envelope/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/envelope/mod_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/envelope/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/envelope/mod_tests.rs diff --git a/crates/tinymemory-cortex/src/envelope/rebuild.rs b/crates/tinymemory-integrations/src/cortex/envelope/rebuild.rs similarity index 100% rename from crates/tinymemory-cortex/src/envelope/rebuild.rs rename to crates/tinymemory-integrations/src/cortex/envelope/rebuild.rs diff --git a/crates/tinymemory-cortex/src/error/mod.rs b/crates/tinymemory-integrations/src/cortex/error/mod.rs similarity index 92% rename from crates/tinymemory-cortex/src/error/mod.rs rename to crates/tinymemory-integrations/src/cortex/error/mod.rs index 4f1f8341..d3504dfa 100644 --- a/crates/tinymemory-cortex/src/error/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/error/mod.rs @@ -1,10 +1,10 @@ //! The engine's errors are the contract's errors. //! -//! This crate does not define a parallel `Error`. Every public operation is a +//! This module does not define a parallel `Error`. Every public operation is a //! [`tinymemory_api::MemoryEngine`] method, and those return //! [`tinymemory_api::Error`]; a second enum would only be converted into it at //! every boundary and would invite variants the host cannot act on. So the -//! contract's enum is re-exported here as the crate-wide [`Error`], and +//! contract's enum is re-exported here as [`Error`], and //! construction and configuration failures use [`Error::Config`]. //! //! # How a CortexDB failure is classified @@ -42,11 +42,11 @@ pub use tinymemory_api::Error; -/// The crate-wide result alias. +/// The result alias for this module's operations. pub type Result = std::result::Result; /// The TinyHumans backend's code for an exhausted credit balance (HTTP 402). -pub const INSUFFICIENT_CREDITS_CODE: &str = "USER_INSUFFICIENT_CREDITS"; +pub(crate) const INSUFFICIENT_CREDITS_CODE: &str = "USER_INSUFFICIENT_CREDITS"; /// The message every variant carries. fn message_of(error: &Error) -> &str { diff --git a/crates/tinymemory-cortex/src/error/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/error/mod_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/error/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/error/mod_tests.rs diff --git a/crates/tinymemory-cortex/src/log/forget.rs b/crates/tinymemory-integrations/src/cortex/log/forget.rs similarity index 94% rename from crates/tinymemory-cortex/src/log/forget.rs rename to crates/tinymemory-integrations/src/cortex/log/forget.rs index 5e12e65d..0f46a818 100644 --- a/crates/tinymemory-cortex/src/log/forget.rs +++ b/crates/tinymemory-integrations/src/cortex/log/forget.rs @@ -10,9 +10,9 @@ use serde_json::json; use super::Log; -use crate::descriptor::{CortexWire, Route}; -use crate::error::{Error, Result}; -use crate::transport::Attempts; +use crate::cortex::descriptor::{CortexWire, Route}; +use crate::cortex::error::{Error, Result}; +use crate::cortex::transport::Attempts; /// The most event ids one removal names, so each body stays small. pub(crate) const FORGET_BATCH: usize = 100; diff --git a/crates/tinymemory-cortex/src/log/mod.rs b/crates/tinymemory-integrations/src/cortex/log/mod.rs similarity index 95% rename from crates/tinymemory-cortex/src/log/mod.rs rename to crates/tinymemory-integrations/src/cortex/log/mod.rs index ba6ba870..e3304608 100644 --- a/crates/tinymemory-cortex/src/log/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/log/mod.rs @@ -41,7 +41,7 @@ pub(crate) use write::Written; use std::time::Duration; -use crate::transport::HttpClient; +use crate::cortex::transport::HttpClient; /// Events one listing page asks for. `limit` counts the engine's duplicate /// copies, so a page holds about half as many distinct events. @@ -100,8 +100,8 @@ impl Log { /// The next poll gap after `current`. fn next_poll(&self, current: Duration) -> Duration { match self.client.wire() { - crate::CortexWire::Direct => current, - crate::CortexWire::TinyHumans => (current * 2).min(HOSTED_POLL_CEILING), + crate::cortex::CortexWire::Direct => current, + crate::cortex::CortexWire::TinyHumans => (current * 2).min(HOSTED_POLL_CEILING), } } } diff --git a/crates/tinymemory-cortex/src/log/read.rs b/crates/tinymemory-integrations/src/cortex/log/read.rs similarity index 98% rename from crates/tinymemory-cortex/src/log/read.rs rename to crates/tinymemory-integrations/src/cortex/log/read.rs index 215a315b..43e759d0 100644 --- a/crates/tinymemory-cortex/src/log/read.rs +++ b/crates/tinymemory-integrations/src/cortex/log/read.rs @@ -6,9 +6,9 @@ use reqwest::Method; use serde_json::Value; use super::{Log, MAX_PAGES, PAGE_SIZE}; -use crate::descriptor::Route; -use crate::error::{Error, Result}; -use crate::transport::{Attempts, urlencode}; +use crate::cortex::descriptor::Route; +use crate::cortex::error::{Error, Result}; +use crate::cortex::transport::{Attempts, urlencode}; /// Most labels one listing names. Labels share one comma-separated /// parameter (the hosted backend refuses a repeated `labels=`), so this diff --git a/crates/tinymemory-cortex/src/log/visibility.rs b/crates/tinymemory-integrations/src/cortex/log/visibility.rs similarity index 98% rename from crates/tinymemory-cortex/src/log/visibility.rs rename to crates/tinymemory-integrations/src/cortex/log/visibility.rs index fa2e59f1..a8512e11 100644 --- a/crates/tinymemory-cortex/src/log/visibility.rs +++ b/crates/tinymemory-integrations/src/cortex/log/visibility.rs @@ -20,8 +20,8 @@ use serde_json::{Value, json}; use super::{Log, PAGE_SIZE}; -use crate::descriptor::CortexWire; -use crate::error::{Error, Result}; +use crate::cortex::descriptor::CortexWire; +use crate::cortex::error::{Error, Result}; /// Longest query the settle probe sends: a distinctive prefix of the stored /// text matches better, and costs less, than a 64 KiB document. diff --git a/crates/tinymemory-cortex/src/log/write.rs b/crates/tinymemory-integrations/src/cortex/log/write.rs similarity index 90% rename from crates/tinymemory-cortex/src/log/write.rs rename to crates/tinymemory-integrations/src/cortex/log/write.rs index 12e48a98..4d5f0dbc 100644 --- a/crates/tinymemory-cortex/src/log/write.rs +++ b/crates/tinymemory-integrations/src/cortex/log/write.rs @@ -29,9 +29,9 @@ use serde_json::{Value, json}; use super::{Log, PAGE_SIZE}; -use crate::descriptor::{CortexWire, Route}; -use crate::error::{Error, Result}; -use crate::transport::{Attempts, fresh_idempotency_key}; +use crate::cortex::descriptor::{CortexWire, Route}; +use crate::cortex::error::{Error, Result}; +use crate::cortex::transport::{Attempts, fresh_idempotency_key}; /// The last event of a write: what a wait for it needs. #[derive(Debug, Clone)] @@ -68,21 +68,11 @@ fn receipt(answer: &Value) -> Result { } impl Log { - /// Appends `requests` (one item's events, in order) and waits until the - /// last is readable. The log is ordered, so the last event being listed - /// implies the earlier ones are: one wait, not one per event. A bulk - /// store uses [`Log::write`] and [`Log::await_written`] instead, to wait - /// once per scope for a whole batch. - pub(crate) async fn append(&self, requests: &[Value]) -> Result<()> { - match self.write(requests).await? { - Some(written) => self.await_written(&written, true).await, - None => Ok(()), - } - } - /// Writes `requests` (one item's events, in order) without waiting, and - /// names the last event so a caller can wait for it, or for a later one - /// in the same scope, which implies it. + /// names the last event so a caller can wait for it with + /// [`Log::await_written`], or for a later one in the same scope, which + /// implies it: the log is ordered, so the last event being listed implies + /// the earlier ones are. pub(crate) async fn write(&self, requests: &[Value]) -> Result> { let Some(last) = requests.last() else { return Ok(None); diff --git a/crates/tinymemory-cortex/src/lib.rs b/crates/tinymemory-integrations/src/cortex/mod.rs similarity index 74% rename from crates/tinymemory-cortex/src/lib.rs rename to crates/tinymemory-integrations/src/cortex/mod.rs index d61b1fe1..f0a2aa79 100644 --- a/crates/tinymemory-cortex/src/lib.rs +++ b/crates/tinymemory-integrations/src/cortex/mod.rs @@ -12,6 +12,11 @@ //! Both declare [`tinymemory_api::FetchMode::Hybrid`] only: CortexDB's //! recall body has no keyword/vector switch (see [`cortexdb_descriptor`]). //! +//! Hosts normally build an engine through [`crate::registry::build_engine`] or +//! [`crate::config::MemoryConfig::build`] rather than naming +//! [`CortexEngine`]. Errors are the contract's [`tinymemory_api::Error`]; see +//! [`error_code`] and [`is_insufficient_credits`] for hosted failures. +//! //! # Storage layout //! //! Items live in one scope per kind under the TinyMemory root: @@ -22,20 +27,21 @@ //! kind, text and full metadata, and each event carries lookup labels (digests //! of the item id and of the exact-match metadata fields) so reads can narrow //! server-side before the full [`tinymemory_api::MetaFilter`] is applied -//! client-side. The crate's `README.md` describes the layout and every engine -//! behaviour it is shaped around. +//! client-side. This module's `README.md` summarises the layout and every +//! engine behaviour it is shaped around; `docs/architecture/cortex.md`, +//! `cortex-wire.md` and `cortex-flows.md` give the full reference. //! //! # Example //! //! ```no_run //! use std::sync::Arc; //! use tinymemory_api::{MemoryEngine, MemoryMeta, SourceKind, StoreItem}; -//! use tinymemory_cortex::{CortexCredential, CortexEngine, StaticBearer, CORTEX_API_ENDPOINT}; +//! use tinymemory_integrations::cortex::{CortexCredential, CortexEngine, StaticBearer, CORTEX_API_ENDPOINT}; //! -//! # async fn demo() -> tinymemory_cortex::Result<()> { +//! # async fn demo() -> tinymemory_integrations::cortex::Result<()> { //! let direct = CortexEngine::direct(CORTEX_API_ENDPOINT, CortexCredential::api_key("ctx_..."))?; //! let hosted = CortexEngine::tinyhumans( -//! tinymemory_cortex::TINYHUMANS_API_ENDPOINT, +//! tinymemory_integrations::cortex::TINYHUMANS_API_ENDPOINT, //! Arc::new(StaticBearer::new("tiny_live_...")), //! )?; //! @@ -68,4 +74,4 @@ pub use descriptor::{ TINYHUMANS_ENGINE_ID, cortexdb_descriptor, tinyhumans_descriptor, }; pub use engine::CortexEngine; -pub use error::{Error, INSUFFICIENT_CREDITS_CODE, Result, error_code, is_insufficient_credits}; +pub use error::{Error, Result, error_code, is_insufficient_credits}; diff --git a/crates/tinymemory-cortex/src/testing/log.rs b/crates/tinymemory-integrations/src/cortex/testing/log.rs similarity index 100% rename from crates/tinymemory-cortex/src/testing/log.rs rename to crates/tinymemory-integrations/src/cortex/testing/log.rs diff --git a/crates/tinymemory-cortex/src/testing/mod.rs b/crates/tinymemory-integrations/src/cortex/testing/mod.rs similarity index 99% rename from crates/tinymemory-cortex/src/testing/mod.rs rename to crates/tinymemory-integrations/src/cortex/testing/mod.rs index afe1ae8e..4c38586d 100644 --- a/crates/tinymemory-cortex/src/testing/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/testing/mod.rs @@ -22,7 +22,7 @@ pub(crate) use log::CortexLog; use tinymemory_api::{LearningKind, MemoryMeta, Role, SourceKind, StoreItem, Turn}; -use crate::{CortexCredential, CortexEngine, StaticBearer}; +use crate::cortex::{CortexCredential, CortexEngine, StaticBearer}; /// The bearer the test engines send. pub(crate) const TEST_TOKEN: &str = "tiny_live_test"; diff --git a/crates/tinymemory-cortex/src/testing/routes.rs b/crates/tinymemory-integrations/src/cortex/testing/routes.rs similarity index 100% rename from crates/tinymemory-cortex/src/testing/routes.rs rename to crates/tinymemory-integrations/src/cortex/testing/routes.rs diff --git a/crates/tinymemory-cortex/src/transport/actor.rs b/crates/tinymemory-integrations/src/cortex/transport/actor.rs similarity index 100% rename from crates/tinymemory-cortex/src/transport/actor.rs rename to crates/tinymemory-integrations/src/cortex/transport/actor.rs diff --git a/crates/tinymemory-cortex/src/transport/actor_tests.rs b/crates/tinymemory-integrations/src/cortex/transport/actor_tests.rs similarity index 100% rename from crates/tinymemory-cortex/src/transport/actor_tests.rs rename to crates/tinymemory-integrations/src/cortex/transport/actor_tests.rs diff --git a/crates/tinymemory-cortex/src/transport/body.rs b/crates/tinymemory-integrations/src/cortex/transport/body.rs similarity index 98% rename from crates/tinymemory-cortex/src/transport/body.rs rename to crates/tinymemory-integrations/src/cortex/transport/body.rs index b9315f02..2806dcee 100644 --- a/crates/tinymemory-cortex/src/transport/body.rs +++ b/crates/tinymemory-integrations/src/cortex/transport/body.rs @@ -7,7 +7,7 @@ use futures::StreamExt; -use crate::error::{Error, Result}; +use crate::cortex::error::{Error, Result}; /// Largest success body accepted. Far above any real page of events, far /// below a size that threatens a process. diff --git a/crates/tinymemory-cortex/src/transport/failure.rs b/crates/tinymemory-integrations/src/cortex/transport/failure.rs similarity index 99% rename from crates/tinymemory-cortex/src/transport/failure.rs rename to crates/tinymemory-integrations/src/cortex/transport/failure.rs index 58e8a0e9..b4461fd8 100644 --- a/crates/tinymemory-cortex/src/transport/failure.rs +++ b/crates/tinymemory-integrations/src/cortex/transport/failure.rs @@ -8,7 +8,7 @@ use reqwest::StatusCode; use serde_json::Value; -use crate::error::{Error, INSUFFICIENT_CREDITS_CODE}; +use crate::cortex::error::{Error, INSUFFICIENT_CREDITS_CODE}; /// Longest excerpt of a backend's error text kept in a message. const MAX_DETAIL_CHARS: usize = 300; diff --git a/crates/tinymemory-cortex/src/transport/failure_tests.rs b/crates/tinymemory-integrations/src/cortex/transport/failure_tests.rs similarity index 98% rename from crates/tinymemory-cortex/src/transport/failure_tests.rs rename to crates/tinymemory-integrations/src/cortex/transport/failure_tests.rs index 11701c36..cf03520c 100644 --- a/crates/tinymemory-cortex/src/transport/failure_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/transport/failure_tests.rs @@ -2,7 +2,7 @@ //! envelope. use super::*; -use crate::error::{error_code, is_insufficient_credits}; +use crate::cortex::error::{error_code, is_insufficient_credits}; #[test] fn a_rustls_handshake_abort_is_named_tls_not_connect() { diff --git a/crates/tinymemory-cortex/src/transport/mod.rs b/crates/tinymemory-integrations/src/cortex/transport/mod.rs similarity index 96% rename from crates/tinymemory-cortex/src/transport/mod.rs rename to crates/tinymemory-integrations/src/cortex/transport/mod.rs index 335ecaeb..6d6d61fd 100644 --- a/crates/tinymemory-cortex/src/transport/mod.rs +++ b/crates/tinymemory-integrations/src/cortex/transport/mod.rs @@ -7,7 +7,7 @@ //! //! **Reads retry; writes do not.** Every call states its [`Attempts`]. A read //! (listing, recall) is retried up to three times with 250ms·2ⁿ backoff on -//! [`crate::Error::Unavailable`]; a write is sent once, because a timeout on a +//! [`crate::cortex::Error::Unavailable`]; a write is sent once, because a timeout on a //! write leaves whether it applied unknown. Hosted writes recover from that //! one level up, with an `Idempotency-Key` claim (see `log::write`). @@ -21,9 +21,9 @@ use reqwest::header::{AUTHORIZATION, HeaderValue}; use reqwest::{Method, RequestBuilder, Url}; use serde_json::Value; -use crate::credential::CortexCredential; -use crate::descriptor::CortexWire; -use crate::error::{Error, Result}; +use crate::cortex::credential::CortexCredential; +use crate::cortex::descriptor::CortexWire; +use crate::cortex::error::{Error, Result}; pub(crate) use failure::health_reason; @@ -132,13 +132,6 @@ impl HttpClient { }) } - /// Rebuilds the client with a different per-request deadline. A retrying - /// read can take about three times this plus 750ms of backoff. - pub(crate) fn set_timeout(&mut self, timeout: Duration) -> Result<()> { - self.inner = build_inner(timeout)?; - Ok(()) - } - /// The wire this client speaks. pub(crate) fn wire(&self) -> CortexWire { self.wire diff --git a/crates/tinymemory-cortex/src/transport/mod_tests.rs b/crates/tinymemory-integrations/src/cortex/transport/mod_tests.rs similarity index 99% rename from crates/tinymemory-cortex/src/transport/mod_tests.rs rename to crates/tinymemory-integrations/src/cortex/transport/mod_tests.rs index 8b1248ae..f46b8377 100644 --- a/crates/tinymemory-cortex/src/transport/mod_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/transport/mod_tests.rs @@ -6,7 +6,7 @@ use super::*; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; -use crate::testing::serve; +use crate::cortex::testing::serve; use axum::Router; use axum::http::StatusCode; use axum::routing::any; diff --git a/crates/tinymemory-cortex/src/transport/transport_test_support.rs b/crates/tinymemory-integrations/src/cortex/transport/transport_test_support.rs similarity index 100% rename from crates/tinymemory-cortex/src/transport/transport_test_support.rs rename to crates/tinymemory-integrations/src/cortex/transport/transport_test_support.rs diff --git a/crates/tinymemory-documents/README.md b/crates/tinymemory-integrations/src/documents/README.md similarity index 83% rename from crates/tinymemory-documents/README.md rename to crates/tinymemory-integrations/src/documents/README.md index c99868ec..56ed00c1 100644 --- a/crates/tinymemory-documents/README.md +++ b/crates/tinymemory-integrations/src/documents/README.md @@ -1,11 +1,13 @@ -# tinymemory-documents +# documents -Document intake for TinyMemory: work out what a file is, turn it into markdown, -and wrap it as the `StoreItem::Document` an engine stores. +Document intake for TinyMemory, the `documents` module of +`tinymemory-integrations` (feature `documents`): work out what a file is, turn +it into markdown, and wrap it as the `StoreItem::Document` an engine stores. -This crate does no I/O. Reading files and fetching URLs belongs to -`tinymemory-sources`, which depends on this crate for conversion and language -detection. +This module does no I/O. Reading files and fetching URLs belongs to the +[`sources`](../sources/README.md) module, which depends on this one for +conversion and language detection. Architecture overview: +[`docs/architecture/integrations.md`](../../../../docs/architecture/integrations.md). ## Three decisions @@ -39,12 +41,12 @@ caller's to state. | `ConvertedDocument` | markdown plus title, source format, language, and converter metadata | | `DocumentConverter` | the conversion seam — object-safe and async | | `NativeConverter` | markdown, text, HTML and code, with no dependencies | -| `OfficeConverter` | PDF, DOCX, PPTX and XLSX, in-process (feature `office`) | +| `OfficeConverter` | PDF, DOCX, PPTX and XLSX, in-process (feature `documents-office`) | | `ConverterChain` | converters in priority order; first claim wins | | `document_item` / `converted_item` | the conversion wrapped as a `StoreItem::Document` | | `markdown_from_text` | the synchronous core, for callers that already hold text | | `html::to_markdown` | the structural HTML converter, usable on its own | -| `Error` / `Result` | the crate error: `Invalid`, `TooLarge`, `UnsupportedFormat`, `Converter` | +| `Error` / `Result` | the module error: `Invalid`, `TooLarge`, `UnsupportedFormat`, `Converter` | ## The item @@ -65,7 +67,7 @@ conversion is a trait a host binds: let chain = ConverterChain::default().prepend(Box::new(MyPdfConverter)); ``` -The `office` feature ships one such binding, `OfficeConverter`: PDF (text +The `documents-office` feature ships one such binding, `OfficeConverter`: PDF (text layer only — a scanned PDF is refused as having no text), DOCX, PPTX (slides in numeric order) and XLSX (one `sheet | cell | cell` line per row), all pure Rust. It refuses hostile input rather than allocating for it: an archive whose @@ -97,6 +99,8 @@ success. ## Features -- `office` — `OfficeConverter` (`pdf-extract`, `calamine`, `zip`, +- `documents` — this module; depends only on `async-trait`, `serde`, + `serde_json` and `thiserror`. +- `documents-office` — `OfficeConverter` (`pdf-extract`, `calamine`, `zip`, `quick-xml`). Off by default; it links a PDF parser and a spreadsheet reader a text-only host has no use for. diff --git a/crates/tinymemory-documents/src/convert/mod.rs b/crates/tinymemory-integrations/src/documents/convert/mod.rs similarity index 95% rename from crates/tinymemory-documents/src/convert/mod.rs rename to crates/tinymemory-integrations/src/documents/convert/mod.rs index 7a555923..35cf66a7 100644 --- a/crates/tinymemory-documents/src/convert/mod.rs +++ b/crates/tinymemory-integrations/src/documents/convert/mod.rs @@ -7,13 +7,13 @@ //! //! ## Why this is a trait //! -//! Text, markdown and HTML convert with no dependencies, and this crate does +//! Text, markdown and HTML convert with no dependencies, and this module does //! them ([`NativeConverter`]). PDF and the Office formats do not: they need a //! real extractor, and which extractor a deployment uses is its own decision — //! an in-process crate, a TinyBus module, a service. So conversion is a trait a //! host binds rather than a fixed table, and [`ConverterChain`] composes the -//! native converter with whatever the host brings — including this crate's own -//! `OfficeConverter` when the `office` feature is on. +//! native converter with whatever the host brings — including this module's own +//! `OfficeConverter` when the `documents-office` feature is on. //! //! Source code is textual too, and [`NativeConverter`] stores it exactly as //! written: reflowing it or running it through the HTML converter would change @@ -27,10 +27,10 @@ mod types; use async_trait::async_trait; -use crate::error::{Error, Result}; -use crate::format::DocumentFormat; -use crate::html; -use crate::language::language_for_path; +use crate::documents::error::{Error, Result}; +use crate::documents::format::DocumentFormat; +use crate::documents::html; +use crate::documents::language::language_for_path; pub use types::{ConvertedDocument, MAX_DOCUMENT_BYTES, RawDocument}; @@ -99,11 +99,11 @@ pub fn markdown_from_text(text: &str, format: DocumentFormat) -> String { } } -/// The formats this crate converts without help: markdown, plain text, HTML +/// The formats this module converts without help: markdown, plain text, HTML /// and source code. /// /// Everything it handles is already text, so the whole implementation is -/// decoding plus, for HTML, [`crate::html::to_markdown`]. PDF and the Office +/// decoding plus, for HTML, [`crate::documents::html::to_markdown`]. PDF and the Office /// formats are deliberately absent — see the module docs. #[derive(Debug, Default, Clone, Copy)] pub struct NativeConverter; diff --git a/crates/tinymemory-documents/src/convert/mod_tests.rs b/crates/tinymemory-integrations/src/documents/convert/mod_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/convert/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/convert/mod_tests.rs diff --git a/crates/tinymemory-documents/src/convert/types.rs b/crates/tinymemory-integrations/src/documents/convert/types.rs similarity index 96% rename from crates/tinymemory-documents/src/convert/types.rs rename to crates/tinymemory-integrations/src/documents/convert/types.rs index 7825a228..a65557fd 100644 --- a/crates/tinymemory-documents/src/convert/types.rs +++ b/crates/tinymemory-integrations/src/documents/convert/types.rs @@ -3,7 +3,7 @@ use serde::{Deserialize, Serialize}; -use crate::format::DocumentFormat; +use crate::documents::format::DocumentFormat; /// Largest document intake will accept, in bytes. /// @@ -95,13 +95,13 @@ pub struct ConvertedDocument { /// Format the source was detected as. pub format: DocumentFormat, /// Programming language, for [`DocumentFormat::Code`] whose filename named - /// one (see [`crate::language_for_path`]). + /// one (see [`crate::documents::language_for_path`]). #[serde(default, skip_serializing_if = "Option::is_none")] pub language: Option, /// Size of the source document in bytes, before conversion. pub source_bytes: usize, /// Anything else the converter learned — page counts, author, the - /// converter's own name. Open on purpose: this crate cannot know what a + /// converter's own name. Open on purpose: this module cannot know what a /// host's converter will find worth keeping. #[serde(default)] pub metadata: serde_json::Value, diff --git a/crates/tinymemory-documents/src/error/mod.rs b/crates/tinymemory-integrations/src/documents/error/mod.rs similarity index 88% rename from crates/tinymemory-documents/src/error/mod.rs rename to crates/tinymemory-integrations/src/documents/error/mod.rs index 9537b87c..7559aa1c 100644 --- a/crates/tinymemory-documents/src/error/mod.rs +++ b/crates/tinymemory-integrations/src/documents/error/mod.rs @@ -1,4 +1,4 @@ -//! The crate-wide error and result alias. +//! The documents module's error and result alias. //! //! Every failure intake can have names what a caller can do about it: fix the //! input ([`Error::Invalid`]), send something smaller ([`Error::TooLarge`]), @@ -12,7 +12,7 @@ pub enum Error { /// textual format, or a conversion that produced no text. #[error("invalid document: {0}")] Invalid(String), - /// The document is over [`crate::MAX_DOCUMENT_BYTES`]. + /// The document is over [`crate::documents::MAX_DOCUMENT_BYTES`]. #[error("document is {size} bytes, over the {limit}-byte intake limit")] TooLarge { /// The document's size in bytes. @@ -26,7 +26,7 @@ pub enum Error { /// A converter claimed the format and then failed. #[error("converter {converter} failed: {message}")] Converter { - /// The converter's [`crate::DocumentConverter::name`]. + /// The converter's [`crate::documents::DocumentConverter::name`]. converter: String, /// What went wrong, as the converter reported it. message: String, @@ -46,7 +46,7 @@ impl From for tinymemory_api::Error { } } -/// Result alias for this crate's fallible operations. +/// Result alias for this module's fallible operations. pub type Result = std::result::Result; #[cfg(test)] diff --git a/crates/tinymemory-documents/src/error/mod_tests.rs b/crates/tinymemory-integrations/src/documents/error/mod_tests.rs similarity index 94% rename from crates/tinymemory-documents/src/error/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/error/mod_tests.rs index 5c5213dd..33aedd12 100644 --- a/crates/tinymemory-documents/src/error/mod_tests.rs +++ b/crates/tinymemory-integrations/src/documents/error/mod_tests.rs @@ -1,4 +1,4 @@ -//! Tests for the crate error and its mapping onto the contract error. +//! Tests for the module error and its mapping onto the contract error. use super::*; diff --git a/crates/tinymemory-documents/src/format/mod.rs b/crates/tinymemory-integrations/src/documents/format/mod.rs similarity index 97% rename from crates/tinymemory-documents/src/format/mod.rs rename to crates/tinymemory-integrations/src/documents/format/mod.rs index 6551901a..0d048137 100644 --- a/crates/tinymemory-documents/src/format/mod.rs +++ b/crates/tinymemory-integrations/src/documents/format/mod.rs @@ -29,9 +29,9 @@ pub enum DocumentFormat { /// HTML. Converted structurally — headings, lists, links, code. Html, /// Source code. Stored as written, never reflowed or HTML-converted; the - /// language comes from [`crate::language_for_path`]. + /// language comes from [`crate::documents::language_for_path`]. Code, - /// PDF. Needs a real extractor; see [`crate::convert::DocumentConverter`]. + /// PDF. Needs a real extractor; see [`crate::documents::convert::DocumentConverter`]. Pdf, /// Office Open XML word processing (`.docx`). Needs a real extractor. Docx, @@ -204,12 +204,12 @@ impl DocumentFormat { /// Map a filename or path onto a format by its extension. /// - /// A name [`crate::language_for_path`] recognises as code is + /// A name [`crate::documents::language_for_path`] recognises as code is /// [`DocumentFormat::Code`]; that check runs first so `CMakeLists.txt` is /// code rather than plain text. HTML stays [`DocumentFormat::Html`]. #[must_use] pub fn from_filename(filename: &str) -> Option { - if crate::language::language_for_path(filename).is_some() { + if crate::documents::language::language_for_path(filename).is_some() { return Some(Self::Code); } let extension = filename.rsplit_once('.')?.1.to_ascii_lowercase(); diff --git a/crates/tinymemory-documents/src/format/mod_tests.rs b/crates/tinymemory-integrations/src/documents/format/mod_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/format/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/format/mod_tests.rs diff --git a/crates/tinymemory-documents/src/format/ooxml.rs b/crates/tinymemory-integrations/src/documents/format/ooxml.rs similarity index 100% rename from crates/tinymemory-documents/src/format/ooxml.rs rename to crates/tinymemory-integrations/src/documents/format/ooxml.rs diff --git a/crates/tinymemory-documents/src/format/ooxml_tests.rs b/crates/tinymemory-integrations/src/documents/format/ooxml_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/format/ooxml_tests.rs rename to crates/tinymemory-integrations/src/documents/format/ooxml_tests.rs diff --git a/crates/tinymemory-documents/src/html/entity.rs b/crates/tinymemory-integrations/src/documents/html/entity.rs similarity index 97% rename from crates/tinymemory-documents/src/html/entity.rs rename to crates/tinymemory-integrations/src/documents/html/entity.rs index 502a78da..4334af43 100644 --- a/crates/tinymemory-documents/src/html/entity.rs +++ b/crates/tinymemory-integrations/src/documents/html/entity.rs @@ -7,7 +7,7 @@ //! nobody notices. /// Decode HTML entities in `text`. -pub(super) fn decode_entities(text: &str) -> String { +pub(crate) fn decode_entities(text: &str) -> String { if !text.contains('&') { return text.to_string(); } diff --git a/crates/tinymemory-documents/src/html/entity_tests.rs b/crates/tinymemory-integrations/src/documents/html/entity_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/html/entity_tests.rs rename to crates/tinymemory-integrations/src/documents/html/entity_tests.rs diff --git a/crates/tinymemory-documents/src/html/mod.rs b/crates/tinymemory-integrations/src/documents/html/mod.rs similarity index 98% rename from crates/tinymemory-documents/src/html/mod.rs rename to crates/tinymemory-integrations/src/documents/html/mod.rs index bf9fe644..af13aea0 100644 --- a/crates/tinymemory-documents/src/html/mod.rs +++ b/crates/tinymemory-integrations/src/documents/html/mod.rs @@ -10,9 +10,9 @@ //! Because the output is prose for a language model to read, and the failure //! modes of a tag-stream walk are all cosmetic: a malformed nesting produces //! slightly wrong emphasis, never wrong text. Pulling in a full DOM parser -//! would cost this crate its "no heavy dependencies" position for output +//! would cost this module its "no heavy dependencies" position for output //! nobody renders. If a host needs fidelity beyond this, it supplies its own -//! [`crate::convert::DocumentConverter`]. +//! [`crate::documents::convert::DocumentConverter`]. //! //! Script and style bodies are removed before anything else, so their contents //! can never reach the output as text. `` goes with them: it is document @@ -22,7 +22,7 @@ mod entity; -use entity::decode_entities; +pub(crate) use entity::decode_entities; /// Convert an HTML document to markdown. /// diff --git a/crates/tinymemory-documents/src/html/mod_tests.rs b/crates/tinymemory-integrations/src/documents/html/mod_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/html/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/html/mod_tests.rs diff --git a/crates/tinymemory-documents/src/item/mod.rs b/crates/tinymemory-integrations/src/documents/item/mod.rs similarity index 79% rename from crates/tinymemory-documents/src/item/mod.rs rename to crates/tinymemory-integrations/src/documents/item/mod.rs index d281e90c..72f00b77 100644 --- a/crates/tinymemory-documents/src/item/mod.rs +++ b/crates/tinymemory-integrations/src/documents/item/mod.rs @@ -4,29 +4,29 @@ //! and what comes out is the item an engine stores — the markdown as //! [`DocumentBody::Text`], a title, the format's MIME type, and the caller's //! [`MemoryMeta`]. Where the item is stored is the host's decision, made by -//! whichever engine it bound; this crate never writes. +//! whichever engine it bound; this module never writes. //! //! The caller owns the metadata. Intake fills exactly one field, and only when //! the caller left it unset: [`MemoryMeta::language`], from the converter or -//! from the file extension ([`crate::language_for_path`]). Provenance — +//! from the file extension ([`crate::documents::language_for_path`]). Provenance — //! source, workspace, URL, observation time — is the caller's to state. use tinymemory_api::{DocumentBody, MemoryMeta, StoreItem}; -use crate::convert::{ConvertedDocument, DocumentConverter, RawDocument}; -use crate::error::Result; -use crate::format::DocumentFormat; -use crate::language::language_for_path; +use crate::documents::convert::{ConvertedDocument, DocumentConverter, RawDocument}; +use crate::documents::error::Result; +use crate::documents::format::DocumentFormat; +use crate::documents::language::language_for_path; /// Convert `document` through `converter` and wrap the result as a /// [`StoreItem::Document`] carrying `meta`. /// /// # Errors /// -/// Whatever the converter returns: [`crate::Error::Invalid`] for an empty or -/// undecodable body, [`crate::Error::TooLarge`] over the size cap, -/// [`crate::Error::UnsupportedFormat`] for a format nothing converts, and -/// [`crate::Error::Converter`] for a converter's own failure. +/// Whatever the converter returns: [`crate::documents::Error::Invalid`] for an empty or +/// undecodable body, [`crate::documents::Error::TooLarge`] over the size cap, +/// [`crate::documents::Error::UnsupportedFormat`] for a format nothing converts, and +/// [`crate::documents::Error::Converter`] for a converter's own failure. pub async fn document_item( converter: &dyn DocumentConverter, document: &RawDocument, diff --git a/crates/tinymemory-documents/src/item/mod_tests.rs b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs similarity index 98% rename from crates/tinymemory-documents/src/item/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/item/mod_tests.rs index c285c134..bb696a90 100644 --- a/crates/tinymemory-documents/src/item/mod_tests.rs +++ b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs @@ -4,8 +4,8 @@ use super::*; use tinymemory_api::{ItemKind, SourceKind}; -use crate::convert::ConverterChain; -use crate::error::Error; +use crate::documents::convert::ConverterChain; +use crate::documents::error::Error; fn parts(item: StoreItem) -> (Option<String>, String, Option<String>, MemoryMeta) { match item { diff --git a/crates/tinymemory-documents/src/language/mod.rs b/crates/tinymemory-integrations/src/documents/language/mod.rs similarity index 97% rename from crates/tinymemory-documents/src/language/mod.rs rename to crates/tinymemory-integrations/src/documents/language/mod.rs index 31a7d7b0..cea66b65 100644 --- a/crates/tinymemory-documents/src/language/mod.rs +++ b/crates/tinymemory-integrations/src/documents/language/mod.rs @@ -8,7 +8,7 @@ //! contract: do not rename one. //! //! Markdown, plain text and HTML are documents, not code, and map to `None`; -//! [`crate::DocumentFormat`] has its own variants for them. +//! [`crate::documents::DocumentFormat`] has its own variants for them. /// Files recognised by their whole name rather than an extension. /// @@ -115,7 +115,7 @@ const EXTENSIONS: &[(&str, &str)] = &[ /// text, HTML), for unknown extensions, and for names without one. /// /// ``` -/// use tinymemory_documents::language_for_path; +/// use tinymemory_integrations::documents::language_for_path; /// /// assert_eq!(language_for_path("src/main.rs"), Some("rust")); /// assert_eq!(language_for_path("web/App.TSX"), Some("typescript")); diff --git a/crates/tinymemory-documents/src/language/mod_tests.rs b/crates/tinymemory-integrations/src/documents/language/mod_tests.rs similarity index 100% rename from crates/tinymemory-documents/src/language/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/language/mod_tests.rs diff --git a/crates/tinymemory-documents/src/lib.rs b/crates/tinymemory-integrations/src/documents/mod.rs similarity index 83% rename from crates/tinymemory-documents/src/lib.rs rename to crates/tinymemory-integrations/src/documents/mod.rs index 92475d7c..ac1f84a4 100644 --- a/crates/tinymemory-documents/src/lib.rs +++ b/crates/tinymemory-integrations/src/documents/mod.rs @@ -11,21 +11,21 @@ //! 2. **Turn it into markdown.** [`DocumentConverter`] is the seam; //! [`NativeConverter`] covers markdown, plain text, HTML and code with no //! dependencies; a host binds its own for PDF and Office documents, or -//! prepends `OfficeConverter` (feature `office`) for PDF, DOCX, PPTX and +//! prepends `OfficeConverter` (feature `documents-office`) for PDF, DOCX, PPTX and //! XLSX. //! 3. **Wrap it as an item.** [`document_item`] produces a //! `StoreItem::Document` with the caller's //! [`MemoryMeta`](tinymemory_api::MemoryMeta), filling `language` from the //! file extension when the caller left it unset. //! -//! This crate does no I/O. Reading files and fetching URLs belongs to -//! `tinymemory-sources`, which depends on this crate for conversion. +//! This module does no I/O. Reading files and fetching URLs belongs to +//! the `sources` module, which depends on this one for conversion. //! //! # Example //! //! ``` //! use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem}; -//! use tinymemory_documents::{ConverterChain, RawDocument, document_item}; +//! use tinymemory_integrations::documents::{ConverterChain, RawDocument, document_item}; //! //! # let runtime = tokio::runtime::Builder::new_current_thread().build()?; //! # runtime.block_on(async { @@ -40,7 +40,7 @@ //! assert_eq!(title.as_deref(), Some("main.rs")); //! assert_eq!(body, DocumentBody::Text("fn main() {}\n".into())); //! assert_eq!(meta.language.as_deref(), Some("rust")); -//! # Ok::<(), tinymemory_documents::Error>(()) +//! # Ok::<(), tinymemory_integrations::documents::Error>(()) //! # })?; //! # Ok::<(), Box<dyn std::error::Error>>(()) //! ``` @@ -51,7 +51,7 @@ pub mod format; pub mod html; pub mod item; pub mod language; -#[cfg(feature = "office")] +#[cfg(feature = "documents-office")] pub mod office; pub use convert::{ @@ -62,5 +62,5 @@ pub use error::{Error, Result}; pub use format::DocumentFormat; pub use item::{converted_item, document_item}; pub use language::language_for_path; -#[cfg(feature = "office")] +#[cfg(feature = "documents-office")] pub use office::OfficeConverter; diff --git a/crates/tinymemory-documents/src/office/mod.rs b/crates/tinymemory-integrations/src/documents/office/mod.rs similarity index 88% rename from crates/tinymemory-documents/src/office/mod.rs rename to crates/tinymemory-integrations/src/documents/office/mod.rs index 46b81524..e529e5be 100644 --- a/crates/tinymemory-documents/src/office/mod.rs +++ b/crates/tinymemory-integrations/src/documents/office/mod.rs @@ -1,12 +1,12 @@ -//! PDF and Office Open XML conversion (the `office` feature). +//! PDF and Office Open XML conversion (the `documents-office` feature). //! -//! [`crate::convert::NativeConverter`] handles what is already text. This is +//! [`crate::documents::convert::NativeConverter`] handles what is already text. This is //! the converter for the formats people actually drop into memory that are not //! — a contract PDF, a spec `.docx`, a pricing `.xlsx`, a deck — so a host //! does not have to bind an extractor of its own for them: //! //! ``` -//! use tinymemory_documents::{ConverterChain, OfficeConverter}; +//! use tinymemory_integrations::documents::{ConverterChain, OfficeConverter}; //! //! let chain = ConverterChain::default().prepend(Box::new(OfficeConverter)); //! ``` @@ -25,7 +25,7 @@ //! //! ## Hostile input //! -//! [`crate::convert::MAX_DOCUMENT_BYTES`] caps the *compressed* upload, but an +//! [`crate::documents::convert::MAX_DOCUMENT_BYTES`] caps the *compressed* upload, but an //! Office file is a zip, and a small highly compressed part can expand without //! limit. Every archive is therefore refused when the uncompressed sizes its //! central directory declares sum past [`MAX_DECOMPRESSED_BYTES`] — checked @@ -57,10 +57,10 @@ mod xlsx; use async_trait::async_trait; #[cfg(test)] -use crate::convert::MAX_DOCUMENT_BYTES; -use crate::convert::{ConvertedDocument, DocumentConverter, RawDocument, check_size}; -use crate::error::{Error, Result}; -use crate::format::DocumentFormat; +use crate::documents::convert::MAX_DOCUMENT_BYTES; +use crate::documents::convert::{ConvertedDocument, DocumentConverter, RawDocument, check_size}; +use crate::documents::error::{Error, Result}; +use crate::documents::format::DocumentFormat; /// The largest uncompressed size an Office archive may declare, in bytes, /// before it is refused as a likely zip bomb. See the module docs. @@ -74,7 +74,7 @@ pub const MAX_SPREADSHEET_DENSE_CELLS: usize = 1_000_000; /// Converts PDF, DOCX, PPTX and XLSX documents to markdown, in-process. /// /// Claims exactly those four formats, so it composes with -/// [`crate::convert::NativeConverter`] in a [`crate::convert::ConverterChain`] +/// [`crate::documents::convert::NativeConverter`] in a [`crate::documents::convert::ConverterChain`] /// without shadowing it. A document whose text cannot be read — malformed, /// over a cap, or a scanned PDF with no text layer — is [`Error::Invalid`] /// saying which, never an empty document. @@ -91,7 +91,7 @@ impl OfficeConverter { /// claim; [`Error::Invalid`] for an empty body, a document that cannot be /// read or exceeds a decoding cap, or one with no extractable text; /// [`Error::TooLarge`] for a body over - /// [`crate::convert::MAX_DOCUMENT_BYTES`]. + /// [`crate::documents::convert::MAX_DOCUMENT_BYTES`]. pub fn convert_blocking(&self, document: &RawDocument) -> Result<ConvertedDocument> { check_size(document)?; let format = document.format(); diff --git a/crates/tinymemory-documents/src/office/mod_tests.rs b/crates/tinymemory-integrations/src/documents/office/mod_tests.rs similarity index 99% rename from crates/tinymemory-documents/src/office/mod_tests.rs rename to crates/tinymemory-integrations/src/documents/office/mod_tests.rs index 43057a77..b41a19c9 100644 --- a/crates/tinymemory-documents/src/office/mod_tests.rs +++ b/crates/tinymemory-integrations/src/documents/office/mod_tests.rs @@ -5,7 +5,7 @@ use super::*; -use crate::convert::ConverterChain; +use crate::documents::convert::ConverterChain; /// A deflated zip archive of `(path, contents)` parts. fn package(parts: &[(&str, &str)]) -> Vec<u8> { diff --git a/crates/tinymemory-documents/src/office/normalize.rs b/crates/tinymemory-integrations/src/documents/office/normalize.rs similarity index 100% rename from crates/tinymemory-documents/src/office/normalize.rs rename to crates/tinymemory-integrations/src/documents/office/normalize.rs diff --git a/crates/tinymemory-documents/src/office/ooxml.rs b/crates/tinymemory-integrations/src/documents/office/ooxml.rs similarity index 99% rename from crates/tinymemory-documents/src/office/ooxml.rs rename to crates/tinymemory-integrations/src/documents/office/ooxml.rs index 161791b1..07652f3b 100644 --- a/crates/tinymemory-documents/src/office/ooxml.rs +++ b/crates/tinymemory-integrations/src/documents/office/ooxml.rs @@ -6,7 +6,7 @@ use quick_xml::events::Event; use zip::ZipArchive; use super::{MAX_DECOMPRESSED_BYTES, unreadable}; -use crate::error::Result; +use crate::documents::error::Result; /// The body text of a Word document. /// diff --git a/crates/tinymemory-documents/src/office/pdf.rs b/crates/tinymemory-integrations/src/documents/office/pdf.rs similarity index 95% rename from crates/tinymemory-documents/src/office/pdf.rs rename to crates/tinymemory-integrations/src/documents/office/pdf.rs index 580d8945..6481b584 100644 --- a/crates/tinymemory-documents/src/office/pdf.rs +++ b/crates/tinymemory-integrations/src/documents/office/pdf.rs @@ -1,7 +1,7 @@ //! A PDF's text layer. use super::unreadable; -use crate::error::Result; +use crate::documents::error::Result; /// Extracts a PDF's text layer, which may be empty. /// diff --git a/crates/tinymemory-documents/src/office/xlsx.rs b/crates/tinymemory-integrations/src/documents/office/xlsx.rs similarity index 98% rename from crates/tinymemory-documents/src/office/xlsx.rs rename to crates/tinymemory-integrations/src/documents/office/xlsx.rs index 89f5cf90..0755ab4f 100644 --- a/crates/tinymemory-documents/src/office/xlsx.rs +++ b/crates/tinymemory-integrations/src/documents/office/xlsx.rs @@ -5,7 +5,7 @@ use std::io::Cursor; use calamine::{Data, Reader, Xlsx}; use super::{MAX_SPREADSHEET_DENSE_CELLS, ooxml, unreadable}; -use crate::error::Result; +use crate::documents::error::Result; /// Extracts every non-empty row of every sheet, in sheet order. pub(super) fn extract(bytes: &[u8]) -> Result<String> { diff --git a/crates/tinymemory-integrations/src/error/mod.rs b/crates/tinymemory-integrations/src/error/mod.rs new file mode 100644 index 00000000..c9be7d82 --- /dev/null +++ b/crates/tinymemory-integrations/src/error/mod.rs @@ -0,0 +1,14 @@ +//! The crate-wide error is the contract's error. +//! +//! Every integration ends at a [`tinymemory_api::MemoryEngine`] call or +//! produces a [`tinymemory_api::StoreItem`] for one, so the error a host acts +//! on is [`tinymemory_api::Error`]. The CortexDB engine and the registry return +//! it directly. `documents`, `sources` and `import` keep a typed error of their +//! own, because their failures (a path escaping its root, a non-v1 workspace) +//! are worth matching on before they reach an engine, and each converts into +//! this one with `From`/`?`. + +pub use tinymemory_api::Error; + +/// The crate-wide result alias. +pub type Result<T> = std::result::Result<T, Error>; diff --git a/crates/tinymemory-import/README.md b/crates/tinymemory-integrations/src/import/README.md similarity index 75% rename from crates/tinymemory-import/README.md rename to crates/tinymemory-integrations/src/import/README.md index ef7026f1..d7db2a73 100644 --- a/crates/tinymemory-import/README.md +++ b/crates/tinymemory-integrations/src/import/README.md @@ -1,12 +1,15 @@ -# tinymemory-import +# import -Reads a legacy (v1, embedded TinyCortex) workspace and yields TinyMemory v2 +The `import` module of `tinymemory-integrations` (feature `legacy-import`): +reads a legacy (v1, embedded TinyCortex) workspace and yields TinyMemory v2 `StoreItem`s, resumably. The v1 engine that wrote the store is not linked: the importer reads its SQLite files directly with `rusqlite`, opened read-only, and chunk bodies with `std::fs`. It never writes to the legacy workspace. -The facade exposes this crate behind its `legacy-import` feature. The crate -itself has no features: being the legacy reader is its whole job. +The module has no sub-features: being the legacy reader is its whole job. It +needs only `rusqlite` (bundled SQLite), `serde`, `serde_json` and `thiserror`. +Architecture overview: +[`docs/architecture/integrations.md`](../../../../docs/architecture/integrations.md). ## Surface @@ -17,7 +20,10 @@ itself has no features: being the legacy reader is its whole job. | `Items::with_page_size(n)` | Keys fetched per query (default `DEFAULT_PAGE_SIZE`, 256). Does not affect output. | | `ImportedItem { item, checkpoint }` | An item and the checkpoint to persist once it is stored. | | `Checkpoint` | Last yielded key per section; `to_json` / `from_json` for the host to persist. | -| `Error` / `Result` | `NotFound`, `NotLegacy`, `Sqlite`, `Io`, `Json`. | +| `migrate(engine, workspace, from)` | Copies every item after `from` into a `MemoryEngine`, in `store_many` batches; returns a `MigrationReport`. | +| `migrate_with(engine, workspace, from, on_batch)` | `migrate`, calling `on_batch(&Checkpoint)` after each stored batch so the host can persist it. | +| `MigrationReport { stored, replayed, batches, checkpoint }` | What a run did, and where to resume. | +| `Error` / `Result` | `NotFound`, `NotLegacy`, `Sqlite`, `Io`, `Json`, `Engine { source, checkpoint }`; `Error::checkpoint()` reads the resume point. | ## Detection @@ -144,3 +150,33 @@ not seen). The iterator fetches one page of keys per query, so memory is bounded by the page size and, for a conversation or chunk source, by that one thread or source. After an error it yields nothing more; resume from the last persisted checkpoint. + +## Migrating into an engine + +`migrate` is the whole backwards-compatibility path: open the v1 workspace, +hand it to `migrate` with the engine built from the host's config (CortexDB, +usually) and the checkpoint persisted by an earlier run, if any. + +```rust,ignore +let workspace = LegacyWorkspace::open(path)?; +let from = saved.map(|json| Checkpoint::from_json(&json)).transpose()?; +let report = migrate_with(engine.as_ref(), workspace, from, |checkpoint| { + save(checkpoint.to_json()); +}) +.await?; +``` + +- Items are read in the order above and sent in batches of at most + `MAX_STORE_MANY` (100). After a batch is stored, its last item's checkpoint + is *committed*: passed to `on_batch` and kept as `report.checkpoint`. +- An engine failure is `Error::Engine { source, checkpoint }`, with the last + committed checkpoint (or `from`, if no batch was stored). Resume by calling + `migrate` again with it. The failed batch may have stored a prefix of its + items; the engine answers those as replays, so a resumed or repeated run + never duplicates (a second full run reports every item as `replayed`). +- A legacy read failure (`Sqlite`, `Io`) is returned as is. Every checkpoint + committed before it has been passed to `on_batch`; resuming from any of + them, or from the start, only replays. +- The workspace is taken by value: its SQLite handle is not `Sync`, and owning + it keeps the returned future `Send`, so a long import can run on a spawned + task. diff --git a/crates/tinymemory-import/src/checkpoint/mod.rs b/crates/tinymemory-integrations/src/import/checkpoint/mod.rs similarity index 91% rename from crates/tinymemory-import/src/checkpoint/mod.rs rename to crates/tinymemory-integrations/src/import/checkpoint/mod.rs index 04695563..72c96096 100644 --- a/crates/tinymemory-import/src/checkpoint/mod.rs +++ b/crates/tinymemory-integrations/src/import/checkpoint/mod.rs @@ -3,7 +3,7 @@ //! An import walks the legacy store in a fixed section order (documents, //! chunks, conversations, learnings, profile) and, within a section, by a //! stable key. A [`Checkpoint`] records the key of the last item yielded in -//! each section; [`crate::LegacyWorkspace::items_from`] skips everything at or +//! each section; [`crate::import::LegacyWorkspace::items_from`] skips everything at or //! before it. Every [`ImportedItem`] carries the checkpoint to persist once //! that item is stored, so a crash between two stores re-yields at most the //! one item that was not acknowledged, which the engine then treats as a @@ -12,7 +12,7 @@ use serde::{Deserialize, Serialize}; use tinymemory_api::StoreItem; -use crate::error::Result; +use crate::import::error::Result; /// The last yielded key in each section of a legacy import. /// @@ -50,7 +50,7 @@ impl Checkpoint { /// /// # Errors /// - /// [`crate::Error::Json`] if serialisation fails, which a checkpoint of + /// [`crate::import::Error::Json`] if serialisation fails, which a checkpoint of /// plain strings does not do in practice. pub fn to_json(&self) -> Result<String> { Ok(serde_json::to_string(self)?) @@ -60,7 +60,7 @@ impl Checkpoint { /// /// # Errors /// - /// [`crate::Error::Json`] if `json` is not a checkpoint. + /// [`crate::import::Error::Json`] if `json` is not a checkpoint. pub fn from_json(json: &str) -> Result<Self> { Ok(serde_json::from_str(json)?) } diff --git a/crates/tinymemory-import/src/checkpoint/mod_tests.rs b/crates/tinymemory-integrations/src/import/checkpoint/mod_tests.rs similarity index 97% rename from crates/tinymemory-import/src/checkpoint/mod_tests.rs rename to crates/tinymemory-integrations/src/import/checkpoint/mod_tests.rs index d1cbad8f..eb292a22 100644 --- a/crates/tinymemory-import/src/checkpoint/mod_tests.rs +++ b/crates/tinymemory-integrations/src/import/checkpoint/mod_tests.rs @@ -1,7 +1,7 @@ //! Checkpoint encoding tests. use super::*; -use crate::error::Error; +use crate::import::error::Error; #[test] fn a_default_checkpoint_is_the_start() { diff --git a/crates/tinymemory-import/src/convert/mod.rs b/crates/tinymemory-integrations/src/import/convert/mod.rs similarity index 100% rename from crates/tinymemory-import/src/convert/mod.rs rename to crates/tinymemory-integrations/src/import/convert/mod.rs diff --git a/crates/tinymemory-import/src/convert/mod_tests.rs b/crates/tinymemory-integrations/src/import/convert/mod_tests.rs similarity index 100% rename from crates/tinymemory-import/src/convert/mod_tests.rs rename to crates/tinymemory-integrations/src/import/convert/mod_tests.rs diff --git a/crates/tinymemory-integrations/src/import/error/mod.rs b/crates/tinymemory-integrations/src/import/error/mod.rs new file mode 100644 index 00000000..8f7c7f3b --- /dev/null +++ b/crates/tinymemory-integrations/src/import/error/mod.rs @@ -0,0 +1,67 @@ +//! The import module's [`Error`] and [`Result`]. + +use std::path::PathBuf; + +use crate::import::checkpoint::Checkpoint; + +/// Everything that can go wrong opening, reading or migrating a legacy +/// workspace. +#[derive(Debug, thiserror::Error)] +#[non_exhaustive] +pub enum Error { + /// The path given to [`crate::import::LegacyWorkspace::open`] does not exist. + #[error("no legacy workspace at {}", path.display())] + NotFound { + /// The path that was looked up. + path: PathBuf, + }, + /// The path exists but is not a v1 TinyCortex workspace. + #[error("{} is not a v1 tinycortex workspace: {reason}", path.display())] + NotLegacy { + /// The path that was inspected. + path: PathBuf, + /// What was missing or wrong. + reason: String, + }, + /// A legacy SQLite database could not be read. + #[error("legacy sqlite read failed: {0}")] + Sqlite(#[from] rusqlite::Error), + /// A file referenced by the legacy store could not be read. + #[error("reading {} failed: {source}", path.display())] + Io { + /// The file that was read. + path: PathBuf, + /// The underlying error. + #[source] + source: std::io::Error, + }, + /// A [`crate::import::Checkpoint`] could not be encoded or decoded as JSON. + #[error("checkpoint json is invalid: {0}")] + Json(#[from] serde_json::Error), + /// The engine refused a batch during [`crate::import::migrate`]. + /// Everything up to `checkpoint` is stored; resume from it. + #[error("migration stopped: the engine failed: {source}")] + Engine { + /// The engine's own error. + #[source] + source: tinymemory_api::Error, + /// The last committed resume point (boxed to keep every `Result` + /// of this module small). + checkpoint: Box<Checkpoint>, + }, +} + +impl Error { + /// The checkpoint to resume a [`crate::import::migrate`] from, when the + /// error carries one ([`Error::Engine`]). + #[must_use] + pub fn checkpoint(&self) -> Option<&Checkpoint> { + match self { + Self::Engine { checkpoint, .. } => Some(checkpoint), + _ => None, + } + } +} + +/// The import module's result alias. +pub type Result<T> = std::result::Result<T, Error>; diff --git a/crates/tinymemory-import/src/items/mod.rs b/crates/tinymemory-integrations/src/import/items/mod.rs similarity index 94% rename from crates/tinymemory-import/src/items/mod.rs rename to crates/tinymemory-integrations/src/import/items/mod.rs index 22222d40..8d461134 100644 --- a/crates/tinymemory-import/src/items/mod.rs +++ b/crates/tinymemory-integrations/src/import/items/mod.rs @@ -14,10 +14,10 @@ use std::collections::VecDeque; use std::iter::FusedIterator; -use crate::checkpoint::{Checkpoint, ImportedItem}; -use crate::error::Result; -use crate::sections::{ORDER, Scanned}; -use crate::workspace::LegacyWorkspace; +use crate::import::checkpoint::{Checkpoint, ImportedItem}; +use crate::import::error::Result; +use crate::import::sections::{ORDER, Scanned}; +use crate::import::workspace::LegacyWorkspace; /// Keys fetched per query unless [`Items::with_page_size`] says otherwise. pub const DEFAULT_PAGE_SIZE: usize = 256; diff --git a/crates/tinymemory-integrations/src/import/migrate/mod.rs b/crates/tinymemory-integrations/src/import/migrate/mod.rs new file mode 100644 index 00000000..2f75e762 --- /dev/null +++ b/crates/tinymemory-integrations/src/import/migrate/mod.rs @@ -0,0 +1,110 @@ +//! [`migrate`]: copy a legacy workspace into an engine, resumably. +//! +//! The driver streams [`LegacyWorkspace::items_from`] in batches of at most +//! [`MAX_STORE_MANY`] items and hands each to +//! [`MemoryEngine::store_many`]. After a batch is stored, the checkpoint of its +//! last item is *committed*: it is reported to the caller's callback +//! ([`migrate_with`]) and becomes the resume point. +//! +//! A failure never loses progress: +//! +//! - an engine failure is [`Error::Engine`], carrying the last committed +//! checkpoint, so the host resumes from exactly there; +//! - a failed batch may have stored some of its items (`store_many` stores in +//! order and stops at the error), and resuming re-sends them, which the +//! engine answers as replays, not duplicates; +//! - a legacy read failure is returned as is; the callback has already seen +//! every committed checkpoint, and resuming from any earlier one (or the +//! start) only replays. +//! +//! The workspace is taken by value so the returned future is `Send` (the +//! legacy store's SQLite handle is not `Sync`), and a host can run a long +//! import on a spawned task. + +use tinymemory_api::{MAX_STORE_MANY, MemoryEngine}; + +use crate::import::checkpoint::Checkpoint; +use crate::import::error::{Error, Result}; +use crate::import::workspace::LegacyWorkspace; + +/// What a [`migrate`] run did. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct MigrationReport { + /// Items the engine stored for the first time. + pub stored: usize, + /// Items the engine already held (a re-run, or a resumed batch). + pub replayed: usize, + /// `store_many` calls made. + pub batches: usize, + /// The resume point after the last stored batch: the checkpoint the run + /// started from when nothing was left to store. + pub checkpoint: Checkpoint, +} + +/// Stores every item of `workspace` after `from` (everything, for `None`) +/// into `engine`, in batches of at most [`MAX_STORE_MANY`]. +/// +/// # Errors +/// +/// - [`Error::Engine`] when `store_many` fails, carrying the checkpoint of +/// the last stored batch (or `from`) to resume from. +/// - [`Error::Sqlite`] or [`Error::Io`] when the legacy store cannot be read. +pub async fn migrate( + engine: &dyn MemoryEngine, + workspace: LegacyWorkspace, + from: Option<Checkpoint>, +) -> Result<MigrationReport> { + migrate_with(engine, workspace, from, |_: &Checkpoint| {}).await +} + +/// [`migrate`], calling `on_batch` with the committed checkpoint after each +/// stored batch so the host can persist it (with +/// [`Checkpoint::to_json`]) as it goes. +/// +/// # Errors +/// +/// As [`migrate`]. +pub async fn migrate_with<F>( + engine: &dyn MemoryEngine, + workspace: LegacyWorkspace, + from: Option<Checkpoint>, + mut on_batch: F, +) -> Result<MigrationReport> +where + F: FnMut(&Checkpoint) + Send, +{ + let mut report = MigrationReport { + checkpoint: from.unwrap_or_default(), + ..MigrationReport::default() + }; + loop { + // Read the batch before awaiting: the iterator borrows the + // workspace, whose SQLite handle must not be held across an await. + let mut items = Vec::with_capacity(MAX_STORE_MANY); + let mut last = None; + for imported in workspace + .items_from(&report.checkpoint) + .take(MAX_STORE_MANY) + { + let imported = imported?; + items.push(imported.item); + last = Some(imported.checkpoint); + } + let Some(last) = last else { + return Ok(report); + }; + let receipts = engine + .store_many(items) + .await + .map_err(|source| Error::Engine { + source, + checkpoint: Box::new(report.checkpoint.clone()), + })?; + let replayed = receipts.iter().filter(|receipt| receipt.replayed).count(); + report.replayed += replayed; + report.stored += receipts.len() - replayed; + report.batches += 1; + report.checkpoint = last; + on_batch(&report.checkpoint); + } +} diff --git a/crates/tinymemory-import/src/lib.rs b/crates/tinymemory-integrations/src/import/mod.rs similarity index 86% rename from crates/tinymemory-import/src/lib.rs rename to crates/tinymemory-integrations/src/import/mod.rs index 1ef4cacd..aa56c1a3 100644 --- a/crates/tinymemory-import/src/lib.rs +++ b/crates/tinymemory-integrations/src/import/mod.rs @@ -18,17 +18,20 @@ //! Every item's `meta.source` is `SourceKind::Import` with a section-scoped //! legacy id (`memory_docs:<document_id>`, `episodic_log:<session_id>`, //! `user_profile:<facet_id>`, `mem_tree_chunks:<kind>:<id>`), and -//! `meta.workspace` is the workspace path. The crate README details every -//! mapping decision. +//! `meta.workspace` is the workspace path. The module's `README.md` details +//! every mapping decision. //! //! Import is resumable: each [`ImportedItem`] carries the [`Checkpoint`] to //! persist once its item is stored, and [`LegacyWorkspace::items_from`] -//! continues after it. +//! continues after it. [`migrate`] (and [`migrate_with`], which reports each +//! committed checkpoint) drives the whole copy into a +//! [`tinymemory_api::MemoryEngine`] in `store_many` batches, and an engine +//! failure carries the checkpoint to resume from. //! //! # Example //! //! ``` -//! use tinymemory_import::{Checkpoint, LegacyWorkspace}; +//! use tinymemory_integrations::import::{Checkpoint, LegacyWorkspace}; //! # let dir = tempfile::tempdir()?; //! # std::fs::create_dir_all(dir.path().join("memory"))?; //! # let db = rusqlite::Connection::open(dir.path().join("memory/memory.db"))?; @@ -67,12 +70,14 @@ mod checkpoint; mod convert; mod error; mod items; +mod migrate; mod sections; mod workspace; pub use checkpoint::{Checkpoint, ChunkCursor, ImportedItem}; pub use error::{Error, Result}; pub use items::{DEFAULT_PAGE_SIZE, Items}; +pub use migrate::{MigrationReport, migrate, migrate_with}; pub use workspace::LegacyWorkspace; /// Re-exported so a host names the same item type the importer yields. diff --git a/crates/tinymemory-import/src/sections/chunks.rs b/crates/tinymemory-integrations/src/import/sections/chunks.rs similarity index 97% rename from crates/tinymemory-import/src/sections/chunks.rs rename to crates/tinymemory-integrations/src/import/sections/chunks.rs index 3bae7764..e188ebf1 100644 --- a/crates/tinymemory-import/src/sections/chunks.rs +++ b/crates/tinymemory-integrations/src/import/sections/chunks.rs @@ -18,10 +18,10 @@ use rusqlite::params; use tinymemory_api::{DocumentBody, Role, StoreItem, Turn, TurnRange}; use super::{Mark, Scanned, import_meta, push_unique, sql_limit}; -use crate::checkpoint::ChunkCursor; -use crate::convert; -use crate::error::{Error, Result}; -use crate::workspace::{ChunkStore, LegacyWorkspace}; +use crate::import::checkpoint::ChunkCursor; +use crate::import::convert; +use crate::import::error::{Error, Result}; +use crate::import::workspace::{ChunkStore, LegacyWorkspace}; /// One chunk with its body resolved. #[derive(Debug)] diff --git a/crates/tinymemory-import/src/sections/chunks_tests.rs b/crates/tinymemory-integrations/src/import/sections/chunks_tests.rs similarity index 100% rename from crates/tinymemory-import/src/sections/chunks_tests.rs rename to crates/tinymemory-integrations/src/import/sections/chunks_tests.rs diff --git a/crates/tinymemory-import/src/sections/episodic.rs b/crates/tinymemory-integrations/src/import/sections/episodic.rs similarity index 96% rename from crates/tinymemory-import/src/sections/episodic.rs rename to crates/tinymemory-integrations/src/import/sections/episodic.rs index 25e885ad..519ff8aa 100644 --- a/crates/tinymemory-import/src/sections/episodic.rs +++ b/crates/tinymemory-integrations/src/import/sections/episodic.rs @@ -10,9 +10,9 @@ use rusqlite::params; use tinymemory_api::{StoreItem, Turn, TurnRange}; use super::{Mark, Scanned, import_meta, sql_limit}; -use crate::convert; -use crate::error::Result; -use crate::workspace::LegacyWorkspace; +use crate::import::convert; +use crate::import::error::Result; +use crate::import::workspace::LegacyWorkspace; /// The next page of threads after `after`. pub(super) fn page( diff --git a/crates/tinymemory-import/src/sections/memory_docs.rs b/crates/tinymemory-integrations/src/import/sections/memory_docs.rs similarity index 98% rename from crates/tinymemory-import/src/sections/memory_docs.rs rename to crates/tinymemory-integrations/src/import/sections/memory_docs.rs index 097d8baf..bfb468a3 100644 --- a/crates/tinymemory-import/src/sections/memory_docs.rs +++ b/crates/tinymemory-integrations/src/import/sections/memory_docs.rs @@ -20,9 +20,9 @@ use serde_json::Value; use tinymemory_api::{DocumentBody, LearningKind, StoreItem}; use super::{Mark, Scanned, import_meta, push_unique, sql_limit}; -use crate::convert; -use crate::error::Result; -use crate::workspace::LegacyWorkspace; +use crate::import::convert; +use crate::import::error::Result; +use crate::import::workspace::LegacyWorkspace; /// v1 section prefixes whose `:` separator the sanitiser turned into `_`. const SECTION_PREFIXES: [&str; 9] = [ diff --git a/crates/tinymemory-import/src/sections/memory_docs_tests.rs b/crates/tinymemory-integrations/src/import/sections/memory_docs_tests.rs similarity index 100% rename from crates/tinymemory-import/src/sections/memory_docs_tests.rs rename to crates/tinymemory-integrations/src/import/sections/memory_docs_tests.rs diff --git a/crates/tinymemory-import/src/sections/mod.rs b/crates/tinymemory-integrations/src/import/sections/mod.rs similarity index 96% rename from crates/tinymemory-import/src/sections/mod.rs rename to crates/tinymemory-integrations/src/import/sections/mod.rs index b7d52344..37ee230c 100644 --- a/crates/tinymemory-import/src/sections/mod.rs +++ b/crates/tinymemory-integrations/src/import/sections/mod.rs @@ -14,9 +14,9 @@ mod profile; use tinymemory_api::{MemoryMeta, SourceKind, StoreItem}; -use crate::checkpoint::{Checkpoint, ChunkCursor}; -use crate::error::Result; -use crate::workspace::LegacyWorkspace; +use crate::import::checkpoint::{Checkpoint, ChunkCursor}; +use crate::import::error::Result; +use crate::import::workspace::LegacyWorkspace; /// One import section, in the order [`ORDER`] walks them. #[derive(Debug, Clone, Copy, PartialEq, Eq)] diff --git a/crates/tinymemory-import/src/sections/profile.rs b/crates/tinymemory-integrations/src/import/sections/profile.rs similarity index 96% rename from crates/tinymemory-import/src/sections/profile.rs rename to crates/tinymemory-integrations/src/import/sections/profile.rs index 675038a5..15515052 100644 --- a/crates/tinymemory-import/src/sections/profile.rs +++ b/crates/tinymemory-integrations/src/import/sections/profile.rs @@ -9,9 +9,9 @@ use rusqlite::params; use tinymemory_api::{LearningKind, StoreItem}; use super::{Mark, Scanned, import_meta, push_unique, sql_limit}; -use crate::convert; -use crate::error::Result; -use crate::workspace::LegacyWorkspace; +use crate::import::convert; +use crate::import::error::Result; +use crate::import::workspace::LegacyWorkspace; /// One `user_profile` row. #[derive(Debug)] diff --git a/crates/tinymemory-import/src/workspace/mod.rs b/crates/tinymemory-integrations/src/import/workspace/mod.rs similarity index 97% rename from crates/tinymemory-import/src/workspace/mod.rs rename to crates/tinymemory-integrations/src/import/workspace/mod.rs index 7b06ece5..4f9614f2 100644 --- a/crates/tinymemory-import/src/workspace/mod.rs +++ b/crates/tinymemory-integrations/src/import/workspace/mod.rs @@ -21,9 +21,9 @@ use rusqlite::{Connection, OpenFlags}; pub(crate) use schema::{ChunkStore, MemorySchema}; -use crate::checkpoint::Checkpoint; -use crate::error::{Error, Result}; -use crate::items::Items; +use crate::import::checkpoint::Checkpoint; +use crate::import::error::{Error, Result}; +use crate::import::items::Items; /// A v1 TinyCortex workspace opened for import. #[derive(Debug)] diff --git a/crates/tinymemory-import/src/workspace/mod_tests.rs b/crates/tinymemory-integrations/src/import/workspace/mod_tests.rs similarity index 100% rename from crates/tinymemory-import/src/workspace/mod_tests.rs rename to crates/tinymemory-integrations/src/import/workspace/mod_tests.rs diff --git a/crates/tinymemory-import/src/workspace/schema.rs b/crates/tinymemory-integrations/src/import/workspace/schema.rs similarity index 99% rename from crates/tinymemory-import/src/workspace/schema.rs rename to crates/tinymemory-integrations/src/import/workspace/schema.rs index 545e1b04..cc07f083 100644 --- a/crates/tinymemory-import/src/workspace/schema.rs +++ b/crates/tinymemory-integrations/src/import/workspace/schema.rs @@ -6,7 +6,7 @@ use std::path::{Path, PathBuf}; use rusqlite::Connection; use super::{is_not_a_database, open_read_only}; -use crate::error::Result; +use crate::import::error::Result; /// Tables a v1 `memory.db` always has. const REQUIRED_TABLES: [&str; 3] = ["memory_docs", "episodic_log", "user_profile"]; diff --git a/crates/tinymemory-integrations/src/lib.rs b/crates/tinymemory-integrations/src/lib.rs new file mode 100644 index 00000000..f5a2111a --- /dev/null +++ b/crates/tinymemory-integrations/src/lib.rs @@ -0,0 +1,62 @@ +//! TinyMemory integrations: everything that connects the core contract +//! ([`tinymemory_api`]) to the outside world. +//! +//! Each integration is a module behind a feature of (nearly) the same name: +//! +//! | Module | Feature | What it does | +//! | --- | --- | --- | +//! | [`cortex`], [`registry`], [`config`] | `cortex` (default) | The CortexDB engine over its two wires, and building one from configuration | +//! | `documents` | `documents`, `documents-office` | Format sniffing and conversion to markdown, emitting `StoreItem::Document` | +//! | `sources` | `sources`, `sources-network` | Readers turning folders, files, links, GitHub, RSS, Composio payloads and conversations into `StoreItem`s | +//! | `safety` | `safety` | Secret and PII scrubbing for a `StoreItem` before it is stored | +//! | `import` | `legacy-import` | Migrating a legacy v1 (embedded TinyCortex) workspace into any engine | +//! +//! A typical write path is source → documents → safety → engine; the +//! agent-facing tools and `context.md` live in `tinymemory-tools`. +//! +//! # Example +//! +//! ```no_run +//! # #[cfg(feature = "cortex")] +//! # async fn demo() -> tinymemory_integrations::Result<()> { +//! use std::sync::Arc; +//! use tinymemory_api::{FetchMode, FetchRequest, MemoryMeta, SourceKind, StoreItem}; +//! use tinymemory_integrations::{EngineCredential, MemoryConfig, cortex::StaticBearer}; +//! +//! let config = MemoryConfig::default(); // engine = "tinyhumans" +//! let engine = config.build(EngineCredential::Dynamic(Arc::new(StaticBearer::new("tiny_live_..."))))?; +//! +//! let meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); +//! engine.store(StoreItem::document("Ownership moves values.", meta)).await?; +//! let page = engine.fetch(FetchRequest::new("ownership", FetchMode::Hybrid, 5)).await?; +//! # let _ = page; +//! # Ok(()) +//! # } +//! ``` + +pub mod error; + +#[cfg(feature = "cortex")] +pub mod config; +#[cfg(feature = "cortex")] +pub mod cortex; +#[cfg(feature = "cortex")] +pub mod registry; + +#[cfg(feature = "documents")] +pub mod documents; +#[cfg(feature = "legacy-import")] +pub mod import; +#[cfg(feature = "safety")] +pub mod safety; +#[cfg(feature = "sources")] +pub mod sources; + +pub use error::{Error, Result}; + +#[cfg(feature = "cortex")] +pub use config::{DEFAULT_ENGINE, EngineSettings, MemoryConfig}; +#[cfg(feature = "cortex")] +pub use cortex::{BearerSource, StaticBearer}; +#[cfg(feature = "cortex")] +pub use registry::{EngineCredential, build_engine, list_engines}; diff --git a/crates/tinymemory-integrations/src/registry/mod.rs b/crates/tinymemory-integrations/src/registry/mod.rs new file mode 100644 index 00000000..564d95d3 --- /dev/null +++ b/crates/tinymemory-integrations/src/registry/mod.rs @@ -0,0 +1,101 @@ +//! The engine registry: [`list_engines`] and [`build_engine`]. +//! +//! Two engines are registered, both served by [`crate::cortex`]: +//! +//! | Id | Engine | Endpoint | Credential | +//! | --- | --- | --- | --- | +//! | `cortexdb` | CortexDB's own `/v1/*` API | defaults to the managed API | API key | +//! | `tinyhumans` | CortexDB behind the TinyHumans backend `/memory/*` | defaults to `api.tinyhumans.ai` | session JWT or `tiny_live_` key, usually dynamic | +//! +//! [`build_engine`] refuses an unknown id, a missing required endpoint or +//! credential, and a credentialed cleartext endpoint that is not loopback, +//! all as [`Error::Config`]. Messages never carry the credential. + +use std::sync::Arc; + +use crate::cortex::{ + BearerSource, CORTEXDB_ENGINE_ID, CortexCredential, CortexEngine, CortexWire, + TINYHUMANS_ENGINE_ID, +}; +use tinymemory_api::{EngineDescriptor, Error, MemoryEngine, Result}; + +use crate::config::EngineSettings; + +/// How an engine authenticates. +#[derive(Clone, Default)] +pub enum EngineCredential { + /// No credential, for an engine that needs none. + #[default] + None, + /// One fixed token, for example an API key. + Static(String), + /// A token resolved before every request, so a refreshed session is used + /// at once. + Dynamic(Arc<dyn BearerSource>), +} + +impl std::fmt::Debug for EngineCredential { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::None => "EngineCredential::None", + Self::Static(_) => "EngineCredential::Static(<redacted>)", + Self::Dynamic(_) => "EngineCredential::Dynamic(<source>)", + }) + } +} + +/// Every engine this build can construct. +#[must_use] +pub fn list_engines() -> Vec<EngineDescriptor> { + vec![ + crate::cortex::cortexdb_descriptor(), + crate::cortex::tinyhumans_descriptor(), + ] +} + +/// Builds the engine `id` from `settings` and `credential`. +/// +/// An unset or blank endpoint falls back to the engine's default. Every +/// registered engine is credentialed, so a credential is always required; +/// the endpoint's own checks (an HTTP(S) URL, and no cleartext off loopback) +/// are [`CortexEngine::new`]'s. +/// +/// # Errors +/// +/// [`Error::Config`] for an unknown id, a missing endpoint or credential, an +/// endpoint that is not an HTTP(S) URL, or a credentialed cleartext +/// (`http://`) endpoint that is not loopback. +pub fn build_engine( + id: &str, + settings: &EngineSettings, + credential: EngineCredential, +) -> Result<Arc<dyn MemoryEngine>> { + let wire = match id { + CORTEXDB_ENGINE_ID => CortexWire::Direct, + TINYHUMANS_ENGINE_ID => CortexWire::TinyHumans, + _ => return Err(Error::Config(format!("unknown memory engine `{id}`"))), + }; + let endpoint = settings + .endpoint + .as_deref() + .map(str::trim) + .filter(|endpoint| !endpoint.is_empty()) + .or(wire.descriptor().default_endpoint) + .ok_or_else(|| Error::Config(format!("memory engine `{id}` needs an endpoint")))?; + let credential = match credential { + EngineCredential::Static(token) if !token.trim().is_empty() => { + CortexCredential::Static(token) + } + EngineCredential::Dynamic(source) => CortexCredential::Dynamic(source), + EngineCredential::Static(_) | EngineCredential::None => { + return Err(Error::Config(format!( + "memory engine `{id}` needs a credential" + ))); + } + }; + Ok(Arc::new(CortexEngine::new(wire, endpoint, credential)?)) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory/src/registry/mod_tests.rs b/crates/tinymemory-integrations/src/registry/mod_tests.rs similarity index 100% rename from crates/tinymemory/src/registry/mod_tests.rs rename to crates/tinymemory-integrations/src/registry/mod_tests.rs diff --git a/crates/tinymemory-integrations/src/safety/README.md b/crates/tinymemory-integrations/src/safety/README.md new file mode 100644 index 00000000..713116b2 --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/README.md @@ -0,0 +1,221 @@ +# `safety` — secret and PII scrubbing + +`tinymemory_integrations::safety` (feature `safety`) removes credentials and +personal identifiers from text before a memory host stores it. It is +conservative by design: it would rather redact a harmless string than let a +token or a national ID into a long-lived store. It runs on-device, uses regular +expressions and checksums only, and makes no network calls. + +The usual call is [`scrub_item`] on each `StoreItem` just before +`MemoryEngine::store`. + +Nothing calls it for you: no engine scrubs on its own, so the host runs it in +the write pipeline (source, documents, safety, engine). See +[`docs/architecture/integrations.md`](../../../../docs/architecture/integrations.md). + +## Layout + +```text +safety/ +├── mod.rs # module docs, `mod` declarations, public re-exports +├── policy/ # Policy, BareCardGate, SanitizationReport, Sanitized +├── sanitize/ # sanitize_text[_with], sanitize_json[_with], has_likely_secret +│ └── patterns.rs # private-key block and credential shape tables +├── markers/ # redact_credential_markers: `/secret/<key>`, `Bearer <value>` +├── pii/ # redact_pii[_with], has_likely_pii, has_likely_email +│ ├── prefilter.rs # byte pass deciding which identifier classes run +│ ├── normalize.rs # fullwidth / zero-width / Arabic-Indic folding +│ └── checks.rs # Luhn, mod-97, Verhoeff, CPF/CNPJ/CUIT, DNI/NIE, SSN, NINO +├── item/ # scrub_item[_with]: applies the scrubber to a StoreItem +└── pattern/ # literal(): the one place a built-in regex is compiled +``` + +## Public surface + +| Item | Purpose | +| --- | --- | +| `scrub_item`, `scrub_item_with` | Scrub every free text a `StoreItem` carries | +| `sanitize_text`, `sanitize_text_with` | Scrub a string: secrets, then PII | +| `sanitize_json`, `sanitize_json_with` | Scrub a JSON value recursively | +| `has_likely_secret` | Boolean check: does this text look like it holds a credential | +| `has_likely_pii` | Strict boundary check used to *reject* namespaces and keys | +| `has_likely_email` | Boundary check for an ordinary email address | +| `redact_credential_markers` | Only the marker rules, without the PII pass | +| `pii::redact_pii`, `pii::redact_pii_with` | Only the PII pass | +| `Policy`, `BareCardGate` | The single tunable | +| `Sanitized<T>`, `SanitizationReport` | Cleaned value plus a tally of changes | + +None of these functions fail or panic at runtime. Every scrubber returns a +`Sanitized<T>`; `report.changed()` says whether anything was replaced. + +## What is blocked and what is redacted + +A text pass in `sanitize_text_with` runs four stages, in this order: + +1. **Private-key blocks are blocked.** A PEM, OpenSSH or PGP private-key block + is replaced in full by `[REDACTED_PRIVATE_KEY]` and counted in + `blocked_secret_hits`. Nothing of the block survives. +2. **Credential markers.** The value after `/secret/` in a one-time-secret URL, + and after a `Bearer ` scheme, becomes `[REDACTED]`. The marker and the + surrounding prose stay, so memory still records that a link or token was + shared (see below). +3. **Credential shapes are redacted.** Provider token prefixes (`sk-`, + `sk-ant-`, `ghp_`, `github_pat_`, `glpat-`, `xox?-`, `AKIA`/`ASIA`, `AIza`, + `npm_`, `SG.`, Stripe `sk_live_`/`rk_test_` and so on), JWTs, and + `key=value` assignments whose key is `api_key`, `token`, `password`, + `secret`, `client_secret` or an OAuth parameter. The matched span becomes + `[REDACTED]`; for `Bearer` and `api_key` the prefix is kept. Counted in + `text_redactions`. +4. **PII is redacted.** The `pii` pass replaces each identifier with a typed + token such as `[REDACTED_PII_CPF]` or `[REDACTED_PII_CREDIT_CARD]`. Counted + in `pii_redactions`. + +`sanitize_json_with` walks objects and arrays: + +- A value whose **key** looks sensitive is replaced by `[REDACTED_SECRET]` + without being read. Keys are compared lowercased with non-alphanumerics + removed, so `API-Key`, `api_key` and `apiKey` match alike. Exact names + (`apikey`, `token`, `authorization`, `password`, `secret`, `clientsecret`, …) + match, as does any key that ends in `token`, `apikey`, `clientsecret` or + `key`, or contains `password` or `secret`. Counted in `key_redactions`. +- Every other string runs through `sanitize_text_with`. +- Numbers, booleans and null pass through untouched. +- Nesting deeper than 128 levels is not walked: the subtree is replaced by + `[REDACTED_SECRET]` and counted in `depth_redactions`. + +The boolean checks are separate from scrubbing. `has_likely_secret` tests the +block and shape tables (not the markers). `has_likely_pii` uses a stricter +pattern set than content scrubbing; see the PII section. + +## The `Policy` knob + +`Policy` has one field, `bare_card: BareCardGate`. It decides how a bare +(separator-free) Luhn-valid 13-19 digit run is judged as a credit card: + +- `BareCardGate::LuhnOnly` (the default) redacts every Luhn-valid run. This is + the strictest setting and what the plain functions use. +- `BareCardGate::Corroborated` additionally requires a real network IIN at an + issued length, or a card keyword within 64 bytes (`card`, `cc`, `pan`, + `cardNumber`, `信用卡`, `カード`, …). `Policy::corroborated()` builds it. + +Separated runs (`4111 1111 1111 1111`) are Luhn-gated under both settings. The +corroborated gate exists because Luhn passes about one in ten arbitrary digit +runs, and 13-digit epoch-millisecond timestamps in stored JSON envelopes were +being corrupted at that rate (opencompany#1201). A host scrubbing items that +carry such timestamps should opt in; a caller that does not never redacts less +than before. + +## PII detection pipeline + +`pii::redact_pii_with` runs in three steps. + +1. **Normalize.** `NormalizedView` builds a copy of the text with zero-width + characters (U+200B/200C/200D/FEFF/2060/180E) removed, fullwidth digits and + `.-/:` folded to ASCII, and Arabic-Indic digits folded to ASCII. It keeps + a byte map back to the original, so `111.444…` or a digit run with + zero-width spaces inside cannot slip past. +2. **Prefilter.** `scan_candidates` makes one cheap pass over the bytes and + sets a flag per identifier class from structural signals: digit-run + lengths, punctuation, letters, `+`, and keyword probes. Every flag is a + necessary condition of its class's precise regex, so it can over-fire but + never under-fire. A class whose flag is unset is skipped, and its regex is + never compiled. +3. **Match and check.** The precise regex of each flagged class runs on the + normalized text, in priority order, and each candidate passes its checksum + or structural gate: + + | Class | Gate | + | --- | --- | + | Brazil CPF / CNPJ (formatted or bare) | mod-11 check digits | + | Argentina CUIT/CUIL (formatted only) | check digit | + | Credit card | Luhn, plus `Policy` for bare runs | + | IBAN | mod-97 | + | India Aadhaar | Verhoeff when grouped; keyword when bare | + | Spain DNI / NIE | check letter | + | US SSN | reserved-range filters | + | UK NINO | reserved-prefix filters | + | Japan My Number | keyword nearby | + | Mexico RFC, India PAN, Korea RRN | format only | + | Phone: E.164, NANP | format (NANP area/exchange rules) | + +Overlapping hits are resolved earliest-and-longest first, so a card number is +not also partly redacted as a phone number. The kept hits are spliced back onto +the **original** bytes through the byte map; text that is not PII, including +fullwidth glyphs a user typed on purpose, is left exactly as it was. + +### The strict boundary check + +`has_likely_pii` decides whether to *reject* a namespace or key, not whether to +rewrite content. It runs the same pipeline but leaves out the patterns whose +only signal is a digit-run shape: bare credit cards, bare CPF/CNPJ, NANP and +E.164 phones. Scanner-built identifiers (WhatsApp JIDs such as +`12025551234-1543890267@g.us`, Telegram peer IDs, millisecond timestamps, +padded counters) would otherwise be rejected constantly. Formatted national IDs +are still rejected. `has_likely_email` is kept apart for the same reason: +identifiers can contain email-like `@` segments legitimately. + +## Credential markers + +Two leaks get past shape-based matching. Both were seen in a live OpenCompany +deployment that remembered every operator message verbatim: + +- **One-time-secret URLs** such as `https://ots.example/secret/<key>`. The key + is the credential, and it doesn't look like one. The value after `/secret/` + is redacted; the match is exact and lowercase. +- **Short `Bearer` values.** `Bearer s3cret` is a valid credential, but the + shape regex needs eight characters. The `Bearer` rule matches the scheme + case-insensitively and is tuned against the English word, so "ring bearer" + and "bearer bond" are left alone. + +Only the value after the marker is replaced. `redact_credential_markers` runs +just these two rules and returns the input borrowed, without allocating, when +neither marker is present. `sanitize_text` runs them before the shape regexes, +and its `[REDACTED]` is not token-shaped, so a later stage cannot match it +again. + +## `scrub_item` per `StoreItem` kind + +`scrub_item_with(item, policy)` runs `sanitize_text_with` over each free text +an item carries and merges the reports: + +| Kind | Scrubbed | Left alone | +| --- | --- | --- | +| `Document` | `title`, a `DocumentBody::Text` body | a `DocumentBody::Uri` body | +| `Conversation` | every turn's `text` | turn roles | +| `Learning` | `text`, `evidence` | the learning kind | +| any (metadata) | `meta.url` (query strings carry tokens) | paths, repository, commit, thread and agent ids | + +Identifiers in the metadata stay as they are because filters match on them, and +rewriting them would make an item impossible to find. A `Uri` body is left +alone because sources resolve it to text before storing, and that text is +scrubbed at that point. `scrub_item` is `scrub_item_with` under the default, +strictest `Policy`. + +## Known limits + +- **Pattern-based only.** Contextual PII ("call me at the usual number"), + combinations (name + employer + city), personal names and free-form dates of + birth need NER or an LLM and are not handled here. +- **Email addresses are detected, not redacted.** `has_likely_email` reports + them; the content scrubber leaves them in place. +- **Bare 10-digit NANP numbers are not redacted.** A NANP phone needs + separators or a leading `1` country code to reach its regex. +- **Unknown credential formats get through.** A token without a known prefix, + a key-like name, or a marker in front of it is not recognised. Under + `sanitize_json` a sensitive *key* name still catches it. +- **False positives are accepted.** Shape rules can rewrite harmless text, such + as a 13-digit timestamp under `LuhnOnly`, or a JSON field whose name ends in + `key`. A redaction inside structured content can corrupt it for whoever wrote + it; `Policy::corroborated()` limits that for card-shaped numbers. +- **Not reversible.** Redacted values are dropped, not stored somewhere else. + The report gives counts, never the original values. + +## Testing + +Unit tests sit beside each module (`mod_tests.rs`, plus +`sanitize/mod_default_policy_tests.rs` and `pii/mod_prefilter_tests.rs`). They +force every `LazyLock` pattern, so a typo in a built-in regex fails CI rather +than a host. Credential fixtures are assembled at run time so repository secret +scanners don't flag them. + +[`scrub_item`]: item/mod.rs diff --git a/crates/tinymemory-safety/src/item.rs b/crates/tinymemory-integrations/src/safety/item/mod.rs similarity index 95% rename from crates/tinymemory-safety/src/item.rs rename to crates/tinymemory-integrations/src/safety/item/mod.rs index e1cedd75..29b9289d 100644 --- a/crates/tinymemory-safety/src/item.rs +++ b/crates/tinymemory-integrations/src/safety/item/mod.rs @@ -10,7 +10,7 @@ use tinymemory_api::{DocumentBody, StoreItem}; -use crate::{sanitize_text_with, Policy, SanitizationReport, Sanitized}; +use crate::safety::{Policy, SanitizationReport, Sanitized, sanitize_text_with}; /// Scrubs every text `item` carries under the default (strictest) [`Policy`]. #[must_use] @@ -58,5 +58,5 @@ pub fn scrub_item_with(mut item: StoreItem, policy: Policy) -> Sanitized<StoreIt } #[cfg(test)] -#[path = "item_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-safety/src/item_tests.rs b/crates/tinymemory-integrations/src/safety/item/mod_tests.rs similarity index 100% rename from crates/tinymemory-safety/src/item_tests.rs rename to crates/tinymemory-integrations/src/safety/item/mod_tests.rs diff --git a/crates/tinymemory-safety/src/markers.rs b/crates/tinymemory-integrations/src/safety/markers/mod.rs similarity index 96% rename from crates/tinymemory-safety/src/markers.rs rename to crates/tinymemory-integrations/src/safety/markers/mod.rs index 817fc533..ebedad07 100644 --- a/crates/tinymemory-safety/src/markers.rs +++ b/crates/tinymemory-integrations/src/safety/markers/mod.rs @@ -1,6 +1,6 @@ //! Credential-marker rules: values that follow an unambiguous marker. //! -//! The regex set in [`crate::sanitize_text`] recognises credentials by their +//! The regex set in [`crate::safety::sanitize_text`] recognises credentials by their //! own shape — a vendor prefix, a JWT's three segments, eight or more token //! characters after `Bearer`. Two leaks get past shape alone, both measured in //! a live OpenCompany deployment where every operator message was remembered @@ -35,11 +35,11 @@ const BEARER_MARKER: &str = "bearer "; /// /// The entry point for a host that scrubs plain text on its way into memory /// and wants exactly these two rules — without the PII pass and the broader -/// token regexes of [`crate::sanitize_text`], which applies these rules too. +/// token regexes of [`crate::safety::sanitize_text`], which applies these rules too. /// Text with neither marker is returned borrowed, without allocating. /// /// ``` -/// use tinymemory_safety::redact_credential_markers; +/// use tinymemory_integrations::safety::redact_credential_markers; /// /// assert_eq!( /// redact_credential_markers("open https://ots.example/secret/AbC123 now"), @@ -59,7 +59,7 @@ pub fn redact_credential_markers(text: &str) -> Cow<'_, str> { } /// [`redact_credential_markers`] plus the number of values it replaced, for -/// the [`crate::SanitizationReport`]. +/// the [`crate::safety::SanitizationReport`]. pub(crate) fn redact_counted(text: &str) -> (Cow<'_, str>, usize) { if !text.contains(SECRET_URL_MARKER) && find_bearer_marker(text).is_none() { return (Cow::Borrowed(text), 0); @@ -191,5 +191,5 @@ fn is_token_char(c: char) -> bool { } #[cfg(test)] -#[path = "markers_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-safety/src/markers_tests.rs b/crates/tinymemory-integrations/src/safety/markers/mod_tests.rs similarity index 100% rename from crates/tinymemory-safety/src/markers_tests.rs rename to crates/tinymemory-integrations/src/safety/markers/mod_tests.rs diff --git a/crates/tinymemory-integrations/src/safety/mod.rs b/crates/tinymemory-integrations/src/safety/mod.rs new file mode 100644 index 00000000..365d305f --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/mod.rs @@ -0,0 +1,54 @@ +//! Secret and PII scrubbing for anything a memory host persists or hands on. +//! +//! Conservative by design — it prefers false positives over leaking +//! credentials into long-lived stores. One copy of this policy is shared by the +//! memory engines and the OpenHuman host; it used to exist three times. The +//! design, what is blocked versus redacted, and the known limits are in this +//! module's `README.md`. +//! +//! [`scrub_item`] applies the policy to every text a +//! [`tinymemory_api::StoreItem`] carries, and is what a host runs on each item +//! before `MemoryEngine::store`. +//! +//! The exhaustive multilingual national-ID PII module ([`pii`]) runs as part +//! of [`sanitize_text`]. The write-rejection boundary ([`has_likely_pii`]) +//! stays stricter than content scrubbing: formatted national IDs are rejected, +//! while phone-like text is scrubbed from content without rejecting every +//! write that mentions it. Email addresses are only detected +//! ([`has_likely_email`]), never redacted from content. +//! +//! Before the shape regexes, [`sanitize_text`] redacts the value after a +//! credential *marker* — a one-time-secret URL's `/secret/<key>` and a `Bearer` +//! value too short for the regexes — keeping the marker and the prose around +//! it. [`redact_credential_markers`] runs just those rules, for a host that +//! scrubs plain text without the PII pass. +//! +//! # The one policy knob +//! +//! The previous copies differed in exactly one behaviour: how a *bare* +//! (separator-less) Luhn-valid 13-19 digit run is treated as a credit card. +//! The OpenHuman host redacted every such run; TinyCortex additionally demanded +//! corroboration (a real network IIN at an issued length, or a card keyword +//! nearby) so 13-digit epoch-millisecond timestamps in stored JSON envelopes +//! stopped being corrupted (opencompany#1201). [`BareCardGate`] names both and +//! the plain functions default to the stricter [`BareCardGate::LuhnOnly`], so no +//! caller that does not opt in redacts less than before. Callers that want the +//! corroborated behaviour use the `*_with` variants and [`Policy::corroborated`]. + +mod item; +mod markers; +mod pattern; +/// Exhaustive checksum-gated multilingual national-ID PII module. Content +/// scrubbing runs from [`sanitize_text`]; the boundary check is re-exported as +/// [`has_likely_pii`]. +pub mod pii; +mod policy; +mod sanitize; + +pub use item::{scrub_item, scrub_item_with}; +pub use markers::redact_credential_markers; +pub use pii::{has_likely_email, has_likely_pii}; +pub use policy::{BareCardGate, Policy, SanitizationReport, Sanitized}; +pub use sanitize::{ + has_likely_secret, sanitize_json, sanitize_json_with, sanitize_text, sanitize_text_with, +}; diff --git a/crates/tinymemory-integrations/src/safety/pattern/mod.rs b/crates/tinymemory-integrations/src/safety/pattern/mod.rs new file mode 100644 index 00000000..e2785798 --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/pattern/mod.rs @@ -0,0 +1,24 @@ +//! Compiling the scrubber's built-in regular expressions. +//! +//! Every credential and PII pattern in this module is a string literal +//! compiled once into a `LazyLock`. They are compiled through [`literal`] so +//! the one place a pattern could fail to compile is named, documented and +//! covered by tests: the scrubbing tests force every `LazyLock`, so a typo in +//! a pattern fails CI rather than a host. + +use regex::Regex; + +/// Compiles a built-in pattern. +/// +/// # Panics +/// +/// Panics if `pattern` is not a valid regular expression. Every caller passes +/// a literal that the safety tests compile, so this is unreachable in a +/// released build. +#[allow( + clippy::expect_used, + reason = "patterns are compile-time literals exercised by the safety tests" +)] +pub(super) fn literal(pattern: &str) -> Regex { + Regex::new(pattern).expect("a built-in safety pattern is a valid regex") +} diff --git a/crates/tinymemory-safety/src/pii/checks.rs b/crates/tinymemory-integrations/src/safety/pii/checks.rs similarity index 97% rename from crates/tinymemory-safety/src/pii/checks.rs rename to crates/tinymemory-integrations/src/safety/pii/checks.rs index 94305e3b..e781295a 100644 --- a/crates/tinymemory-safety/src/pii/checks.rs +++ b/crates/tinymemory-integrations/src/safety/pii/checks.rs @@ -1,10 +1,7 @@ //! Checksum and structural validators for PII candidates. pub(crate) fn digits(s: &str) -> Vec<u32> { - s.chars() - .filter(|c| c.is_ascii_digit()) - .map(|c| c.to_digit(10).expect("ascii digit")) - .collect() + s.chars().filter_map(|c| c.to_digit(10)).collect() } pub(crate) fn valid_cpf(d: &[u32]) -> bool { @@ -65,11 +62,7 @@ pub(crate) fn valid_luhn(s: &str) -> bool { for x in d.iter().rev() { let v = if alt { let doubled = x * 2; - if doubled > 9 { - doubled - 9 - } else { - doubled - } + if doubled > 9 { doubled - 9 } else { doubled } } else { *x }; diff --git a/crates/tinymemory-safety/src/pii/checks_tests.rs b/crates/tinymemory-integrations/src/safety/pii/checks_tests.rs similarity index 99% rename from crates/tinymemory-safety/src/pii/checks_tests.rs rename to crates/tinymemory-integrations/src/safety/pii/checks_tests.rs index e05291e9..ebcafadf 100644 --- a/crates/tinymemory-safety/src/pii/checks_tests.rs +++ b/crates/tinymemory-integrations/src/safety/pii/checks_tests.rs @@ -1,3 +1,5 @@ +//! Checksum validators and the card-network plausibility check. + use super::*; #[test] diff --git a/crates/tinymemory-safety/src/pii.rs b/crates/tinymemory-integrations/src/safety/pii/mod.rs similarity index 86% rename from crates/tinymemory-safety/src/pii.rs rename to crates/tinymemory-integrations/src/safety/pii/mod.rs index 5be220dd..6e313510 100644 --- a/crates/tinymemory-safety/src/pii.rs +++ b/crates/tinymemory-integrations/src/safety/pii/mod.rs @@ -30,18 +30,16 @@ use regex::Regex; use std::sync::LazyLock; -use super::{BareCardGate, Policy, SanitizationReport, Sanitized}; +use crate::safety::pattern::literal; +use crate::safety::policy::{BareCardGate, Policy, SanitizationReport, Sanitized}; mod checks; -use checks::*; +mod normalize; +mod prefilter; -// Flattened test-only re-exports so the crate's test modules can exercise the -// internals (checksum validators, the normalization pass, the candidate scan). -#[cfg(test)] -pub(crate) use checks::{ - digits, valid_cnpj, valid_cpf, valid_cuit, valid_dni_es, valid_iban, valid_luhn, valid_nie_es, - valid_nino, valid_ssn, valid_verhoeff, -}; +use checks::*; +pub(crate) use normalize::NormalizedView; +pub(crate) use prefilter::{Candidates, scan_candidates}; // ---------- Replacement tokens ---------- @@ -63,49 +61,40 @@ pub(crate) const PII_RRN: &str = "[REDACTED_PII_RRN]"; // ---------- Patterns ---------- // Brazilian CPF, formatted: NNN.NNN.NNN-NN -static CPF_FMT_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{3}\.\d{3}\.\d{3}-\d{2}\b").expect("cpf fmt")); +static CPF_FMT_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{3}\.\d{3}\.\d{3}-\d{2}\b")); // Brazilian CPF, bare: 11 consecutive digits. Checksum-gated; ~1% raw FP. -static CPF_BARE_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{11}\b").expect("cpf bare")); +static CPF_BARE_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{11}\b")); // Brazilian CNPJ, formatted: NN.NNN.NNN/NNNN-NN static CNPJ_FMT_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{2}\.\d{3}\.\d{3}/\d{4}-\d{2}\b").expect("cnpj fmt")); + LazyLock::new(|| literal(r"\b\d{2}\.\d{3}\.\d{3}/\d{4}-\d{2}\b")); // Brazilian CNPJ, bare: 14 consecutive digits. -static CNPJ_BARE_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{14}\b").expect("cnpj bare")); +static CNPJ_BARE_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{14}\b")); // Argentine CUIT/CUIL: NN-NNNNNNNN-N (formatted only — bare 11-digit with // single check digit has ~9% FP on random IDs, too noisy without context). -static CUIT_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{2}-\d{8}-\d\b").expect("cuit")); +static CUIT_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{2}-\d{8}-\d\b")); // Mexican RFC: 3-4 letters (incl. Ñ &) + 6 digits + 3 alphanumeric homoclave. -static RFC_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"(?i)\b[A-ZÑ&]{3,4}\d{6}[A-Z0-9]{3}\b").expect("rfc")); +static RFC_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"(?i)\b[A-ZÑ&]{3,4}\d{6}[A-Z0-9]{3}\b")); // Japan My Number (12 digits) gated by a Japanese or English keyword within // ~30 chars. Bare 12-digit runs without keyword are too noisy. static MYNUM_RE: LazyLock<Regex> = LazyLock::new(|| { - Regex::new(r"(?:マイナンバー|個人番号|My\s?Number)[\s:はがを、.\-]{0,12}(\d{12})\b") - .expect("my number") + literal(r"(?:マイナンバー|個人番号|My\s?Number)[\s:はがを、.\-]{0,12}(\d{12})\b") }); // E.164 phone: + followed by 7-15 digits, no separators. -static PHONE_E164_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\+\d{7,15}\b").expect("e164")); +static PHONE_E164_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\+\d{7,15}\b")); // NANP (US/Canada) formatted phone. Area code must start 2-9; first digit of // central-office code also 2-9 (real NANP rule). static PHONE_NANP_RE: LazyLock<Regex> = LazyLock::new(|| { - Regex::new(r"\b(?:\+?1[\s.\-]?)?\(?([2-9]\d{2})\)?[\s.\-]?([2-9]\d{2})[\s.\-]?(\d{4})\b") - .expect("nanp phone") + literal(r"\b(?:\+?1[\s.\-]?)?\(?([2-9]\d{2})\)?[\s.\-]?([2-9]\d{2})[\s.\-]?(\d{4})\b") }); // US SSN: NNN-NN-NNNN. Range filter applied below. -static SSN_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{3}-\d{2}-\d{4}\b").expect("ssn")); +static SSN_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{3}-\d{2}-\d{4}\b")); // Credit card: 13-19 digits with optional spaces/dashes every 4. Every match // is Luhn-gated; a match with no separators at all additionally needs @@ -114,8 +103,7 @@ static SSN_RE: LazyLock<Regex> = // 13-digit epoch-millisecond timestamps were being redacted out of stored // JSON envelopes at exactly that rate (opencompany#1201). Same split as // Aadhaar below: formatted keeps the checksum-only gate, bare needs more. -static CC_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b(?:\d[\s\-]?){13,19}\b").expect("credit card")); +static CC_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b(?:\d[\s\-]?){13,19}\b")); // Card keyword corroborating a bare digit run. Three tiers, matched // case-insensitively: @@ -136,52 +124,38 @@ static CC_RE: LazyLock<Regex> = // directly attached: there `CC_RE`'s own leading `\b` already fails // (CJK is `\w`), so the run is never a candidate in the first place. static CC_KEYWORD_RE: LazyLock<Regex> = LazyLock::new(|| { - Regex::new( + literal( r"(?i)(?:^|[\W_])(?:card|credit|debit|visa|mastercard|amex|american\s?express|discover|jcb|diners|unionpay|hipercard|rupay|cvv|cvc|cc|pan|tarjeta|cart[aã]o|carte|karte|карта|карты|карту|картой|карте|кредитка)(?:[\W_]|$)|(?i:cardnumber|creditcard|ccnum|cardno|pannumber|カード|信用卡|卡号|银行卡|카드)", ) - .expect("cc keyword") }); // IBAN: 2 letter country code + 2 check digits + 11-30 alphanumeric. // Allow optional spaces every 4 chars (common human format). static IBAN_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b[A-Z]{2}\d{2}(?:[\s]?[A-Z0-9]){11,30}\b").expect("iban")); + LazyLock::new(|| literal(r"\b[A-Z]{2}\d{2}(?:[\s]?[A-Z0-9]){11,30}\b")); // India Aadhaar: 4-4-4 digit groups (space or hyphen) OR contiguous 12 digits // gated by keyword. Verhoeff-checksum-gated when grouped, keyword-gated when // bare (Verhoeff alone has ~10% raw FP rate on random 12-digit runs). static AADHAAR_FMT_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{4}[\s\-]\d{4}[\s\-]\d{4}\b").expect("aadhaar formatted")); -static AADHAAR_KW_RE: LazyLock<Regex> = LazyLock::new(|| { - Regex::new(r"(?i)(?:aadhaar|aadhar|आधार|uidai|uid)[\s:#\-no.]{0,10}(\d{12})\b") - .expect("aadhaar keyword") -}); + LazyLock::new(|| literal(r"\b\d{4}[\s\-]\d{4}[\s\-]\d{4}\b")); +static AADHAAR_KW_RE: LazyLock<Regex> = + LazyLock::new(|| literal(r"(?i)(?:aadhaar|aadhar|आधार|uidai|uid)[\s:#\-no.]{0,10}(\d{12})\b")); // India PAN: 5 letters, 4 digits, 1 letter. Very high signal — no checksum. -static PAN_IN_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z]{5}\d{4}[A-Z]\b").expect("pan-in")); +static PAN_IN_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"(?i)\b[A-Z]{5}\d{4}[A-Z]\b")); // UK NINO: 2 letters + 6 digits + suffix A/B/C/D. -static NINO_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z]{2}\d{6}[A-D]\b").expect("nino")); +static NINO_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"(?i)\b[A-Z]{2}\d{6}[A-D]\b")); // Spain DNI: 8 digits + check letter. NIE: starts X/Y/Z, then 7 digits + letter. -static DNI_RE: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"(?i)\b\d{8}[A-Z]\b").expect("dni")); -static NIE_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"(?i)\b[XYZ]\d{7}[A-Z]\b").expect("nie")); +static DNI_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"(?i)\b\d{8}[A-Z]\b")); +static NIE_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"(?i)\b[XYZ]\d{7}[A-Z]\b")); // South Korea RRN: NNNNNN-CXXXXXX where C is gender/century digit (1-4). -static RRN_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"\b\d{6}-[1-4]\d{6}\b").expect("rrn")); +static RRN_RE: LazyLock<Regex> = LazyLock::new(|| literal(r"\b\d{6}-[1-4]\d{6}\b")); static EMAIL_RE: LazyLock<Regex> = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b").expect("email")); - -// ---------- Byte-oriented candidate pre-filter ---------- -// -// The single cheap byte pass that replaces the always-resident combined -// `RegexSet`. Lives in its own module — see `prefilter.rs` for the full rationale. -mod prefilter; -pub(crate) use prefilter::{scan_candidates, Candidates}; + LazyLock::new(|| literal(r"(?i)\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b")); // ---------- Public API ---------- @@ -581,15 +555,10 @@ fn splice_redactions( } } -// ---------- Unicode normalization for matching ---------- - -// Fullwidth / zero-width normalization used before matching. Lives in its own -// module — see `normalize.rs`. -mod normalize; -pub(crate) use normalize::NormalizedView; - -// ---------- Checksum helpers ---------- - #[cfg(test)] -#[path = "pii_tests.rs"] +#[path = "mod_tests.rs"] mod tests; + +#[cfg(test)] +#[path = "mod_prefilter_tests.rs"] +mod prefilter_tests; diff --git a/crates/tinymemory-safety/src/default_policy_prefilter_tests.rs b/crates/tinymemory-integrations/src/safety/pii/mod_prefilter_tests.rs similarity index 97% rename from crates/tinymemory-safety/src/default_policy_prefilter_tests.rs rename to crates/tinymemory-integrations/src/safety/pii/mod_prefilter_tests.rs index a6fcf0e6..d956d515 100644 --- a/crates/tinymemory-safety/src/default_policy_prefilter_tests.rs +++ b/crates/tinymemory-integrations/src/safety/pii/mod_prefilter_tests.rs @@ -1,3 +1,6 @@ +//! The byte prefilter against the legacy screen, and the checksum validators +//! at their length, range and repetition bounds. + use super::*; /// Parity oracle: the new byte prefilter must be a SUPERSET of the legacy diff --git a/crates/tinymemory-safety/src/pii_tests.rs b/crates/tinymemory-integrations/src/safety/pii/mod_tests.rs similarity index 99% rename from crates/tinymemory-safety/src/pii_tests.rs rename to crates/tinymemory-integrations/src/safety/pii/mod_tests.rs index f7df65dc..df2899ef 100644 --- a/crates/tinymemory-safety/src/pii_tests.rs +++ b/crates/tinymemory-integrations/src/safety/pii/mod_tests.rs @@ -1,3 +1,6 @@ +//! PII redaction per identifier class, normalization bypasses, the strict +//! boundary check and the candidate prefilter, under the corroborated policy. + use super::*; /// These tests were written against the TinyCortex engine, whose content diff --git a/crates/tinymemory-safety/src/pii/normalize.rs b/crates/tinymemory-integrations/src/safety/pii/normalize.rs similarity index 100% rename from crates/tinymemory-safety/src/pii/normalize.rs rename to crates/tinymemory-integrations/src/safety/pii/normalize.rs diff --git a/crates/tinymemory-safety/src/pii/prefilter.rs b/crates/tinymemory-integrations/src/safety/pii/prefilter.rs similarity index 100% rename from crates/tinymemory-safety/src/pii/prefilter.rs rename to crates/tinymemory-integrations/src/safety/pii/prefilter.rs diff --git a/crates/tinymemory-integrations/src/safety/policy/mod.rs b/crates/tinymemory-integrations/src/safety/policy/mod.rs new file mode 100644 index 00000000..aedfabc5 --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/policy/mod.rs @@ -0,0 +1,89 @@ +//! The scrubber's one tunable and the types every scrubbing pass returns. +//! +//! [`Policy`] carries the single knob the historical copies of this scrubber +//! disagreed on — [`BareCardGate`] — and defaults to the strictest setting. +//! [`Sanitized`] pairs a cleaned value with the [`SanitizationReport`] that +//! counts what was changed to produce it. + +/// How a bare (no separators) Luhn-valid 13-19 digit run is judged as a credit +/// card by the content scrubber. Separated runs (`4111 1111 1111 1111`) are +/// always Luhn-gated only. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum BareCardGate { + /// Redact every Luhn-valid run. The strictest behaviour and the default. + #[default] + LuhnOnly, + /// Also require a plausible network IIN at an issued length, or a card + /// keyword within 64 bytes, so machine identifiers such as 13-digit + /// epoch-millisecond timestamps are left alone. + Corroborated, +} + +/// Tunables for content scrubbing. The default never redacts less than +/// [`BareCardGate::LuhnOnly`]. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct Policy { + /// Gate applied to bare credit-card-shaped digit runs. + pub bare_card: BareCardGate, +} + +impl Policy { + /// The policy the TinyCortex engine has always applied: bare card runs need + /// corroboration beyond their checksum. + pub const fn corroborated() -> Self { + Self { + bare_card: BareCardGate::Corroborated, + } + } +} + +/// Tally of what a sanitization pass changed. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct SanitizationReport { + /// Count of secret/token pattern matches rewritten in string text by the + /// text-pattern redaction pass. + pub text_redactions: usize, + /// Count of JSON object entries dropped wholesale because their key was + /// classified as sensitive by the key classifier. + pub key_redactions: usize, + /// Count of full private-key blocks replaced; these are + /// the most severe hits since the entire block is removed. + pub blocked_secret_hits: usize, + /// Count of nodes collapsed because JSON nesting reached + /// the JSON traversal depth cap; the subtree is replaced rather than walked. + pub depth_redactions: usize, + /// Count of personal-identifier matches replaced by the + /// lightweight PII screen. + pub pii_redactions: usize, +} + +impl SanitizationReport { + /// True when any field recorded a redaction. + pub fn changed(&self) -> bool { + self.text_redactions > 0 + || self.key_redactions > 0 + || self.blocked_secret_hits > 0 + || self.depth_redactions > 0 + || self.pii_redactions > 0 + } + + /// Sum two reports field-wise. + pub fn merge(self, rhs: Self) -> Self { + Self { + text_redactions: self.text_redactions + rhs.text_redactions, + key_redactions: self.key_redactions + rhs.key_redactions, + blocked_secret_hits: self.blocked_secret_hits + rhs.blocked_secret_hits, + depth_redactions: self.depth_redactions + rhs.depth_redactions, + pii_redactions: self.pii_redactions + rhs.pii_redactions, + } + } +} + +/// A sanitized value plus the [`SanitizationReport`] describing the changes. +#[derive(Debug, Clone)] +pub struct Sanitized<T> { + /// The cleaned value with secrets and PII removed. + pub value: T, + /// Tally of what the sanitization pass changed to produce `value`. + pub report: SanitizationReport, +} diff --git a/crates/tinymemory-integrations/src/safety/sanitize/mod.rs b/crates/tinymemory-integrations/src/safety/sanitize/mod.rs new file mode 100644 index 00000000..ce0b5ed1 --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/sanitize/mod.rs @@ -0,0 +1,197 @@ +//! Scrubbing free text and JSON values of secrets and PII. +//! +//! [`sanitize_text_with`] runs, in order: private-key blocks +//! ([`patterns::BLOCK_PATTERNS`], replaced in full), credential markers +//! ([`crate::safety::redact_credential_markers`]), credential shapes +//! ([`patterns::REDACTION_PATTERNS`]) and finally the PII pass +//! ([`crate::safety::pii`]). [`sanitize_json_with`] walks a JSON value, +//! replacing the value under a sensitive-looking key wholesale and running +//! every other string through [`sanitize_text_with`]. + +use serde_json::Value; + +use crate::safety::policy::{Policy, SanitizationReport, Sanitized}; +use crate::safety::{markers, pii}; + +mod patterns; + +use patterns::{BLOCK_PATTERNS, REDACTION_PATTERNS}; + +/// Replacement for a JSON value under a sensitive key, or a subtree beyond +/// [`MAX_JSON_SANITIZE_DEPTH`]. +pub(crate) const REDACTED_SECRET: &str = "[REDACTED_SECRET]"; +/// Replacement for a whole private-key block. +pub(crate) const REDACTED_PRIVATE_KEY: &str = "[REDACTED_PRIVATE_KEY]"; +/// Nesting depth at which [`sanitize_json_with`] stops walking and replaces +/// the subtree with [`REDACTED_SECRET`]. +pub(crate) const MAX_JSON_SANITIZE_DEPTH: usize = 128; + +/// True when `value` looks like it contains a credential. +pub fn has_likely_secret(value: &str) -> bool { + BLOCK_PATTERNS.iter().any(|p| p.is_match(value)) + || REDACTION_PATTERNS.iter().any(|(p, _)| p.is_match(value)) +} + +/// Scrub secrets and PII from free text, returning the cleaned text plus a +/// [`SanitizationReport`]. +pub fn sanitize_text(value: &str) -> Sanitized<String> { + sanitize_text_with(value, Policy::default()) +} + +/// [`sanitize_text`] under an explicit [`Policy`]. +pub fn sanitize_text_with(value: &str, policy: Policy) -> Sanitized<String> { + let mut out = value.to_string(); + let mut report = SanitizationReport::default(); + + for pattern in BLOCK_PATTERNS.iter() { + let hits = pattern.find_iter(&out).count(); + if hits > 0 { + report.blocked_secret_hits += hits; + out = pattern.replace_all(&out, REDACTED_PRIVATE_KEY).into_owned(); + } + } + + // Values after a credential marker (`/secret/<key>`, `Bearer <value>`), + // before the shape regexes: it catches what they cannot — a one-time key, + // a short bearer value — and its `[REDACTED]` is not token-shaped, so no + // regex below fires on it again. Only ever replaces, so the pass makes the + // scrubber strictly stricter. + let (marked, hits) = markers::redact_counted(&out); + if hits > 0 { + report.text_redactions += hits; + out = marked.into_owned(); + } + + for (pattern, replacement) in REDACTION_PATTERNS.iter() { + let hits = pattern.find_iter(&out).count(); + if hits > 0 { + report.text_redactions += hits; + out = pattern.replace_all(&out, *replacement).into_owned(); + } + } + + // Full multilingual national-ID PII scrub (checksum-gated, normalization + // pre-pass) — runs after secret redaction so every call site that scrubs + // secrets also scrubs PII. + let pii = pii::redact_pii_with(&out, policy); + report = report.merge(pii.report); + out = pii.value; + + Sanitized { value: out, report } +} + +/// Recursively scrub a JSON value: sensitive keys are replaced wholesale and +/// every string value runs through `sanitize_text`. +pub fn sanitize_json(value: &Value) -> Sanitized<Value> { + sanitize_json_with(value, Policy::default()) +} + +/// [`sanitize_json`] under an explicit [`Policy`]. +pub fn sanitize_json_with(value: &Value, policy: Policy) -> Sanitized<Value> { + sanitize_json_inner(value, 0, policy) +} + +/// Recursive worker behind [`sanitize_json`]. +/// +/// `depth` counts nesting from the call in `sanitize_json` (which starts at +/// `0`); once it reaches [`MAX_JSON_SANITIZE_DEPTH`] the whole subtree at that +/// point is replaced by a single redaction marker rather than walked further, +/// bounding recursion against pathologically deep or adversarial JSON. +fn sanitize_json_inner(value: &Value, depth: usize, policy: Policy) -> Sanitized<Value> { + if depth >= MAX_JSON_SANITIZE_DEPTH { + return Sanitized { + value: Value::String(REDACTED_SECRET.to_string()), + report: SanitizationReport { + depth_redactions: 1, + ..SanitizationReport::default() + }, + }; + } + + match value { + Value::Object(map) => { + let mut out = serde_json::Map::new(); + let mut report = SanitizationReport::default(); + for (key, value) in map { + if is_sensitive_key(key) { + report.key_redactions += 1; + out.insert(key.clone(), Value::String(REDACTED_SECRET.to_string())); + continue; + } + let sanitized = sanitize_json_inner(value, depth + 1, policy); + report = report.merge(sanitized.report); + out.insert(key.clone(), sanitized.value); + } + Sanitized { + value: Value::Object(out), + report, + } + } + Value::Array(items) => { + let mut out = Vec::with_capacity(items.len()); + let mut report = SanitizationReport::default(); + for item in items { + let sanitized = sanitize_json_inner(item, depth + 1, policy); + report = report.merge(sanitized.report); + out.push(sanitized.value); + } + Sanitized { + value: Value::Array(out), + report, + } + } + Value::String(value) => { + let sanitized = sanitize_text_with(value, policy); + Sanitized { + value: Value::String(sanitized.value), + report: sanitized.report, + } + } + _ => Sanitized { + value: value.clone(), + report: SanitizationReport::default(), + }, + } +} + +/// True when a JSON object key's name itself suggests it holds a secret +/// (`api_key`, `token`, `password`, …), independent of the value's contents. +/// +/// Matching keys are redacted wholesale in [`sanitize_json_inner`] — the +/// value is replaced rather than scanned, since a key named e.g. `password` +/// is assumed sensitive even if its value doesn't match any +/// [`REDACTION_PATTERNS`] regex. Matching is on the key with all +/// non-alphanumeric characters stripped and lowercased, so `API-Key`, +/// `api_key`, and `apiKey` are all treated identically. +fn is_sensitive_key(key: &str) -> bool { + let normalized: String = key + .chars() + .filter(|c| c.is_ascii_alphanumeric()) + .map(|c| c.to_ascii_lowercase()) + .collect(); + + matches!( + normalized.as_str(), + "apikey" + | "token" + | "accesstoken" + | "refreshtoken" + | "authorization" + | "password" + | "secret" + | "clientsecret" + ) || normalized.ends_with("token") + || normalized.ends_with("apikey") + || normalized.ends_with("clientsecret") + || normalized.contains("password") + || normalized.contains("secret") + || normalized.ends_with("key") +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; + +#[cfg(test)] +#[path = "mod_default_policy_tests.rs"] +mod default_policy_tests; diff --git a/crates/tinymemory-safety/src/default_policy_sanitize_tests.rs b/crates/tinymemory-integrations/src/safety/sanitize/mod_default_policy_tests.rs similarity index 85% rename from crates/tinymemory-safety/src/default_policy_sanitize_tests.rs rename to crates/tinymemory-integrations/src/safety/sanitize/mod_default_policy_tests.rs index 3bcd1ca2..5ace6d39 100644 --- a/crates/tinymemory-safety/src/default_policy_sanitize_tests.rs +++ b/crates/tinymemory-integrations/src/safety/sanitize/mod_default_policy_tests.rs @@ -1,20 +1,86 @@ +//! The scrubber under the default (strictest) policy: secret redaction, JSON +//! walking, and the full PII pass reached through [`sanitize_text`]. + use super::*; +use serde_json::json; + +use crate::safety::pii::{ + PII_AADHAAR, PII_CC, PII_CNPJ, PII_CPF, PII_CUIT, PII_DNI, PII_IBAN, PII_MYNUM, PII_NINO, + PII_PAN_IN, PII_PHONE, PII_RFC, PII_RRN, PII_SSN, redact_pii, scan_candidates, +}; +use crate::safety::{BareCardGate, has_likely_email, has_likely_pii}; + +/// Assembled rather than written out so a repository secret scanner does +/// not read the fixture as a real key block. +fn private_key_fixture(kind: &str, body: &str) -> String { + format!("-----BEGIN {kind}-----\n{body}\n-----END {kind}-----") +} + +fn redacts(input: &str, token: &str) { + let out = redact_pii(input); + assert!( + out.value.contains(token), + "expected {token} in output. input={input:?} output={out:?}" + ); +} + +fn unchanged(input: &str) { + let out = redact_pii(input); + assert_eq!( + out.value, input, + "expected no change; report={:?}", + out.report + ); + assert_eq!(out.report.pii_redactions, 0); +} + +/// The one place the two historical copies differed: a bare Luhn-valid run that +/// is neither a real network IIN nor near a card keyword (here a 13-digit +/// epoch-millisecond timestamp). The default policy is the strictest and +/// redacts it; the TinyCortex policy leaves it alone. +#[test] +fn bare_card_gate_is_the_only_policy_difference() { + let ts = "1700000000004"; + let json = format!("{{\"ts\": {ts}}}"); + + let strict = redact_pii(&json); + assert!( + strict.value.contains(PII_CC), + "default policy must redact: {strict:?}" + ); + assert_eq!( + crate::safety::pii::redact_pii_with(&json, Policy::default()).value, + strict.value + ); + assert_eq!(Policy::default().bare_card, BareCardGate::LuhnOnly); + + let corroborated = crate::safety::pii::redact_pii_with(&json, Policy::corroborated()); + assert_eq!( + corroborated.value, json, + "corroborated policy keeps timestamps" + ); + + // Real card, bare, real IIN: both policies redact. + let visa = "4111111111111111"; + assert!(redact_pii(visa).value.contains(PII_CC)); + assert!( + crate::safety::pii::redact_pii_with(visa, Policy::corroborated()) + .value + .contains(PII_CC) + ); + + // The JSON and text entry points thread the policy through. + let value = json!({ "ts": ts }); + assert_ne!( + sanitize_json(&value).value, + sanitize_json_with(&value, Policy::corroborated()).value + ); + assert_ne!( + sanitize_text(ts).value, + sanitize_text_with(ts, Policy::corroborated()).value + ); +} -use crate::pii::redact_pii; -use crate::pii::PII_AADHAAR; -use crate::pii::PII_CC; -use crate::pii::PII_CNPJ; -use crate::pii::PII_CPF; -use crate::pii::PII_CUIT; -use crate::pii::PII_DNI; -use crate::pii::PII_IBAN; -use crate::pii::PII_MYNUM; -use crate::pii::PII_NINO; -use crate::pii::PII_PAN_IN; -use crate::pii::PII_PHONE; -use crate::pii::PII_RFC; -use crate::pii::PII_RRN; -use crate::pii::PII_SSN; #[test] fn sanitize_text_redacts_bearer_and_openai_key() { let input = "Authorization: Bearer abcdefghijklmnop and sk-1234567890123456789012345"; @@ -42,10 +108,12 @@ fn sanitize_json_redacts_sensitive_keys_and_nested_strings() { let sanitized = sanitize_json(&input); assert_eq!(sanitized.value["token"], json!(REDACTED_SECRET)); assert_eq!(sanitized.value["nested"]["ok"], json!("hello")); - assert!(sanitized.value["nested"]["notes"] - .as_str() - .unwrap_or_default() - .contains("[REDACTED]")); + assert!( + sanitized.value["nested"]["notes"] + .as_str() + .unwrap_or_default() + .contains("[REDACTED]") + ); assert!(sanitized.report.key_redactions >= 1); assert!(sanitized.report.text_redactions >= 2); } @@ -130,14 +198,18 @@ fn sanitize_json_propagates_pii_redaction_into_nested_strings() { "meta": { "cuit": "20-11111111-2" } }); let sanitized = sanitize_json(&input); - assert!(sanitized.value["note"] - .as_str() - .unwrap_or_default() - .contains("[REDACTED_PII_RFC]")); - assert!(sanitized.value["meta"]["cuit"] - .as_str() - .unwrap_or_default() - .contains("[REDACTED_PII_CUIT]")); + assert!( + sanitized.value["note"] + .as_str() + .unwrap_or_default() + .contains("[REDACTED_PII_RFC]") + ); + assert!( + sanitized.value["meta"]["cuit"] + .as_str() + .unwrap_or_default() + .contains("[REDACTED_PII_CUIT]") + ); assert!(sanitized.report.pii_redactions >= 2); } @@ -149,10 +221,12 @@ fn sanitize_json_redacts_values_beyond_max_depth() { } let sanitized = sanitize_json(&nested); assert!(sanitized.report.depth_redactions >= 1); - assert!(sanitized - .value - .to_string() - .contains(&format!("\"{REDACTED_SECRET}\""))); + assert!( + sanitized + .value + .to_string() + .contains(&format!("\"{REDACTED_SECRET}\"")) + ); } #[test] diff --git a/crates/tinymemory-safety/src/safety_tests.rs b/crates/tinymemory-integrations/src/safety/sanitize/mod_tests.rs similarity index 91% rename from crates/tinymemory-safety/src/safety_tests.rs rename to crates/tinymemory-integrations/src/safety/sanitize/mod_tests.rs index ad032300..92fbfdb0 100644 --- a/crates/tinymemory-safety/src/safety_tests.rs +++ b/crates/tinymemory-integrations/src/safety/sanitize/mod_tests.rs @@ -1,4 +1,8 @@ +//! The scrubber under the corroborated (TinyCortex) policy: secret patterns, +//! sensitive JSON keys, the depth cap and the credential-marker pass. + use super::*; +use crate::safety::has_likely_pii; use serde_json::json; /// Assembled at run time so a repository secret scanner does not read the @@ -47,10 +51,12 @@ fn sanitize_json_redacts_sensitive_keys_and_nested_strings() { let sanitized = sanitize_json(&input); assert_eq!(sanitized.value["token"], json!(REDACTED_SECRET)); assert_eq!(sanitized.value["nested"]["ok"], json!("hello")); - assert!(sanitized.value["nested"]["notes"] - .as_str() - .unwrap_or_default() - .contains("[REDACTED]")); + assert!( + sanitized.value["nested"]["notes"] + .as_str() + .unwrap_or_default() + .contains("[REDACTED]") + ); assert!(sanitized.report.key_redactions >= 1); assert!(sanitized.report.text_redactions >= 2); } @@ -114,10 +120,12 @@ fn sanitize_json_redacts_values_beyond_max_depth() { } let sanitized = sanitize_json(&nested); assert!(sanitized.report.depth_redactions >= 1); - assert!(sanitized - .value - .to_string() - .contains(&format!("\"{REDACTED_SECRET}\""))); + assert!( + sanitized + .value + .to_string() + .contains(&format!("\"{REDACTED_SECRET}\"")) + ); } #[test] diff --git a/crates/tinymemory-integrations/src/safety/sanitize/patterns.rs b/crates/tinymemory-integrations/src/safety/sanitize/patterns.rs new file mode 100644 index 00000000..d4284e1f --- /dev/null +++ b/crates/tinymemory-integrations/src/safety/sanitize/patterns.rs @@ -0,0 +1,80 @@ +//! The credential shape tables the text scrubber runs. +//! +//! [`BLOCK_PATTERNS`] match whole private-key blocks, which are replaced +//! wholesale. [`REDACTION_PATTERNS`] match a credential's shape — a provider +//! token prefix, a `key=value` assignment, a JWT — and rewrite only the +//! matched span, keeping a captured prefix where the replacement names one. + +use std::sync::LazyLock; + +use regex::Regex; + +use crate::safety::pattern::literal; + +/// Private-key blocks (PEM, OpenSSH, PGP), replaced in full. +pub(super) static BLOCK_PATTERNS: LazyLock<Vec<Regex>> = LazyLock::new(|| { + vec![ + literal( + r"(?is)-----BEGIN(?: [A-Z]+)? PRIVATE KEY-----.*?-----END(?: [A-Z]+)? PRIVATE KEY-----", + ), + literal(r"(?is)-----BEGIN OPENSSH PRIVATE KEY-----.*?-----END OPENSSH PRIVATE KEY-----"), + literal( + r"(?is)-----BEGIN PGP PRIVATE KEY BLOCK-----.*?-----END PGP PRIVATE KEY BLOCK-----", + ), + ] +}); + +/// Credential shapes paired with the replacement each match is rewritten to. +pub(super) static REDACTION_PATTERNS: LazyLock<Vec<(Regex, &'static str)>> = LazyLock::new(|| { + vec![ + ( + literal(r"(?i)(bearer\s+)[A-Za-z0-9._~+/=-]{8,}"), + "${1}[REDACTED]", + ), + ( + literal(r#"(?i)(api[_-]?key\s*[=:\s]\s*["']?)[^\s"']+"#), + "${1}[REDACTED]", + ), + ( + literal( + r#"(?i)\b(token|access[_-]?token|refresh[_-]?token|client[_-]?secret|password|secret)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, + ), + "[REDACTED]", + ), + (literal(r"\bsk-[A-Za-z0-9]{20,}\b"), "[REDACTED]"), + (literal(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b"), "[REDACTED]"), + (literal(r"\bAKIA[0-9A-Z]{16}\b"), "[REDACTED]"), + (literal(r"\bASIA[0-9A-Z]{16}\b"), "[REDACTED]"), + ( + literal(r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9._-]{8,}\.[A-Za-z0-9._-]{8,}\b"), + "[REDACTED]", + ), + ( + literal( + r#"(?i)\b(access_token|refresh_token|id_token|authorization_code|code_verifier|code_challenge)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, + ), + "[REDACTED]", + ), + (literal(r"\bAIza[0-9A-Za-z\-_]{35}\b"), "[REDACTED]"), + (literal(r"\bsk-ant-[A-Za-z0-9\-_]{16,}\b"), "[REDACTED]"), + ( + literal(r"\bsk-(?:proj|org)-[A-Za-z0-9\-_]{12,}\b"), + "[REDACTED]", + ), + ( + literal(r"\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}\b"), + "[REDACTED]", + ), + ( + literal(r"\bxox(?:a|b|p|s|r)-[A-Za-z0-9-]{10,}\b"), + "[REDACTED]", + ), + (literal(r"\bgithub_pat_[A-Za-z0-9_]{20,}\b"), "[REDACTED]"), + (literal(r"\bglpat-[A-Za-z0-9\-_]{16,}\b"), "[REDACTED]"), + (literal(r"\bnpm_[A-Za-z0-9]{20,}\b"), "[REDACTED]"), + ( + literal(r"\bSG\.[A-Za-z0-9_\-]{16,}\.[A-Za-z0-9_\-]{16,}\b"), + "[REDACTED]", + ), + ] +}); diff --git a/crates/tinymemory-integrations/src/sources/README.md b/crates/tinymemory-integrations/src/sources/README.md new file mode 100644 index 00000000..dca6befa --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/README.md @@ -0,0 +1,80 @@ +# sources + +`tinymemory_integrations::sources` (features `sources` and `sources-network`): +readers that turn a source into `StoreItem`s — a folder, a single file, a web +page, a GitHub repository, an RSS feed, a Composio toolkit payload, or the +host's local conversation threads. Conversion to markdown and language +detection come from the sibling [`documents`](../documents/README.md) module. +Architecture overview: +[`docs/architecture/integrations.md`](../../../../docs/architecture/integrations.md). + +Where a host stores its configured sources, and how it edits them, is the +host's business: this module reads a `MemorySourceEntry` it is handed and +checks it with `MemorySourceEntry::validate`, nothing more. + +## Layers + +| Module | Owns | +| --- | --- | +| `types` | the configuration a host persists: `MemorySourceEntry` keyed by `SourceKind`, its field rules, and the reader output types (`SourceItem`, `SourceContent`, `ContentType`) | +| `readers` | `SourceReader` (list, read, read as a `StoreItem`) and one reader per kind, each in its own module directory; `local_file` holds the shared size-capped read and the path-containment guard | +| `fetch` | one URL into a `RawDocument` or a link item (`sources-network`); the RSS and web-page readers fetch through it with their own body caps | +| `fetch::ssrf` | the SSRF guard: scheme and host policy, one address classifier for literal and resolved addresses, a public-only DNS resolver, per-hop redirect checks, and a capped body reader | +| `items` | reader output to `StoreItem`s with `MemoryMeta` filled per kind; `collect_items` drives a reader end to end | +| `composio` | toolkit normalisers (Gmail, Slack, GitHub, Linear, Notion, ClickUp), the `fields::pick_str` lookup they share, `normalise_payload` and `payload_items`. `readers::composio::ComposioReader` is only a placeholder reader | +| `error` | the module `Error`, mapped onto `tinymemory_api::Error` | + +## Kinds and metadata + +Every item's `meta.source` is `SourceRef { kind, id: Some(entry.id) }`. + +| Config kind | `SourceKind` | Item | Metadata | +| --- | --- | --- | --- | +| `folder` | `Folder` | document | `workspace`, `folder` (containing directory), `file_path`, `language`, `observed_at` (mtime), `mime` | +| `file` | `File` | document | as `folder`; `file_item` reads a path with no configured source | +| `web_page` | `Link` | document | `url` | +| `github_repo` | `Github` | document | `repo` (`owner/name`), `commit` (commits), `url` (issues, PRs), `observed_at` | +| `rss_feed` | `Rss` | document | `url` (the entry's link), `observed_at` (published) | +| `composio` | `Composio` | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | +| `conversation` | `Conversation` | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | + +## Folder selection + +With a glob, a folder source takes exactly the matching files. Without one it +takes markdown, plain text and source code (`is_default_candidate`). Either way +it skips hidden files and directories and `target`, `node_modules`, +`__pycache__` and `venv`, never follows symlinks while walking, refuses files +over `FOLDER_FILE_SIZE_CAP_BYTES` (10 MiB), and confines reads to the folder +root (`readers::local_file::ensure_within_base`). A relative path is anchored on the workspace, not +the process working directory. + +## Who decides when + +`readers::reader_for` hands out only the local readers (folder, file, +conversation), which are safe to drive on a timer. Network readers are +constructed explicitly, or through `reader_for_request` (feature +`sources-network`) for an explicit user request. Scheduling, credentials, OAuth and egress budgets stay with the host. + +## Fetching + +Every network fetch of a user-configured URL goes through `fetch` and its +SSRF guard. A hostname is checked as text (private and reserved IP literals, +`localhost`, `.local`/`.internal`, single-label names), its resolved addresses +are checked again by the client's resolver, which pins the connection to an +address it has vetted, and every redirect hop is re-checked. Bodies are read +against a cap while streaming: 32 MiB for `fetch_url`, 10 MiB for a web page, +5 MiB for a feed. The client sends the user agent `openhuman`, times out after +20 seconds, and allows only `http(s)`. An IPv6 literal URL is always refused, +even a public address (the bracketed host fails the IP parse and falls into the +single-label rule): fail closed. Failures are typed — `Invalid` for a refused or malformed +URL, `Unreachable`, `Upstream` for a failure status, `TooLarge`. + +Page titles and feed text are decoded with the `documents::html` helpers, so +named and numeric entities decode the same way everywhere. + +## Features + +- `sources` — the local readers, `items`, `composio` and `types`; implies + `documents`. Links no HTTP stack. +- `sources-network` — adds the GitHub, RSS and web-page readers, `fetch`, and + the SSRF guard. diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs new file mode 100644 index 00000000..b4ce64e5 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs @@ -0,0 +1,69 @@ +//! ClickUp host normalization helpers — result extraction and task-title and +//! timestamp extraction. +//! +//! ClickUp's REST API (and therefore Composio's wrapping of it) returns +//! task lists in a small handful of shapes depending on which endpoint +//! is called. The functions here walk the union of common shapes so the +//! provider doesn't have to branch per Composio envelope variant. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for ClickUp task list results. +/// +/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }` +/// at the top level; Composio re-wraps the upstream payload under +/// `data` or `data.data` depending on the action. We probe each shape +/// in order and return the first array we find. +pub fn extract_tasks(data: &Value) -> Vec<Value> { + let candidates = [ + data.pointer("/data/tasks"), + data.pointer("/tasks"), + data.pointer("/data/data/tasks"), + data.pointer("/data/results"), + data.pointer("/results"), + data.pointer("/data/items"), + data.pointer("/items"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract a human-readable title from a ClickUp task object. +/// +/// ClickUp tasks store the name at `name` (or `data.name` after Composio +/// envelope wrapping). When the name is missing we fall back to the +/// task ID so chunks remain identifiable. +pub fn extract_task_name(task: &Value) -> Option<String> { + pick_str(task, &["name", "data.name", "title", "data.title"]) +} + +/// Extract a stable cursor timestamp (milliseconds since epoch as a +/// string) from a ClickUp task object. +/// +/// The ClickUp API returns `date_updated` as a stringified epoch ms +/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic +/// comparison against the stored cursor remains valid as long as the +/// length doesn't change (it won't until year 33658). +pub fn extract_task_updated(task: &Value) -> Option<String> { + pick_str( + task, + &[ + "date_updated", + "data.date_updated", + "updated_at", + "data.updated_at", + "dateUpdated", + "data.dateUpdated", + ], + ) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-sources/src/composio/clickup_tests.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs similarity index 60% rename from crates/tinymemory-sources/src/composio/clickup_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs index f8d0b5f9..5a520401 100644 --- a/crates/tinymemory-sources/src/composio/clickup_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs @@ -56,43 +56,3 @@ fn extract_task_updated_handles_nested_data() { Some("1700000000000".to_string()) ); } - -#[test] -fn extract_user_id_handles_numeric_id() { - let data = json!({ "user": { "id": 12345 } }); - assert_eq!(extract_user_id(&data), Some("12345".to_string())); -} - -#[test] -fn extract_user_id_handles_wrapped_payload() { - let data = json!({ "data": { "user": { "id": "777" } } }); - assert_eq!(extract_user_id(&data), Some("777".to_string())); -} - -#[test] -fn extract_user_id_none_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_user_id(&data).is_none()); -} - -#[test] -fn extract_workspace_ids_from_teams_array() { - let data = json!({ - "teams": [ - { "id": "ws1", "name": "Personal" }, - { "id": "ws2", "name": "Acme" }, - ] - }); - assert_eq!(extract_workspace_ids(&data), vec!["ws1", "ws2"]); -} - -#[test] -fn extract_workspace_ids_empty_when_no_teams() { - let data = json!({ "foo": "bar" }); - assert!(extract_workspace_ids(&data).is_empty()); -} - -#[test] -fn now_ms_returns_nonzero() { - assert!(now_ms() > 0); -} diff --git a/crates/tinymemory-sources/src/composio/documents.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs similarity index 99% rename from crates/tinymemory-sources/src/composio/documents.rs rename to crates/tinymemory-integrations/src/sources/composio/documents/mod.rs index 5fbcff3f..04753e4b 100644 --- a/crates/tinymemory-sources/src/composio/documents.rs +++ b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs @@ -7,12 +7,12 @@ //! rather than dropped. [`payload_items`] wraps the documents as //! `StoreItem::Document`s. +use crate::documents::{DocumentFormat, markdown_from_text}; use chrono::{DateTime, TimeZone, Utc}; use serde_json::Value; use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem}; -use tinymemory_documents::{markdown_from_text, DocumentFormat}; -use super::helpers::pick_str; +use super::fields::pick_str; use super::{clickup, github, gmail_post_process, linear, notion}; /// One record of a Composio payload, normalised: an email, a message, an @@ -339,5 +339,5 @@ fn generic_records(data: &Value) -> Vec<ComposioDocument> { } #[cfg(test)] -#[path = "documents_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/composio/documents_tests.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs similarity index 100% rename from crates/tinymemory-sources/src/composio/documents_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs new file mode 100644 index 00000000..4c257085 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs @@ -0,0 +1,44 @@ +//! Field lookup shared by the Composio normalisers: pull a string out of a +//! payload by trying several dotted paths, because Composio wraps the same +//! upstream field at different depths depending on the action and version. + +/// Walk a JSON object using a list of dotted-path candidates and return the +/// first non-empty **string** match, trimmed. +/// +/// Each path is split on `.` and followed with `Value::get`, so it only +/// descends through objects — it never indexes into an array. A leaf that is +/// not a string (a number, a bool) is rejected rather than coerced, so a +/// payload whose `id` is `42` rather than `"42"` yields `None` here. That +/// differs from the private `scalar` lookup in the `documents` mapping, which +/// renders numbers; the normalisers were written against the +/// reject-non-strings behaviour and `pick_str_rejects_non_string_values` pins +/// it. +pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option<String> { + for path in paths { + let mut cur = value; + let mut ok = true; + for segment in path.split('.') { + match cur.get(segment) { + Some(next) => cur = next, + None => { + ok = false; + break; + } + } + } + if !ok { + continue; + } + if let Some(s) = cur.as_str() { + let trimmed = s.trim(); + if !trimmed.is_empty() { + return Some(trimmed.to_string()); + } + } + } + None +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-sources/src/composio/helpers_tests.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs similarity index 80% rename from crates/tinymemory-sources/src/composio/helpers_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs index c8f9dd00..330ec404 100644 --- a/crates/tinymemory-sources/src/composio/helpers_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs @@ -1,4 +1,4 @@ -//! Tests for the shared normaliser helpers. +//! Tests for the shared Composio field lookup. use super::*; use serde_json::json; @@ -24,9 +24,9 @@ fn pick_str_respects_path_order() { assert_eq!(pick_str(&v, &["b", "a"]), Some("second".into())); } -/// The drift guard for the divergence documented on [`pick_str`]. If this -/// ever starts returning `Some("42")`, someone has re-pointed the -/// normalisers at `common::pick_str` and changed their output. +/// The drift guard for the behaviour documented on [`pick_str`]. If this +/// ever starts returning `Some("42")`, the normalisers' emitted ids have +/// changed. #[test] fn pick_str_rejects_non_string_values() { let v = json!({"count": 42, "flag": true, "empty": "", "whitespace": " "}); diff --git a/crates/tinymemory-sources/src/composio/github.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs similarity index 81% rename from crates/tinymemory-sources/src/composio/github.rs rename to crates/tinymemory-integrations/src/sources/composio/github/mod.rs index 393e93e2..12e7a615 100644 --- a/crates/tinymemory-sources/src/composio/github.rs +++ b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs @@ -1,13 +1,14 @@ -//! GitHub host normalization helpers — result extraction, identity helpers, and time utilities. +//! GitHub host normalization helpers — issue extraction and issue id, title and +//! timestamp helpers. //! -//! GitHub's REST API (proxied through Composio) returns search results and -//! authenticated-user payloads in a small number of shapes. The functions here +//! GitHub's REST API (proxied through Composio) returns search results in a +//! small number of shapes. The functions here //! walk the union of common Composio envelope variants so the provider stays //! clean and branch-free. use serde_json::Value; -use super::helpers::pick_str; +use super::fields::pick_str; /// Walk the Composio response envelope for GitHub search issue results. /// @@ -50,10 +51,10 @@ pub fn extract_issue_id(issue: &Value) -> Option<String> { } // Fallback: parse owner/repo/number from html_url path segments. // URL shape: https://github.com/{owner}/{repo}/issues/{number} - if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"]) { - if let Some(slug) = github_url_to_slug(&url) { - return Some(slug); - } + if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"]) + && let Some(slug) = github_url_to_slug(&url) + { + return Some(slug); } None } @@ -110,21 +111,6 @@ pub fn extract_issue_updated_at(issue: &Value) -> Option<String> { ) } -/// Extract the authenticated user's login handle from a -/// `GITHUB_GET_THE_AUTHENTICATED_USER` response. -pub fn extract_user_login(data: &Value) -> Option<String> { - pick_str(data, &["login", "data.login"]) -} - -/// Current wall-clock time in milliseconds since the UNIX epoch. -pub fn now_ms() -> u64 { - use std::time::{SystemTime, UNIX_EPOCH}; - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0) -} - #[cfg(test)] -#[path = "github_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/composio/github_tests.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs similarity index 82% rename from crates/tinymemory-sources/src/composio/github_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs index 6e2a9e0b..0b7322f1 100644 --- a/crates/tinymemory-sources/src/composio/github_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs @@ -95,26 +95,3 @@ fn extract_issue_updated_at_none_when_missing() { let issue = json!({ "id": 1u64 }); assert!(extract_issue_updated_at(&issue).is_none()); } - -#[test] -fn extract_user_login_from_top_level() { - let data = json!({ "login": "octocat" }); - assert_eq!(extract_user_login(&data), Some("octocat".to_string())); -} - -#[test] -fn extract_user_login_from_data_wrapper() { - let data = json!({ "data": { "login": "monalisa" } }); - assert_eq!(extract_user_login(&data), Some("monalisa".to_string())); -} - -#[test] -fn extract_user_login_none_when_missing() { - let data = json!({ "id": 1u64 }); - assert!(extract_user_login(&data).is_none()); -} - -#[test] -fn now_ms_returns_nonzero() { - assert!(now_ms() > 0); -} diff --git a/crates/tinymemory-sources/src/composio/gmail_post_process.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs similarity index 93% rename from crates/tinymemory-sources/src/composio/gmail_post_process.rs rename to crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs index 327479ab..c6c78ae7 100644 --- a/crates/tinymemory-sources/src/composio/gmail_post_process.rs +++ b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs @@ -51,9 +51,10 @@ //! more slugs they should live in this file, branched from //! [`post_process`]. -use serde_json::{json, Map, Value}; +use serde_json::{Map, Value, json}; -/// Entry point called from `GmailProvider::post_process_action_result`. +/// Entry point a host calls on each Gmail action response (the slug names +/// the action) before handing it to `normalise_payload`. /// /// Dispatches on the Composio action slug. Unknown Gmail slugs fall /// through to a no-op. @@ -97,7 +98,7 @@ pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) { } // Presence is checked immutably first, then fetched mutably. The original // form re-fetched with `unwrap()` after a mutable probe, which is sound but - // relies on the reader to see why; this crate forbids `unwrap`, and the + // relies on the reader to see why; this crate lints against `unwrap`, and the // immutable probe expresses the same reasoning to the compiler. let container = if data.get("messages").is_some() { data @@ -162,16 +163,10 @@ pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) { /// each segment really does belong to the message at the same index. /// Mismatches force a fallback so we never write a wrong-message body /// to the raw archive. -pub fn split_response_markdown_per_message(md: &str, expected_count: usize) -> Option<Vec<String>> { - split_response_markdown_per_message_with_hint(md, expected_count, None) -} - -/// Split a response-level markdown blob into one slice per message. /// -/// `hint` carries the message ids in response order, which is what makes the -/// split reliable: the blob's own section headings are backend-rendered and -/// have changed shape between versions, so matching on them alone silently -/// mis-attributed bodies. +/// The hint is what makes the split reliable: the blob's own section headings +/// are backend-rendered and have changed shape between versions, so matching +/// on them alone silently mis-attributed bodies. pub fn split_response_markdown_per_message_with_hint( md: &str, expected_count: usize, @@ -223,15 +218,15 @@ pub fn split_response_markdown_per_message_with_hint( // pair fails, we treat the split as unreliable and try the // next pattern. Empty / null subjects skip validation (e.g. // notification mails where the subject is ""). - if let Some(hints) = messages_hint { - if !validate_segments_against_hints(&segments, hints) { - tracing::debug!( - expected = expected_count, - sep = sep, - "[composio:gmail][post-process] split candidate failed subject check" - ); - continue; - } + if let Some(hints) = messages_hint + && !validate_segments_against_hints(&segments, hints) + { + tracing::debug!( + expected = expected_count, + sep = sep, + "[composio:gmail][post-process] split candidate failed subject check" + ); + continue; } return Some(segments); } @@ -436,10 +431,10 @@ fn pick_header(msg: &Map<String, Value>, name: &str) -> Option<Value> { let headers = msg.get("payload")?.get("headers")?.as_array()?; for h in headers { let hn = h.get("name").and_then(|v| v.as_str()).unwrap_or(""); - if hn.eq_ignore_ascii_case(name) { - if let Some(v) = h.get("value").and_then(|v| v.as_str()) { - return Some(Value::String(v.to_string())); - } + if hn.eq_ignore_ascii_case(name) + && let Some(v) = h.get("value").and_then(|v| v.as_str()) + { + return Some(Value::String(v.to_string())); } } None @@ -499,5 +494,5 @@ fn extract_attachments(msg: &Map<String, Value>) -> Vec<Value> { } #[cfg(test)] -#[path = "gmail_post_process_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/composio/gmail_post_process_tests.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs similarity index 95% rename from crates/tinymemory-sources/src/composio/gmail_post_process_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs index 0f114d1e..dd1b44af 100644 --- a/crates/tinymemory-sources/src/composio/gmail_post_process_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs @@ -190,14 +190,14 @@ fn empty_markdown_formatted_falls_through_to_message_text() { assert!(md.contains("real body")); } -// ── split_response_markdown_per_message ───────────────────────────────── +// ── split_response_markdown_per_message_with_hint ─────────────────────── #[test] fn split_response_markdown_uses_horizontal_rule_marker() { // The confirmed backend marker is `\n---\n`. Three messages → // expect three slices when there's no preamble. let md = "## Alice's update\n\nbody A with https://gh.io/abc\n---\n## Bob's reply\n\nbody B\n---\n## Carol\n\nbody C"; - let slices = super::split_response_markdown_per_message(md, 3).unwrap(); + let slices = super::split_response_markdown_per_message_with_hint(md, 3, None).unwrap(); assert_eq!(slices.len(), 3); assert!(slices[0].contains("Alice's update")); assert!(slices[1].contains("Bob's reply")); @@ -213,7 +213,7 @@ fn split_response_markdown_drops_preamble() { // When a preamble like `# Inbox` precedes the first marker, we // see N+1 parts after split — the preamble must be dropped. let md = "# Inbox (2 messages)\n---\n## A\n\nbody A\n---\n## B\n\nbody B"; - let slices = super::split_response_markdown_per_message(md, 2).unwrap(); + let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); assert_eq!(slices.len(), 2); assert!(slices[0].contains("body A")); assert!(slices[1].contains("body B")); @@ -226,7 +226,7 @@ fn split_response_markdown_drops_preamble() { fn split_response_markdown_falls_back_to_h2_marker() { // No `---` rules — backend used h2 headings as boundaries. let md = "## Alice\n\nbody A\n\n## Bob\n\nbody B"; - let slices = super::split_response_markdown_per_message(md, 2).unwrap(); + let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); assert_eq!(slices.len(), 2); assert!(slices[0].contains("body A")); assert!(slices[1].contains("body B")); @@ -235,13 +235,13 @@ fn split_response_markdown_falls_back_to_h2_marker() { #[test] fn split_response_markdown_returns_none_on_count_mismatch() { let md = "## only one section here"; - assert!(super::split_response_markdown_per_message(md, 3).is_none()); + assert!(super::split_response_markdown_per_message_with_hint(md, 3, None).is_none()); } #[test] fn split_response_markdown_single_message_returns_whole_input() { let md = "## solo\n\nthe whole body"; - let slices = super::split_response_markdown_per_message(md, 1).unwrap(); + let slices = super::split_response_markdown_per_message_with_hint(md, 1, None).unwrap(); assert_eq!(slices, vec![md.to_string()]); } diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs new file mode 100644 index 00000000..ea52092a --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs @@ -0,0 +1,79 @@ +//! Linear host normalization helpers — result extraction and issue-title and +//! timestamp extraction. +//! +//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns +//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top +//! level or nested under `data`. The functions here walk the union of +//! common shapes so the provider does not have to branch per Composio +//! envelope variant. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for Linear issue list results. +/// +/// Linear's list endpoints return `{ nodes: [...] }` or +/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the +/// upstream payload under `data` or `data.data`. We probe each shape +/// in order and return the first array we find. +pub fn extract_issues(data: &Value) -> Vec<Value> { + let candidates = [ + data.pointer("/data/nodes"), + data.pointer("/nodes"), + data.pointer("/data/issues/nodes"), + data.pointer("/issues/nodes"), + data.pointer("/data/data/nodes"), + data.pointer("/data/data/issues/nodes"), + data.pointer("/data/results"), + data.pointer("/results"), + data.pointer("/data/items"), + data.pointer("/items"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract a human-readable title from a Linear issue object. +/// +/// Linear issues store the name at `title` (or `data.title` after +/// Composio envelope wrapping). Falls back to `name` / `identifier` +/// so the chunk remains identifiable even for unusual response shapes. +pub fn extract_issue_title(issue: &Value) -> Option<String> { + pick_str( + issue, + &[ + "title", + "data.title", + "name", + "data.name", + "identifier", + "data.identifier", + ], + ) +} + +/// Extract a stable cursor timestamp from a Linear issue object. +/// +/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep +/// the value as a string so lexicographic comparison against the stored +/// cursor is valid. +pub fn extract_issue_updated(issue: &Value) -> Option<String> { + pick_str( + issue, + &[ + "updatedAt", + "data.updatedAt", + "updated_at", + "data.updated_at", + ], + ) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-sources/src/composio/linear_tests.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs similarity index 50% rename from crates/tinymemory-sources/src/composio/linear_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs index b71a7f7b..c0acff63 100644 --- a/crates/tinymemory-sources/src/composio/linear_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs @@ -91,96 +91,3 @@ fn extract_issue_updated_falls_back_to_snake_case() { Some("2026-01-15T08:30:00.000Z".to_string()) ); } - -// ── extract_viewer ─────────────────────────────────────────────── - -#[test] -fn extract_viewer_from_data_nodes() { - let data = json!({ "data": { "nodes": [{ "id": "usr_1", "email": "a@b.com" }] } }); - let v = extract_viewer(&data).expect("should find viewer"); - assert_eq!(v["id"], "usr_1"); -} - -#[test] -fn extract_viewer_from_top_level_nodes() { - let data = json!({ "nodes": [{ "id": "usr_2" }] }); - let v = extract_viewer(&data).expect("should find viewer"); - assert_eq!(v["id"], "usr_2"); -} - -#[test] -fn extract_viewer_fallback_direct_object() { - let data = json!({ "id": "usr_direct", "name": "Direct User" }); - let v = extract_viewer(&data).expect("should return direct object"); - assert_eq!(v["id"], "usr_direct"); -} - -#[test] -fn extract_viewer_returns_none_when_absent() { - let data = json!({ "foo": "bar" }); - assert!(extract_viewer(&data).is_none()); -} - -// ── extract_pagination_cursor ──────────────────────────────────── - -#[test] -fn extract_pagination_cursor_returns_cursor_when_has_next_page() { - let data = json!({ - "data": { - "pageInfo": { - "hasNextPage": true, - "endCursor": "cursor_abc" - } - } - }); - assert_eq!( - extract_pagination_cursor(&data), - Some("cursor_abc".to_string()) - ); -} - -#[test] -fn extract_pagination_cursor_returns_none_when_last_page() { - let data = json!({ - "pageInfo": { - "hasNextPage": false, - "endCursor": "cursor_xyz" - } - }); - assert!(extract_pagination_cursor(&data).is_none()); -} - -#[test] -fn extract_pagination_cursor_from_doubly_nested_issues() { - // The same `data.data.issues` shape `extract_issues` reads must also - // expose its pageInfo cursor, or a doubly-nested payload never pages. - let data = json!({ - "data": { - "data": { - "issues": { - "pageInfo": { - "hasNextPage": true, - "endCursor": "cursor_issue_2" - } - } - } - } - }); - assert_eq!( - extract_pagination_cursor(&data), - Some("cursor_issue_2".to_string()) - ); -} - -#[test] -fn extract_pagination_cursor_returns_none_when_absent() { - let data = json!({ "nodes": [{"id": "i1"}] }); - assert!(extract_pagination_cursor(&data).is_none()); -} - -// ── now_ms ─────────────────────────────────────────────────────── - -#[test] -fn now_ms_returns_nonzero() { - assert!(now_ms() > 0); -} diff --git a/crates/tinymemory-sources/src/composio/mod.rs b/crates/tinymemory-integrations/src/sources/composio/mod.rs similarity index 73% rename from crates/tinymemory-sources/src/composio/mod.rs rename to crates/tinymemory-integrations/src/sources/composio/mod.rs index 8a5eae9a..4df32030 100644 --- a/crates/tinymemory-sources/src/composio/mod.rs +++ b/crates/tinymemory-integrations/src/sources/composio/mod.rs @@ -8,8 +8,7 @@ //! and pull out the tasks, issues, pages or messages //! ([`clickup`], [`github`], [`linear`], [`notion`]), or rewrite a verbose //! response into a slim one in place ([`gmail_post_process`], -//! [`slack_post_process`]). [`email_clean`] and [`email_markdown`] render -//! email bodies and threads. +//! [`slack_post_process`]). [`fields`] holds the path lookup they share. //! 2. [`normalise_payload`] turns one (post-processed) response into //! [`ComposioDocument`]s, and [`payload_items`] turns those into //! [`StoreItem::Document`](tinymemory_api::StoreItem::Document)s with @@ -20,20 +19,17 @@ //! Nothing here holds a credential, opens a socket or decides when to sync. //! //! One caveat on "pure": [`gmail_post_process::format_email_local_time`] -//! renders in `chrono::Local`, so it reads the host's timezone, and the -//! `now_ms` helpers read the clock. The raw UTC fields are preserved -//! alongside, so ordering and identity stay UTC-based. +//! renders in `chrono::Local`, so it reads the host's timezone. The raw UTC +//! fields are preserved alongside, so ordering and identity stay UTC-based. pub mod clickup; -pub mod email_clean; -pub mod email_markdown; +pub mod fields; pub mod github; pub mod gmail_post_process; -pub mod helpers; pub mod linear; pub mod notion; pub mod slack_post_process; mod documents; -pub use documents::{normalise_payload, payload_items, ComposioDocument}; +pub use documents::{ComposioDocument, normalise_payload, payload_items}; diff --git a/crates/tinymemory-sources/src/composio/notion.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs similarity index 58% rename from crates/tinymemory-sources/src/composio/notion.rs rename to crates/tinymemory-integrations/src/sources/composio/notion/mod.rs index f13c17fc..00c1df50 100644 --- a/crates/tinymemory-sources/src/composio/notion.rs +++ b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs @@ -1,9 +1,9 @@ -//! Notion host normalization helpers — result extraction, pagination cursor, -//! page title extraction, and time utilities. +//! Notion host normalization helpers — result extraction, page markdown and +//! page title extraction. use serde_json::Value; -use super::helpers::pick_str; +use super::fields::pick_str; /// Walk the Composio response envelope for Notion page results. pub fn extract_results(data: &Value) -> Vec<Value> { @@ -41,29 +41,10 @@ pub fn extract_page_markdown(data: &Value) -> Option<String> { "/data/text", ]; for p in PATHS { - if let Some(s) = data.pointer(p).and_then(Value::as_str) { - if !s.trim().is_empty() { - return Some(s.to_string()); - } - } - } - None -} - -/// Extract the Notion pagination cursor (for `start_cursor` on the -/// next request). -pub fn extract_notion_cursor(data: &Value) -> Option<String> { - let candidates = [ - data.pointer("/data/next_cursor"), - data.pointer("/next_cursor"), - data.pointer("/data/data/next_cursor"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(s) = cand.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } + if let Some(s) = data.pointer(p).and_then(Value::as_str) + && !s.trim().is_empty() + { + return Some(s.to_string()); } } None @@ -82,16 +63,16 @@ pub fn extract_page_title(page: &Value) -> Option<String> { // Walk all properties looking for a "title" type field. if let Some(obj) = props.as_object() { for (_key, val) in obj { - if val.get("type").and_then(Value::as_str) == Some("title") { - if let Some(arr) = val.get("title").and_then(Value::as_array) { - let text: String = arr - .iter() - .filter_map(|t| t.get("plain_text").and_then(Value::as_str)) - .collect::<Vec<_>>() - .join(""); - if !text.is_empty() { - return Some(text); - } + if val.get("type").and_then(Value::as_str) == Some("title") + && let Some(arr) = val.get("title").and_then(Value::as_array) + { + let text: String = arr + .iter() + .filter_map(|t| t.get("plain_text").and_then(Value::as_str)) + .collect::<Vec<_>>() + .join(""); + if !text.is_empty() { + return Some(text); } } } @@ -102,19 +83,6 @@ pub fn extract_page_title(page: &Value) -> Option<String> { pick_str(page, &["title", "data.title", "name", "data.name"]) } -/// Milliseconds since the Unix epoch. -/// -/// The one clock read in this crate. Notion payloads carry no ingestion -/// timestamp, so the normaliser stamps one; everything else here is a function -/// of its input alone. -pub fn now_ms() -> u64 { - use std::time::{SystemTime, UNIX_EPOCH}; - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0) -} - #[cfg(test)] -#[path = "notion_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/composio/notion_tests.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs similarity index 82% rename from crates/tinymemory-sources/src/composio/notion_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs index 76b5d18a..ab2e79bc 100644 --- a/crates/tinymemory-sources/src/composio/notion_tests.rs +++ b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs @@ -61,29 +61,6 @@ fn extract_results_empty_when_no_match() { assert!(extract_results(&data).is_empty()); } -#[test] -fn extract_notion_cursor_from_data() { - let data = json!({"data": {"next_cursor": "cur123"}}); - assert_eq!(extract_notion_cursor(&data), Some("cur123".into())); -} - -#[test] -fn extract_notion_cursor_from_top_level() { - let data = json!({"next_cursor": "abc"}); - assert_eq!(extract_notion_cursor(&data), Some("abc".into())); -} - -#[test] -fn extract_notion_cursor_none_when_empty() { - let data = json!({"data": {"next_cursor": " "}}); - assert_eq!(extract_notion_cursor(&data), None); -} - -#[test] -fn extract_notion_cursor_none_when_missing() { - assert_eq!(extract_notion_cursor(&json!({})), None); -} - #[test] fn extract_page_title_from_properties_title_type() { let page = json!({ @@ -132,8 +109,3 @@ fn extract_page_title_none_when_no_title_field() { let page = json!({"id": "123"}); assert!(extract_page_title(&page).is_none()); } - -#[test] -fn now_ms_returns_nonzero() { - assert!(now_ms() > 0); -} diff --git a/crates/tinymemory-sources/src/composio/slack_post_process.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs similarity index 98% rename from crates/tinymemory-sources/src/composio/slack_post_process.rs rename to crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs index 8e217b36..764b885a 100644 --- a/crates/tinymemory-sources/src/composio/slack_post_process.rs +++ b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs @@ -36,7 +36,8 @@ use serde_json::{Map, Value}; -/// Entry point called from `SlackProvider::post_process_action_result`. +/// Entry point a host calls on each Slack action response (the slug names +/// the action) before handing it to `normalise_payload`. /// /// Dispatches on the Composio action slug and rewrites `data` in place. /// Unknown slugs are silently ignored. @@ -62,7 +63,7 @@ pub fn post_process(slug: &str, _arguments: Option<&Value>, data: &mut Value) { /// shape under a top-level `messages[]` key. The consumed nested array is /// removed from the payload so the raw verbose rows don't linger alongside /// the slim copy. The caller injects `channel_id` via -/// [`super::sync::extract_messages`]. +/// the host's Slack sync pipeline. fn reshape_fetch_history(data: &mut Value) { let arr = take_array( data, @@ -319,5 +320,5 @@ fn with_object(data: &mut Value, edit: impl FnOnce(&mut Map<String, Value>)) { } #[cfg(test)] -#[path = "slack_post_process_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/composio/slack_post_process_tests.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs similarity index 100% rename from crates/tinymemory-sources/src/composio/slack_post_process_tests.rs rename to crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs diff --git a/crates/tinymemory-sources/src/error/mod.rs b/crates/tinymemory-integrations/src/sources/error/mod.rs similarity index 88% rename from crates/tinymemory-sources/src/error/mod.rs rename to crates/tinymemory-integrations/src/sources/error/mod.rs index e5aa0639..3d04ab85 100644 --- a/crates/tinymemory-sources/src/error/mod.rs +++ b/crates/tinymemory-integrations/src/sources/error/mod.rs @@ -1,4 +1,4 @@ -//! The crate-wide error and result alias. +//! The sources module's error and result alias. //! //! Variants say what went wrong in terms a host can act on: bad //! configuration or input ([`Error::Invalid`]), something that is not there @@ -37,9 +37,6 @@ pub enum Error { /// A network reader's own diagnostic, carried verbatim. #[error("{0}")] Reader(String), - /// The source registry file could not be read, parsed or written. - #[error("source registry error: {0}")] - Registry(String), /// A filesystem operation failed. #[error("io error: {0}")] Io(#[from] std::io::Error), @@ -48,7 +45,7 @@ pub enum Error { Json(#[from] serde_json::Error), /// Converting a body to markdown failed. #[error(transparent)] - Document(#[from] tinymemory_documents::Error), + Document(#[from] crate::documents::Error), } impl From<Error> for tinymemory_api::Error { @@ -60,14 +57,13 @@ impl From<Error> for tinymemory_api::Error { } Error::NotFound(_) => Self::NotFound(error.to_string()), Error::Unreachable(_) => Self::Unavailable(error.to_string()), - Error::Registry(_) => Self::Config(error.to_string()), Error::Document(inner) => inner.into(), Error::Upstream(_) | Error::Reader(_) | Error::Io(_) => Self::Engine(error.to_string()), } } } -/// Result alias for this crate's fallible operations. +/// Result alias for this module's fallible operations. pub type Result<T> = std::result::Result<T, Error>; #[cfg(test)] diff --git a/crates/tinymemory-sources/src/error/mod_tests.rs b/crates/tinymemory-integrations/src/sources/error/mod_tests.rs similarity index 87% rename from crates/tinymemory-sources/src/error/mod_tests.rs rename to crates/tinymemory-integrations/src/sources/error/mod_tests.rs index 0df2b635..4a9928ec 100644 --- a/crates/tinymemory-sources/src/error/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/error/mod_tests.rs @@ -1,4 +1,4 @@ -//! Tests for the crate error and its mapping onto the contract error. +//! Tests for the module error and its mapping onto the contract error. use super::*; @@ -31,14 +31,13 @@ fn every_variant_maps_onto_the_contract_error_a_host_can_act_on() { (Error::Unreachable("x".into()), |e| { matches!(e, Api::Unavailable(_)) }), - (Error::Registry("x".into()), |e| matches!(e, Api::Config(_))), (Error::Upstream("x".into()), |e| matches!(e, Api::Engine(_))), (Error::Reader("x".into()), |e| matches!(e, Api::Engine(_))), (Error::Io(std::io::Error::other("disk")), |e| { matches!(e, Api::Engine(_)) }), ( - Error::Document(tinymemory_documents::Error::UnsupportedFormat("pdf".into())), + Error::Document(crate::documents::Error::UnsupportedFormat("pdf".into())), |e| matches!(e, Api::Unsupported(_)), ), ]; diff --git a/crates/tinymemory-sources/src/fetch/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/mod.rs similarity index 80% rename from crates/tinymemory-sources/src/fetch/mod.rs rename to crates/tinymemory-integrations/src/sources/fetch/mod.rs index 403b1106..6063b5f7 100644 --- a/crates/tinymemory-sources/src/fetch/mod.rs +++ b/crates/tinymemory-integrations/src/sources/fetch/mod.rs @@ -3,19 +3,21 @@ //! A URL a user types is an SSRF vector: `http://169.254.169.254/` is a cloud //! metadata endpoint, `http://localhost:6379/` is somebody's Redis, and a //! hostname that resolves publicly on the first lookup can resolve to a private -//! address on the second. Every fetch here goes through the same guard as the -//! RSS and web-page readers ([`crate::readers::ssrf`]): a scheme and host -//! policy, a resolver that pins connections to globally routable addresses, -//! and per-hop redirect re-checks. +//! address on the second. Every fetch goes through the guard in [`ssrf`]: a +//! scheme and host policy, a resolver that pins connections to globally +//! routable addresses, and per-hop redirect re-checks. The RSS and web-page +//! readers fetch through here too, each with its own body cap. //! //! No scheduling, no retries, no credentials, no robots.txt: this fetches one -//! URL, once, when asked. Conversion to markdown is `tinymemory-documents`'. +//! URL, once, when asked. Conversion to markdown is the `documents` module's. +use crate::documents::{DocumentConverter, MAX_DOCUMENT_BYTES, RawDocument, document_item}; use tinymemory_api::{MemoryMeta, SourceKind, StoreItem}; -use tinymemory_documents::{document_item, DocumentConverter, RawDocument, MAX_DOCUMENT_BYTES}; -use crate::error::{Error, Result}; -use crate::readers::ssrf::{build_client, is_url_allowed, read_body_capped}; +use crate::sources::error::{Error, Result}; +use ssrf::{build_client, is_url_allowed, read_body_capped}; + +pub mod ssrf; /// Fetch `url` and return its body as a [`RawDocument`]. /// @@ -32,6 +34,16 @@ use crate::readers::ssrf::{build_client, is_url_allowed, read_body_capped}; /// - [`Error::Upstream`] for a non-success status. /// - [`Error::TooLarge`] for a body over [`MAX_DOCUMENT_BYTES`]. pub async fn fetch_url(url: &str) -> Result<RawDocument> { + fetch_url_capped(url, MAX_DOCUMENT_BYTES as u64).await +} + +/// [`fetch_url`] with a caller-chosen body cap, for the readers whose sources +/// warrant a tighter one than [`MAX_DOCUMENT_BYTES`]. +/// +/// # Errors +/// +/// As [`fetch_url`], with [`Error::TooLarge`] for a body over `max_bytes`. +pub(crate) async fn fetch_url_capped(url: &str, max_bytes: u64) -> Result<RawDocument> { let parsed = reqwest::Url::parse(url) .map_err(|error| Error::Invalid(format!("invalid url {url:?}: {error}")))?; if !is_url_allowed(&parsed) { @@ -51,7 +63,7 @@ pub async fn fetch_url(url: &str) -> Result<RawDocument> { .await .map_err(|error| Error::Unreachable(format!("fetching {url:?}: {error}")))?; - response_to_document(url, parsed, response).await + response_to_document(url, parsed, response, max_bytes).await } /// Fetch `url`, convert it through `converter`, and wrap it as a @@ -79,6 +91,7 @@ async fn response_to_document( url: &str, parsed: reqwest::Url, response: reqwest::Response, + max_bytes: u64, ) -> Result<RawDocument> { let status = response.status(); if !status.is_success() { @@ -95,7 +108,7 @@ async fn response_to_document( // The cap is applied while reading, not after: a body that would not fit is // one this process should never have finished buffering. - let bytes = read_body_capped(response, MAX_DOCUMENT_BYTES as u64) + let bytes = read_body_capped(response, max_bytes) .await .map_err(|error| read_error(url, &error))?; diff --git a/crates/tinymemory-sources/src/fetch/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs similarity index 83% rename from crates/tinymemory-sources/src/fetch/mod_tests.rs rename to crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs index 517e238b..a42f4917 100644 --- a/crates/tinymemory-sources/src/fetch/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs @@ -87,9 +87,14 @@ async fn completed_response_preserves_body_type_origin_and_filename() { ) .await; let url = reqwest::Url::parse("https://example.com/guides/readme.md").unwrap(); - let document = response_to_document(url.as_str(), url.clone(), response) - .await - .unwrap(); + let document = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap(); assert_eq!(document.bytes, b"# title"); assert_eq!(document.origin.as_deref(), Some(url.as_str())); assert_eq!(document.filename.as_deref(), Some("readme.md")); @@ -104,21 +109,36 @@ async fn completed_response_handles_status_empty_body_and_filename_absence() { let response = local_response(b"HTTP/1.1 503 Service Unavailable\r\nContent-Length: 0\r\n\r\n").await; let url = reqwest::Url::parse("https://example.com/unavailable").unwrap(); - let error = response_to_document(url.as_str(), url.clone(), response) - .await - .unwrap_err(); + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); assert!(matches!(error, Error::Upstream(_))); let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n").await; - let error = response_to_document(url.as_str(), url.clone(), response) - .await - .unwrap_err(); + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); assert!(matches!(error, Error::Invalid(_))); let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\ntext").await; - let document = response_to_document(url.as_str(), url.clone(), response) - .await - .unwrap(); + let document = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap(); assert_eq!(document.bytes, b"text"); assert!(document.filename.is_none()); assert!(document.declared_mime.is_none()); @@ -136,15 +156,20 @@ async fn completed_response_maps_declared_oversize_to_budget_exceeded() { ) .await; let url = reqwest::Url::parse("https://example.com/huge.bin").unwrap(); - let error = response_to_document(url.as_str(), url.clone(), response) - .await - .unwrap_err(); + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); assert!(matches!(error, Error::TooLarge(_))); } #[tokio::test] async fn a_link_item_refuses_a_private_target_before_fetching() { - let chain = tinymemory_documents::ConverterChain::default(); + let chain = crate::documents::ConverterChain::default(); let error = link_item("http://127.0.0.1/", Some("src_link".into()), &chain) .await .unwrap_err(); diff --git a/crates/tinymemory-sources/src/readers/ssrf.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs similarity index 59% rename from crates/tinymemory-sources/src/readers/ssrf.rs rename to crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs index d7f56ee3..49c28453 100644 --- a/crates/tinymemory-sources/src/readers/ssrf.rs +++ b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs @@ -1,7 +1,9 @@ //! Shared SSRF guard and fetch hygiene for the network source readers. //! -//! The web-page and RSS readers both fetch user-configured URLs, so they share -//! the policy in this module. +//! [`super::fetch_url`] and the web-page and RSS readers built on it all fetch +//! user-configured URLs, so they share the policy in this module. It is public +//! so a host fetching a user-supplied URL by other means applies the same +//! policy rather than a second, weaker one. //! //! The hostname *text* check (`is_blocked_host`) rejects private IP literals //! (including their IPv4-mapped IPv6 forms, e.g. `::ffff:127.0.0.1`), @@ -17,16 +19,25 @@ //! `read_body_capped` streams a response body and stops at a byte cap, so a //! hostile or gigantic page/feed cannot OOM the process before the size check //! runs. - -use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr}; +//! +//! An IPv6 literal URL such as `http://[::1]/` is refused whatever its address, +//! including a public one: `reqwest::Url::host_str` keeps the brackets, the +//! text no longer parses as an IP address, and a name with no dot is treated +//! as a single-label internal name. That is a fail-closed limitation, not a +//! classification: the address classifier (`is_public_ip`) does handle IPv6 +//! (and IPv4-mapped forms) for *resolved* addresses, which is how a hostname +//! with an AAAA record is still vetted. + +use std::net::{IpAddr, Ipv4Addr, SocketAddr}; use std::sync::Arc; use futures::stream::StreamExt; use reqwest::dns::{Addrs, Name, Resolve, Resolving}; /// Build an HTTP client with a redirect policy that re-applies the SSRF -/// host/scheme check to every redirect hop, and a DNS resolver that only -/// yields globally routable addresses. +/// host/scheme check to every redirect hop, a DNS resolver that only yields +/// globally routable addresses, a 20-second timeout and the `openhuman` +/// user agent the network readers have always sent. /// /// # Errors /// @@ -34,6 +45,7 @@ use reqwest::dns::{Addrs, Name, Resolve, Resolving}; pub fn build_client() -> Result<reqwest::Client, String> { reqwest::Client::builder() .timeout(std::time::Duration::from_secs(20)) + .user_agent("openhuman") .redirect(reqwest::redirect::Policy::custom(|attempt| { if is_url_allowed(attempt.url()) { attempt.follow() @@ -62,12 +74,12 @@ pub fn build_client() -> Result<reqwest::Client, String> { pub async fn read_body_capped(resp: reqwest::Response, max: u64) -> Result<Vec<u8>, String> { // Trust a truthful Content-Length up front so a known-huge body is // rejected before the first byte is read. - if let Some(len) = resp.content_length() { - if len > max { - return Err(format!( - "response body exceeds {max}-byte limit (Content-Length={len})" - )); - } + if let Some(len) = resp.content_length() + && len > max + { + return Err(format!( + "response body exceeds {max}-byte limit (Content-Length={len})" + )); } let mut body = Vec::new(); @@ -124,54 +136,59 @@ fn box_err( Box::new(e) } -/// Whether `ip` is a globally routable address — the resolved-address half of -/// the SSRF guard. Mirrors the literal/name policy in `is_blocked_host`: -/// loopback, private, link-local, unique-local, multicast, broadcast, -/// unspecified, and documentation/reserved ranges are not fetchable. +/// Whether `ip` is a globally routable address — the one address classifier +/// behind both halves of the SSRF guard (literal hosts in `is_blocked_host` +/// and resolved addresses in `PublicOnlyResolver`). +/// +/// Not fetchable: loopback, private, link-local, unspecified, CGNAT +/// (`100.64.0.0/10`), `192.0.0.0/16` (IETF protocol assignments and the +/// `192.0.2.0/24` documentation range), multicast, broadcast, documentation +/// (`198.51.100.0/24`, `203.0.113.0/24`, `2001:db8::/32`), benchmarking +/// (`198.18.0.0/15`), +/// reserved (`240.0.0.0/4`), and IPv6 unique-local (`fc00::/7`) and +/// link-local (`fe80::/10`). An IPv6 address carrying an IPv4 one — mapped +/// (`::ffff:a.b.c.d`) or the deprecated compatible form (`::a.b.c.d`) — is +/// judged by its IPv4 part, so a mapped loopback stays blocked. fn is_public_ip(ip: IpAddr) -> bool { match ip { - IpAddr::V4(v4) => is_public_ipv4(v4), - IpAddr::V6(v6) => is_public_ipv6(v6), - } -} - -fn is_public_ipv4(ip: Ipv4Addr) -> bool { - if is_private_ipv4(ip) || ip.is_multicast() || ip.is_broadcast() { - return false; - } - let o = ip.octets(); - // Documentation (192.0.2.0/24, 198.51.100.0/24, 203.0.113.0/24), - // benchmarking (198.18.0.0/15), and reserved (240.0.0.0/4) ranges are not - // globally routable. - !((o[0] == 192 && o[1] == 0 && o[2] == 2) - || (o[0] == 198 && o[1] == 51 && o[2] == 100) - || (o[0] == 203 && o[1] == 0 && o[2] == 113) - || (o[0] == 198 && o[1] == 18) - || o[0] >= 240) -} - -fn is_public_ipv6(ip: Ipv6Addr) -> bool { - if is_private_ipv6(ip) || ip.is_multicast() { - return false; - } - let o = ip.octets(); - // Documentation prefix 2001:db8::/32. - if o[0] == 0x20 && o[1] == 0x01 && o[2] == 0x0d && o[3] == 0xb8 { - return false; - } - // IPv4-mapped (`::ffff:a.b.c.d`) delegate to the embedded IPv4, so a - // mapped loopback/private address stays blocked. - if let Some(v4) = ip.to_ipv4_mapped() { - return is_public_ipv4(v4); - } - // The deprecated IPv4-compatible form (`::a.b.c.d`) carries an IPv4 - // address too, and `to_ipv4_mapped` answers `None` for it — so without - // this, `::127.0.0.1` and `::169.254.169.254` read as public. `::` and - // `::1` are judged as themselves above, before this reading applies. - if o[..12].iter().all(|byte| *byte == 0) { - return is_public_ipv4(Ipv4Addr::new(o[12], o[13], o[14], o[15])); + IpAddr::V4(v4) => { + let o = v4.octets(); + !(v4.is_loopback() + || v4.is_private() + || v4.is_link_local() + || v4.is_unspecified() + || v4.is_multicast() + || v4.is_broadcast() + || (o[0] == 100 && o[1] & 0xc0 == 0x40) + || (o[0] == 192 && o[1] == 0) + || (o[0] == 198 && o[1] == 51 && o[2] == 100) + || (o[0] == 203 && o[1] == 0 && o[2] == 113) + || (o[0] == 198 && o[1] & 0xfe == 18) + || o[0] >= 240) + } + IpAddr::V6(v6) => { + if v6.is_loopback() || v6.is_unspecified() || v6.is_multicast() { + return false; + } + let o = v6.octets(); + if (o[0] & 0xfe == 0xfc) + || (o[0] == 0xfe && o[1] & 0xc0 == 0x80) + || (o[0] == 0x20 && o[1] == 0x01 && o[2] == 0x0d && o[3] == 0xb8) + { + return false; + } + if let Some(v4) = v6.to_ipv4_mapped() { + return is_public_ip(IpAddr::V4(v4)); + } + // `to_ipv4_mapped` answers `None` for the compatible form, so + // without this `::127.0.0.1` would read as public. `::` and `::1` + // were judged as themselves above. + if o[..12].iter().all(|byte| *byte == 0) { + return is_public_ip(IpAddr::V4(Ipv4Addr::new(o[12], o[13], o[14], o[15]))); + } + true + } } - true } /// Whether a URL may be fetched: `http(s)` scheme against a public host. @@ -196,21 +213,11 @@ fn is_blocked_host(host: &str) -> bool { if host.is_empty() { return true; } - if let Ok(ip) = host.parse::<std::net::Ipv4Addr>() { - // Use the same public-address classification as the resolved-address - // guard (and the IPv6 literal branch) so reserved/multicast/broadcast/ - // documentation/benchmarking literals are rejected too. A literal never - // goes through DNS resolution, so the `PublicOnlyResolver` never sees - // it — this text check is the only line of defense for it. - return !is_public_ipv4(ip); - } - if let Ok(ip) = host.parse::<std::net::Ipv6Addr>() { - // Use the same public-address classification as the resolved-address - // guard so an IPv4-mapped literal (`::ffff:127.0.0.1`, - // `::ffff:10.0.0.1`) is rejected like its bare IPv4 counterpart. A - // literal never goes through DNS resolution, so the `PublicOnlyResolver` - // never sees it — this text check is the only line of defense for it. - return !is_public_ipv6(ip); + if let Ok(ip) = host.parse::<IpAddr>() { + // A literal never goes through DNS resolution, so `PublicOnlyResolver` + // never sees it — this text check is its only line of defense, and it + // uses the same classification as the resolver. + return !is_public_ip(ip); } if host == "localhost" || host.ends_with(".local") || host.ends_with(".internal") { return true; @@ -219,24 +226,6 @@ fn is_blocked_host(host: &str) -> bool { !host.contains('.') } -fn is_private_ipv4(ip: std::net::Ipv4Addr) -> bool { - if ip.is_loopback() || ip.is_private() || ip.is_link_local() || ip.is_unspecified() { - return true; - } - let o = ip.octets(); - // 100.64.0.0/10 CGNAT and 192.0.0.0/24 (IETF protocol assignments). - (o[0] == 100 && o[1] & 0xc0 == 0x40) || (o[0] == 192 && o[1] == 0) -} - -fn is_private_ipv6(ip: std::net::Ipv6Addr) -> bool { - if ip.is_loopback() || ip.is_unspecified() { - return true; - } - let o = ip.octets(); - // Unique-local fc00::/7 and link-local fe80::/10. - (o[0] == 0xfc || o[0] == 0xfd) || (o[0] == 0xfe && o[1] & 0xc0 == 0x80) -} - #[cfg(test)] -#[path = "ssrf_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/ssrf_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs similarity index 98% rename from crates/tinymemory-sources/src/readers/ssrf_tests.rs rename to crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs index 7df2bbb2..98f02183 100644 --- a/crates/tinymemory-sources/src/readers/ssrf_tests.rs +++ b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs @@ -1,3 +1,6 @@ +//! Tests for the SSRF guard: the address classifier, host and URL policy, +//! the capped body reader and the hardened client. + use super::*; async fn local_response(response: &'static [u8]) -> reqwest::Response { diff --git a/crates/tinymemory-sources/src/items/mod.rs b/crates/tinymemory-integrations/src/sources/items/mod.rs similarity index 93% rename from crates/tinymemory-sources/src/items/mod.rs rename to crates/tinymemory-integrations/src/sources/items/mod.rs index 5175da7c..40205a2e 100644 --- a/crates/tinymemory-sources/src/items/mod.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod.rs @@ -2,7 +2,7 @@ //! //! Every item a source produces names its reader in //! `meta.source = SourceRef { kind, id: Some(entry.id) }`, with the config kind -//! mapped through [`SourceKind::api_kind`](crate::SourceKind::api_kind). The +//! mapped through [`SourceKind::api_kind`](crate::sources::SourceKind::api_kind). The //! rest of the metadata depends on the kind: //! //! | Kind | Item | Metadata filled | @@ -11,10 +11,10 @@ //! | github | document | `repo` (`owner/name`), `commit` (commit items), `url` (issues and PRs), `observed_at` | //! | link | document | `url` | //! | rss | document | `url` (the entry's link), `observed_at` (published) | -//! | composio | document | `tags = [toolkit]` (payloads: see [`crate::composio`]) | +//! | composio | document | `tags = [toolkit]` (payloads: see [`crate::sources::composio`]) | //! | conversation | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | //! -//! Every document body is markdown, converted through `tinymemory-documents`: +//! Every document body is markdown, converted through `crate::documents`: //! local files through the host's [`DocumentConverter`] (so a bound PDF or //! DOCX converter applies), reader bodies through //! [`markdown_from_text`]. @@ -24,18 +24,18 @@ use std::path::Path; +use crate::documents::{ + DocumentConverter, DocumentFormat, document_item, language_for_path, markdown_from_text, +}; use chrono::{DateTime, TimeZone, Utc}; use tinymemory_api::{DocumentBody, MemoryMeta, SourceRef, StoreItem, TurnRange}; -use tinymemory_documents::{ - document_item, language_for_path, markdown_from_text, DocumentConverter, DocumentFormat, -}; -use crate::error::{Error, Result}; -use crate::readers::conversation::Thread; -use crate::readers::file::FileReader; -use crate::readers::local_file::LocalFile; -use crate::readers::SourceReader; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceKind}; +use crate::sources::error::{Error, Result}; +use crate::sources::readers::SourceReader; +use crate::sources::readers::conversation::Thread; +use crate::sources::readers::file::FileReader; +use crate::sources::readers::local_file::LocalFile; +use crate::sources::types::{ContentType, MemorySourceEntry, SourceContent, SourceKind}; /// Metadata naming `entry` as the source: `source.kind` is the entry's kind /// mapped onto the contract, `source.id` its id. A Composio entry also gets @@ -49,10 +49,10 @@ pub fn base_meta(entry: &MemorySourceEntry) -> MemoryMeta { }, ..MemoryMeta::default() }; - if entry.kind == SourceKind::Composio { - if let Some(toolkit) = entry.toolkit.as_deref().filter(|t| !t.is_empty()) { - meta.tags = vec![toolkit.to_string()]; - } + if entry.kind == SourceKind::Composio + && let Some(toolkit) = entry.toolkit.as_deref().filter(|t| !t.is_empty()) + { + meta.tags = vec![toolkit.to_string()]; } meta } diff --git a/crates/tinymemory-sources/src/items/mod_tests.rs b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs similarity index 98% rename from crates/tinymemory-sources/src/items/mod_tests.rs rename to crates/tinymemory-integrations/src/sources/items/mod_tests.rs index e2c4b57e..b050d8ea 100644 --- a/crates/tinymemory-sources/src/items/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs @@ -5,15 +5,15 @@ use super::*; use std::fs; +use crate::documents::ConverterChain; use async_trait::async_trait; use tempfile::TempDir; use tinymemory_api::{ItemKind, Role, SourceKind as Api, Turn}; -use tinymemory_documents::ConverterChain; -use crate::readers::conversation::ConversationReader; -use crate::readers::file::FileReader; -use crate::readers::folder::FolderReader; -use crate::types::SourceItem; +use crate::sources::readers::conversation::ConversationReader; +use crate::sources::readers::file::FileReader; +use crate::sources::readers::folder::FolderReader; +use crate::sources::types::SourceItem; fn entry(kind: SourceKind) -> MemorySourceEntry { MemorySourceEntry::new("src_test", kind, "Test") @@ -373,7 +373,7 @@ async fn a_file_no_converter_handles_is_skipped_not_fatal() { assert_eq!(collected.skipped[0].id, "scan.pdf"); assert!(matches!( collected.skipped[0].error, - Error::Document(tinymemory_documents::Error::UnsupportedFormat(_)) + Error::Document(crate::documents::Error::UnsupportedFormat(_)) )); } diff --git a/crates/tinymemory-sources/src/lib.rs b/crates/tinymemory-integrations/src/sources/mod.rs similarity index 70% rename from crates/tinymemory-sources/src/lib.rs rename to crates/tinymemory-integrations/src/sources/mod.rs index aed0e1da..dad654b8 100644 --- a/crates/tinymemory-sources/src/lib.rs +++ b/crates/tinymemory-integrations/src/sources/mod.rs @@ -3,28 +3,28 @@ //! into [`StoreItem`](tinymemory_api::StoreItem)s. //! //! - **Configuration** — what a source *is* ([`MemorySourceEntry`], keyed by -//! [`SourceKind`]), its partial updates ([`MemorySourcePatch`]), field rules -//! ([`validation`]), the host's persisted registry ([`SourceRegistry`]) and -//! Composio reconciliation ([`reconcile`]). +//! [`SourceKind`], checked by [`MemorySourceEntry::validate`]). Where the +//! host stores its sources, and how it edits them, is the host's business. //! - **Readers** — [`readers::SourceReader`] lists a source's items and reads //! one. Local readers (folder, file, conversation) are always compiled; the //! network readers (GitHub, RSS, web page) and `fetch` sit behind the -//! `network` feature, behind one SSRF guard (`readers::ssrf`). +//! `sources-network` feature; RSS and web pages fetch through `fetch`, +//! behind its one SSRF guard (`fetch::ssrf`). //! - **Items** — [`items`] maps reader output to `StoreItem`s with //! [`MemoryMeta`](tinymemory_api::MemoryMeta) filled per kind; every text -//! body is converted to markdown through `tinymemory-documents`. +//! body is converted to markdown through [`crate::documents`]. //! - **Composio** — [`composio`] normalises toolkit payloads (Gmail, Slack, //! GitHub, Linear, Notion, ClickUp) and maps them to items. //! -//! Scheduling, credentials and egress budgets stay with the host: this crate +//! Scheduling, credentials and egress budgets stay with the host: this module //! reads when asked. //! //! # Example //! //! ``` //! use tinymemory_api::{SourceKind as ApiKind, StoreItem}; -//! use tinymemory_documents::ConverterChain; -//! use tinymemory_sources::{items, readers, MemorySourceEntry, SourceKind}; +//! use tinymemory_integrations::documents::ConverterChain; +//! use tinymemory_integrations::sources::{items, readers, MemorySourceEntry, SourceKind}; //! //! # let runtime = tokio::runtime::Builder::new_current_thread().build()?; //! # runtime.block_on(async { @@ -57,30 +57,23 @@ //! //! # Feature flags //! -//! - `network` — the GitHub, RSS and web-page readers, `fetch`, and the -//! SSRF guard. Off by default, so a host that only reads local sources -//! links no HTTP stack. +//! - `sources` — everything above except the network pieces. Implies +//! `documents`; links no HTTP stack. +//! - `sources-network` — the GitHub, RSS and web-page readers, `fetch`, +//! `readers::reader_for_request` and the SSRF guard. Without it, a host that +//! only reads local sources links no HTTP stack. pub mod composio; pub mod error; -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] pub mod fetch; pub mod items; -pub mod raw_kind; pub mod readers; -pub mod reconcile; -pub mod registry; pub mod types; -pub mod validation; /// Largest file a folder or file source will read. pub const FOLDER_FILE_SIZE_CAP_BYTES: u64 = 10 * 1024 * 1024; pub use error::{Error, Result}; -pub use items::{collect_items, content_item, conversation_item, file_item, Collected}; -pub use registry::{ - apply_kind_defaults, memory_sync_defaults_for_toolkit, ComposioUpsertTarget, SourceRegistry, -}; -pub use types::{ - ContentType, MemorySourceEntry, MemorySourcePatch, SourceContent, SourceItem, SourceKind, -}; +pub use items::{Collected, collect_items, content_item, conversation_item, file_item}; +pub use types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; diff --git a/crates/tinymemory-sources/src/readers/composio.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs similarity index 81% rename from crates/tinymemory-sources/src/readers/composio.rs rename to crates/tinymemory-integrations/src/sources/readers/composio/mod.rs index d1903331..5d0c3d84 100644 --- a/crates/tinymemory-sources/src/readers/composio.rs +++ b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs @@ -1,26 +1,28 @@ //! Composio source reader — a placeholder over the provider pipeline. //! //! Composio data does not arrive item by item: the host runs toolkit actions -//! with its credentials and hands the responses to [`crate::composio`], which +//! with its credentials and hands the responses to [`crate::sources::composio`], which //! normalises them and maps them to `StoreItem`s. For a Composio source, //! `list_items` returns the connection as one sync target and `read_item` -//! describes that pipeline. The reader exists so the registry can query every -//! source kind uniformly. +//! describes that pipeline. The reader exists so `reader_for_request` can hand +//! out a reader for every source kind uniformly. use std::path::Path; use async_trait::async_trait; use super::SourceReader; -use crate::error::Result; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; +use crate::sources::error::Result; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; /// Lists a Composio connection as a single sync target. /// /// Composio data arrives through the provider sync pipeline rather than /// item-by-item, so `read_item` returns a description of that rather than -/// content. The reader exists so the registry can query every source kind -/// uniformly. +/// content. The reader exists so `reader_for_request` can serve every source +/// kind uniformly. #[derive(Debug, Clone, Copy, Default)] pub struct ComposioReader; @@ -72,5 +74,5 @@ impl SourceReader for ComposioReader { } #[cfg(test)] -#[path = "composio_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/composio_tests.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/composio_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs diff --git a/crates/tinymemory-sources/src/readers/conversation.rs b/crates/tinymemory-integrations/src/sources/readers/conversation/mod.rs similarity index 95% rename from crates/tinymemory-sources/src/readers/conversation.rs rename to crates/tinymemory-integrations/src/sources/readers/conversation/mod.rs index 01fd59b1..eeba0b1c 100644 --- a/crates/tinymemory-sources/src/readers/conversation.rs +++ b/crates/tinymemory-integrations/src/sources/readers/conversation/mod.rs @@ -4,25 +4,28 @@ //! Threads are JSON files under `<workspace>/threads/`, shaped //! `{ title, messages: [{ role, content, created_at? }] }`. As a //! [`SourceContent`] a thread renders to markdown; as a store item it becomes -//! a [`StoreItem::Conversation`] with one [`Turn`] per non-empty message. +//! a [`StoreItem::Conversation`] with one [`Turn`] per non-empty message whose +//! role is known (`user`, `assistant`, `system`, `tool` and their usual aliases); +//! a message with any other role is skipped. //! //! Safety: `item_id` is rejected if it contains path separators or `..`, and the //! resolved file is re-checked for containment within the threads directory. use std::path::{Path, PathBuf}; +use crate::documents::DocumentConverter; use async_trait::async_trait; use chrono::{DateTime, TimeZone, Utc}; use tinymemory_api::{Role, StoreItem, Turn}; -use tinymemory_documents::DocumentConverter; -use crate::error::{Error, Result}; -use crate::items; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; -use crate::validation::ensure_within_base; +use crate::sources::error::{Error, Result}; +use crate::sources::items; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; -use super::local_file::modified_at; use super::SourceReader; +use super::local_file::{ensure_within_base, modified_at}; /// One thread read from disk, parsed into turns. #[derive(Debug, Clone, PartialEq)] @@ -270,5 +273,5 @@ fn format_thread_as_markdown(thread: &serde_json::Value) -> String { } #[cfg(test)] -#[path = "conversation_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/conversation_tests.rs b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs similarity index 96% rename from crates/tinymemory-sources/src/readers/conversation_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs index 1376d630..8f49976f 100644 --- a/crates/tinymemory-sources/src/readers/conversation_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs @@ -193,19 +193,23 @@ async fn read_item_rejects_path_traversal() { let result = reader.read_item(&source, "../config", config).await; assert!(result.is_err()); - assert!(result - .unwrap_err() - .to_string() - .contains("path traversal denied")); + assert!( + result + .unwrap_err() + .to_string() + .contains("path traversal denied") + ); let result = reader .read_item(&source, "foo/../../etc/passwd", config) .await; assert!(result.is_err()); - assert!(result - .unwrap_err() - .to_string() - .contains("path traversal denied")); + assert!( + result + .unwrap_err() + .to_string() + .contains("path traversal denied") + ); } #[test] diff --git a/crates/tinymemory-sources/src/readers/file.rs b/crates/tinymemory-integrations/src/sources/readers/file/mod.rs similarity index 90% rename from crates/tinymemory-sources/src/readers/file.rs rename to crates/tinymemory-integrations/src/sources/readers/file/mod.rs index 7f03fc68..4f76367f 100644 --- a/crates/tinymemory-sources/src/readers/file.rs +++ b/crates/tinymemory-integrations/src/sources/readers/file/mod.rs @@ -6,20 +6,22 @@ //! //! [`FileReader::read_path`] reads a file with no configured source at all, //! for a host that was handed a path (a drag-and-drop, a CLI argument); -//! [`crate::items::file_item`] turns that straight into a `StoreItem`. +//! [`crate::sources::items::file_item`] turns that straight into a `StoreItem`. use std::path::{Path, PathBuf}; +use crate::documents::DocumentConverter; use async_trait::async_trait; use tinymemory_api::StoreItem; -use tinymemory_documents::DocumentConverter; -use crate::error::{Error, Result}; -use crate::items; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; +use crate::sources::error::{Error, Result}; +use crate::sources::items; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; -use super::local_file::{modified_at, read_capped, resolve_base, LocalFile}; use super::SourceReader; +use super::local_file::{LocalFile, modified_at, read_capped, resolve_base}; /// A reader over one local file. #[derive(Debug, Clone, Copy, Default)] @@ -48,7 +50,7 @@ impl FileReader { /// /// [`Error::NotFound`] for a missing file, [`Error::Invalid`] for a path /// that is not a regular file, [`Error::TooLarge`] for one over - /// [`crate::FOLDER_FILE_SIZE_CAP_BYTES`], and [`Error::Io`] for a read + /// [`crate::sources::FOLDER_FILE_SIZE_CAP_BYTES`], and [`Error::Io`] for a read /// failure. pub fn read_path(path: &Path) -> Result<LocalFile> { if !path.exists() { @@ -155,5 +157,5 @@ impl SourceReader for FileReader { } #[cfg(test)] -#[path = "file_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/file_tests.rs b/crates/tinymemory-integrations/src/sources/readers/file/mod_tests.rs similarity index 97% rename from crates/tinymemory-sources/src/readers/file_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/file/mod_tests.rs index 1daa243c..ecd8fb0f 100644 --- a/crates/tinymemory-sources/src/readers/file_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/file/mod_tests.rs @@ -71,7 +71,8 @@ fn read_path_refuses_directories_and_oversized_files() { let huge = dir.path().join("huge.txt"); let file = fs::File::create(&huge).unwrap(); - file.set_len(crate::FOLDER_FILE_SIZE_CAP_BYTES + 1).unwrap(); + file.set_len(crate::sources::FOLDER_FILE_SIZE_CAP_BYTES + 1) + .unwrap(); drop(file); let error = FileReader::read_path(&huge).unwrap_err(); assert!(matches!(error, Error::TooLarge(_)), "got {error:?}"); diff --git a/crates/tinymemory-sources/src/readers/folder.rs b/crates/tinymemory-integrations/src/sources/readers/folder/mod.rs similarity index 96% rename from crates/tinymemory-sources/src/readers/folder.rs rename to crates/tinymemory-integrations/src/sources/readers/folder/mod.rs index e4def36c..64030f09 100644 --- a/crates/tinymemory-sources/src/readers/folder.rs +++ b/crates/tinymemory-integrations/src/sources/readers/folder/mod.rs @@ -19,20 +19,21 @@ use std::path::{Path, PathBuf}; +use crate::documents::{DocumentConverter, DocumentFormat, language_for_path}; use async_trait::async_trait; use regex::Regex; use tinymemory_api::StoreItem; -use tinymemory_documents::{language_for_path, DocumentConverter, DocumentFormat}; use walkdir::WalkDir; -use crate::error::{Error, Result}; -use crate::items; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; -use crate::validation::ensure_within_base; -use crate::FOLDER_FILE_SIZE_CAP_BYTES; +use crate::sources::FOLDER_FILE_SIZE_CAP_BYTES; +use crate::sources::error::{Error, Result}; +use crate::sources::items; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; -use super::local_file::{modified_at, read_capped, resolve_base, LocalFile}; use super::SourceReader; +use super::local_file::{LocalFile, ensure_within_base, modified_at, read_capped, resolve_base}; /// Directory names never descended into, wherever they appear. const IGNORED_DIRS: &[&str] = &[ @@ -352,5 +353,5 @@ fn glob_to_regex(pattern: &str) -> Result<Regex> { } #[cfg(test)] -#[path = "folder_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/folder_tests.rs b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs similarity index 96% rename from crates/tinymemory-sources/src/readers/folder_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs index 8a7b630e..32060ca7 100644 --- a/crates/tinymemory-sources/src/readers/folder_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs @@ -110,7 +110,7 @@ async fn list_items_skips_hidden_and_build_directories() { .await .unwrap_err(); assert!( - matches!(error, crate::Error::Invalid(_)), + matches!(error, crate::sources::Error::Invalid(_)), "{hidden}: {error:?}" ); } @@ -180,10 +180,12 @@ async fn read_item_enforces_configured_glob() { source.glob = Some("docs/**/*.md".into()); let reader = FolderReader; - assert!(reader - .read_item(&source, "docs/allowed.md", config()) - .await - .is_ok()); + assert!( + reader + .read_item(&source, "docs/allowed.md", config()) + .await + .is_ok() + ); let err = reader .read_item(&source, "docs/secret.env", config()) .await @@ -250,11 +252,13 @@ async fn oversized_files_are_not_listed_and_cannot_be_read() { let source = folder_source(&tmp.path().to_string_lossy()); let reader = FolderReader; - assert!(reader - .list_items(&source, config()) - .await - .unwrap() - .is_empty()); + assert!( + reader + .list_items(&source, config()) + .await + .unwrap() + .is_empty() + ); let error = reader .read_item(&source, "huge.md", config()) .await @@ -309,17 +313,19 @@ async fn symlinks_cannot_escape_the_configured_folder() { let source = folder_source(&base.path().to_string_lossy()); let reader = FolderReader; - assert!(reader - .list_items(&source, config()) - .await - .unwrap() - .is_empty()); + assert!( + reader + .list_items(&source, config()) + .await + .unwrap() + .is_empty() + ); let error = reader .read_item(&source, "escape.md", config()) .await .unwrap_err(); assert!( - matches!(error, crate::Error::PathEscape(_)), + matches!(error, crate::sources::Error::PathEscape(_)), "got {error:?}" ); } diff --git a/crates/tinymemory-sources/src/readers/github/api.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs similarity index 96% rename from crates/tinymemory-sources/src/readers/github/api.rs rename to crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs index 81268702..0ec4a8b6 100644 --- a/crates/tinymemory-sources/src/readers/github/api.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs @@ -12,16 +12,15 @@ use std::collections::HashSet; -use crate::types::{ContentType, SourceContent, SourceItem}; +use crate::sources::types::{ContentType, SourceContent, SourceItem}; use super::types::GhCommit; -use super::{parse_iso_ts, GH_CLI_TIMEOUT}; +use super::{GH_CLI_TIMEOUT, parse_iso_ts}; -// Keep the production transport at its established source locations. This file -// is compiled both as the standalone sources crate and through downstream -// workspace consumers, and LLVM merges their regions by source coordinate. -// Moving these functions would turn otherwise identical regions into apparent -// duplicate production lines. Only the deterministic response queue belongs in +// Keep the production transport at its established source locations. Coverage +// tools merge regions by source coordinate, so moving these functions would +// turn otherwise identical regions into apparent duplicate production lines. +// Only the deterministic response queue belongs in // the selected external module below; the actual transport remains here. // // The deliberately expanded explanation also occupies the source range that @@ -32,10 +31,10 @@ use super::{parse_iso_ts, GH_CLI_TIMEOUT}; // Its behavior is unchanged; only the test override storage moved. // #[cfg(not(test))] -#[path = "api/transport_override.rs"] +#[path = "transport_override.rs"] mod response_override; #[cfg(test)] -#[path = "api/transport_tests.rs"] +#[path = "transport_tests.rs"] mod response_override; #[cfg(test)] diff --git a/crates/tinymemory-sources/src/readers/github/api/transport_override.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/github/api/transport_override.rs rename to crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs diff --git a/crates/tinymemory-sources/src/readers/github/api/transport_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/github/api/transport_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs diff --git a/crates/tinymemory-sources/src/readers/github/git.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs similarity index 99% rename from crates/tinymemory-sources/src/readers/github/git.rs rename to crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs index 0a330bcb..b9b181ae 100644 --- a/crates/tinymemory-sources/src/readers/github/git.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs @@ -14,7 +14,7 @@ use std::path::{Path, PathBuf}; use std::time::Duration; -use crate::types::{ContentType, SourceContent, SourceItem}; +use crate::sources::types::{ContentType, SourceContent, SourceItem}; use super::parse_iso_ts; @@ -316,5 +316,5 @@ pub(super) async fn read_commit_git( } #[cfg(test)] -#[path = "git_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/github/git_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs similarity index 93% rename from crates/tinymemory-sources/src/readers/github/git_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs index 9adc9e61..af28f58d 100644 --- a/crates/tinymemory-sources/src/readers/github/git_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs @@ -1,3 +1,6 @@ +//! Tests for the local bare-clone helpers: `git log` arguments, cache +//! lifecycle, and process failures. + use super::*; use std::process::Command; @@ -181,10 +184,12 @@ async fn local_bare_clone_lists_filters_and_renders_commits() { async fn git_helpers_surface_missing_cache_ref_and_process_failures() { let tmp = tempfile::tempdir().expect("tempdir"); let missing = tmp.path().join("missing.git"); - assert!(read_commit_git("owner", "repo", "deadbeef", &missing) - .await - .expect_err("missing cache") - .contains("not present")); + assert!( + read_commit_git("owner", "repo", "deadbeef", &missing) + .await + .expect_err("missing cache") + .contains("not present") + ); let src = tmp.path().join("src"); init_repo(&src); @@ -199,10 +204,12 @@ async fn git_helpers_surface_missing_cache_ref_and_process_failures() { cache.to_str().expect("cache path"), ], ); - assert!(read_commit_git("owner", "repo", "not-a-ref", &cache) - .await - .expect_err("unknown ref") - .contains("git show exited")); + assert!( + read_commit_git("owner", "repo", "not-a-ref", &cache) + .await + .expect_err("unknown ref") + .contains("git show exited") + ); assert!( list_commits_git("owner", "repo", 10, &cache, Some("missing"), &[]) .await diff --git a/crates/tinymemory-sources/src/readers/github/issues.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs similarity index 98% rename from crates/tinymemory-sources/src/readers/github/issues.rs rename to crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs index 0937d3d2..cb19b2d0 100644 --- a/crates/tinymemory-sources/src/readers/github/issues.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs @@ -8,9 +8,9 @@ use serde::Deserialize; -use crate::types::{ContentType, SourceContent, SourceItem}; +use crate::sources::types::{ContentType, SourceContent, SourceItem}; -use super::api::{fetch_all_pages, fetch_github, GH_MAX_PAGES, GH_PAGE_SIZE}; +use super::api::{GH_MAX_PAGES, GH_PAGE_SIZE, fetch_all_pages, fetch_github}; use super::types::{CachedItem, GhIssue, GhPr, GhUser, IssueComment}; use super::{parse_iso_ts, unique_handles}; @@ -300,5 +300,5 @@ async fn fetch_issue_comments( } #[cfg(test)] -#[path = "issues_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/github/issues_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs similarity index 95% rename from crates/tinymemory-sources/src/readers/github/issues_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs index 154e598b..5a521680 100644 --- a/crates/tinymemory-sources/src/readers/github/issues_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs @@ -1,8 +1,8 @@ //! Offline behavioral tests for issue and pull-request list/read orchestration. use super::*; -use crate::readers::github::api::with_test_responses; -use crate::readers::github::types::LIST_CACHE; +use crate::sources::readers::github::api::with_test_responses; +use crate::sources::readers::github::types::LIST_CACHE; fn issue_json(number: u64) -> serde_json::Value { serde_json::json!({ @@ -96,9 +96,10 @@ async fn lists_cache_and_render_issues_and_pull_requests_without_network() { ) .await .expect("read cached pull request despite malformed comments"); - assert!(pr - .body - .contains("**State:** closed (merged at 2026-01-05T00:00:00Z)")); + assert!( + pr.body + .contains("**State:** closed (merged at 2026-01-05T00:00:00Z)") + ); assert!(pr.body.contains("**Participants:** @bob")); assert!(!pr.body.contains("## Comments")); assert_eq!(pr.metadata["merged"], true); diff --git a/crates/tinymemory-sources/src/readers/github.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs similarity index 81% rename from crates/tinymemory-sources/src/readers/github.rs rename to crates/tinymemory-integrations/src/sources/readers/github/mod.rs index 576204e9..4ace4648 100644 --- a/crates/tinymemory-sources/src/readers/github.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs @@ -1,14 +1,17 @@ //! GitHub repo source reader. //! //! Pulls **project activity** (commits, issues, PRs) from a GitHub -//! repository — not source code. Uses the `gh` CLI when available for -//! authenticated, higher-rate-limit access; falls back to the public -//! GitHub REST API for unauthenticated reads. +//! repository — not source code. Commits are read from a local bare clone +//! under `<workspace>/git_cache/` (`git` must be on `PATH`), falling back to +//! the API when the clone fails. Issues and pull requests, and that fallback, +//! go through the `gh` CLI when it is available (authenticated, higher rate +//! limit) and otherwise the public, unauthenticated GitHub REST API. +//! `gh_available` is probed once per process. //! //! ## Module layout //! //! - [`self`] — [`GithubReader`] orchestration: item listing/reading, URL -//! parsing, raw-archive coordinates, shared utilities, and the cached +//! parsing, shared utilities, and the cached //! `gh`-availability probe. //! - `types` — API response models and the `gh`-fallback list cache. //! - `git` — local bare-clone + `git log` / `git show` helpers. @@ -21,16 +24,15 @@ mod issues; mod types; #[cfg(test)] -#[path = "github_tests.rs"] +#[path = "mod_tests.rs"] mod tests; use std::time::Duration; use async_trait::async_trait; -use crate::error::{Error, Result}; -use crate::raw_kind::RawKind; -use crate::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind}; +use crate::sources::error::{Error, Result}; +use crate::sources::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind}; use super::SourceReader; @@ -73,7 +75,9 @@ async fn gh_available() -> bool { } /// Reader for a GitHub repository source: lists and fetches commits, issues -/// and pull requests via the REST API, and file content via a shallow clone. +/// and pull requests. Item ids are `commit:<sha>`, `issue:<n>` and `pr:<n>`. +/// Commits come from a local bare clone with an API fallback; issues and pull +/// requests come from `gh api` or the REST API. #[derive(Debug, Clone, Copy, Default)] pub struct GithubReader; @@ -100,48 +104,6 @@ pub(crate) fn parse_github_url(url: &str) -> std::result::Result<(String, String Ok((parts[0].to_string(), parts[1].to_string())) } -// ── Raw-archive coordinates ───────────────────────────────────────── - -/// Slugifiable raw-archive source id for a repo URL. -/// -/// Returns `github.com/<owner>/<repo>`, which slugifies (via -/// `slugify_source_id`) to `github-com-<owner>-<repo>` so a source's -/// commits/issues/PRs land under -/// `raw/github-com-<owner>-<repo>/{commits,issues,prs}/`. -pub fn repo_archive_source_id(url: &str) -> Option<String> { - let (owner, repo) = parse_github_url(url).ok()?; - Some(format!("github.com/{owner}/{repo}")) -} - -/// Chunk-store source id for a single repo item (dedup key). -/// -/// `github:<owner>/<repo>:<item_id>` keeps per-item uniqueness for the -/// `mem_tree_ingested_sources` dedup table while the separate -/// [`repo_chunk_scope`] drives a shared directory. -pub fn chunk_source_id(url: &str, item_id: &str) -> Option<String> { - let (owner, repo) = parse_github_url(url).ok()?; - Some(format!("github:{owner}/{repo}:{item_id}")) -} - -/// Repo-scoped chunk path scope so all items from one repo share a -/// single directory in the content store (e.g. `document/github-org-repo/`). -pub fn repo_chunk_scope(url: &str) -> Option<String> { - let (owner, repo) = parse_github_url(url).ok()?; - Some(format!("github:{owner}/{repo}")) -} - -/// Map a [`SourceItem`] id (`commit:<sha>`, `issue:<n>`, `pr:<n>`) to its -/// raw-archive [`RawKind`] and the clean uid used as the filename suffix. -pub fn raw_archive_coords(item_id: &str) -> Option<(RawKind, String)> { - let (kind, rest) = ItemKind::from_id(item_id)?; - let raw_kind = match kind { - ItemKind::Commit => RawKind::Commit, - ItemKind::Issue => RawKind::Issue, - ItemKind::PullRequest => RawKind::PullRequest, - }; - Some((raw_kind, rest.to_string())) -} - // ── Reader implementation ─────────────────────────────────────────── #[async_trait] diff --git a/crates/tinymemory-sources/src/readers/github_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs similarity index 91% rename from crates/tinymemory-sources/src/readers/github_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs index a646c55d..66168524 100644 --- a/crates/tinymemory-sources/src/readers/github_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs @@ -1,6 +1,8 @@ +//! Tests for the GitHub reader: orchestration over cached clones and the +//! test transport, commit queries and merging, and URL and item-id parsing. + use super::*; -use crate::raw_kind::RawKind; -use crate::readers::SourceReader; +use crate::sources::readers::SourceReader; fn github_source(url: Option<&str>) -> MemorySourceEntry { MemorySourceEntry { @@ -93,17 +95,21 @@ async fn reader_rejects_missing_urls_and_malformed_item_ids_before_network() { let reader = GithubReader; let missing = github_source(None); assert!(reader.list_items(&missing, workspace.path()).await.is_err()); - assert!(reader - .read_item(&missing, "commit:abc", workspace.path()) - .await - .is_err()); + assert!( + reader + .read_item(&missing, "commit:abc", workspace.path()) + .await + .is_err() + ); let configured = github_source(Some("https://github.com/local/fixture")); for item_id in ["unknown", "issue:not-a-number", "pr:not-a-number"] { - assert!(reader - .read_item(&configured, item_id, workspace.path()) - .await - .is_err()); + assert!( + reader + .read_item(&configured, item_id, workspace.path()) + .await + .is_err() + ); } } @@ -258,9 +264,11 @@ async fn api_commit_fallback_lists_merges_and_renders_without_network() { .await .expect("read deterministic commit"); assert_eq!(content.title, "newer commit"); - assert!(content - .body - .contains("Test Author <author@example.com> (@octocat)")); + assert!( + content + .body + .contains("Test Author <author@example.com> (@octocat)") + ); assert_eq!(content.metadata["author_handle"], "octocat"); } @@ -497,15 +505,16 @@ async fn fetch_all_pages_stops_at_a_short_page() { // A short page (fewer than GH_PAGE_SIZE rows) is the last page; the walk // must not request page 2 after it. let mut requested: Vec<u32> = Vec::new(); - let pages = crate::readers::github::api::collect_pages::<u64, _, _>("commits", 1000, |page| { - requested.push(page); - async move { - // Page 1 is short (3 rows) — stop after it even though max is large. - Ok("[1,2,3]".to_string()) - } - }) - .await - .unwrap(); + let pages = + crate::sources::readers::github::api::collect_pages::<u64, _, _>("commits", 1000, |page| { + requested.push(page); + async move { + // Page 1 is short (3 rows) — stop after it even though max is large. + Ok("[1,2,3]".to_string()) + } + }) + .await + .unwrap(); assert_eq!(requested, vec![1]); assert_eq!(pages, vec![1, 2, 3]); @@ -604,28 +613,6 @@ fn item_kind_rejects_invalid() { assert!(ItemKind::from_id("noprefix").is_none()); } -#[test] -fn repo_archive_source_id_slugs_to_repo_folder() { - // `github.com/<owner>/<repo>` → slugify → `github-com-<owner>-<repo>`. - assert_eq!( - repo_archive_source_id("https://github.com/tinyhumansai/openhuman").as_deref(), - Some("github.com/tinyhumansai/openhuman") - ); - assert!(repo_archive_source_id("not-a-url").is_none()); -} - -#[test] -fn chunk_source_id_is_clean_and_per_item() { - assert_eq!( - chunk_source_id("https://github.com/org/repo", "commit:abc123").as_deref(), - Some("github:org/repo:commit:abc123") - ); - assert_eq!( - chunk_source_id("https://github.com/org/repo", "pr:42").as_deref(), - Some("github:org/repo:pr:42") - ); -} - #[test] fn unique_handles_dedups_and_skips_unknown() { assert_eq!( @@ -635,20 +622,3 @@ fn unique_handles_dedups_and_skips_unknown() { assert_eq!(unique_handles(["unknown", ""].into_iter()), "none"); assert_eq!(unique_handles(std::iter::empty()), "none"); } - -#[test] -fn raw_archive_coords_maps_kind_and_uid() { - assert_eq!( - raw_archive_coords("commit:deadbeef"), - Some((RawKind::Commit, "deadbeef".to_string())) - ); - assert_eq!( - raw_archive_coords("issue:7"), - Some((RawKind::Issue, "7".to_string())) - ); - assert_eq!( - raw_archive_coords("pr:99"), - Some((RawKind::PullRequest, "99".to_string())) - ); - assert!(raw_archive_coords("bogus:1").is_none()); -} diff --git a/crates/tinymemory-sources/src/readers/github/types.rs b/crates/tinymemory-integrations/src/sources/readers/github/types.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/github/types.rs rename to crates/tinymemory-integrations/src/sources/readers/github/types.rs diff --git a/crates/tinymemory-sources/src/readers/local_file.rs b/crates/tinymemory-integrations/src/sources/readers/local_file/mod.rs similarity index 73% rename from crates/tinymemory-sources/src/readers/local_file.rs rename to crates/tinymemory-integrations/src/sources/readers/local_file/mod.rs index 0d5dc722..71721b2f 100644 --- a/crates/tinymemory-sources/src/readers/local_file.rs +++ b/crates/tinymemory-integrations/src/sources/readers/local_file/mod.rs @@ -3,15 +3,17 @@ //! The folder and file readers share this: both resolve a configured path //! against the workspace, both refuse files over //! [`FOLDER_FILE_SIZE_CAP_BYTES`], and both hand the raw bytes on so a host -//! converter can handle formats that are not UTF-8 text (PDF, DOCX). +//! converter can handle formats that are not UTF-8 text (PDF, DOCX). The +//! path-containment guard every local reader applies, [`ensure_within_base`], +//! lives here too. use std::path::{Path, PathBuf}; +use crate::documents::RawDocument; use chrono::{DateTime, Utc}; -use tinymemory_documents::RawDocument; -use crate::error::{Error, Result}; -use crate::FOLDER_FILE_SIZE_CAP_BYTES; +use crate::sources::FOLDER_FILE_SIZE_CAP_BYTES; +use crate::sources::error::{Error, Result}; /// A file read from disk, before any conversion. #[derive(Debug, Clone)] @@ -66,6 +68,26 @@ pub(crate) fn resolve_base(base_path: &str, workspace: &Path) -> PathBuf { } } +/// Canonicalize `target` and ensure it stays within canonicalized `base`. +/// +/// This is the shared path-traversal guard for local readers. Both paths must +/// exist (they are passed through [`std::fs::canonicalize`], which resolves +/// symlinks and `..` segments). If the resolved target escapes the base +/// directory, the guard refuses it. +/// +/// # Errors +/// +/// [`Error::PathEscape`] carrying `"path traversal denied"` when the target +/// escapes, [`Error::Io`] when either path cannot be canonicalised. +pub fn ensure_within_base(base: &Path, target: &Path) -> Result<PathBuf> { + let canonical_base = std::fs::canonicalize(base)?; + let canonical_target = std::fs::canonicalize(target)?; + if !canonical_target.starts_with(&canonical_base) { + return Err(Error::PathEscape("path traversal denied".to_string())); + } + Ok(canonical_target) +} + /// The modification time of `metadata`, as a UTC instant. pub(crate) fn modified_at(metadata: &std::fs::Metadata) -> Option<DateTime<Utc>> { metadata.modified().ok().map(DateTime::<Utc>::from) @@ -89,3 +111,7 @@ pub(crate) fn read_capped(canonical: PathBuf, id: String) -> Result<LocalFile> { bytes, }) } + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/local_file/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/local_file/mod_tests.rs new file mode 100644 index 00000000..09e8f6cc --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/local_file/mod_tests.rs @@ -0,0 +1,23 @@ +//! Tests for the path-containment guard the local readers share. + +use super::*; +use std::fs; +use tempfile::TempDir; + +#[test] +fn ensure_within_base_accepts_contained_file() { + let tmp = TempDir::new().unwrap(); + fs::write(tmp.path().join("ok.md"), "hi").unwrap(); + let resolved = ensure_within_base(tmp.path(), &tmp.path().join("ok.md")).unwrap(); + assert!(resolved.ends_with("ok.md")); +} + +#[test] +fn ensure_within_base_rejects_escape() { + let tmp = TempDir::new().unwrap(); + fs::write(tmp.path().join("ok.md"), "hi").unwrap(); + // Build a target that escapes the base via `..`. + let escaping = tmp.path().join("../../etc/hosts"); + let result = ensure_within_base(tmp.path(), &escaping); + assert!(result.is_err()); +} diff --git a/crates/tinymemory-sources/src/readers/mod.rs b/crates/tinymemory-integrations/src/sources/readers/mod.rs similarity index 81% rename from crates/tinymemory-sources/src/readers/mod.rs rename to crates/tinymemory-integrations/src/sources/readers/mod.rs index b15fd801..503a21ee 100644 --- a/crates/tinymemory-sources/src/readers/mod.rs +++ b/crates/tinymemory-integrations/src/sources/readers/mod.rs @@ -10,9 +10,9 @@ //! //! The local kinds ([`folder::FolderReader`], [`file::FileReader`], //! [`conversation::ConversationReader`]) are always compiled. The network -//! kinds (`github`, `rss`, `web_page`, plus `fetch`) sit -//! behind the `network` feature. What this crate does **not** own is *when* -//! a network read happens: scheduling, polling cadence, OAuth, credentials, +//! kinds (`github`, `rss`, `web_page`) sit behind the `sources-network` +//! feature; `rss` and `web_page` fetch through `sources::fetch`. What this module does **not** own is *when* a +//! network read happens: scheduling, polling cadence, OAuth, credentials, //! and egress/cost budgeting stay with the host. //! //! That is why [`reader_for`] and [`is_locally_readable`] draw their line at @@ -25,7 +25,7 @@ //! //! `composio` is represented by a placeholder reader //! ([`composio::ComposioReader`]): its data arrives through the credentialed -//! provider pipeline, and [`crate::composio`] turns those payloads into items. +//! provider pipeline, and [`crate::sources::composio`] turns those payloads into items. //! //! A host servicing an *explicit user request* (not a timer) that wants one //! reader for any kind uses `reader_for_request`. @@ -34,30 +34,22 @@ pub mod composio; pub mod conversation; pub mod file; pub mod folder; -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] pub mod github; pub mod local_file; -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] pub mod rss; -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] pub mod web_page; -/// SSRF guard + fetch hygiene shared by the network readers and -/// [`crate::fetch`]. See the `ssrf` module docs. -/// -/// Public so a host fetching a user-supplied URL by other means applies the -/// same policy rather than a second, weaker one. -#[cfg(feature = "network")] -pub mod ssrf; - use std::path::Path; +use crate::documents::DocumentConverter; use async_trait::async_trait; use tinymemory_api::StoreItem; -use tinymemory_documents::DocumentConverter; -use crate::error::Result; -use crate::items; +use crate::sources::error::Result; +use crate::sources::items; use super::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind}; @@ -74,8 +66,8 @@ pub trait SourceReader: Send + Sync + std::fmt::Debug { /// /// # Errors /// - /// The reader's failure: missing configuration ([`crate::Error::Invalid`]), - /// a missing root ([`crate::Error::NotFound`]), or a network failure. + /// The reader's failure: missing configuration ([`crate::sources::Error::Invalid`]), + /// a missing root ([`crate::sources::Error::NotFound`]), or a network failure. async fn list_items( &self, source: &MemorySourceEntry, @@ -86,8 +78,8 @@ pub trait SourceReader: Send + Sync + std::fmt::Debug { /// /// # Errors /// - /// The reader's failure: an unknown item ([`crate::Error::NotFound`]), a - /// path that escapes its root ([`crate::Error::PathEscape`]), a body over + /// The reader's failure: an unknown item ([`crate::sources::Error::NotFound`]), a + /// path that escapes its root ([`crate::sources::Error::PathEscape`]), a body over /// the size cap, or a network failure. async fn read_item( &self, @@ -105,8 +97,8 @@ pub trait SourceReader: Send + Sync + std::fmt::Debug { /// /// # Errors /// - /// Whatever [`Self::read_item`] returns, plus [`crate::Error::Document`] - /// when conversion fails and [`crate::Error::Invalid`] for an item with no + /// Whatever [`Self::read_item`] returns, plus [`crate::sources::Error::Document`] + /// when conversion fails and [`crate::sources::Error::Invalid`] for an item with no /// text. async fn read_store_item( &self, @@ -160,7 +152,7 @@ pub fn reader_for(kind: &SourceKind) -> Option<Box<dyn SourceReader>> { /// already decided the fetch is allowed. **Do not reuse it from a polling /// loop**: the host stays in charge of egress, OAuth and cost budgeting by /// constructing a network reader deliberately there. -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] #[must_use] pub fn reader_for_request(kind: &SourceKind) -> Box<dyn SourceReader> { match kind { diff --git a/crates/tinymemory-sources/src/readers/rss.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs similarity index 77% rename from crates/tinymemory-sources/src/readers/rss.rs rename to crates/tinymemory-integrations/src/sources/readers/rss/mod.rs index e9a54e45..ec042713 100644 --- a/crates/tinymemory-sources/src/readers/rss.rs +++ b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs @@ -4,9 +4,9 @@ //! source items. Uses a lightweight XML parser (`quick-xml` via //! manual parsing) to avoid pulling in heavy feed crates. //! -//! Fetches go through the shared `ssrf` guard (scheme/host policy, a DNS -//! resolver that pins connections to globally routable addresses, and -//! per-hop redirect re-checks), and the parsed feed is cached briefly so a +//! Fetches go through `sources::fetch` and its SSRF guard (scheme/host +//! policy, a DNS resolver that pins connections to globally routable +//! addresses, and per-hop redirect re-checks), and the parsed feed is cached briefly so a //! list-then-read sync pass downloads it once rather than once per entry. mod types; @@ -16,11 +16,14 @@ use std::time::{Duration, Instant}; use async_trait::async_trait; -use crate::error::{Error, Result}; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; +use crate::sources::error::{Error, Result}; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; -use super::ssrf::{build_client, is_url_allowed, read_body_capped}; use super::SourceReader; +use crate::documents::html::decode_entities; +use crate::sources::fetch::fetch_url_capped; use types::{FeedCache, FeedEntry}; const DEFAULT_MAX_ITEMS: u32 = 50; @@ -55,21 +58,26 @@ impl RssReader { /// that is N+1 downloads of the same feed per sync (and a rate-limit /// risk against the feed host); the cache turns it into one fetch whose /// results are reused for the read phase. - async fn fetch_entries(&self, url: &str) -> std::result::Result<Vec<FeedEntry>, String> { + async fn fetch_entries(&self, url: &str) -> Result<Vec<FeedEntry>> { // Read the cache in a nested scope so the mutex guard is dropped before // the await below — the guard is not `Send`, and holding it across an // await would make the reader's async methods non-`Send`. { let cache = self.cache.lock().unwrap_or_else(|e| e.into_inner()); - if let Some(cached) = cache.as_ref() { - if cached.url == url && cached.fetched_at.elapsed() < FEED_CACHE_TTL { - return Ok(cached.entries.clone()); - } + if let Some(cached) = cache.as_ref() + && cached.url == url + && cached.fetched_at.elapsed() < FEED_CACHE_TTL + { + return Ok(cached.entries.clone()); } } - let body = fetch_url(url).await?; - let entries = parse_feed_full(&body)?; + // `fetch_url_capped` applies the SSRF guard and streams the body + // against the cap, so a pathological feed cannot exhaust memory. + let document = fetch_url_capped(url, MAX_FEED_BYTES).await?; + let body = String::from_utf8(document.bytes) + .map_err(|e| Error::Reader(format!("feed body is not valid UTF-8: {e}")))?; + let entries = parse_feed_full(&body).map_err(Error::Reader)?; *self.cache.lock().unwrap_or_else(|e| e.into_inner()) = Some(FeedCache { url: url.to_string(), fetched_at: Instant::now(), @@ -94,34 +102,11 @@ impl SourceReader for RssReader { } async fn list_items( - &self, - source: &MemorySourceEntry, - workspace: &std::path::Path, - ) -> Result<Vec<SourceItem>> { - self.list_items_inner(source, workspace) - .await - .map_err(Error::Reader) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - workspace: &std::path::Path, - ) -> Result<SourceContent> { - self.read_item_inner(source, item_id, workspace) - .await - .map_err(Error::Reader) - } -} - -impl RssReader { - async fn list_items_inner( &self, source: &MemorySourceEntry, _workspace: &std::path::Path, - ) -> std::result::Result<Vec<SourceItem>, String> { - let url = source.url.as_deref().ok_or("rss source requires a url")?; + ) -> Result<Vec<SourceItem>> { + let url = configured_url(source)?; let max_items = source.max_items.unwrap_or(DEFAULT_MAX_ITEMS) as usize; tracing::debug!( @@ -145,13 +130,13 @@ impl RssReader { .collect()) } - async fn read_item_inner( + async fn read_item( &self, source: &MemorySourceEntry, item_id: &str, _workspace: &std::path::Path, - ) -> std::result::Result<SourceContent, String> { - let url = source.url.as_deref().ok_or("rss source requires a url")?; + ) -> Result<SourceContent> { + let url = configured_url(source)?; tracing::debug!( host = %url_host(url), @@ -163,7 +148,7 @@ impl RssReader { let entry = entries .into_iter() .find(|e| e.id == item_id) - .ok_or_else(|| format!("item '{item_id}' not found in feed"))?; + .ok_or_else(|| Error::NotFound(format!("item '{item_id}' not found in feed")))?; let content_type = if entry.body.contains('<') { ContentType::Html @@ -184,6 +169,14 @@ impl RssReader { } } +/// The configured feed URL. +fn configured_url(source: &MemorySourceEntry) -> Result<&str> { + source + .url + .as_deref() + .ok_or_else(|| Error::Invalid("rss source requires a url".to_string())) +} + /// Extract just the host portion of a URL for debug-log redaction so we /// don't leak query params, paths, or embedded credentials (userinfo). fn url_host(url: &str) -> String { @@ -211,35 +204,6 @@ fn url_host(url: &str) -> String { }) } -async fn fetch_url(url: &str) -> std::result::Result<String, String> { - // SSRF guard: validate scheme and host, reject private/internal targets, - // and refuse redirects that would escape that policy. - let parsed = reqwest::Url::parse(url).map_err(|e| format!("invalid URL: {e}"))?; - if !is_url_allowed(&parsed) { - return Err(format!( - "rss source requires an http(s) URL to a public host, got: {}", - url.chars().take(64).collect::<String>() - )); - } - - let client = build_client()?; - let resp = client - .get(parsed) - .header("User-Agent", "openhuman") - .send() - .await - .map_err(|e| format!("failed to fetch feed: {e}"))?; - - if !resp.status().is_success() { - return Err(format!("feed returned {}", resp.status())); - } - - // Stream the body with a cap so a pathological feed can't OOM us before - // the size check runs (`Content-Length` can be omitted or understated). - let bytes = read_body_capped(resp, MAX_FEED_BYTES).await?; - String::from_utf8(bytes).map_err(|e| format!("feed body is not valid UTF-8: {e}")) -} - fn parse_feed_full(xml: &str) -> std::result::Result<Vec<FeedEntry>, String> { // Detect RSS vs Atom by looking for <rss or <feed if xml.contains("<rss") || xml.contains("<channel") { @@ -366,7 +330,7 @@ fn extract_tag(xml: &str, tag: &str) -> Option<String> { if trimmed.starts_with("<![CDATA[") { Some(unwrapped.to_string()) } else { - Some(decode_xml_entities(unwrapped)) + Some(decode_entities(unwrapped)) } } @@ -386,16 +350,6 @@ fn extract_attr(xml: &str, tag: &str, attr: &str) -> Option<String> { Some(tag_str[attr_start..attr_end].to_string()) } -fn decode_xml_entities(s: &str) -> String { - // `&` is decoded last so escaped entity text (`&lt;` → `<`) - // survives as literal text instead of being decoded a second time. - s.replace("<", "<") - .replace(">", ">") - .replace(""", "\"") - .replace("'", "'") - .replace("&", "&") -} - #[cfg(test)] -#[path = "rss_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/rss_tests.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs similarity index 89% rename from crates/tinymemory-sources/src/readers/rss_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs index 3520adb2..cf6fc2ca 100644 --- a/crates/tinymemory-sources/src/readers/rss_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs @@ -1,6 +1,9 @@ +//! Tests for the RSS/Atom reader: the cached list-then-read flow, URL +//! refusal, and the feed parser. + use super::*; -use crate::readers::SourceReader; +use crate::sources::readers::SourceReader; fn cached_reader(url: &str) -> RssReader { RssReader { @@ -93,24 +96,30 @@ async fn cached_feed_drives_list_and_read_without_network() { async fn rss_reader_reports_missing_configuration_and_items() { let reader = RssReader::new(); let missing_url = rss_source(None, None); - assert!(reader - .list_items(&missing_url, std::path::Path::new(".")) - .await - .is_err()); - assert!(reader - .read_item(&missing_url, "anything", std::path::Path::new(".")) - .await - .is_err()); + assert!( + reader + .list_items(&missing_url, std::path::Path::new(".")) + .await + .is_err() + ); + assert!( + reader + .read_item(&missing_url, "anything", std::path::Path::new(".")) + .await + .is_err() + ); let url = "https://example.com/feed.xml"; - assert!(cached_reader(url) - .read_item( - &rss_source(Some(url), None), - "missing", - std::path::Path::new("."), - ) - .await - .is_err()); + assert!( + cached_reader(url) + .read_item( + &rss_source(Some(url), None), + "missing", + std::path::Path::new("."), + ) + .await + .is_err() + ); } #[tokio::test] @@ -351,17 +360,27 @@ fn url_host_fallback_strips_userinfo_without_scheme() { // ── Entity decoding ───────────────────────────────────────────────── #[test] -fn decode_xml_entities_decodes_amp_last() { +fn feed_text_entities_decode_exactly_once() { // `&lt;` is the escaped form of `<`; it must decode once to `<`, // not twice to `<`. - assert_eq!(decode_xml_entities("&lt;"), "<"); - assert_eq!(decode_xml_entities("&amp;"), "&"); + assert_eq!( + extract_tag("<title>&lt;", "title").as_deref(), + Some("<") + ); + assert_eq!( + extract_tag("&amp;", "title").as_deref(), + Some("&") + ); } #[test] -fn decode_xml_entities_handles_all_named() { +fn feed_text_decodes_every_predefined_xml_entity_and_numeric_references() { assert_eq!( - decode_xml_entities("<b> "q" 'a' & more"), - " \"q\" 'a' & more" + extract_tag( + "<b> "q" 'a' & more ’", + "title" + ) + .as_deref(), + Some(" \"q\" 'a' & more \u{2019}") ); } diff --git a/crates/tinymemory-sources/src/readers/rss/types.rs b/crates/tinymemory-integrations/src/sources/readers/rss/types.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/rss/types.rs rename to crates/tinymemory-integrations/src/sources/readers/rss/types.rs diff --git a/crates/tinymemory-sources/src/readers/web_page.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs similarity index 78% rename from crates/tinymemory-sources/src/readers/web_page.rs rename to crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs index 177b64b6..f7841dac 100644 --- a/crates/tinymemory-sources/src/readers/web_page.rs +++ b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs @@ -3,25 +3,31 @@ //! Fetches a single URL and extracts its content. When a CSS `selector` is //! configured, only the text of matching elements is included (plain text); //! otherwise the whole page is converted to markdown through -//! `tinymemory_documents::html::to_markdown`, keeping its headings, lists and +//! `tinymemory_integrations::documents::html::to_markdown`, keeping its headings, lists and //! links. //! -//! The fetch-side SSRF guard (scheme/host policy plus a DNS resolver that -//! pins connections to globally routable addresses) lives in the shared -//! `ssrf` module, which the RSS reader uses too. +//! The page is fetched through `sources::fetch`, behind its SSRF guard +//! (scheme/host policy plus a DNS resolver that pins connections to globally +//! routable addresses), with a 10 MiB body cap. mod types; use async_trait::async_trait; -use super::ssrf::{build_client, is_url_allowed, read_body_capped}; use types::SelectorSpec; -use crate::error::{Error, Result}; -use crate::types::{ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind}; +use crate::sources::fetch::fetch_url_capped; + +use crate::sources::error::{Error, Result}; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; use super::SourceReader; +/// Largest page body the reader will buffer. +const MAX_BODY_BYTES: u64 = 10 * 1024 * 1024; + /// Reader for a single-page web source: fetches one URL and extracts its /// readable text. #[derive(Debug, Clone, Copy, Default)] @@ -34,38 +40,11 @@ impl SourceReader for WebPageReader { } async fn list_items( - &self, - source: &MemorySourceEntry, - workspace: &std::path::Path, - ) -> Result> { - self.list_items_inner(source, workspace) - .await - .map_err(Error::Reader) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - workspace: &std::path::Path, - ) -> Result { - self.read_item_inner(source, item_id, workspace) - .await - .map_err(Error::Reader) - } -} - -impl WebPageReader { - async fn list_items_inner( &self, source: &MemorySourceEntry, _workspace: &std::path::Path, - ) -> std::result::Result, String> { - let url = source - .url - .as_deref() - .ok_or("web_page source requires a url")?; - + ) -> Result> { + let url = configured_url(source)?; Ok(vec![SourceItem { id: url.to_string(), title: source.label.clone(), @@ -73,60 +52,34 @@ impl WebPageReader { }]) } - async fn read_item_inner( + async fn read_item( &self, source: &MemorySourceEntry, item_id: &str, _workspace: &std::path::Path, - ) -> std::result::Result { + ) -> Result { let url = if item_id.starts_with("http") { item_id.to_string() } else { - source.url.clone().ok_or("web_page source requires a url")? + configured_url(source)?.to_string() }; - // SSRF guard: validate scheme and host, reject private/internal - // targets, and refuse redirects that would escape that policy. - let parsed = reqwest::Url::parse(&url).map_err(|e| format!("invalid URL: {e}"))?; - if !is_url_allowed(&parsed) { - return Err(format!( - "web_page source requires an http(s) URL to a public host, got: {}", - url.chars().take(64).collect::() - )); - } - tracing::debug!( - host = %parsed.host_str().unwrap_or(""), selector = ?source.selector, "[memory_sources:web_page] reading item" ); - let client = build_client()?; - let resp = client - .get(parsed) - .header("User-Agent", "openhuman") - .send() - .await - .map_err(|e| format!("failed to fetch page: {e}"))?; - - if !resp.status().is_success() { - return Err(format!("page returned {}", resp.status())); - } + // `fetch_url_capped` applies the SSRF guard (scheme and host policy, + // public-only DNS, per-hop redirect checks) and streams the body + // against the cap, so a hostile or giant page cannot exhaust memory. + let document = fetch_url_capped(&url, MAX_BODY_BYTES).await?; + let body = String::from_utf8_lossy(&document.bytes).into_owned(); - // Cap response body to 10 MiB so a hostile/giant page can't OOM us. - // The read is streamed so the cap is enforced while downloading, not - // after the whole body has been buffered into memory. - const MAX_BODY_BYTES: u64 = 10 * 1024 * 1024; - let bytes = read_body_capped(resp, MAX_BODY_BYTES).await?; - let body = String::from_utf8_lossy(&bytes).into_owned(); - - let title = tinymemory_documents::html::extract_title(&body) - .or_else(|| extract_title(&body)) - .unwrap_or_else(|| url.clone()); + let title = crate::documents::html::extract_title(&body).unwrap_or_else(|| url.clone()); let (extracted, content_type) = match source.selector.as_deref() { Some(selector) => (extract_by_selector(&body, selector), ContentType::Plaintext), None => ( - tinymemory_documents::html::to_markdown(&body), + crate::documents::html::to_markdown(&body), ContentType::Markdown, ), }; @@ -141,15 +94,16 @@ impl WebPageReader { } } -// ── Text extraction ───────────────────────────────────────────────── - -fn extract_title(html: &str) -> Option { - let start = html.find("')? + start + 1; - let end = html[content_start..].find("")? + content_start; - Some(html[content_start..end].trim().to_string()) +/// The configured page URL. +fn configured_url(source: &MemorySourceEntry) -> Result<&str> { + source + .url + .as_deref() + .ok_or_else(|| Error::Invalid("web_page source requires a url".to_string())) } +// ── Text extraction ───────────────────────────────────────────────── + fn parse_selector(selector: &str) -> Option { let last = selector .trim() @@ -291,11 +245,11 @@ fn find_next_element( continue; } let tag = &after[..tag_len]; - if let Some(expected) = &spec.tag { - if !tag.eq_ignore_ascii_case(expected) { - offset = abs + 1; - continue; - } + if let Some(expected) = &spec.tag + && !tag.eq_ignore_ascii_case(expected) + { + offset = abs + 1; + continue; } let gt = lower_html[abs..] @@ -304,11 +258,11 @@ fn find_next_element( .unwrap_or(lower_html.len()); let open_tag = &lower_html[abs..gt]; let orig_open_tag = &orig_html[abs..gt]; - if let Some(expected_id) = &spec.id { - if attr_value(open_tag, orig_open_tag, "id").as_deref() != Some(expected_id.as_str()) { - offset = abs + 1; - continue; - } + if let Some(expected_id) = &spec.id + && attr_value(open_tag, orig_open_tag, "id").as_deref() != Some(expected_id.as_str()) + { + offset = abs + 1; + continue; } if !spec.classes.is_empty() { let class_attr = attr_value(open_tag, orig_open_tag, "class").unwrap_or_default(); @@ -368,10 +322,10 @@ fn attr_value(open_tag: &str, orig_open_tag: &str, name: &str) -> Option if let Some(end_rel) = v.find('"') { return Some(orig_open_tag[value_abs + 1..value_abs + 1 + end_rel].to_string()); } - } else if let Some(v) = eq_trimmed.strip_prefix('\'') { - if let Some(end_rel) = v.find('\'') { - return Some(orig_open_tag[value_abs + 1..value_abs + 1 + end_rel].to_string()); - } + } else if let Some(v) = eq_trimmed.strip_prefix('\'') + && let Some(end_rel) = v.find('\'') + { + return Some(orig_open_tag[value_abs + 1..value_abs + 1 + end_rel].to_string()); } } rest = trimmed; @@ -486,5 +440,5 @@ fn strip_html_tags(html: &str) -> String { } #[cfg(test)] -#[path = "web_page_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/readers/web_page_tests.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs similarity index 96% rename from crates/tinymemory-sources/src/readers/web_page_tests.rs rename to crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs index b2bbfc1e..39c0baba 100644 --- a/crates/tinymemory-sources/src/readers/web_page_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs @@ -1,3 +1,6 @@ +//! Tests for the web-page reader: listing, URL refusal, and the simple CSS +//! selector extraction. + use super::*; fn web_source(url: Option<&str>, selector: Option<&str>) -> MemorySourceEntry { @@ -38,20 +41,24 @@ async fn reader_lists_one_configured_page_and_rejects_missing_or_private_reads() let missing = web_source(None, None); assert!(reader.list_items(&missing, workspace.path()).await.is_err()); - assert!(reader - .read_item(&missing, "not-an-http-id", workspace.path()) - .await - .is_err()); + assert!( + reader + .read_item(&missing, "not-an-http-id", workspace.path()) + .await + .is_err() + ); for item in [ "http://[", "http://127.0.0.1/private", "http://service.internal/private", ] { - assert!(reader - .read_item(&source, item, workspace.path()) - .await - .is_err()); + assert!( + reader + .read_item(&source, item, workspace.path()) + .await + .is_err() + ); } } @@ -61,12 +68,6 @@ fn strip_html_tags_removes_tags() { assert_eq!(strip_html_tags(html), "Hello world"); } -#[test] -fn extract_title_finds_title_tag() { - let html = "My Page"; - assert_eq!(extract_title(html).as_deref(), Some("My Page")); -} - #[test] fn extract_by_selector_finds_tag_content() { let html = "

Important content

skip
"; diff --git a/crates/tinymemory-sources/src/readers/web_page/types.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs similarity index 100% rename from crates/tinymemory-sources/src/readers/web_page/types.rs rename to crates/tinymemory-integrations/src/sources/readers/web_page/types.rs diff --git a/crates/tinymemory-sources/src/types.rs b/crates/tinymemory-integrations/src/sources/types/mod.rs similarity index 52% rename from crates/tinymemory-sources/src/types.rs rename to crates/tinymemory-integrations/src/sources/types/mod.rs index ad60cc59..156f2589 100644 --- a/crates/tinymemory-sources/src/types.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod.rs @@ -1,15 +1,14 @@ //! Core types for memory sources. //! //! A *memory source* answers the question "what feeds my memory?". Each -//! configured source is a [`MemorySourceEntry`] persisted in `config.toml` -//! under `[[memory_sources]]`. The [`SourceKind`] discriminator selects which -//! kind-specific fields are required; required-field checks live in -//! [`crate::validation`] and are surfaced via -//! [`MemorySourceEntry::validate`]. +//! configured source is a [`MemorySourceEntry`], which the host persists +//! wherever it keeps configuration. The [`SourceKind`] discriminator selects +//! which kind-specific fields are required; [`MemorySourceEntry::validate`] +//! checks them. //! //! Reader output contracts ([`SourceItem`], [`SourceContent`], [`ContentType`]) //! are shared across every reader implementation so the host can ingest source -//! payloads uniformly regardless of where they came from; [`crate::items`] +//! payloads uniformly regardless of where they came from; [`crate::sources::items`] //! turns them into `StoreItem`s. //! //! Wire strings are snake_case and are part of the persisted contract — do not @@ -18,7 +17,7 @@ use schemars::JsonSchema; use serde::{Deserialize, Serialize}; -use crate::error::{Error, Result}; +use crate::sources::error::{Error, Result}; pub(crate) fn default_true() -> bool { true @@ -27,13 +26,13 @@ pub(crate) fn default_true() -> bool { /// The kind of a configured memory source. /// /// The wire representation is snake_case (`github_repo`, `rss_feed`, …) and is -/// persisted in `config.toml`; it must stay stable across versions. Each maps +/// persisted by hosts; it must stay stable across versions. Each maps /// onto one [`tinymemory_api::SourceKind`] through [`SourceKind::api_kind`]. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "snake_case")] pub enum SourceKind { /// A Composio OAuth connector (Gmail, Slack, Notion, …). Network-backed; - /// the live fetch is owned by the host, not this crate. + /// the live fetch is owned by the host, not this module. Composio, /// Local agent conversation transcripts stored in the workspace. Conversation, @@ -93,12 +92,11 @@ impl SourceKind { } } -/// A configured memory source entry persisted in `config.toml`. +/// A configured memory source entry. /// /// All kind-specific fields are flattened onto the struct as `Option`s. The /// [`kind`](MemorySourceEntry::kind) discriminator determines which fields are -/// required; validation is enforced at add/update time via -/// [`MemorySourceEntry::validate`]. +/// required; [`MemorySourceEntry::validate`] checks them. #[derive(Debug, Clone, Serialize, Deserialize, JsonSchema)] pub struct MemorySourceEntry { /// Stable unique id (e.g. `src_`). @@ -200,201 +198,49 @@ impl MemorySourceEntry { } } - /// Validate required fields for this entry's [`SourceKind`]. + /// Validate the fields this entry's [`SourceKind`] requires. /// - /// Delegates to [`crate::validation::validate_entry`]. + /// `id` and `label` are required for every kind, and `id` must not contain + /// `:` or control characters. Composio needs `toolkit` and + /// `connection_id`; folders and files need `path`; GitHub repositories, RSS + /// feeds and web pages need `url`. An empty string counts as missing. /// /// # Errors /// /// [`Error::Invalid`] naming the first failing rule. pub fn validate(&self) -> Result<()> { - crate::validation::validate_entry(self) - } -} - -fn deserialize_double_option<'de, D, T>( - deserializer: D, -) -> std::result::Result>, D::Error> -where - D: serde::Deserializer<'de>, - T: serde::Deserialize<'de>, -{ - as serde::Deserialize>::deserialize(deserializer).map(Some) -} - -/// Partial update payload for a source entry. -/// -/// An absent field leaves the current value unchanged. For optional source -/// properties, an explicit JSON `null` clears the value while a concrete value -/// replaces it. -#[derive(Debug, Default, Deserialize)] -pub struct MemorySourcePatch { - /// New human-readable label for the source. - #[serde(default)] - pub label: Option, - /// Toggle whether the source participates in sync. - #[serde(default)] - pub enabled: Option, - /// Composio toolkit slug (e.g. `gmail`, `slack`). - #[serde(default, deserialize_with = "deserialize_double_option")] - pub toolkit: Option>, - /// Composio connection id this source binds to. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub connection_id: Option>, - /// Filesystem root for a local-files source. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub path: Option>, - /// Glob filter applied under [`MemorySourcePatch::path`]. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub glob: Option>, - /// Remote URL for a git/web source. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub url: Option>, - /// Git branch to track. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub branch: Option>, - /// Explicit path allowlist within a repo source. - #[serde(default)] - pub paths: Option>, - /// Cap on the number of items pulled per sync. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_items: Option>, - /// Source-specific selector (e.g. a CSS selector for a web page). - #[serde(default, deserialize_with = "deserialize_double_option")] - pub selector: Option>, - /// Token budget per sync run. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_tokens_per_sync: Option>, - /// Cost budget per sync run, in USD. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_cost_per_sync_usd: Option>, - /// History depth in days for tree/summary backfill. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub sync_depth_days: Option>, - /// Cap on commits ingested from a git source. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_commits: Option>, - /// Cap on issues ingested from a repo source. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_issues: Option>, - /// Cap on pull requests ingested from a repo source. - #[serde(default, deserialize_with = "deserialize_double_option")] - pub max_prs: Option>, -} - -impl MemorySourcePatch { - /// Reject fields that do not apply to `kind`. - /// - /// A patch is a partial update, so a caller can set a field the source's - /// kind has no use for — a git branch on an RSS feed. Catching that here - /// keeps a nonsensical value out of the registry rather than letting the - /// reader discover it later. - /// - /// # Errors - /// - /// [`Error::Invalid`] naming the first inapplicable field. - pub fn validate_for_kind(&self, kind: SourceKind) -> Result<()> { - let reject = |field: &str| { - Err(Error::Invalid(format!( - "field '{field}' is not applicable to source kind '{}'", - kind.as_str() - ))) - }; - if (self.toolkit.is_some() || self.connection_id.is_some()) && kind != SourceKind::Composio - { - return reject("toolkit/connection_id"); - } - if self.path.is_some() && !matches!(kind, SourceKind::Folder | SourceKind::File) { - return reject("path"); - } - if self.glob.is_some() && kind != SourceKind::Folder { - return reject("glob"); - } - if (self.branch.is_some() - || self.paths.is_some() - || self.max_commits.is_some() - || self.max_issues.is_some() - || self.max_prs.is_some()) - && kind != SourceKind::GithubRepo - { - return reject("github repository fields"); + if self.id.trim().is_empty() { + return Err(Error::Invalid("id is required".to_string())); } - if self.selector.is_some() && kind != SourceKind::WebPage { - return reject("selector"); + if self.id.contains(':') || self.id.chars().any(char::is_control) { + return Err(Error::Invalid( + "id must not contain ':' or control characters".to_string(), + )); } - // `max_items` is the per-run ingest cap. It applies to RSS feeds and to - // Composio connections — the host UI (`SourceSettingsPanel`) exposes it - // for both, and a Composio source is created with a toolkit default, so - // rejecting it on edit desynced the UI from the store. Other kinds have - // no per-run item cap. - if matches!(self.max_items, Some(Some(_))) - && !matches!(kind, SourceKind::RssFeed | SourceKind::Composio) - { - return reject("max_items"); + if self.label.is_empty() { + return Err(Error::Invalid("label is required".to_string())); } - if self.url.is_some() - && kind != SourceKind::GithubRepo - && kind != SourceKind::RssFeed - && kind != SourceKind::WebPage - { - return reject("url"); + match self.kind { + SourceKind::Composio => { + require_field(&self.toolkit, "toolkit")?; + require_field(&self.connection_id, "connection_id") + } + SourceKind::Conversation => Ok(()), + SourceKind::Folder | SourceKind::File => require_field(&self.path, "path"), + SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => { + require_field(&self.url, "url") + } } - Ok(()) } +} - /// Apply each present field of this patch onto `entry` in place. - pub fn apply_to(self, entry: &mut MemorySourceEntry) { - if let Some(value) = self.label { - entry.label = value; - } - if let Some(value) = self.enabled { - entry.enabled = value; - } - if let Some(value) = self.toolkit { - entry.toolkit = value; - } - if let Some(value) = self.connection_id { - entry.connection_id = value; - } - if let Some(value) = self.path { - entry.path = value; - } - if let Some(value) = self.glob { - entry.glob = value; - } - if let Some(value) = self.url { - entry.url = value; - } - if let Some(value) = self.branch { - entry.branch = value; - } - if let Some(value) = self.paths { - entry.paths = value; - } - if let Some(value) = self.max_items { - entry.max_items = value; - } - if let Some(value) = self.selector { - entry.selector = value; - } - if let Some(value) = self.max_tokens_per_sync { - entry.max_tokens_per_sync = value; - } - if let Some(value) = self.max_cost_per_sync_usd { - entry.max_cost_per_sync_usd = value; - } - if let Some(value) = self.sync_depth_days { - entry.sync_depth_days = value; - } - if let Some(value) = self.max_commits { - entry.max_commits = value; - } - if let Some(value) = self.max_issues { - entry.max_issues = value; - } - if let Some(value) = self.max_prs { - entry.max_prs = value; - } +/// Require that `value` is present and non-empty, naming it `name` in errors. +fn require_field(value: &Option, name: &str) -> Result<()> { + match value { + Some(v) if !v.is_empty() => Ok(()), + _ => Err(Error::Invalid(format!( + "{name} is required for this source kind" + ))), } } @@ -442,5 +288,5 @@ pub struct SourceContent { } #[cfg(test)] -#[path = "types_tests.rs"] +#[path = "mod_tests.rs"] mod tests; diff --git a/crates/tinymemory-sources/src/types_tests.rs b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs similarity index 88% rename from crates/tinymemory-sources/src/types_tests.rs rename to crates/tinymemory-integrations/src/sources/types/mod_tests.rs index 3b11596f..2ea2171b 100644 --- a/crates/tinymemory-sources/src/types_tests.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs @@ -111,23 +111,6 @@ fn every_config_kind_maps_onto_a_contract_source_kind() { ); } -#[test] -fn path_applies_to_folders_and_files_but_glob_only_to_folders() { - let path = MemorySourcePatch { - path: Some(Some("a".into())), - ..Default::default() - }; - assert!(path.validate_for_kind(SourceKind::Folder).is_ok()); - assert!(path.validate_for_kind(SourceKind::File).is_ok()); - assert!(path.validate_for_kind(SourceKind::RssFeed).is_err()); - let glob = MemorySourcePatch { - glob: Some(Some("*.md".into())), - ..Default::default() - }; - assert!(glob.validate_for_kind(SourceKind::Folder).is_ok()); - assert!(glob.validate_for_kind(SourceKind::File).is_err()); -} - #[test] fn validate_rss_and_web_page_require_url() { let rss = MemorySourceEntry { @@ -280,24 +263,7 @@ pub(super) fn default_entry() -> MemorySourceEntry { } } -#[test] -fn max_items_is_applicable_to_composio_and_rss_but_not_other_kinds() { - // The host UI exposes `max_items` for Composio sources and creates them with - // a toolkit default, so editing one must not be rejected — the regression - // this guards ("field 'max_items' is not applicable to source kind - // 'composio'"). RSS keeps it; kinds with no per-run item cap still reject. - let patch = || MemorySourcePatch { - max_items: Some(Some(100)), - ..Default::default() - }; - assert!(patch().validate_for_kind(SourceKind::Composio).is_ok()); - assert!(patch().validate_for_kind(SourceKind::RssFeed).is_ok()); - assert!(patch().validate_for_kind(SourceKind::Folder).is_err()); - assert!(patch().validate_for_kind(SourceKind::GithubRepo).is_err()); - assert!(patch().validate_for_kind(SourceKind::WebPage).is_err()); -} - -/// Hosts persist these types in their `config.toml` and exchange them over +/// Hosts persist these types in their configuration and exchange them over /// RPC as JSON, so a renamed field or a new `SourceKind` variant is not a /// compile error anywhere: it is a runtime failure the first time a host reads /// a config written by another version. @@ -426,3 +392,27 @@ fn source_content_wire_format_is_pinned() { }) ); } + +#[test] +fn validate_treats_an_empty_string_field_as_missing() { + let entry = MemorySourceEntry { + id: "src_folder".into(), + label: "Folder".into(), + path: Some(String::new()), + ..default_entry() + }; + assert!(entry.validate().is_err()); +} + +#[test] +fn validate_rejects_an_id_with_a_colon_or_control_character() { + for id in ["src:x", "src\nx"] { + let entry = MemorySourceEntry { + id: id.into(), + label: "Conversation".into(), + kind: SourceKind::Conversation, + ..default_entry() + }; + assert!(entry.validate().is_err(), "{id:?} must be rejected"); + } +} diff --git a/crates/tinymemory/tests/documents_office.rs b/crates/tinymemory-integrations/tests/documents_office.rs similarity index 70% rename from crates/tinymemory/tests/documents_office.rs rename to crates/tinymemory-integrations/tests/documents_office.rs index 88e95901..34eb2b22 100644 --- a/crates/tinymemory/tests/documents_office.rs +++ b/crates/tinymemory-integrations/tests/documents_office.rs @@ -1,8 +1,10 @@ -//! The `documents-office` feature reaches `OfficeConverter` through the -//! facade, and it composes with the default converter chain. +//! The `documents-office` feature reaches `OfficeConverter` through +//! `documents`, and it composes with the default converter chain. #![cfg(feature = "documents-office")] -use tinymemory::documents::{ConverterChain, DocumentConverter, DocumentFormat, OfficeConverter}; +use tinymemory_integrations::documents::{ + ConverterChain, DocumentConverter, DocumentFormat, OfficeConverter, +}; #[test] fn documents_office_feature_exposes_the_office_converter() { diff --git a/crates/tinymemory/tests/feature_surface.rs b/crates/tinymemory-integrations/tests/feature_surface.rs similarity index 50% rename from crates/tinymemory/tests/feature_surface.rs rename to crates/tinymemory-integrations/tests/feature_surface.rs index f75c7a8f..fa1e358c 100644 --- a/crates/tinymemory/tests/feature_surface.rs +++ b/crates/tinymemory-integrations/tests/feature_surface.rs @@ -1,14 +1,14 @@ -//! With every feature on, each optional crate is reachable through the facade, +//! With every feature on, each integration is reachable through its module, //! and the pieces compose: scrub an item, store it in the reference engine, //! run the conformance suite, and compile a context from what is left. #![cfg(feature = "full")] -use tinymemory::{LearningKind, MemoryEngine, MemoryMeta, StoreItem}; +use tinymemory_api::{LearningKind, MemoryEngine, MemoryMeta, StoreItem}; #[tokio::test] -async fn the_optional_crates_compose_through_the_facade() { - let engine = tinymemory::conformance::ReferenceEngine::new(); - tinymemory::conformance::run(&engine) +async fn the_integration_modules_compose_into_one_write_path() { + let engine = tinymemory_api::conformance::ReferenceEngine::new(); + tinymemory_api::conformance::run(&engine) .await .expect("the reference engine conforms"); @@ -18,13 +18,16 @@ async fn the_optional_crates_compose_through_the_facade() { 0.8, MemoryMeta::default(), ); - let scrubbed = tinymemory::safety::scrub_item(item); + let scrubbed = tinymemory_integrations::safety::scrub_item(item); assert!(scrubbed.report.changed()); engine.store(scrubbed.value).await.expect("store"); - let doc = tinymemory::context::compile(&engine, &tinymemory::context::ContextSpec::default()) - .await - .expect("compile"); + let doc = tinymemory_tools::context::compile( + &engine, + &tinymemory_tools::context::ContextSpec::default(), + ) + .await + .expect("compile"); assert!(doc.markdown.contains("## Learnings")); assert!(!doc.markdown.contains("sk-proj-")); assert_eq!(doc.engine, "reference"); @@ -33,9 +36,9 @@ async fn the_optional_crates_compose_through_the_facade() { #[test] fn the_reader_and_converter_crates_are_reachable() { assert_eq!( - tinymemory::documents::language_for_path("src/main.rs"), + tinymemory_integrations::documents::language_for_path("src/main.rs"), Some("rust") ); - let _ = std::any::type_name::(); - let _ = std::any::type_name::(); + let _ = std::any::type_name::(); + let _ = std::any::type_name::(); } diff --git a/crates/tinymemory-import/tests/legacy_import.rs b/crates/tinymemory-integrations/tests/legacy_import.rs similarity index 72% rename from crates/tinymemory-import/tests/legacy_import.rs rename to crates/tinymemory-integrations/tests/legacy_import.rs index f1a6504a..3cdc3e16 100644 --- a/crates/tinymemory-import/tests/legacy_import.rs +++ b/crates/tinymemory-integrations/tests/legacy_import.rs @@ -12,7 +12,9 @@ use support::{OLD_MEMORY_DDL, chunk, chunk_store, doc, facet, turn, workspace}; use tinymemory_api::{ DocumentBody, LearningKind, Role, SourceKind, StoreItem, ToolCallRef, TurnRange, }; -use tinymemory_import::{Checkpoint, ChunkCursor, Error, ImportedItem, LegacyWorkspace}; +use tinymemory_integrations::import::{ + Checkpoint, ChunkCursor, Error, ImportedItem, LegacyWorkspace, +}; const T0: f64 = 1_700_000_000.0; @@ -741,3 +743,223 @@ fn a_row_sqlite_cannot_decode_is_a_sqlite_error() { assert!(matches!(items.next(), Some(Err(Error::Sqlite(_))))); assert!(items.next().is_none()); } + +// --- migrate: a v1 workspace into an engine, in resumable batches --- + +mod migration { + use std::sync::atomic::{AtomicUsize, Ordering}; + + use tinymemory_api::conformance::ReferenceEngine; + use tinymemory_api::{ + EngineDescriptor, EngineHealth, FetchPage, FetchRequest, ForgetReport, ForgetTarget, + ListPage, ListRequest, MAX_STORE_MANY, MemoryEngine, RecallAnswer, RecallRequest, + StoreItem, StoreReceipt, async_trait, + }; + use tinymemory_integrations::import::{ + Checkpoint, Error, LegacyWorkspace, MigrationReport, migrate, migrate_with, + }; + + use super::T0; + use super::support::{doc, workspace}; + + /// More documents than two full `store_many` batches hold. + const COUNT: usize = 2 * MAX_STORE_MANY + 50; + + /// A workspace of `COUNT` documents, `d000` to `d249` in key order. + fn documents() -> (tempfile::TempDir, LegacyWorkspace) { + let (dir, conn) = workspace(super::support::MEMORY_DDL); + for index in 0..COUNT { + doc( + &conn, + &format!("d{index:03}"), + "document_notes", + None, + &format!("Note {index}"), + &format!("Body of note {index}."), + "[]", + "{}", + T0 + index as f64, + ); + } + drop(conn); + let legacy = LegacyWorkspace::open(dir.path()).unwrap(); + (dir, legacy) + } + + fn key(checkpoint: &Checkpoint) -> Option<&str> { + checkpoint.documents.as_deref() + } + + /// Delegates to a [`ReferenceEngine`], failing the `fail_on`th + /// `store_many` call (1-based) without storing anything. + struct FailingOn { + inner: ReferenceEngine, + fail_on: usize, + calls: AtomicUsize, + } + + #[async_trait] + impl MemoryEngine for FailingOn { + fn descriptor(&self) -> &EngineDescriptor { + self.inner.descriptor() + } + async fn health(&self) -> EngineHealth { + self.inner.health().await + } + async fn recall(&self, req: RecallRequest) -> tinymemory_api::Result { + self.inner.recall(req).await + } + async fn fetch(&self, req: FetchRequest) -> tinymemory_api::Result { + self.inner.fetch(req).await + } + async fn store(&self, item: StoreItem) -> tinymemory_api::Result { + self.inner.store(item).await + } + async fn store_many( + &self, + items: Vec, + ) -> tinymemory_api::Result> { + if self.calls.fetch_add(1, Ordering::SeqCst) + 1 == self.fail_on { + return Err(tinymemory_api::Error::Unavailable("engine down".into())); + } + self.inner.store_many(items).await + } + async fn forget(&self, target: ForgetTarget) -> tinymemory_api::Result { + self.inner.forget(target).await + } + async fn list(&self, req: ListRequest) -> tinymemory_api::Result { + self.inner.list(req).await + } + } + + #[tokio::test] + async fn a_full_migration_stores_every_item_in_bounded_batches() { + let (_dir, legacy) = documents(); + let engine = ReferenceEngine::new(); + let report = migrate(&engine, legacy, None).await.unwrap(); + assert_eq!( + report, + MigrationReport { + stored: COUNT, + replayed: 0, + batches: 3, + checkpoint: Checkpoint { + documents: Some("d249".into()), + ..Checkpoint::default() + }, + } + ); + assert_eq!(engine.len(), COUNT); + } + + #[tokio::test] + async fn a_second_run_is_all_replays() { + let (dir, legacy) = documents(); + let engine = ReferenceEngine::new(); + migrate(&engine, legacy, None).await.unwrap(); + let again = migrate(&engine, LegacyWorkspace::open(dir.path()).unwrap(), None) + .await + .unwrap(); + assert_eq!((again.stored, again.replayed), (0, COUNT)); + assert_eq!(engine.len(), COUNT, "a re-run replays, it never duplicates"); + } + + #[tokio::test] + async fn resuming_from_a_checkpoint_stores_only_the_rest() { + let (_dir, legacy) = documents(); + let mid = legacy.items().nth(149).unwrap().unwrap().checkpoint; + assert_eq!(key(&mid), Some("d149")); + let engine = ReferenceEngine::new(); + let report = migrate(&engine, legacy, Some(mid)).await.unwrap(); + assert_eq!((report.stored, report.batches), (COUNT - 150, 1)); + assert_eq!(engine.len(), COUNT - 150); + assert_eq!(key(&report.checkpoint), Some("d249")); + } + + #[tokio::test] + async fn the_callback_sees_each_committed_checkpoint_in_order() { + let (_dir, legacy) = documents(); + let engine = ReferenceEngine::new(); + let mut seen = Vec::new(); + let report = migrate_with(&engine, legacy, None, |checkpoint: &Checkpoint| { + seen.push(checkpoint.clone()); + }) + .await + .unwrap(); + let keys: Vec> = seen.iter().map(key).collect(); + assert_eq!(keys, [Some("d099"), Some("d199"), Some("d249")]); + assert_eq!(seen.last(), Some(&report.checkpoint)); + } + + #[tokio::test] + async fn an_engine_failure_carries_the_last_committed_checkpoint() { + let (dir, legacy) = documents(); + let engine = FailingOn { + inner: ReferenceEngine::new(), + fail_on: 2, + calls: AtomicUsize::new(0), + }; + let error = migrate(&engine, legacy, None).await.unwrap_err(); + let checkpoint = error.checkpoint().cloned().unwrap(); + assert_eq!(key(&checkpoint), Some("d099")); + match &error { + Error::Engine { source, .. } => { + assert!(matches!(source, tinymemory_api::Error::Unavailable(_))); + } + other => panic!("expected an engine error, got {other}"), + } + assert!(error.to_string().contains("engine down"), "{error}"); + assert_eq!(engine.inner.len(), MAX_STORE_MANY); + + let resumed = migrate( + &engine, + LegacyWorkspace::open(dir.path()).unwrap(), + Some(checkpoint), + ) + .await + .unwrap(); + assert_eq!( + (resumed.stored, resumed.replayed), + (COUNT - MAX_STORE_MANY, 0) + ); + assert_eq!(engine.inner.len(), COUNT); + } + + #[tokio::test] + async fn an_engine_failure_on_the_first_batch_carries_the_starting_checkpoint() { + let (_dir, legacy) = documents(); + let engine = FailingOn { + inner: ReferenceEngine::new(), + fail_on: 1, + calls: AtomicUsize::new(0), + }; + let start = Checkpoint { + documents: Some("d009".into()), + ..Checkpoint::default() + }; + let error = migrate(&engine, legacy, Some(start.clone())) + .await + .unwrap_err(); + assert_eq!(error.checkpoint(), Some(&start)); + } + + #[tokio::test] + async fn an_empty_workspace_migrates_nothing() { + let (dir, conn) = workspace(super::support::MEMORY_DDL); + drop(conn); + let engine = ReferenceEngine::new(); + let report = migrate(&engine, LegacyWorkspace::open(dir.path()).unwrap(), None) + .await + .unwrap(); + assert_eq!(report, MigrationReport::default()); + assert_eq!(engine.len(), 0); + } + + #[test] + fn a_migration_can_run_on_a_spawned_task() { + fn assert_send(_: &T) {} + let (_dir, legacy) = documents(); + let engine = ReferenceEngine::new(); + assert_send(&migrate(&engine, legacy, None)); + } +} diff --git a/crates/tinymemory-cortex/tests/live_cortexdb.rs b/crates/tinymemory-integrations/tests/live_cortexdb.rs similarity index 97% rename from crates/tinymemory-cortex/tests/live_cortexdb.rs rename to crates/tinymemory-integrations/tests/live_cortexdb.rs index 183eb9a8..da4ac013 100644 --- a/crates/tinymemory-cortex/tests/live_cortexdb.rs +++ b/crates/tinymemory-integrations/tests/live_cortexdb.rs @@ -20,8 +20,8 @@ use tinymemory_api::{ MemoryMeta, MetaFilter, RecallRequest, Role, SourceKind, SourceRef, StoreItem, ToolCallRef, Turn, }; -use tinymemory_context::{ContextSpec, compile}; -use tinymemory_cortex::{CortexCredential, CortexEngine}; +use tinymemory_integrations::cortex::{CortexCredential, CortexEngine}; +use tinymemory_tools::context::{ContextSpec, compile}; const DEFAULT_KEY: &str = "tinymemory-cortex-test"; @@ -76,7 +76,7 @@ async fn the_live_server_upholds_the_contract() { eprintln!("TINYMEMORY_LIVE_CORTEXDB_URL unset; skipping"); return; }; - tinymemory_conformance::run(&engine) + tinymemory_api::conformance::run(&engine) .await .expect("the live CortexDB conforms"); } diff --git a/crates/tinymemory/tests/office_live.rs b/crates/tinymemory-integrations/tests/office_live.rs similarity index 89% rename from crates/tinymemory/tests/office_live.rs rename to crates/tinymemory-integrations/tests/office_live.rs index 05d4ff2b..d7d6d2bb 100644 --- a/crates/tinymemory/tests/office_live.rs +++ b/crates/tinymemory-integrations/tests/office_live.rs @@ -1,15 +1,17 @@ -//! Exercises Office conversion through the facade and into a live CortexDB. -#![cfg(feature = "documents-office")] +//! Exercises Office conversion through `documents` and into a live CortexDB. +#![cfg(all(feature = "documents-office", feature = "cortex"))] #![allow(clippy::expect_used)] use std::io::Write; use std::time::{Duration, Instant}; -use tinymemory::cortex::{CortexCredential, CortexEngine}; -use tinymemory::documents::{ConverterChain, OfficeConverter, RawDocument, document_item}; -use tinymemory::{ +use tinymemory_api::{ ItemKind, ListRequest, MemoryEngine, MemoryMeta, MetaFilter, SourceKind, SourceRef, }; +use tinymemory_integrations::cortex::{CortexCredential, CortexEngine}; +use tinymemory_integrations::documents::{ + ConverterChain, OfficeConverter, RawDocument, document_item, +}; const DEFAULT_KEY: &str = "tinymemory-cortex-test"; diff --git a/crates/tinymemory-sources/tests/reader_dispatch.rs b/crates/tinymemory-integrations/tests/reader_dispatch.rs similarity index 85% rename from crates/tinymemory-sources/tests/reader_dispatch.rs rename to crates/tinymemory-integrations/tests/reader_dispatch.rs index 1d2cbed9..6cb12752 100644 --- a/crates/tinymemory-sources/tests/reader_dispatch.rs +++ b/crates/tinymemory-integrations/tests/reader_dispatch.rs @@ -1,8 +1,8 @@ //! Public reader-dispatch policy tests. -use tinymemory_sources::{ - readers::{is_locally_readable, reader_for}, +use tinymemory_integrations::sources::{ SourceKind, + readers::{is_locally_readable, reader_for}, }; #[test] @@ -27,10 +27,10 @@ fn timer_dispatch_constructs_only_readers_that_never_need_network() { } } -#[cfg(feature = "network")] +#[cfg(feature = "sources-network")] #[test] fn request_dispatch_hands_out_a_reader_for_every_kind() { - use tinymemory_sources::readers::reader_for_request; + use tinymemory_integrations::sources::readers::reader_for_request; for kind in SourceKind::ALL { assert_eq!(reader_for_request(&kind).kind(), kind); diff --git a/crates/tinymemory-import/tests/support/mod.rs b/crates/tinymemory-integrations/tests/support/mod.rs similarity index 100% rename from crates/tinymemory-import/tests/support/mod.rs rename to crates/tinymemory-integrations/tests/support/mod.rs diff --git a/crates/tinymemory-safety/Cargo.toml b/crates/tinymemory-safety/Cargo.toml deleted file mode 100644 index 82495fbe..00000000 --- a/crates/tinymemory-safety/Cargo.toml +++ /dev/null @@ -1,26 +0,0 @@ -[package] -name = "tinymemory-safety" -publish = false -version = "0.1.0" -edition = "2021" -rust-version = "1.96" -license = "GPL-3.0-only" -description = "Secret and PII scrubbing for memory writes: credential patterns, sensitive-key classifier, checksum-gated multilingual national-ID redaction" -repository = "https://github.com/tinyhumansai/tinymemory" - -# Deliberately tiny: a scrubber is regexes over text and JSON. No engine, no -# runtime, no storage, so an engine and a host can both depend on it without -# pulling anything else in. -[dependencies] -# `scrub_item` cleans a `StoreItem` before it is stored. The contract performs -# no I/O, so this adds no runtime or storage dependency. -tinymemory-api = { path = "../tinymemory-api" } -regex = "1" -serde_json = "1" -log = "0.4" - -[lints.rust] -unsafe_code = "forbid" - -[lints.clippy] -all = { level = "warn", priority = -1 } diff --git a/crates/tinymemory-safety/src/default_policy_tests.rs b/crates/tinymemory-safety/src/default_policy_tests.rs deleted file mode 100644 index a550a7de..00000000 --- a/crates/tinymemory-safety/src/default_policy_tests.rs +++ /dev/null @@ -1,86 +0,0 @@ -use super::*; -use serde_json::json; - -use crate::pii::{redact_pii, PII_CC}; -// `pii`'s internals (checksum validators, the normalization pass) are test-only -// re-exports at the `pii` module level; pull them in here so the nested test -// submodules below can reach them through their own `use super::*;`. -use crate::pii::{ - digits, scan_candidates, valid_cnpj, valid_cpf, valid_cuit, valid_dni_es, valid_iban, - valid_luhn, valid_nie_es, valid_nino, valid_ssn, valid_verhoeff, NormalizedView, -}; -use crate::{MAX_JSON_SANITIZE_DEPTH, REDACTED_PRIVATE_KEY, REDACTED_SECRET}; - -/// Assembled rather than written out so a repository secret scanner does -/// not read the fixture as a real key block. -fn private_key_fixture(kind: &str, body: &str) -> String { - format!("-----BEGIN {kind}-----\n{body}\n-----END {kind}-----") -} - -fn redacts(input: &str, token: &str) { - let out = redact_pii(input); - assert!( - out.value.contains(token), - "expected {token} in output. input={input:?} output={out:?}" - ); -} - -fn unchanged(input: &str) { - let out = redact_pii(input); - assert_eq!( - out.value, input, - "expected no change; report={:?}", - out.report - ); - assert_eq!(out.report.pii_redactions, 0); -} - -#[path = "default_policy_prefilter_tests.rs"] -mod default_policy_prefilter_tests; -#[path = "default_policy_sanitize_tests.rs"] -mod default_policy_sanitize_tests; - -/// The one place the two historical copies differed: a bare Luhn-valid run that -/// is neither a real network IIN nor near a card keyword (here a 13-digit -/// epoch-millisecond timestamp). The default policy is the strictest and -/// redacts it; the TinyCortex policy leaves it alone. -#[test] -fn bare_card_gate_is_the_only_policy_difference() { - let ts = "1700000000004"; - let json = format!("{{\"ts\": {ts}}}"); - - let strict = redact_pii(&json); - assert!( - strict.value.contains(PII_CC), - "default policy must redact: {strict:?}" - ); - assert_eq!( - crate::pii::redact_pii_with(&json, Policy::default()).value, - strict.value - ); - assert_eq!(Policy::default().bare_card, BareCardGate::LuhnOnly); - - let corroborated = crate::pii::redact_pii_with(&json, Policy::corroborated()); - assert_eq!( - corroborated.value, json, - "corroborated policy keeps timestamps" - ); - - // Real card, bare, real IIN: both policies redact. - let visa = "4111111111111111"; - assert!(redact_pii(visa).value.contains(PII_CC)); - assert!(crate::pii::redact_pii_with(visa, Policy::corroborated()) - .value - .contains(PII_CC)); - - // The JSON and text entry points thread the policy through. - let value = json!({ "ts": ts }); - assert_ne!( - sanitize_json(&value).value, - sanitize_json_with(&value, Policy::corroborated()).value - ); - assert_ne!( - sanitize_text(ts).value, - sanitize_text_with(ts, Policy::corroborated()).value - ); -} diff --git a/crates/tinymemory-safety/src/lib.rs b/crates/tinymemory-safety/src/lib.rs deleted file mode 100644 index df477b27..00000000 --- a/crates/tinymemory-safety/src/lib.rs +++ /dev/null @@ -1,417 +0,0 @@ -//! `tinymemory-safety` — secret and PII scrubbing for anything a memory host -//! persists or hands on. -//! -//! Conservative by design — it prefers false positives over leaking -//! credentials into long-lived stores. One copy of this policy is shared by the -//! memory engines and the OpenHuman host; it used to exist three times. -//! -//! [`scrub_item`] applies the policy to every text a -//! [`tinymemory_api::StoreItem`] carries, and is what a host runs on each item -//! before `MemoryEngine::store`. -//! -//! The exhaustive multilingual national-ID PII module ([`pii`], ~1k lines of -//! checksum logic) runs as part of [`sanitize_text`]. The write-rejection -//! boundary ([`has_likely_pii`]) stays stricter than content scrubbing: -//! formatted national IDs are rejected, while phone/email-like text is -//! scrubbed from content without rejecting every write that mentions them. -//! -//! Before the shape regexes, [`sanitize_text`] redacts the value after a -//! credential *marker* — a one-time-secret URL's `/secret/` and a `Bearer` -//! value too short for the regexes — keeping the marker and the prose around -//! it. [`redact_credential_markers`] runs just those rules, for a host that -//! scrubs plain text without the PII pass. -//! -//! # The one policy knob -//! -//! The previous copies differed in exactly one behaviour: how a *bare* -//! (separator-less) Luhn-valid 13-19 digit run is treated as a credit card. -//! The OpenHuman host redacted every such run; TinyCortex additionally demanded -//! corroboration (a real network IIN at an issued length, or a card keyword -//! nearby) so 13-digit epoch-millisecond timestamps in stored JSON envelopes -//! stopped being corrupted (opencompany#1201). [`BareCardGate`] names both and -//! the plain functions default to the stricter [`BareCardGate::LuhnOnly`], so no -//! caller that does not opt in redacts less than before. Callers that want the -//! corroborated behaviour use the `*_with` variants and [`Policy::corroborated`]. - -use std::sync::LazyLock; - -use regex::Regex; -use serde_json::Value; - -/// Exhaustive checksum-gated multilingual national-ID PII module. Content -/// scrubbing runs from [`sanitize_text`]; the boundary check is re-exported as -/// [`has_likely_pii`]. -pub mod pii; - -pub use pii::{has_likely_email, has_likely_pii}; - -/// Scrubbing a whole [`tinymemory_api::StoreItem`] before it is stored. -mod item; - -/// One-time-secret URLs and `Bearer` values, including short ones. -mod markers; - -pub use markers::redact_credential_markers; - -pub use item::{scrub_item, scrub_item_with}; - -pub(crate) const REDACTED_SECRET: &str = "[REDACTED_SECRET]"; -pub(crate) const REDACTED_PRIVATE_KEY: &str = "[REDACTED_PRIVATE_KEY]"; -pub(crate) const MAX_JSON_SANITIZE_DEPTH: usize = 128; - -/// How a bare (no separators) Luhn-valid 13-19 digit run is judged as a credit -/// card by the content scrubber. Separated runs (`4111 1111 1111 1111`) are -/// always Luhn-gated only. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub enum BareCardGate { - /// Redact every Luhn-valid run. The strictest behaviour and the default. - #[default] - LuhnOnly, - /// Also require a plausible network IIN at an issued length, or a card - /// keyword within 64 bytes, so machine identifiers such as 13-digit - /// epoch-millisecond timestamps are left alone. - Corroborated, -} - -/// Tunables for content scrubbing. The default never redacts less than -/// [`BareCardGate::LuhnOnly`]. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct Policy { - /// Gate applied to bare credit-card-shaped digit runs. - pub bare_card: BareCardGate, -} - -impl Policy { - /// The policy the TinyCortex engine has always applied: bare card runs need - /// corroboration beyond their checksum. - pub const fn corroborated() -> Self { - Self { - bare_card: BareCardGate::Corroborated, - } - } -} - -/// Tally of what a sanitization pass changed. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct SanitizationReport { - /// Count of secret/token pattern matches rewritten in string text by the - /// text-pattern redaction pass. - pub text_redactions: usize, - /// Count of JSON object entries dropped wholesale because their key was - /// classified as sensitive by the key classifier. - pub key_redactions: usize, - /// Count of full private-key blocks replaced; these are - /// the most severe hits since the entire block is removed. - pub blocked_secret_hits: usize, - /// Count of nodes collapsed because JSON nesting reached - /// the JSON traversal depth cap; the subtree is replaced rather than walked. - pub depth_redactions: usize, - /// Count of personal-identifier matches replaced by the - /// lightweight PII screen. - pub pii_redactions: usize, -} - -impl SanitizationReport { - /// True when any field recorded a redaction. - pub fn changed(&self) -> bool { - self.text_redactions > 0 - || self.key_redactions > 0 - || self.blocked_secret_hits > 0 - || self.depth_redactions > 0 - || self.pii_redactions > 0 - } - - /// Sum two reports field-wise. - pub fn merge(self, rhs: Self) -> Self { - Self { - text_redactions: self.text_redactions + rhs.text_redactions, - key_redactions: self.key_redactions + rhs.key_redactions, - blocked_secret_hits: self.blocked_secret_hits + rhs.blocked_secret_hits, - depth_redactions: self.depth_redactions + rhs.depth_redactions, - pii_redactions: self.pii_redactions + rhs.pii_redactions, - } - } -} - -/// A sanitized value plus the [`SanitizationReport`] describing the changes. -#[derive(Debug, Clone)] -pub struct Sanitized { - /// The cleaned value with secrets and PII removed. - pub value: T, - /// Tally of what the sanitization pass changed to produce `value`. - pub report: SanitizationReport, -} - -static BLOCK_PATTERNS: LazyLock> = LazyLock::new(|| { - vec![ - Regex::new( - r"(?is)-----BEGIN(?: [A-Z]+)? PRIVATE KEY-----.*?-----END(?: [A-Z]+)? PRIVATE KEY-----", - ) - .expect("valid private key block"), - Regex::new(r"(?is)-----BEGIN OPENSSH PRIVATE KEY-----.*?-----END OPENSSH PRIVATE KEY-----") - .expect("valid openssh private key block"), - Regex::new( - r"(?is)-----BEGIN PGP PRIVATE KEY BLOCK-----.*?-----END PGP PRIVATE KEY BLOCK-----", - ) - .expect("valid pgp private key block"), - ] -}); - -static REDACTION_PATTERNS: LazyLock> = LazyLock::new(|| { - vec![ - ( - Regex::new(r"(?i)(bearer\s+)[A-Za-z0-9._~+/=-]{8,}").expect("valid bearer redaction"), - "${1}[REDACTED]", - ), - ( - Regex::new(r#"(?i)(api[_-]?key\s*[=:\s]\s*["']?)[^\s"']+"#) - .expect("valid api key redaction"), - "${1}[REDACTED]", - ), - ( - Regex::new( - r#"(?i)\b(token|access[_-]?token|refresh[_-]?token|client[_-]?secret|password|secret)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, - ) - .expect("valid token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-[A-Za-z0-9]{20,}\b").expect("valid openai key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b").expect("valid github token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bAKIA[0-9A-Z]{16}\b").expect("valid aws key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bASIA[0-9A-Z]{16}\b").expect("valid aws sts key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9._-]{8,}\.[A-Za-z0-9._-]{8,}\b") - .expect("valid jwt redaction"), - "[REDACTED]", - ), - ( - Regex::new( - r#"(?i)\b(access_token|refresh_token|id_token|authorization_code|code_verifier|code_challenge)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, - ) - .expect("valid oauth token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bAIza[0-9A-Za-z\-_]{35}\b").expect("valid google api key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-ant-[A-Za-z0-9\-_]{16,}\b").expect("valid anthropic key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-(?:proj|org)-[A-Za-z0-9\-_]{12,}\b") - .expect("valid openai scoped key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}\b") - .expect("valid stripe key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bxox(?:a|b|p|s|r)-[A-Za-z0-9-]{10,}\b") - .expect("valid slack token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bgithub_pat_[A-Za-z0-9_]{20,}\b").expect("valid github pat redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bglpat-[A-Za-z0-9\-_]{16,}\b").expect("valid gitlab pat redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bnpm_[A-Za-z0-9]{20,}\b").expect("valid npm token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bSG\.[A-Za-z0-9_\-]{16,}\.[A-Za-z0-9_\-]{16,}\b") - .expect("valid sendgrid key redaction"), - "[REDACTED]", - ), - ] -}); - -/// True when `value` looks like it contains a credential. -pub fn has_likely_secret(value: &str) -> bool { - BLOCK_PATTERNS.iter().any(|p| p.is_match(value)) - || REDACTION_PATTERNS.iter().any(|(p, _)| p.is_match(value)) -} - -/// Scrub secrets and PII from free text, returning the cleaned text plus a -/// [`SanitizationReport`]. -pub fn sanitize_text(value: &str) -> Sanitized { - sanitize_text_with(value, Policy::default()) -} - -/// [`sanitize_text`] under an explicit [`Policy`]. -pub fn sanitize_text_with(value: &str, policy: Policy) -> Sanitized { - let mut out = value.to_string(); - let mut report = SanitizationReport::default(); - - for pattern in BLOCK_PATTERNS.iter() { - let hits = pattern.find_iter(&out).count(); - if hits > 0 { - report.blocked_secret_hits += hits; - out = pattern.replace_all(&out, REDACTED_PRIVATE_KEY).into_owned(); - } - } - - // Values after a credential marker (`/secret/`, `Bearer `), - // before the shape regexes: it catches what they cannot — a one-time key, - // a short bearer value — and its `[REDACTED]` is not token-shaped, so no - // regex below fires on it again. Only ever replaces, so the pass makes the - // scrubber strictly stricter. - let (marked, hits) = markers::redact_counted(&out); - if hits > 0 { - report.text_redactions += hits; - out = marked.into_owned(); - } - - for (pattern, replacement) in REDACTION_PATTERNS.iter() { - let hits = pattern.find_iter(&out).count(); - if hits > 0 { - report.text_redactions += hits; - out = pattern.replace_all(&out, *replacement).into_owned(); - } - } - - // Full multilingual national-ID PII scrub (checksum-gated, normalization - // pre-pass) — runs after secret redaction so every call site that scrubs - // secrets also scrubs PII. - let pii = pii::redact_pii_with(&out, policy); - report = report.merge(pii.report); - out = pii.value; - - Sanitized { value: out, report } -} - -/// Recursively scrub a JSON value: sensitive keys are replaced wholesale and -/// every string value runs through `sanitize_text`. -pub fn sanitize_json(value: &Value) -> Sanitized { - sanitize_json_with(value, Policy::default()) -} - -/// [`sanitize_json`] under an explicit [`Policy`]. -pub fn sanitize_json_with(value: &Value, policy: Policy) -> Sanitized { - sanitize_json_inner(value, 0, policy) -} - -/// Recursive worker behind [`sanitize_json`]. -/// -/// `depth` counts nesting from the call in `sanitize_json` (which starts at -/// `0`); once it reaches [`MAX_JSON_SANITIZE_DEPTH`] the whole subtree at that -/// point is replaced by a single redaction marker rather than walked further, -/// bounding recursion against pathologically deep or adversarial JSON. -fn sanitize_json_inner(value: &Value, depth: usize, policy: Policy) -> Sanitized { - if depth >= MAX_JSON_SANITIZE_DEPTH { - return Sanitized { - value: Value::String(REDACTED_SECRET.to_string()), - report: SanitizationReport { - depth_redactions: 1, - ..SanitizationReport::default() - }, - }; - } - - match value { - Value::Object(map) => { - let mut out = serde_json::Map::new(); - let mut report = SanitizationReport::default(); - for (key, value) in map { - if is_sensitive_key(key) { - report.key_redactions += 1; - out.insert(key.clone(), Value::String(REDACTED_SECRET.to_string())); - continue; - } - let sanitized = sanitize_json_inner(value, depth + 1, policy); - report = report.merge(sanitized.report); - out.insert(key.clone(), sanitized.value); - } - Sanitized { - value: Value::Object(out), - report, - } - } - Value::Array(items) => { - let mut out = Vec::with_capacity(items.len()); - let mut report = SanitizationReport::default(); - for item in items { - let sanitized = sanitize_json_inner(item, depth + 1, policy); - report = report.merge(sanitized.report); - out.push(sanitized.value); - } - Sanitized { - value: Value::Array(out), - report, - } - } - Value::String(value) => { - let sanitized = sanitize_text_with(value, policy); - Sanitized { - value: Value::String(sanitized.value), - report: sanitized.report, - } - } - _ => Sanitized { - value: value.clone(), - report: SanitizationReport::default(), - }, - } -} - -/// True when a JSON object key's name itself suggests it holds a secret -/// (`api_key`, `token`, `password`, …), independent of the value's contents. -/// -/// Matching keys are redacted wholesale in [`sanitize_json_inner`] — the -/// value is replaced rather than scanned, since a key named e.g. `password` -/// is assumed sensitive even if its value doesn't match any -/// [`REDACTION_PATTERNS`] regex. Matching is on the key with all -/// non-alphanumeric characters stripped and lowercased, so `API-Key`, -/// `api_key`, and `apiKey` are all treated identically. -fn is_sensitive_key(key: &str) -> bool { - let normalized: String = key - .chars() - .filter(|c| c.is_ascii_alphanumeric()) - .map(|c| c.to_ascii_lowercase()) - .collect(); - - matches!( - normalized.as_str(), - "apikey" - | "token" - | "accesstoken" - | "refreshtoken" - | "authorization" - | "password" - | "secret" - | "clientsecret" - ) || normalized.ends_with("token") - || normalized.ends_with("apikey") - || normalized.ends_with("clientsecret") - || normalized.contains("password") - || normalized.contains("secret") - || normalized.ends_with("key") -} - -#[cfg(test)] -#[path = "safety_tests.rs"] -mod tests; - -#[cfg(test)] -#[path = "default_policy_tests.rs"] -mod default_policy_tests; diff --git a/crates/tinymemory-sources/Cargo.toml b/crates/tinymemory-sources/Cargo.toml deleted file mode 100644 index 58a9e0c8..00000000 --- a/crates/tinymemory-sources/Cargo.toml +++ /dev/null @@ -1,89 +0,0 @@ -[package] -name = "tinymemory-sources" -version = "0.1.0" -edition = "2021" -rust-version = "1.96" -license = "GPL-3.0-only" -repository = "https://github.com/tinyhumansai/tinymemory" -description = "Source readers for TinyMemory: folders, files, links, GitHub, RSS, Composio payloads and conversations turned into StoreItems" -publish = false - -[dependencies] -# The contract: every reader's output ends as a `StoreItem` carrying -# `MemoryMeta`, and `tinymemory_api::Error` is what a host maps reader failures -# onto. -tinymemory-api = { path = "../tinymemory-api" } -# Conversion to markdown (`markdown_from_text`, `document_item`), the size cap -# a fetch reads up to, and `language_for_path` for code files. Sources sit -# above documents: documents does no I/O, sources does all of it. -tinymemory-documents = { path = "../tinymemory-documents" } -# The crate-wide `Error`. -thiserror = "2" -# The source types are serde shapes: they are persisted in the host's source -# registry and cross the RPC surface. -serde = { version = "1", features = ["derive"] } -# `MemorySourceEntry` and friends appear in generated schemas, same as the -# contract crate's own types. -schemars = "1.2" -# `MemorySourceEntry::metadata`-style open values, thread files, and every -# Composio payload are JSON. -serde_json = "1" -# `SourceReader` is an object-safe async trait so network and local readers -# share one surface. -async-trait = "0.1" -# The folder reader compiles a source's glob to a regex. -regex = "1.10" -# The folder reader walks the directory tree. -walkdir = "2" -# Timestamps: `MemoryMeta::observed_at`, file mtimes, feed and issue dates, -# Gmail `Date:` headers. `clock` is needed by the Composio email normaliser, -# which renders a message time in the host's local timezone. -chrono = { version = "0.4", features = ["clock", "serde"] } -# The registry is the host's `sources.toml`: it reads, mutates and rewrites it. -toml = "1.1" -# Diagnostics on the local readers, the registry and the Slack normaliser. -log = "0.4" -# Diagnostics on the network readers and the Gmail normaliser. -tracing = "0.1" -# New sources get a generated id; registry temp files get a unique name. -uuid = { version = "1", features = ["v4"] } -# The readers that fetch over the network — GitHub, RSS, web pages, URL fetch. -futures = { version = "0.3", optional = true } -reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "stream"], optional = true } -# The GitHub reader shells out to `git` and `gh`; the SSRF resolver looks up -# hosts. -tokio = { version = "1", features = ["process", "io-util", "net", "time"], optional = true } - -[dev-dependencies] -tempfile = "3" -# The reader tests are async. -tokio = { version = "1", features = ["macros", "rt", "rt-multi-thread", "net", "io-util"] } - -[features] -# Nothing by default: a host that only reads local folders, files and -# conversations links no HTTP stack. -default = [] -# The readers that fetch over the network — GitHub, RSS, web pages — and -# `fetch::fetch_url`, all behind the shared SSRF guard. -network = ["dep:reqwest", "dep:futures", "dep:tokio"] - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory-sources/README.md b/crates/tinymemory-sources/README.md deleted file mode 100644 index f9d10b31..00000000 --- a/crates/tinymemory-sources/README.md +++ /dev/null @@ -1,54 +0,0 @@ -# tinymemory-sources - -Readers that turn a source into `StoreItem`s: a folder, a single file, a web -page, a GitHub repository, an RSS feed, a Composio toolkit payload, or the -host's local conversation threads. Conversion to markdown and language -detection come from `tinymemory-documents`. - -## Layers - -| Module | Owns | -| --- | --- | -| `types`, `validation`, `registry`, `reconcile` | the configuration a host persists: `MemorySourceEntry` keyed by `SourceKind`, `MemorySourcePatch`, field rules, the `[[memory_sources]]` TOML registry, Composio reconciliation | -| `readers` | `SourceReader` (list, read, read as a `StoreItem`) and one reader per kind; the SSRF guard (`readers::ssrf`) | -| `fetch` | one URL into a `RawDocument` or a link item, behind the SSRF guard (`network`) | -| `items` | reader output to `StoreItem`s with `MemoryMeta` filled per kind; `collect_items` drives a reader end to end | -| `composio` | toolkit normalisers (Gmail, Slack, GitHub, Linear, Notion, ClickUp) and `payload_items` | -| `error` | the crate `Error`, mapped onto `tinymemory_api::Error` | - -## Kinds and metadata - -Every item's `meta.source` is `SourceRef { kind, id: Some(entry.id) }`. - -| Config kind | `SourceKind` | Item | Metadata | -| --- | --- | --- | --- | -| `folder` | `Folder` | document | `workspace`, `folder` (containing directory), `file_path`, `language`, `observed_at` (mtime), `mime` | -| `file` | `File` | document | as `folder`; `file_item` reads a path with no configured source | -| `web_page` | `Link` | document | `url` | -| `github_repo` | `Github` | document | `repo` (`owner/name`), `commit` (commits), `url` (issues, PRs), `observed_at` | -| `rss_feed` | `Rss` | document | `url` (the entry's link), `observed_at` (published) | -| `composio` | `Composio` | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | -| `conversation` | `Conversation` | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | - -## Folder selection - -With a glob, a folder source takes exactly the matching files. Without one it -takes markdown, plain text and source code (`is_default_candidate`). Either way -it skips hidden files and directories and `target`, `node_modules`, -`__pycache__` and `venv`, never follows symlinks while walking, refuses files -over `FOLDER_FILE_SIZE_CAP_BYTES` (10 MiB), and confines reads to the folder -root (`ensure_within_base`). A relative path is anchored on the workspace, not -the process working directory. - -## Who decides when - -`readers::reader_for` hands out only the local readers (folder, file, -conversation), which are safe to drive on a timer. Network readers are -constructed explicitly, or through `reader_for_request` for an explicit user -request. Scheduling, credentials, OAuth and egress budgets stay with the host. - -## Features - -- `network` — the GitHub, RSS and web-page readers, `fetch`, and the SSRF - guard. Off by default, so a host that only reads local sources links no HTTP - stack. diff --git a/crates/tinymemory-sources/src/composio/clickup.rs b/crates/tinymemory-sources/src/composio/clickup.rs deleted file mode 100644 index ae340b67..00000000 --- a/crates/tinymemory-sources/src/composio/clickup.rs +++ /dev/null @@ -1,133 +0,0 @@ -//! ClickUp host normalization helpers — result extraction, task-title extraction, -//! and time utilities. -//! -//! ClickUp's REST API (and therefore Composio's wrapping of it) returns -//! task lists in a small handful of shapes depending on which endpoint -//! is called. The functions here walk the union of common shapes so the -//! provider doesn't have to branch per Composio envelope variant. - -use serde_json::Value; - -use super::helpers::pick_str; - -/// Walk the Composio response envelope for ClickUp task list results. -/// -/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }` -/// at the top level; Composio re-wraps the upstream payload under -/// `data` or `data.data` depending on the action. We probe each shape -/// in order and return the first array we find. -pub fn extract_tasks(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/tasks"), - data.pointer("/tasks"), - data.pointer("/data/data/tasks"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a ClickUp task object. -/// -/// ClickUp tasks store the name at `name` (or `data.name` after Composio -/// envelope wrapping). When the name is missing we fall back to the -/// task ID so chunks remain identifiable. -pub fn extract_task_name(task: &Value) -> Option { - pick_str(task, &["name", "data.name", "title", "data.title"]) -} - -/// Extract a stable cursor timestamp (milliseconds since epoch as a -/// string) from a ClickUp task object. -/// -/// The ClickUp API returns `date_updated` as a stringified epoch ms -/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic -/// comparison against the stored cursor remains valid as long as the -/// length doesn't change (it won't until year 33658). -pub fn extract_task_updated(task: &Value) -> Option { - pick_str( - task, - &[ - "date_updated", - "data.date_updated", - "updated_at", - "data.updated_at", - "dateUpdated", - "data.dateUpdated", - ], - ) -} - -/// Current wall-clock time in milliseconds since the UNIX epoch. -pub fn now_ms() -> u64 { - use std::time::{SystemTime, UNIX_EPOCH}; - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0) -} - -/// Extract the authorized user's numeric ID from the -/// `CLICKUP_GET_AUTHORIZED_USER` response. -/// -/// Composio wraps the upstream `{"user": {"id": …}}` shape; this walker -/// is defensive against both raw and wrapped payloads. Returns the ID -/// as a string because `CLICKUP_GET_FILTERED_TEAM_TASKS` accepts the -/// `assignees` filter as a string array. -pub fn extract_user_id(data: &Value) -> Option { - let candidates = [ - data.pointer("/user/id"), - data.pointer("/data/user/id"), - data.pointer("/id"), - data.pointer("/data/id"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(n) = cand.as_u64() { - return Some(n.to_string()); - } - if let Some(n) = cand.as_i64() { - return Some(n.to_string()); - } - if let Some(s) = cand.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - None -} - -/// Extract a list of workspace (team) IDs from the -/// `CLICKUP_GET_AUTHORIZED_TEAMS_WORKSPACES` response. -/// -/// ClickUp returns `{"teams": [{"id": "...", "name": "..."}, …]}`. We -/// keep the IDs as strings — `CLICKUP_GET_FILTERED_TEAM_TASKS` requires -/// a `team_id` (string) argument. -pub fn extract_workspace_ids(data: &Value) -> Vec { - let candidates = [ - data.pointer("/teams"), - data.pointer("/data/teams"), - data.pointer("/workspaces"), - data.pointer("/data/workspaces"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr - .iter() - .filter_map(|t| pick_str(t, &["id", "team_id", "workspace_id"])) - .collect(); - } - } - Vec::new() -} - -#[cfg(test)] -#[path = "clickup_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/composio/email_clean.rs b/crates/tinymemory-sources/src/composio/email_clean.rs deleted file mode 100644 index 37671f80..00000000 --- a/crates/tinymemory-sources/src/composio/email_clean.rs +++ /dev/null @@ -1,264 +0,0 @@ -//! Shared email rendering + cleaning helpers. -//! -//! Used by [`super::email_markdown`] when rendering canonical email markdown. The module -//! is intentionally pure-string-oriented plus a single `serde_json::Value` -//! helper (`parse_message_date`) for callers that work directly off slim -//! envelope JSON. Nothing here depends on the chunk-store types, which keeps the -//! helpers reusable. - -use chrono::{DateTime, NaiveDate, Utc}; -use serde_json::Value; - -/// Two-stage cleanup applied to each message body before it gets rendered into -/// a digest: -/// -/// 1. **Drop quoted reply chains** — once a message contains a -/// `On , wrote:` preamble, an `Original Message` / -/// `Forwarded message` separator, or a run of three+ consecutive -/// `>`-prefixed lines, everything from that point onward is the parent -/// message we already render directly above. -/// 2. **Drop footer noise** — `Unsubscribe`, `View in browser`, copyright -/// lines, legal disclaimers, and address blocks. We cut at the first line -/// containing a known footer trigger. -/// -/// The two passes run in order so a quoted-chain preamble below a -/// "view in browser" line still gets stripped on its own merits even if the -/// footer pass missed it. -pub fn clean_body(raw: &str) -> String { - let stage1 = drop_reply_chain(raw); - let stage2 = drop_footer_noise(&stage1); - collapse_blank_runs(stage2.trim()) -} - -/// Substrings that, when matched (case-insensitive) anywhere on a line, mark -/// the start of footer / boilerplate territory. Conservative list — every entry -/// should be unambiguous noise that wouldn't reasonably appear inside real -/// prose. -const FOOTER_TRIGGERS: &[&str] = &[ - "unsubscribe", - "view in browser", - "view this email in your browser", - "view it in your browser", - "update your email settings", - "manage your subscription", - "manage preferences", - "email preferences", - "you are receiving this email because", - "you received this email because", - "you're receiving this email because", - "to stop receiving", - "all rights reserved", - "© 20", - "(c) 20", - "copyright 20", - "powered by mailchimp", - "sent via sendgrid", - "this email and any files", - "confidentiality notice", - "if you are not the intended recipient", - "this communication may contain", -]; - -/// Strip quoted reply chains. See [`clean_body`] for details. -pub fn drop_reply_chain(s: &str) -> String { - let mut offset = 0usize; - let mut quoted_run_start: Option = None; - let mut quoted_run_len = 0u32; - - for line in s.split_inclusive('\n') { - let trimmed = line.trim(); - let lower = trimmed.to_ascii_lowercase(); - - // Explicit reply / forward markers. - let is_preamble = (lower.starts_with("on ") && lower.contains(" wrote:")) - || lower.contains("---------- forwarded message") - || lower.contains("----- original message") - || lower.contains("--------- original message") - || lower.contains("--- forwarded by"); - if is_preamble { - debug_assert!(s.is_char_boundary(offset)); - return s[..offset].trim_end().to_string(); - } - - // Three+ consecutive lines starting with `>` is a quoted reply chain in - // disguise (some clients de-quote on send). Treat the start of the run - // as the cut point. - if trimmed.starts_with('>') { - if quoted_run_start.is_none() { - quoted_run_start = Some(offset); - quoted_run_len = 1; - } else { - quoted_run_len += 1; - } - if quoted_run_len >= 3 { - let cut = quoted_run_start.unwrap_or(offset); - debug_assert!(s.is_char_boundary(cut)); - return s[..cut].trim_end().to_string(); - } - } else if !trimmed.is_empty() { - // Reset on a non-empty, non-quoted line. Blank lines don't break a - // quote run because senders often interleave them. - quoted_run_start = None; - quoted_run_len = 0; - } - - offset += line.len(); - } - s.to_string() -} - -/// Strip everything from the first line containing a footer trigger onward. -/// Uses the module's known footer-trigger list. -pub fn drop_footer_noise(s: &str) -> String { - let mut offset = 0usize; - for line in s.split_inclusive('\n') { - let lower = line.to_ascii_lowercase(); - if FOOTER_TRIGGERS.iter().any(|t| lower.contains(t)) { - debug_assert!(s.is_char_boundary(offset)); - return s[..offset].trim_end().to_string(); - } - offset += line.len(); - } - s.to_string() -} - -/// Collapse runs of 2+ blank lines into a single blank line. Trims trailing -/// newlines. -pub fn collapse_blank_runs(s: &str) -> String { - let mut out = String::with_capacity(s.len()); - let mut blank = 0u32; - for line in s.lines() { - if line.trim().is_empty() { - blank += 1; - if blank <= 1 { - out.push('\n'); - } - } else { - blank = 0; - out.push_str(line); - out.push('\n'); - } - } - while out.ends_with('\n') { - out.pop(); - } - out -} - -/// Truncate a body to at most `max_chars` characters, appending `…` when the -/// body is longer. Trims first so leading/trailing whitespace doesn't count -/// against the budget. -pub fn truncate_body(body: &str, max_chars: usize) -> String { - let trimmed = body.trim(); - if trimmed.chars().count() <= max_chars { - return trimmed.to_string(); - } - let mut out: String = trimmed.chars().take(max_chars).collect(); - out.push('…'); - out -} - -/// Escape only the few markdown chars that would visibly break the -/// header/inline contexts we use (#, |, *, _, `). Newlines collapse to spaces. -/// We leave most punctuation alone — the body is rendered as a blockquote -/// anyway. -pub fn md_escape(s: &str) -> String { - let mut out = String::with_capacity(s.len()); - for ch in s.chars() { - match ch { - '\\' | '`' | '*' | '_' | '|' => { - out.push('\\'); - out.push(ch); - } - '\n' | '\r' => out.push(' '), - _ => out.push(ch), - } - } - out -} - -/// Pull the `` portion out of a `From` header, returning just the -/// bare email address. Falls back to `None` when no `<…>` brackets exist; in -/// that case the caller may use the raw From field. -pub fn extract_email(from: &str) -> Option { - let s = from.trim(); - if let (Some(start), Some(end)) = (s.rfind('<'), s.rfind('>')) { - if start < end { - debug_assert!(s.is_char_boundary(start + 1)); - debug_assert!(s.is_char_boundary(end)); - let inner = s[start + 1..end].trim(); - if inner.contains('@') { - return Some(inner.to_string()); - } - } - } - if s.contains('@') && !s.contains(' ') { - return Some(s.to_string()); - } - None -} - -/// If `s` starts with a 3-letter day-of-week prefix (`Mon, `, `Tue, `, …), -/// return the remainder; otherwise `None`. Used to feed a strict-rfc2822 reject -/// into a lenient retry. -fn strip_day_of_week_prefix(s: &str) -> Option<&str> { - const DAYS: &[&str] = &["Mon", "Tue", "Wed", "Thu", "Fri", "Sat", "Sun"]; - let (prefix, rest) = s.split_once(", ")?; - if DAYS.iter().any(|d| d.eq_ignore_ascii_case(prefix)) { - Some(rest) - } else { - None - } -} - -/// Try a sequence of common date formats. The slim envelope sets `date` from -/// `messageTimestamp` (often ISO 8601 or epoch ms) when present, falling back -/// to the raw `Date:` header (RFC 2822). Operates on the raw `serde_json::Value` -/// so callers that work off the slim envelope JSON don't have to reshape it -/// first. -pub fn parse_message_date(m: &Value) -> Option> { - if let Some(dt) = m.get("date").and_then(parse_date_value) { - return Some(dt); - } - if let Some(dt) = m.get("internalDate").and_then(parse_date_value) { - return Some(dt); - } - m.get("data") - .and_then(|data| data.get("internalDate")) - .and_then(parse_date_value) -} - -fn parse_date_value(raw: &Value) -> Option> { - if let Some(s) = raw.as_str() { - let s = s.trim(); - if s.is_empty() { - return None; - } - // Epoch millis as a string? Gmail's `internalDate` uses this form. - if let Ok(ms) = s.parse::() { - return DateTime::from_timestamp_millis(ms); - } - if let Ok(dt) = DateTime::parse_from_rfc3339(s) { - return Some(dt.with_timezone(&Utc)); - } - if let Ok(dt) = DateTime::parse_from_rfc2822(s) { - return Some(dt.with_timezone(&Utc)); - } - // Lenient RFC 2822 fallback: strict `parse_from_rfc2822` rejects - // mismatched day-of-week. Strip a `, ` prefix and retry with - // the rfc2822 body format. - if let Some(rest) = strip_day_of_week_prefix(s) { - if let Ok(dt) = DateTime::parse_from_str(rest, "%d %b %Y %H:%M:%S %z") { - return Some(dt.with_timezone(&Utc)); - } - } - if let Ok(d) = NaiveDate::parse_from_str(s, "%Y-%m-%d") { - return d.and_hms_opt(0, 0, 0).map(|n| n.and_utc()); - } - } - raw.as_i64().and_then(DateTime::from_timestamp_millis) -} - -#[cfg(test)] -#[path = "email_clean_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/composio/email_clean_tests.rs b/crates/tinymemory-sources/src/composio/email_clean_tests.rs deleted file mode 100644 index 2dc4a57e..00000000 --- a/crates/tinymemory-sources/src/composio/email_clean_tests.rs +++ /dev/null @@ -1,145 +0,0 @@ -//! Tests for the email body cleaning helpers. - -use super::*; -use serde_json::json; - -#[test] -fn drop_reply_chain_strips_on_x_wrote_preamble() { - let body = "Sounds good — let's do Tuesday.\n\nOn Mon, Apr 22, 2026 at 10:00 AM, Alice wrote:\n> Tuesday or Wednesday?\n> Let me know."; - let cleaned = drop_reply_chain(body); - assert_eq!(cleaned.trim(), "Sounds good — let's do Tuesday."); -} - -#[test] -fn drop_reply_chain_strips_forwarded_separator() { - let body = "FYI.\n\n---------- Forwarded message ---------\nFrom: bob\nSubject: hi"; - assert_eq!(drop_reply_chain(body).trim(), "FYI."); -} - -#[test] -fn drop_reply_chain_strips_consecutive_quoted_run() { - let body = "Thanks for the update.\n\n> earlier line 1\n> earlier line 2\n> earlier line 3\n> earlier line 4"; - assert_eq!(drop_reply_chain(body).trim(), "Thanks for the update."); -} - -#[test] -fn drop_reply_chain_keeps_short_quote() { - let body = "I think:\n> That sounds reasonable\n\nLet's proceed."; - let cleaned = drop_reply_chain(body); - assert!(cleaned.contains("Let's proceed")); - assert!(cleaned.contains("That sounds reasonable")); -} - -#[test] -fn drop_footer_noise_strips_unsubscribe_block() { - let body = - "Big news: GPT-5.5 is here.\n\nRead more at example.com\n\nUnsubscribe | © 2026 OpenAI"; - let cleaned = drop_footer_noise(body); - assert!(cleaned.contains("GPT-5.5")); - assert!(!cleaned.to_ascii_lowercase().contains("unsubscribe")); - assert!(!cleaned.contains("©")); -} - -#[test] -fn drop_footer_noise_strips_legal_disclaimer() { - let body = "Action item — review by Friday.\n\nThis email and any files transmitted with it are confidential and intended solely for the use of the individual to whom they are addressed."; - let cleaned = drop_footer_noise(body); - assert_eq!(cleaned.trim(), "Action item — review by Friday."); -} - -#[test] -fn clean_body_combines_passes() { - let body = - "Real content here.\n\nOn Mon, Apr 22, 2026, Alice wrote:\n> old stuff\n\nUnsubscribe"; - let cleaned = clean_body(body); - assert_eq!(cleaned, "Real content here."); -} - -#[test] -fn collapse_blank_runs_keeps_paragraph_breaks() { - let s = "a\n\n\n\nb\n\n\nc\n"; - assert_eq!(collapse_blank_runs(s), "a\n\nb\n\nc"); -} - -#[test] -fn truncate_body_adds_ellipsis() { - let s = "x".repeat(2000); - let t = truncate_body(&s, 1200); - assert!(t.ends_with('…')); - assert_eq!(t.chars().count(), 1201); -} - -#[test] -fn truncate_body_passthrough_when_short() { - let s = "hello"; - let t = truncate_body(s, 1200); - assert_eq!(t, "hello"); -} - -#[test] -fn md_escape_handles_special_chars() { - assert_eq!(md_escape("a*b_c"), "a\\*b\\_c"); - assert_eq!(md_escape("foo|bar"), "foo\\|bar"); - assert_eq!(md_escape("line1\nline2"), "line1 line2"); - assert_eq!(md_escape("plain text"), "plain text"); -} - -#[test] -fn extract_email_handles_both_forms() { - assert_eq!( - extract_email("Alice ").as_deref(), - Some("alice@example.com") - ); - assert_eq!( - extract_email("notify@github.com").as_deref(), - Some("notify@github.com") - ); - assert_eq!( - extract_email("\"Bot Name\" ").as_deref(), - Some("bot@x.io") - ); - assert!(extract_email("Alice").is_none()); -} - -#[test] -fn parse_message_date_handles_iso_and_rfc2822() { - let iso = json!({"date": "2026-04-21T10:00:00Z"}); - let rfc = json!({"date": "Mon, 21 Apr 2026 10:00:00 +0000"}); - let ms = json!({"date": 1745236800000_i64}); - let ms_str = json!({"date": "1745236800000"}); - let internal_ms_str = json!({"internalDate": "1745236800000"}); - let nested_internal_ms_str = json!({"data": {"internalDate": "1745236800000"}}); - let date_only = json!({"date": "2026-04-21"}); - assert!(parse_message_date(&iso).is_some()); - assert!(parse_message_date(&rfc).is_some()); - assert!(parse_message_date(&ms).is_some()); - assert!(parse_message_date(&ms_str).is_some()); - assert!(parse_message_date(&internal_ms_str).is_some()); - assert!(parse_message_date(&nested_internal_ms_str).is_some()); - assert!(parse_message_date(&date_only).is_some()); -} - -#[test] -fn parse_message_date_returns_none_when_missing_or_blank() { - assert!(parse_message_date(&json!({})).is_none()); - assert!(parse_message_date(&json!({"date": ""})).is_none()); - assert!(parse_message_date(&json!({"date": " "})).is_none()); -} - -#[test] -fn drop_reply_chain_handles_zwnj_in_body() { - let zwnj = "\u{200c}"; - let body = format!( - "سلام{}دوست عزیز، لطفاً بررسی کنید.\n\nOn Mon, Apr 22, 2026, Alice wrote:\n> old content", - zwnj - ); - - let cleaned = drop_reply_chain(&body); - - assert!(!cleaned.contains("old content")); - assert!( - cleaned.contains(zwnj), - "ZWNJ was incorrectly removed from real content" - ); - assert!(std::str::from_utf8(cleaned.as_bytes()).is_ok()); -} diff --git a/crates/tinymemory-sources/src/composio/email_markdown.rs b/crates/tinymemory-sources/src/composio/email_markdown.rs deleted file mode 100644 index 99eb13eb..00000000 --- a/crates/tinymemory-sources/src/composio/email_markdown.rs +++ /dev/null @@ -1,180 +0,0 @@ -//! Email thread → markdown, shared shape with the engine's canonicaliser -//! (#18 §B1). -//! -//! The gmail pipeline stores one markdown document per message; the engine's -//! ingest path canonicalises full threads with the same header block and -//! body-cleaning rules. The exact output format is load-bearing twice over: -//! the chunker splits at `---\nFrom:` boundaries, and the engine writes the -//! same shape from its copy — `thread_markdown_format_is_pinned` holds the -//! two to one form. - -use chrono::{DateTime, Utc}; -use serde::{Deserialize, Deserializer, Serialize}; - -use super::email_clean; - -/// One message of a thread, in the canonicaliser's input shape. -#[derive(Clone, Debug, Serialize, Deserialize)] -pub struct EmailMessage { - /// Sender, as the provider renders it. - pub from: String, - /// Direct recipients. - #[serde(default)] - pub to: Vec, - /// Carbon-copy recipients. - #[serde(default)] - pub cc: Vec, - /// Subject line. - pub subject: String, - #[serde( - default = "chrono_now", - serialize_with = "chrono::serde::ts_milliseconds::serialize", - deserialize_with = "deserialize_flexible_timestamp" - )] - /// When the message was sent. Accepts epoch-ms or RFC 3339 on the wire. - pub sent_at: DateTime, - /// Body text, best rendering the provider offers. - pub body: String, - /// Opaque pointer back to the raw source record. - #[serde(default)] - pub source_ref: Option, - /// `List-Unsubscribe` header, when present. - #[serde(default)] - pub list_unsubscribe: Option, -} - -/// A thread of messages from one provider. -#[derive(Clone, Debug, Serialize, Deserialize)] -pub struct EmailThread { - /// Provider slug, e.g. `gmail`. - pub provider: String, - /// The thread's subject. - pub thread_subject: String, - /// Messages, any order; rendering sorts oldest-first. - pub messages: Vec, -} - -fn chrono_now() -> DateTime { - Utc::now() -} - -/// The thread as canonical markdown: one `---\nFrom:` block per message, -/// oldest first, bodies through [`email_clean::clean_body`]. `None` for an -/// empty thread. -pub fn thread_markdown(thread: EmailThread) -> Option { - if thread.messages.is_empty() { - return None; - } - let mut messages = thread.messages; - messages.sort_by_key(|m| m.sent_at); - - let mut md = String::new(); - // No leading `# Email thread — ...` header. Provider / subject info belongs - // in the MD front-matter. The chunker splits this output at `---\nFrom:` - // boundaries so each message becomes one chunk. - for msg in &messages { - md.push_str("---\n"); - md.push_str(&format!("From: {}\n", email_clean::md_escape(&msg.from))); - if !msg.to.is_empty() { - md.push_str(&format!( - "To: {}\n", - email_clean::md_escape(&msg.to.join(", ")) - )); - } - if !msg.cc.is_empty() { - md.push_str(&format!( - "Cc: {}\n", - email_clean::md_escape(&msg.cc.join(", ")) - )); - } - md.push_str(&format!( - "Subject: {}\n", - email_clean::md_escape(&msg.subject) - )); - md.push_str(&format!("Date: {}\n", msg.sent_at.to_rfc3339())); - - if let Some(unsub) = &msg.list_unsubscribe { - md.push_str(&format!( - "List-Unsubscribe: {}\n", - email_clean::md_escape(unsub) - )); - } - md.push('\n'); - let cleaned = email_clean::clean_body(msg.body.trim()); - if cleaned.is_empty() { - md.push('\n'); - } else { - let safe_body = cleaned - .lines() - .map(|line| { - if line.trim_end() == "---" { - format!("\\{line}") - } else { - line.to_string() - } - }) - .collect::>() - .join("\n"); - md.push_str(&safe_body); - } - md.push_str("\n\n"); - } - Some(md) -} - -/// Deserialise a `DateTime` from either: -/// - a JSON integer = epoch **milliseconds** (legacy callers — back-compat), -/// - a JSON string = RFC 3339 / ISO-8601 (e.g. `"2026-05-17T19:30:00Z"`), or -/// a decimal string containing epoch milliseconds. -/// -/// On an unparseable string a serde error is returned (no silent default). -/// Shared across chat, email, and document canonicalisers. -/// -fn deserialize_flexible_timestamp<'de, D>(deserializer: D) -> Result, D::Error> -where - D: Deserializer<'de>, -{ - #[derive(Deserialize)] - #[serde(untagged)] - enum RawTs { - Millis(i64), - Text(String), - Null, - } - - fn epoch_millis(ms: i64) -> Result, E> { - // Contemporary epoch seconds are ten digits while epoch milliseconds - // are thirteen. Reject the ambiguous near-epoch range so a seconds - // value cannot silently poison ordering and staleness calculations. - const MIN_PLAUSIBLE_EPOCH_MILLIS: u64 = 100_000_000_000; - if ms.unsigned_abs() < MIN_PLAUSIBLE_EPOCH_MILLIS { - return Err(E::custom(format!( - "epoch-ms value {ms} is too small; pass milliseconds, not seconds" - ))); - } - chrono::TimeZone::timestamp_millis_opt(&Utc, ms) - .single() - .ok_or_else(|| E::custom(format!("invalid epoch-ms: {ms}"))) - } - - let raw = RawTs::deserialize(deserializer)?; - match raw { - RawTs::Null => Ok(Utc::now()), - RawTs::Millis(ms) => epoch_millis(ms), - RawTs::Text(s) => { - if let Ok(dt) = DateTime::parse_from_rfc3339(&s) { - return Ok(dt.with_timezone(&Utc)); - } - if let Ok(ms) = s.parse::() { - return epoch_millis(ms); - } - Err(serde::de::Error::custom(format!( - "cannot parse '{s}' as RFC 3339 or epoch-ms" - ))) - } - } -} - -#[cfg(test)] -#[path = "email_markdown_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/composio/email_markdown_tests.rs b/crates/tinymemory-sources/src/composio/email_markdown_tests.rs deleted file mode 100644 index 0ee088dc..00000000 --- a/crates/tinymemory-sources/src/composio/email_markdown_tests.rs +++ /dev/null @@ -1,220 +0,0 @@ -//! Tests for email thread markdown rendering. - -use super::*; - -/// The engine's canonicaliser emits this exact shape from its own copy of -/// this assembly, and the chunker splits on `---\nFrom:`. A failure here -/// is a coordinated format change, never a local edit. -#[test] -fn thread_markdown_format_is_pinned() { - let thread = EmailThread { - provider: "gmail".into(), - thread_subject: "Hello".into(), - messages: vec![EmailMessage { - from: "a@example.com".into(), - to: vec!["b@example.com".into()], - cc: Vec::new(), - subject: "Hello".into(), - sent_at: DateTime::parse_from_rfc3339("2026-01-02T03:04:05Z") - .unwrap() - .with_timezone(&Utc), - body: "Hi there".into(), - source_ref: Some("gmail:m1".into()), - list_unsubscribe: None, - }], - }; - assert_eq!( - thread_markdown(thread).unwrap(), - "---\nFrom: a@example.com\nTo: b@example.com\nSubject: Hello\nDate: 2026-01-02T03:04:05+00:00\n\nHi there\n\n" - ); -} - -#[test] -fn empty_thread_is_none_and_body_separators_are_escaped() { - assert!(thread_markdown(EmailThread { - provider: "gmail".into(), - thread_subject: String::new(), - messages: Vec::new(), - }) - .is_none()); - - let thread = EmailThread { - provider: "gmail".into(), - thread_subject: "s".into(), - messages: vec![EmailMessage { - from: "a".into(), - to: Vec::new(), - cc: Vec::new(), - subject: "s".into(), - sent_at: DateTime::parse_from_rfc3339("2026-01-02T03:04:05Z") - .unwrap() - .with_timezone(&Utc), - body: "x\n---\ny".into(), - source_ref: None, - list_unsubscribe: None, - }], - }; - let md = thread_markdown(thread).unwrap(); - assert!( - md.contains("\\---"), - "chunk separator must be escaped: {md}" - ); -} - -fn message_json(sent_at: serde_json::Value) -> serde_json::Value { - serde_json::json!({ - "from": "sender", - "subject": "subject", - "sent_at": sent_at, - "body": "body" - }) -} - -#[test] -fn flexible_timestamp_accepts_rfc3339_and_numeric_or_string_milliseconds() { - for value in [ - serde_json::json!("2026-01-02T03:04:05Z"), - serde_json::json!(1_767_323_045_000_i64), - serde_json::json!("1767323045000"), - ] { - let message: EmailMessage = serde_json::from_value(message_json(value)).unwrap(); - assert_eq!(message.sent_at.timestamp_millis(), 1_767_323_045_000); - } -} - -#[test] -fn flexible_timestamp_rejects_seconds_and_malformed_text() { - for value in [ - serde_json::json!(1_767_322_245_i64), - serde_json::json!("1767322245"), - serde_json::json!("last Tuesday"), - ] { - let error = serde_json::from_value::(message_json(value)).unwrap_err(); - assert!( - error.to_string().contains("milliseconds") - || error.to_string().contains("cannot parse"), - "unexpected error: {error}" - ); - } -} - -#[test] -fn rendering_sorts_oldest_first_and_escapes_header_markdown() { - let at = |timestamp: &str| { - DateTime::parse_from_rfc3339(timestamp) - .unwrap() - .with_timezone(&Utc) - }; - let thread = EmailThread { - provider: "gmail".into(), - thread_subject: "thread".into(), - messages: vec![ - EmailMessage { - from: "*new*".into(), - to: Vec::new(), - cc: Vec::new(), - subject: "[later]".into(), - sent_at: at("2026-02-01T00:00:00Z"), - body: "new".into(), - source_ref: None, - list_unsubscribe: None, - }, - EmailMessage { - from: "_old_".into(), - to: vec!["a|b".into()], - cc: vec!["c`d".into()], - subject: "# first".into(), - sent_at: at("2026-01-01T00:00:00Z"), - body: "old".into(), - source_ref: None, - list_unsubscribe: Some("".into()), - }, - ], - }; - - let markdown = thread_markdown(thread).unwrap(); - assert!(markdown.find("old").unwrap() < markdown.find("new").unwrap()); - assert!(markdown.contains("From: \\_old\\_")); - assert!(markdown.contains("To: a\\|b")); - assert!(markdown.contains("Cc: c\\`d")); - assert!(markdown.contains("Subject: # first")); - assert!(markdown.contains("List-Unsubscribe: ")); -} - -#[test] -fn message_serde_defaults_optional_fields_and_uses_epoch_milliseconds() { - let before = Utc::now().timestamp_millis(); - let message: EmailMessage = serde_json::from_value(serde_json::json!({ - "from": "sender", - "subject": "subject", - "sent_at": null, - "body": "body" - })) - .unwrap(); - let after = Utc::now().timestamp_millis(); - assert!(message.sent_at.timestamp_millis() >= before); - assert!(message.sent_at.timestamp_millis() <= after); - assert!(message.to.is_empty()); - assert!(message.cc.is_empty()); - assert!(message.source_ref.is_none()); - assert!(message.list_unsubscribe.is_none()); - - let serialized = serde_json::to_value(&message).unwrap(); - assert_eq!(serialized["sent_at"], message.sent_at.timestamp_millis()); -} - -#[test] -fn empty_cleaned_bodies_still_preserve_message_boundaries() { - let timestamp = DateTime::parse_from_rfc3339("2026-01-02T03:04:05Z") - .unwrap() - .with_timezone(&Utc); - let markdown = thread_markdown(EmailThread { - provider: "gmail".into(), - thread_subject: "empty body".into(), - messages: vec![EmailMessage { - from: "sender".into(), - to: Vec::new(), - cc: Vec::new(), - subject: "empty".into(), - sent_at: timestamp, - body: " \n\t".into(), - source_ref: None, - list_unsubscribe: None, - }], - }) - .unwrap(); - assert!(markdown.ends_with("\n\n\n"), "{markdown:?}"); -} - -#[test] -fn flexible_timestamp_rejects_out_of_range_milliseconds() { - let error = serde_json::from_value::(message_json(serde_json::json!(i64::MAX))) - .unwrap_err(); - assert!(error.to_string().contains("invalid epoch-ms"), "{error}"); -} - -#[test] -fn thread_shape_round_trips_without_losing_message_metadata() { - let raw = serde_json::json!({ - "provider": "imap", - "thread_subject": "subject", - "messages": [{ - "from": "a@example.com", - "to": ["b@example.com"], - "cc": ["c@example.com"], - "subject": "subject", - "sent_at": "2026-01-02T03:04:05Z", - "body": "body", - "source_ref": "imap:1", - "list_unsubscribe": "mailto:unsubscribe@example.com" - }] - }); - let thread: EmailThread = serde_json::from_value(raw).unwrap(); - let encoded = serde_json::to_value(&thread).unwrap(); - assert_eq!(encoded["provider"], "imap"); - assert_eq!(encoded["messages"][0]["source_ref"], "imap:1"); - assert_eq!( - encoded["messages"][0]["list_unsubscribe"], - "mailto:unsubscribe@example.com" - ); -} diff --git a/crates/tinymemory-sources/src/composio/helpers.rs b/crates/tinymemory-sources/src/composio/helpers.rs deleted file mode 100644 index 101239ed..00000000 --- a/crates/tinymemory-sources/src/composio/helpers.rs +++ /dev/null @@ -1,50 +0,0 @@ -//! Shared helpers for the provider normalisers in this module. - -/// Walk a JSON object using a list of dotted-path candidates and return the -/// first non-empty **string** match. -/// -/// # This is deliberately NOT `super::super::common::pick_str` -/// -/// The crate carries two `pick_str` functions with the same name and -/// genuinely different behaviour. Do not "deduplicate" them: -/// -/// | | this one (`normalize::helpers`) | `common::pick_str` | -/// |---|---|---| -/// | traversal | `Value::get` per `.`-separated segment — objects only | `Value::pointer` — also indexes into arrays | -/// | non-string leaf | rejected, returns `None` | `Number` is coerced via `to_string()` | -/// -/// The number case is the one that bites. A payload whose `id` is `42` -/// rather than `"42"` yields `None` here and `Some("42")` there, which -/// silently changes what a normaliser emits as a document id. The callers of -/// this function were written against the reject-non-strings behaviour and -/// have a test pinning it (`pick_str_rejects_non_string_values` below, and -/// the host-side mirror of it). -pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option { - for path in paths { - let mut cur = value; - let mut ok = true; - for segment in path.split('.') { - match cur.get(segment) { - Some(next) => cur = next, - None => { - ok = false; - break; - } - } - } - if !ok { - continue; - } - if let Some(s) = cur.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - None -} - -#[cfg(test)] -#[path = "helpers_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/composio/linear.rs b/crates/tinymemory-sources/src/composio/linear.rs deleted file mode 100644 index 843f98ab..00000000 --- a/crates/tinymemory-sources/src/composio/linear.rs +++ /dev/null @@ -1,157 +0,0 @@ -//! Linear host normalization helpers — result extraction, issue-title extraction, -//! viewer identity, cursor extraction, and time utilities. -//! -//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns -//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top -//! level or nested under `data`. The functions here walk the union of -//! common shapes so the provider does not have to branch per Composio -//! envelope variant. - -use serde_json::Value; - -use super::helpers::pick_str; - -/// Walk the Composio response envelope for Linear issue list results. -/// -/// Linear's list endpoints return `{ nodes: [...] }` or -/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the -/// upstream payload under `data` or `data.data`. We probe each shape -/// in order and return the first array we find. -pub fn extract_issues(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/nodes"), - data.pointer("/nodes"), - data.pointer("/data/issues/nodes"), - data.pointer("/issues/nodes"), - data.pointer("/data/data/nodes"), - data.pointer("/data/data/issues/nodes"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a Linear issue object. -/// -/// Linear issues store the name at `title` (or `data.title` after -/// Composio envelope wrapping). Falls back to `name` / `identifier` -/// so the chunk remains identifiable even for unusual response shapes. -pub fn extract_issue_title(issue: &Value) -> Option { - pick_str( - issue, - &[ - "title", - "data.title", - "name", - "data.name", - "identifier", - "data.identifier", - ], - ) -} - -/// Extract a stable cursor timestamp from a Linear issue object. -/// -/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep -/// the value as a string so lexicographic comparison against the stored -/// cursor is valid. -pub fn extract_issue_updated(issue: &Value) -> Option { - pick_str( - issue, - &[ - "updatedAt", - "data.updatedAt", - "updated_at", - "data.updated_at", - ], - ) -} - -/// Extract the viewer (authenticated user) object from a -/// `LINEAR_LIST_LINEAR_USERS { isMe: true }` response. -/// -/// Linear's GraphQL viewer endpoint returns `{ nodes: [{ id, email, … }] }`. -/// Composio may wrap this under `data` or `data.data`. We probe each -/// shape and return the first element of the nodes array, falling back -/// to the payload itself if it looks like a direct user object (has -/// `id` or `email`). -pub fn extract_viewer(data: &Value) -> Option { - let array_candidates = [ - data.pointer("/data/nodes"), - data.pointer("/nodes"), - data.pointer("/data/data/nodes"), - data.pointer("/data/users/nodes"), - ]; - for cand in array_candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - if let Some(first) = arr.first() { - return Some(first.clone()); - } - } - } - // Fallback: if the payload itself looks like a user object, return it. - if data.get("id").is_some() || data.get("email").is_some() { - return Some(data.clone()); - } - None -} - -/// Extract the viewer's ID string from a `LINEAR_LIST_LINEAR_USERS` -/// response. Returns `None` if the payload does not contain a -/// recognizable user ID. -pub fn extract_viewer_id(data: &Value) -> Option { - let viewer = extract_viewer(data)?; - pick_str(&viewer, &["id", "data.id"]) -} - -/// Extract a pagination cursor from a Linear connection `pageInfo` block. -/// -/// Returns `Some(endCursor)` only when `hasNextPage` is `true`; -/// `None` when the last page has been reached or when the envelope does -/// not carry `pageInfo` at all. -pub fn extract_pagination_cursor(data: &Value) -> Option { - // Mirrors the `extract_issues` envelope shapes, so every shape that can - // carry a node list can also carry its `pageInfo` cursor. - let page_info_candidates = [ - data.pointer("/data/pageInfo"), - data.pointer("/pageInfo"), - data.pointer("/data/data/pageInfo"), - data.pointer("/data/issues/pageInfo"), - data.pointer("/data/data/issues/pageInfo"), - ]; - for cand in page_info_candidates.into_iter().flatten() { - let has_next = cand - .get("hasNextPage") - .and_then(|v| v.as_bool()) - .unwrap_or(false); - if has_next { - if let Some(cursor) = cand.get("endCursor").and_then(|v| v.as_str()) { - let trimmed = cursor.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - } - None -} - -/// Current wall-clock time in milliseconds since the UNIX epoch. -pub fn now_ms() -> u64 { - use std::time::{SystemTime, UNIX_EPOCH}; - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0) -} - -#[cfg(test)] -#[path = "linear_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/raw_kind.rs b/crates/tinymemory-sources/src/raw_kind.rs deleted file mode 100644 index b58eba52..00000000 --- a/crates/tinymemory-sources/src/raw_kind.rs +++ /dev/null @@ -1,53 +0,0 @@ -//! Which sub-directory of the raw archive an item belongs in. -//! -//! Moved with the readers (#18 §B4) rather than imported, because importing it -//! would mean depending on the engine for one dependency-free enum — the exact -//! coupling this move removes. -//! -//! This is a deliberate second definition, and it is safe because it never -//! crosses a boundary: `raw_archive_coords` is the only consumer, nothing -//! outside this crate calls it, and the engine keeps its own copy for its -//! storage layer. If a caller ever needs to exchange one, that is the moment to -//! lift it into the contract instead. - -/// Category of a raw item, selecting the per-kind subdirectory under -/// `raw///`. -#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)] -pub enum RawKind { - /// Email messages (Gmail, Outlook, …). - Email, - /// Chat / DM messages (Slack, Telegram, WhatsApp, Discord, …). - Chat, - /// Standalone documents — Notion pages, Drive files, attachments. - Document, - /// One file per person reachable via this source. - Contact, - /// Long-form posts — LinkedIn posts, tweets, blog entries. - Post, - /// Git commits (one file per commit) — GitHub repo sources. - Commit, - /// Issues with their conversation + metadata — GitHub repo sources. - Issue, - /// Pull requests with their body + metadata — GitHub repo sources. - PullRequest, -} - -impl RawKind { - /// Directory name used on disk for this kind (plural). - pub const fn as_dir(&self) -> &'static str { - match self { - Self::Email => "emails", - Self::Chat => "chats", - Self::Document => "documents", - Self::Contact => "contacts", - Self::Post => "posts", - Self::Commit => "commits", - Self::Issue => "issues", - Self::PullRequest => "prs", - } - } -} - -#[cfg(test)] -#[path = "raw_kind_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/raw_kind_tests.rs b/crates/tinymemory-sources/src/raw_kind_tests.rs deleted file mode 100644 index f38708f0..00000000 --- a/crates/tinymemory-sources/src/raw_kind_tests.rs +++ /dev/null @@ -1,25 +0,0 @@ -//! Tests for stable raw archive directory names. - -use super::RawKind; - -#[test] -fn every_raw_kind_has_a_distinct_plural_directory() { - let cases = [ - (RawKind::Email, "emails"), - (RawKind::Chat, "chats"), - (RawKind::Document, "documents"), - (RawKind::Contact, "contacts"), - (RawKind::Post, "posts"), - (RawKind::Commit, "commits"), - (RawKind::Issue, "issues"), - (RawKind::PullRequest, "prs"), - ]; - let mut directories = std::collections::HashSet::new(); - for (kind, expected) in cases { - assert_eq!(kind.as_dir(), expected); - assert!( - directories.insert(kind.as_dir()), - "duplicate directory {expected}" - ); - } -} diff --git a/crates/tinymemory-sources/src/reconcile.rs b/crates/tinymemory-sources/src/reconcile.rs deleted file mode 100644 index e4998512..00000000 --- a/crates/tinymemory-sources/src/reconcile.rs +++ /dev/null @@ -1,90 +0,0 @@ -//! The pure half of Composio reconciliation: what a scanned connection looks -//! like as a registry row, and which caps a cap-less row gets on migration. -//! -//! Scanning live connections and persisting the registry are the host's (they -//! need its credentials, config file and write lock); the functions here are -//! the decisions, with no I/O, so they are unit-tested directly. - -use crate::registry::{ - apply_kind_defaults, memory_sync_defaults_for_toolkit, ComposioUpsertTarget, -}; -use crate::types::{MemorySourceEntry, SourceKind}; - -/// Build the `(toolkit, connection_id, label)` upsert target for one scanned -/// Composio connection. -/// -/// The label is a title-cased toolkit name plus the truncated connection id so -/// distinct accounts of the same toolkit (e.g. two Gmail logins) don't all show -/// as "Gmail connection". -pub fn composio_upsert_target(toolkit: &str, connection_id: &str) -> ComposioUpsertTarget { - let label = format!("{} · {}", title_case(toolkit), short_id(connection_id)); - (toolkit.to_string(), connection_id.to_string(), label) -} - -fn title_case(s: &str) -> String { - let mut chars = s.chars(); - match chars.next() { - None => String::new(), - Some(c) => c.to_uppercase().chain(chars).collect(), - } -} - -fn short_id(id: &str) -> &str { - // Show only the last 8 Unicode scalar values to keep labels compact. - // Byte-slicing would panic if the cut point isn't a UTF-8 boundary. - let n = id.chars().count(); - if n <= 8 { - return id; - } - let skip = n - 8; - let start = id.char_indices().nth(skip).map(|(idx, _)| idx).unwrap_or(0); - &id[start..] -} - -/// Apply conservative default caps in place to every cap-less source. -/// -/// For a Composio source with no `max_items` / `sync_depth_days`, writes the -/// per-toolkit defaults **and enables it** (a no-op when already enabled) — an -/// already-enabled, cap-less source would otherwise sync at the provider's -/// large internal ceiling instead of the cheap default, which is the cost this -/// migration exists to avoid. For other kinds it fills any unset kind-specific -/// caps through [`apply_kind_defaults`]. Caps the user has -/// customised (any non-`None` value) are never overwritten. -/// -/// Returns the number of Composio entries that received defaults. Pure (no -/// I/O) so it can be unit-tested directly. -pub fn apply_caps_defaults_to_entries(sources: &mut [MemorySourceEntry]) -> u32 { - let mut applied = 0u32; - for source in sources.iter_mut() { - match source.kind { - SourceKind::Composio => { - // Applies to enabled AND disabled cap-less sources; skips - // entries the user has already customised (any non-None cap). - if source.max_items.is_none() && source.sync_depth_days.is_none() { - let toolkit = source.toolkit.as_deref().unwrap_or(""); - let (max_items, sync_depth_days) = memory_sync_defaults_for_toolkit(toolkit); - log::debug!( - "[memory_sources:reconcile] caps migration: applying conservative defaults \ - id={} toolkit={toolkit} was_enabled={} max_items={max_items:?} \ - sync_depth_days={sync_depth_days:?}", - source.id, - source.enabled - ); - source.enabled = true; - source.max_items = max_items; - source.sync_depth_days = sync_depth_days; - applied += 1; - } - } - // Non-composio kinds get their kind defaults through the same - // helper the CRUD path uses, so one table of conservative values - // serves both. - _ => apply_kind_defaults(source), - } - } - applied -} - -#[cfg(test)] -#[path = "reconcile_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/reconcile_tests.rs b/crates/tinymemory-sources/src/reconcile_tests.rs deleted file mode 100644 index b824df83..00000000 --- a/crates/tinymemory-sources/src/reconcile_tests.rs +++ /dev/null @@ -1,164 +0,0 @@ -use super::*; - -#[test] -fn composio_upsert_target_formats_label_and_carries_ids_verbatim() { - let out = composio_upsert_target("gmail", "ca_WaktIDFlZwXO"); - // (toolkit, connection_id, label) - assert_eq!(out.0, "gmail"); - assert_eq!(out.1, "ca_WaktIDFlZwXO"); - assert_eq!(out.2, "Gmail · IDFlZwXO"); - let out = composio_upsert_target("slack", "short"); - assert_eq!(out.2, "Slack · short"); -} - -#[test] -fn title_case_handles_empty_and_non_ascii() { - assert_eq!(title_case(""), ""); - assert_eq!(title_case("éclair"), "Éclair"); -} - -#[test] -fn short_id_truncates_ascii() { - assert_eq!(short_id("ca_WaktIDFlZwXO"), "IDFlZwXO"); -} - -#[test] -fn short_id_short_input_passthrough() { - assert_eq!(short_id("abc"), "abc"); - assert_eq!(short_id("12345678"), "12345678"); -} - -#[test] -fn short_id_utf8_safe() { - // Multi-byte chars would have panicked with byte-slicing. - let s = "🦀🐢🐙🦊🐼🐰🐯🐸🦁"; - let out = short_id(s); - assert_eq!(out.chars().count(), 8); -} - -// ── Caps migration ────────────────────────────────────────────────────────── -// -// The transform came home with `apply_composio_source_caps_migration` (#5560), -// so its predicate is this crate's to pin. These exercise -// `apply_caps_defaults_to_entries` — the real production function, not a -// re-statement of it — because the migration's whole risk is in which entries -// it decides to touch: an over-eager pass overwrites a cap the user chose, and -// a shy one leaves an enabled, cap-less connector syncing at the provider's -// internal ceiling. - -fn composio_entry( - id: &str, - toolkit: &str, - enabled: bool, - max_items: Option, - sync_depth_days: Option, -) -> MemorySourceEntry { - MemorySourceEntry { - id: id.to_string(), - kind: SourceKind::Composio, - label: toolkit.to_string(), - enabled, - toolkit: Some(toolkit.to_string()), - connection_id: Some(format!("conn_{id}")), - path: None, - glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days, - } -} - -#[test] -fn migration_flips_disabled_capless_entry_to_enabled_with_caps() { - let mut sources = vec![composio_entry("s1", "gmail", false, None, None)]; - let count = apply_caps_defaults_to_entries(&mut sources); - assert_eq!(count, 1); - assert!(sources[0].enabled); - assert_eq!(sources[0].max_items, Some(100)); - assert_eq!(sources[0].sync_depth_days, Some(30)); -} - -#[test] -fn migration_applies_defaults_to_enabled_capless_entry() { - // An already-enabled but cap-less source must also receive defaults — - // otherwise its first sync runs at the provider's large internal ceiling. - let mut sources = vec![composio_entry("s2", "slack", true, None, None)]; - let count = apply_caps_defaults_to_entries(&mut sources); - assert_eq!(count, 1); - assert!(sources[0].enabled); - assert_eq!(sources[0].max_items, Some(50)); - assert_eq!(sources[0].sync_depth_days, Some(14)); -} - -#[test] -fn migration_leaves_user_customised_caps_untouched() { - // The user set max_items explicitly, so the migration must not override it - // — and must not flip `enabled` on its way past either. - let mut sources = vec![composio_entry("s3", "notion", false, Some(5), None)]; - let count = apply_caps_defaults_to_entries(&mut sources); - assert_eq!(count, 0, "entry with user-set caps must not be migrated"); - assert!(!sources[0].enabled, "enabled must not be flipped"); - assert_eq!(sources[0].max_items, Some(5), "user cap must be preserved"); -} - -#[test] -fn migration_is_noop_on_empty_list() { - let mut sources: Vec = vec![]; - assert_eq!(apply_caps_defaults_to_entries(&mut sources), 0); -} - -#[test] -fn migration_applies_correct_defaults_per_toolkit() { - let toolkits = [ - ("gmail", Some(100u32), Some(30u32)), - ("slack", Some(50), Some(14)), - ("notion", Some(30), Some(30)), - ("linear", Some(50), Some(30)), - ("clickup", Some(50), Some(30)), - ("github", Some(50), Some(30)), - ("unknown", Some(30), Some(14)), - ]; - for (toolkit, exp_items, exp_days) in &toolkits { - let mut sources = vec![composio_entry("sid", toolkit, false, None, None)]; - apply_caps_defaults_to_entries(&mut sources); - assert_eq!( - sources[0].max_items, *exp_items, - "max_items mismatch for toolkit={toolkit}" - ); - assert_eq!( - sources[0].sync_depth_days, *exp_days, - "sync_depth_days mismatch for toolkit={toolkit}" - ); - } -} - -/// The non-Composio arm goes through `apply_kind_defaults`, which fills the -/// kind's own caps and — unlike the Composio arm — never enables anything and -/// never counts towards the returned tally. -#[test] -fn migration_fills_kind_defaults_without_counting_them() { - let mut repo = composio_entry("s4", "github", false, None, None); - repo.kind = SourceKind::GithubRepo; - repo.toolkit = None; - repo.connection_id = None; - repo.url = Some("https://github.com/tinyhumansai/openhuman".to_string()); - - let mut sources = vec![repo]; - let count = apply_caps_defaults_to_entries(&mut sources); - - assert_eq!(count, 0, "only composio entries count as migrated"); - assert!( - !sources[0].enabled, - "kind defaults must not enable a source" - ); - assert_eq!(sources[0].max_prs, Some(10)); - assert!(sources[0].max_issues.is_some()); -} diff --git a/crates/tinymemory-sources/src/registry.rs b/crates/tinymemory-sources/src/registry.rs deleted file mode 100644 index d7bd6ca9..00000000 --- a/crates/tinymemory-sources/src/registry.rs +++ /dev/null @@ -1,500 +0,0 @@ -//! The configured-source registry. -//! -//! Sources are persisted as `[[memory_sources]]` entries in a TOML config file -//! (typically `config.toml`). In OpenHuman this lived on a large shared `Config` -//! struct loaded through an async RPC; this crate does not own that global -//! config, so the registry here is a small self-contained reader/writer over a -//! single TOML file. Other top-level keys in the file are preserved across -//! writes — only the `memory_sources` array is rewritten. -//! -//! Every mutation follows the spec's atomic load-modify-validate-save cycle: -//! load the current file, apply the change in memory, validate, and persist. -//! Each on-disk write (`SourceRegistry::atomic_write`) is atomic (temp file + -//! rename), so a crash mid-write cannot leave a truncated `config.toml`. -//! -//! Because that rename replaces a file the *host* also writes, this registry -//! owes the host its permission contract as well as its contents: the temp file -//! is created owner-only, so a source mutation cannot hand back a config that is -//! more permissive than the one it replaced. See `create_owner_only`, which is -//! private, so this is a plain reference rather than an intra-doc link. -//! -//! The complete load-modify-save cycle is guarded by a process-wide mutation -//! lock, so separate [`SourceRegistry`] handles cannot overwrite one another's -//! in-process updates. Atomic rename protects each individual disk write. - -use std::path::{Path, PathBuf}; -use std::sync::{LazyLock, Mutex}; - -use crate::error::{Error, Result}; - -use super::types::{MemorySourceEntry, MemorySourcePatch, SourceKind}; - -/// Wrap a registry I/O or codec failure, naming what was being done. -fn registry_error(action: impl std::fmt::Display, error: impl std::fmt::Display) -> Error { - Error::Registry(format!("{action}: {error}")) -} - -/// Serializes each registry load-modify-save transaction in this process. -/// -/// A single lock deliberately covers every path: registry mutation is rare, -/// and correctness is more important than allowing unrelated config files to -/// race through their atomic renames. The on-disk rename remains the crash- -/// safety boundary; this mutex closes the in-process lost-update window. -static REGISTRY_MUTATION_LOCK: LazyLock> = LazyLock::new(|| Mutex::new(())); - -fn mutation_guard() -> std::sync::MutexGuard<'static, ()> { - // A poisoned lock means a previous mutation panicked while holding it. The - // guard's data is `()`, so there is nothing torn to inherit -- recover and - // continue rather than cascading the panic into every later mutation. - REGISTRY_MUTATION_LOCK - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) -} - -/// Conservative default sync caps for a Composio toolkit, keyed by toolkit slug. -/// -/// Single source of truth for the cheap out-of-the-box sync volume. Applied to a -/// source entry when it is first registered. Never overwrites a user-customised -/// cap. Returns `(max_items, sync_depth_days)`. -#[must_use] -pub fn memory_sync_defaults_for_toolkit(toolkit: &str) -> (Option, Option) { - match toolkit { - "gmail" => (Some(100), Some(30)), - "slack" => (Some(50), Some(14)), - "notion" => (Some(30), Some(30)), - "linear" => (Some(50), Some(30)), - "clickup" => (Some(50), Some(30)), - "github" => (Some(50), Some(30)), - // Generic fallback for any toolkit not listed above. - _ => (Some(30), Some(14)), - } -} - -/// Apply conservative per-kind cap defaults to a new source entry. -/// -/// Only fills fields that are still `None` — never overwrites a caller-supplied -/// value, so re-running it over an entry a user has already tuned is a no-op. -/// The retroactive Composio migration applies the same reasoning through -/// [`memory_sync_defaults_for_toolkit`], which is why the two live together: -/// creation time and migration time must agree, and they only do if the policy -/// has one address. -/// -/// Folder, web-page and Composio kinds have nothing to fill here. Composio caps -/// are set at upsert time from the toolkit slug, which this function does not -/// have. -pub fn apply_kind_defaults(entry: &mut MemorySourceEntry) { - match entry.kind { - SourceKind::GithubRepo => { - if entry.max_prs.is_none() { - entry.max_prs = Some(10); - } - if entry.max_issues.is_none() { - entry.max_issues = Some(10); - } - if entry.max_commits.is_none() { - entry.max_commits = Some(50); - } - } - SourceKind::RssFeed if entry.max_items.is_none() => { - entry.max_items = Some(20); - } - _ => {} - } -} - -/// A registry of [`MemorySourceEntry`] values backed by a TOML config file. -/// -/// Construct one with [`SourceRegistry::new`], pointing at the `config.toml` -/// path. The file need not exist yet — reads return an empty list and the first -/// write creates it (and any missing parent directories). -#[derive(Debug, Clone)] -pub struct SourceRegistry { - path: PathBuf, -} - -/// Create `path` for writing, restricted to the owner on platforms that have -/// file modes, and refusing to reuse anything already at that path. -/// -/// The mode is part of the `open(2)` call rather than a `chmod` afterwards, so -/// the file is never even momentarily group- or world-readable. That matters -/// because [`SourceRegistry::atomic_write`] renames this temp file over the -/// host's `config.toml`, and a rename carries the *source* file's mode onto the -/// destination: a temp file created at the default `0o666 & ~umask` (0644 under -/// the usual 022) silently re-widens the live config on every source mutation, -/// undoing any hardening the host applied when it wrote that file itself. -/// -/// `create_new` is deliberate too. The caller already names the temp file with a -/// fresh UUID, so a collision means something else put a file — or a symlink — -/// where this one was about to go, and failing is the safe answer. -/// -/// Note that `mode` is masked by the process umask, so a pathological umask can -/// make the result *narrower* than `0o600`. That is not a weakening, and it is -/// the same property every other umask-respecting create in the tree has. -fn create_owner_only(path: &Path) -> std::io::Result { - let mut options = std::fs::OpenOptions::new(); - options.write(true).create_new(true); - #[cfg(unix)] - { - use std::os::unix::fs::OpenOptionsExt; - options.mode(0o600); - } - options.open(path) -} - -impl SourceRegistry { - /// Create a registry persisted at `config_path`. - #[must_use] - pub fn new(config_path: impl Into) -> Self { - Self { - path: config_path.into(), - } - } - - /// The config file path this registry reads and writes. - #[must_use] - pub fn path(&self) -> &Path { - &self.path - } - - /// Read the whole config file into a TOML table (empty if it doesn't exist). - fn read_table(&self) -> Result { - if !self.path.exists() { - return Ok(toml::Table::new()); - } - let text = std::fs::read_to_string(&self.path) - .map_err(|e| registry_error(format!("failed to read {}", self.path.display()), e))?; - let table: toml::Table = toml::from_str(&text) - .map_err(|e| registry_error(format!("failed to parse {}", self.path.display()), e))?; - Ok(table) - } - - /// List all configured sources. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or decoded. - pub fn list(&self) -> Result> { - let table = self.read_table()?; - match table.get("memory_sources") { - Some(value) => value - .clone() - .try_into() - .map_err(|e| registry_error("failed to decode [[memory_sources]]", e)), - None => Ok(Vec::new()), - } - } - - /// List enabled sources of a given [`SourceKind`]. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or decoded. - pub fn list_enabled_by_kind(&self, kind: SourceKind) -> Result> { - Ok(self - .list()? - .into_iter() - .filter(|s| s.kind == kind && s.enabled) - .collect()) - } - - /// Get a single source by id, if present. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or decoded. - pub fn get(&self, id: &str) -> Result> { - Ok(self.list()?.into_iter().find(|s| s.id == id)) - } - - /// Persist the full source list, preserving any other top-level config - /// keys. - /// - /// Writes are atomic: the new TOML is written to a same-directory temp file - /// and then renamed over the config. This keeps a failed/crashed write from - /// leaving a truncated `config.toml`, matching the OpenHuman source - /// registry contract. The temp file is created owner-only so the rename - /// cannot widen the live config — see [`create_owner_only`]. - /// - /// Mutation callers hold [`REGISTRY_MUTATION_LOCK`] across their initial - /// read and this preserving re-read, keeping the two snapshots ordered with - /// respect to every other in-process writer. - fn write_all(&self, entries: &[MemorySourceEntry]) -> Result<()> { - let mut table = self.read_table()?; - let value = toml::Value::try_from(entries) - .map_err(|e| registry_error("failed to encode memory_sources", e))?; - table.insert("memory_sources".to_string(), value); - let text = toml::to_string_pretty(&table) - .map_err(|e| registry_error("failed to serialize config", e))?; - if let Some(parent) = self.path.parent() { - if !parent.as_os_str().is_empty() { - std::fs::create_dir_all(parent).map_err(|e| { - registry_error(format!("failed to create {}", parent.display()), e) - })?; - } - } - self.atomic_write(text.as_bytes())?; - Ok(()) - } - - fn atomic_write(&self, bytes: &[u8]) -> Result<()> { - let parent = self - .path - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or_else(|| Path::new(".")); - let filename = self - .path - .file_name() - .and_then(|n| n.to_str()) - .ok_or_else(|| { - Error::Registry(format!( - "config path has no file name: {}", - self.path.display() - )) - })?; - let tmp_path = parent.join(format!( - ".{filename}.tmp-{}", - uuid::Uuid::new_v4().as_simple() - )); - - let write_result = (|| -> Result<()> { - { - let mut file = create_owner_only(&tmp_path).map_err(|e| { - registry_error(format!("failed to create {}", tmp_path.display()), e) - })?; - use std::io::Write; - file.write_all(bytes).map_err(|e| { - registry_error(format!("failed to write {}", tmp_path.display()), e) - })?; - file.sync_all().map_err(|e| { - registry_error(format!("failed to sync {}", tmp_path.display()), e) - })?; - } - std::fs::rename(&tmp_path, &self.path).map_err(|e| { - registry_error( - format!( - "failed to atomically replace {} with {}", - self.path.display(), - tmp_path.display() - ), - e, - ) - })?; - Ok(()) - })(); - - if write_result.is_err() { - let _ = std::fs::remove_file(&tmp_path); - } - write_result - } - - /// Validate and add a new source. Fails if the id already exists. - /// - /// # Errors - /// - /// [`Error::Invalid`] for an entry that fails validation or reuses an id, - /// [`Error::Registry`] when the file cannot be read or written. - pub fn add(&self, entry: MemorySourceEntry) -> Result { - let _guard = mutation_guard(); - entry.validate()?; - let mut sources = self.list()?; - if sources.iter().any(|s| s.id == entry.id) { - return Err(Error::Invalid(format!( - "source with id '{}' already exists", - entry.id - ))); - } - sources.push(entry.clone()); - self.write_all(&sources)?; - Ok(entry) - } - - /// Apply a [`MemorySourcePatch`] to an existing source, then re-validate and - /// save. Fails if no source has the given id. - /// - /// # Errors - /// - /// [`Error::NotFound`] for an unknown id, [`Error::Invalid`] for a patch - /// field the kind does not use or a result that fails validation, - /// [`Error::Registry`] when the file cannot be read or written. - pub fn update(&self, id: &str, patch: MemorySourcePatch) -> Result { - let _guard = mutation_guard(); - let mut sources = self.list()?; - let entry = sources - .iter_mut() - .find(|s| s.id == id) - .ok_or_else(|| Error::NotFound(format!("source '{id}' not found")))?; - - patch.validate_for_kind(entry.kind.clone())?; - patch.apply_to(entry); - entry.validate()?; - let updated = entry.clone(); - self.write_all(&sources)?; - Ok(updated) - } - - /// Remove a source by id. Returns `true` if an entry was removed. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or written. - pub fn remove(&self, id: &str) -> Result { - let _guard = mutation_guard(); - let mut sources = self.list()?; - let before = sources.len(); - sources.retain(|s| s.id != id); - let removed = sources.len() < before; - if removed { - self.write_all(&sources)?; - } - Ok(removed) - } - - /// Remove every composio source bound to `connection_id`. Returns the count - /// removed. Mirrors [`SourceRegistry::upsert_composio_source`], which keys - /// composio sources on `connection_id` rather than the `src_*` id. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or written. - pub fn remove_composio_source_by_connection_id(&self, connection_id: &str) -> Result { - let _guard = mutation_guard(); - let mut sources = self.list()?; - let before = sources.len(); - sources.retain(|s| { - !(s.kind == SourceKind::Composio && s.connection_id.as_deref() == Some(connection_id)) - }); - let removed = before - sources.len(); - if removed > 0 { - self.write_all(&sources)?; - } - Ok(removed) - } - - /// Upsert a composio source keyed on `connection_id`. - /// - /// If a source with the same `connection_id` exists, its label is updated; - /// otherwise a new entry is inserted with conservative per-toolkit caps. The - /// update path never clobbers user-customised caps. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or written. - pub fn upsert_composio_source( - &self, - toolkit: &str, - connection_id: &str, - label: &str, - ) -> Result { - let _guard = mutation_guard(); - let mut sources = self.list()?; - let (entry, _was_insert) = - upsert_composio_entry_in_place(&mut sources, toolkit, connection_id, label); - self.write_all(&sources)?; - Ok(entry) - } - - /// Batch-upsert Composio sources with one load and one atomic save. - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or written. - pub fn upsert_composio_sources_batch(&self, targets: &[ComposioUpsertTarget]) -> Result { - if targets.is_empty() { - return Ok(0); - } - let _guard = mutation_guard(); - let mut sources = self.list()?; - for (toolkit, connection_id, label) in targets { - upsert_composio_entry_in_place(&mut sources, toolkit, connection_id, label); - } - self.write_all(&sources)?; - Ok(targets.len().min(u32::MAX as usize) as u32) - } - - /// Replace the whole registry with `entries`, validating each first. - /// - /// The write-through behind a host-config view whose `memory_sources_json` - /// reads this file: a setter that only updated an in-memory snapshot would - /// be invisible to the very next getter (openhuman#5820). Same atomic - /// load-modify-validate-save cycle as the other mutations, so other - /// top-level keys in the file are preserved. - /// - /// # Errors - /// - /// [`Error::Invalid`] when an entry fails validation, [`Error::Registry`] - /// when the file cannot be read, parsed, serialized or atomically replaced. - pub fn replace_all(&self, entries: &[MemorySourceEntry]) -> Result<()> { - let _guard = mutation_guard(); - for entry in entries { - entry.validate().map_err(|reason| { - Error::Invalid(format!("invalid memory source `{}`: {reason}", entry.id)) - })?; - } - self.write_all(entries) - } - - /// Enable every source and clear all per-source caps ("All In" mode). - /// - /// # Errors - /// - /// [`Error::Registry`] when the file cannot be read or written. - pub fn apply_all_in(&self) -> Result> { - let _guard = mutation_guard(); - let mut sources = self.list()?; - for source in &mut sources { - source.enabled = true; - source.max_items = None; - source.sync_depth_days = None; - source.max_commits = None; - source.max_issues = None; - source.max_prs = None; - source.max_tokens_per_sync = None; - source.max_cost_per_sync_usd = None; - } - self.write_all(&sources)?; - Ok(sources) - } -} - -/// `(toolkit, account_id, label)` — the three fields that identify which -/// Composio account a source upserts into. -pub type ComposioUpsertTarget = (String, String, String); - -/// Apply a single composio upsert to an in-memory source list. -/// -/// Pure (no I/O) so the registry path and unit tests share one find-or-push -/// predicate. Returns the resulting entry and whether it was a fresh insert. -pub(crate) fn upsert_composio_entry_in_place( - sources: &mut Vec, - toolkit: &str, - connection_id: &str, - label: &str, -) -> (MemorySourceEntry, bool) { - if let Some(existing) = sources.iter_mut().find(|s| { - s.kind == SourceKind::Composio && s.connection_id.as_deref() == Some(connection_id) - }) { - existing.label = label.to_string(); - return (existing.clone(), false); - } - - let (default_max_items, default_sync_depth_days) = memory_sync_defaults_for_toolkit(toolkit); - let entry = MemorySourceEntry { - toolkit: Some(toolkit.to_string()), - connection_id: Some(connection_id.to_string()), - max_items: default_max_items, - sync_depth_days: default_sync_depth_days, - ..MemorySourceEntry::new( - format!("src_{}", uuid::Uuid::new_v4().as_simple()), - SourceKind::Composio, - label, - ) - }; - sources.push(entry.clone()); - (entry, true) -} - -#[cfg(test)] -#[path = "registry_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/registry_tests.rs b/crates/tinymemory-sources/src/registry_tests.rs deleted file mode 100644 index 643663a2..00000000 --- a/crates/tinymemory-sources/src/registry_tests.rs +++ /dev/null @@ -1,498 +0,0 @@ -//! Tests for the TOML-backed source registry. - -use super::*; -use crate::types::SourceKind; -use tempfile::TempDir; - -fn registry() -> (TempDir, SourceRegistry) { - let tmp = TempDir::new().unwrap(); - let reg = SourceRegistry::new(tmp.path().join("config.toml")); - (tmp, reg) -} - -fn folder_entry(id: &str) -> MemorySourceEntry { - let mut e = MemorySourceEntry { - id: id.into(), - kind: SourceKind::Folder, - label: "Notes".into(), - enabled: true, - toolkit: None, - connection_id: None, - path: Some("/tmp/notes".into()), - glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - }; - e.glob = Some("**/*.md".into()); - e -} - -#[test] -fn list_is_empty_for_missing_file() { - let (_tmp, reg) = registry(); - assert!(reg.list().unwrap().is_empty()); - assert!(reg.get("anything").unwrap().is_none()); -} - -#[test] -fn add_get_list_round_trip() { - let (_tmp, reg) = registry(); - let added = reg.add(folder_entry("src_1")).unwrap(); - assert_eq!(added.id, "src_1"); - - let got = reg.get("src_1").unwrap().unwrap(); - assert_eq!(got.kind, SourceKind::Folder); - assert_eq!(got.path.as_deref(), Some("/tmp/notes")); - - let all = reg.list().unwrap(); - assert_eq!(all.len(), 1); -} - -#[test] -fn add_rejects_duplicate_id() { - let (_tmp, reg) = registry(); - reg.add(folder_entry("src_dup")).unwrap(); - assert!(reg.add(folder_entry("src_dup")).is_err()); -} - -#[test] -fn add_rejects_invalid_entry() { - let (_tmp, reg) = registry(); - let mut bad = folder_entry("src_bad"); - bad.path = None; // folder requires a path - assert!(reg.add(bad).is_err()); - assert!(reg.list().unwrap().is_empty()); -} - -#[test] -fn concurrent_registry_adds_preserve_every_source() { - let (_tmp, reg) = registry(); - let writers = 24; - let barrier = std::sync::Arc::new(std::sync::Barrier::new(writers)); - let mut threads = Vec::new(); - for index in 0..writers { - let reg = reg.clone(); - let barrier = barrier.clone(); - threads.push(std::thread::spawn(move || { - barrier.wait(); - reg.add(folder_entry(&format!("src_concurrent_{index}"))) - .unwrap(); - })); - } - for thread in threads { - thread.join().unwrap(); - } - - let mut ids: Vec<_> = reg - .list() - .unwrap() - .into_iter() - .map(|entry| entry.id) - .collect(); - ids.sort(); - assert_eq!(ids.len(), writers); - for index in 0..writers { - assert!(ids.contains(&format!("src_concurrent_{index}"))); - } -} - -#[test] -fn update_applies_patch_and_persists() { - let (_tmp, reg) = registry(); - reg.add(folder_entry("src_u")).unwrap(); - - let patch = MemorySourcePatch { - label: Some("Renamed".into()), - enabled: Some(false), - ..Default::default() - }; - let updated = reg.update("src_u", patch).unwrap(); - assert_eq!(updated.label, "Renamed"); - assert!(!updated.enabled); - - // Re-read from disk to confirm persistence. - let got = reg.get("src_u").unwrap().unwrap(); - assert_eq!(got.label, "Renamed"); - assert!(!got.enabled); -} - -#[test] -fn update_missing_id_errors() { - let (_tmp, reg) = registry(); - assert!(reg.update("nope", MemorySourcePatch::default()).is_err()); -} - -#[test] -fn remove_returns_whether_anything_was_removed() { - let (_tmp, reg) = registry(); - reg.add(folder_entry("src_r")).unwrap(); - assert!(reg.remove("src_r").unwrap()); - assert!(!reg.remove("src_r").unwrap()); - assert!(reg.list().unwrap().is_empty()); -} - -#[test] -fn list_enabled_by_kind_filters() { - let (_tmp, reg) = registry(); - reg.add(folder_entry("src_a")).unwrap(); - let mut disabled = folder_entry("src_b"); - disabled.enabled = false; - reg.add(disabled).unwrap(); - - let enabled = reg.list_enabled_by_kind(SourceKind::Folder).unwrap(); - assert_eq!(enabled.len(), 1); - assert_eq!(enabled[0].id, "src_a"); - assert!(reg - .list_enabled_by_kind(SourceKind::Conversation) - .unwrap() - .is_empty()); -} - -#[test] -fn write_preserves_other_top_level_config_keys() { - let (tmp, reg) = registry(); - let path = tmp.path().join("config.toml"); - std::fs::write(&path, "workspace = \"/data\"\n").unwrap(); - - reg.add(folder_entry("src_keep")).unwrap(); - - let text = std::fs::read_to_string(&path).unwrap(); - assert!(text.contains("workspace = \"/data\"")); - assert!(text.contains("[[memory_sources]]")); -} - -#[test] -fn write_uses_atomic_temp_file_without_leaving_stale_temp() { - let (tmp, reg) = registry(); - reg.add(folder_entry("src_atomic")).unwrap(); - - let text = std::fs::read_to_string(reg.path()).unwrap(); - assert!(text.contains("src_atomic")); - - let stale_temp_files: Vec<_> = std::fs::read_dir(tmp.path()) - .unwrap() - .filter_map(std::result::Result::ok) - .filter(|entry| { - entry - .file_name() - .to_string_lossy() - .starts_with(".config.toml.tmp-") - }) - .collect(); - assert!(stale_temp_files.is_empty()); -} - -// ── Composio upsert ── - -#[test] -fn composio_defaults_for_known_and_unknown_toolkits() { - assert_eq!( - memory_sync_defaults_for_toolkit("gmail"), - (Some(100), Some(30)) - ); - assert_eq!( - memory_sync_defaults_for_toolkit("slack"), - (Some(50), Some(14)) - ); - assert_eq!( - memory_sync_defaults_for_toolkit("unknown_xyz"), - (Some(30), Some(14)) - ); -} - -#[test] -fn in_place_upsert_inserts_then_updates_label_only() { - let mut sources: Vec = vec![]; - let (entry, was_insert) = - upsert_composio_entry_in_place(&mut sources, "gmail", "conn_a", "Gmail · conn_a"); - assert!(was_insert); - assert_eq!(entry.toolkit.as_deref(), Some("gmail")); - assert_eq!(entry.max_items, Some(100)); - assert_eq!(entry.sync_depth_days, Some(30)); - - // User customises a cap, then a second upsert updates label only. - sources[0].max_items = Some(7); - let (entry, was_insert) = - upsert_composio_entry_in_place(&mut sources, "gmail", "conn_a", "new label"); - assert!(!was_insert); - assert_eq!(sources.len(), 1); - assert_eq!(entry.label, "new label"); - assert_eq!(entry.max_items, Some(7)); -} - -#[test] -fn upsert_composio_source_persists_and_disconnect_removes() { - let (_tmp, reg) = registry(); - reg.upsert_composio_source("gmail", "conn_a", "Gmail") - .unwrap(); - reg.upsert_composio_source("slack", "conn_b", "Slack") - .unwrap(); - assert_eq!(reg.list().unwrap().len(), 2); - - let removed = reg - .remove_composio_source_by_connection_id("conn_a") - .unwrap(); - assert_eq!(removed, 1); - assert_eq!(reg.list().unwrap().len(), 1); -} - -#[test] -fn apply_all_in_enables_and_clears_caps() { - let (_tmp, reg) = registry(); - let mut capped = folder_entry("src_capped"); - capped.enabled = false; - capped.max_items = Some(5); - capped.sync_depth_days = Some(3); - reg.add(capped).unwrap(); - - let updated = reg.apply_all_in().unwrap(); - assert_eq!(updated.len(), 1); - assert!(updated[0].enabled); - assert!(updated[0].max_items.is_none()); - assert!(updated[0].sync_depth_days.is_none()); -} - -#[test] -fn memory_source_patch_deserializes_partial_and_github_fields() { - let json = serde_json::json!({ - "label": "New label", - "enabled": false, - "max_commits": 100, - "max_issues": 50, - "max_prs": 25 - }); - let patch: MemorySourcePatch = serde_json::from_value(json).unwrap(); - assert_eq!(patch.label.as_deref(), Some("New label")); - assert_eq!(patch.enabled, Some(false)); - assert_eq!(patch.max_commits, Some(Some(100))); - assert_eq!(patch.max_issues, Some(Some(50))); - assert_eq!(patch.max_prs, Some(Some(25))); - assert!(patch.toolkit.is_none()); -} - -#[test] -fn memory_source_patch_can_clear_optional_fields_with_null() { - let (_tmp, reg) = registry(); - let mut entry = folder_entry("src_clear"); - entry.glob = Some("**/*.md".into()); - entry.max_items = Some(10); - reg.add(entry).unwrap(); - - let patch: MemorySourcePatch = serde_json::from_value(serde_json::json!({ - "glob": null, - "max_items": null - })) - .unwrap(); - let updated = reg.update("src_clear", patch).unwrap(); - assert!(updated.glob.is_none()); - assert!(updated.max_items.is_none()); -} - -#[test] -fn update_rejects_fields_that_do_not_apply_to_source_kind() { - let (_tmp, reg) = registry(); - reg.add(folder_entry("src_kind")).unwrap(); - let patch: MemorySourcePatch = serde_json::from_value(serde_json::json!({ - "url": "https://example.com/repo" - })) - .unwrap(); - assert!(reg.update("src_kind", patch).is_err()); -} - -// ── File permissions ──────────────────────────────────────────────────────── -// -// `atomic_write` renames its temp file over the host's `config.toml`, so the -// temp file's mode becomes the live config's mode. Created with plain -// `File::create` that was `0o666 & ~umask` — 0644 under the usual 022 — which -// silently re-widened a config the host had deliberately written owner-only, -// on every single source mutation. These tests pin the mode of the file this -// registry leaves behind, not the mode it was handed. - -/// The mode bits of `path`, or `None` on a platform without file modes. -#[cfg(unix)] -fn mode_of(path: &std::path::Path) -> u32 { - use std::os::unix::fs::PermissionsExt; - std::fs::metadata(path).unwrap().permissions().mode() & 0o777 -} - -#[cfg(unix)] -#[test] -fn first_write_creates_an_owner_only_config() { - let (tmp, reg) = registry(); - let path = tmp.path().join("config.toml"); - assert!( - !path.exists(), - "precondition: the config does not exist yet" - ); - - reg.add(folder_entry("src_1")).unwrap(); - - let mode = mode_of(&path); - assert_eq!( - mode & 0o077, - 0, - "a freshly created config.toml must not be group- or world-accessible, got {mode:o}" - ); -} - -#[cfg(unix)] -#[test] -fn a_mutation_does_not_widen_an_owner_only_config() { - use std::os::unix::fs::PermissionsExt; - - let (tmp, reg) = registry(); - let path = tmp.path().join("config.toml"); - - // Stand in for a host that wrote the config itself and hardened it, which - // is exactly what OpenHuman's `Config::save` does. - std::fs::write(&path, "[some_other_section]\nkept = true\n").unwrap(); - std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600)).unwrap(); - - reg.add(folder_entry("src_1")).unwrap(); - - let mode = mode_of(&path); - assert_eq!( - mode & 0o077, - 0, - "a source mutation must not re-widen a config the host hardened, got {mode:o}" - ); - // The whole point of the preserving re-read: unrelated keys survive. - let text = std::fs::read_to_string(&path).unwrap(); - assert!( - text.contains("[some_other_section]"), - "unrelated config sections must survive the rewrite" - ); -} - -#[cfg(unix)] -#[test] -fn a_mutation_narrows_a_config_that_was_already_world_readable() { - use std::os::unix::fs::PermissionsExt; - - let (tmp, reg) = registry(); - let path = tmp.path().join("config.toml"); - - // A config left at 0644 by an older build. The mode under test belongs to - // the temp file, not to this one, so the pre-existing width must not be - // inherited through the rename. - std::fs::write(&path, "[some_other_section]\nkept = true\n").unwrap(); - std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o644)).unwrap(); - assert_eq!( - mode_of(&path) & 0o077, - 0o044, - "precondition: starts at 0644" - ); - - reg.add(folder_entry("src_1")).unwrap(); - - let mode = mode_of(&path); - assert_eq!( - mode & 0o077, - 0, - "the rename must not carry the old file's 0644 onto the new one, got {mode:o}" - ); -} - -#[cfg(unix)] -#[test] -fn every_mutation_path_leaves_the_config_owner_only() { - use std::os::unix::fs::PermissionsExt; - - let (tmp, reg) = registry(); - let path = tmp.path().join("config.toml"); - - reg.add(folder_entry("src_1")).unwrap(); - reg.add(folder_entry("src_2")).unwrap(); - - // Re-widen between mutations so each assertion is about the write that - // follows it rather than about a mode set once at creation. - let widen = |p: &std::path::Path| { - std::fs::set_permissions(p, std::fs::Permissions::from_mode(0o644)).unwrap() - }; - - widen(&path); - reg.update( - "src_1", - MemorySourcePatch { - enabled: Some(false), - ..Default::default() - }, - ) - .unwrap(); - assert_eq!(mode_of(&path) & 0o077, 0, "update() widened the config"); - - widen(&path); - assert!(reg.remove("src_2").unwrap()); - assert_eq!(mode_of(&path) & 0o077, 0, "remove() widened the config"); -} - -// ── apply_kind_defaults ───────────────────────────────────────────────────── -// -// Moved here from the engine crate in #5560 so a host can fill a new entry's -// caps without linking the engine. The defaults are the ones the retroactive -// Composio caps migration also applies, so any change here is a change to what -// already-registered sources are reconciled against. - -fn entry_of_kind(kind: SourceKind) -> MemorySourceEntry { - let mut entry = folder_entry("defaults"); - entry.kind = kind; - entry -} - -#[test] -fn github_defaults_fill_only_the_caps_left_unset() { - let mut entry = entry_of_kind(SourceKind::GithubRepo); - entry.max_issues = Some(3); - apply_kind_defaults(&mut entry); - assert_eq!(entry.max_prs, Some(10)); - assert_eq!(entry.max_issues, Some(3), "a user-set cap must survive"); - assert_eq!(entry.max_commits, Some(50)); -} - -#[test] -fn an_rss_feed_gets_an_item_cap() { - let mut entry = entry_of_kind(SourceKind::RssFeed); - apply_kind_defaults(&mut entry); - assert_eq!(entry.max_items, Some(20)); -} - -#[test] -fn kinds_with_no_defaults_are_left_alone() { - // Composio caps come from the toolkit slug at upsert time, which this - // function does not have; folders and web pages have no caps at all. - for kind in [ - SourceKind::Composio, - SourceKind::Conversation, - SourceKind::Folder, - SourceKind::File, - SourceKind::WebPage, - ] { - let mut entry = entry_of_kind(kind.clone()); - apply_kind_defaults(&mut entry); - assert!(entry.max_items.is_none(), "{kind:?} gained an item cap"); - assert!( - entry.max_prs.is_none(), - "{kind:?} gained a pull-request cap" - ); - } -} - -#[test] -fn applying_the_defaults_twice_changes_nothing() { - let mut once = entry_of_kind(SourceKind::GithubRepo); - apply_kind_defaults(&mut once); - let mut twice = once.clone(); - apply_kind_defaults(&mut twice); - assert_eq!(twice.max_prs, once.max_prs); - assert_eq!(twice.max_issues, once.max_issues); - assert_eq!(twice.max_commits, once.max_commits); -} diff --git a/crates/tinymemory-sources/src/validation.rs b/crates/tinymemory-sources/src/validation.rs deleted file mode 100644 index 92962736..00000000 --- a/crates/tinymemory-sources/src/validation.rs +++ /dev/null @@ -1,86 +0,0 @@ -//! Field rules for a configured source, and the path-containment guard the -//! local readers share. - -use std::path::{Path, PathBuf}; - -use crate::error::{Error, Result}; - -use super::types::{MemorySourceEntry, SourceKind}; - -/// Validate required fields for `entry` based on its [`SourceKind`]. -/// -/// `id` and `label` are required for every kind; kind-specific fields follow. -/// -/// # Errors -/// -/// [`Error::Invalid`] with a human-readable message naming the first failing -/// rule. -pub fn validate_entry(entry: &MemorySourceEntry) -> Result<()> { - if entry.id.trim().is_empty() { - return Err(Error::Invalid("id is required".to_string())); - } - if entry.id.contains(':') || entry.id.chars().any(char::is_control) { - return Err(Error::Invalid( - "id must not contain ':' or control characters".to_string(), - )); - } - if entry.label.is_empty() { - return Err(Error::Invalid("label is required".to_string())); - } - match entry.kind { - SourceKind::Composio => { - require_field(&entry.toolkit, "toolkit")?; - require_field(&entry.connection_id, "connection_id")?; - } - SourceKind::Conversation => { - // No kind-specific required fields — just enabled/disabled. - } - SourceKind::Folder | SourceKind::File => { - require_field(&entry.path, "path")?; - } - SourceKind::GithubRepo => { - require_field(&entry.url, "url")?; - } - SourceKind::RssFeed => { - require_field(&entry.url, "url")?; - } - SourceKind::WebPage => { - require_field(&entry.url, "url")?; - } - } - Ok(()) -} - -/// Require that `value` is present and non-empty, naming it `name` in errors. -fn require_field(value: &Option, name: &str) -> Result<()> { - match value { - Some(v) if !v.is_empty() => Ok(()), - _ => Err(Error::Invalid(format!( - "{name} is required for this source kind" - ))), - } -} - -/// Canonicalize `target` and ensure it stays within canonicalized `base`. -/// -/// This is the shared path-traversal guard for local readers. Both paths must -/// exist (they are passed through [`std::fs::canonicalize`], which resolves -/// symlinks and `..` segments). If the resolved target escapes the base -/// directory, the guard refuses it. -/// -/// # Errors -/// -/// [`Error::PathEscape`] carrying `"path traversal denied"` when the target -/// escapes, [`Error::Io`] when either path cannot be canonicalised. -pub fn ensure_within_base(base: &Path, target: &Path) -> Result { - let canonical_base = std::fs::canonicalize(base)?; - let canonical_target = std::fs::canonicalize(target)?; - if !canonical_target.starts_with(&canonical_base) { - return Err(Error::PathEscape("path traversal denied".to_string())); - } - Ok(canonical_target) -} - -#[cfg(test)] -#[path = "validation_tests.rs"] -mod tests; diff --git a/crates/tinymemory-sources/src/validation_tests.rs b/crates/tinymemory-sources/src/validation_tests.rs deleted file mode 100644 index 11e86c78..00000000 --- a/crates/tinymemory-sources/src/validation_tests.rs +++ /dev/null @@ -1,75 +0,0 @@ -//! Tests for required-field validation and the path-traversal guard. - -use super::*; -use crate::types::SourceKind; -use std::fs; -use tempfile::TempDir; - -fn entry(kind: SourceKind) -> MemorySourceEntry { - MemorySourceEntry { - id: "src_x".into(), - kind, - label: "Label".into(), - enabled: true, - toolkit: None, - connection_id: None, - path: None, - glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - } -} - -#[test] -fn empty_id_or_label_is_rejected_for_every_kind() { - let mut e = entry(SourceKind::Conversation); - e.id = String::new(); - assert!(validate_entry(&e).is_err()); - - let mut e = entry(SourceKind::Conversation); - e.label = String::new(); - assert!(validate_entry(&e).is_err()); -} - -#[test] -fn empty_string_field_counts_as_missing() { - let mut e = entry(SourceKind::Folder); - e.path = Some(String::new()); - assert!(validate_entry(&e).is_err()); -} - -#[test] -fn composio_requires_both_toolkit_and_connection() { - let mut e = entry(SourceKind::Composio); - e.toolkit = Some("gmail".into()); - assert!(validate_entry(&e).is_err()); - e.connection_id = Some("conn".into()); - assert!(validate_entry(&e).is_ok()); -} - -#[test] -fn ensure_within_base_accepts_contained_file() { - let tmp = TempDir::new().unwrap(); - fs::write(tmp.path().join("ok.md"), "hi").unwrap(); - let resolved = ensure_within_base(tmp.path(), &tmp.path().join("ok.md")).unwrap(); - assert!(resolved.ends_with("ok.md")); -} - -#[test] -fn ensure_within_base_rejects_escape() { - let tmp = TempDir::new().unwrap(); - fs::write(tmp.path().join("ok.md"), "hi").unwrap(); - // Build a target that escapes the base via `..`. - let escaping = tmp.path().join("../../etc/hosts"); - let result = ensure_within_base(tmp.path(), &escaping); - assert!(result.is_err()); -} diff --git a/crates/tinymemory-tools/Cargo.toml b/crates/tinymemory-tools/Cargo.toml new file mode 100644 index 00000000..b77a22ed --- /dev/null +++ b/crates/tinymemory-tools/Cargo.toml @@ -0,0 +1,35 @@ +[package] +name = "tinymemory-tools" +description = "Agent-facing memory tools over any TinyMemory engine, and the context.md compiler" +readme = "README.md" +keywords = ["memory", "agent", "llm", "tools"] +categories = ["database"] +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +publish.workspace = true + +[dependencies] +# The engine every tool calls and the types its arguments and results map to. +tinymemory-api = { path = "../tinymemory-api" } +# Tool parameters are JSON Schema documents, and arguments and results are +# JSON values, so a host can hand them to any tool runtime unchanged. +serde = { version = "1", features = ["derive"] } +serde_json = "1" +# `ContextDoc::generated_at` is stamped from the system clock. +chrono = { version = "0.4", default-features = false, features = ["clock", "std", "serde"] } +# A context brief whose recall fails is skipped and logged, not fatal. +log = "0.4" +# `context::Error`. +thiserror = "2" + +[dev-dependencies] +# Every tool and the context compiler run against the reference engine. +tinymemory-api = { path = "../tinymemory-api", features = ["conformance"] } +async-trait = "0.1" +tokio = { version = "1", features = ["macros", "rt"] } + +[lints] +workspace = true diff --git a/crates/tinymemory-tools/README.md b/crates/tinymemory-tools/README.md new file mode 100644 index 00000000..0ca92865 --- /dev/null +++ b/crates/tinymemory-tools/README.md @@ -0,0 +1,197 @@ +# tinymemory-tools + +The agent-facing side of TinyMemory, over any `tinymemory_api::MemoryEngine`: + +- **`tools`**: seven memory tools a host offers its model, as + runtime-neutral specs (name, description, JSON Schema) plus one `call` + entry point that runs them and returns compact JSON. +- **`context`**: the `context.md` compiler, a token-budgeted brief a host + injects at the start of a session. + +The crate has no tool-runtime dependency (no MCP, no `tinytools`). A host +adapts `ToolSpec` to whatever runtime it uses; see +[Adapting to a tool runtime](#adapting-to-a-tool-runtime). The design is +written up in [`docs/architecture/tools.md`](../../docs/architecture/tools.md). + +## The tools + +| Tool | Kind | Arguments | Result | +| --- | --- | --- | --- | +| `memory_recall` | read | `question`, `filter?`, `limit?`, `instructions?` | `{answer, citations: [...]}` | +| `memory_fetch` | read | `query`, `mode?`, `filter?`, `limit?`, `cursor?` | `{hits: [...], next_cursor?}` | +| `memory_list` | read | `filter?`, `limit?`, `cursor?` | `{items: [...], next_cursor?}` | +| `memory_get` | read | `ids` | `{items: [...], missing: [ids]}` | +| `memory_explore` | read | `facet`, `filter?`, `limit?` | `{facet, buckets: [{value, count}], total, missing, more_buckets, truncated}` | +| `memory_store` | write | one of `learning` / `document` / `conversation`, `tags?` | `{id, replayed}` | +| `memory_forget` | write | `ids` **or** a non-empty `filter` | `{forgotten, skipped: [ids]}` | + +Details: + +- `limit` is `1..=50`, default `10`. `ids` lists `1..=200` ids. +- `memory_fetch`'s `mode` enum lists exactly the engine's + `EngineDescriptor::fetch_modes`; with no mode named, `hybrid` is used when + served, otherwise the engine's first mode. An engine serving no mode gets no + `memory_fetch` tool at all. +- `memory_store` takes `learning: {text, learning_kind?, confidence?, + evidence?}` (kind defaults to `fact`, confidence to `0.8`), + `document: {title?, text}` or `conversation: {turns: [{role, text}]}`. + Storing the same memory twice is a replay, not a duplicate. +- `memory_explore`'s facets are `kind`, `source`, `source_id`, `workspace`, + `folder`, `file_path`, `language`, `repo`, `url`, `thread`, `agent`, + `tool_call` and `tag`. The `namespace` facet is deliberately absent. +- The model-facing `filter` is a subset of `MetaFilter`: `kinds`, `sources`, + `tags_any`, `workspace`, `folder`, `file_path`, `repo`, `url`, `thread_id`, + `agent_id`, `observed_after` and `observed_before` (RFC 3339). It never has + a namespace or a reach. +- A hit renders as `{id, kind, text, score, confidence?, meta}` where `meta` + is a subset: `source {kind, id?}`, `file_path`, `url`, `thread_id`, `tags`, + `observed_at`. Scores are rounded to four places. The namespace is never + rendered. + +The exact names and schemas are frozen in +[`tests/fixtures/tool_contracts.json`](tests/fixtures/tool_contracts.json). +A change to them changes what every host's model is told, so the +`tool_contracts` test fails until the fixture is regenerated on purpose: + +```sh +BLESS_TOOL_CONTRACTS=1 cargo test -p tinymemory-tools --test tool_contracts +``` + +## Scoping invariants + +This crate exists so that a model can never choose whose memory it touches. +An earlier tool layer let the model pass a namespace, which let one tenant's +agent read another's memory. Here, the namespace and the reach are fixed by +the host in a `ToolScope` and are never tool arguments: + +| Field | Meaning | +| --- | --- | +| `place: Namespace` | the node every stored item is written to | +| `reach: Option` | the nodes every read (and forget) is confined to; `None` reads every namespace | +| `writes: bool` | whether `memory_store` and `memory_forget` are offered | + +What the tools enforce: + +1. **Writes land at `place`.** `memory_store` builds the item's metadata + itself: namespace `place`, source `agent`, the model's tags, and + `observed_at` set to the time of the call. +2. **Reads use the scope's reach.** Every recall, fetch, list and explore + filter has its `reach` overwritten with the scope's; `memory_get` passes it + as `GetRequest::reach`, and ids outside it come back as `missing`. +3. **Forget cannot reach out.** By ids, the ids are first read back with + `get` under the reach and only those found are forgotten; the rest are + returned as `skipped` and survive. By filter, the model's filter must set + at least one field (a reach alone would mean "everything in reach") and is + then confined to the reach. +4. **Attempts are refused, not ignored.** A `namespace` or `reach` key + anywhere in the arguments, at any depth, is an `Error::InvalidRequest` + saying the host fixes it. Every schema object sets + `additionalProperties: false`, and any other unknown key is refused too. + +Note that a reach with `inherit` (the default from `Reach::of`) includes the +node's ancestors, so a forget may remove memory the agent shares with its +team or the root. A host that wants forgets confined to the agent's own node +sets `Reach::exact(place)`, or offers read-only tools. + +### Building a scope + +```rust,ignore +use std::sync::Arc; +use tinymemory_api::{Namespace, Reach}; +use tinymemory_tools::MemoryTools; + +// Single tenant: root, reads everything, writes enabled. +let tools = MemoryTools::new(engine.clone()); + +// One agent: writes to its node, reads it and its ancestors, never a sibling. +let agent = MemoryTools::new(engine.clone()).placed_at("team:acme/agent:writer".parse()?); + +// Read-only, team-wide. +let auditor = MemoryTools::new(engine) + .placed_at("team:acme".parse()?) + .reach(Reach::subtree("team:acme".parse()?)) + .read_only(); +``` + +`placed_at` resets the reach to `Reach::of(place)`, so a placed scope never +reads more than its own branch by accident; call `reach` after it to change +that. `MemoryTools::with_scope` takes a `ToolScope` directly. + +## Errors + +`MemoryTools::call` returns `tinymemory_api::Result`: + +- `Error::InvalidRequest` for an unknown tool name, arguments that are not an + object, a host-fixed key, an unknown key, or a missing, mistyped or + out-of-range value. Messages are lowercase and name the tool and the field + (`memory_list: \`filter.kinds\` \`memo\` is not an item kind`), so they can + be shown to the model as the tool's error output. +- `Error::Unsupported` for a write tool on read-only tools, and for + `memory_fetch` on an engine serving no fetch mode. The call is well formed; + the tools simply do not offer the operation, which is what this variant + means across the contract. +- Anything the engine returns. + +## Adapting to a tool runtime + +`ToolSpec` is three fields: `name`, `description` and `parameters` (a JSON +Schema object). Most runtimes want exactly that: + +- **MCP**: `name` → `name`, `description` → `description`, + `parameters` → `inputSchema`. On `tools/call`, pass `arguments` to + `MemoryTools::call` and return the result serialised as text content; + return an error result with the error's message on `Err`. +- **OpenAI-style function calling**: `{"type": "function", "function": + {"name", "description", "parameters"}}`. The schemas are closed objects, so + they also suit strict mode where the runtime accepts optional properties. +- **Anthropic tool use**: `name`, `description`, `input_schema`. + +A typical adapter: + +```rust,ignore +for spec in tools.specs() { + runtime.register(spec.name, spec.description, spec.parameters); +} +// When the model calls a tool: +let output = match tools.call(&call.name, call.arguments).await { + Ok(value) => value.to_string(), + Err(error) => format!("error: {error}"), +}; +``` + +Build one `MemoryTools` per agent session from the session's identity, never +from anything the model said. `specs()` is cheap; call it per session so a +read-only or differently scoped agent sees only its own tools. + +## `context.md` + +`tinymemory_tools::context::compile(&engine, &ContextSpec::default())` recalls +one question per `Brief` (four defaults: about the user, active work, +preferences and standing instructions, recent important events), lists the +stored learnings newest and most confident first, and renders markdown with +`generated_at`, `engine`, `tokens` and `refs` frontmatter. The document fits +`budget_tokens` (default 1,500, four characters per token): learnings are +trimmed first, then the last brief. A failing brief is skipped and logged, and +an empty engine yields an empty document. `ContextSpec::reach` confines the +brief to one agent's part of the tree. Details: +[`docs/architecture/tools.md`](../../docs/architecture/tools.md#contextmd). + +## Layout + +```text +src/ +├── lib.rs # crate docs and re-exports +├── context/ # the context.md compiler +└── tools/ + ├── mod.rs # MemoryTools, ToolScope, dispatch + ├── spec/ # ToolSpec, tool names, JSON Schemas, limits + ├── args/ # strict argument reading and the model filter + ├── read/ # recall, fetch, list, get, explore + ├── write/ # store, forget + └── render/ # compact result JSON +tests/ +├── tools_roundtrip.rs # every tool against the reference engine +├── tools_scoping.rs # the scoping invariants +├── tool_contracts.rs # frozen names and schemas +└── fixtures/tool_contracts.json +``` diff --git a/crates/tinymemory-context/src/compile/mod.rs b/crates/tinymemory-tools/src/context/compile/mod.rs similarity index 96% rename from crates/tinymemory-context/src/compile/mod.rs rename to crates/tinymemory-tools/src/context/compile/mod.rs index 043acd63..9baa3bb7 100644 --- a/crates/tinymemory-context/src/compile/mod.rs +++ b/crates/tinymemory-tools/src/context/compile/mod.rs @@ -14,8 +14,8 @@ use tinymemory_api::{ Hit, ItemId, ItemKind, ListRequest, MemoryEngine, MetaFilter, Reach, RecallRequest, }; -use crate::error::Result; -use crate::spec::ContextSpec; +use crate::context::error::Result; +use crate::context::spec::ContextSpec; use render::{BriefSection, LearningLine, Sections}; pub use render::estimate_tokens; @@ -74,7 +74,7 @@ impl ContextCompiler { /// /// # Errors /// - /// [`crate::Error::InvalidSpec`] when the spec cannot produce a document. + /// [`crate::context::Error::InvalidSpec`] when the spec cannot produce a document. /// Engine failures are not errors: see the module docs. pub async fn compile( &self, @@ -108,7 +108,7 @@ impl ContextCompiler { /// /// # Errors /// -/// [`crate::Error::InvalidSpec`] when the spec cannot produce a document. +/// [`crate::context::Error::InvalidSpec`] when the spec cannot produce a document. pub async fn compile(engine: &dyn MemoryEngine, spec: &ContextSpec) -> Result { ContextCompiler::new().compile(engine, spec).await } diff --git a/crates/tinymemory-context/src/compile/mod_tests.rs b/crates/tinymemory-tools/src/context/compile/mod_tests.rs similarity index 96% rename from crates/tinymemory-context/src/compile/mod_tests.rs rename to crates/tinymemory-tools/src/context/compile/mod_tests.rs index eab8238e..d7485933 100644 --- a/crates/tinymemory-context/src/compile/mod_tests.rs +++ b/crates/tinymemory-tools/src/context/compile/mod_tests.rs @@ -2,14 +2,14 @@ use async_trait::async_trait; use chrono::TimeZone; +use tinymemory_api::conformance::ReferenceEngine; use tinymemory_api::{ EngineDescriptor, EngineHealth, Error as ApiError, FetchPage, FetchRequest, ForgetReport, ForgetTarget, LearningKind, ListPage, MemoryMeta, RecallAnswer, StoreItem, StoreReceipt, }; -use tinymemory_conformance::ReferenceEngine; use super::*; -use crate::spec::Brief; +use crate::context::spec::Brief; fn at() -> DateTime { Utc.with_ymd_and_hms(2026, 10, 2, 12, 0, 0).unwrap() @@ -85,7 +85,7 @@ async fn briefs_and_learnings_fill_the_document_in_order() { .collect(); assert!(order.windows(2).all(|pair| pair[0] < pair[1]), "{md}"); assert!(!doc.refs.is_empty()); - assert_eq!(doc.tokens, crate::estimate_tokens(md)); + assert_eq!(doc.tokens, crate::context::estimate_tokens(md)); } #[tokio::test] @@ -156,7 +156,7 @@ async fn an_invalid_spec_is_refused() { }; assert!(matches!( compile(&engine, &spec).await, - Err(crate::Error::InvalidSpec(_)) + Err(crate::context::Error::InvalidSpec(_)) )); } diff --git a/crates/tinymemory-context/src/compile/render.rs b/crates/tinymemory-tools/src/context/compile/render.rs similarity index 98% rename from crates/tinymemory-context/src/compile/render.rs rename to crates/tinymemory-tools/src/context/compile/render.rs index 610c61ae..25e1c7b6 100644 --- a/crates/tinymemory-context/src/compile/render.rs +++ b/crates/tinymemory-tools/src/context/compile/render.rs @@ -19,7 +19,7 @@ const MIN_BRIEF_CHARS: usize = 40; const ELLIPSIS: char = '…'; /// The estimated token count of `text`: four characters per token, rounded -/// up — the estimate every budget in this crate uses. +/// up — the estimate every context budget uses. pub fn estimate_tokens(text: &str) -> usize { text.chars().count().div_ceil(CHARS_PER_TOKEN) } diff --git a/crates/tinymemory-context/src/compile/render_tests.rs b/crates/tinymemory-tools/src/context/compile/render_tests.rs similarity index 100% rename from crates/tinymemory-context/src/compile/render_tests.rs rename to crates/tinymemory-tools/src/context/compile/render_tests.rs diff --git a/crates/tinymemory-context/src/error/mod.rs b/crates/tinymemory-tools/src/context/error/mod.rs similarity index 93% rename from crates/tinymemory-context/src/error/mod.rs rename to crates/tinymemory-tools/src/context/error/mod.rs index c8d80dbe..2a85695a 100644 --- a/crates/tinymemory-context/src/error/mod.rs +++ b/crates/tinymemory-tools/src/context/error/mod.rs @@ -12,5 +12,5 @@ pub enum Error { InvalidSpec(String), } -/// The crate-wide result alias. +/// The context module's result alias. pub type Result = std::result::Result; diff --git a/crates/tinymemory-context/src/lib.rs b/crates/tinymemory-tools/src/context/mod.rs similarity index 60% rename from crates/tinymemory-context/src/lib.rs rename to crates/tinymemory-tools/src/context/mod.rs index 7eb24725..ea35aa42 100644 --- a/crates/tinymemory-context/src/lib.rs +++ b/crates/tinymemory-tools/src/context/mod.rs @@ -7,20 +7,31 @@ //! and most confident first, and renders a markdown document with frontmatter //! recording `generated_at`, `engine`, `tokens` and `refs`. //! +//! The document is `---` frontmatter, a `# Context` heading, one `## ` +//! section per brief that cited something, and a `## Learnings` bullet list. +//! `tokens` in the frontmatter is the document's own estimate, frontmatter +//! included, and `refs` lists every item the document cites in order of first +//! citation. [`ContextSpec::reach`] confines every brief's recall and the +//! learnings listing to one agent's part of the memory tree. +//! //! Rules, from the spec: //! //! - The whole document fits `budget_tokens`, estimated at four characters //! per token ([`estimate_tokens`]). Briefs keep their order; learnings are -//! trimmed first, then the last brief shrinks and is dropped. -//! - An engine with nothing stored yields an empty document, not an error. -//! - A brief that fails is skipped and logged; it does not fail the document. +//! trimmed first (one line at a time from the end), then the last brief +//! shrinks and is dropped once too little of it is left. +//! - An engine with nothing stored yields an empty document, not an error. So +//! does a budget too small for anything to survive trimming. +//! - A brief that fails, or cites nothing, is skipped (a failure is logged); a +//! failed learnings listing leaves the learnings out. Neither fails the +//! document. The only error is an invalid spec ([`ContextSpec::validate`]). //! //! # Example //! //! ``` //! use tinymemory_api::{LearningKind, MemoryEngine, MemoryMeta, StoreItem}; -//! use tinymemory_conformance::ReferenceEngine; -//! use tinymemory_context::{ContextSpec, compile}; +//! use tinymemory_api::conformance::ReferenceEngine; +//! use tinymemory_tools::context::{ContextSpec, compile}; //! //! # let runtime = tokio::runtime::Builder::new_current_thread().build()?; //! # runtime.block_on(async { @@ -31,11 +42,11 @@ //! engine //! .store(StoreItem::learning("prefers short answers", LearningKind::Preference, 0.9, MemoryMeta::default())) //! .await -//! .map_err(|e| tinymemory_context::Error::InvalidSpec(e.to_string()))?; +//! .map_err(|e| tinymemory_tools::context::Error::InvalidSpec(e.to_string()))?; //! let doc = compile(&engine, &ContextSpec::default()).await?; //! assert!(doc.markdown.contains("## Learnings")); //! assert!(doc.tokens <= ContextSpec::default().budget_tokens); -//! # Ok::<(), tinymemory_context::Error>(()) +//! # Ok::<(), tinymemory_tools::context::Error>(()) //! # })?; //! # Ok::<(), Box>(()) //! ``` diff --git a/crates/tinymemory-context/src/spec.rs b/crates/tinymemory-tools/src/context/spec.rs similarity index 98% rename from crates/tinymemory-context/src/spec.rs rename to crates/tinymemory-tools/src/context/spec.rs index 6229e294..6e253cf1 100644 --- a/crates/tinymemory-context/src/spec.rs +++ b/crates/tinymemory-tools/src/context/spec.rs @@ -3,7 +3,7 @@ use serde::{Deserialize, Serialize}; use tinymemory_api::{MetaFilter, Reach}; -use crate::error::{Error, Result}; +use crate::context::error::{Error, Result}; /// Default token budget for the whole document. pub const DEFAULT_BUDGET_TOKENS: usize = 1_500; diff --git a/crates/tinymemory-context/src/spec_tests.rs b/crates/tinymemory-tools/src/context/spec_tests.rs similarity index 100% rename from crates/tinymemory-context/src/spec_tests.rs rename to crates/tinymemory-tools/src/context/spec_tests.rs diff --git a/crates/tinymemory-tools/src/lib.rs b/crates/tinymemory-tools/src/lib.rs new file mode 100644 index 00000000..e259646d --- /dev/null +++ b/crates/tinymemory-tools/src/lib.rs @@ -0,0 +1,57 @@ +//! The agent-facing side of TinyMemory, over any +//! [`tinymemory_api::MemoryEngine`]. +//! +//! - [`tools`] offers a model seven memory tools (recall, fetch, list, get, +//! explore, store, forget) as runtime-neutral [`ToolSpec`]s, and runs them +//! through [`MemoryTools::call`]. The namespace a model writes to and the +//! reach it reads with are fixed by the host in a [`ToolScope`]; no tool +//! argument can name either. +//! - [`context`] compiles `context.md`, a token-budgeted brief a host injects +//! at the start of a session. +//! +//! # Example +//! +//! ``` +//! use std::sync::Arc; +//! use serde_json::json; +//! use tinymemory_api::conformance::ReferenceEngine; +//! use tinymemory_api::Namespace; +//! use tinymemory_tools::{MEMORY_RECALL, MEMORY_STORE, MemoryTools}; +//! +//! # let runtime = tokio::runtime::Builder::new_current_thread().build()?; +//! # runtime.block_on(async { +//! let engine = Arc::new(ReferenceEngine::new()); +//! let tools = MemoryTools::new(engine).placed_at(Namespace::agent("researcher")); +//! +//! // Hand these to the tool runtime: name, description, JSON Schema. +//! let names: Vec<&str> = tools.specs().iter().map(|spec| spec.name).collect(); +//! assert!(names.contains(&MEMORY_STORE)); +//! +//! // Run what the model asked for. +//! tools +//! .call(MEMORY_STORE, json!({ +//! "learning": { "text": "The user prefers short answers", "learning_kind": "preference" }, +//! "tags": ["style"] +//! })) +//! .await?; +//! let answer = tools +//! .call(MEMORY_RECALL, json!({ "question": "How long should answers be?" })) +//! .await?; +//! assert_eq!(answer["citations"][0]["meta"]["tags"], json!(["style"])); +//! +//! // Read-only tools neither list nor run the write tools. +//! let reader = tools.clone().read_only(); +//! assert!(reader.specs().iter().all(|spec| spec.name != MEMORY_STORE)); +//! assert!(reader.call(MEMORY_STORE, json!({})).await.is_err()); +//! # Ok::<(), tinymemory_api::Error>(()) +//! # })?; +//! # Ok::<(), Box>(()) +//! ``` + +pub mod context; +pub mod tools; + +pub use tools::{ + MEMORY_EXPLORE, MEMORY_FETCH, MEMORY_FORGET, MEMORY_GET, MEMORY_LIST, MEMORY_RECALL, + MEMORY_STORE, MemoryTools, TOOL_NAMES, ToolScope, ToolSpec, WRITE_TOOL_NAMES, +}; diff --git a/crates/tinymemory-tools/src/tools/args/filter.rs b/crates/tinymemory-tools/src/tools/args/filter.rs new file mode 100644 index 00000000..470a3de6 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/args/filter.rs @@ -0,0 +1,107 @@ +//! Reading the model-facing filter, the explore facet and the fetch mode. +//! +//! The filter is a subset of [`MetaFilter`]: the fields in +//! [`FILTER_FIELDS`]. Its `reach` is never read from the arguments; the +//! caller sets it from the host's scope afterwards. + +use chrono::{DateTime, Utc}; +use serde::de::DeserializeOwned; +use serde_json::Value; +use tinymemory_api::{Facet, FetchMode, ItemKind, MetaFilter, Result, SourceKind}; + +use super::Args; +use crate::tools::spec::schema::{EXPLORE_FACETS, FILTER_FIELDS}; + +/// The filter at `key`, or an empty filter when absent. Its `reach` is unset. +/// +/// # Errors +/// +/// [`tinymemory_api::Error::InvalidRequest`] naming the offending `filter.*` field. +pub(crate) fn meta_filter(args: &Args<'_>, key: &str) -> Result { + let Some(filter) = args.object(key, "filter.", &FILTER_FIELDS)? else { + return Ok(MetaFilter::default()); + }; + Ok(MetaFilter { + kinds: wire_list::(&filter, "kinds", "an item kind")?, + sources: wire_list::(&filter, "sources", "a source kind")?, + tags_any: filter.strings("tags_any")?, + workspace: filter.string("workspace")?, + folder: filter.string("folder")?, + file_path: filter.string("file_path")?, + repo: filter.string("repo")?, + url: filter.string("url")?, + thread_id: filter.string("thread_id")?, + agent_id: filter.string("agent_id")?, + observed_after: timestamp(&filter, "observed_after")?, + observed_before: timestamp(&filter, "observed_before")?, + ..MetaFilter::default() + }) +} + +/// The required facet at `key`, one of [`EXPLORE_FACETS`]. +/// +/// # Errors +/// +/// [`tinymemory_api::Error::InvalidRequest`] for a missing value or one outside the list. +pub(crate) fn facet(args: &Args<'_>, key: &str) -> Result { + let name = args.required_string(key)?; + if !EXPLORE_FACETS.contains(&name.as_str()) { + return Err(args.field_error( + key, + &format!("must be one of {}", EXPLORE_FACETS.join(", ")), + )); + } + wire(&name).ok_or_else(|| args.field_error(key, "is not a facet")) +} + +/// The fetch mode at `key`, one of `modes`; `default` when absent. +/// +/// # Errors +/// +/// [`tinymemory_api::Error::InvalidRequest`] for a value that is not one of `modes`. +pub(crate) fn fetch_mode( + args: &Args<'_>, + key: &str, + modes: &[FetchMode], + default: FetchMode, +) -> Result { + let Some(name) = args.string(key)? else { + return Ok(default); + }; + modes + .iter() + .copied() + .find(|mode| mode.as_str() == name) + .ok_or_else(|| { + let names: Vec<&str> = modes.iter().map(|mode| mode.as_str()).collect(); + args.field_error(key, &format!("must be one of {}", names.join(", "))) + }) +} + +fn wire_list(args: &Args<'_>, key: &str, what: &str) -> Result> { + args.strings(key)? + .iter() + .map(|name| { + wire(name).ok_or_else(|| args.field_error(key, &format!("`{name}` is not {what}"))) + }) + .collect() +} + +/// A snake_case wire string read as the enum it names. +fn wire(name: &str) -> Option { + serde_json::from_value(Value::String(name.to_string())).ok() +} + +fn timestamp(args: &Args<'_>, key: &str) -> Result>> { + args.string(key)? + .map(|value| { + DateTime::parse_from_rfc3339(&value) + .map(|at| at.with_timezone(&Utc)) + .map_err(|_| args.field_error(key, "must be an rfc 3339 timestamp")) + }) + .transpose() +} + +#[cfg(test)] +#[path = "filter_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/args/filter_tests.rs b/crates/tinymemory-tools/src/tools/args/filter_tests.rs new file mode 100644 index 00000000..4cae4759 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/args/filter_tests.rs @@ -0,0 +1,129 @@ +//! The model-facing filter, the explore facet and the fetch mode. + +use super::*; +use serde_json::json; +use tinymemory_api::Error; + +fn invalid_message(result: Result) -> String { + match result { + Err(Error::InvalidRequest(message)) => message, + other => panic!("expected an invalid request, got {other:?}"), + } +} + +#[test] +fn every_filter_field_maps_onto_the_meta_filter_and_reach_stays_unset() { + let value = json!({ "filter": { + "kinds": ["learning", "document"], + "sources": ["folder"], + "tags_any": ["rust"], + "workspace": "ws", + "folder": "/notes", + "file_path": "/notes/a.md", + "repo": "o/r", + "url": "https://example.com", + "thread_id": "t1", + "agent_id": "a1", + "observed_after": "2026-01-01T00:00:00Z", + "observed_before": "2026-02-01T00:00:00+01:00" + }}); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + let filter = meta_filter(&args, "filter").unwrap(); + assert_eq!(filter.kinds, [ItemKind::Learning, ItemKind::Document]); + assert_eq!(filter.sources, [SourceKind::Folder]); + assert_eq!(filter.tags_any, ["rust"]); + assert_eq!(filter.workspace.as_deref(), Some("ws")); + assert_eq!(filter.folder.as_deref(), Some("/notes")); + assert_eq!(filter.file_path.as_deref(), Some("/notes/a.md")); + assert_eq!(filter.repo.as_deref(), Some("o/r")); + assert_eq!(filter.url.as_deref(), Some("https://example.com")); + assert_eq!(filter.thread_id.as_deref(), Some("t1")); + assert_eq!(filter.agent_id.as_deref(), Some("a1")); + assert_eq!( + filter.observed_after.map(|at| at.to_rfc3339()), + Some("2026-01-01T00:00:00+00:00".to_string()) + ); + assert_eq!( + filter.observed_before.map(|at| at.to_rfc3339()), + Some("2026-01-31T23:00:00+00:00".to_string()) + ); + assert!(filter.reach.is_none()); +} + +#[test] +fn an_absent_filter_is_empty() { + let value = json!({}); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + assert!(meta_filter(&args, "filter").unwrap().is_empty()); +} + +#[test] +fn a_filter_field_outside_the_subset_is_refused() { + let value = json!({ "filter": { "commit": "abc" } }); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + let text = invalid_message(meta_filter(&args, "filter")); + assert!( + text.contains("`filter.commit` is not an argument"), + "{text}" + ); +} + +#[test] +fn a_namespace_inside_the_filter_is_refused() { + let value = json!({ "filter": { "namespace": "agent:other" } }); + let text = invalid_message(Args::parse("t", &value, &["filter"])); + assert!( + text.contains("`filter.namespace` is fixed by the host"), + "{text}" + ); +} + +#[test] +fn a_bad_kind_or_timestamp_is_refused_by_field() { + let value = json!({ "filter": { "kinds": ["memo"] } }); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + let text = invalid_message(meta_filter(&args, "filter")); + assert!( + text.contains("`filter.kinds` `memo` is not an item kind"), + "{text}" + ); + + let value = json!({ "filter": { "observed_after": "yesterday" } }); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + let text = invalid_message(meta_filter(&args, "filter")); + assert!( + text.contains("`filter.observed_after` must be an rfc 3339 timestamp"), + "{text}" + ); +} + +#[test] +fn facets_are_read_and_the_namespace_facet_is_refused() { + let value = json!({ "a": "thread", "b": "namespace", "c": "colour" }); + let args = Args::parse("t", &value, &["a", "b", "c"]).unwrap(); + assert_eq!(facet(&args, "a").unwrap(), Facet::Thread); + assert!(invalid_message(facet(&args, "b")).contains("`b` must be one of")); + assert!(facet(&args, "c").is_err()); + assert!(facet(&args, "missing").is_err()); +} + +#[test] +fn fetch_modes_follow_the_engine() { + let value = json!({ "mode": "vector" }); + let args = Args::parse("t", &value, &["mode"]).unwrap(); + let text = invalid_message(fetch_mode( + &args, + "mode", + &[FetchMode::Keyword], + FetchMode::Keyword, + )); + assert!(text.contains("`mode` must be one of keyword"), "{text}"); + assert_eq!( + fetch_mode(&args, "mode", &FetchMode::ALL, FetchMode::Hybrid).unwrap(), + FetchMode::Vector + ); + assert_eq!( + fetch_mode(&args, "absent", &FetchMode::ALL, FetchMode::Hybrid).unwrap(), + FetchMode::Hybrid + ); +} diff --git a/crates/tinymemory-tools/src/tools/args/mod.rs b/crates/tinymemory-tools/src/tools/args/mod.rs new file mode 100644 index 00000000..a7e6a142 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/args/mod.rs @@ -0,0 +1,270 @@ +//! Reading a tool call's JSON arguments, strictly. +//! +//! [`Args`] wraps one JSON object and refuses, with +//! [`tinymemory_api::Error::InvalidRequest`] naming the tool and the field: +//! +//! - a `namespace` or `reach` key anywhere in the arguments, nested objects +//! and arrays included, with a message saying the host fixes it (a model +//! that tries to pick a namespace is told so, not quietly ignored); +//! - any other key the tool's schema does not list; +//! - a value of the wrong type or out of range. +//! +//! An explicit `null` reads as absent, since many models send `null` for an +//! optional argument they mean to leave out. + +mod filter; + +pub(crate) use filter::{facet, fetch_mode, meta_filter}; + +use serde_json::{Map, Value}; +use tinymemory_api::{Error, Result}; + +/// Keys a model may never pass, at any depth. +const HOST_FIXED: [&str; 2] = ["namespace", "reach"]; + +/// One JSON object of arguments, checked against the keys a tool accepts. +#[derive(Debug, Clone, Copy)] +pub(crate) struct Args<'a> { + tool: &'static str, + path: &'a str, + map: &'a Map, +} + +/// An empty object, what `null` arguments read as. +static EMPTY: std::sync::LazyLock> = std::sync::LazyLock::new(Map::new); + +impl<'a> Args<'a> { + /// The top-level arguments of `tool`, accepting only `allowed` keys. + /// `null` reads as no arguments. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when `value` is not an object, names a + /// host-fixed key, or names a key outside `allowed`. + pub(crate) fn parse(tool: &'static str, value: &'a Value, allowed: &[&str]) -> Result { + let map = match value { + Value::Null => &*EMPTY, + Value::Object(map) => map, + _ => return Err(invalid(tool, "arguments must be a json object")), + }; + let args = Self { + tool, + path: "", + map, + }; + if let Some(path) = host_fixed_key(value, "") { + return Err(invalid( + tool, + &format!("`{path}` is fixed by the host and cannot be passed to a memory tool"), + )); + } + args.check_keys(allowed)?; + Ok(args) + } + + /// The nested object at `key`, accepting only `allowed` keys; `None` when + /// absent. `path` is how errors name it, normally `"{key}."`. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not an object, or its keys + /// break the rules [`Args::parse`] enforces. + pub(crate) fn object( + &self, + key: &str, + path: &'a str, + allowed: &[&str], + ) -> Result>> { + match self.get(key) { + None => Ok(None), + Some(Value::Object(map)) => { + let nested = Args { + tool: self.tool, + path, + map, + }; + nested.check_keys(allowed)?; + Ok(Some(nested)) + } + Some(_) => Err(self.field_error(key, "must be an object")), + } + } + + /// One element of the array at `key`, read as an object accepting only + /// `allowed` keys. `path` is how errors name its fields. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the element is not an object, or its + /// keys break the rules [`Args::parse`] enforces. + pub(crate) fn element<'b>( + &self, + key: &str, + value: &'b Value, + path: &'b str, + allowed: &[&str], + ) -> Result> { + let Value::Object(map) = value else { + return Err(self.field_error(key, "must hold only objects")); + }; + let element = Args { + tool: self.tool, + path, + map, + }; + element.check_keys(allowed)?; + Ok(element) + } + + /// Whether `key` is present and not `null`. + pub(crate) fn has(&self, key: &str) -> bool { + self.get(key).is_some() + } + + /// An optional string. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not a string. + pub(crate) fn string(&self, key: &str) -> Result> { + match self.get(key) { + None => Ok(None), + Some(Value::String(value)) => Ok(Some(value.clone())), + Some(_) => Err(self.field_error(key, "must be a string")), + } + } + + /// A required, non-blank string. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is missing, blank or not a + /// string. + pub(crate) fn required_string(&self, key: &str) -> Result { + match self.string(key)? { + Some(value) if !value.trim().is_empty() => Ok(value), + _ => Err(self.field_error(key, "is required and must not be blank")), + } + } + + /// An integer in `1..=max`, `default` when absent. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not an integer in range. + pub(crate) fn count(&self, key: &str, default: usize, max: usize) -> Result { + let Some(value) = self.get(key) else { + return Ok(default); + }; + value + .as_u64() + .and_then(|n| usize::try_from(n).ok()) + .filter(|n| (1..=max).contains(n)) + .ok_or_else(|| self.field_error(key, &format!("must be an integer from 1 to {max}"))) + } + + /// A number in `0.0..=1.0`, `default` when absent. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not a number in range. + pub(crate) fn unit(&self, key: &str, default: f32) -> Result { + let Some(value) = self.get(key) else { + return Ok(default); + }; + value + .as_f64() + .filter(|n| (0.0..=1.0).contains(n)) + // In range, so the narrowing loses only precision. + .map(|n| n as f32) + .ok_or_else(|| self.field_error(key, "must be a number from 0 to 1")) + } + + /// An optional list of strings; empty when absent. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not an array of strings. + pub(crate) fn strings(&self, key: &str) -> Result> { + match self.get(key) { + None => Ok(Vec::new()), + Some(Value::Array(values)) => values + .iter() + .map(|value| { + value + .as_str() + .map(str::to_string) + .ok_or_else(|| self.field_error(key, "must be an array of strings")) + }) + .collect(), + Some(_) => Err(self.field_error(key, "must be an array of strings")), + } + } + + /// An optional array; empty when absent. + /// + /// # Errors + /// + /// [`Error::InvalidRequest`] when the value is not an array. + pub(crate) fn array(&self, key: &str) -> Result<&'a [Value]> { + match self.get(key) { + None => Ok(&[]), + Some(Value::Array(values)) => Ok(values), + Some(_) => Err(self.field_error(key, "must be an array")), + } + } + + /// An [`Error::InvalidRequest`] naming `key` as `{tool}: `{path}{key}` {problem}`. + pub(crate) fn field_error(&self, key: &str, problem: &str) -> Error { + invalid(self.tool, &format!("`{}{key}` {problem}", self.path)) + } + + fn get(&self, key: &str) -> Option<&'a Value> { + self.map.get(key).filter(|value| !value.is_null()) + } + + fn check_keys(&self, allowed: &[&str]) -> Result<()> { + if let Some(key) = self + .map + .keys() + .find(|key| HOST_FIXED.contains(&key.as_str())) + { + return Err(self.field_error( + key, + "is fixed by the host and cannot be passed to a memory tool", + )); + } + if let Some(key) = self.map.keys().find(|key| !allowed.contains(&key.as_str())) { + return Err(self.field_error(key, "is not an argument of this tool")); + } + Ok(()) + } +} + +/// The path of the first host-fixed key anywhere in `value`, searched depth +/// first, so a `namespace` is refused even under a key the tool would +/// otherwise reject as unknown. +fn host_fixed_key(value: &Value, path: &str) -> Option { + match value { + Value::Object(map) => map.iter().find_map(|(key, nested)| { + if HOST_FIXED.contains(&key.as_str()) { + Some(format!("{path}{key}")) + } else { + host_fixed_key(nested, &format!("{path}{key}.")) + } + }), + Value::Array(values) => values.iter().find_map(|nested| { + host_fixed_key(nested, &format!("{}[].", path.trim_end_matches('.'))) + }), + _ => None, + } +} + +/// An [`Error::InvalidRequest`] prefixed with the tool's name. +pub(crate) fn invalid(tool: &str, message: &str) -> Error { + Error::InvalidRequest(format!("{tool}: {message}")) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/args/mod_tests.rs b/crates/tinymemory-tools/src/tools/args/mod_tests.rs new file mode 100644 index 00000000..09f363f4 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/args/mod_tests.rs @@ -0,0 +1,139 @@ +//! Strict argument reading: host-fixed keys, unknown keys, types and ranges. + +use super::*; +use serde_json::json; + +fn message(error: Error) -> String { + match error { + Error::InvalidRequest(message) => message, + other => panic!("expected an invalid request, got {other:?}"), + } +} + +#[test] +fn null_reads_as_no_arguments() { + let value = Value::Null; + let args = Args::parse("memory_list", &value, &["limit"]).unwrap(); + assert!(!args.has("limit")); + assert_eq!(args.count("limit", 7, 50).unwrap(), 7); +} + +#[test] +fn arguments_that_are_not_an_object_are_refused() { + let value = json!([1, 2]); + let error = Args::parse("memory_list", &value, &[]).unwrap_err(); + assert_eq!( + message(error), + "memory_list: arguments must be a json object" + ); +} + +#[test] +fn a_namespace_or_reach_key_is_refused_as_host_fixed() { + for key in ["namespace", "reach"] { + let value = json!({ key: "agent:other" }); + let error = Args::parse("memory_list", &value, &["limit"]).unwrap_err(); + let text = message(error); + assert!(text.contains(&format!("`{key}`")), "{text}"); + assert!(text.contains("fixed by the host"), "{text}"); + } +} + +#[test] +fn a_host_fixed_key_is_refused_even_when_listed_as_allowed() { + let value = json!({ "namespace": "root" }); + assert!(Args::parse("memory_list", &value, &["namespace"]).is_err()); +} + +#[test] +fn a_nested_host_fixed_key_is_refused_with_its_path() { + let value = json!({ "filter": { "reach": { "at": "root" } } }); + let error = Args::parse("memory_list", &value, &["filter"]).unwrap_err(); + assert!(message(error).contains("`filter.reach` is fixed by the host")); +} + +#[test] +fn an_unknown_key_is_refused_by_name() { + let value = json!({ "limt": 3 }); + let error = Args::parse("memory_list", &value, &["limit"]).unwrap_err(); + assert_eq!( + message(error), + "memory_list: `limt` is not an argument of this tool" + ); +} + +#[test] +fn counts_default_and_are_range_checked() { + let value = json!({ "a": 5, "b": 0, "c": 51, "d": "5", "e": -1 }); + let args = Args::parse("t", &value, &["a", "b", "c", "d", "e"]).unwrap(); + assert_eq!(args.count("a", 1, 50).unwrap(), 5); + assert_eq!(args.count("missing", 9, 50).unwrap(), 9); + for key in ["b", "c", "d", "e"] { + let text = message(args.count(key, 1, 50).unwrap_err()); + assert!( + text.contains(&format!("`{key}` must be an integer from 1 to 50")), + "{text}" + ); + } +} + +#[test] +fn units_default_and_are_range_checked() { + let value = json!({ "ok": 0.25, "high": 1.5, "text": "high" }); + let args = Args::parse("t", &value, &["ok", "high", "text"]).unwrap(); + assert!((args.unit("ok", 0.8).unwrap() - 0.25).abs() < f32::EPSILON); + assert!((args.unit("missing", 0.8).unwrap() - 0.8).abs() < f32::EPSILON); + assert!(args.unit("high", 0.8).is_err()); + assert!(args.unit("text", 0.8).is_err()); +} + +#[test] +fn strings_are_type_checked() { + let value = json!({ "s": 1, "blank": " ", "list": ["a", 2], "notlist": "a", "null": null }); + let args = Args::parse("t", &value, &["s", "blank", "list", "notlist", "null"]).unwrap(); + assert!(message(args.string("s").unwrap_err()).contains("`s` must be a string")); + assert!(message(args.required_string("blank").unwrap_err()).contains("must not be blank")); + assert!(args.required_string("null").is_err()); + assert!(args.strings("list").is_err()); + assert!(args.strings("notlist").is_err()); + assert!(args.strings("null").unwrap().is_empty()); + assert!(args.array("notlist").is_err()); +} + +#[test] +fn a_nested_value_that_is_not_an_object_is_refused() { + let value = json!({ "filter": "kinds=learning" }); + let args = Args::parse("t", &value, &["filter"]).unwrap(); + assert!( + message(args.object("filter", "filter.", &[]).unwrap_err()).contains("must be an object") + ); +} + +#[test] +fn an_array_element_that_is_not_an_object_is_refused() { + let value = json!({ "turns": ["hi"] }); + let args = Args::parse("t", &value, &["turns"]).unwrap(); + let element = &args.array("turns").unwrap()[0]; + let error = args + .element("turns", element, "turns[].", &["text"]) + .unwrap_err(); + assert!(message(error).contains("`turns` must hold only objects")); +} + +#[test] +fn a_host_fixed_key_is_found_at_any_depth() { + let value = json!({ "conversation": { "turns": [{ "role": "user", "reach": "x" }] } }); + let error = Args::parse("memory_store", &value, &["conversation"]).unwrap_err(); + let text = message(error); + assert!( + text.contains("`conversation.turns[].reach` is fixed by the host"), + "{text}" + ); + + let value = json!({ "unknown": { "namespace": "root" } }); + let text = message(Args::parse("memory_get", &value, &["ids"]).unwrap_err()); + assert!( + text.contains("`unknown.namespace` is fixed by the host"), + "{text}" + ); +} diff --git a/crates/tinymemory-tools/src/tools/mod.rs b/crates/tinymemory-tools/src/tools/mod.rs new file mode 100644 index 00000000..9156ea4f --- /dev/null +++ b/crates/tinymemory-tools/src/tools/mod.rs @@ -0,0 +1,242 @@ +//! Agent-facing memory tools over any [`MemoryEngine`], with no tool-runtime +//! dependency. +//! +//! [`MemoryTools`] offers seven tools (see [`TOOL_NAMES`]): `memory_recall`, +//! `memory_fetch`, `memory_list`, `memory_get` and `memory_explore` read; +//! `memory_store` and `memory_forget` write. [`MemoryTools::specs`] describes +//! them as [`ToolSpec`]s (a name, a description, a JSON Schema) for a host to +//! hand its tool runtime, and [`MemoryTools::call`] runs one by name with the +//! JSON arguments the model produced, returning compact JSON. +//! +//! # Scoping +//! +//! Which memory node a model writes to and how far it reads are fixed by the +//! host in a [`ToolScope`], never chosen by the model: +//! +//! - Every stored item's namespace is the scope's `place`. +//! - Every read's reach is the scope's `reach`, overwriting anything else; +//! `memory_get` passes it as [`tinymemory_api::GetRequest::reach`]. +//! - `memory_forget` by ids reads the ids back under the reach first and +//! forgets only those found, reporting the rest as `skipped`; by filter, the +//! filter is confined to the reach. +//! - Arguments naming `namespace` or `reach`, at any depth, are refused with +//! [`Error::InvalidRequest`] rather than ignored, and every schema sets +//! `additionalProperties: false`. +//! +//! # Errors +//! +//! An unknown tool name and malformed arguments are +//! [`Error::InvalidRequest`], with a lowercase message naming the tool and +//! the field. A write tool called on read-only tools is +//! [`Error::Unsupported`]: the call is well formed, but these tools do not +//! offer the operation, which is what that variant means across the contract. + +mod args; +mod read; +mod render; +mod spec; +mod write; + +use std::sync::Arc; + +use serde_json::Value; +use tinymemory_api::{Error, MemoryEngine, MetaFilter, Namespace, Reach, Result}; + +pub use spec::{ + MEMORY_EXPLORE, MEMORY_FETCH, MEMORY_FORGET, MEMORY_GET, MEMORY_LIST, MEMORY_RECALL, + MEMORY_STORE, TOOL_NAMES, ToolSpec, WRITE_TOOL_NAMES, +}; + +/// Where a model's memory tools write and how far they read, fixed by the +/// host. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ToolScope { + /// The node every stored item is written to. + pub place: Namespace, + /// The reach every read is confined to; `None` reads every namespace, + /// which suits only a host whose engine serves a single tenant. + pub reach: Option, + /// Whether `memory_store` and `memory_forget` are offered. + pub writes: bool, +} + +impl Default for ToolScope { + /// The root, reading every namespace, with writes enabled: a + /// single-tenant host's scope. + fn default() -> Self { + Self { + place: Namespace::ROOT, + reach: None, + writes: true, + } + } +} + +impl ToolScope { + /// The scope of an agent at `place`: writes land at `place`, and reads + /// see `place` and its ancestors ([`Reach::of`]), never a sibling. + #[must_use] + pub fn at(place: Namespace) -> Self { + Self { + reach: Some(Reach::of(place.clone())), + place, + writes: true, + } + } + + /// Overwrites `filter.reach` with the scope's reach. + pub(crate) fn confine(&self, filter: &mut MetaFilter) { + filter.reach = self.reach.clone(); + } +} + +/// The memory tools a host offers a model, over one engine and one +/// [`ToolScope`]. +/// +/// # Example +/// +/// ``` +/// use std::sync::Arc; +/// use serde_json::json; +/// use tinymemory_api::conformance::ReferenceEngine; +/// use tinymemory_api::Namespace; +/// use tinymemory_tools::MemoryTools; +/// +/// # let runtime = tokio::runtime::Builder::new_current_thread().build()?; +/// # runtime.block_on(async { +/// let tools = MemoryTools::new(Arc::new(ReferenceEngine::new())) +/// .placed_at(Namespace::agent("writer")); +/// let stored = tools +/// .call("memory_store", json!({ "learning": { "text": "prefers tabs" } })) +/// .await?; +/// assert_eq!(stored["replayed"], json!(false)); +/// +/// // A model cannot pick a namespace. +/// let escape = tools +/// .call("memory_list", json!({ "filter": { "namespace": "agent:other" } })) +/// .await; +/// assert!(escape.is_err()); +/// # Ok::<(), tinymemory_api::Error>(()) +/// # })?; +/// # Ok::<(), Box>(()) +/// ``` +#[derive(Clone)] +pub struct MemoryTools { + engine: Arc, + scope: ToolScope, +} + +impl std::fmt::Debug for MemoryTools { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("MemoryTools") + .field("engine", &self.engine.descriptor().id) + .field("scope", &self.scope) + .finish() + } +} + +impl MemoryTools { + /// Tools over `engine` with [`ToolScope::default`]: written to the root, + /// reading every namespace, writes enabled. A multi-tenant host narrows + /// this with [`MemoryTools::placed_at`] or [`MemoryTools::with_scope`]. + #[must_use] + pub fn new(engine: Arc) -> Self { + Self::with_scope(engine, ToolScope::default()) + } + + /// Tools over `engine` with an explicit scope. + #[must_use] + pub fn with_scope(engine: Arc, scope: ToolScope) -> Self { + Self { engine, scope } + } + + /// Writes land at `place`, and the reach becomes [`Reach::of`]`(place)`: + /// `place` and its ancestors, never a sibling. Call + /// [`MemoryTools::reach`] afterwards to read differently; placing resets + /// any reach set before, so a placed scope is never left reading more + /// than its own branch by accident. + #[must_use] + pub fn placed_at(mut self, place: Namespace) -> Self { + let writes = self.scope.writes; + self.scope = ToolScope { + writes, + ..ToolScope::at(place) + }; + self + } + + /// Confines every read (and every forget) to `reach`. + #[must_use] + pub fn reach(mut self, reach: Reach) -> Self { + self.scope.reach = Some(reach); + self + } + + /// Leaves out `memory_store` and `memory_forget`: [`MemoryTools::specs`] + /// omits them and [`MemoryTools::call`] refuses them. + #[must_use] + pub fn read_only(mut self) -> Self { + self.scope.writes = false; + self + } + + /// The scope the tools run in. + #[must_use] + pub fn scope(&self) -> &ToolScope { + &self.scope + } + + /// The tools to offer the model, reads first. The write tools appear only + /// when writes are enabled; `memory_fetch`'s `mode` enum lists exactly the + /// engine's [`tinymemory_api::EngineDescriptor::fetch_modes`], and the tool + /// is left out for an engine serving none. + #[must_use] + pub fn specs(&self) -> Vec { + spec::specs(&self.engine.descriptor().fetch_modes, self.scope.writes) + } + + /// Runs the tool `name` with the model's JSON `args` and returns its + /// compact JSON result. `null` args read as no arguments. + /// + /// # Errors + /// + /// - [`Error::InvalidRequest`] for an unknown tool, arguments that are + /// not an object, name `namespace` or `reach`, name a key the tool does + /// not take, or carry a missing, mistyped or out-of-range value. + /// - [`Error::Unsupported`] for a write tool on read-only tools, and for + /// `memory_fetch` on an engine serving no fetch mode. + /// - Whatever the engine returns for the request. + pub async fn call(&self, name: &str, args: Value) -> Result { + if !TOOL_NAMES.contains(&name) { + return Err(unknown_tool(name)); + } + if spec::is_write_tool(name) && !self.scope.writes { + return Err(Error::Unsupported(format!( + "{name} is not offered: these memory tools are read-only" + ))); + } + let engine = self.engine.as_ref(); + let scope = &self.scope; + match name { + MEMORY_RECALL => read::recall(engine, scope, &args).await, + MEMORY_FETCH => read::fetch(engine, scope, &args).await, + MEMORY_LIST => read::list(engine, scope, &args).await, + MEMORY_GET => read::get(engine, scope, &args).await, + MEMORY_EXPLORE => read::explore(engine, scope, &args).await, + MEMORY_STORE => write::store(engine, scope, &args).await, + MEMORY_FORGET => write::forget(engine, scope, &args).await, + _ => Err(unknown_tool(name)), + } + } +} + +fn unknown_tool(name: &str) -> Error { + Error::InvalidRequest(format!( + "unknown memory tool `{name}`; expected one of {}", + TOOL_NAMES.join(", ") + )) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/mod_tests.rs b/crates/tinymemory-tools/src/tools/mod_tests.rs new file mode 100644 index 00000000..dea4a59b --- /dev/null +++ b/crates/tinymemory-tools/src/tools/mod_tests.rs @@ -0,0 +1,114 @@ +//! `MemoryTools`: dispatch, the read-only refusal, and the scope builders. + +use super::*; +use serde_json::json; +use tinymemory_api::conformance::ReferenceEngine; + +fn tools() -> MemoryTools { + MemoryTools::new(Arc::new(ReferenceEngine::new())) +} + +#[test] +fn new_writes_to_the_root_reads_everything_and_allows_writes() { + assert_eq!(tools().scope(), &ToolScope::default()); + assert_eq!(ToolScope::default().place, Namespace::ROOT); + assert!(ToolScope::default().reach.is_none()); + assert!(ToolScope::default().writes); +} + +#[test] +fn placing_sets_the_reach_to_the_place_and_its_ancestors() { + let place = Namespace::agent("a"); + let placed = tools().placed_at(place.clone()); + assert_eq!(placed.scope().place, place); + assert_eq!(placed.scope().reach, Some(Reach::of(place.clone()))); + + let narrowed = tools() + .placed_at(place.clone()) + .reach(Reach::exact(place.clone())); + assert_eq!(narrowed.scope().reach, Some(Reach::exact(place.clone()))); + + let reset = tools() + .reach(Reach::subtree(Namespace::ROOT)) + .placed_at(place.clone()); + assert_eq!(reset.scope().reach, Some(Reach::of(place))); +} + +#[test] +fn placing_keeps_read_only() { + let tools = tools().read_only().placed_at(Namespace::agent("a")); + assert!(!tools.scope().writes); +} + +#[test] +fn with_scope_uses_the_scope_given() { + let scope = ToolScope { + writes: false, + ..ToolScope::at(Namespace::agent("a")) + }; + let tools = MemoryTools::with_scope(Arc::new(ReferenceEngine::new()), scope.clone()); + assert_eq!(tools.scope(), &scope); +} + +#[test] +fn debug_names_the_engine_and_scope() { + let text = format!("{:?}", tools()); + assert!( + text.contains("reference") && text.contains("scope"), + "{text}" + ); +} + +#[tokio::test] +async fn an_unknown_tool_is_an_invalid_request() { + let error = tools() + .call("memory_delete_all", json!({})) + .await + .unwrap_err(); + assert!( + matches!(error, Error::InvalidRequest(message) if message.contains("memory_delete_all")) + ); +} + +#[tokio::test] +async fn write_tools_are_unsupported_when_read_only() { + let tools = tools().read_only(); + for name in WRITE_TOOL_NAMES { + let error = tools + .call(name, json!({ "learning": { "text": "x" } })) + .await + .unwrap_err(); + assert!(matches!(error, Error::Unsupported(_)), "{name}: {error:?}"); + } +} + +#[tokio::test] +async fn every_tool_dispatches() { + let tools = tools(); + let stored = tools + .call( + MEMORY_STORE, + json!({ "learning": { "text": "rust is fast" } }), + ) + .await + .unwrap(); + let id = stored["id"].clone(); + let calls = [ + (MEMORY_RECALL, json!({ "question": "rust" })), + (MEMORY_FETCH, json!({ "query": "rust" })), + (MEMORY_LIST, Value::Null), + (MEMORY_GET, json!({ "ids": [id] })), + (MEMORY_EXPLORE, json!({ "facet": "kind" })), + (MEMORY_FORGET, json!({ "ids": [id] })), + ]; + for (name, args) in calls { + assert!(tools.call(name, args).await.is_ok(), "{name}"); + } +} + +#[test] +fn the_call_future_is_send() { + fn assert_send(_: T) {} + let tools = tools(); + assert_send(tools.call(MEMORY_LIST, Value::Null)); +} diff --git a/crates/tinymemory-tools/src/tools/read/mod.rs b/crates/tinymemory-tools/src/tools/read/mod.rs new file mode 100644 index 00000000..dc7187c8 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/read/mod.rs @@ -0,0 +1,191 @@ +//! The read tools: `memory_recall`, `memory_fetch`, `memory_list`, +//! `memory_get` and `memory_explore`. +//! +//! Each parses its arguments strictly ([`crate::tools::args`]), confines the +//! request to the host's reach ([`ToolScope::confine`]), calls the engine, +//! and renders the compact result ([`crate::tools::render`]). The reach in +//! the request is always the scope's, whatever the arguments said: a model +//! cannot name one, and a filter it builds cannot widen one. + +use serde_json::Value; +use tinymemory_api::{ + Error, ExploreRequest, FetchRequest, GetRequest, Hit, ItemId, ListRequest, MemoryEngine, + RecallRequest, Result, +}; + +use super::ToolScope; +use super::args::{Args, facet, fetch_mode, meta_filter}; +use super::render; +use super::spec::schema::{DEFAULT_LIMIT, MAX_IDS, MAX_LIMIT, default_mode}; +use super::spec::{MEMORY_EXPLORE, MEMORY_FETCH, MEMORY_GET, MEMORY_LIST, MEMORY_RECALL}; + +/// `memory_recall`. +/// +/// # Errors +/// +/// Invalid arguments, and the engine's own failures. +pub(crate) async fn recall( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse( + MEMORY_RECALL, + value, + &["question", "filter", "limit", "instructions"], + )?; + let mut filter = meta_filter(&args, "filter")?; + scope.confine(&mut filter); + let request = RecallRequest { + question: args.required_string("question")?, + filter, + limit: args.count("limit", DEFAULT_LIMIT, MAX_LIMIT)?, + instructions: args.string("instructions")?, + }; + Ok(render::recall(&engine.recall(request).await?)) +} + +/// `memory_fetch`. +/// +/// # Errors +/// +/// [`Error::Unsupported`] when the engine serves no fetch mode, invalid +/// arguments (a mode the engine does not serve among them), and the engine's +/// own failures. +pub(crate) async fn fetch( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let modes = &engine.descriptor().fetch_modes; + let Some(default) = default_mode(modes) else { + return Err(Error::Unsupported(format!( + "{MEMORY_FETCH}: engine `{}` serves no fetch mode", + engine.descriptor().id + ))); + }; + let args = Args::parse( + MEMORY_FETCH, + value, + &["query", "mode", "filter", "limit", "cursor"], + )?; + let mut filter = meta_filter(&args, "filter")?; + scope.confine(&mut filter); + let request = FetchRequest { + query: args.required_string("query")?, + mode: fetch_mode(&args, "mode", modes, default)?, + filter, + limit: args.count("limit", DEFAULT_LIMIT, MAX_LIMIT)?, + cursor: args.string("cursor")?, + }; + Ok(render::fetch(&engine.fetch(request).await?)) +} + +/// `memory_list`. +/// +/// # Errors +/// +/// Invalid arguments, and the engine's own failures. +pub(crate) async fn list( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse(MEMORY_LIST, value, &["filter", "limit", "cursor"])?; + let mut filter = meta_filter(&args, "filter")?; + scope.confine(&mut filter); + let request = ListRequest { + filter, + limit: args.count("limit", DEFAULT_LIMIT, MAX_LIMIT)?, + cursor: args.string("cursor")?, + }; + Ok(render::list(&engine.list(request).await?)) +} + +/// `memory_get`: the items in reach, and the ids that named nothing in reach +/// as `missing`. +/// +/// # Errors +/// +/// Invalid arguments, and the engine's own failures. +pub(crate) async fn get( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse(MEMORY_GET, value, &["ids"])?; + let ids = item_ids(&args, "ids")?; + let found = resolve(engine, scope, &ids).await?; + let missing: Vec = ids + .into_iter() + .filter(|id| !found.iter().any(|hit| &hit.id == id)) + .collect(); + Ok(render::get(&found, &missing)) +} + +/// `memory_explore`. +/// +/// # Errors +/// +/// Invalid arguments, and the engine's own failures. +pub(crate) async fn explore( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse(MEMORY_EXPLORE, value, &["facet", "filter", "limit"])?; + let mut request = ExploreRequest::new( + facet(&args, "facet")?, + args.count("limit", DEFAULT_LIMIT, MAX_LIMIT)?, + ); + request.filter = meta_filter(&args, "filter")?; + scope.confine(&mut request.filter); + Ok(render::explore(&engine.explore(request).await?)) +} + +/// Reads `ids` with [`MemoryEngine::get`] under the scope's reach: what comes +/// back is exactly what the scope may see. +/// +/// # Errors +/// +/// The engine's own failures. +pub(crate) async fn resolve( + engine: &dyn MemoryEngine, + scope: &ToolScope, + ids: &[ItemId], +) -> Result> { + engine + .get(GetRequest { + ids: ids.to_vec(), + reach: scope.reach.clone(), + }) + .await +} + +/// The required id list at `key`: `1..=`[`MAX_IDS`] non-blank strings, +/// duplicates dropped. +/// +/// # Errors +/// +/// [`Error::InvalidRequest`] for a missing, empty, oversized or blank list. +pub(crate) fn item_ids(args: &Args<'_>, key: &str) -> Result> { + let raw = args.strings(key)?; + if raw.is_empty() || raw.len() > MAX_IDS { + return Err(args.field_error(key, &format!("must list from 1 to {MAX_IDS} ids"))); + } + if raw.iter().any(|id| id.trim().is_empty()) { + return Err(args.field_error(key, "must not hold a blank id")); + } + let mut ids: Vec = Vec::with_capacity(raw.len()); + for id in raw { + let id = ItemId::new(id); + if !ids.contains(&id) { + ids.push(id); + } + } + Ok(ids) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/read/mod_tests.rs b/crates/tinymemory-tools/src/tools/read/mod_tests.rs new file mode 100644 index 00000000..0e21c6b4 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/read/mod_tests.rs @@ -0,0 +1,117 @@ +//! The read tools against the reference engine, and id-list reading. + +use super::*; +use serde_json::json; +use tinymemory_api::conformance::ReferenceEngine; +use tinymemory_api::{LearningKind, MemoryMeta, Namespace, StoreItem}; + +async fn engine_with(items: &[(&str, Namespace)]) -> (ReferenceEngine, Vec) { + let engine = ReferenceEngine::new(); + let mut ids = Vec::new(); + for (text, namespace) in items { + let meta = MemoryMeta { + namespace: namespace.clone(), + ..MemoryMeta::default() + }; + let receipt = engine + .store(StoreItem::learning(*text, LearningKind::Fact, 0.9, meta)) + .await + .unwrap(); + ids.push(receipt.id); + } + (engine, ids) +} + +#[test] +fn item_ids_are_deduplicated_in_order() { + let value = json!({ "ids": ["b", "a", "b"] }); + let args = Args::parse("t", &value, &["ids"]).unwrap(); + assert_eq!( + item_ids(&args, "ids").unwrap(), + [ItemId::new("b"), ItemId::new("a")] + ); +} + +#[test] +fn item_ids_refuse_empty_oversized_and_blank_lists() { + let too_many: Vec = (0..=MAX_IDS).map(|n| n.to_string()).collect(); + for value in [ + json!({}), + json!({ "ids": [] }), + json!({ "ids": too_many }), + json!({ "ids": ["a", " "] }), + ] { + let args = Args::parse("t", &value, &["ids"]).unwrap(); + assert!( + matches!(item_ids(&args, "ids"), Err(Error::InvalidRequest(_))), + "{value}" + ); + } +} + +#[tokio::test] +async fn get_reports_ids_outside_the_reach_as_missing() { + let own = Namespace::agent("a"); + let (engine, ids) = + engine_with(&[("mine", own.clone()), ("theirs", Namespace::agent("b"))]).await; + let scope = ToolScope::at(own); + let result = get( + &engine, + &scope, + &json!({ "ids": [ids[0].as_str(), ids[1].as_str()] }), + ) + .await + .unwrap(); + assert_eq!(result["items"][0]["text"], json!("mine")); + assert_eq!(result["items"].as_array().map(Vec::len), Some(1)); + assert_eq!(result["missing"], json!([ids[1].as_str()])); +} + +#[tokio::test] +async fn list_overwrites_the_reach_whatever_the_filter_says() { + let own = Namespace::agent("a"); + let (engine, _) = + engine_with(&[("mine", own.clone()), ("theirs", Namespace::agent("b"))]).await; + let scope = ToolScope::at(own); + let result = list( + &engine, + &scope, + &json!({ "filter": { "kinds": ["learning"] } }), + ) + .await + .unwrap(); + let texts: Vec<&Value> = result["items"] + .as_array() + .unwrap() + .iter() + .map(|item| &item["text"]) + .collect(); + assert_eq!(texts, [&json!("mine")]); +} + +#[tokio::test] +async fn fetch_refuses_a_mode_the_engine_does_not_list_by_name() { + let (engine, _) = engine_with(&[]).await; + let error = fetch( + &engine, + &ToolScope::default(), + &json!({ "query": "x", "mode": "semantic" }), + ) + .await + .unwrap_err(); + assert!(matches!(error, Error::InvalidRequest(message) if message.contains("`mode`"))); +} + +#[tokio::test] +async fn recall_and_explore_need_their_required_arguments() { + let (engine, _) = engine_with(&[]).await; + let scope = ToolScope::default(); + assert!(matches!( + recall(&engine, &scope, &json!({})).await, + Err(Error::InvalidRequest(message)) if message.contains("`question`") + )); + assert!(matches!( + explore(&engine, &scope, &json!({})).await, + Err(Error::InvalidRequest(message)) if message.contains("`facet`") + )); +} diff --git a/crates/tinymemory-tools/src/tools/render/mod.rs b/crates/tinymemory-tools/src/tools/render/mod.rs new file mode 100644 index 00000000..081432d3 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/render/mod.rs @@ -0,0 +1,142 @@ +//! Turning engine responses into the compact JSON a model reads. +//! +//! Results carry what a model can act on and nothing else: a hit is its id, +//! kind, text, score (rounded to four places), a learning's confidence, and a +//! metadata subset (the source kind and id, `file_path`, `url`, `thread_id`, +//! `tags`, `observed_at`). The namespace is never rendered; it is the host's. +//! Absent optional fields are left out rather than written as `null`. + +use chrono::SecondsFormat; +use serde_json::{Map, Value, json}; +use tinymemory_api::{ + Citation, ExplorePage, FetchPage, ForgetReport, Hit, ItemId, ListPage, MemoryMeta, + RecallAnswer, StoreReceipt, +}; + +/// `memory_recall`'s result: `{answer, citations: [...]}`. +pub(crate) fn recall(answer: &RecallAnswer) -> Value { + json!({ + "answer": answer.answer, + "citations": answer.citations.iter().map(citation).collect::>(), + }) +} + +/// `memory_fetch`'s result: `{hits: [...], next_cursor?}`. +pub(crate) fn fetch(page: &FetchPage) -> Value { + paged("hits", &page.hits, page.next_cursor.as_deref()) +} + +/// `memory_list`'s result: `{items: [...], next_cursor?}`. +pub(crate) fn list(page: &ListPage) -> Value { + paged("items", &page.items, page.next_cursor.as_deref()) +} + +/// `memory_get`'s result: `{items: [...], missing: [ids]}`. +pub(crate) fn get(found: &[Hit], missing: &[ItemId]) -> Value { + json!({ + "items": found.iter().map(hit).collect::>(), + "missing": missing, + }) +} + +/// `memory_explore`'s result: the facet, its buckets and the counts. +pub(crate) fn explore(page: &ExplorePage) -> Value { + json!({ + "facet": page.facet.as_str(), + "buckets": page.buckets, + "total": page.total, + "missing": page.missing, + "more_buckets": page.more_buckets, + "truncated": page.truncated, + }) +} + +/// `memory_store`'s result: `{id, replayed}`. +pub(crate) fn store(receipt: &StoreReceipt) -> Value { + json!({ "id": receipt.id, "replayed": receipt.replayed }) +} + +/// `memory_forget`'s result: the [`ForgetReport`] fields plus the ids that +/// were skipped because they named nothing in reach. +pub(crate) fn forget(report: &ForgetReport, skipped: &[ItemId]) -> Value { + json!({ "forgotten": report.forgotten, "skipped": skipped }) +} + +/// One hit. +pub(crate) fn hit(hit: &Hit) -> Value { + let mut out = Map::new(); + out.insert("id".to_string(), json!(hit.id)); + out.insert("kind".to_string(), json!(hit.kind.as_str())); + out.insert("text".to_string(), json!(hit.text)); + out.insert("score".to_string(), score(hit.score)); + if let Some(confidence) = hit.confidence { + out.insert("confidence".to_string(), score(confidence)); + } + out.insert("meta".to_string(), meta(&hit.meta)); + Value::Object(out) +} + +fn citation(citation: &Citation) -> Value { + let mut out = Map::new(); + out.insert("id".to_string(), json!(citation.id)); + out.insert("kind".to_string(), json!(citation.kind.as_str())); + out.insert("snippet".to_string(), json!(citation.snippet)); + if let Some(value) = citation.score { + out.insert("score".to_string(), score(value)); + } + out.insert("meta".to_string(), meta(&citation.meta)); + Value::Object(out) +} + +fn paged(key: &str, hits: &[Hit], next_cursor: Option<&str>) -> Value { + let mut out = Map::new(); + out.insert( + key.to_string(), + Value::Array(hits.iter().map(hit).collect()), + ); + if let Some(cursor) = next_cursor { + out.insert("next_cursor".to_string(), json!(cursor)); + } + Value::Object(out) +} + +/// The metadata subset a model sees. +fn meta(meta: &MemoryMeta) -> Value { + let mut source = Map::new(); + source.insert("kind".to_string(), json!(meta.source.kind.as_str())); + if let Some(id) = &meta.source.id { + source.insert("id".to_string(), json!(id)); + } + let mut out = Map::new(); + out.insert("source".to_string(), Value::Object(source)); + let optional = [ + ("file_path", &meta.file_path), + ("url", &meta.url), + ("thread_id", &meta.thread_id), + ]; + for (key, value) in optional { + if let Some(value) = value { + out.insert(key.to_string(), json!(value)); + } + } + if !meta.tags.is_empty() { + out.insert("tags".to_string(), json!(meta.tags)); + } + if let Some(at) = meta.observed_at { + out.insert( + "observed_at".to_string(), + json!(at.to_rfc3339_opts(SecondsFormat::Secs, true)), + ); + } + Value::Object(out) +} + +/// A score rounded to four decimal places, so `f32` noise does not reach the +/// model (`0.1` rather than `0.10000000149011612`). +fn score(value: f32) -> Value { + json!((f64::from(value) * 10_000.0).round() / 10_000.0) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/render/mod_tests.rs b/crates/tinymemory-tools/src/tools/render/mod_tests.rs new file mode 100644 index 00000000..8d205be7 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/render/mod_tests.rs @@ -0,0 +1,136 @@ +//! Compact result rendering. + +use super::*; +use tinymemory_api::chrono::{TimeZone, Utc}; +use tinymemory_api::{Facet, FacetBucket, ItemKind, Namespace, SourceKind, SourceRef}; + +fn sample_hit() -> Hit { + let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); + meta.namespace = Namespace::agent("writer"); + meta.file_path = Some("/notes/a.md".into()); + meta.workspace = Some("hidden-from-the-model".into()); + meta.tags = vec!["rust".into()]; + meta.observed_at = Utc.with_ymd_and_hms(2026, 3, 4, 5, 6, 7).single(); + Hit { + id: ItemId::new("abc"), + kind: ItemKind::Learning, + text: "ownership moves".into(), + meta, + score: 0.1, + confidence: Some(0.9), + } +} + +#[test] +fn a_hit_carries_the_subset_and_never_the_namespace() { + let rendered = hit(&sample_hit()); + assert_eq!( + rendered, + json!({ + "id": "abc", + "kind": "learning", + "text": "ownership moves", + "score": 0.1, + "confidence": 0.9, + "meta": { + "source": { "kind": "folder", "id": "notes" }, + "file_path": "/notes/a.md", + "tags": ["rust"], + "observed_at": "2026-03-04T05:06:07Z" + } + }) + ); +} + +#[test] +fn a_bare_hit_leaves_optional_fields_out() { + let bare = Hit { + id: ItemId::new("x"), + kind: ItemKind::Document, + text: "t".into(), + meta: MemoryMeta { + source: SourceRef::default(), + ..MemoryMeta::default() + }, + score: 0.0, + confidence: None, + }; + assert_eq!( + hit(&bare), + json!({ "id": "x", "kind": "document", "text": "t", "score": 0.0, + "meta": { "source": { "kind": "agent" } } }) + ); +} + +#[test] +fn pages_carry_a_cursor_only_when_there_is_more() { + let more = FetchPage { + hits: vec![sample_hit()], + next_cursor: Some("1".into()), + }; + assert_eq!(fetch(&more)["next_cursor"], json!("1")); + let end = ListPage { + items: vec![sample_hit()], + next_cursor: None, + }; + let rendered = list(&end); + assert_eq!(rendered["items"].as_array().map(Vec::len), Some(1)); + assert!(rendered.get("next_cursor").is_none()); +} + +#[test] +fn recall_renders_answer_and_citations() { + let source = sample_hit(); + let answer = RecallAnswer { + answer: "it moves".into(), + citations: vec![Citation { + id: source.id, + kind: source.kind, + snippet: source.text, + meta: source.meta, + score: None, + }], + model: Some("m".into()), + }; + let rendered = recall(&answer); + assert_eq!(rendered["answer"], json!("it moves")); + assert_eq!( + rendered["citations"][0]["snippet"], + json!("ownership moves") + ); + assert!(rendered["citations"][0].get("score").is_none()); + assert!(rendered.get("model").is_none()); +} + +#[test] +fn writes_and_explore_render_their_fields() { + let receipt = StoreReceipt { + id: ItemId::new("i"), + replayed: true, + }; + assert_eq!(store(&receipt), json!({ "id": "i", "replayed": true })); + assert_eq!( + forget(&ForgetReport { forgotten: 2 }, &[ItemId::new("gone")]), + json!({ "forgotten": 2, "skipped": ["gone"] }) + ); + assert_eq!( + get(&[], &[ItemId::new("m")]), + json!({ "items": [], "missing": ["m"] }) + ); + let page = ExplorePage { + facet: Facet::Tag, + buckets: vec![FacetBucket { + value: "rust".into(), + count: 3, + }], + total: 4, + missing: 1, + more_buckets: 0, + truncated: false, + }; + assert_eq!( + explore(&page), + json!({ "facet": "tag", "buckets": [{ "value": "rust", "count": 3 }], + "total": 4, "missing": 1, "more_buckets": 0, "truncated": false }) + ); +} diff --git a/crates/tinymemory-tools/src/tools/spec/mod.rs b/crates/tinymemory-tools/src/tools/spec/mod.rs new file mode 100644 index 00000000..903d4daf --- /dev/null +++ b/crates/tinymemory-tools/src/tools/spec/mod.rs @@ -0,0 +1,128 @@ +//! What the model is told about each tool: its name, a description, and a +//! JSON Schema for its arguments. +//! +//! [`ToolSpec`] is deliberately runtime-neutral: a host maps `name`, +//! `description` and `parameters` onto whatever its tool runtime calls them +//! (an MCP tool, an OpenAI function, a `tinytools` tool). The schemas never +//! mention a namespace or a reach; those are fixed by the host on +//! [`super::MemoryTools`] and are not the model's to choose. + +pub(crate) mod schema; + +use serde::Serialize; +use tinymemory_api::FetchMode; + +/// `memory_recall`: a question in, a synthesised answer with citations out. +pub const MEMORY_RECALL: &str = "memory_recall"; +/// `memory_fetch`: raw keyword, vector or hybrid retrieval. +pub const MEMORY_FETCH: &str = "memory_fetch"; +/// `memory_list`: page through stored memories with no query. +pub const MEMORY_LIST: &str = "memory_list"; +/// `memory_get`: read memories whole by id. +pub const MEMORY_GET: &str = "memory_get"; +/// `memory_explore`: count stored memories per value of one facet. +pub const MEMORY_EXPLORE: &str = "memory_explore"; +/// `memory_store`: store a learning, a document or a conversation. +pub const MEMORY_STORE: &str = "memory_store"; +/// `memory_forget`: remove memories by id or by a non-empty filter. +pub const MEMORY_FORGET: &str = "memory_forget"; + +/// Every tool name, reads first, in the order [`super::MemoryTools::specs`] +/// lists them. +pub const TOOL_NAMES: [&str; 7] = [ + MEMORY_RECALL, + MEMORY_FETCH, + MEMORY_LIST, + MEMORY_GET, + MEMORY_EXPLORE, + MEMORY_STORE, + MEMORY_FORGET, +]; + +/// The tools that change what memory holds; a read-only +/// [`super::MemoryTools`] neither lists nor runs them. +pub const WRITE_TOOL_NAMES: [&str; 2] = [MEMORY_STORE, MEMORY_FORGET]; + +/// One tool as a model sees it. +#[derive(Debug, Clone, PartialEq, Serialize)] +pub struct ToolSpec { + /// The tool's name, one of [`TOOL_NAMES`]. + pub name: &'static str, + /// What the tool does and when to use it, written for the model. + pub description: &'static str, + /// A JSON Schema object describing the arguments. Every object in it sets + /// `additionalProperties: false`. + pub parameters: serde_json::Value, +} + +/// Whether `name` is one of the write tools. +pub(crate) fn is_write_tool(name: &str) -> bool { + WRITE_TOOL_NAMES.contains(&name) +} + +/// The specs for an engine serving `fetch_modes`, with the write tools only +/// when `writes`. `memory_fetch` is left out when the engine serves no mode. +pub(crate) fn specs(fetch_modes: &[FetchMode], writes: bool) -> Vec { + let mut specs = vec![ToolSpec { + name: MEMORY_RECALL, + description: "Answer a question from long-term memory. Returns a synthesised answer \ + and the memories it cites. Use this first when you need to know what \ + is remembered about something.", + parameters: schema::recall(), + }]; + if !fetch_modes.is_empty() { + specs.push(ToolSpec { + name: MEMORY_FETCH, + description: "Search long-term memory and return the raw matching memories, best \ + first. Use it when you need the stored text itself rather than an \ + answer. Pass `cursor` from a previous result to get the next page.", + parameters: schema::fetch(fetch_modes), + }); + } + specs.extend([ + ToolSpec { + name: MEMORY_LIST, + description: "Page through stored memories without a query, optionally narrowed \ + by a filter. Pass `cursor` from a previous result to get the next \ + page.", + parameters: schema::list(), + }, + ToolSpec { + name: MEMORY_GET, + description: "Read memories whole by id (ids come from recall citations, fetch \ + and list results). Ids that name nothing you can see are reported \ + as missing.", + parameters: schema::get(), + }, + ToolSpec { + name: MEMORY_EXPLORE, + description: "Count stored memories per value of one facet (kind, source, \ + folder, thread, tag, ...), largest first, to see what memory \ + holds before narrowing a filter.", + parameters: schema::explore(), + }, + ]); + if writes { + specs.extend([ + ToolSpec { + name: MEMORY_STORE, + description: "Store one memory: exactly one of `learning` (a distilled \ + statement worth remembering: a preference, fact, procedure or \ + correction), `document` (a titled text) or `conversation` \ + (ordered turns). Storing the same memory twice is harmless.", + parameters: schema::store(), + }, + ToolSpec { + name: MEMORY_FORGET, + description: "Remove memories, either by `ids` or by a non-empty `filter` \ + (never both). Ids you cannot see are skipped and reported.", + parameters: schema::forget(), + }, + ]); + } + specs +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/spec/mod_tests.rs b/crates/tinymemory-tools/src/tools/spec/mod_tests.rs new file mode 100644 index 00000000..b4d837e8 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/spec/mod_tests.rs @@ -0,0 +1,103 @@ +//! Tool specs: names, the read-only subset, closed schemas, and the fetch +//! mode enum following the engine. + +use super::*; +use serde_json::Value; + +fn names(specs: &[ToolSpec]) -> Vec<&'static str> { + specs.iter().map(|spec| spec.name).collect() +} + +fn find<'a>(specs: &'a [ToolSpec], name: &str) -> &'a ToolSpec { + specs + .iter() + .find(|spec| spec.name == name) + .unwrap_or_else(|| panic!("no spec named {name}")) +} + +/// Every object schema reachable from `schema`, with the path to it. +fn objects<'a>(schema: &'a Value, path: String, out: &mut Vec<(String, &'a Value)>) { + if schema.get("type") == Some(&Value::from("object")) { + out.push((path.clone(), schema)); + } + if let Some(properties) = schema.get("properties").and_then(Value::as_object) { + for (name, property) in properties { + objects(property, format!("{path}.{name}"), out); + } + } + if let Some(items) = schema.get("items") { + objects(items, format!("{path}[]"), out); + } +} + +#[test] +fn with_writes_every_tool_is_listed_in_order() { + let specs = specs(&FetchMode::ALL, true); + assert_eq!(names(&specs), TOOL_NAMES); +} + +#[test] +fn without_writes_the_write_tools_are_left_out() { + let specs = specs(&FetchMode::ALL, false); + let listed = names(&specs); + assert_eq!(listed.len(), TOOL_NAMES.len() - WRITE_TOOL_NAMES.len()); + assert!(WRITE_TOOL_NAMES.iter().all(|name| !listed.contains(name))); + assert!(is_write_tool(MEMORY_STORE) && is_write_tool(MEMORY_FORGET)); + assert!(!is_write_tool(MEMORY_RECALL)); +} + +#[test] +fn every_object_is_closed_and_none_names_a_namespace_or_reach() { + for spec in specs(&FetchMode::ALL, true) { + let mut found = Vec::new(); + objects(&spec.parameters, spec.name.to_string(), &mut found); + assert!(!found.is_empty(), "{} has no object schema", spec.name); + for (path, object) in found { + assert_eq!( + object.get("additionalProperties"), + Some(&Value::Bool(false)), + "{path} is not closed" + ); + let properties = object["properties"].as_object().expect("properties"); + for forbidden in ["namespace", "reach"] { + assert!( + !properties.contains_key(forbidden), + "{path} names {forbidden}" + ); + } + } + } +} + +#[test] +fn the_fetch_mode_enum_lists_exactly_the_engine_modes() { + let keyword_only = specs(&[FetchMode::Keyword], true); + let mode = &find(&keyword_only, MEMORY_FETCH).parameters["properties"]["mode"]; + assert_eq!(mode["enum"], serde_json::json!(["keyword"])); + assert_eq!(mode["default"], serde_json::json!("keyword")); + + let all = specs(&FetchMode::ALL, true); + let mode = &find(&all, MEMORY_FETCH).parameters["properties"]["mode"]; + assert_eq!( + mode["enum"], + serde_json::json!(["keyword", "vector", "hybrid"]) + ); + assert_eq!(mode["default"], serde_json::json!("hybrid")); +} + +#[test] +fn an_engine_without_fetch_modes_gets_no_fetch_tool() { + assert!(!names(&specs(&[], true)).contains(&MEMORY_FETCH)); +} + +#[test] +fn the_explore_facets_leave_out_the_namespace() { + let all = specs(&FetchMode::ALL, true); + let facets = &find(&all, MEMORY_EXPLORE).parameters["properties"]["facet"]["enum"]; + assert!( + !facets + .as_array() + .expect("enum") + .contains(&Value::from("namespace")) + ); +} diff --git a/crates/tinymemory-tools/src/tools/spec/schema.rs b/crates/tinymemory-tools/src/tools/spec/schema.rs new file mode 100644 index 00000000..ee53df1b --- /dev/null +++ b/crates/tinymemory-tools/src/tools/spec/schema.rs @@ -0,0 +1,366 @@ +//! The JSON Schema for each tool's arguments, and the limits and defaults the +//! schemas advertise and the argument parser enforces. +//! +//! Every object sets `additionalProperties: false`, and no schema has a +//! `namespace` or `reach` property: those are fixed by the host. + +use serde_json::{Map, Value, json}; +use tinymemory_api::explore::MAX_GET_IDS; +use tinymemory_api::{FetchMode, ItemKind, SourceKind}; + +/// The `limit` a read uses when the model passes none. +pub(crate) const DEFAULT_LIMIT: usize = 10; + +/// The largest `limit` a read accepts. +pub(crate) const MAX_LIMIT: usize = 50; + +/// The most ids one `memory_get` or `memory_forget` call names. +pub(crate) const MAX_IDS: usize = MAX_GET_IDS; + +/// A learning's confidence when the model gives none. +pub(crate) const DEFAULT_CONFIDENCE: f32 = 0.8; + +/// A learning's kind when the model gives none. +pub(crate) const DEFAULT_LEARNING_KIND: &str = "fact"; + +/// The facets `memory_explore` groups by. The namespace facet is left out on +/// purpose: the namespace is the host's, not the model's. +pub(crate) const EXPLORE_FACETS: [&str; 13] = [ + "kind", + "source", + "source_id", + "workspace", + "folder", + "file_path", + "language", + "repo", + "url", + "thread", + "agent", + "tool_call", + "tag", +]; + +/// The learning kinds a model may name. +pub(crate) const LEARNING_KINDS: [&str; 5] = + ["preference", "fact", "procedure", "correction", "other"]; + +/// The conversation roles a model may name. +pub(crate) const ROLES: [&str; 4] = ["user", "assistant", "system", "tool"]; + +/// The fields of the model-facing filter, a subset of +/// [`tinymemory_api::MetaFilter`]. +pub(crate) const FILTER_FIELDS: [&str; 12] = [ + "kinds", + "sources", + "tags_any", + "workspace", + "folder", + "file_path", + "repo", + "url", + "thread_id", + "agent_id", + "observed_after", + "observed_before", +]; + +/// The fetch mode used when the model names none: hybrid when the engine +/// serves it, otherwise the first mode it lists. +pub(crate) fn default_mode(modes: &[FetchMode]) -> Option { + if modes.contains(&FetchMode::Hybrid) { + Some(FetchMode::Hybrid) + } else { + modes.first().copied() + } +} + +/// `memory_recall`'s arguments. +pub(crate) fn recall() -> Value { + object( + [ + ( + "question", + text("The question to answer, in natural language."), + ), + ("filter", filter()), + ( + "limit", + limit("Most memories the answer may cite.", DEFAULT_LIMIT), + ), + ( + "instructions", + text("Optional extra instructions for how to answer (length, format, focus)."), + ), + ], + &["question"], + ) +} + +/// `memory_fetch`'s arguments; `mode` lists exactly `modes`. +pub(crate) fn fetch(modes: &[FetchMode]) -> Value { + let names: Vec<&str> = modes.iter().map(|mode| mode.as_str()).collect(); + let mut mode = json!({ + "type": "string", + "enum": names, + "description": "How to rank: `keyword` (lexical match), `vector` (meaning) or \ + `hybrid` (both), as this memory offers them.", + }); + if let (Some(default), Some(schema)) = (default_mode(modes), mode.as_object_mut()) { + schema.insert("default".to_string(), json!(default.as_str())); + } + object( + [ + ("query", text("What to search for.")), + ("mode", mode), + ("filter", filter()), + ("limit", limit("Most memories to return.", DEFAULT_LIMIT)), + ("cursor", cursor()), + ], + &["query"], + ) +} + +/// `memory_list`'s arguments. +pub(crate) fn list() -> Value { + object( + [ + ("filter", filter()), + ("limit", limit("Most memories to return.", DEFAULT_LIMIT)), + ("cursor", cursor()), + ], + &[], + ) +} + +/// `memory_get`'s arguments. +pub(crate) fn get() -> Value { + object([("ids", ids("The ids of the memories to read."))], &["ids"]) +} + +/// `memory_explore`'s arguments. +pub(crate) fn explore() -> Value { + object( + [ + ( + "facet", + json!({ + "type": "string", + "enum": EXPLORE_FACETS, + "description": "The dimension to group memories by.", + }), + ), + ("filter", filter()), + ("limit", limit("Most values to return.", DEFAULT_LIMIT)), + ], + &["facet"], + ) +} + +/// `memory_store`'s arguments. +pub(crate) fn store() -> Value { + let learning = object( + [ + ("text", text("The statement, self-contained and specific.")), + ( + "learning_kind", + json!({ + "type": "string", + "enum": LEARNING_KINDS, + "default": DEFAULT_LEARNING_KIND, + "description": "What kind of statement it is.", + }), + ), + ( + "confidence", + json!({ + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": DEFAULT_CONFIDENCE, + "description": "How sure you are, from 0 to 1.", + }), + ), + ("evidence", text("What supports the statement.")), + ], + &["text"], + ); + let document = object( + [ + ("title", text("The document's title.")), + ("text", text("The document's body, normally markdown.")), + ], + &["text"], + ); + let turn = object( + [ + ( + "role", + json!({ "type": "string", "enum": ROLES, "description": "Who spoke." }), + ), + ("text", text("What was said.")), + ], + &["role", "text"], + ); + let conversation = object( + [( + "turns", + json!({ + "type": "array", + "items": turn, + "minItems": 1, + "description": "The turns, in order.", + }), + )], + &["turns"], + ); + object( + [ + ( + "learning", + described(learning, "A distilled statement worth remembering."), + ), + ("document", described(document, "A text to remember whole.")), + ( + "conversation", + described(conversation, "An exchange to remember."), + ), + ("tags", strings("Free-form tags to file the memory under.")), + ], + &[], + ) +} + +/// `memory_forget`'s arguments. +pub(crate) fn forget() -> Value { + object( + [ + ("ids", ids("The ids of the memories to remove.")), + ( + "filter", + described( + filter(), + "Remove every memory matching this filter; it must set at least one field.", + ), + ), + ], + &[], + ) +} + +/// The model-facing filter: [`FILTER_FIELDS`], never a namespace or reach. +fn filter() -> Value { + let kinds: Vec<&str> = ItemKind::ALL.iter().map(|kind| kind.as_str()).collect(); + let sources: Vec<&str> = SourceKind::ALL.iter().map(|kind| kind.as_str()).collect(); + let mut schema = object( + [ + ( + "kinds", + enum_list(&kinds, "Only these kinds of memory; empty means all."), + ), + ( + "sources", + enum_list( + &sources, + "Only memories from these kinds of source; empty means all.", + ), + ), + ( + "tags_any", + strings("Only memories carrying at least one of these tags."), + ), + ("workspace", text("Exact workspace.")), + ("folder", text("Folder, exact or as a path prefix.")), + ("file_path", text("File path, exact or as a path prefix.")), + ("repo", text("Exact repository, as `owner/name` or a URL.")), + ("url", text("Exact URL the memory was read from.")), + ("thread_id", text("Exact conversation thread id.")), + ("agent_id", text("Exact id of the agent that produced it.")), + ( + "observed_after", + timestamp("Only memories observed at or after this RFC 3339 time."), + ), + ( + "observed_before", + timestamp("Only memories observed before this RFC 3339 time."), + ), + ], + &[], + ); + if let Some(map) = schema.as_object_mut() { + map.insert( + "description".to_string(), + json!("Narrows which memories are considered; every field set must match."), + ); + } + schema +} + +/// A closed object with these properties, `required` listed only when +/// non-empty. +fn object(properties: [(&str, Value); N], required: &[&str]) -> Value { + let properties: Map = properties + .into_iter() + .map(|(name, schema)| (name.to_string(), schema)) + .collect(); + let mut schema = json!({ + "type": "object", + "properties": properties, + "additionalProperties": false, + }); + if let (false, Some(map)) = (required.is_empty(), schema.as_object_mut()) { + map.insert("required".to_string(), json!(required)); + } + schema +} + +fn described(mut schema: Value, description: &str) -> Value { + if let Some(map) = schema.as_object_mut() { + map.insert("description".to_string(), json!(description)); + } + schema +} + +fn text(description: &str) -> Value { + json!({ "type": "string", "description": description }) +} + +fn timestamp(description: &str) -> Value { + json!({ "type": "string", "format": "date-time", "description": description }) +} + +fn strings(description: &str) -> Value { + json!({ "type": "array", "items": { "type": "string" }, "description": description }) +} + +fn enum_list(values: &[&str], description: &str) -> Value { + json!({ + "type": "array", + "items": { "type": "string", "enum": values }, + "description": description, + }) +} + +fn ids(description: &str) -> Value { + json!({ + "type": "array", + "items": { "type": "string" }, + "minItems": 1, + "maxItems": MAX_IDS, + "description": description, + }) +} + +fn limit(description: &str, default: usize) -> Value { + json!({ + "type": "integer", + "minimum": 1, + "maximum": MAX_LIMIT, + "default": default, + "description": description, + }) +} + +fn cursor() -> Value { + text("The `next_cursor` of a previous result, to continue from it.") +} diff --git a/crates/tinymemory-tools/src/tools/write/mod.rs b/crates/tinymemory-tools/src/tools/write/mod.rs new file mode 100644 index 00000000..932ff7c2 --- /dev/null +++ b/crates/tinymemory-tools/src/tools/write/mod.rs @@ -0,0 +1,194 @@ +//! The write tools: `memory_store` and `memory_forget`. +//! +//! - `memory_store` stores exactly one learning, document or conversation. +//! Its metadata is built here, never read from the arguments: the +//! namespace is the scope's `place`, the source is +//! [`SourceKind::Agent`], the tags are the model's, and `observed_at` is +//! the time of the call. +//! - `memory_forget` removes by ids or by a non-empty filter. Ids are first +//! read back with [`MemoryEngine::get`] under the scope's reach, and only +//! those found are forgotten; the rest are reported as `skipped`, so an id +//! from outside the reach is never removed. A filter must set at least one +//! model-facing field before the reach is added (a reach alone would mean +//! "everything in reach"), and is then confined to the reach. + +use chrono::Utc; +use serde_json::Value; +use tinymemory_api::{ + DocumentBody, ForgetReport, ForgetTarget, ItemId, LearningKind, MemoryEngine, MemoryMeta, + Result, Role, SourceKind, SourceRef, StoreItem, Turn, +}; + +use super::ToolScope; +use super::args::{Args, invalid, meta_filter}; +use super::read::{item_ids, resolve}; +use super::render; +use super::spec::schema::{DEFAULT_CONFIDENCE, DEFAULT_LEARNING_KIND, LEARNING_KINDS, ROLES}; +use super::spec::{MEMORY_FORGET, MEMORY_STORE}; + +/// The three shapes `memory_store` takes, exactly one per call. +const STORE_SHAPES: [&str; 3] = ["learning", "document", "conversation"]; + +/// `memory_store`. +/// +/// # Errors +/// +/// Invalid arguments (none or several of the three shapes among them), an +/// item the engine refuses as invalid, and the engine's own failures. +pub(crate) async fn store( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse( + MEMORY_STORE, + value, + &["learning", "document", "conversation", "tags"], + )?; + let shapes: Vec<&str> = STORE_SHAPES + .into_iter() + .filter(|shape| args.has(shape)) + .collect(); + let [shape] = shapes.as_slice() else { + return Err(invalid( + MEMORY_STORE, + "pass exactly one of `learning`, `document` or `conversation`", + )); + }; + let meta = MemoryMeta { + namespace: scope.place.clone(), + source: SourceRef { + kind: SourceKind::Agent, + id: None, + }, + tags: args.strings("tags")?, + observed_at: Some(Utc::now()), + ..MemoryMeta::default() + }; + let item = match *shape { + "learning" => learning(&args, meta)?, + "document" => document(&args, meta)?, + _ => conversation(&args, meta)?, + }; + Ok(render::store(&engine.store(item).await?)) +} + +/// `memory_forget`. +/// +/// # Errors +/// +/// Invalid arguments (neither or both of `ids` and `filter`, or a filter that +/// sets nothing), and the engine's own failures. +pub(crate) async fn forget( + engine: &dyn MemoryEngine, + scope: &ToolScope, + value: &Value, +) -> Result { + let args = Args::parse(MEMORY_FORGET, value, &["ids", "filter"])?; + match (args.has("ids"), args.has("filter")) { + (true, false) => { + let ids = item_ids(&args, "ids")?; + let found: Vec = resolve(engine, scope, &ids) + .await? + .into_iter() + .map(|hit| hit.id) + .collect(); + let skipped: Vec = ids.into_iter().filter(|id| !found.contains(id)).collect(); + let report = if found.is_empty() { + ForgetReport::default() + } else { + engine.forget(ForgetTarget::Ids(found)).await? + }; + Ok(render::forget(&report, &skipped)) + } + (false, true) => { + let mut filter = meta_filter(&args, "filter")?; + if filter.is_empty() { + return Err(args.field_error( + "filter", + "must set at least one field; an empty filter would mean everything", + )); + } + scope.confine(&mut filter); + let report = engine.forget(ForgetTarget::Filter(filter)).await?; + Ok(render::forget(&report, &[])) + } + _ => Err(invalid( + MEMORY_FORGET, + "pass exactly one of `ids` or `filter`", + )), + } +} + +fn learning(args: &Args<'_>, meta: MemoryMeta) -> Result { + let Some(learning) = args.object( + "learning", + "learning.", + &["text", "learning_kind", "confidence", "evidence"], + )? + else { + return Err(args.field_error("learning", "is required")); + }; + let kind = learning + .string("learning_kind")? + .unwrap_or_else(|| DEFAULT_LEARNING_KIND.to_string()); + Ok(StoreItem::Learning { + text: learning.required_string("text")?, + kind: learning_kind(&kind).ok_or_else(|| { + learning.field_error( + "learning_kind", + &format!("must be one of {}", LEARNING_KINDS.join(", ")), + ) + })?, + confidence: learning.unit("confidence", DEFAULT_CONFIDENCE)?, + evidence: learning.string("evidence")?, + meta, + }) +} + +fn document(args: &Args<'_>, meta: MemoryMeta) -> Result { + let Some(document) = args.object("document", "document.", &["title", "text"])? else { + return Err(args.field_error("document", "is required")); + }; + Ok(StoreItem::Document { + title: document.string("title")?, + body: DocumentBody::Text(document.required_string("text")?), + mime: None, + meta, + }) +} + +fn conversation(args: &Args<'_>, meta: MemoryMeta) -> Result { + let Some(conversation) = args.object("conversation", "conversation.", &["turns"])? else { + return Err(args.field_error("conversation", "is required")); + }; + let raw = conversation.array("turns")?; + if raw.is_empty() { + return Err(conversation.field_error("turns", "must hold at least one turn")); + } + let turns = raw + .iter() + .map(|value| turn(&conversation, value)) + .collect::>>()?; + Ok(StoreItem::Conversation { turns, meta }) +} + +fn turn(conversation: &Args<'_>, value: &Value) -> Result { + let turn = conversation.element("turns", value, "conversation.turns[].", &["role", "text"])?; + let role_name = turn.required_string("role")?; + let role = role(&role_name) + .ok_or_else(|| turn.field_error("role", &format!("must be one of {}", ROLES.join(", "))))?; + Ok(Turn::new(role, turn.required_string("text")?)) +} + +fn learning_kind(name: &str) -> Option { + serde_json::from_value(Value::String(name.to_string())).ok() +} + +fn role(name: &str) -> Option { + serde_json::from_value(Value::String(name.to_string())).ok() +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-tools/src/tools/write/mod_tests.rs b/crates/tinymemory-tools/src/tools/write/mod_tests.rs new file mode 100644 index 00000000..0899789e --- /dev/null +++ b/crates/tinymemory-tools/src/tools/write/mod_tests.rs @@ -0,0 +1,192 @@ +//! The write tools against the reference engine: shapes, metadata, and +//! forget's id resolution and filter rules. + +use super::*; +use serde_json::json; +use tinymemory_api::conformance::ReferenceEngine; +use tinymemory_api::{Error, ListRequest, MetaFilter, Namespace}; + +fn scope() -> ToolScope { + ToolScope::at(Namespace::agent("writer")) +} + +fn invalid(result: Result) -> String { + match result { + Err(Error::InvalidRequest(message)) => message, + other => panic!("expected an invalid request, got {other:?}"), + } +} + +async fn everything(engine: &ReferenceEngine) -> Vec { + engine + .list(ListRequest::new(MetaFilter::default(), 100)) + .await + .unwrap() + .items +} + +#[tokio::test] +async fn store_needs_exactly_one_shape() { + let engine = ReferenceEngine::new(); + for value in [ + json!({}), + json!({ "tags": ["x"] }), + json!({ "learning": { "text": "a" }, "document": { "text": "b" } }), + ] { + let text = invalid(store(&engine, &scope(), &value).await); + assert!(text.contains("exactly one of"), "{text}"); + } + assert!(engine.is_empty()); +} + +#[tokio::test] +async fn store_builds_the_metadata_itself() { + let engine = ReferenceEngine::new(); + let result = store( + &engine, + &scope(), + &json!({ "learning": { "text": "likes tea", "learning_kind": "preference", + "confidence": 0.5, "evidence": "said so" }, + "tags": ["drink"] }), + ) + .await + .unwrap(); + assert_eq!(result["replayed"], json!(false)); + let stored = everything(&engine).await; + let meta = &stored[0].meta; + assert_eq!(meta.namespace, Namespace::agent("writer")); + assert_eq!(meta.source.kind, SourceKind::Agent); + assert_eq!(meta.tags, ["drink"]); + assert!(meta.observed_at.is_some()); + assert_eq!(stored[0].confidence, Some(0.5)); +} + +#[tokio::test] +async fn store_takes_documents_and_conversations() { + let engine = ReferenceEngine::new(); + store( + &engine, + &scope(), + &json!({ "document": { "title": "T", "text": "body" } }), + ) + .await + .unwrap(); + store( + &engine, + &scope(), + &json!({ "conversation": { "turns": [ + { "role": "user", "text": "hi" }, { "role": "assistant", "text": "hello" } + ] } }), + ) + .await + .unwrap(); + let texts: Vec = everything(&engine) + .await + .into_iter() + .map(|h| h.text) + .collect(); + assert_eq!(texts, ["# T\n\nbody", "user: hi\nassistant: hello"]); +} + +#[tokio::test] +async fn store_refuses_bad_shapes_by_field() { + let engine = ReferenceEngine::new(); + let cases = [ + ( + json!({ "learning": { "text": "x", "learning_kind": "rumour" } }), + "`learning.learning_kind`", + ), + ( + json!({ "learning": { "text": "x", "confidence": 2 } }), + "`learning.confidence`", + ), + (json!({ "learning": { "text": " " } }), "`learning.text`"), + (json!({ "learning": "x" }), "`learning` must be an object"), + ( + json!({ "document": { "title": "no body" } }), + "`document.text`", + ), + ( + json!({ "conversation": { "turns": [] } }), + "`conversation.turns`", + ), + ( + json!({ "conversation": { "turns": [{ "role": "robot", "text": "x" }] } }), + "`conversation.turns[].role`", + ), + ( + json!({ "conversation": { "turns": [{ "role": "user" }] } }), + "`conversation.turns[].text`", + ), + ( + json!({ "learning": { "text": "x", "namespace": "root" } }), + "`learning.namespace` is fixed by the host", + ), + ]; + for (value, expected) in cases { + let text = invalid(store(&engine, &scope(), &value).await); + assert!(text.contains(expected), "{value}: {text}"); + } +} + +#[tokio::test] +async fn forget_needs_exactly_one_target_and_a_non_empty_filter() { + let engine = ReferenceEngine::new(); + for value in [ + json!({}), + json!({ "ids": ["a"], "filter": { "tags_any": ["x"] } }), + ] { + assert!(invalid(forget(&engine, &scope(), &value).await).contains("exactly one of")); + } + let text = invalid(forget(&engine, &scope(), &json!({ "filter": {} })).await); + assert!( + text.contains("`filter` must set at least one field"), + "{text}" + ); +} + +#[tokio::test] +async fn forget_by_ids_skips_what_is_out_of_reach() { + let engine = ReferenceEngine::new(); + let mine = store( + &engine, + &scope(), + &json!({ "learning": { "text": "mine" } }), + ) + .await + .unwrap(); + let other = ToolScope::at(Namespace::agent("other")); + let theirs = store( + &engine, + &other, + &json!({ "learning": { "text": "theirs" } }), + ) + .await + .unwrap(); + let result = forget( + &engine, + &scope(), + &json!({ "ids": [mine["id"], theirs["id"], "nothing"] }), + ) + .await + .unwrap(); + assert_eq!( + result, + json!({ "forgotten": 1, "skipped": [theirs["id"], "nothing"] }) + ); + let left: Vec = everything(&engine) + .await + .into_iter() + .map(|h| h.text) + .collect(); + assert_eq!(left, ["theirs"]); +} + +#[tokio::test] +async fn forget_by_ids_with_nothing_in_reach_forgets_nothing() { + let engine = ReferenceEngine::new(); + let result = forget(&engine, &scope(), &json!({ "ids": ["nothing"] })) + .await + .unwrap(); + assert_eq!(result, json!({ "forgotten": 0, "skipped": ["nothing"] })); +} diff --git a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json new file mode 100644 index 00000000..1a595e8d --- /dev/null +++ b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json @@ -0,0 +1,690 @@ +[ + { + "description": "Answer a question from long-term memory. Returns a synthesised answer and the memories it cites. Use this first when you need to know what is remembered about something.", + "name": "memory_recall", + "parameters": { + "additionalProperties": false, + "properties": { + "filter": { + "additionalProperties": false, + "description": "Narrows which memories are considered; every field set must match.", + "properties": { + "agent_id": { + "description": "Exact id of the agent that produced it.", + "type": "string" + }, + "file_path": { + "description": "File path, exact or as a path prefix.", + "type": "string" + }, + "folder": { + "description": "Folder, exact or as a path prefix.", + "type": "string" + }, + "kinds": { + "description": "Only these kinds of memory; empty means all.", + "items": { + "enum": [ + "document", + "conversation", + "learning" + ], + "type": "string" + }, + "type": "array" + }, + "observed_after": { + "description": "Only memories observed at or after this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "observed_before": { + "description": "Only memories observed before this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "repo": { + "description": "Exact repository, as `owner/name` or a URL.", + "type": "string" + }, + "sources": { + "description": "Only memories from these kinds of source; empty means all.", + "items": { + "enum": [ + "folder", + "file", + "link", + "github", + "rss", + "composio", + "conversation", + "agent", + "import" + ], + "type": "string" + }, + "type": "array" + }, + "tags_any": { + "description": "Only memories carrying at least one of these tags.", + "items": { + "type": "string" + }, + "type": "array" + }, + "thread_id": { + "description": "Exact conversation thread id.", + "type": "string" + }, + "url": { + "description": "Exact URL the memory was read from.", + "type": "string" + }, + "workspace": { + "description": "Exact workspace.", + "type": "string" + } + }, + "type": "object" + }, + "instructions": { + "description": "Optional extra instructions for how to answer (length, format, focus).", + "type": "string" + }, + "limit": { + "default": 10, + "description": "Most memories the answer may cite.", + "maximum": 50, + "minimum": 1, + "type": "integer" + }, + "question": { + "description": "The question to answer, in natural language.", + "type": "string" + } + }, + "required": [ + "question" + ], + "type": "object" + } + }, + { + "description": "Search long-term memory and return the raw matching memories, best first. Use it when you need the stored text itself rather than an answer. Pass `cursor` from a previous result to get the next page.", + "name": "memory_fetch", + "parameters": { + "additionalProperties": false, + "properties": { + "cursor": { + "description": "The `next_cursor` of a previous result, to continue from it.", + "type": "string" + }, + "filter": { + "additionalProperties": false, + "description": "Narrows which memories are considered; every field set must match.", + "properties": { + "agent_id": { + "description": "Exact id of the agent that produced it.", + "type": "string" + }, + "file_path": { + "description": "File path, exact or as a path prefix.", + "type": "string" + }, + "folder": { + "description": "Folder, exact or as a path prefix.", + "type": "string" + }, + "kinds": { + "description": "Only these kinds of memory; empty means all.", + "items": { + "enum": [ + "document", + "conversation", + "learning" + ], + "type": "string" + }, + "type": "array" + }, + "observed_after": { + "description": "Only memories observed at or after this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "observed_before": { + "description": "Only memories observed before this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "repo": { + "description": "Exact repository, as `owner/name` or a URL.", + "type": "string" + }, + "sources": { + "description": "Only memories from these kinds of source; empty means all.", + "items": { + "enum": [ + "folder", + "file", + "link", + "github", + "rss", + "composio", + "conversation", + "agent", + "import" + ], + "type": "string" + }, + "type": "array" + }, + "tags_any": { + "description": "Only memories carrying at least one of these tags.", + "items": { + "type": "string" + }, + "type": "array" + }, + "thread_id": { + "description": "Exact conversation thread id.", + "type": "string" + }, + "url": { + "description": "Exact URL the memory was read from.", + "type": "string" + }, + "workspace": { + "description": "Exact workspace.", + "type": "string" + } + }, + "type": "object" + }, + "limit": { + "default": 10, + "description": "Most memories to return.", + "maximum": 50, + "minimum": 1, + "type": "integer" + }, + "mode": { + "default": "hybrid", + "description": "How to rank: `keyword` (lexical match), `vector` (meaning) or `hybrid` (both), as this memory offers them.", + "enum": [ + "keyword", + "vector", + "hybrid" + ], + "type": "string" + }, + "query": { + "description": "What to search for.", + "type": "string" + } + }, + "required": [ + "query" + ], + "type": "object" + } + }, + { + "description": "Page through stored memories without a query, optionally narrowed by a filter. Pass `cursor` from a previous result to get the next page.", + "name": "memory_list", + "parameters": { + "additionalProperties": false, + "properties": { + "cursor": { + "description": "The `next_cursor` of a previous result, to continue from it.", + "type": "string" + }, + "filter": { + "additionalProperties": false, + "description": "Narrows which memories are considered; every field set must match.", + "properties": { + "agent_id": { + "description": "Exact id of the agent that produced it.", + "type": "string" + }, + "file_path": { + "description": "File path, exact or as a path prefix.", + "type": "string" + }, + "folder": { + "description": "Folder, exact or as a path prefix.", + "type": "string" + }, + "kinds": { + "description": "Only these kinds of memory; empty means all.", + "items": { + "enum": [ + "document", + "conversation", + "learning" + ], + "type": "string" + }, + "type": "array" + }, + "observed_after": { + "description": "Only memories observed at or after this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "observed_before": { + "description": "Only memories observed before this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "repo": { + "description": "Exact repository, as `owner/name` or a URL.", + "type": "string" + }, + "sources": { + "description": "Only memories from these kinds of source; empty means all.", + "items": { + "enum": [ + "folder", + "file", + "link", + "github", + "rss", + "composio", + "conversation", + "agent", + "import" + ], + "type": "string" + }, + "type": "array" + }, + "tags_any": { + "description": "Only memories carrying at least one of these tags.", + "items": { + "type": "string" + }, + "type": "array" + }, + "thread_id": { + "description": "Exact conversation thread id.", + "type": "string" + }, + "url": { + "description": "Exact URL the memory was read from.", + "type": "string" + }, + "workspace": { + "description": "Exact workspace.", + "type": "string" + } + }, + "type": "object" + }, + "limit": { + "default": 10, + "description": "Most memories to return.", + "maximum": 50, + "minimum": 1, + "type": "integer" + } + }, + "type": "object" + } + }, + { + "description": "Read memories whole by id (ids come from recall citations, fetch and list results). Ids that name nothing you can see are reported as missing.", + "name": "memory_get", + "parameters": { + "additionalProperties": false, + "properties": { + "ids": { + "description": "The ids of the memories to read.", + "items": { + "type": "string" + }, + "maxItems": 200, + "minItems": 1, + "type": "array" + } + }, + "required": [ + "ids" + ], + "type": "object" + } + }, + { + "description": "Count stored memories per value of one facet (kind, source, folder, thread, tag, ...), largest first, to see what memory holds before narrowing a filter.", + "name": "memory_explore", + "parameters": { + "additionalProperties": false, + "properties": { + "facet": { + "description": "The dimension to group memories by.", + "enum": [ + "kind", + "source", + "source_id", + "workspace", + "folder", + "file_path", + "language", + "repo", + "url", + "thread", + "agent", + "tool_call", + "tag" + ], + "type": "string" + }, + "filter": { + "additionalProperties": false, + "description": "Narrows which memories are considered; every field set must match.", + "properties": { + "agent_id": { + "description": "Exact id of the agent that produced it.", + "type": "string" + }, + "file_path": { + "description": "File path, exact or as a path prefix.", + "type": "string" + }, + "folder": { + "description": "Folder, exact or as a path prefix.", + "type": "string" + }, + "kinds": { + "description": "Only these kinds of memory; empty means all.", + "items": { + "enum": [ + "document", + "conversation", + "learning" + ], + "type": "string" + }, + "type": "array" + }, + "observed_after": { + "description": "Only memories observed at or after this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "observed_before": { + "description": "Only memories observed before this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "repo": { + "description": "Exact repository, as `owner/name` or a URL.", + "type": "string" + }, + "sources": { + "description": "Only memories from these kinds of source; empty means all.", + "items": { + "enum": [ + "folder", + "file", + "link", + "github", + "rss", + "composio", + "conversation", + "agent", + "import" + ], + "type": "string" + }, + "type": "array" + }, + "tags_any": { + "description": "Only memories carrying at least one of these tags.", + "items": { + "type": "string" + }, + "type": "array" + }, + "thread_id": { + "description": "Exact conversation thread id.", + "type": "string" + }, + "url": { + "description": "Exact URL the memory was read from.", + "type": "string" + }, + "workspace": { + "description": "Exact workspace.", + "type": "string" + } + }, + "type": "object" + }, + "limit": { + "default": 10, + "description": "Most values to return.", + "maximum": 50, + "minimum": 1, + "type": "integer" + } + }, + "required": [ + "facet" + ], + "type": "object" + } + }, + { + "description": "Store one memory: exactly one of `learning` (a distilled statement worth remembering: a preference, fact, procedure or correction), `document` (a titled text) or `conversation` (ordered turns). Storing the same memory twice is harmless.", + "name": "memory_store", + "parameters": { + "additionalProperties": false, + "properties": { + "conversation": { + "additionalProperties": false, + "description": "An exchange to remember.", + "properties": { + "turns": { + "description": "The turns, in order.", + "items": { + "additionalProperties": false, + "properties": { + "role": { + "description": "Who spoke.", + "enum": [ + "user", + "assistant", + "system", + "tool" + ], + "type": "string" + }, + "text": { + "description": "What was said.", + "type": "string" + } + }, + "required": [ + "role", + "text" + ], + "type": "object" + }, + "minItems": 1, + "type": "array" + } + }, + "required": [ + "turns" + ], + "type": "object" + }, + "document": { + "additionalProperties": false, + "description": "A text to remember whole.", + "properties": { + "text": { + "description": "The document's body, normally markdown.", + "type": "string" + }, + "title": { + "description": "The document's title.", + "type": "string" + } + }, + "required": [ + "text" + ], + "type": "object" + }, + "learning": { + "additionalProperties": false, + "description": "A distilled statement worth remembering.", + "properties": { + "confidence": { + "default": 0.800000011920929, + "description": "How sure you are, from 0 to 1.", + "maximum": 1.0, + "minimum": 0.0, + "type": "number" + }, + "evidence": { + "description": "What supports the statement.", + "type": "string" + }, + "learning_kind": { + "default": "fact", + "description": "What kind of statement it is.", + "enum": [ + "preference", + "fact", + "procedure", + "correction", + "other" + ], + "type": "string" + }, + "text": { + "description": "The statement, self-contained and specific.", + "type": "string" + } + }, + "required": [ + "text" + ], + "type": "object" + }, + "tags": { + "description": "Free-form tags to file the memory under.", + "items": { + "type": "string" + }, + "type": "array" + } + }, + "type": "object" + } + }, + { + "description": "Remove memories, either by `ids` or by a non-empty `filter` (never both). Ids you cannot see are skipped and reported.", + "name": "memory_forget", + "parameters": { + "additionalProperties": false, + "properties": { + "filter": { + "additionalProperties": false, + "description": "Remove every memory matching this filter; it must set at least one field.", + "properties": { + "agent_id": { + "description": "Exact id of the agent that produced it.", + "type": "string" + }, + "file_path": { + "description": "File path, exact or as a path prefix.", + "type": "string" + }, + "folder": { + "description": "Folder, exact or as a path prefix.", + "type": "string" + }, + "kinds": { + "description": "Only these kinds of memory; empty means all.", + "items": { + "enum": [ + "document", + "conversation", + "learning" + ], + "type": "string" + }, + "type": "array" + }, + "observed_after": { + "description": "Only memories observed at or after this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "observed_before": { + "description": "Only memories observed before this RFC 3339 time.", + "format": "date-time", + "type": "string" + }, + "repo": { + "description": "Exact repository, as `owner/name` or a URL.", + "type": "string" + }, + "sources": { + "description": "Only memories from these kinds of source; empty means all.", + "items": { + "enum": [ + "folder", + "file", + "link", + "github", + "rss", + "composio", + "conversation", + "agent", + "import" + ], + "type": "string" + }, + "type": "array" + }, + "tags_any": { + "description": "Only memories carrying at least one of these tags.", + "items": { + "type": "string" + }, + "type": "array" + }, + "thread_id": { + "description": "Exact conversation thread id.", + "type": "string" + }, + "url": { + "description": "Exact URL the memory was read from.", + "type": "string" + }, + "workspace": { + "description": "Exact workspace.", + "type": "string" + } + }, + "type": "object" + }, + "ids": { + "description": "The ids of the memories to remove.", + "items": { + "type": "string" + }, + "maxItems": 200, + "minItems": 1, + "type": "array" + } + }, + "type": "object" + } + } +] diff --git a/crates/tinymemory-tools/tests/tool_contracts.rs b/crates/tinymemory-tools/tests/tool_contracts.rs new file mode 100644 index 00000000..992b172d --- /dev/null +++ b/crates/tinymemory-tools/tests/tool_contracts.rs @@ -0,0 +1,43 @@ +//! Freezes the tool names and argument schemas a model sees. +//! +//! `fixtures/tool_contracts.json` is the serialised `specs()` of writable +//! tools over the reference engine (every fetch mode). A schema change is a +//! change to what every host's model is told, so it must be deliberate. +//! To regenerate after an intended change: +//! +//! ```sh +//! BLESS_TOOL_CONTRACTS=1 cargo test -p tinymemory-tools --test tool_contracts +//! ``` + +use std::path::PathBuf; +use std::sync::Arc; + +use tinymemory_api::conformance::ReferenceEngine; +use tinymemory_tools::MemoryTools; + +fn fixture() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("tests/fixtures/tool_contracts.json") +} + +#[test] +fn specs_match_the_frozen_tool_contracts() { + let specs = MemoryTools::new(Arc::new(ReferenceEngine::new())).specs(); + let actual = serde_json::to_value(&specs).unwrap(); + if std::env::var_os("BLESS_TOOL_CONTRACTS").is_some() { + let mut text = serde_json::to_string_pretty(&actual).unwrap(); + text.push('\n'); + std::fs::write(fixture(), text).unwrap(); + return; + } + let frozen: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(fixture()).unwrap()).unwrap(); + assert!( + actual == frozen, + "the memory tool specs no longer match tests/fixtures/tool_contracts.json.\n\ + If the change is deliberate, regenerate the fixture with\n\ + `BLESS_TOOL_CONTRACTS=1 cargo test -p tinymemory-tools --test tool_contracts`\n\ + and review the diff: every host's model sees these schemas.\n\ + actual:\n{}", + serde_json::to_string_pretty(&actual).unwrap() + ); +} diff --git a/crates/tinymemory-tools/tests/tools_roundtrip.rs b/crates/tinymemory-tools/tests/tools_roundtrip.rs new file mode 100644 index 00000000..215c6205 --- /dev/null +++ b/crates/tinymemory-tools/tests/tools_roundtrip.rs @@ -0,0 +1,340 @@ +//! Every memory tool round-trips through `MemoryTools::call` against the +//! reference engine, and read-only tools neither list nor run the writes. + +use std::sync::Arc; + +use serde_json::{Value, json}; +use tinymemory_api::conformance::ReferenceEngine; +use tinymemory_api::{ + EngineDescriptor, EngineHealth, Error, FetchMode, FetchPage, FetchRequest, ForgetReport, + ForgetTarget, ListPage, ListRequest, MemoryEngine, RecallAnswer, RecallRequest, Result, + StoreItem, StoreReceipt, async_trait, +}; +use tinymemory_tools::{ + MEMORY_EXPLORE, MEMORY_FETCH, MEMORY_FORGET, MEMORY_GET, MEMORY_LIST, MEMORY_RECALL, + MEMORY_STORE, MemoryTools, TOOL_NAMES, WRITE_TOOL_NAMES, +}; + +fn tools() -> MemoryTools { + MemoryTools::new(Arc::new(ReferenceEngine::new())) +} + +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +async fn store(tools: &MemoryTools, args: Value) -> String { + let receipt = tools.call(MEMORY_STORE, args).await.unwrap(); + receipt["id"].as_str().unwrap().to_string() +} + +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +fn texts(items: &Value) -> Vec { + items + .as_array() + .unwrap() + .iter() + .map(|item| item["text"].as_str().unwrap().to_string()) + .collect() +} + +#[tokio::test] +async fn store_is_idempotent_and_reports_replays() { + let tools = tools(); + let args = json!({ "learning": { "text": "the user prefers tabs" } }); + let first = tools.call(MEMORY_STORE, args.clone()).await.unwrap(); + let again = tools.call(MEMORY_STORE, args).await.unwrap(); + assert_eq!(first["replayed"], json!(false)); + assert_eq!(again["replayed"], json!(true)); + assert_eq!(first["id"], again["id"]); +} + +#[tokio::test] +async fn recall_answers_with_citations() { + let tools = tools(); + store( + &tools, + json!({ "learning": { "text": "rust ownership moves values" }, "tags": ["rust"] }), + ) + .await; + let answer = tools + .call( + MEMORY_RECALL, + json!({ "question": "rust ownership", "limit": 3, + "instructions": "be brief" }), + ) + .await + .unwrap(); + assert!(answer["answer"].as_str().unwrap().contains("ownership")); + assert_eq!(answer["citations"][0]["kind"], json!("learning")); + assert_eq!(answer["citations"][0]["meta"]["tags"], json!(["rust"])); +} + +#[tokio::test] +async fn fetch_pages_with_a_cursor() { + let tools = tools(); + for n in 0..3 { + store( + &tools, + json!({ "document": { "text": format!("rust note {n}") } }), + ) + .await; + } + let first = tools + .call( + MEMORY_FETCH, + json!({ "query": "rust", "mode": "keyword", "limit": 2 }), + ) + .await + .unwrap(); + assert_eq!(first["hits"].as_array().unwrap().len(), 2); + let cursor = first["next_cursor"].clone(); + assert!(cursor.is_string()); + let rest = tools + .call( + MEMORY_FETCH, + json!({ "query": "rust", "mode": "keyword", "limit": 2, "cursor": cursor }), + ) + .await + .unwrap(); + assert_eq!(rest["hits"].as_array().unwrap().len(), 1); + assert!(rest.get("next_cursor").is_none()); +} + +#[tokio::test] +async fn list_narrows_by_the_model_filter() { + let tools = tools(); + store( + &tools, + json!({ "learning": { "text": "a learning" }, "tags": ["keep"] }), + ) + .await; + store(&tools, json!({ "document": { "text": "a document" } })).await; + let learnings = tools + .call(MEMORY_LIST, json!({ "filter": { "kinds": ["learning"] } })) + .await + .unwrap(); + assert_eq!(texts(&learnings["items"]), ["a learning"]); + let tagged = tools + .call( + MEMORY_LIST, + json!({ "filter": { "tags_any": ["keep"], "sources": ["agent"] } }), + ) + .await + .unwrap(); + assert_eq!(texts(&tagged["items"]), ["a learning"]); +} + +#[tokio::test] +async fn get_reads_whole_items_and_reports_missing_ids() { + let tools = tools(); + let id = store( + &tools, + json!({ "conversation": { "turns": [ + { "role": "user", "text": "hello" }, { "role": "assistant", "text": "hi" } + ] } }), + ) + .await; + let result = tools + .call(MEMORY_GET, json!({ "ids": [id, "unknown"] })) + .await + .unwrap(); + assert_eq!(texts(&result["items"]), ["user: hello\nassistant: hi"]); + assert_eq!(result["missing"], json!(["unknown"])); +} + +#[tokio::test] +async fn explore_counts_per_facet() { + let tools = tools(); + store( + &tools, + json!({ "learning": { "text": "one" }, "tags": ["a", "b"] }), + ) + .await; + store( + &tools, + json!({ "learning": { "text": "two" }, "tags": ["a"] }), + ) + .await; + let page = tools + .call(MEMORY_EXPLORE, json!({ "facet": "tag" })) + .await + .unwrap(); + assert_eq!(page["facet"], json!("tag")); + assert_eq!( + page["buckets"], + json!([{ "value": "a", "count": 2 }, { "value": "b", "count": 1 }]) + ); + assert_eq!(page["total"], json!(2)); +} + +#[tokio::test] +async fn forget_by_ids_and_by_filter() { + let tools = tools(); + let id = store(&tools, json!({ "learning": { "text": "first" } })).await; + store( + &tools, + json!({ "learning": { "text": "second" }, "tags": ["drop"] }), + ) + .await; + store(&tools, json!({ "learning": { "text": "third" } })).await; + + let by_id = tools + .call(MEMORY_FORGET, json!({ "ids": [id] })) + .await + .unwrap(); + assert_eq!(by_id, json!({ "forgotten": 1, "skipped": [] })); + let by_filter = tools + .call(MEMORY_FORGET, json!({ "filter": { "tags_any": ["drop"] } })) + .await + .unwrap(); + assert_eq!(by_filter, json!({ "forgotten": 1, "skipped": [] })); + + let left = tools.call(MEMORY_LIST, json!({})).await.unwrap(); + assert_eq!(texts(&left["items"]), ["third"]); +} + +#[tokio::test] +async fn read_only_tools_omit_and_refuse_the_writes() { + let tools = tools().read_only(); + let names: Vec<&str> = tools.specs().iter().map(|spec| spec.name).collect(); + assert_eq!( + names, + [ + MEMORY_RECALL, + MEMORY_FETCH, + MEMORY_LIST, + MEMORY_GET, + MEMORY_EXPLORE + ] + ); + for name in WRITE_TOOL_NAMES { + let error = tools.call(name, json!({ "ids": ["x"] })).await.unwrap_err(); + assert!(matches!(error, Error::Unsupported(_)), "{name}: {error:?}"); + } + assert!(tools.call(MEMORY_LIST, json!({})).await.is_ok()); +} + +#[tokio::test] +async fn unknown_tools_and_bad_arguments_are_invalid_requests() { + let tools = tools(); + assert!(matches!( + tools.call("memory_nuke", json!({})).await, + Err(Error::InvalidRequest(_)) + )); + let cases = [ + (MEMORY_RECALL, json!({ "question": "" }), "`question`"), + (MEMORY_LIST, json!({ "limit": 0 }), "`limit`"), + (MEMORY_LIST, json!({ "limit": 1000 }), "`limit`"), + (MEMORY_LIST, json!("not an object"), "json object"), + (MEMORY_GET, json!({ "ids": [] }), "`ids`"), + (MEMORY_EXPLORE, json!({ "facet": "namespace" }), "`facet`"), + (MEMORY_FETCH, json!({ "query": "x", "extra": 1 }), "`extra`"), + ]; + for (name, args, field) in cases { + match tools.call(name, args).await { + Err(Error::InvalidRequest(message)) => { + assert!( + message.starts_with(name) || name == "memory_nuke", + "{message}" + ); + assert!(message.contains(field), "{name}: {message}"); + assert_eq!(message, message.to_lowercase(), "{message}"); + } + other => panic!("{name}: expected an invalid request, got {other:?}"), + } + } +} + +/// The reference engine advertising only keyword fetch. +struct KeywordOnly { + inner: ReferenceEngine, + descriptor: EngineDescriptor, +} + +impl KeywordOnly { + fn new() -> Self { + let inner = ReferenceEngine::new(); + let descriptor = EngineDescriptor { + fetch_modes: vec![FetchMode::Keyword], + ..inner.descriptor().clone() + }; + Self { inner, descriptor } + } +} + +#[async_trait] +impl MemoryEngine for KeywordOnly { + fn descriptor(&self) -> &EngineDescriptor { + &self.descriptor + } + async fn health(&self) -> EngineHealth { + self.inner.health().await + } + async fn recall(&self, req: RecallRequest) -> Result { + self.inner.recall(req).await + } + async fn fetch(&self, req: FetchRequest) -> Result { + self.descriptor.ensure_mode(req.mode)?; + self.inner.fetch(req).await + } + async fn store(&self, item: StoreItem) -> Result { + self.inner.store(item).await + } + async fn forget(&self, target: ForgetTarget) -> Result { + self.inner.forget(target).await + } + async fn list(&self, req: ListRequest) -> Result { + self.inner.list(req).await + } +} + +#[tokio::test] +async fn the_fetch_mode_enum_matches_the_engine_descriptor() { + let reference = tools(); + let fetch = reference + .specs() + .into_iter() + .find(|spec| spec.name == MEMORY_FETCH) + .unwrap(); + let modes: Vec<&str> = FetchMode::ALL.iter().map(|mode| mode.as_str()).collect(); + assert_eq!(fetch.parameters["properties"]["mode"]["enum"], json!(modes)); + + let keyword = MemoryTools::new(Arc::new(KeywordOnly::new())); + let fetch = keyword + .specs() + .into_iter() + .find(|spec| spec.name == MEMORY_FETCH) + .unwrap(); + assert_eq!( + fetch.parameters["properties"]["mode"]["enum"], + json!(["keyword"]) + ); + + // With no mode named, the engine's only mode is used. + store( + &keyword, + json!({ "document": { "text": "keyword search works" } }), + ) + .await; + let hits = keyword + .call(MEMORY_FETCH, json!({ "query": "keyword" })) + .await + .unwrap(); + assert_eq!(texts(&hits["hits"]), ["keyword search works"]); + assert!(matches!( + keyword + .call(MEMORY_FETCH, json!({ "query": "x", "mode": "vector" })) + .await, + Err(Error::InvalidRequest(_)) + )); +} + +#[test] +fn every_tool_name_is_listed_once() { + let names: Vec<&str> = tools().specs().iter().map(|spec| spec.name).collect(); + assert_eq!(names, TOOL_NAMES); +} diff --git a/crates/tinymemory-tools/tests/tools_scoping.rs b/crates/tinymemory-tools/tests/tools_scoping.rs new file mode 100644 index 00000000..f5cfbb76 --- /dev/null +++ b/crates/tinymemory-tools/tests/tools_scoping.rs @@ -0,0 +1,269 @@ +//! The security invariants: the namespace and reach are the host's, and a +//! model cannot read, write or forget outside its scope. +//! +//! The tree: `team:acme/agent:a` (the tools' place), its sibling +//! `team:acme/agent:b`, their shared parent `team:acme`, and the root. + +use std::sync::Arc; + +use serde_json::{Value, json}; +use tinymemory_api::conformance::ReferenceEngine; +use tinymemory_api::{ + Error, LearningKind, ListRequest, MemoryEngine, MemoryMeta, MetaFilter, Namespace, Reach, + StoreItem, +}; +use tinymemory_tools::{ + MEMORY_EXPLORE, MEMORY_FETCH, MEMORY_FORGET, MEMORY_GET, MEMORY_LIST, MEMORY_RECALL, + MEMORY_STORE, MemoryTools, TOOL_NAMES, +}; + +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +fn ns(path: &str) -> Namespace { + path.parse().unwrap() +} + +struct World { + engine: Arc, + tools: MemoryTools, + sibling_id: String, + team_id: String, +} + +/// An engine holding one secret at the sibling, one note at the team node, +/// and tools placed at `team:acme/agent:a`. +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +async fn world() -> World { + let engine = Arc::new(ReferenceEngine::new()); + let put = |text: &str, at: &str| { + let meta = MemoryMeta { + namespace: ns(at), + tags: vec!["shared-tag".into()], + ..MemoryMeta::default() + }; + StoreItem::learning(text, LearningKind::Fact, 0.9, meta) + }; + let sibling = engine + .store(put("sibling secret password", "team:acme/agent:b")) + .await + .unwrap(); + let team = engine + .store(put("team note password", "team:acme")) + .await + .unwrap(); + let tools = MemoryTools::new(engine.clone()).placed_at(ns("team:acme/agent:a")); + World { + engine, + tools, + sibling_id: sibling.id.0, + team_id: team.id.0, + } +} + +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +fn texts(items: &Value) -> Vec { + items + .as_array() + .unwrap() + .iter() + .map(|item| { + item["text"] + .as_str() + .or(item["snippet"].as_str()) + .unwrap() + .to_string() + }) + .collect() +} + +#[allow( + clippy::unwrap_used, + reason = "a helper outside `#[test]` fails its test by panicking, as the tests do" +)] +async fn held(engine: &ReferenceEngine) -> Vec<(String, Namespace)> { + engine + .list(ListRequest::new(MetaFilter::default(), 100)) + .await + .unwrap() + .items + .into_iter() + .map(|hit| (hit.text, hit.meta.namespace)) + .collect() +} + +#[tokio::test] +async fn a_namespace_or_reach_in_the_arguments_is_refused_by_every_tool() { + let world = world().await; + for name in TOOL_NAMES { + for key in ["namespace", "reach"] { + let top = json!({ key: "team:acme/agent:b" }); + let nested = json!({ "filter": { key: "team:acme/agent:b" } }); + for args in [top, nested] { + match world.tools.call(name, args.clone()).await { + Err(Error::InvalidRequest(message)) => { + assert!( + message.contains("fixed by the host"), + "{name} {args}: {message}" + ); + } + other => panic!("{name} {args}: expected a refusal, got {other:?}"), + } + } + } + } + let nested_store = json!({ "learning": { "text": "x", "namespace": "root" } }); + assert!(matches!( + world.tools.call(MEMORY_STORE, nested_store).await, + Err(Error::InvalidRequest(_)) + )); +} + +#[tokio::test] +async fn store_lands_at_the_place() { + let world = world().await; + world + .tools + .call( + MEMORY_STORE, + json!({ "document": { "title": "mine", "text": "agent a doc" } }), + ) + .await + .unwrap(); + let placed: Vec = held(&world.engine) + .await + .into_iter() + .filter(|(text, _)| text.contains("agent a doc")) + .map(|(_, namespace)| namespace) + .collect(); + assert_eq!(placed, [ns("team:acme/agent:a")]); +} + +#[tokio::test] +async fn reads_never_return_the_siblings_item() { + let world = world().await; + let tools = &world.tools; + + let listed = tools + .call(MEMORY_LIST, json!({ "limit": 50 })) + .await + .unwrap(); + assert_eq!(texts(&listed["items"]), ["team note password"]); + + for mode in ["keyword", "vector", "hybrid"] { + let fetched = tools + .call( + MEMORY_FETCH, + json!({ "query": "password secret", "mode": mode }), + ) + .await + .unwrap(); + assert!( + !texts(&fetched["hits"]) + .iter() + .any(|t| t.contains("sibling")), + "{mode}: {fetched}" + ); + } + + let recalled = tools + .call( + MEMORY_RECALL, + json!({ "question": "what is the sibling secret password" }), + ) + .await + .unwrap(); + assert!( + !recalled.to_string().contains("sibling secret"), + "{recalled}" + ); + assert_eq!(texts(&recalled["citations"]), ["team note password"]); + + let got = tools + .call( + MEMORY_GET, + json!({ "ids": [world.sibling_id, world.team_id] }), + ) + .await + .unwrap(); + assert_eq!(texts(&got["items"]), ["team note password"]); + assert_eq!(got["missing"], json!([world.sibling_id])); + + let explored = tools + .call(MEMORY_EXPLORE, json!({ "facet": "tag" })) + .await + .unwrap(); + assert_eq!( + explored["buckets"], + json!([{ "value": "shared-tag", "count": 1 }]) + ); +} + +#[tokio::test] +async fn forgetting_the_siblings_id_is_skipped_and_the_item_survives() { + let world = world().await; + let result = world + .tools + .call(MEMORY_FORGET, json!({ "ids": [world.sibling_id] })) + .await + .unwrap(); + assert_eq!( + result, + json!({ "forgotten": 0, "skipped": [world.sibling_id] }) + ); + assert_eq!(held(&world.engine).await.len(), 2); +} + +#[tokio::test] +async fn forgetting_by_filter_is_confined_to_the_reach() { + let world = world().await; + let tools = MemoryTools::new(world.engine.clone()) + .placed_at(ns("team:acme/agent:a")) + .reach(Reach::exact(ns("team:acme/agent:a"))); + tools + .call( + MEMORY_STORE, + json!({ "learning": { "text": "mine to drop" }, "tags": ["shared-tag"] }), + ) + .await + .unwrap(); + let result = tools + .call( + MEMORY_FORGET, + json!({ "filter": { "tags_any": ["shared-tag"] } }), + ) + .await + .unwrap(); + assert_eq!(result["forgotten"], json!(1)); + let mut left: Vec = held(&world.engine) + .await + .into_iter() + .map(|(t, _)| t) + .collect(); + left.sort(); + assert_eq!(left, ["sibling secret password", "team note password"]); +} + +#[tokio::test] +async fn an_explicit_reach_replaces_the_placed_one() { + let world = world().await; + let team_wide = MemoryTools::new(world.engine.clone()) + .placed_at(ns("team:acme/agent:a")) + .reach(Reach::subtree(ns("team:acme"))); + let listed = team_wide.call(MEMORY_LIST, json!({})).await.unwrap(); + assert_eq!(texts(&listed["items"]).len(), 2); +} + +#[tokio::test] +async fn results_never_render_the_namespace() { + let world = world().await; + let listed = world.tools.call(MEMORY_LIST, json!({})).await.unwrap(); + assert!(!listed.to_string().contains("team:acme"), "{listed}"); +} diff --git a/crates/tinymemory/Cargo.toml b/crates/tinymemory/Cargo.toml deleted file mode 100644 index e4cafce8..00000000 --- a/crates/tinymemory/Cargo.toml +++ /dev/null @@ -1,91 +0,0 @@ -[package] -name = "tinymemory" -# Not published: hosts take this repository by git or path. -publish = false -version = "1.22.4" -edition = "2024" -rust-version = "1.96" -license = "GPL-3.0-only" -description = "TinyMemory: recall, fetch and store over pluggable memory engines" -repository = "https://github.com/tinyhumansai/tinymemory" -readme = "../../README.md" -keywords = ["memory", "agent", "llm", "retrieval"] -categories = ["database"] - -[dependencies] -# The contract, re-exported wholesale so a host takes one dependency and -# `tinymemory::MemoryEngine` is `tinymemory_api::MemoryEngine`. -tinymemory-api = { path = "../tinymemory-api" } -# The engines the registry builds (`cortexdb`, `tinyhumans`), and the -# `BearerSource` seam `EngineCredential::Dynamic` carries. Not optional: a -# registry with no engine has nothing to build. A host that wants only the -# contract types depends on `tinymemory-api` alone. -tinymemory-cortex = { path = "../tinymemory-cortex" } -# `MemoryConfig` is read out of a host's config file. -serde = { version = "1", features = ["derive"] } -# The optional crates, each behind the feature named after it, re-exported as a -# module of the same name. -tinymemory-documents = { path = "../tinymemory-documents", optional = true } -tinymemory-sources = { path = "../tinymemory-sources", optional = true } -tinymemory-safety = { path = "../tinymemory-safety", optional = true } -tinymemory-context = { path = "../tinymemory-context", optional = true } -tinymemory-import = { path = "../tinymemory-import", optional = true } -tinymemory-conformance = { path = "../tinymemory-conformance", optional = true } - -[dev-dependencies] -# `MemoryConfig` round-trips through the TOML and JSON a host stores it in. -toml = "1" -serde_json = "1" -# The registry tests implement `BearerSource`. -async-trait = "0.1" -tokio = { version = "1", features = ["macros", "rt", "time"] } -# The live Office pipeline test converts a real OOXML archive and stores the -# resulting document through the CortexDB engine. -zip = { version = "8", default-features = false, features = ["deflate"] } - -# One feature per optional crate, all additive and none on by default: naming -# no feature gets the contract, the registry and the CortexDB engines. -[features] -default = [] -# Format sniffing and conversion to markdown (`tinymemory::documents`). -documents = ["dep:tinymemory-documents"] -# `OfficeConverter`: PDF, DOCX, PPTX and XLSX to markdown in-process, for a -# host to prepend to its `ConverterChain`. Implies `documents`. -documents-office = ["documents", "tinymemory-documents/office"] -# Source readers that emit `StoreItem`s (`tinymemory::sources`). -sources = ["dep:tinymemory-sources"] -# The network readers in `sources` (GitHub, RSS, web pages, URL fetch). -sources-network = ["sources", "tinymemory-sources/network"] -# Secret and PII scrubbing applied before `store` (`tinymemory::safety`). -safety = ["dep:tinymemory-safety"] -# The `context.md` compiler (`tinymemory::context`). -context = ["dep:tinymemory-context"] -# The legacy v1 workspace reader (`tinymemory::import`). -import = ["dep:tinymemory-import"] -# The spec's name for `import`. -legacy-import = ["import"] -# The behavioural suite and reference engine (`tinymemory::conformance`). -conformance = ["dep:tinymemory-conformance"] -# Everything. -full = ["documents", "documents-office", "sources-network", "safety", "context", "legacy-import", "conformance"] - -[lints.rust] -unsafe_code = "forbid" -missing_docs = "warn" -missing_debug_implementations = "warn" -unreachable_pub = "warn" -rust_2018_idioms = { level = "warn", priority = -1 } - -[lints.clippy] -all = { level = "warn", priority = -1 } -unwrap_used = "warn" -expect_used = "warn" -panic = "warn" -todo = "warn" -unimplemented = "warn" -missing_errors_doc = "warn" -missing_panics_doc = "warn" - -[lints.rustdoc] -broken_intra_doc_links = "warn" -private_intra_doc_links = "warn" diff --git a/crates/tinymemory/src/lib.rs b/crates/tinymemory/src/lib.rs deleted file mode 100644 index 43cf6850..00000000 --- a/crates/tinymemory/src/lib.rs +++ /dev/null @@ -1,73 +0,0 @@ -//! TinyMemory: recall, fetch and store over pluggable memory engines. -//! -//! The facade a host depends on. It re-exports the contract -//! ([`MemoryEngine`], [`StoreItem`], [`MetaFilter`], ...), registers the -//! engines this build can construct ([`list_engines`]), and builds one from -//! configuration ([`MemoryConfig`], [`build_engine`]). Every other crate of -//! the workspace is reachable through a feature named after it: -//! -//! | Feature | Module | What it adds | -//! | --- | --- | --- | -//! | `documents` | `documents` | format sniffing and conversion to markdown | -//! | `documents-office` | `documents` | `OfficeConverter`: PDF, DOCX, PPTX and XLSX to markdown | -//! | `sources` / `sources-network` | `sources` | source readers emitting `StoreItem`s | -//! | `safety` | `safety` | secret and PII scrubbing before `store` | -//! | `context` | `context` | the `context.md` compiler | -//! | `import` / `legacy-import` | `import` | the legacy v1 workspace reader | -//! | `conformance` | `conformance` | the behavioural suite and reference engine | -//! | `full` | | all of the above | -//! -//! With no feature the facade is the contract, the registry and the CortexDB -//! engines. -//! -//! # Example -//! -//! ``` -//! use tinymemory::{EngineCredential, MemoryConfig, list_engines}; -//! -//! let ids: Vec<&str> = list_engines().iter().map(|d| d.id).collect(); -//! assert_eq!(ids, ["cortexdb", "tinyhumans"]); -//! -//! let config: MemoryConfig = serde_json::from_str(r#"{ "engine": "cortexdb" }"#)?; -//! let engine = config.build(EngineCredential::Static("cortex-api-key".into()))?; -//! assert_eq!(engine.descriptor().id, "cortexdb"); -//! # Ok::<(), Box>(()) -//! ``` - -pub mod config; -pub mod registry; - -pub use config::{DEFAULT_ENGINE, EngineSettings, MemoryConfig}; -pub use registry::{EngineCredential, build_engine, list_engines}; -pub use tinymemory_api::*; -pub use tinymemory_cortex::{BearerSource, StaticBearer}; - -/// The contract crate, by name. -pub use tinymemory_api as api; -/// The CortexDB engines (`cortexdb`, `tinyhumans`). -pub use tinymemory_cortex as cortex; - -/// Format sniffing and conversion to markdown; `documents::OfficeConverter` -/// (PDF, DOCX, PPTX, XLSX) needs `documents-office` as well. -#[cfg(feature = "documents")] -pub use tinymemory_documents as documents; - -/// Source readers that emit `StoreItem`s. -#[cfg(feature = "sources")] -pub use tinymemory_sources as sources; - -/// Secret and PII scrubbing applied before `store`. -#[cfg(feature = "safety")] -pub use tinymemory_safety as safety; - -/// The `context.md` compiler. -#[cfg(feature = "context")] -pub use tinymemory_context as context; - -/// The legacy v1 workspace reader. -#[cfg(feature = "import")] -pub use tinymemory_import as import; - -/// The behavioural suite and reference engine. -#[cfg(feature = "conformance")] -pub use tinymemory_conformance as conformance; diff --git a/crates/tinymemory/src/registry/mod.rs b/crates/tinymemory/src/registry/mod.rs deleted file mode 100644 index e09690d1..00000000 --- a/crates/tinymemory/src/registry/mod.rs +++ /dev/null @@ -1,166 +0,0 @@ -//! The engine registry: [`list_engines`] and [`build_engine`]. -//! -//! Two engines are registered, both served by `tinymemory-cortex`: -//! -//! | Id | Engine | Endpoint | Credential | -//! | --- | --- | --- | --- | -//! | `cortexdb` | CortexDB's own `/v1/*` API | defaults to the managed API | API key | -//! | `tinyhumans` | CortexDB behind the TinyHumans backend `/memory/*` | defaults to `api.tinyhumans.ai` | session JWT or `tiny_live_` key, usually dynamic | -//! -//! [`build_engine`] refuses an unknown id, a missing required endpoint or -//! credential, and a credentialed cleartext endpoint that is not loopback, -//! all as [`Error::Config`]. Messages never carry the credential. - -use std::net::IpAddr; -use std::sync::Arc; - -use tinymemory_api::{EngineDescriptor, Error, MemoryEngine, Result}; -use tinymemory_cortex::{ - BearerSource, CORTEXDB_ENGINE_ID, CortexCredential, CortexEngine, StaticBearer, - TINYHUMANS_ENGINE_ID, -}; - -use crate::config::EngineSettings; - -/// How an engine authenticates. -#[derive(Clone, Default)] -pub enum EngineCredential { - /// No credential, for an engine that needs none. - #[default] - None, - /// One fixed token, for example an API key. - Static(String), - /// A token resolved before every request, so a refreshed session is used - /// at once. - Dynamic(Arc), -} - -impl std::fmt::Debug for EngineCredential { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(match self { - Self::None => "EngineCredential::None", - Self::Static(_) => "EngineCredential::Static()", - Self::Dynamic(_) => "EngineCredential::Dynamic()", - }) - } -} - -impl EngineCredential { - fn is_present(&self) -> bool { - match self { - Self::None => false, - Self::Static(token) => !token.trim().is_empty(), - Self::Dynamic(_) => true, - } - } -} - -/// Every engine this build can construct. -#[must_use] -pub fn list_engines() -> Vec { - vec![ - tinymemory_cortex::cortexdb_descriptor(), - tinymemory_cortex::tinyhumans_descriptor(), - ] -} - -/// Builds the engine `id` from `settings` and `credential`. -/// -/// # Errors -/// -/// [`Error::Config`] for an unknown id, a missing required endpoint or -/// credential, an endpoint that is not an HTTP(S) URL, or a credentialed -/// cleartext (`http://`) endpoint that is not loopback. -pub fn build_engine( - id: &str, - settings: &EngineSettings, - credential: EngineCredential, -) -> Result> { - let descriptor = list_engines() - .into_iter() - .find(|descriptor| descriptor.id == id) - .ok_or_else(|| Error::Config(format!("unknown memory engine `{id}`")))?; - let endpoint = settings - .endpoint - .as_deref() - .map(str::trim) - .filter(|endpoint| !endpoint.is_empty()) - .or(descriptor.default_endpoint) - .ok_or_else(|| Error::Config(format!("memory engine `{id}` needs an endpoint")))?; - if descriptor.needs_endpoint - && settings - .endpoint - .as_deref() - .is_none_or(|e| e.trim().is_empty()) - { - return Err(Error::Config(format!( - "memory engine `{id}` needs an endpoint" - ))); - } - if descriptor.needs_key && !credential.is_present() { - return Err(Error::Config(format!( - "memory engine `{id}` needs a credential" - ))); - } - if credential.is_present() { - ensure_secure_endpoint(endpoint)?; - } - let engine = match (id, credential) { - (CORTEXDB_ENGINE_ID, EngineCredential::Static(key)) => { - CortexEngine::direct(endpoint, CortexCredential::Static(key))? - } - (CORTEXDB_ENGINE_ID, EngineCredential::Dynamic(source)) => { - CortexEngine::direct(endpoint, CortexCredential::Dynamic(source))? - } - (TINYHUMANS_ENGINE_ID, EngineCredential::Static(token)) => { - CortexEngine::tinyhumans(endpoint, Arc::new(StaticBearer::new(token)))? - } - (TINYHUMANS_ENGINE_ID, EngineCredential::Dynamic(source)) => { - CortexEngine::tinyhumans(endpoint, source)? - } - (_, EngineCredential::None) => { - return Err(Error::Config(format!( - "memory engine `{id}` needs a credential" - ))); - } - _ => return Err(Error::Config(format!("unknown memory engine `{id}`"))), - }; - Ok(Arc::new(engine)) -} - -/// Refuses a cleartext endpoint off loopback, and anything that is not an -/// HTTP(S) URL with a host. -fn ensure_secure_endpoint(endpoint: &str) -> Result<()> { - let invalid = || Error::Config("memory endpoint is not an http(s) url".to_string()); - let (scheme, rest) = endpoint.split_once("://").ok_or_else(invalid)?; - let authority = rest.split(['/', '?', '#']).next().unwrap_or_default(); - let host_port = authority.rsplit('@').next().unwrap_or_default(); - let host = if let Some(bracketed) = host_port.strip_prefix('[') { - bracketed.split(']').next().unwrap_or_default() - } else { - host_port.split(':').next().unwrap_or_default() - }; - if host.is_empty() { - return Err(invalid()); - } - match scheme.to_ascii_lowercase().as_str() { - "https" => Ok(()), - "http" => { - let loopback = host.eq_ignore_ascii_case("localhost") - || host.parse::().is_ok_and(|ip| ip.is_loopback()); - if loopback { - Ok(()) - } else { - Err(Error::Config( - "credentialed memory endpoints must use https unless they are loopback" - .to_string(), - )) - } - } - _ => Err(invalid()), - } -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/docs/README.md b/docs/README.md index a6ebe425..4508902a 100644 --- a/docs/README.md +++ b/docs/README.md @@ -10,11 +10,15 @@ where it cannot drift. ```text docs/ ├── README.md # this index +├── architecture/ # how the code is built, one document per concern ├── specs/ # behavior and architecture specifications ├── plans/ # implementation plans derived from approved specs └── adr/ # architecture decision records, numbered and immutable ``` +- **[`architecture/`](architecture/README.md)** — the shape of the three crates: + the core contract, operation semantics, namespaces, the CortexDB engine, the + tools, the integrations and the test strategy. - **[`specs/`](specs/README.md)** — one file per feature, module, or subsystem, describing its behavior, public surface, invariants, and acceptance criteria. - **[`plans/`](plans/README.md)** — implementation-ordered, test-first steps for diff --git a/docs/architecture/README.md b/docs/architecture/README.md new file mode 100644 index 00000000..109925b3 --- /dev/null +++ b/docs/architecture/README.md @@ -0,0 +1,22 @@ +# Architecture + +How TinyMemory is built, one document per concern. The accepted behaviour (the +"what and why") is [`specs/memory-v2.md`](../specs/memory-v2.md); these pages +describe the shape of the code that delivers it. Item-level reference lives in +rustdoc next to the code. + +| Document | Read it for | +| --- | --- | +| [overview.md](overview.md) | The three crates, their dependency graph, the feature map, and the end-to-end write and read paths | +| [api.md](api.md) | The core contract: the `MemoryEngine` trait, descriptor, health, errors, limits and the wire format | +| [api-items.md](api-items.md) | Items, metadata and filters in detail: every field, validation rule and JSON shape | +| [operations.md](operations.md) | Step-by-step semantics of store, store_many, fetch, recall, list, forget, explore and get | +| [namespaces.md](namespaces.md) | The memory tree: `Namespace`, `Segment`, `Reach`, and what each operation does with them | +| [cortex.md](cortex.md) | The CortexDB engine: wires, scopes, envelopes, recall | +| [cortex-wire.md](cortex-wire.md), [cortex-flows.md](cortex-flows.md) | The CortexDB wire formats and the step-by-step request flows | +| [tools.md](tools.md) | `tinymemory-tools`: the seven agent tools, host-fixed scoping, `context.md` | +| [integrations.md](integrations.md) | `tinymemory-integrations`: registry and config, documents, sources, safety, legacy import | +| [testing.md](testing.md) | The conformance suite, the reference engine and the test layout | + +Reading order for a newcomer: overview, then api, then operations. Read +namespaces before writing anything that serves more than one agent. diff --git a/docs/architecture/api-items.md b/docs/architecture/api-items.md new file mode 100644 index 00000000..0c1eb095 --- /dev/null +++ b/docs/architecture/api-items.md @@ -0,0 +1,219 @@ +# Items, metadata and filters + +The data model of [`tinymemory-api`](api.md): what is stored (`StoreItem`), +what describes it (`MemoryMeta`) and how reads select it (`MetaFilter`). + +## `StoreItem` + +The unit of `MemoryEngine::store`. Three variants, internally tagged by +`"type"` (`document`, `conversation`, `learning`). Each carries a `meta` +(`#[serde(default)]`, so it may be omitted on the wire). + +### Document + +| Field | Type | Notes | +| --- | --- | --- | +| `title` | `Option` | Omitted when `None`. | +| `body` | `DocumentBody` | `Text(String)` or `Uri(String)`; must be `Text` when it reaches an engine. | +| `mime` | `Option` | The body's MIME type, when known. | +| `meta` | `MemoryMeta` | | + +```json +{ + "type": "document", + "title": "Ownership", + "body": { "text": "Ownership moves values." }, + "mime": "text/markdown", + "meta": { + "source": { "kind": "folder", "id": "notes" }, + "file_path": "/notes/rust/ownership.md", + "language": "en" + } +} +``` + +`DocumentBody::Uri` serialises as `{ "uri": "https://..." }`. It exists so a +source can describe a document before reading it; `validate` refuses it. + +### Conversation + +`turns: Vec`, in order. A `Turn` is `{ role, text, at?, tool_calls }`: +`role` is `user | assistant | system | tool`; `at` is an RFC 3339 instant; +`tool_calls` (`Vec`, omitted when empty) lists the calls the turn +made. + +```json +{ + "type": "conversation", + "turns": [ + { "role": "user", "text": "Which port does the API use?", "at": "2026-10-04T09:30:00Z" }, + { + "role": "assistant", + "text": "8080.", + "tool_calls": [{ "name": "read_config", "id": "call_1" }] + } + ], + "meta": { "source": { "kind": "conversation", "id": "thread-42" }, "thread_id": "thread-42" } +} +``` + +### Learning + +| Field | Type | Notes | +| --- | --- | --- | +| `text` | `String` | The statement. | +| `kind` | `LearningKind` | `preference | fact | procedure | correction | other`. | +| `confidence` | `f32` | `0.0..=1.0`. | +| `evidence` | `Option` | What supports it; omitted when `None`. | +| `meta` | `MemoryMeta` | | + +```json +{ + "type": "learning", + "text": "prefers tabs over spaces", + "kind": "preference", + "confidence": 0.8, + "evidence": "said so in review 12", + "meta": { "source": { "kind": "agent" }, "tags": ["style"] } +} +``` + +### Methods + +| Method | Behaviour | +| --- | --- | +| `StoreItem::document(text, meta)` | A text document with no title or MIME. | +| `StoreItem::learning(text, kind, confidence, meta)` | A learning with no evidence. | +| `kind() -> ItemKind` | `Document`, `Conversation` or `Learning` (`ItemKind::ALL`, `as_str`). | +| `confidence() -> Option` | A learning's confidence; `None` otherwise. | +| `meta()` / `meta_mut()` | The metadata. | +| `render_text() -> String` | A document: `# {title}\n\n{body}` when the title is non-blank, else the body. A conversation: one `Turn::render()` line per turn, joined by `\n`. A learning: its text. This is what `Hit::text` carries. | +| `validate() -> Result<()>` | See below. | +| `fingerprint() -> String` | See below. | + +`Turn::render()` is `role: text`, followed by ` [tools: name (id), name]` when +the turn made tool calls (the id only when one was assigned), so tool calls +stay searchable in fetch and list results. + +### Validation + +`StoreItem::validate` fails with `Error::InvalidRequest` for: + +- a document whose body is blank text (`document body must not be empty`) or + an unresolved `Uri`; +- a conversation with no turns, or any turn whose text is blank; +- a learning whose text is blank, or whose confidence is outside `0.0..=1.0` + (NaN included, as it is not in the range). + +Nothing else is checked: metadata is not validated beyond what its types +enforce (`Namespace` is checked when parsed). + +### Fingerprint + +`fingerprint()` is a stable 40-character lowercase hex string: the first 20 +bytes of the SHA-256 of the item's JSON serialisation, with +`meta.observed_at` cleared first. It covers **everything else**: kind, text, +title, turns, learning kind, confidence, evidence, and every metadata field +including `namespace` (a root namespace is not serialised, so root items hash +as they did before namespaces existed). + +`observed_at` is excluded because it records *when* the item was seen, not +*what* it is. A host stamps it on every store; hashing it would turn a retried +learning, or an unchanged file re-synced, into a new item each time. + +Two items with the same fingerprint are the same item. Engines derive +idempotency from it: the reference engine uses the fingerprint as the item id. +See [operations.md](operations.md#idempotency-and-fingerprints). + +## `ItemId` + +A transparent newtype over `String` (a bare JSON string), assigned by the +engine and opaque to the host. `ItemId::new`, `as_str`, `Display`, and `From` +for `&str` and `String`. + +## `StoreReceipt` + +`{ id: ItemId, replayed: bool }`. `replayed` is `true` when the engine already +held this exact item and wrote nothing. + +## `MemoryMeta` + +Where an item came from and what it is about. `#[serde(default)]`, so every +field may be omitted on the wire; unset options, an empty `tags` and a root +namespace are omitted on serialisation. + +| Field | Type | Meaning | +| --- | --- | --- | +| `namespace` | `Namespace` | The memory node the item lives at; the root by default. See [namespaces.md](namespaces.md). | +| `workspace` | `Option` | Absolute path or logical workspace id. | +| `folder` | `Option` | Containing folder, absolute or workspace-relative. | +| `file_path` | `Option` | The file the item was read from. | +| `language` | `Option` | Code language (`rust`) or natural-language tag (`en`). | +| `repo` | `Option` | `owner/name` or a remote URL. | +| `commit` | `Option` | Commit the item was read at. | +| `url` | `Option` | URL the item was read from. | +| `thread_id` | `Option` | Conversation thread. | +| `turns` | `Option` | `{ first, last }`, zero-based and inclusive. | +| `agent_id` | `Option` | Agent that produced the item. | +| `tool_call` | `Option` | `{ name, id? }` of the producing tool call. | +| `source` | `SourceRef` | `{ kind: SourceKind, id? }`; always present, defaults to kind `agent`. | +| `tags` | `Vec` | Free-form tags. | +| `observed_at` | `Option>` | When the underlying fact was observed, as opposed to stored. Excluded from the fingerprint. | + +`MemoryMeta::from_source(kind, id)` sets only the source. `SourceKind` is +`folder | file | link | github | rss | composio | conversation | agent | +import` (`SourceKind::ALL`, `as_str`); `agent` is the default. + +## `MetaFilter` + +Selects items for recall, fetch, list, explore and forget-by-filter. It is +`#[serde(default)]` and omits unset fields, so `{}` is the empty filter. + +Matching rules (`MetaFilter::matches(kind, &meta)`): **every set field must +match**; an empty filter matches everything. + +| Field | Type | Rule | +| --- | --- | --- | +| `reach` | `Option` | Item's namespace must be in reach; `None` admits every namespace. | +| `workspace`, `language`, `repo`, `commit`, `url`, `thread_id`, `agent_id` | `Option` | Exact match. An item with no value never matches a set field. | +| `folder`, `file_path` | `Option` | Exact, or a path prefix on a `/` boundary: `/a/b` matches `/a/b` and `/a/b/c.rs`, not `/a/bc`. A trailing `/` on the filter value is ignored. | +| `turns` | `Option` | Exact range. | +| `tool_call` | `Option` | Exact tool name of the item's `tool_call`. | +| `source_id` | `Option` | Exact `meta.source.id`. | +| `kinds` | `Vec` | Item kind is in the list; empty means all. | +| `sources` | `Vec` | `meta.source.kind` is in the list; empty means all. | +| `tags_any` | `Vec` | Item has at least one of these tags; empty means no constraint. | +| `observed_after` | `Option>` | `observed_at >= observed_after` (inclusive). | +| `observed_before` | `Option>` | `observed_at < observed_before` (exclusive). | + +An item with no `observed_at` never matches when either bound is set. + +Helpers: `MetaFilter::kinds(iter)`, `is_empty()` (true exactly when the filter +equals the default, so a filter holding only a `reach` is **not** empty), +`admits_kind(kind)`. + +```json +{ + "reach": { "at": "team:acme/agent:writer", "inherit": true, "descendants": false }, + "kinds": ["document", "learning"], + "sources": ["folder", "github"], + "folder": "/notes/rust", + "tags_any": ["style", "review"], + "observed_after": "2026-01-01T00:00:00Z", + "observed_before": "2026-10-01T00:00:00Z" +} +``` + +### Matching example + +```rust +use tinymemory_api::{ItemKind, MemoryMeta, MetaFilter, SourceKind}; + +let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("notes".into())); +meta.file_path = Some("/notes/rust/ownership.md".into()); + +let under = MetaFilter { file_path: Some("/notes/rust".into()), ..MetaFilter::default() }; +let lookalike = MetaFilter { file_path: Some("/notes/ru".into()), ..MetaFilter::default() }; +assert!(under.matches(ItemKind::Document, &meta)); +assert!(!lookalike.matches(ItemKind::Document, &meta)); // not on a `/` boundary +``` diff --git a/docs/architecture/api.md b/docs/architecture/api.md new file mode 100644 index 00000000..0dd05339 --- /dev/null +++ b/docs/architecture/api.md @@ -0,0 +1,254 @@ +# The core contract: `tinymemory-api` + +`tinymemory-api` is the contract between a host, the tools and an engine. It +performs no I/O. Everything here is re-exported from the crate root +(`tinymemory_api::MemoryEngine`, ...). + +Items, metadata and filters are in [api-items.md](api-items.md); the behaviour +of each operation is in [operations.md](operations.md); namespaces are in +[namespaces.md](namespaces.md). + +## Modules + +| Module | Holds | +| --- | --- | +| `engine` | `MemoryEngine`, `EngineDescriptor`, `EngineHealth`, `MAX_STORE_MANY`, `validate_many` | +| `error` | `Error`, `Result` | +| `item` | `StoreItem`, `ItemKind`, `ItemId`, `DocumentBody`, `Turn`, `Role`, `LearningKind`, `StoreReceipt` | +| `meta` | `MemoryMeta`, `MetaFilter`, `SourceKind`, `SourceRef`, `ToolCallRef`, `TurnRange` | +| `namespace` | `Namespace`, `Segment`, `SegmentKind`, `Reach` | +| `query` | Requests and responses for recall, fetch, list and forget | +| `explore` | `Facet`, explore and get requests, the listing-based defaults, limits | +| `conformance` | (feature `conformance`) `run`, `ReferenceEngine` | + +The crate also re-exports `async_trait` and `chrono` so an engine and the +contract name the same versions. + +## `MemoryEngine` + +An object-safe `#[async_trait]` trait, `Send + Sync`; hosts hold it as +`Arc`. + +| Method | Required? | Behaviour | +| --- | --- | --- | +| `descriptor(&self) -> &EngineDescriptor` | required | What the engine is and offers. | +| `health(&self) -> EngineHealth` | required | Whether it can serve now. Infallible: trouble is reported as `Degraded` or `Down`. | +| `recall(RecallRequest) -> Result` | required | A synthesised answer with citations. | +| `fetch(FetchRequest) -> Result` | required | Ranked raw retrieval in one `FetchMode`. | +| `store(StoreItem) -> Result` | required | Store one item; an identical item is a replay. | +| `forget(ForgetTarget) -> Result` | required | Remove by ids or by a non-empty filter. | +| `list(ListRequest) -> Result` | required | Query-free paging. | +| `store_many(Vec) -> Result>` | **default** | Calls `validate_many`, then `store` one item at a time, in order, stopping at the first error. An engine overrides it to batch. | +| `explore(ExploreRequest) -> Result` | **default** | `explore_by_listing`: pages through `list`. An engine that can aggregate server-side overrides it. | +| `get(GetRequest) -> Result>` | **default** | `get_by_listing`: pages through `list` until every id is found. An engine that can look ids up directly overrides it. | + +The trait documents the rule every method follows: **validate first**, using +the `validate` method of the request type, so every engine refuses the same +malformed call with the same `Error::InvalidRequest`. A `FetchMode` the +descriptor does not list fails with `Error::Unsupported` +(`EngineDescriptor::ensure_mode`). Note `FetchRequest::validate` does not +check the mode; the engine does, by calling `ensure_mode`. + +### Constants and limits + +| Constant | Value | Where it applies | +| --- | --- | --- | +| `MAX_STORE_MANY` | 100 | Items per `store_many` call (1 to 100). | +| `MAX_GET_IDS` | 200 | Ids per `GetRequest` (1 to 200). | +| `MAX_BUCKETS` | 500 | `ExploreRequest::limit` (1 to 500). | +| `MAX_SCAN_LIMIT` | 50 000 | `ExploreRequest::scan_limit` (1 to 50 000). The default when omitted is 5 000. | + +Other limits live in the types they bound: a namespace nests at most 8 deep, +and a segment id is 1 to 128 characters ([namespaces.md](namespaces.md)). +`recall`, `fetch` and `list` take a `limit` that must be positive; the contract +sets no upper bound for them. + +### `validate_many` + +`validate_many(&[StoreItem]) -> Result<()>` checks a batch: `1..=MAX_STORE_MANY` +items, each passing `StoreItem::validate`. Engines overriding `store_many` +call it first. It returns `Error::InvalidRequest` for an empty or oversized +batch, otherwise the first invalid item's error. + +### Free helpers + +| Function | Purpose | +| --- | --- | +| `explore_by_listing(&engine, req)` | The default `explore`: scan `list`, count facet values, build the page. | +| `get_by_listing(&engine, req)` | The default `get`. | +| `in_request_order(&ids, found)` | Orders a `BTreeMap` by the requested ids, each once. | + +`explore_by_listing` and `get_by_listing` accept any `E: MemoryEngine + ?Sized`. +`in_request_order` is public in `explore` but not re-exported from the crate +root. + +## `EngineDescriptor` + +A value an engine returns from `descriptor()`; it is `Serialize` only (it holds +`&'static str` fields). + +| Field | Meaning | +| --- | --- | +| `id: &'static str` | Stable id used in configuration (`cortexdb`, `tinyhumans`, `reference`). | +| `label` | Human-readable name. | +| `description` | One sentence. | +| `hosted: bool` | A third party runs the engine. | +| `needs_endpoint: bool` | Configuration must name an endpoint. | +| `needs_key: bool` | Configuration must supply a credential. | +| `default_endpoint: Option<&'static str>` | Used when configuration names none. | +| `fetch_modes: Vec` | The modes the engine serves. | + +Methods: `supports(mode) -> bool`, and `ensure_mode(mode) -> Result<()>`, which +fails with `Error::Unsupported("engine `` does not offer fetch")`. + +## `EngineHealth` + +| Variant | Meaning | +| --- | --- | +| `Ok` | Serving. | +| `Degraded(String)` | Serving, impaired (rate limited, partially available). | +| `Down(String)` | Not serving. | + +`is_serving()` is `false` only for `Down`. Wire form is adjacently tagged: + +```json +{ "state": "ok" } +{ "state": "degraded", "reason": "rate limited" } +{ "state": "down", "reason": "connection refused" } +``` + +## `Error` + +One enum, built with `thiserror`. Variants classify a failure by what a host +can do about it. Messages are lowercase, carry no trailing punctuation, and +never carry a credential: an engine sanitises its own failure before it becomes +`Error::Engine`. `Error` is `Clone + PartialEq + Eq`. + +| Variant | Display prefix | Used when | Raised by `tinymemory-api` itself? | +| --- | --- | --- | --- | +| `Unsupported(String)` | `unsupported:` | The engine does not offer the operation or fetch mode; the host should have read the descriptor. Also what `tinymemory-tools` returns for a write tool on read-only tools. | yes (`ensure_mode`) | +| `InvalidRequest(String)` | `invalid request:` | The request is malformed: a blank query, zero limit, empty forget target, unresolved document URI, bad namespace, out-of-range confidence, unknown cursor. | yes (every `validate`) | +| `Unauthorized(String)` | `unauthorized:` | The credential was missing, expired or rejected. | no, engines | +| `NotFound(String)` | `not found:` | The addressed item or route does not exist. | no, engines | +| `Conflict(String)` | `conflict:` | The write conflicts with what the engine holds. | no, engines | +| `Unavailable(String)` | `unavailable:` | Transient (timeout, rate limit, unavailable upstream); the same call may succeed later. | no, engines | +| `Engine(String)` | `engine error:` | The engine's own failure, already sanitised. | only by the reference engine (poisoned lock) | +| `Config(String)` | `configuration error:` | The engine was configured wrongly (unknown id, missing endpoint or key, credentialed cleartext endpoint). | no, the registry in `tinymemory-integrations` | + +`Error::is_transient()` is `true` only for `Unavailable`; hosts retry on it. +`get` of an unknown id is **not** an error: the id is left out of the result. + +`Result` is `std::result::Result`. The conformance feature has +its own `conformance::Error` (`Check` and `Engine` variants) naming the check +that failed. + +## Requests and responses + +All derive `Debug, Clone, PartialEq, Serialize, Deserialize`. `filter` fields +default to the empty filter when absent on the wire. + +| Request | Response | Validation (`Error::InvalidRequest`) | +| --- | --- | --- | +| `RecallRequest { question, filter, limit, instructions? }` | `RecallAnswer { answer, citations, model? }` | blank question; `limit == 0` | +| `FetchRequest { query, mode, filter, limit, cursor? }` | `FetchPage { hits, next_cursor? }` | blank query; `limit == 0` | +| `ListRequest { filter, limit, cursor? }` | `ListPage { items, next_cursor? }` | `limit == 0` | +| `ForgetTarget::Ids(Vec)` or `Filter(MetaFilter)` | `ForgetReport { forgotten }` | no ids; an empty filter | +| `ExploreRequest { facet, filter, limit, scan_limit }` | `ExplorePage { facet, buckets, total, missing, more_buckets, truncated }` | limit not in `1..=500`; scan_limit not in `1..=50 000` | +| `GetRequest { ids, reach? }` | `Vec` | no ids, more than 200, or a blank id | +| `StoreItem` | `StoreReceipt { id, replayed }` | see [api-items.md](api-items.md) | + +Constructors: `RecallRequest::new(question, limit)`, +`FetchRequest::new(query, mode, limit)`, `ListRequest::new(filter, limit)`, +`ExploreRequest::new(facet, limit)` (scan limit 5 000); each starts with an +empty filter and no cursor. `GetRequest` has no constructor. + +`Citation { id, kind, snippet, meta, score? }` is one item an answer drew on; +its id resolves through `list`. `Hit { id, kind, text, meta, score, +confidence? }` is one stored item as a read returns it: `text` is +`StoreItem::render_text()`, `score` is `0.0` in a listing, `confidence` is a +learning's confidence and absent for other kinds. + +`FacetBucket { value, count }`; `FetchMode` is `Keyword | Vector | Hybrid` +(`FetchMode::ALL`, `as_str`). + +### Wire shapes + +Serde names are `snake_case`. Optional fields are omitted when `None`, and +`meta` omits unset fields, an empty `tags` list and a root namespace. + +A `Hit`: + +```json +{ + "id": "9f2c1c6e0a8b4d3e7f5a1b2c3d4e5f6a7b8c9d0e", + "kind": "learning", + "text": "prefers tabs", + "meta": { + "namespace": "team:acme/agent:writer", + "source": { "kind": "agent" }, + "tags": ["style"], + "observed_at": "2026-10-04T09:30:00Z" + }, + "score": 0.0, + "confidence": 0.8 +} +``` + +A `FetchRequest` page two: + +```json +{ + "query": "ownership", + "mode": "hybrid", + "filter": { "kinds": ["document"], "folder": "/notes/rust" }, + "limit": 10, + "cursor": "10" +} +``` + +A `RecallAnswer`: + +```json +{ + "answer": "The user prefers tabs.", + "citations": [ + { + "id": "9f2c1c6e0a8b4d3e7f5a1b2c3d4e5f6a7b8c9d0e", + "kind": "learning", + "snippet": "prefers tabs", + "meta": { "source": { "kind": "agent" } }, + "score": 0.91 + } + ], + "model": "reference" +} +``` + +`ForgetTarget` is externally tagged; `ForgetReport` counts only items actually +removed: + +```json +{ "ids": ["9f2c1c6e0a8b4d3e7f5a1b2c3d4e5f6a7b8c9d0e"] } +{ "filter": { "workspace": "scratch" } } +{ "forgotten": 1 } +``` + +`ExploreRequest` and `ExplorePage` ([operations.md](operations.md#explore)): + +```json +{ "facet": "folder", "filter": { "kinds": ["document"] }, "limit": 20 } +{ + "facet": "folder", + "buckets": [{ "value": "/notes/rust", "count": 12 }], + "total": 14, "missing": 2, "more_buckets": 0, "truncated": false +} +``` + +`StoreReceipt`: `{ "id": "...", "replayed": false }`. + +## Conformance feature + +With `features = ["conformance"]`, `tinymemory_api::conformance` provides +`run(&dyn MemoryEngine) -> conformance::Result<()>` and `ReferenceEngine` +(id `reference`, an in-memory engine serving every fetch mode). See +[testing.md](testing.md). diff --git a/docs/architecture/cortex-flows.md b/docs/architecture/cortex-flows.md new file mode 100644 index 00000000..73dcdc24 --- /dev/null +++ b/docs/architecture/cortex-flows.md @@ -0,0 +1,232 @@ +# CortexDB engine: operation flows + +Step by step, what each `MemoryEngine` method does on the CortexDB engine. +Part of the CortexDB engine docs: [overview and transport](cortex.md) · +[the wire](cortex-wire.md) · this page. Sources are under +`crates/tinymemory-integrations/src/cortex/engine/` and `.../log/`. + +Every method first validates its request with the contract's `validate` +methods, so a malformed call fails as `Error::InvalidRequest` before any +request is sent. Read [the wire](cortex-wire.md) for what "scope", "label" and +"envelope" mean here. + +## Store and store_many + +`store(item)` is `store_items(vec![item])` and returns the single receipt. +There is **one** path, so a single store gets exactly the batch's guarantees: +listed on return, and ranked recall awaited for its final event. + +`store_many` first runs `validate_many` (a batch of 1 to `MAX_STORE_MANY` (100) +valid items; an empty or oversized batch is `Error::InvalidRequest`). Then: + +1. **Fingerprint.** Each item's id is `StoreItem::fingerprint()`, a digest of the + whole item, metadata and namespace included but `meta.observed_at` not, so + the same text at two nodes is two items and a re-sync that only restamps + `observed_at` is a replay. +2. **Group** the items by scope (kind at namespace node). +3. **Replay detection.** One id lookup per scope: the listing narrowed by the + items' `tm:i:` labels (batches of up to 50 labels), each hit re-checked + against the envelope's real id. The result is, per id, which turn indexes + are already held (`None` for a document or learning). +4. **Write, in item order, without waiting.** For each item, build its + envelopes (one per event) and skip every event already held: + - every event present: a **replay**; nothing is written and the receipt has + `replayed: true`; + - some turns of a conversation present: a previous store failed part way; + only the missing turns are written, in order (and the receipt is not a + replay); + - nothing present: every event is written. + + An item repeated inside the batch is a replay of its first copy. Each + write uses a fresh idempotency key (see below). The wire call is made per + item: Direct sends one experience, or one ordered bulk when two or more + events are due; TinyHumans sends the events one at a time. +5. **Wait, once per scope.** For the **last event written** in each scope the + engine waits until it is listed (the log is ordered, so its being listed + implies the earlier ones are). Ranked recall is awaited for one event + only: the last event written to the most recently written scope. See + [waiting](#waiting-for-a-write-to-be-readable). + +Receipts come back in item order, each `{id, replayed}`. On an error the items +before the failing one are stored, and storing them again is a replay. + +**Fresh idempotency keys.** Writes use a fresh `tm---` key, +never one derived from content. CortexDB never releases a key on forget, so a +content key would make re-storing a forgotten item a silent no-op; and the +hosted memory API answers every replay of a claim with 409. Replay detection is +done by the engine, by looking the item up, before it writes. + +### Waiting for a write to be readable + +The contract requires read-after-write; CortexDB indexes after it accepts. A +write waits in up to two stages: + +1. **Listed** (fatal on timeout, 30s). Poll the scope's listing narrowed to the + item label until it carries the event's id. Failure is `Error::Unavailable` + ("did not become readable"). One page suffices since the listing is newest + first. +2. **Settled** (best-effort, 10s). Poll ranked recall (query is the first 256 + characters of the stored text) until the event is in the pack. A recall + that is down, slow, or errors ends the wait quietly: the write is durable + and listed, so it is not reported as failed. + +Polling starts at 250ms. Direct keeps that gap; TinyHumans doubles it up to a +2s ceiling (a fixed 250ms poll would spend a fifth of the backend's 300 +requests per minute on one write). On TinyHumans a 429 or 5xx while waiting +for the listing means "not yet" and the wait continues to its deadline; on +Direct such an error is returned. + +### Hosted writes and outcome-unknown recovery + +Each TinyHumans write carries a random `Idempotency-Key` claim, reused across +that write's own retries (3 attempts, 250ms then 500ms apart, on a transient +error). The memory API takes the claim before forwarding and answers any replay +of a claimed key with 409 without forwarding it. So: + +- a transient fault (429, 5xx, timeout) is retried under the same claim; a + fault raised before the memory API (its own rate limiter) leaves the claim + free, and the retry is simply forwarded; +- a **409 on a retry** means the earlier attempt reached the engine and may + have been applied: the outcome is unknown, not failed. The engine looks for + the event (same scope, same item label, exactly the same stored text) until + the 30s budget runs out. If it is found its id is the receipt. If not, the + error is `Unavailable` and says the outcome is unknown. + +A 409 on the *first* attempt is returned as `Error::Conflict`. Direct writes +are sent once: a timeout leaves the outcome unknown, and the replay detection +makes a retry by the caller safe. + +## List + +`list_page` reads a cursor over the listings of the scopes the filter reads, +in `ItemKind::ALL` order and then by namespace, each newest first. + +1. Resolve the scopes (see [scope layout](cortex-wire.md#scope-layout)); none + means an empty page. Decode the cursor, or start at the first scope. +2. If the cursor names a scope that no longer exists, resume at the next scope + in order from its first page. +3. Page the scope with `limit=200`, narrowed by one label when the filter has + a labelled field. For each raw event: skip a copy equal to the previous + event id (the engine emits each event twice; the cursor remembers the last + id so this works across page boundaries), decode it, and keep it when it is + an envelope of the scope's kind and the **full** `MetaFilter` matches. +4. **Each item once.** A document or learning is one event. A conversation is + emitted only on the page holding its **turn 0** event; its text is + assembled from all its turns by one label lookup for all the conversations + on the page. A conversation whose store failed part way still has turn 0 and + lists with the turns it holds. +5. Stop when `limit` items are collected and return a cursor, unless the end + of the last scope was reached (then there is none). + +Scores are `0`. The whole call reads at most 500 engine pages; past that it +fails with `Error::Engine` rather than answer from a truncated log. A cursor +the wrong operation produced, or any malformed one, is `Error::InvalidRequest`. + +**The cursor** is opaque: a one-letter tag (`l` list, `f` fetch) followed by +hex-encoded JSON, so a host stores it and passes it back but cannot usefully +edit it. A list cursor holds the scope's path (not a position, so a scope +created between pages cannot shift the listing), the engine's cursor for the +page being read, the offset of events already consumed on it, and the last +event id. + +## Fetch + +Only `Hybrid`. Other modes fail `Error::Unsupported` before any request. + +1. Decode the cursor (an offset into the merged ranking) and compute the + page end `offset + limit`. +2. For **each scope** the filter reads, ask recall for a pack with + `events = min((end + 1) * 3, 1000)`, narrowed by one label when possible. + (Three raw events per wanted hit, because a conversation contributes + several turns and the client-side filter drops some. 1000 events is the + deepest a fetch page can go.) +3. Decode each pack's events, apply the **full filter** client-side, keep each + item once at its best rank. +4. **Interleave** the scopes rank by rank: every scope's best, then every + scope's second, and so on. +5. Take the page `[offset, end)`. A conversation hit carries the whole + conversation, assembled from all its turns (one lookup per namespace node). +6. Score each hit `1 / (1 + rank)`, since CortexDB reports no score. +7. `next_cursor` is `offset = end` when the merged ranking held more than `end` + items, else none. The next page asks again with a larger budget. + +## Recall + +Recall builds a pack, asks the answer route **once** with `use_pack_id`, and +cites from the pack. + +1. Resolve the scopes. Then choose the packs: + - **one scope**: one pack over it; + - **no reach** (an unscoped, administrative read), or a filter that admits + no kinds: one pack over `app:tinymemory` with `view: "descend"`, which + recalls the root and every scope under it; + - **a reach over several scopes**: one pack per scope, built four at a + time, exact (never server-side traversal), so a sibling agent's scope is + never in the pack. Scopes are ordered most specific node first. +2. Each pack's events budget is `2 * limit`, with the derived layers sharing + `limit` (see [the wire](cortex-wire.md#recall-recall)). +3. Decode and filter each pack's events with the full `MetaFilter` (reach + included). +4. The answer comes from the pack holding the **most admitted events**, the + most specific node on a tie. A missing `pack_id` is `Error::Engine`. +5. Ask the answer route with that pack's scope and `use_pack_id`. A response + without `answer` text is `Error::Engine`. `model` is + `diagnostics.answer_model`. +6. **Citations** come from the packs' decoded events, one per item, the most + specific node's first, capped at `limit`, with `score: None` and the + envelope's text as the snippet. A pack with no decodable events still + returns the answer, with no citations. + +## Forget + +`forget` looks the items' events up, then removes them by `memory_ids`. The +target is validated first, so an empty id list or an **empty filter is refused +and sends nothing**. + +- **`Ids`**: for each scope the engine holds (the root's and every discovered + namespace node, all kinds), find the ids' events by their labels (re-checked + against the envelope) and remove every event of a found item. Ids are not + confined to a reach; a confined caller reads them with `get` first. +- **`Filter`** (must be non-empty): walk each scope the filter reads (its + kinds within its reach), narrowed by one label when possible, collect the + events of every item the **full** filter matches, then remove them. + +Either way the matched event ids are removed per scope with +`selector.memory_ids`, in batches of 100 (see +[the wire](cortex-wire.md#forget-forget)). A scope with nothing to remove sends +no request, so the engine can never send an empty selector, and it never +sends `confirm_all`. TinyHumans retries a transient failure up to 3 times +(removal of named events is idempotent; a 404 on a retry counts as done); +Direct sends once. `ForgetReport.forgotten` counts **items**, not events. + +## Get + +`get` is overridden to look the ids up directly rather than by scanning, the +contract default. It takes the scopes the request's `reach` reads (every +kind), and per scope looks the ids up by their `tm:i:` labels, stopping as +soon as every id is found. Each item is rebuilt from its events (a +conversation from all its turns). Hits have score `0`, come back in the order +asked, and an id that names nothing is left out. + +## Explore + +`explore` is **not** overridden: the engine uses the contract's default, +`explore_by_listing`, which pages through `list` up to the request's +`scan_limit` and reports whether it stopped early. CortexDB has no +server-side aggregation the engine relies on. + +## Scope discovery + +Needed only for a subtree reach or no reach (see +[scope layout](cortex-wire.md#scope-layout)). The engine asks the scopes +route (`v1/scopes/list` or `memory/scopes`) with `prefix` set to +`app:tinymemory`, or to the reach's own node path when it is not the root, and +`limit=1000`. Each returned path is parsed with `parse_scope`; paths that are +not TinyMemory kind scopes are skipped, and those whose namespace the reach +admits and whose kind the filter admits are added to the known nodes. A `404` +from the scopes route yields no extra scopes. Discovery runs once per call. + +## Health + +See [the wire](cortex-wire.md#health). It is one request and never carries the +backend's own error text into the reason. diff --git a/docs/architecture/cortex-wire.md b/docs/architecture/cortex-wire.md new file mode 100644 index 00000000..be3baf45 --- /dev/null +++ b/docs/architecture/cortex-wire.md @@ -0,0 +1,354 @@ +# CortexDB engine: the wire + +What `CortexEngine` sends to CortexDB and how it lays an item out as events. +Part of the CortexDB engine docs: [overview and transport](cortex.md) · +this page · [operation flows](cortex-flows.md). The source is +`crates/tinymemory-integrations/src/cortex/`. + +CortexDB is an append-only event log with ranked recall and a grounded answer +route. TinyMemory stores each item as one or more events in that log, and +reads them back through the listing and recall routes. + +## Two wires + +One engine type, `CortexEngine`, speaks two HTTP surfaces. `CortexWire` +selects the surface; `CortexWire::path` is the only place a route name lives. + +| | `Direct` (`cortexdb`) | `TinyHumans` (`tinyhumans`) | +| --- | --- | --- | +| Constructor | `CortexEngine::direct(endpoint, CortexCredential)` | `CortexEngine::tinyhumans(base_url, Arc)` | +| Default endpoint | `https://api-v1.cortexdb.ai` (`CORTEX_API_ENDPOINT`) | `https://api.tinyhumans.ai` (`TINYHUMANS_API_ENDPOINT`) | +| Route prefix | `/v1/*` | `/memory/*` | +| Success body | bare JSON | `{"success": true, "data": ...}`; `data` is unwrapped | +| Failure body | any text (an excerpt is kept) | `{"success": false, "error": "...", "errorCode": "CODE"}` | +| Credential | API key (static) or a bearer source | bearer source (session JWT or `tiny_live_` key) | +| `X-Cortex-Actor` | learned from `v1/auth/whoami` | not sent (the backend names the actor) | +| Write extras | `?wait=indexed`, a bulk route | none: one event per request, an `Idempotency-Key` claim | +| Health route | `v1/admin/health` | none: lists one scope under a prefix | + +Both descriptors declare `fetch_modes = [Hybrid]`. CortexDB's recall body +accepts only `scope`, `query`, `budgets`, `view`, `include`, `temporal` and +`filters`; nothing switches between lexical and embedding retrieval, so +declaring `Keyword` or `Vector` would promise a ranking the wire cannot ask +for. Both fail with `Error::Unsupported` before any request. + +### Routes + +| Logical route | Direct | TinyHumans | Method | +| --- | --- | --- | --- | +| Experience (append one event) | `v1/experience` | `memory/experience` | POST | +| Bulk (append an ordered batch) | `v1/experience/bulk` | `memory/experience` (never used for a batch) | POST | +| Events (list a scope) | `v1/events` | `memory/events` | GET | +| Recall (build a pack) | `v1/recall` | `memory/recall` | POST | +| Forget | `v1/forget` | `memory/forget` | POST | +| Answer | `v1/answer` | `memory/answer` | POST | +| Health | `v1/admin/health` | `memory/scopes` | GET | +| Scopes (registered scopes under a prefix) | `v1/scopes/list` | `memory/scopes` | GET | +| Whoami (Direct only) | `v1/auth/whoami` | | GET | + +The endpoint is joined with the route, so a base URL with a path prefix keeps +it (a trailing `/` is added when missing). + +## Endpoints and their shapes + +Only the fields the engine reads or writes are listed. Unlisted response +fields are ignored. + +### Append: `experience` and `bulk` + +Request body (one event). It is the same on both wires: + +```json +{ + "scope": "app:tinymemory/agent:researcher/app:documents", + "modality": "document", + "idempotency_key": "tm---", + "content": { "kind": "message", "role": "user", "text": "" }, + "context": { + "labels": ["tm:i:<16 hex>", "tm:k:<16 hex>"], + "observed_at": "2026-01-02T03:04:05+00:00" + } +} +``` + +- `modality` is `document` for a document, `observation` for a learning, and + `conversation` for a turn. `content.role` is `user` for documents and + learnings and the turn's speaker (`user`, `assistant`, `system`, `tool`) + for a conversation turn. +- `idempotency_key` is a fresh value on every write, never derived from + content (see [flows: store](cortex-flows.md#store-and-store_many)). +- `context.observed_at` is the turn's `at`, else the item's + `meta.observed_at`; it is omitted when neither is set. +- `context.labels[0]` is always the item label; the writer relies on that. + +Response: `{"event_id": "..."}` (Direct answers `202`, with `status` and +`replayed_from_idempotency` the engine does not read). A response without +`event_id` is `Error::Engine`. + +Direct appends with `?wait=indexed`. A single event goes to `v1/experience`; +**two or more** go to `v1/experience/bulk` with + +```json +{ "items": [ ...experience bodies... ], "ordering": "strict_temporal" } +``` + +and the response must carry `results` with one entry per request, the last +naming `event_id`. A missing `results`, or a count that differs from the +number sent, is `Error::Engine`. Note that a conversation of one turn, or one +with a single missing turn, goes the single-event route. + +TinyHumans always sends one event per request, in order, each under an +`Idempotency-Key` header claim (see [transport](cortex.md#idempotency-claims)). + +### List: `events` + +```text +GET {events}?scope=&limit=200[&labels=][&cursor=] +``` + +Response: + +```json +{ "items": [ { "id": "evt_1", "scope": "...", "content": { "text": "..." }, + "context": { "labels": [], "observed_at": "..." } } ], + "has_more": true, "next_cursor": "..." } +``` + +- Newest first. The engine emits **every event twice** and `limit` counts the + copies, so a page of 200 holds about 100 distinct events. Readers dedupe. +- `labels` is **one** comma-separated parameter (the hosted backend refuses a + repeated `labels=`); at most 50 labels per request. An event matches when it + carries any one of them. +- A next page exists only when `has_more` is `true` **and** `next_cursor` is + present. A `next_cursor` equal to the cursor just sent is `Error::Engine` + (a listing that does not advance). +- Unknown query parameters are ignored by the engine, so the paging parameter + is exactly `cursor`; a misspelling would serve page one for ever. + +### Recall: `recall` + +```json +{ "scope": "app:tinymemory/app:documents", "query": "...", + "budgets": { "per_layer_limits": { "events": 30 } }, + "filters": { "metadata": { "labels": ["tm:t:<16 hex>"] } }, + "view": "descend" } +``` + +`filters` is present only when the metadata filter has a labelled field; +`view` is only `"descend"`, for one case (an unscoped multi-scope recall). +Response: `{"pack_id": "...", "layers": {"events": [...]}}`. Events in a pack +render their text for a reader as `[role] {...}`; the decoder strips that +prefix. A pack's events are read from `/layers/events` and decoded exactly +like listing events. + +For `recall` (the answer path) the budget also names the derived layers: +`events` is `2 * limit`, and `facts`, `beliefs`, `episodes` and +`understanding` share `limit` between them (the remainder goes to the first +ones). + +### Answer: `answer` + +```json +{ "scope": "...", "question": "...", "use_pack_id": "pack_...", + "cite_sources": true, "include_context": true, + "answer_instructions": "..." } +``` + +Response fields read: `answer` (required, string) and +`diagnostics.answer_model` (optional, becomes `RecallAnswer.model`). + +`answer_instructions` is the request's instructions when set. When unset, +Direct sends `null` and TinyHumans **omits the key**: its answer schema is +strict (an unknown key, or a `null` instructions, is a 400). + +### Forget: `forget` + +```json +{ "scope": "...", "layers": ["events"], + "selector": { "memory_ids": ["evt_1", "evt_2"] }, + "audit_note": "tinymemory: forget" } +``` + +At most 100 ids per request. The id field is exactly `memory_ids`: an +unrecognised or empty selector means *the whole scope* to CortexDB (an empty +selector needs `confirm_all`, and `confirm_all` beside a selector is refused). +The engine never sends an empty selector and never sends `confirm_all`; a +scope with nothing to remove sends no request at all. + +### Scopes: `v1/scopes/list` and `memory/scopes` + +```text +GET {scopes}?prefix=&limit=1000 +``` + +The reader accepts either `{"items": [{"path": "..."}]}` (Direct) or +`{"scopes": ["..."]}` (hosted), and for each entry either a bare string or an +object with `path`. A `404` means "no scope listing" and is treated as no +scopes. + +### Health + +Direct: `GET v1/admin/health`. TinyHumans has no health route, so it lists one +scope under a prefix the engine never writes: +`GET memory/scopes?prefix=tmh%3Aprobe&limit=1`. The memory API refuses a +prefix that is not `type:id` segments (a bare word is a 400, which would +report a healthy service as broken). This proves reachability and the +credential in one round trip. `Error::Unavailable` is `Degraded`, any other +failure is `Down`; the reason keeps the message head and withholds the +backend's own text (everything after a spaced em-dash). + +### Whoami (Direct only) + +`GET v1/auth/whoami` returns `{"caller": "user:local"}`. See +[the actor header](cortex.md#the-actor-header). + +## Scope layout + +Every item lives in the scope of its **kind** at its **namespace node**, +under the TinyMemory root `app:tinymemory` (`envelope::scope_path`): + +```text +app:tinymemory/app:{documents,conversations,learnings} the root node +app:tinymemory/agent:researcher/app:{documents,conversations,learnings} an agent +app:tinymemory/team:acme/agent:writer/app:learnings a team member +``` + +So within every node, documents, conversations and learnings are separate +scopes and CortexDB can recall, retain and erase each on its own. The +hosted backend re-roots every scope under the caller's tenant, which is +invisible to the engine except that scope paths it reads back may carry a +prefix: `parse_scope` finds `app:tinymemory` wherever it sits. + +**Scope-type mapping.** A namespace segment `kind:id` becomes the CortexDB +scope segment of the same text, using the contract's prefixes: + +| Namespace segment | Scope segment type | +| --- | --- | +| Agent | `agent` | +| Team | `team` | +| User | `user` | +| Workspace | `ws` | +| Project | `project` | +| TinyMemory root and each kind leaf | `app` | + +These are CortexDB's built-in types, chosen on purpose. From CortexDB v0.10 a +deployment admits only the types in its policy's `allowed_scope_types` +(`org, dept, team, app, user, agent, service, ws, project, global, system, +source` in every shipped preset) and refuses any other with +`422 UNREGISTERED_SCOPE_TYPE`. A private type such as `tm:` would need every +operator to register it first, so the engine uses only types that every +preset allows. A namespace nests at most 8 deep, which keeps the path far +inside the hosted grammar (at most 31 `type:id` segments, as the hosted +double enforces). + +**Which scopes a read touches.** `MetaFilter.kinds` picks the kinds and +`MetaFilter.reach` the nodes (`engine/scopes.rs`). Ordering is by kind +(`ItemKind::ALL`) and then namespace, so a cursor can resume by position. + +- A reach **without descendants** reads `at` and, when it inherits, each + ancestor. The nodes are known, so no request is made; a node nothing was + written to simply lists empty. +- A **subtree reach, or no reach**, needs the nodes below. They are + discovered once per call from the scopes registered under the TinyMemory + root (or under the reach's own node), and the root's kind scopes are always + read. +- Reads are always exact. Server-side traversal (`view: "descend"`) is used by + one case only, an unscoped multi-scope recall, so one agent's read never + reaches a sibling's scope. +- A filter whose `kinds` admits nothing reads no scopes. + +## The v2 envelope + +A document or learning is one event; a conversation is one event per turn, +appended in order. CortexDB's experience schema is closed (an unknown field is +a 422), so the structured data rides in the one free-form field: the event's +`content.text` is a JSON **envelope**: + +```json +{ "v": 2, "id": "<40-hex fingerprint>", "kind": "conversation", + "text": "", + "meta": { "...": "the item's whole MemoryMeta, on every event" }, + "title": "...", "mime": "...", + "learning_kind": "preference", "confidence": 0.8, "evidence": "...", + "turn": { "index": 0, "count": 3, "role": "user", "at": "...", "tool_calls": [] } } +``` + +| Field | Present on | Meaning | +| --- | --- | --- | +| `v` | every event | always `2`; any other value is ignored | +| `id` | every event | the item id, `StoreItem::fingerprint()` (a content digest) | +| `kind` | every event | `document`, `conversation` or `learning` | +| `text` | every event | body, the turn's text, or the learning's statement | +| `meta` | every event | the whole `MemoryMeta`, including the namespace | +| `title`, `mime` | documents, when set | | +| `learning_kind`, `confidence`, `evidence` | learnings (`evidence` when set) | | +| `turn` | conversation turns | `index` (0-based), `count`, `role`, `at`, `tool_calls` | + +Text that is not a v2 envelope is someone else's event and is ignored by +every reader. Decoding first tries the text as written (`/events` returns it +as stored), then, failing that, strips a `[role] ` prefix (`/recall` renders +text for a reader). A document whose body is still an unresolved URI is +refused at write time as `Error::InvalidRequest`. + +Rebuilding an item from envelopes: a document or learning takes the first +envelope; a conversation orders turns by `index` and keeps one per index (so a +duplicated or re-written turn does not repeat, and a turn that was never +written is absent). A learning with no `learning_kind` reads back as `Other`. + +## Lookup labels and digests + +Each event carries up to eight `context.labels`, each `tm::` followed by +the first 16 lowercase hex digits (64 bits) of the SHA-256 of the value: + +| Label | Value hashed | On | +| --- | --- | --- | +| `tm:i:` | the item id | every event | +| `tm:t:` | `meta.thread_id` | when set | +| `tm:s:` | `meta.source.id` | when set | +| `tm:r:` | `meta.repo` | when set | +| `tm:w:` | `meta.workspace` | when set | +| `tm:a:` | `meta.agent_id` | when set | +| `tm:l:` | `meta.language` | when set | +| `tm:k:` | the source kind (`meta.source.kind`) | every event | + +A label holds a **digest**, not the value, because the engine splits a label +filter on commas and bounds a label's length, and a path or source id may be +long or hold a comma. + +Reads use labels two ways: + +- **Item lookup.** Replay detection, `get`, conversation assembly and forget + by id ask the listing for `tm:i:` labels. Because a label is a + digest, every hit is re-checked against the envelope's real `id`. +- **Narrowing.** A read whose filter has a labelled field sends **one** label + filter to narrow server-side: the first set field of thread, source id, repo, + workspace, agent, language (in that order, most selective first), else the + filter's source kinds (several `tm:k:` labels, which the engine reads as + any-of). Only labels of one field may be sent together, since the engine + keeps events carrying *any* of the labels. + +The label only ever narrows. Every reader **always** re-applies the full +`MetaFilter` to the decoded envelope, so a digest collision costs a wasted row +and never a wrong answer. `folder` and `file_path` match as prefixes, which a +digest cannot, so they are never labelled and are filtered client-side only. + +## CortexDB behaviours the engine is shaped around + +Each was measured against a live CortexDB and was wrong in the first adapter. +The loopback doubles reproduce all of them (see [testing](testing.md)). + +- **Append-only.** There is no update route. Forget removes events but **not** + their idempotency records, so a reused body `idempotency_key` after a forget + is swallowed as a replay. +- **Accepted is not readable.** An append answers `202` and indexes afterwards. + The status route and the lifecycle stream are not readiness signals, so the + engine waits on the listing and on recall itself (see + [flows](cortex-flows.md#waiting-for-a-write-to-be-readable)). +- **The listing emits every event twice**, and `limit` counts the copies. +- **Unknown query parameters are ignored.** +- **Recall and the listing return different bytes** (`[role] {...}` versus the + stored text). +- **The forget selector field is `memory_ids`**, and anything else reads as + empty, which means the whole scope. +- **TinyHumans** rate-limits a user to 300 requests a minute and has no bulk, + `?wait=indexed` or health route. diff --git a/docs/architecture/cortex.md b/docs/architecture/cortex.md new file mode 100644 index 00000000..a53f910a --- /dev/null +++ b/docs/architecture/cortex.md @@ -0,0 +1,280 @@ +# CortexDB engine + +`CortexEngine` is the one `MemoryEngine` implementation TinyMemory ships: it +stores, lists, fetches, recalls and forgets over CortexDB's append-only event +log. It lives in `tinymemory-integrations`, module `cortex`, behind the +`cortex` feature (on by default), with the registry (`registry`) and the +configuration type (`config`) that build it. + +This is the overview. The detail is split into focused pages: + +- **this page**: surface, credentials, transport, failure mapping, endpoint + security, the registry and `MemoryConfig`; +- [cortex-wire.md](cortex-wire.md): the two wires, every endpoint and its + request and response shape, the scope layout, the v2 envelope and the lookup + labels; +- [cortex-flows.md](cortex-flows.md): step-by-step store, list, fetch, recall, + forget, get, explore, scope discovery and health; +- [testing.md](testing.md): the loopback doubles, the conformance suite and + the live tests. + +The module README (`crates/tinymemory-integrations/src/cortex/README.md`) is +the short in-tree version of this. + +## Surface + +```rust +use std::sync::Arc; +use tinymemory_integrations::cortex::{ + CortexCredential, CortexEngine, StaticBearer, CORTEX_API_ENDPOINT, TINYHUMANS_API_ENDPOINT, +}; + +// CortexDB's own /v1/* API, API key. +let direct = CortexEngine::direct(CORTEX_API_ENDPOINT, CortexCredential::api_key("ctx_..."))?; +// CortexDB behind the TinyHumans backend (/memory/*), bearer resolved per request. +let hosted = CortexEngine::tinyhumans( + TINYHUMANS_API_ENDPOINT, + Arc::new(StaticBearer::new("tiny_live_...")), +)?; +``` + +| Item | What it is | +| --- | --- | +| `CortexEngine::{new, direct, tinyhumans, wire}` | constructors (all fallible with `Error::Config`) and the wire accessor | +| `CortexWire { Direct, TinyHumans }` | which HTTP surface; `descriptor()` gives its registration | +| `CortexCredential { Static, Dynamic }` | how an engine authenticates; `api_key(..)` builds a static one | +| `BearerSource` (async `bearer()`), `StaticBearer` | a per-request token source, and a fixed token as one | +| `CORTEXDB_ENGINE_ID`, `TINYHUMANS_ENGINE_ID` | the config ids `cortexdb` and `tinyhumans` | +| `CORTEX_API_ENDPOINT`, `TINYHUMANS_API_ENDPOINT` | the default endpoints | +| `cortexdb_descriptor()`, `tinyhumans_descriptor()` | the `EngineDescriptor`s | +| `Error`, `Result`, `error_code`, `is_insufficient_credits` | the contract's error and two helpers for hosted failures | + +`Debug` on the engine shows the wire (by id) and the endpoint origin, never the +credential. A `CortexEngine` is `Clone` and cheap to share. + +| Engine id | `hosted` | `needs_endpoint` | `needs_key` | Default endpoint | `fetch_modes` | +| --- | --- | --- | --- | --- | --- | +| `cortexdb` | no | no | yes | `https://api-v1.cortexdb.ai` | `[Hybrid]` | +| `tinyhumans` | yes | no | yes | `https://api.tinyhumans.ai` | `[Hybrid]` | + +## Credentials + +Both wires authenticate with `Authorization: Bearer `. + +- **`CortexCredential::Static(String)`** (`CortexCredential::api_key`): one + fixed token, normally a CortexDB API key for the direct wire. A blank key is + `Error::Config` at construction. +- **`CortexCredential::Dynamic(Arc)`**: a token source the + engine consults on **every request attempt**. TinyHumans takes the host's + session JWT or `tiny_live_` API key, which rotates, so a refreshed token is + used at once without rebuilding the engine. `From>` is + implemented. +- **`BearerSource`**: `async fn bearer(&self) -> Result`. + Implementations must not log the token, and should return an error (not an + empty string) when no credential is available, for example when the host is + signed out. +- **`StaticBearer`**: a fixed token as a `BearerSource`. + +**Per-request bearer resolution.** The transport resolves the credential inside +each attempt (so every read retry, every poll, and every hosted write retry +re-asks the source). A source failure, a blank token, or a token that cannot +be an HTTP header value (CR or LF, any other byte a header may not carry) is +`Error::Unauthorized` and **no request is sent**. The refusal message carries +no part of the token. The token is trimmed before use. + +**Sensitive headers.** The `Authorization` value is marked sensitive on the +header (`HeaderValue::set_sensitive`), so nothing that formats the request +prints it. `Debug` on `CortexCredential` prints `Static()` or +`Dynamic()`, `StaticBearer` prints `StaticBearer()`, and +`EngineCredential` (below) is redacted the same way. No error message is built +from a credential. + +### The actor header + +On the direct wire every request also carries `X-Cortex-Actor`. CortexDB +serves every request as an actor; a minted token (the CortexDB cloud signs one +per account) is accepted only when the request names its subject, and +otherwise answers `401 ACTOR_MISMATCH`. The actor is the `caller` that +`GET v1/auth/whoami` reports for the key, which the client asks once and +caches (shared across clones): + +- **known**: `whoami` answered; the caller is sent on every request. (A + static operator key is served as `user:local`.) +- **absent**: the route is 404 or 405 (a server before the actor model); no + header, and `whoami` is not asked again. +- **unknown**: nothing learned yet, or a credential was just rejected (401 or + 403 clears the cache so a replaced key is looked up again). The next request + asks `whoami` again. A failed lookup is not cached: the request goes out + without the header and reports its own failure. + +The TinyHumans wire never sends the header; the backend names the actor. + +## Transport + +`HttpClient` (`cortex/transport/`) is shared by both wires. + +| Aspect | Behaviour | +| --- | --- | +| Request timeout | 60s per request | +| Connect timeout | 10s (or the request timeout if smaller) | +| Reads | `Attempts::RetryTransient`: 3 attempts, 250ms then 500ms apart, only on `Error::Unavailable` | +| Writes | `Attempts::Once`: one attempt, because a timeout leaves it unknown whether the write applied | +| Success body cap | 64 MiB (also checked against `Content-Length`); larger is `Error::Engine` | +| Error body cap | 64 KiB, read lossily, never failing; only a 300-character excerpt reaches a message | +| TinyHumans bodies | `{success,data}` is unwrapped; see below | +| Direct bodies | bare JSON; an empty success body is `null` | + +Bodies are read chunk by chunk and the cap is checked **before** each chunk is +appended, so a server that omits or understates `Content-Length` cannot +exhaust the host's memory. A body cut off mid-read is `Error::Unavailable`; a +body that is not valid JSON is `Error::Engine`. + +**Reads retry, writes do not**, at this level. Layers above add what each +operation needs: the hosted write claim and recovery, the hosted forget retry +and the visibility polls (see [flows](cortex-flows.md#hosted-writes-and-outcome-unknown-recovery)). +Recall and listings are the reads; the answer route and forget are sent once. + +### Idempotency claims + +On TinyHumans, every `POST` sent as a single attempt carries a fresh +`Idempotency-Key` header: experience writes (under a claim the writer chooses +and reuses across its own retries), the answer route, and forget. Recall and +listings, which retry, carry none. The Direct wire sends no such header; +writes there carry the body `idempotency_key` only. + +### TinyHumans envelope + +A 2xx body must be `{"success": true, "data": ...}`. `success: false` is +reported as a hosted failure (below); a missing `data`, a body without +`success`, or invalid JSON is `Error::Engine`. + +## Failure mapping + +Every message names the route (without its query string, which carries +scopes and cursors) and the endpoint **host**, never a credential. Anything the +backend itself said follows a spaced em-dash (` — `) and is cut to 300 +characters, so a status surface can keep the head and withhold the backend's +text. + +| HTTP status | `Error` variant | Notes | +| --- | --- | --- | +| 401, 403 | `Unauthorized` | message tells the user to check the API key (direct) or re-authenticate (hosted) | +| 402 | `Engine` | hosted: prefixed `[USER_INSUFFICIENT_CREDITS]`; see below | +| 404 | `NotFound` | | +| 400, 413, 422 | `InvalidRequest` | | +| 409 | `Conflict` | on a hosted write retry it triggers recovery instead | +| 429, 500, 502, 503, 504 | `Unavailable` | retried for reads; `is_transient()` is true | +| any other non-2xx | `Engine` | | +| timeout, DNS, TLS, connect, reset | `Unavailable` | message names the class, for example "TLS failed" or "the host could not be resolved; check the URL" | +| request could not be built | `Engine` | no retry will change it | +| response over the cap, invalid JSON, malformed envelope | `Engine` | | +| bearer source failure, blank or invalid token | `Unauthorized` | no request sent | + +**The `[CODE]` prefix.** The TinyHumans backend names every failure with an +`errorCode`. The contract's `Error` has no field for it, so a hosted failure's +message starts with `[CODE] ` (the code uppercased, restricted to ASCII +letters, digits and `_`, at most 64 characters). A failure with no +`errorCode` is filed under `UNAUTHORIZED` (401, 403), `USER_INSUFFICIENT_CREDITS` +(402), `RATE_LIMITED` (429) or `HTTP_`. `error_code(&Error) -> +Option<&str>` reads the code back, and returns `None` for a direct failure, a +local refusal, or a message that no longer starts with a well-formed prefix. + +**402 is `Engine`.** An exhausted credit balance is not transient +(`Unavailable` would invite a retry loop that cannot succeed until someone tops +up) and not a credential fault (`Unauthorized` would send the host to its +sign-in flow). It is the engine refusing to serve, which is what `Engine` +means, and the code lets a host tell it apart: +`is_insufficient_credits(&Error)` is true for an `Engine` error whose code is +`USER_INSUFFICIENT_CREDITS`, so a host can show a top-up prompt. + +Other errors the engine raises itself: `Error::Unsupported` for a fetch mode +other than `Hybrid`; `Error::InvalidRequest` for a malformed cursor or an +empty or oversized store batch; `Error::Config` for construction; and +`Error::Engine` for a listing past 500 pages, a cursor that does not advance, +or a write receipt that lacks `event_id`. + +## Endpoint security + +Every engine here is credentialed, so a cleartext endpoint would put the +bearer on the network. `CortexEngine::new` (and so `direct`, `tinyhumans` and +the registry) returns `Error::Config` for: + +- a URL that does not parse, or whose scheme is not `http` or `https`; +- an `http://` endpoint whose host is not loopback (`localhost`, or an IP that + `is_loopback()`, IPv6 `[::1]` included): "credentialed memory endpoints must + use https unless they are loopback"; +- a blank static credential. + +Loopback `http://` is allowed so local servers and the test doubles work. The +endpoint is operator supplied, which is why response bodies are capped. + +## Registry + +`registry` (feature `cortex`) is how a host turns configuration into an engine +without naming `CortexEngine`: + +- `list_engines() -> Vec`: every engine this build can + construct, `cortexdb` then `tinyhumans`. A host uses it to render a picker + (`needs_endpoint`, `needs_key`, `default_endpoint`, `fetch_modes`). +- `build_engine(id, &EngineSettings, EngineCredential) -> + Result>`. +- `EngineCredential`: `None` (default), `Static(String)`, or + `Dynamic(Arc)`. `Debug` is redacted. + +`build_engine` picks the wire from the id (`cortexdb` is `Direct`, +`tinyhumans` is `TinyHumans`) and resolves the endpoint: the setting, trimmed, +if it is not blank, else the engine's default. It returns `Error::Config` for +an unknown id; a missing credential (`None`, or a blank `Static`); and +everything `CortexEngine::new` refuses (not an HTTP(S) URL, cleartext off +loopback). Messages never carry the credential. These are re-exported at the +crate root: `tinymemory_integrations::{build_engine, list_engines, +EngineCredential}`. + +## MemoryConfig + +`config::MemoryConfig` says which engine a host uses and how each is reached. +It holds **no credential**: a host keeps keys in its own secret store and +passes one to `build`, so a config file can be shared or logged. + +| Field | Type | Meaning | +| --- | --- | --- | +| `engine` | string | the selected engine id; `DEFAULT_ENGINE` is `tinyhumans` | +| `engines` | map id to `EngineSettings` | per-engine settings; optional; an absent engine uses its defaults | +| `engines..endpoint` | string, optional | base URL; absent or blank uses the engine's default | + +TOML: + +```toml +engine = "cortexdb" + +[engines.cortexdb] +endpoint = "https://cortex.example.com" + +# An engine with no entry uses its defaults; an empty table is fine too. +[engines.tinyhumans] +``` + +JSON (the same shape): + +```json +{ "engine": "cortexdb", + "engines": { "cortexdb": { "endpoint": "https://cortex.example.com" } } } +``` + +`MemoryConfig::default()` is `engine = "tinyhumans"` with no settings. +`settings()` returns the selected engine's `EngineSettings` (or the defaults), +and `build(credential)` is `build_engine(&self.engine, &self.settings(), +credential)`. Unknown fields in a config are ignored on read. + +```rust +use std::sync::Arc; +use tinymemory_integrations::{EngineCredential, MemoryConfig, cortex::StaticBearer}; + +let config: MemoryConfig = toml::from_str(r#"engine = "tinyhumans""#)?; +let engine = config.build(EngineCredential::Dynamic(Arc::new(StaticBearer::new("tiny_live_..."))))?; +``` + +The crate-level `Error` (`tinymemory_integrations::Error`) is the contract's +`tinymemory_api::Error`: the engine and the registry return it directly, and +the `documents`, `sources` and `import` modules keep a typed error of their +own that converts into it. diff --git a/docs/architecture/integrations-sources.md b/docs/architecture/integrations-sources.md new file mode 100644 index 00000000..98be38d1 --- /dev/null +++ b/docs/architecture/integrations-sources.md @@ -0,0 +1,188 @@ +# Integrations: sources + +The `sources` module of `tinymemory-integrations` (features `sources` and +`sources-network`) turns a configured source into `StoreItem`s. Overview of the +crate: [integrations.md](integrations.md). Module README: +[`src/sources/README.md`](../../crates/tinymemory-integrations/src/sources/README.md). + +The module reads what it is handed. Where a host stores its sources, when it +syncs them, credentials, OAuth and egress budgets all stay with the host. + +## Configuration: MemorySourceEntry and SourceKind + +`MemorySourceEntry` is the configuration a host persists (it derives serde and +`JsonSchema`). Its `kind` (`SourceKind`, snake_case on the wire) selects which +optional fields are required; `validate()` checks them. + +| `SourceKind` | Required | Other fields | Item's `SourceKind` | +| --- | --- | --- | --- | +| `folder` | `path` | `glob` | `Folder` | +| `file` | `path` | | `File` | +| `conversation` | | | `Conversation` | +| `web_page` | `url` | `selector` | `Link` | +| `github_repo` | `url` | `branch`, `paths`, `max_commits`, `max_issues`, `max_prs` (default 1000 each) | `Github` | +| `rss_feed` | `url` | `max_items` (default 50) | `Rss` | +| `composio` | `toolkit`, `connection_id` | | `Composio` | + +Every entry also needs a non-blank `id` (no `:` or control characters) and a +non-empty `label`, and carries `enabled` and optional sync-budget fields +(`max_tokens_per_sync`, `max_cost_per_sync_usd`, `sync_depth_days`) that the +host's sync runner interprets. An empty string counts as missing. Failure is +`Error::Invalid` naming the first failing rule. + +## Readers + +`SourceReader` is the narrow trait: `kind`, `list_items(source, workspace)`, +`read_item(source, item_id, workspace)` and `read_store_item(source, item, +workspace, converter)`. The default `read_store_item` reads the content and maps +it through `items::content_item`; local readers override it to work from raw +bytes so a bound converter can handle PDF or DOCX. + +`readers::reader_for(kind)` returns only the local readers (folder, file, +conversation), which are safe to drive on a timer. The network kinds return +`None`, meaning "route through the host's sync runner". With +`sources-network`, `reader_for_request(kind)` returns a reader for every kind, +for a host servicing an explicit user request, never a polling loop. + +| Reader | Items | Notes | +| --- | --- | --- | +| `FolderReader` | one per selected file; id is the folder-relative slash path | With a `glob`, exactly the matches (compiled to a regex over the relative path). Without one, markdown, plain text and source code (`is_default_candidate`). Skips hidden files and directories and `.git`, `.hg`, `.svn`, `target`, `node_modules`, `__pycache__`, `venv`; never follows symlinks; refuses files over `FOLDER_FILE_SIZE_CAP_BYTES` (10 MiB). A relative `path` is anchored on the workspace. | +| `FileReader` | exactly one, id is the file name | Same size cap. `FileReader::read_path` reads a path with no configured source. | +| `ConversationReader` | one per `/threads/.json` | Threads are `{title, messages: [{role, content, created_at?}]}`. Messages with blank text or an unknown role are dropped. Timestamps: RFC 3339, or epoch seconds or milliseconds. The id may not contain separators or `..`. | +| `WebPageReader` | one: the page URL | With a CSS `selector`, only the text of matching elements (plain text; only the last compound of a descendant chain is honoured). Otherwise the whole page as markdown. 10 MiB body cap. | +| `RssReader` | one per feed entry (RSS or Atom), up to `max_items` | The parsed feed is cached for 60 seconds so a list-then-read pass downloads it once. 5 MiB cap; non-UTF-8 bodies are refused. | +| `GithubReader` | `commit:`, `issue:`, `pr:` | See below. | +| `ComposioReader` | one: the connection | A placeholder; see Composio. | + +### local_file and ensure_within_base + +`readers::local_file` is shared by the folder and file readers: a size-capped +whole-file read into `LocalFile { path, id, bytes, modified }`, and the +path-traversal guard `ensure_within_base(base, target)`. The guard +canonicalises both paths (resolving symlinks and `..`) and returns +`Error::PathEscape("path traversal denied")` when the target is outside the +base, or `Error::Io` when either cannot be canonicalised. The folder reader +applies it on every read; the conversation reader applies it within the +threads directory. + +### GitHub transports + +The reader pulls **project activity**, not source code, from +`https://github.com//` (extra path segments such as `/tree/main` +are rejected). It combines three transports: + +- **Commits:** a bare clone under `/git_cache//.git`, + fetched with an explicit refspec and listed with `git log`, honouring + `branch` and `paths`. `git` must be on `PATH`. If the clone fails, it falls + back to the commits API. +- **Issues and pull requests:** `gh api` when the `gh` CLI is available + (authenticated, higher rate limit; probed once per process), otherwise the + unauthenticated REST API at `api.github.com`. The list pass caches full rows + so reads do not refetch. +- Limits: `max_commits`, `max_issues`, `max_prs` per sync, default 1000 each. + Listing fails only when every call failed; otherwise the partial list is + returned. Failures surface as `Error::Reader`. + +## Items mapping + +`items` maps reader output to `StoreItem`s. Every item's `meta.source` is +`SourceRef { kind, id: Some(entry.id) }`, and every document body is markdown +(local files through the host's converter, reader bodies through +`markdown_from_text`). + +| Kind | Item | Metadata filled | +| --- | --- | --- | +| folder, file | document | `workspace`, `folder`, `file_path` (canonical), `language`, `observed_at` (mtime), `mime` | +| github | document | `repo` (`owner/name`), `commit` (commits), `url` (issues and PRs), `observed_at` | +| web page | document | `url` | +| rss | document | `url` (entry link), `observed_at` (published) | +| composio | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | +| conversation | conversation | `workspace`, `thread_id`, `turns` (`0..=n-1`), `observed_at` (last turn, else mtime) | + +`collect_items(reader, entry, workspace, converter)` lists and reads every +item, returning `Collected { items, skipped }`. One bad item lands in `skipped` +with its error and the pass continues; only a listing failure is an `Err`. +Other entry points: `file_item` (a path with no source), `conversation_item`, +`content_item` and `items::local_file_item`. + +## Composio + +Composio data does not arrive item by item. A host runs toolkit actions with +its own credentials and hands the raw responses to `sources::composio`, which +holds no credential, opens no socket and decides nothing about when to sync. + +1. **Normalisers**, pure `serde_json::Value` transforms, one module per + toolkit. They walk Composio's envelope variants (top level, under `data`, + under `data.data`) and return the first array found: `clickup` + (`extract_tasks`), `github` (`extract_issues`), `linear` + (`extract_issues`), `notion` (`extract_results`, `extract_page_markdown`), + each with title, id and updated-time helpers. `fields::pick_str` is the + shared lookup: it tries dotted paths, descends only through objects, and + rejects non-string leaves. +2. **Post-processors the host must call** for two toolkits, because their raw + responses are too verbose: + - `gmail_post_process::post_process(slug, arguments, &mut data)` rewrites a + `GMAIL_FETCH_EMAILS` response into slim `messages[]` (other Gmail slugs + pass through; `raw_html: true` in the arguments skips the reshape). If the + response carries a response-level `markdownFormatted` string, call + `apply_response_level_markdown(&mut data, markdown)` **before** + `post_process`; it is a no-op unless the split count matches the message + count. `format_email_local_time` renders in the host's local timezone; + the raw UTC fields are preserved. + - `slack_post_process::post_process(slug, arguments, &mut data)` reshapes + `SLACK_FETCH_CONVERSATION_HISTORY`, `SLACK_LIST_CONVERSATIONS` and + `SLACK_SEARCH_MESSAGES`; unknown slugs are no-ops. `channel_id` for history + is injected by the host (it is in the request, not the response), and user + ids are resolved by the host. +3. `normalise_payload(toolkit, &data)` returns `ComposioDocument`s (id, title, + markdown body, url, `observed_at`, `thread_id`, `repo`), dispatching on the + case-insensitive toolkit slug: `gmail`, `slack`, `github`, `linear`, + `notion`, `clickup`; any other toolkit falls back to each record as fenced + JSON, so a new toolkit is ingested verbosely rather than dropped. Records + with no text are skipped. `payload_items(toolkit, source_id, &data)` wraps + them as `StoreItem::Document` with `source.kind = Composio`, + `source.id = source_id` and `tags = [toolkit]`. + +`readers::composio::ComposioReader` is only a placeholder so +`reader_for_request` can serve every kind: `list_items` returns the connection +as one sync target. + +## Fetching and the SSRF guard (sources-network) + +`fetch::fetch_url(url)` fetches one URL into a `RawDocument`: the `Content-Type` +becomes the declared MIME, the URL the origin, and the last path segment (if +it has an extension) the filename. `fetch::link_item(url, source_id, +converter)` converts it into a document with `source.kind = Link` and `url` +set. The cap is `MAX_DOCUMENT_BYTES` (32 MiB), applied while streaming. No +retries, no robots.txt, no scheduling. Errors: `Invalid` (malformed or refused +URL, empty body), `Unreachable` (never completed, interrupted read, may +succeed later), `Upstream` (non-success status), `TooLarge`. + +The web-page and RSS readers fetch through the same path with tighter caps +(10 MiB and 5 MiB). `fetch::ssrf` is public so a host fetching a user-supplied +URL by other means applies the same policy. + +Guard rules: + +- **Scheme:** `http` and `https` only. +- **Host text:** refused are empty hosts, `localhost`, `.local` and + `.internal` names, single-label names (internal service names such as + `redis`), and IP literals that are not globally routable. +- **One address classifier** serves literals and resolved addresses. Not + fetchable: loopback, private, link-local (including `169.254.169.254`), + unspecified, CGNAT (`100.64.0.0/10`), `192.0.0.0/16`, multicast, broadcast, + documentation and benchmarking ranges, reserved `240.0.0.0/4`, IPv6 + unique-local and link-local; IPv4-mapped and IPv4-compatible IPv6 addresses + are judged by their IPv4 part. +- **IPv6 literals fail closed.** A URL such as `http://[2606:4700::1111]/` is + refused whatever the address: the host text keeps its brackets, does not parse + as an IP, and has no dot, so the single-label rule blocks it. A hostname with + an AAAA record is still vetted when resolved. +- **Resolver:** `PublicOnlyResolver` keeps only public addresses and fails the + request when none remain, pinning the connection to a vetted address with no + re-resolution between check and connect. +- **Redirects** are re-checked per hop; a refused hop is not followed, and the + read fails on the resulting non-success status. +- **Client:** 20-second timeout and the user agent `openhuman`. +- **Body caps** are enforced while streaming (`read_body_capped`), with an early + rejection on a truthful `Content-Length`. diff --git a/docs/architecture/integrations.md b/docs/architecture/integrations.md new file mode 100644 index 00000000..520b0094 --- /dev/null +++ b/docs/architecture/integrations.md @@ -0,0 +1,268 @@ +# Integrations + +`tinymemory-integrations` holds everything that connects the contract +([`tinymemory-api`](api.md)) to the outside world. Each integration is a module +behind a Cargo feature, so a host links only what it uses. This page covers the +module map, documents, safety and the legacy import. Sources are large enough +for their own page: [integrations-sources.md](integrations-sources.md). The +CortexDB engine and the registry are in [cortex.md](cortex.md). + +| Module | Feature | Page | +| --- | --- | --- | +| `cortex`, `registry`, `config` | `cortex` (default) | [cortex.md](cortex.md) | +| `documents` | `documents`, `documents-office` | [Documents](#documents) | +| `sources` | `sources`, `sources-network` | [integrations-sources.md](integrations-sources.md) | +| `safety` | `safety` | [Safety](#safety) | +| `import` | `legacy-import` | [Legacy v1 import](#legacy-v1-import) | + +Per-feature dependency weight is in the +[crate README](../../crates/tinymemory-integrations/README.md). Each module also +has its own README with the full detail: +[`documents`](../../crates/tinymemory-integrations/src/documents/README.md), +[`sources`](../../crates/tinymemory-integrations/src/sources/README.md), +[`safety`](../../crates/tinymemory-integrations/src/safety/README.md) and +[`import`](../../crates/tinymemory-integrations/src/import/README.md). + +## How the pieces compose + +The modules do not call each other's engines or schedule anything; the host +composes them into the write path: + +```text +sources ──▶ documents ──▶ safety ──▶ engine.store +``` + +`sources` depends on `documents` for conversion (the `sources` feature implies +`documents`). `safety` and `import` stand alone. Nothing here scrubs, converts +or schedules on an engine's behalf: scheduling, credentials, OAuth and egress +budgets are the host's. + +## Errors + +The crate's `Error` is `tinymemory_api::Error`. `documents`, `sources` and +`import` keep a typed error each, because their failures are worth matching on +before they reach an engine, and each converts into the contract error with +`From`: + +| Module error | Maps to | +| --- | --- | +| `documents::Error::Invalid`, `TooLarge` | `InvalidRequest` | +| `documents::Error::UnsupportedFormat` | `Unsupported` | +| `documents::Error::Converter` | `Engine` | +| `sources::Error::Invalid`, `PathEscape`, `TooLarge`, `Json` | `InvalidRequest` | +| `sources::Error::NotFound` | `NotFound` | +| `sources::Error::Unreachable` | `Unavailable` | +| `sources::Error::Upstream`, `Reader`, `Io` | `Engine` | +| `sources::Error::Document(e)` | whatever `e` maps to | + +## Documents + +Feature `documents` (and `documents-office`). The module turns bytes into +markdown and wraps the result as a `StoreItem::Document`. It does no I/O. + +### Format detection + +`DocumentFormat::sniff(bytes, filename, mime)` consults three signals in order +of trustworthiness: + +1. **Magic bytes.** `%PDF-` is a PDF. A zip (`PK\x03\x04`) is an Office package + of some kind, refined by its part names read from the central directory + (`word/`, `xl/`, `ppt/`); failing that, a MIME type or filename naming an + Office format; failing that, `Docx`. +2. **The declared MIME type**, parameters stripped and compared + case-insensitively. Legacy binary types (`application/msword`, + `application/vnd.ms-excel`, `application/vnd.ms-powerpoint`) are deliberately + not claimed. +3. **The filename.** A name `language_for_path` recognises is `Code` (checked + first, so `CMakeLists.txt` is code, not plain text); otherwise the extension + (`md`, `txt`, `html`, `pdf`, `docx`, `xlsx`/`xlsm`, `pptx`). + +With none of those, a buffer opening with `` whose report tallies what changed. It runs on-device with +regular expressions and checksums, makes no network calls, never fails, and +errs toward redacting a harmless string rather than letting a secret into a +long-lived store. It is not called by any engine: the host runs it between +conversion and `store`. + +What it does, in order, on a text: + +1. blocks private-key blocks whole (`[REDACTED_PRIVATE_KEY]`); +2. redacts credential markers: the value after `/secret/` in a one-time-secret + URL and after a `Bearer ` scheme; +3. redacts credential shapes: provider token prefixes, JWTs and `key=value` + assignments with a sensitive key; +4. redacts PII with typed tokens (`[REDACTED_PII_CPF]`, ...): checksum-gated + national IDs, credit cards, IBANs and phone numbers. + +Per kind, `scrub_item` scrubs a document's title and text body, every +conversation turn's text, a learning's text and evidence, and `meta.url` +(query strings carry tokens). Metadata identifiers (paths, repo, commit, +thread, agent ids) and `DocumentBody::Uri` bodies are left alone, because +filters match on the identifiers. The single tunable is `Policy`'s +`BareCardGate` (`LuhnOnly`, the default and strictest, or `Corroborated`). +JSON values are scrubbed with `sanitize_json`, which also redacts by key name. +Email addresses are detected (`has_likely_email`) but not redacted. See +[`safety/README.md`](../../crates/tinymemory-integrations/src/safety/README.md) +for the pipeline, the strict `has_likely_pii` boundary check and the known +limits. + +## Legacy v1 import + +Feature `legacy-import`. The `import` module reads a v1 (embedded TinyCortex) +workspace and yields v2 `StoreItem`s, resumably. The v1 engine is not linked: +the importer reads its SQLite files with `rusqlite` (bundled), opened +read-only, plus chunk bodies from disk. It never writes to the legacy +workspace. + +### Detection + +`LegacyWorkspace::open(path)` requires `/memory/memory.db` to be a SQLite +database with the `memory_docs`, `episodic_log` and `user_profile` tables and +the columns the importer reads. A missing path is `Error::NotFound`; anything +else that is not a v1 store is `Error::NotLegacy` with the reason. Columns added +by later v1 migrations are probed and used when present. +`memory_tree/chunks.db` is optional and skipped silently when absent or +unusable. Per-profile stores (`memory-/memory.db`) are not read; open each +as its own workspace. + +### Mapping v1 to v2 + +Every item gets `meta.source = { kind: Import, id: }` and +`meta.workspace` set to the workspace path. + +| Section (in order) | Legacy rows | Legacy id | v2 item | +| --- | --- | --- | --- | +| documents | `memory_docs` in document namespaces | `memory_docs:` | `Document` | +| chunks | `mem_tree_chunks` grouped by source | `mem_tree_chunks::` | `Conversation` for `chat`, else `Document` | +| conversations | `episodic_log` grouped by session | `episodic_log:` | `Conversation` | +| learnings | `memory_docs` in `learning:*` and `global` | `memory_docs:` | `Learning` | +| profile | live `user_profile` facets | `user_profile:` | `Learning(Preference)` | + +Notable decisions: `event` namespaces and the `kv_*` tables are not imported +(bookkeeping, not recall material); unknown conversation roles become `User`; +learning classes map onto kinds (`style`/`channel` to `Preference`, `identity` +to `Fact`, `tooling` to `Procedure`, `veto` to `Correction`, others to +`Other`); dropped or forgotten profile facets and blank rows are skipped. The +full rules are in +[`import/README.md`](../../crates/tinymemory-integrations/src/import/README.md). + +### Checkpoint and resumption + +Sections run in a fixed order and, within one, keys ascend in SQLite `TEXT` +order. A `Checkpoint` records the last yielded key per section; every +`ImportedItem { item, checkpoint }` carries the checkpoint covering it and +everything before. `LegacyWorkspace::items_from(&checkpoint)` yields exactly +what `items()` yields after that item (given the legacy store did not change +in between). A checkpoint serialises with `to_json` and `from_json` for the +host to persist. Pages of keys are fetched `DEFAULT_PAGE_SIZE` (256) at a time, +so memory is bounded. + +### migrate and migrate_with + +```rust,ignore +let workspace = LegacyWorkspace::open(path)?; +let from = saved.map(|json| Checkpoint::from_json(&json)).transpose()?; +let report = migrate_with(engine.as_ref(), workspace, from, |checkpoint| { + save(checkpoint.to_json()); +}) +.await?; +``` + +`migrate(engine, workspace, from)` stores every item after `from` (all of them +for `None`) in `store_many` batches of at most `MAX_STORE_MANY` (100) and +returns a `MigrationReport { stored, replayed, batches, checkpoint }`. +`migrate_with` also calls `on_batch(&Checkpoint)` after each stored batch. + +Resume semantics: + +- After a batch is stored, its last item's checkpoint is committed: passed to + `on_batch` and kept as `report.checkpoint`. +- An engine failure is `Error::Engine { source, checkpoint }`, carrying the last + committed checkpoint (or `from` if no batch was stored). Call `migrate` again + with it. The failed batch may have stored a prefix; the engine answers those + as replays, so a resumed or repeated run never duplicates (a second full run + reports every item as `replayed`). +- A legacy read failure (`Sqlite`, `Io`) is returned as is. Every checkpoint + committed before it has already reached `on_batch`. +- The workspace is taken by value: its SQLite handle is not `Sync`, and owning + it keeps the future `Send` so a long import can run on a spawned task. + +## Engine and registry + +For `CortexEngine`, `list_engines`, `build_engine`, `EngineCredential` and +`MemoryConfig`, see [cortex.md](cortex.md). diff --git a/docs/architecture/namespaces.md b/docs/architecture/namespaces.md new file mode 100644 index 00000000..a32b71da --- /dev/null +++ b/docs/architecture/namespaces.md @@ -0,0 +1,204 @@ +# Namespaces + +Memory is a **tree of nodes**. The root holds what every agent shares; below it +sit agents, teams, users, workspaces and projects, nested as deep as a host +needs. Every stored item lives at exactly one node. A reader names a `Reach` +saying which nodes it sees. All of it is in `tinymemory_api::namespace`. + +Inside a node each item kind (learnings, documents, conversations) is kept +apart, so an engine can hold, recall and erase each on its own +([cortex.md](cortex.md) maps this onto scopes). + +## Syntax + +A `Namespace` is the path from the root, written as `/`-separated +`kind:id` segments: + +```text +root the root (the empty path) +agent:researcher +team:acme/agent:writer +team:acme/agent:writer/agent:helper +``` + +- `""` (after trimming) and `root` both parse to the root; the root always + prints as `root`. +- A segment is `kind:id`, split at the first `:`. A missing `:` or an unknown + kind is `Error::InvalidRequest`. +- On the wire (serde) a namespace is that string. `MemoryMeta` omits it when + it is the root, and old envelopes without it read as root. + +### Segment kinds + +| `SegmentKind` | Prefix in a path | Names | +| --- | --- | --- | +| `Agent` | `agent` | An agent, or a sub-agent nested under its parent | +| `Team` | `team` | A team of agents sharing memory | +| `User` | `user` | A human user | +| `Workspace` | `ws` | A shared workspace | +| `Project` | `project` | A project | + +`SegmentKind::as_str()` gives the path prefix (`ws` for `Workspace`). Note +that the enum's own serde form is `snake_case` of the variant, so +`Workspace` serialises as `"workspace"`; the `ws` prefix is only the +namespace path spelling. + +### Limits + +| Limit | Value | Error when exceeded | +| --- | --- | --- | +| Depth | at most 8 segments (`Namespace::new`, parsing) | `InvalidRequest("a namespace nests at most 8 deep")` | +| Segment id length | 1 to 128 characters | `InvalidRequest` | +| Segment id charset | `A-Z a-z 0-9 _ -` | `InvalidRequest` | + +Because `:` and `/` are outside the charset, an id can never be confused +with the path syntax. + +### Constructing + +| Constructor | Behaviour | +| --- | --- | +| `Namespace::ROOT` / `Namespace::default()` | The root. | +| `Namespace::new(Vec)` | Checks depth only. | +| `"team:acme/agent:writer".parse::()` | Parses and checks every segment. | +| `Namespace::agent("writer")` | One agent directly under the root, id [sanitised](#sanitising-host-ids). | +| `Segment::new(kind, id)` | Checks the id; rejects an invalid one. | +| `Segment::sanitized(kind, raw)` | Never fails; see below. | + +Accessors: `is_root()`, `depth()` (root is 0), `segments()`; `Segment` has +`kind()` and `id()`. + +### Sanitising host ids + +Host identifiers (an email, a UUID with unusual characters, a display name) +rarely fit the charset. `Segment::sanitized(kind, raw)` maps any string onto a +valid id: + +- a valid `raw` is kept unchanged; +- otherwise each illegal character becomes `-`, the result is cut to 119 + characters, and `-` plus the 8-hex-digit FNV-1a hash of the **original** + is appended, so two different raw ids that clean to the same text stay + distinct; +- an empty `raw` becomes `_`. + +```text +"writer" → agent:writer +"o'neil@acme.com" → agent:o-neil-acme-com-<8 hex of the original> +"" → agent:_ +``` + +The hash is a stable, dependency-free disambiguator, not a security +boundary. (The empty-input result `_` is also a valid raw id, so `""` and +`"_"` produce the same segment.) + +## Placement + +An item's node is `MemoryMeta::namespace` (default: root). Placing an item is +just setting that field on the `StoreItem`'s metadata before `store`. + +The namespace is part of the item's [fingerprint](operations.md#idempotency-and-fingerprints) +(a root namespace is not serialised, so root fingerprints are unchanged). So +the **same text learned by two agents is two items**, one per node, and each +can be forgotten on its own. + +## Reach + +```rust +pub struct Reach { pub at: Namespace, pub inherit: bool, pub descendants: bool } +``` + +| Field | Default | Meaning | +| --- | --- | --- | +| `at` | root | The node read from | +| `inherit` | `true` | Also read every ancestor of `at` (so an agent sees what its team and the root share) | +| `descendants` | `false` | Also read everything below `at` | + +`Reach::admits(ns)` is true when `ns == at`, or `inherit` and `ns` is an +ancestor of `at`, or `descendants` and `ns` lies below `at`. **A sibling is +never admitted**: one agent's memory is invisible to another unless written to +a node both inherit. + +| Constructor | `inherit` | `descendants` | Sees | +| --- | --- | --- | --- | +| `Reach::of(at)` | yes | no | `at` and its ancestors: an agent's ordinary reach | +| `Reach::exact(at)` | no | no | exactly one node | +| `Reach::subtree(at)` | no | yes | `at` and everything below it, no ancestors | +| `Reach::default()` | yes | no | `Reach::of(root)`: **only the root** | + +`Reach::nodes()` lists the nodes read exactly, root first (`at` and, when +`inherit`, its ancestors). Descendants cannot be enumerated from the reach; +an engine reads them as one subtree below `at`. + +Two different notions of "everything": + +- `MetaFilter::reach == None` reads **every namespace**. +- `Reach::default()` reads **only the root**. Reading everything with a reach + takes `Reach::subtree(Namespace::ROOT)`. + +### Worked example + +```text +root R shared by everyone +└── team:acme T shared by the team + ├── agent:writer W + │ └── agent:helper H a sub-agent of the writer + └── agent:editor E +``` + +Items exist at R, T, W, H and E. What each reach sees: + +| Reach | Sees | +| --- | --- | +| `of(team:acme/agent:writer)` | R, T, W | +| `of(team:acme/agent:editor)` | R, T, E (never W or H) | +| `exact(team:acme/agent:writer)` | W | +| `of(team:acme/agent:writer/agent:helper)` | R, T, W, H | +| `subtree(team:acme/agent:writer)` | W, H | +| `subtree(team:acme)` | T, W, H, E (not R) | +| `Reach { at: team:acme, inherit: true, descendants: true }` | R, T, W, H, E | +| `of(root)` / `default()` | R | +| `subtree(root)` | R, T, W, H, E | +| no reach (`None`) | R, T, W, H, E | + +## What each operation does with reach + +| Operation | Namespace handling | +| --- | --- | +| `store`, `store_many` | Write at `meta.namespace`. No reach involved. | +| `recall` | `filter.reach` confines the items the answer may draw on. | +| `fetch` | `filter.reach` confines the ranked items. | +| `list` | `filter.reach` confines the listing. | +| `explore` | `filter.reach` confines the items counted. `Facet::Namespace` groups by node (the path string, `root` for the root); narrowing a bucket sets `Reach::exact(node)`, overwriting any reach. | +| `get` | `GetRequest::reach` leaves out ids outside it, as if they named nothing. | +| `forget` by filter | `filter.reach` confines what is removed. | +| `forget` by ids | **Not scoped.** The ids are removed wherever they live. | + +Because forget by ids is unscoped, a caller confined to a reach must first +`get` the ids under its reach and forget only the ids that came back. A +filter whose only field is a `reach` is non-empty, so it is a valid forget +target meaning "everything in reach". + +Reach with `inherit` (the default) includes ancestors, so a forget by filter +under `Reach::of(agent)` can remove memory the agent shares with its team or +the root; use `Reach::exact` to confine removal to the agent's own node. + +`MetaFilter::matches` applies the reach to the item's namespace like any +other field, and every engine must agree: the conformance `namespaces` check +covers reaches, `get`, `fetch`, the namespace facet and a forget scoped to one +node. + +## How tools pin it + +A model never chooses a namespace or a reach. `tinymemory-tools` takes them +from the host in a `ToolScope { place, reach, writes }`: + +- every item a tool stores gets `meta.namespace = place`; +- every read filter's `reach` is **overwritten** with the scope's reach, and + `memory_get` passes it as `GetRequest::reach`; +- `memory_forget` by ids reads them back under the reach first and forgets + only those found; +- a `namespace` or `reach` key in a tool's arguments is refused with + `InvalidRequest`, and the `namespace` facet is not offered to the model. + +See [tools.md](tools.md) for the full contract. `context.md` takes the same +reach through `ContextSpec::reach`. diff --git a/docs/architecture/operations.md b/docs/architecture/operations.md new file mode 100644 index 00000000..399f7a5a --- /dev/null +++ b/docs/architecture/operations.md @@ -0,0 +1,247 @@ +# Operations + +What each `MemoryEngine` operation does, independent of engine. For the +types see [api.md](api.md) and [api-items.md](api-items.md); for how CortexDB +realises them see [cortex.md](cortex.md). + +Every operation starts the same way: **validate the request** (its `validate` +method) and return `Error::InvalidRequest` before touching storage. Filters +are applied identically everywhere through `MetaFilter::matches`, including +the `reach` that confines which [namespaces](namespaces.md) are visible. + +## store + +`store(item) -> StoreReceipt { id, replayed }` + +1. `item.validate()`. A blank body, unresolved `Uri`, empty conversation, + blank learning or out-of-range confidence is `InvalidRequest`. +2. Derive the item's identity from `item.fingerprint()`. +3. If the engine already holds that item, write nothing and return the + existing id with `replayed: true`. +4. Otherwise write it at `meta.namespace` (the item lands at exactly one + node) and return `replayed: false`. + +Which fields are in the fingerprint, and why `observed_at` is not, is in +[Idempotency](#idempotency-and-fingerprints). + +## store_many + +`store_many(items) -> Vec`, for imports, backfills and syncs. + +1. `validate_many`: `1..=100` items (`MAX_STORE_MANY`), each valid. Empty or + oversized is `InvalidRequest`; so is the first invalid item. +2. Store the items **in order**. The default implementation calls `store` per + item; an engine may batch. +3. Receipts come back in item order. An item repeated within the batch is a + replay of its first copy. +4. On return every item is readable through `list`, `get` and `forget`. + Ranked `fetch` and `recall` **may lag** a moment behind for all but the + last item; that is what lets an engine skip a per-item wait. +5. On an error, the items before the failing one are stored. Sending the batch + again is safe: the stored ones come back as replays. + +## fetch + +`fetch(req) -> FetchPage { hits, next_cursor }`: raw retrieval, no synthesis. + +1. The engine checks `req.mode` against its descriptor + (`EngineDescriptor::ensure_mode`): a mode it does not serve is + `Error::Unsupported`. Hosts read `fetch_modes` and never offer one the + engine lacks. +2. `req.validate()`: blank query or zero limit is `InvalidRequest`. +3. Rank items admitted by `req.filter` in `Keyword` (lexical), `Vector` + (embedding) or `Hybrid` (the engine's blend) mode. Hits are best first, + each with a `score`. +4. Return up to `limit` hits and a `next_cursor` when more remain + ([cursors](#cursors-and-paging)). + +### Fetch modes and descriptor gating + +`EngineDescriptor::fetch_modes` is the engine's declaration of what it +serves. An engine need not serve all three (CortexDB declares only `Hybrid`). +Gating is by declaration: `supports(mode)` answers, `ensure_mode(mode)` +fails with `Unsupported("engine `` does not offer fetch")`. +`tinymemory-tools` mirrors this: the `memory_fetch` tool's `mode` enum lists +exactly the engine's modes, and an engine serving none gets no such tool. +The conformance suite checks that every declared mode works and every +undeclared one is `Unsupported`. + +## recall + +`recall(req) -> RecallAnswer { answer, citations, model? }` + +1. `req.validate()`: blank question or zero limit is `InvalidRequest`. +2. Gather at most `limit` citations from items admitted by `req.filter`. +3. Synthesise an answer, optionally steered by `req.instructions`. How the + engine answers is its own business. +4. Return the answer text, its `Citation`s and, when the engine reports it, + the model. + +Every citation's `id` must resolve through `list`; the conformance +suite checks it. + +## list + +`list(req) -> ListPage { items, next_cursor }`: a query-free listing. + +1. `req.validate()`: zero limit is `InvalidRequest`. +2. Return up to `limit` items admitted by `req.filter`, each a `Hit` with + `score == 0.0`, plus a `next_cursor` when more remain. + +The contract does not promise an order across engines, only that following +cursors visits every matching item and that paging terminates. `list` is the +primitive the default `explore` and `get` are built on. + +## forget + +`forget(target) -> ForgetReport { forgotten }` + +`ForgetTarget::validate` runs first: + +| Target | Rule | +| --- | --- | +| `Ids(ids)` | At least one id, else `InvalidRequest("forget needs at least one id")`. | +| `Filter(filter)` | Must not be empty (`MetaFilter::is_empty`), else `InvalidRequest`: an empty filter would mean everything, so the contract refuses it. | + +- **By ids**: remove those items, **wherever they live**. Ids are not scoped + by namespace. Ids that name nothing are skipped and not counted. A caller + confined to a reach reads the ids first with `get` under that reach and + forgets only what came back (this is what `memory_forget` does). +- **By filter**: remove every item the filter admits. The filter's `reach` + confines it. A filter holding only a `reach` is not empty, so + `Filter(MetaFilter { reach: Some(..), .. })` forgets everything in that + reach. Callers that take filters from untrusted input should require a + second field, as `tinymemory-tools` does. + +`forgotten` counts items actually removed. + +## explore + +`explore(req) -> ExplorePage`: counts of stored items per value of one facet, +for explorers (a UI tree, a CLI, an audit script). + +The **facet** is a metadata dimension fixed by the contract (`Kind`, +`Source`, `SourceId`, `Workspace`, `Folder`, `FilePath`, `Language`, `Repo`, +`Url`, `Thread`, `Agent`, `ToolCall`, `Tag`, `Namespace`), so one explorer +works on every engine. `Facet::values(kind, &meta)` gives an item's values +for a facet: none when the field is unset, several only for `Tag`. + +Semantics of the default (`explore_by_listing`), which every engine gets +unless it overrides `explore`: + +1. `req.validate()`: `limit` in `1..=500`, `scan_limit` in `1..=50 000` + (default 5 000). +2. Page through `list` with `req.filter`, 200 at a time, reading at most + `scan_limit` items. +3. For each item read: `total += 1`; if the facet has no value for it, + `missing += 1`; each value it has increments that value's count. A tagged + item counts once per tag, so for `Tag` bucket counts can sum to more than + `total`. +4. Sort buckets by count descending, ties by value ascending; cut to `limit`; + `more_buckets` is the number of distinct values cut. +5. `truncated` is `true` when the scan stopped at `scan_limit` with more + items remaining; counts are then a **lower bound**, and `total` is the + number read. + +An engine that aggregates server-side overrides `explore` and may ignore +`scan_limit`. + +### Facet::narrow: drilling down + +`facet.narrow(&mut filter, value)` turns a chosen bucket back into a filter +field, so drilling down is: `explore` → pick a bucket → `narrow` → `explore` +(another facet) or `list`. + +| Facet | Sets on the filter | +| --- | --- | +| `Kind` | `kinds = [value]` (must name an item kind) | +| `Source` | `sources = [value]` (must name a source kind) | +| `SourceId` | `source_id` | +| `Workspace`, `Language`, `Repo`, `Url`, `Agent`, `ToolCall` | `workspace`, `language`, `repo`, `url`, `agent_id`, `tool_call` | +| `Folder`, `FilePath` | `folder`, `file_path` (**prefix** match, so a folder also admits its subfolders) | +| `Thread` | `thread_id` | +| `Tag` | `tags_any = [value]` | +| `Namespace` | `reach = Reach::exact(value.parse()?)`: exactly that node | + +`narrow` **replaces** the one field it targets (a list field is replaced by a +one-element list; `Namespace` replaces any existing reach) and leaves others +alone. It fails with `InvalidRequest` for a blank value, an unknown kind or +source value, or a namespace that does not parse. + +Drill-down example: + +```text +explore(facet=source) → folder: 40, github: 7 +Source.narrow(filter, "folder") +explore(facet=folder, filter) → /notes: 31, /docs: 9 +Folder.narrow(filter, "/notes") +list(filter) → the 31 items under /notes (and subfolders) +``` + +Because `Folder` matches by prefix, a bucket count for `/notes` (items whose +`folder` is exactly that value) can be smaller than the number of items a +narrowed `list` returns, since subfolder items match the prefix too. + +## get + +`get(req) -> Vec`: read whole items by id. + +1. `req.validate()`: `1..=200` ids (`MAX_GET_IDS`), none blank. +2. Look each id up. The default pages through `list` (200 at a time, + confined to `req.reach` when set) until every id is found or the listing + ends; an engine that can look an id up directly overrides it. +3. Return hits **in the order the ids were named**, each at most once. An id + that names nothing is left out, with no error. So is an id whose item lies + outside `req.reach`: it is indistinguishable from a missing one. + +## Idempotency and fingerprints + +Storing an identical item twice must not create a second item. The contract +expresses "identical" as `StoreItem::fingerprint`: + +- **What is hashed**: SHA-256 over the item's JSON, keeping the first 20 bytes + as 40 hex characters. That JSON holds the whole item: kind, title, body, + mime, turns (with tool calls), learning kind, confidence, evidence, and all + of `meta` (including `namespace`, `tags` and `source`). +- **What is excluded**: `meta.observed_at`, set to `None` before hashing. It + says when the item was seen, which a host stamps on every store. + Including it would make a retried learning, or an unchanged file re-synced, + a new item every time. +- **Namespace is included**: the same text at two nodes is two items. + Root-namespace items serialise without a namespace, so their fingerprints + match those from before namespaces existed. +- **Replay**: an engine that finds the fingerprint already stored writes + nothing and returns `StoreReceipt { id, replayed: true }`, with the same id + as the first store. A retry after a timeout or a partial `store_many` is + therefore safe. +- **Changing anything else is a new item**: editing one tag, one character of + text, or the confidence produces a different fingerprint and a second + item; the old one is not replaced. + +Engines choose how the id relates to the fingerprint (`ItemId` is opaque); the +reference engine uses the fingerprint itself. + +## Cursors and paging + +`fetch` and `list` page with an opaque `cursor: Option`. + +- A first request has no cursor. A page that is not the last carries + `next_cursor: Some(token)`; the last page has `None`. +- Pass `next_cursor` unchanged as the next request's `cursor`, with the same + filter, query and mode. The format is the engine's business; do not parse or + construct one. An engine rejects a cursor it does not recognise with + `InvalidRequest`. +- `limit` is the page size and must be positive. A page may hold fewer items + than `limit`; only a missing `next_cursor` means the end. +- Cursors must make progress: a repeated cursor means paging never ends, and + the conformance suite fails an engine that does that. +- `explore` and `get` take no cursor: they page internally through `list`. + +## health + +`health() -> EngineHealth` (`Ok`, `Degraded(reason)`, `Down(reason)`) is +infallible and cheap to call. `Degraded` still serves; `Down` does not +(`is_serving()`). The method returns no `Result`, so an engine reports +trouble through the variant and its reason (which must not carry a +credential), never as an error. The conformance suite's first check is +that the engine reports itself serving. diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md new file mode 100644 index 00000000..e5e62f00 --- /dev/null +++ b/docs/architecture/overview.md @@ -0,0 +1,143 @@ +# Overview + +TinyMemory gives an agent host memory it can write to, search and answer +questions from, without binding the host to one storage engine. A host needs +three operations (see [the spec](../specs/memory-v2.md)): + +| Operation | Meaning | +| --- | --- | +| **Recall** | A question in, a synthesised answer with citations out. | +| **Fetch** | Raw keyword, vector or hybrid retrieval, filtered by metadata. | +| **Store** | Ingest a document, a conversation or a learning, each with typed metadata. | + +plus `list`, `forget`, `explore` and `get` for paging, removal and browsing. + +## The three parts + +The workspace has three crates under `crates/`, split by what each is allowed +to depend on. + +| Crate | Role | Why it is separate | +| --- | --- | --- | +| `tinymemory-api` | The **contract**: `MemoryEngine`, items, metadata, filters, namespaces, errors. With the `conformance` feature, the suite every engine must pass and an in-memory reference engine. | It performs no I/O and links no runtime, HTTP stack or storage engine, so an engine, a tool layer or a host can depend on it without inheriting anything else. | +| `tinymemory-tools` | The **agent surface**: `MemoryTools` (seven model-callable tools with JSON Schemas and host-fixed scoping) and the `context.md` compiler. | It works over any `MemoryEngine`, has no tool-runtime dependency, and is where "a model must never choose whose memory it touches" is enforced. | +| `tinymemory-integrations` | Everything that touches the **outside world**: the CortexDB engine, the engine registry and `MemoryConfig`, document conversion, source readers, safety scrubbing and the legacy v1 import. | Each integration is a Cargo feature, so a host pays only for the ones it uses. | + +## Dependency graph + +```mermaid +graph TD + api["tinymemory-api
(contract; feature: conformance)"] + tools["tinymemory-tools
(MemoryTools, context.md)"] + integ["tinymemory-integrations
(cortex, documents, sources,
safety, legacy-import)"] + host["host application"] + + tools --> api + integ --> api + host --> api + host --> tools + host --> integ + tools -. "dev: conformance" .-> api + integ -. "dev: conformance, tinymemory-tools" .-> api +``` + +`tinymemory-tools` and `tinymemory-integrations` do not depend on each other at +runtime; the only link is a dev-dependency (`integrations` compiles +`context.md` from a live server in one test). Both depend on `tinymemory-api` +alone, so the contract is the only coupling point. + +## Feature map + +| Crate | Feature | Enables | +| --- | --- | --- | +| `tinymemory-api` | `conformance` | `conformance::run`, `conformance::ReferenceEngine`. No extra dependency. | +| `tinymemory-tools` | (none) | `tools` and `context` are always built. | +| `tinymemory-integrations` | `cortex` (default) | `cortex::CortexEngine` (both wires), `registry` (`list_engines`, `build_engine`), `config` (`MemoryConfig`) | +| | `documents` | Format sniffing and conversion to markdown | +| | `documents-office` | PDF, DOCX, PPTX, XLSX conversion (implies `documents`) | +| | `sources` | Readers for folders, files, conversations; Composio normalisers (implies `documents`) | +| | `sources-network` | GitHub, RSS and web-page readers behind the SSRF guard (implies `sources`) | +| | `safety` | Secret and PII scrubbing of a `StoreItem` | +| | `legacy-import` | Reading a v1 workspace; `import::migrate` and `migrate_with` | +| | `full` | `cortex`, `documents-office`, `sources-network`, `safety`, `legacy-import` | + +## Write path + +A write is a pipeline; every stage but the last is optional and lives in +`tinymemory-integrations`. + +```text +source reader ──▶ documents ──▶ safety ──▶ engine.store ──▶ CortexDB + (sources) (conversion) (scrub) (contract) (cortex) +``` + +1. **Source reader** (`sources`): lists a configured source (folder, file, + link, GitHub, RSS, Composio payload, conversation) and reads each entry. + `sources::collect_items` does this for one source; one bad entry lands in + `Collected::skipped` instead of aborting the pass. +2. **Documents conversion** (`documents`): sniffs the format and converts the + bytes to markdown, producing a `StoreItem::Document` whose body is + `DocumentBody::Text` and whose `MemoryMeta` records where it came from + (`file_path`, `language`, `source`, ...). The contract refuses a + `DocumentBody::Uri`, so resolving a URI to text is a source's job. +3. **Safety scrub** (`safety`): `safety::scrub_item` redacts secrets and + PII in every text the item carries and returns a `Sanitized` + with a report. The host decides to run it; the engine does not. +4. **`engine.store`** (or `store_many` for batches): the engine calls + `StoreItem::validate`, derives the item's identity from + `StoreItem::fingerprint`, and answers with a `StoreReceipt { id, replayed }`. + See [operations.md](operations.md). +5. **CortexDB** (`cortex`): `CortexEngine` maps the item onto an experience + in the scope for its kind and namespace. See [cortex.md](cortex.md). + +An agent writing through `memory_store` skips steps 1 to 3: `MemoryTools` +builds a learning, document or conversation itself, stamps the host's +namespace and `observed_at`, and calls `engine.store`. + +The legacy import is the same pipeline with a different source: +`import::migrate` reads a v1 workspace and feeds `store_many` in batches of at +most `MAX_STORE_MANY`. + +## Read path + +```text +model tool call ──▶ MemoryTools ──▶ engine.recall / fetch / list / get / explore ──▶ render + (scope pins (contract) (compact JSON) + reach) +``` + +1. **Tool call**: the host forwards a model's call to + `MemoryTools::call(name, args)`. +2. **Scoping** (`tinymemory-tools`): arguments are read strictly (unknown + keys, `namespace` and `reach` are refused), the model's `filter` is + narrowed to a safe subset, and then `filter.reach` is **overwritten** with + the host's `ToolScope::reach`. See [tools.md](tools.md) and + [namespaces.md](namespaces.md). +3. **Engine call**: `recall`, `fetch`, `list`, `get` or `explore` on the + `MemoryEngine`. The engine validates the request, then applies the filter, + including its reach, to decide which items are visible. +4. **Render**: results become compact JSON (`{hits: [...]}`, + `{answer, citations}`, ...) with scores rounded and only a subset of + metadata; the namespace is never rendered. + +`context.md` takes the same engine calls (`recall` per brief, `list` of +learnings) from a host rather than a model, with `ContextSpec::reach` playing +the role of the scope. + +## Where each concern lives + +| Concern | Lives in | +| --- | --- | +| The operations, item model, metadata and filters | `tinymemory-api` (`engine`, `item`, `meta`, `query`, `explore`) | +| Whose memory an item is and who can read it | `tinymemory-api::namespace`; enforced by the engine on `filter.reach`, pinned for models by `tinymemory-tools` | +| Validation of requests | `validate` methods in `tinymemory-api`; every engine calls them first | +| Idempotency (replay) | `StoreItem::fingerprint` in `tinymemory-api`; each engine derives its ids from it | +| Error classification | `tinymemory_api::Error` | +| Does an engine obey the contract | `tinymemory_api::conformance` | +| What a model may do | `tinymemory-tools::tools` | +| Session briefing | `tinymemory-tools::context` | +| Choosing and building an engine | `tinymemory-integrations::{registry, config}` | +| The CortexDB engine | `tinymemory-integrations::cortex` | +| Turning files and feeds into items | `tinymemory-integrations::{documents, sources}` | +| Secrets and PII | `tinymemory-integrations::safety` | +| v1 migration | `tinymemory-integrations::import` | diff --git a/docs/architecture/testing.md b/docs/architecture/testing.md new file mode 100644 index 00000000..df4f0fc1 --- /dev/null +++ b/docs/architecture/testing.md @@ -0,0 +1,356 @@ +# Testing + +How TinyMemory is tested, what CI runs, and how to add tests for a new engine +or integration. The rules for where tests live are in `AGENTS.md`; this page +is the map. + +## The four contract commands + +CI runs exactly these, so a green local run should mean a green CI run. Run +them from the repository root. + +```sh +cargo fmt --all -- --check +cargo clippy --all-targets --all-features -- -D warnings +cargo build --all-targets --all-features +cargo test --all-features +``` + +Supporting commands: + +```sh +cargo test # default features only +cargo test -p tinymemory-integrations # a focused subset +cargo test --doc -p tinymemory-integrations --all-features # doctests alone +RUSTDOCFLAGS="-D warnings" cargo doc --no-deps --all-features +cargo run -p tinymemory-integrations --example basic +``` + +Never skip, ignore or delete a failing test to get a green run. Lints are one +workspace table (`[workspace.lints]`, opted into per crate): `unsafe_code` +is forbidden, `missing_docs` warns (and CI turns warnings into errors), and +clippy denies `unwrap`/`expect`/`panic` in library code. `clippy.toml` allows +them in tests. + +## Where tests live + +- **Unit tests** are never inline. They sit in a sibling `_tests.rs` + (`mod_tests.rs` beside `mod.rs`, `lib_tests.rs` beside `lib.rs`; a second + group is `__tests.rs`), declared at the bottom of the module: + + ```rust + #[cfg(test)] + #[path = "foo_tests.rs"] + mod tests; + ``` + + The file starts with a `//!` line and `use super::*;`, carries no + `#[cfg(test)]` of its own, and is a child module so it reaches private + items. +- **Test support** that is not a test lives in `_test_support.rs` (for + example `engine_test_support.rs`, which gives `CortexEngine` a + `with_test_timing` knob) or a `test_support/` directory. +- **Integration tests** are in the crate's `tests/` directory and use only the + public API. They are the regression suite for the contract. +- **Doctests** are compiled and run by `cargo test`, so examples cannot drift. + An example that needs a network is `no_run`. + +## CI + +`.github/workflows/ci.yml` runs on every push and pull request. Jobs: + +| Job | What it enforces | +| --- | --- | +| **Rust** | the four contract commands; `cargo test` with default features; `cargo run -p tinymemory-integrations --example basic`; and the **contract-crate dependency guard** | +| **Feature powerset and coverage** | `cargo hack --feature-powerset --depth 2 --workspace check --all-targets`; **coverage of at least 80% of lines** (`cargo llvm-cov --all-features --workspace --fail-under-lines 80`, ignoring `tests/`, `*_tests.rs`, `*_test_support.rs` and `test_support/`); and the **inline-test refusal** | +| **Test (feature matrix)** | `cargo test -p ` for each row below | +| **Docs** | `cargo doc --no-deps --all-features` with `RUSTDOCFLAGS=-D warnings` | +| **MSRV** | reads `rust-version` of `tinymemory-api` and builds `--all-targets --all-features` with that toolchain (1.96) | +| **Supply chain** | `cargo-deny check all` (advisories, licenses, bans, sources; `deny.toml`) | +| **CortexDB live** | `./scripts/cortexdb-live.sh` against a pinned real server (see [live tests](#live-tests)) | + +**Feature matrix.** Each row is its own `cargo test -p ...` so a feature +works on its own, not only in the union: + +| Package | Features | +| --- | --- | +| `tinymemory-integrations` | `--no-default-features` | +| | `--no-default-features --features cortex` | +| | `--no-default-features --features documents` | +| | `--no-default-features --features documents-office` | +| | `--no-default-features --features sources-network` (proves it implies `sources`) | +| | `--no-default-features --features safety` | +| | `--no-default-features --features legacy-import` | +| | `--no-default-features --features full` | +| `tinymemory-api` | `--no-default-features` | +| | `--features conformance` | +| `tinymemory-tools` | defaults | + +**Contract-crate dependency guard.** `tinymemory-api` is what engines and hosts +compile against, so it must stay free of storage engines, native libraries, +HTTP clients and async runtimes. The job runs `cargo tree -p tinymemory-api +-e normal,build --prefix none` and fails if it lists `rusqlite`, `libsqlite`, +`git2`, `reqwest`, `regex` or `tokio`. The forward form of `cargo tree` is +required: `cargo tree -i -p ...` discards the `-p` scope and looks +clean even when this crate is at fault. + +**Inline-test refusal.** Any `#[cfg(test)]` (or `#[cfg(any(test, ...))]`) in a +`.rs` file other than `*_tests.rs` and `*_test_support.rs` must guard a `mod` +or `use` declaration. Anything else is test code inline in a production file, +and the job fails with `inline test-only executable code must live in a +*_tests.rs file`. + +**Coverage.** Production-source line coverage across the workspace must be at +least 80%. Add tests with every behaviour change, and note a deliberately +untested edge case in the pull request. + +## Release + +`.github/workflows/release.yml` is a manual `workflow_dispatch` with a +`patch`, `minor` or `major` bump, and only runs on `main`. It re-runs +`cargo fmt --check`, clippy, `cargo test --all-features` and rustdoc. It then +reads the current version of `tinymemory-api` (every crate inherits +`[workspace.package] version`, so any one names it), computes the next +version, refuses an existing tag, and rewrites `version` in the +`[workspace.package]` table of the root `Cargo.toml` and refreshes +`Cargo.lock` (`cargo update --workspace`). Before the rewrite it fails if any +intra-workspace path dependency carries a `version = "..."` requirement, +since nothing is published and the one workspace version is enough. It +checks the bump took, commits `Release vX.Y.Z`, tags it, pushes both, and +creates a GitHub release. There are no binary artifacts, and nothing goes to +crates.io (`publish = false`). Do not hand-edit the workspace `version`. + +## The conformance suite + +`tinymemory_api::conformance` (feature `conformance` of `tinymemory-api`) +holds the behavioural suite every engine must pass: + +```rust +tinymemory_api::conformance::run(&engine).await?; +``` + +`run(&dyn MemoryEngine) -> conformance::Result<()>` writes only under a +workspace unique to the run (`tinymemory-conformance/`), filters by it, +and forgets it afterwards, so it can run against an engine that already holds +data. It stops at the first failed check. `Error::Check { check, detail }` +means the engine answered wrongly; `Error::Engine { check, source }` means it +failed a call it must serve. Cleanup runs even after a failure. + +The checks, in order (`suite/checks.rs`, `explore.rs`, `bulk.rs`, +`namespaces.rs`): + +| # | Check | What it proves | +| --- | --- | --- | +| 1 | `health` | the engine reports itself serving | +| 2 | `round_trip` | one item of each kind stores and lists back with the same kind, metadata and rendered text; paging terminates (the suite lists two at a time and fails on a repeated cursor or a non-zero listing score) | +| 3 | `replay` | storing an identical item again is a replay with the same id | +| 4 | `explore` | per-kind and per-workspace counts agree with `list`, buckets are largest first, and each bucket narrows to exactly its count | +| 5 | `get` | the run's items read back by id in the order asked, equal to their listing, with an unknown id left out | +| 6 | `store_many` | a batch stores in order, every item is listed on return, a repeat is all replays, an empty batch is refused | +| 7 | `fetch_filters` | for **every declared fetch mode**, a filter on each metadata field selects exactly the item carrying it (and `list` agrees) | +| 8 | `unsupported_modes` | every undeclared fetch mode fails `Unsupported` | +| 9 | `namespaces` | items at the root, two sibling agents and a sub-agent: each reach (own and inherited, exact, subtree) lists exactly its nodes, never a sibling's; `get` and `fetch` honour the reach; the same text in two namespaces is two items; the namespace facet counts each node; a forget scoped to one node removes only it | +| 10 | `empty_forget` | a forget with no ids or an empty filter is refused and removes nothing | +| 11 | `forget_by_id`, `forget_by_filter` | forgotten items stop listing and are counted; others stay | +| 12 | `recall` | an answer cites items that resolve through `list` | +| end | cleanup | forget by workspace filter; nothing survives | + +### ReferenceEngine + +`conformance::ReferenceEngine` (id `reference`) is the calibration subject: an +in-memory engine that is obvious by inspection, so a failure against it means +the assertion is wrong, not the engine. It serves every fetch mode (a trivial +keyword scorer and a deterministic 64-dimension toy vector), and answers +recall by quoting its best hybrid hits. It is also what the tools tests and +the import driver tests run against. `len()` and `is_empty()` let a test +assert that the suite cleaned up. + +`crates/tinymemory-api/tests/conformance_reference.rs` runs the suite against +the reference engine **and** against deliberately broken wrappers of it, each +fault expected to be caught by the check written for it. Without the second +half a suite that asserted nothing would also be green. + +## CortexDB engine tests + +Inside `crates/tinymemory-integrations/src/cortex/`: + +- **Unit tests** per module: `credential`, `descriptor`, `envelope` (and + `labels`), `error`, `transport` (`actor`, `failure`, body cap, retry), + and `engine` (`cursor`, `fetch`, `recall`, `scopes`, `store`, plus + `mod_tests`, `mod_list_tests`, `mod_direct_tests` and `mod_hosted_tests` + for each wire's behaviour through the doubles). +- **Conformance** (`conformance_tests.rs`): the shared suite against both + wires, `the_direct_wire_upholds_the_contract` and + `the_tinyhumans_wire_upholds_the_contract`. +- **The doubles** (`cortex/testing/`, compiled only under `cfg(test)`): + real HTTP servers on an ephemeral loopback port, built with axum. + +### The HTTP doubles + +`direct_double()` and `hosted_double()` start a double and return its base URL +and shared state; `direct_engine(url)` and `hosted_engine(url)` build an engine +with fast test timing (5ms polls and backoff, a 2s visibility budget); +`both()` gives one engine per wire. `sample_items()` is one item of each kind. + +Both doubles serve the same in-memory `CortexLog`, which is deliberately +**unaccommodating**, because a tidy double proves nothing. It reproduces every +behaviour in [the wire page](cortex-wire.md#cortexdb-behaviours-the-engine-is-shaped-around): + +- append-only, a body `idempotency_key` remembered for ever (same key, same + body is a replay; same key, different body is `409 IDEMPOTENCY_CONFLICT`; + forget does not release it); +- the listing is newest first, emits **every event twice**, counts the copies + in `limit`, ignores unknown query parameters, and pages by offset cursor; +- the forget selector reads only `memory_ids`; an empty selector without + `confirm_all` is refused, and one with `confirm_all` is refused as + ambiguous; +- recall renders text as `[role] {...}`, honours `view: "descend"`, metadata + label filters, and the events budget, and returns `pack_id: "pack_test"`; + the answer route requires that `use_pack_id`. + +The **hosted** double additionally wraps bodies in `{success,data}`, reports +failures with `errorCode`, refuses a scope outside the memory API's grammar +(`type:id` segments), takes an `Idempotency-Key` claim per write (any replay of +a claimed key is a 409, never forwarded), refuses a repeated `labels=` +parameter, and enforces the strict answer schema. + +**Knobs** make either fail the ways the real stacks fail, so tests aim at one +failure at a time: `fail_all` (every request), `accept_token` (the only token +accepted), `hide_listing_for` (empty listings), `rate_limit_events`, +`rate_limit_experience` (the backend's own 429 before any claim), +`apply_then_fail` (applied then 503), `claim_then_fail` (claimed, not applied, +502), `fail_nth_experience`, `rate_limit_forget`, `recall_down`, and +`arm_after_write` (rate limit or hide the reads a write makes *after* it is +sent). `Seen` records every request, the `Authorization` headers, the +idempotency pairs, and every recall, answer and forget body for assertions. + +## Tools tests + +`crates/tinymemory-tools/tests/`: + +- `tool_contracts.rs` freezes the **tool names and argument schemas a model + sees**. `fixtures/tool_contracts.json` is the serialised `specs()` of the + writable tools over the reference engine (every fetch mode). A schema + change tells every host's model something new, so it must be deliberate. To + regenerate after an intended change: + + ```sh + BLESS_TOOL_CONTRACTS=1 cargo test -p tinymemory-tools --test tool_contracts + ``` + + then review the diff. Without the variable the test fails on any mismatch + and prints the new specs. +- `tools_roundtrip.rs`: every tool round-trips through `MemoryTools::call` + against the reference engine, and read-only tools neither list nor run the + writes. +- `tools_scoping.rs`: the security invariants: the namespace and reach are the + host's, and a model cannot read, write or forget outside its scope. Its tree + is `team:acme/agent:a`, its sibling `agent:b`, their parent and the root. + +## Integrations tests + +`crates/tinymemory-integrations/tests/` (public API only): + +| File | Needs | Covers | +| --- | --- | --- | +| `feature_surface.rs` | `full` | the modules compose: scrub an item, store it in the reference engine, run the conformance suite, compile a context | +| `documents_office.rs` | `documents-office` | `OfficeConverter` prepends to the default `ConverterChain` and supports PDF, DOCX, XLSX, PPTX | +| `reader_dispatch.rs` | `sources` (`sources-network` for one test) | local readers are constructed for timers, network readers only for requests | +| `legacy_import.rs` | `legacy-import` | importing v1 workspaces, every mapping, ordering and resumption | +| `live_cortexdb.rs`, `office_live.rs` | `cortex` (+ `documents-office`) | a real CortexDB; skipped unless configured | + +### Legacy import fixtures + +`tests/support/mod.rs` builds v1 TinyCortex workspaces in temporary +directories (`tempfile`) using the **verbatim v1 DDL**: `MEMORY_DDL` (the +current `memory.db`), `OLD_MEMORY_DDL` (an early one without the `taint`, +`logical_namespace` and `tool_calls_json` columns and the later profile +columns), and `CHUNKS_DDL` (`memory_tree/chunks.db`, plus the migrated +`content_path` column). Helpers: `workspace(ddl)`, `doc`, `turn`, `facet`, +`chunk_store(root)` and `chunk` insert rows. `legacy_import.rs` builds a `rich()` +workspace exercising every section and edge case and imports it into the +reference engine. + +## Live tests + +The doubles check the wire cheaply; the live tests prove it against a real +CortexDB, pinned in `integration/cortexdb/` (see its README). They are **skipped +unless configured**, so a plain `cargo test` never needs Docker. + +| Variable | Used by | Meaning | +| --- | --- | --- | +| `TINYMEMORY_LIVE_CORTEXDB_URL` | `live_cortexdb.rs`, `office_live.rs` | base URL of a live CortexDB; unset skips the test | +| `TINYMEMORY_TEST_CORTEX_KEY` | both | the bearer; defaults to `tinymemory-cortex-test`, the harness's key | +| `CORTEXDB_VERSION` | `scripts/cortexdb-live.sh` | server release to boot (default `v0.10.4`; `v0.9.9` checks the older one) | +| `CORTEXDB_PORT` | the script and compose file | published port (script default 3142; compose default 3141) | +| `KEEP` | the script | leave the server running afterwards | + +```sh +./scripts/cortexdb-live.sh # boot, test, tear down +KEEP=1 ./scripts/cortexdb-live.sh # leave it running +TINYMEMORY_LIVE_CORTEXDB_URL=http://127.0.0.1:3141 \ + cargo test -p tinymemory-integrations --test live_cortexdb -- --nocapture +TINYMEMORY_LIVE_CORTEXDB_URL=http://127.0.0.1:3141 \ + cargo test -p tinymemory-integrations --features documents-office --test office_live -- --nocapture +``` + +- `live_cortexdb.rs`: (1) `the_live_server_upholds_the_contract` runs the + conformance suite; (2) `documents_conversations_and_learnings_round_trip_into_context` + stores a document, a conversation with a tool call and a learning, lists + them back (polling up to 60s, since CortexDB indexes asynchronously), checks + metadata filters narrow, `fetch` finds the document, `recall` answers with + citations, compiles `context.md` carrying the learning (within the token + budget), and forgets the run (3 items). +- `office_live.rs`: converts a generated DOCX with `OfficeConverter`, stores + it through the engine, and lists it back with the extracted text. + +The harness runs `mock-inference`, a deterministic OpenAI-compatible double for +CortexDB's embeddings, extraction and answer models, so it needs no +credential; it is a wiring fixture, not a quality benchmark. Live tests name +their file `live_*` (or `*_live`) so they are easy to exclude, and write only +under a workspace unique to the run. + +## Adding tests + +**A new engine** (any `MemoryEngine`): + +1. Run the conformance suite against it in a test. Against a local or + in-process engine, a test in the engine's `*_tests.rs` is enough: + + ```rust + #[tokio::test] + async fn the_engine_upholds_the_contract() { + tinymemory_api::conformance::run(&engine).await.unwrap(); + } + ``` + + Enable `tinymemory-api`'s `conformance` feature as a dev-dependency. For an + HTTP engine, run it against a loopback double, as `cortex/conformance_tests.rs` + does. +2. Reproduce the backend's quirks in the double rather than tidying them away, + and add a knob per failure mode. +3. Add unit tests per module and test each error variant's mapping (every new + `Error` source needs a test that produces it); cover the failure paths, not + just the happy one. +4. Gate any live or network test behind an environment variable, skip cleanly + when it is unset, and name it `live_*`. +5. Register the engine in the registry and add it to `list_engines` tests + (`registry/mod_tests.rs`). + +**A new integration module** in `tinymemory-integrations`: + +1. Put it behind a feature of (nearly) the same name in `Cargo.toml`, gate the + module in `lib.rs`, add the feature to `full`, and add a row to the CI + feature matrix so it is tested on its own. +2. Unit tests in `_tests.rs` beside each `mod.rs` (each starting with a + `//!` line), integration tests in `tests/` with `required-features`. +3. Keep dependencies optional and gated by the feature; do not add anything the + contract crate must not have (see the dependency guard). +4. Document it: `//!` on every `mod.rs`, rustdoc with `# Errors` and `# Panics` + on public items, and a module `README.md` if it is complex. + +**Tools**: when a change alters a tool's name or schema, regenerate and review +`tool_contracts.json` as above. + +**Doc changes**: run `RUSTDOCFLAGS="-D warnings" cargo doc --no-deps +--all-features` and `cargo test --doc -p --all-features`. diff --git a/docs/architecture/tools.md b/docs/architecture/tools.md new file mode 100644 index 00000000..38cae757 --- /dev/null +++ b/docs/architecture/tools.md @@ -0,0 +1,412 @@ +# Tools and context.md + +`tinymemory-tools` is the agent-facing side of TinyMemory. It works over any +[`MemoryEngine`](api.md) and has no tool-runtime dependency (no MCP, no +`tinytools`): it produces runtime-neutral tool specs and runs tool calls, and a +host adapts them to whatever its model runtime expects. It also holds the +`context.md` compiler. + +| Part | Module | Purpose | +| --- | --- | --- | +| Tools | `tinymemory_tools::tools` | `MemoryTools`, `ToolScope`, `ToolSpec`, the seven tool names | +| Context | `tinymemory_tools::context` | `compile`, `ContextCompiler`, `ContextSpec`, `Brief`, `ContextDoc` | + +Reference for the item level lives in rustdoc; the crate README +([`crates/tinymemory-tools/README.md`](../../crates/tinymemory-tools/README.md)) +is the quick tour. This page explains how it works and why. + +## The design constraint + +A model must never choose whose memory it touches. An earlier tool layer let +the model pass a namespace, which let one tenant's agent read another's memory. +Here the host fixes two things in a `ToolScope` when it builds the tools, and +neither is a tool argument: + +- where writes land (`place`), and +- how far reads reach (`reach`, see [namespaces.md](namespaces.md)). + +Everything below follows from that. + +## ToolSpec + +```rust,ignore +pub struct ToolSpec { + pub name: &'static str, // one of TOOL_NAMES + pub description: &'static str, // written for the model + pub parameters: serde_json::Value, // a JSON Schema object +} +``` + +`ToolSpec` derives `Serialize`. Every object in `parameters` sets +`additionalProperties: false`, and no schema has a `namespace` or `reach` +property. `MemoryTools::specs()` returns the specs to offer, reads first: + +- the write tools (`memory_store`, `memory_forget`) appear only when writes are + enabled; +- `memory_fetch` is left out entirely for an engine whose + `EngineDescriptor::fetch_modes` is empty, and its `mode` enum lists exactly + the modes the engine serves. + +`TOOL_NAMES` lists all seven in spec order and `WRITE_TOOL_NAMES` the two +write tools. The names are exported as constants (`MEMORY_RECALL`, +`MEMORY_FETCH`, `MEMORY_LIST`, `MEMORY_GET`, `MEMORY_EXPLORE`, `MEMORY_STORE`, +`MEMORY_FORGET`). + +## ToolScope + +```rust,ignore +pub struct ToolScope { + pub place: Namespace, // the node every stored item is written to + pub reach: Option, // the nodes reads and forgets are confined to + pub writes: bool, // whether store and forget are offered +} +``` + +| Constructor | Result | +| --- | --- | +| `ToolScope::default()` | `place` is the root, `reach` is `None` (every namespace), `writes` is `true`. A single-tenant host's scope. | +| `ToolScope::at(place)` | `place` set, `reach = Some(Reach::of(place))`, writes on. | + +`reach: None` reads every namespace, which suits only a host whose engine +serves a single tenant. + +`MemoryTools` is built over an `Arc` and a scope: + +| Builder | Effect | +| --- | --- | +| `MemoryTools::new(engine)` | `ToolScope::default()` | +| `MemoryTools::with_scope(engine, scope)` | an explicit scope | +| `.placed_at(place)` | writes land at `place`, **and the reach resets to `Reach::of(place)`**: `place` and its ancestors, never a sibling. Writes stay as they were. | +| `.reach(reach)` | replaces the reach; call it after `placed_at` to read differently | +| `.read_only()` | `writes = false` | +| `.scope()` | the current scope | + +Because `placed_at` resets the reach, a placed scope can never end up reading +more than its own branch by accident. The order matters: `.reach(r).placed_at(p)` +discards `r`. + +```rust,ignore +// One agent: writes to its node, reads it and its ancestors. +let agent = MemoryTools::new(engine.clone()).placed_at(Namespace::agent("writer")); + +// A team-wide, read-only auditor. +let team: Namespace = "team:acme".parse()?; +let auditor = MemoryTools::new(engine) + .placed_at(team.clone()) + .reach(Reach::subtree(team)) + .read_only(); +``` + +Build one `MemoryTools` per agent session from the session's identity, never +from anything the model said. `specs()` is cheap; call it per session so a +read-only or differently scoped agent sees only its own tools. + +## Dispatch: MemoryTools::call + +`call(name, args: serde_json::Value) -> Result` runs one tool. + +1. An unknown `name` is `Error::InvalidRequest` listing the valid names. +2. A write tool on read-only tools is `Error::Unsupported`. This is checked + before the arguments are read, so the call never reaches the engine. +3. The call is dispatched to the tool's implementation (reads in + `tools/read`, writes in `tools/write`), which parses the arguments strictly, + confines the request to the scope, calls the engine and renders the result. + +`null` arguments read as no arguments. Arguments that are not a JSON object are +`InvalidRequest`. An explicit `null` for an optional field reads as absent, since +many models send `null` for an argument they mean to leave out. + +### Argument validation + +Argument reading (`tools/args`) is deliberately strict. It refuses: + +- a `namespace` or `reach` key **anywhere** in the arguments, at any depth, + nested objects and arrays included. The message says the host fixes it, so + the model is told rather than quietly ignored. The check runs before the + unknown-key check, so a `namespace` under a key the tool would otherwise + reject is still reported as host-fixed; +- any other key the tool's schema does not list; +- a value of the wrong type or out of range. + +Messages are lowercase and name the tool and the field, for example +``memory_list: `filter.kinds` `memo` is not an item kind``, so they can be shown +to the model as the tool's error output. + +Limits: `limit` is `1..=50`, default `10`. `ids` lists `1..=200` non-blank ids +(`MAX_GET_IDS`), with duplicates dropped. Timestamps are RFC 3339. + +## The seven tools + +The exact names and schemas are frozen in +[`crates/tinymemory-tools/tests/fixtures/tool_contracts.json`](../../crates/tinymemory-tools/tests/fixtures/tool_contracts.json). +That file is the source of truth for parameter schemas; the summaries below +are for orientation. + +| Tool | Kind | Arguments (`?` optional) | Result | +| --- | --- | --- | --- | +| `memory_recall` | read | `question`, `filter?`, `limit?`, `instructions?` | `{answer, citations: [...]}` | +| `memory_fetch` | read | `query`, `mode?`, `filter?`, `limit?`, `cursor?` | `{hits: [...], next_cursor?}` | +| `memory_list` | read | `filter?`, `limit?`, `cursor?` | `{items: [...], next_cursor?}` | +| `memory_get` | read | `ids` | `{items: [...], missing: [ids]}` | +| `memory_explore` | read | `facet`, `filter?`, `limit?` | `{facet, buckets: [{value, count}], total, missing, more_buckets, truncated}` | +| `memory_store` | write | exactly one of `learning` / `document` / `conversation`, `tags?` | `{id, replayed}` | +| `memory_forget` | write | `ids` **or** a non-empty `filter` | `{forgotten, skipped: [ids]}` | + +A change to a name or schema changes what every host's model is told, so the +`tool_contracts` test fails until the fixture is regenerated on purpose: + +```sh +BLESS_TOOL_CONTRACTS=1 cargo test -p tinymemory-tools --test tool_contracts +``` + +### The model-facing filter + +`filter` is a subset of `MetaFilter`: `kinds`, `sources` (enum lists of item +and source kinds), `tags_any`, `workspace`, `folder`, `file_path`, `repo`, +`url`, `thread_id`, `agent_id`, `observed_after` and `observed_before`. It +never has a namespace or a reach; the tool sets `reach` itself from the scope, +overwriting anything else. + +### memory_recall + +Asks the engine a question and returns its synthesised answer with citations. +`question` is required and non-blank. `limit` bounds how many memories the +answer may cite; `instructions` steers length or format. A citation renders as +`{id, kind, snippet, score?, meta}`. + +### memory_fetch + +Raw retrieval. `query` is required. `mode` is one of the modes the engine +serves; when the model names none, `hybrid` is used if served, otherwise the +engine's first mode. A mode the engine does not serve is `InvalidRequest`; an +engine serving no mode gets `Error::Unsupported` (and no spec). Pass back +`next_cursor` as `cursor` for the next page. + +### memory_list + +Pages through stored memories with no query, optionally narrowed by a filter, +with `cursor` paging as above. + +### memory_get + +Reads memories whole by id. The request carries the scope's reach, so an id +outside it comes back in `missing`, as does an id that names nothing. A model +therefore cannot tell "does not exist" from "not yours". + +### memory_explore + +Counts memories per value of one `facet`, largest first. Facets: `kind`, +`source`, `source_id`, `workspace`, `folder`, `file_path`, `language`, `repo`, +`url`, `thread`, `agent`, `tool_call` and `tag`. The `namespace` facet is +deliberately absent. The result is the engine's explore page: `buckets` of +`{value, count}`, `total` (items the filter admitted), `missing` (those with no +value for the facet), `more_buckets` (values beyond `limit`) and `truncated` +(the engine's scan stopped early, so counts are a lower bound). + +### memory_store + +Stores exactly one memory; none or several of the three shapes is +`InvalidRequest`. + +- `learning: {text, learning_kind?, confidence?, evidence?}`: kind is one of + `preference`, `fact`, `procedure`, `correction`, `other` (default `fact`); + confidence is `0..=1` (default `0.8`). +- `document: {title?, text}`: stored as a text document. +- `conversation: {turns: [{role, text}]}`: at least one turn; roles are + `user`, `assistant`, `system`, `tool`. + +The item's metadata is built by the tool and never read from arguments: +namespace is the scope's `place`, source is `agent`, `tags` are the model's, +and `observed_at` is the time of the call. Storing the same memory twice is a +replay (`replayed: true`), not a duplicate. + +### memory_forget + +Exactly one of `ids` or `filter`, else `InvalidRequest`. + +- **By ids:** the ids are first read back with `get` under the scope's reach, + and only those found are forgotten. The rest are returned as `skipped` and + survive. If none were found, the engine is not called at all. +- **By filter:** the filter must set at least one field (a reach alone would + mean "everything in reach"); an empty filter is `InvalidRequest`. It is then + confined to the reach and forgotten as a filter target. `skipped` is empty. + +### Result shape + +Results carry what a model can act on and nothing else. A hit renders as +`{id, kind, text, score, confidence?, meta}`, where `meta` is a subset: +`source {kind, id?}`, `file_path`, `url`, `thread_id`, `tags`, `observed_at`. +Scores and confidences are rounded to four places so `f32` noise does not reach +the model. Absent optional fields are omitted rather than written as `null`. +The namespace is never rendered. + +## Errors + +`call` returns `tinymemory_api::Result`: + +| Error | When | +| --- | --- | +| `InvalidRequest` | unknown tool; arguments not an object; a host-fixed key; an unknown key; a missing, mistyped or out-of-range value; a bad store shape; a forget with neither or both selectors or an empty filter | +| `Unsupported` | a write tool on read-only tools; `memory_fetch` on an engine serving no fetch mode | +| anything else | whatever the engine returns for the request | + +`Unsupported` means "the call is well formed but these tools do not offer the +operation", which is what the variant means across the contract. + +## Scoping and security invariants + +1. **Namespace and reach are refused at any depth.** A `namespace` or `reach` + key anywhere in the arguments is `InvalidRequest`, never ignored. +2. **Stores are forced to `place`.** The tool builds the metadata; the model + supplies only content and tags. +3. **Reads are confined to the reach.** Every recall, fetch, list and explore + filter has its `reach` overwritten with the scope's; `get` passes it as + `GetRequest::reach`. +4. **Forget cannot reach out.** By id it goes through `get` under the reach + first; by filter the reach is applied and an empty filter is refused. +5. **Schemas are closed.** `additionalProperties: false` everywhere. +6. **Read-only means read-only.** The write tools are neither listed nor run. + +### The ancestor-forget caveat + +A reach built with `Reach::of(place)` (the default from `placed_at`) includes +the node's **ancestors**. Reads there are the point: an agent sees what its team +and the root share. But forget uses the same reach, so an agent may forget +memory it can see that lives at an ancestor, which includes memory shared with +its team or the root. A host that wants an agent's forgets confined to its own +node sets `Reach::exact(place)` (which also narrows reads to that node), or +offers read-only tools. There is no separate "forget reach". + +## Adapting ToolSpec to a tool runtime + +`ToolSpec` is three fields, and most runtimes want exactly those: + +| Runtime | Mapping | +| --- | --- | +| MCP | `name` to `name`, `description` to `description`, `parameters` to `inputSchema`. On `tools/call`, pass `arguments` to `MemoryTools::call` and return the result serialised as text content, or an error result with the error's message on `Err`. | +| OpenAI-style function calling | `{"type": "function", "function": {"name", "description", "parameters"}}`. The schemas are closed objects, so they suit strict mode where the runtime accepts optional properties. | +| Anthropic tool use | `name`, `description`, `input_schema`. | +| `tinytools` or a custom registry | register `(name, description, parameters)` and route calls to `call`. | + +```rust,ignore +for spec in tools.specs() { + runtime.register(spec.name, spec.description, spec.parameters); +} +// When the model calls a tool: +let output = match tools.call(&call.name, call.arguments).await { + Ok(value) => value.to_string(), + Err(error) => format!("error: {error}"), +}; +``` + +## context.md + +`context.md` is a token-budgeted brief a host injects at the start of a +session so the agent begins with what memory knows about its user. The +compiler works over any engine and is stateless. + +### ContextSpec + +| Field | Default | Meaning | +| --- | --- | --- | +| `budget_tokens` | `1_500` (`DEFAULT_BUDGET_TOKENS`) | the most tokens the whole document may take | +| `briefs` | `Brief::defaults()` | the sections, in order; each is answered by one recall | +| `learnings_limit` | `20` (`DEFAULT_LEARNINGS_LIMIT`) | the most learnings listed after the briefs | +| `reach` | `None` | whose memory the document is about; `None` reads every namespace | + +`ContextSpec` is serde-serialisable, so a host can keep it in its config. +`validate()` checks the spec: a zero budget, or a brief with a blank heading or +question, is `context::Error::InvalidSpec`. It is the only error `compile` can +return. + +A `Brief` is a `heading`, a `question` and a `filter` (a `MetaFilter` +restricting what the answer may draw on). `Brief::new(heading, question)` has no +filter. The four default briefs, in order: + +1. About the user: identity, role, how they like to work +2. Active work: current projects, workspaces and repositories +3. Preferences and standing instructions +4. Recent important events + +### Compilation + +`compile(&engine, &spec)` (or `ContextCompiler::new().compile(...)`; use +`ContextCompiler::at(timestamp)` for reproducible `generated_at`) does: + +1. **Validate** the spec. +2. **Gather briefs:** one recall per brief, with the brief's filter (and the + spec's `reach` overwriting the filter's reach when set), up to 8 citations, + and a fixed instruction to answer briefly as markdown bullets stating only + what the stored items support. A brief whose recall fails is logged and + skipped. A brief whose answer is blank or cites nothing is skipped too. +3. **Gather learnings:** list `Learning` items under the spec's reach, 100 per + page up to 50 pages, then rank newest first, then most confident first + (undated learnings after dated ones; ties keep the engine's order) and take + `learnings_limit`. A limit of `0` skips the listing. A failed listing leaves + the learnings out. +4. **Render** within the budget. + +Engine failures are never errors of `compile`: an engine that holds nothing, +or that fails everything, yields an empty document. + +### The document + +```markdown +--- +generated_at: 2026-10-04T09:30:00Z +engine: tinyhumans +tokens: 412 +refs: [id1, id2, id3] +--- + +# Context + +## About the user + +- ... + +## Learnings + +- The user prefers short answers +``` + +`ContextDoc` carries the `markdown`, its estimated `tokens`, `generated_at`, +the `engine` id and `refs`: every item the document cites, in order of first +citation. A document with no briefs and no learnings is the empty string, with +zero tokens and no frontmatter. Learning lines are collapsed to a single line. + +### Budget and trim rules + +Tokens are estimated at four characters each, rounded up (`estimate_tokens`). +The frontmatter counts toward the budget, and the `tokens` it reports is +exact: the compiler settles the number to a fixed point. When the document is +over budget, `render` trims in a fixed order until it fits: + +1. **Learnings go first**, one line at a time from the end (so the oldest or + least confident leave first). +2. **Then the last remaining brief shrinks.** Its body is cut back by the + overflow, to a word boundary when one is near, and marked with an ellipsis. + Once fewer than 40 characters of it would remain, the brief is dropped + instead of kept as a stub. +3. Earlier briefs keep their text longest and every surviving brief keeps its + place. If everything is trimmed away the result is the empty document, + which always fits. + +## Where things live + +```text +crates/tinymemory-tools/src/ +├── lib.rs # crate docs and re-exports +├── tools/ +│ ├── mod.rs # MemoryTools, ToolScope, dispatch +│ ├── spec/ # ToolSpec, tool names, JSON Schemas, limits +│ ├── args/ # strict argument reading and the model filter +│ ├── read/ # recall, fetch, list, get, explore +│ ├── write/ # store, forget +│ └── render/ # compact result JSON +└── context/ # ContextSpec, Brief, compile, render, Error +``` + +Tests: `tests/tools_roundtrip.rs` runs every tool against the reference +engine, `tests/tools_scoping.rs` pins the invariants above, and +`tests/tool_contracts.rs` guards the frozen schemas. See [testing.md](testing.md). diff --git a/docs/specs/memory-v2.md b/docs/specs/memory-v2.md index 33644dab..8db14007 100644 --- a/docs/specs/memory-v2.md +++ b/docs/specs/memory-v2.md @@ -17,32 +17,36 @@ things from memory, plus a way to choose who provides them: On top of the engine sits one engine-neutral product: **`context.md`**, a token-budgeted brief compiled from Recall and Fetch that a host injects at the -start of a session. +start of a session. A second, agent-facing surface, the memory **tools**, lets +a model call the same operations under a scope the host fixes (see +[Tools](#tools-tinymemory-tools)). ## Crates +Three crates, one directory each under `crates/`. Their relationships are in +[`docs/architecture/overview.md`](../architecture/overview.md). + | Crate | Owns | | --- | --- | -| `tinymemory-api` | The contract: `MemoryEngine`, request/response types, `MemoryMeta`, `MetaFilter`, `EngineDescriptor`, `Error`. No I/O. | -| `tinymemory-cortex` | The CortexDB engine, registered twice: `cortexdb` (direct `/v1/*`, endpoint + key) and `tinyhumans` (CortexDB behind the TinyHumans backend `/memory/*`, host bearer). | -| `tinymemory-documents` | Format sniffing and conversion to markdown (Markdown, plain text, HTML, code, PDF/DOCX via a host `DocumentConverter`). Emits `StoreItem::Document`. | -| `tinymemory-sources` | Readers that turn a source into `StoreItem`s: folder, file, link (web page), GitHub repo, RSS, Composio toolkit payloads. Includes the SSRF guard. | -| `tinymemory-safety` | Secret/PII scrubbing applied to every item before `store`. | -| `tinymemory-context` | `ContextCompiler`: builds `context.md` from an engine. | -| `tinymemory-import` | Reads a legacy (v1, embedded TinyCortex) workspace and yields `StoreItem`s. | -| `tinymemory-conformance` | Behavioural suite every engine must pass, plus a reference in-memory engine. | -| `tinymemory` | Facade: engine registry, `MemoryConfig`, `build_engine`, re-exports. One feature per optional crate. | - -Deleted: `tinymemory-bus`, `tinymemory-core`, `tinymemory-tinycortex`, -`tinymemory-remote` (CortexDB moves to `tinymemory-cortex`; mem0, supermemory, -cognee, agentmemory, livingbrain are dropped), `tinymemory-tools`, +| `tinymemory-api` | The contract: `MemoryEngine`, request/response types, `MemoryMeta`, `MetaFilter`, namespaces, `EngineDescriptor`, `Error`. No I/O. Feature `conformance`: the behavioural suite every engine must pass, plus a reference in-memory engine. | +| `tinymemory-tools` | The agent tool spec `MemoryTools` (seven tools, host-fixed namespace and reach) and `context`, the `ContextCompiler` that builds `context.md` from an engine. | +| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` and `sources-network` (readers for folder, file, link, GitHub, RSS, Composio payloads and conversations, with the SSRF guard), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | + +Deleted: the earlier `tinymemory` facade, `tinymemory-cortex`, +`tinymemory-documents`, `tinymemory-sources`, `tinymemory-safety`, +`tinymemory-context`, `tinymemory-import` and `tinymemory-conformance` as +separate crates (their code moved into the three above: the engine, registry +and config, documents, sources, safety and import into +`tinymemory-integrations`; context into `tinymemory-tools`; conformance into +the `conformance` feature of `tinymemory-api`). Earlier still: `tinymemory-bus`, +`tinymemory-core`, `tinymemory-tinycortex`, `tinymemory-remote` (CortexDB is +the engine; mem0, supermemory, cognee, agentmemory, livingbrain are dropped), `tinymemory-conversations`, `tinymemory-guard`, `tinymemory-gate`, -`tinymemory-sync` (normalisers move into `tinymemory-sources`), -`tinymemory-module`, `tinymemory-testing-ui`, and the `vendor/tinycortex`, -`vendor/tinybus` and `vendor/tinyinference` submodules (`tinymemory-import` -reads the v1 on-disk layout directly, so it needs no engine dependency). -`tinymemory-conversations` (the chat thread store) moves to -`tinyagents-session::threads` in tinyagents. +`tinymemory-sync` (normalisers moved into `sources`), `tinymemory-module`, +`tinymemory-testing-ui`, and the `vendor/tinycortex`, `vendor/tinybus` and +`vendor/tinyinference` submodules (`legacy-import` reads the v1 on-disk layout +directly, so it needs no engine dependency). `tinymemory-conversations` (the +chat thread store) moved to `tinyagents-session::threads` in tinyagents. ## Contract (`tinymemory-api`) @@ -158,11 +162,11 @@ contract rather than by an engine's storage layout, so one explorer works on every engine: ```rust -pub enum Facet { Kind, Source, SourceId, Workspace, Folder, FilePath, Language, Repo, Url, Thread, Agent, ToolCall, Tag } +pub enum Facet { Kind, Source, SourceId, Workspace, Folder, FilePath, Language, Repo, Url, Thread, Agent, ToolCall, Tag, Namespace } pub struct ExploreRequest { pub facet: Facet, pub filter: MetaFilter, pub limit: usize /* 1..=500 buckets */, pub scan_limit: usize /* 1..=50_000, default 5_000 */ } pub struct FacetBucket { pub value: String, pub count: u64 } pub struct ExplorePage { pub facet: Facet, pub buckets: Vec, pub total: u64, pub missing: u64, pub more_buckets: u64, pub truncated: bool } -pub struct GetRequest { pub ids: Vec /* 1..=200 */ } +pub struct GetRequest { pub ids: Vec /* 1..=200 */, pub reach: Option } ``` - `explore` groups the items `filter` admits by one facet: buckets largest @@ -212,7 +216,10 @@ There is one `Error` enum: `Unsupported`, `InvalidRequest`, `Unauthorized`, `NotFound`, `Conflict`, `Unavailable` (transient), `Engine` (the engine's own failure, already sanitised), and `Config`. Messages never carry credentials. -## Facade (`tinymemory`) +## Registry and config (`tinymemory-integrations`, feature `cortex`) + +There is no facade crate: a host depends on `tinymemory-api` and the +integrations it wants, and builds an engine with the registry. ```rust pub struct MemoryConfig { pub engine: String, pub engines: BTreeMap } @@ -220,12 +227,13 @@ pub struct EngineSettings { pub endpoint: Option } pub enum EngineCredential { None, Static(String), Dynamic(Arc) } pub fn list_engines() -> Vec; pub fn build_engine(id: &str, settings: &EngineSettings, credential: EngineCredential) -> Result>; +impl MemoryConfig { pub fn build(&self, credential: EngineCredential) -> Result>; } ``` `build_engine` refuses an unknown id, a missing required endpoint or key, and a -credentialed cleartext non-loopback endpoint. +credentialed cleartext non-loopback endpoint, all as `Error::Config`. -## Engine: CortexDB (`tinymemory-cortex`) +## Engine: CortexDB (`tinymemory-integrations`, module `cortex`) - **Wires.** `Direct` (`v1/experience`, `v1/events`, `v1/recall`, `v1/forget`, `v1/answer`) and `TinyHumans` (`memory/*` with `{success,data}` envelopes), as in the v1 adapter. - **Store.** @@ -239,7 +247,41 @@ credentialed cleartext non-loopback endpoint. - **Recall.** Pack, then answer, as in v1. One scope: one pack over it. An unscoped read over several scopes: one pack over `app:tinymemory` with `view: "descend"`. A reach over several scopes: one pack per scope, built concurrently, and the answer route is asked once with the pack holding the most admitted events. Citations come from the packs' `layers.events`, decoded back to `Hit`s, the most specific node's first. - **List / forget.** These use `v1/events` paging and `v1/forget` by `memory_ids`. `ForgetTarget::Filter` lists first, then forgets ids, and never sends an empty selector. -## Context (`tinymemory-context`) +## Tools (`tinymemory-tools`) + +`MemoryTools` exposes seven tools over any `MemoryEngine`, with no tool-runtime +dependency. `MemoryTools::specs()` returns `Vec` (`name`, +`description`, `parameters` as a JSON Schema); `MemoryTools::call(name, args)` +runs one by name with the model's JSON arguments and returns compact JSON. + +| Tool | Kind | Maps to | +| --- | --- | --- | +| `memory_recall` | read | `recall` | +| `memory_fetch` | read | `fetch` (the `mode` enum lists exactly the engine's `fetch_modes`; absent when the engine serves none) | +| `memory_list` | read | `list` | +| `memory_get` | read | `get` (reports unknown or out-of-reach ids as `missing`) | +| `memory_explore` | read | `explore` (every facet except `namespace`) | +| `memory_store` | write | `store` of one learning, document or conversation | +| `memory_forget` | write | `forget` by ids or a non-empty filter (reports `skipped` ids) | + +**The model never chooses whose memory it touches.** The host fixes a +`ToolScope { place: Namespace, reach: Option, writes: bool }` +(`MemoryTools::new`, `placed_at`, `reach`, `read_only`, `with_scope`): + +- writes land at `place`: the tool builds the metadata itself (namespace, + source `agent`, the model's tags, `observed_at` now); +- every read filter's `reach` is overwritten with the scope's, and `memory_get` + passes it as `GetRequest::reach`; +- forget by ids first reads the ids back under the reach and forgets only + those found; by filter, the filter must set a field besides the reach; +- a `namespace` or `reach` key anywhere in the arguments, or any unknown key, + is `Error::InvalidRequest`; write tools on read-only tools are + `Error::Unsupported`. + +Names and schemas are frozen by a fixture test. See +[`docs/architecture/tools.md`](../architecture/tools.md). + +## Context (`tinymemory-tools`, module `context`) ```rust pub struct ContextSpec { pub budget_tokens: usize, pub briefs: Vec, pub learnings_limit: usize } @@ -261,7 +303,7 @@ Output rules: - Frontmatter records `generated_at`, `engine`, `tokens` and `refs`. - An engine with nothing stored yields an empty document (`markdown` is empty), not an error. A brief that fails is skipped and logged; it does not fail the document. -## Import (`tinymemory-import`) +## Import (`tinymemory-integrations`, feature `legacy-import`) `LegacyWorkspace::open(path)` detects a v1 TinyCortex store. `items()` yields `StoreItem`s: @@ -271,11 +313,21 @@ Output rules: - Profile facets become `Learning(Preference)`. Every item gets `source.kind = Import`. A `Checkpoint` (last yielded cursor per -section, persisted by the host) makes import resumable. Behind `legacy-import`. +section, persisted by the host) makes import resumable. + +`import::migrate(engine, workspace, from)` copies a workspace into any engine: +it streams `items_from(checkpoint)` in batches of at most `MAX_STORE_MANY`, +hands each batch to `store_many`, and returns a `MigrationReport { stored, +replayed, batches, checkpoint }`. `migrate_with` also calls an `on_batch` +callback with the committed checkpoint after every stored batch, so the host +can persist it. A failed `store_many` is `Error::Engine`, carrying the last +committed checkpoint; resuming re-sends the failed batch, whose stored items +come back as replays. A legacy read failure is returned as is, and resuming +from any earlier checkpoint only replays. ## Testing -`tinymemory-conformance::run(engine)` covers: +`tinymemory_api::conformance::run(engine)` (feature `conformance`) covers: - store/list round-trip for each kind; - replay idempotency; - `explore` counts agreeing with `list` per kind and per workspace, and each diff --git a/integration/cortexdb/README.md b/integration/cortexdb/README.md index 18eecdd6..38c7eeca 100644 --- a/integration/cortexdb/README.md +++ b/integration/cortexdb/README.md @@ -1,9 +1,9 @@ # Live CortexDB harness A real CortexDB server for the `cortexdb` engine's live tests -(`crates/tinymemory-cortex/tests/live_cortexdb.rs`), so the wire is proven +(`crates/tinymemory-integrations/tests/live_cortexdb.rs`), so the wire is proven against the server and not only against the HTTP doubles in -`crates/tinymemory-cortex/src/testing/`. +`crates/tinymemory-integrations/src/cortex/testing/`. ```sh ./scripts/cortexdb-live.sh # boot, test, tear down @@ -16,7 +16,7 @@ Or by hand: ```sh docker compose -f integration/cortexdb/docker-compose.yml up -d --build TINYMEMORY_LIVE_CORTEXDB_URL=http://127.0.0.1:3141 \ - cargo test -p tinymemory-cortex --test live_cortexdb + cargo test -p tinymemory-integrations --test live_cortexdb docker compose -f integration/cortexdb/docker-compose.yml down --volumes ``` @@ -42,12 +42,12 @@ server you started by hand (which uses the compose file's own project on ## What the tests prove -- The shared conformance suite (`tinymemory_conformance::run`) passes against +- The shared conformance suite (`tinymemory_api::conformance::run`) passes against the live server. - A document, a conversation with a tool call and a learning, each with its metadata, store and list back; metadata filters narrow; hybrid `fetch` finds the document; `recall` (CortexDB's answer route) answers with citations; - `tinymemory_context::compile` builds a `context.md` that carries the + `tinymemory_tools::context::compile` builds a `context.md` that carries the learning; and a filter `forget` removes all three. - With `documents-office`, a generated DOCX is converted by `OfficeConverter`, stored through the CortexDB engine, and listed back with its extracted text. diff --git a/scripts/cortexdb-live.sh b/scripts/cortexdb-live.sh index 456c9289..82269f5d 100755 --- a/scripts/cortexdb-live.sh +++ b/scripts/cortexdb-live.sh @@ -49,5 +49,5 @@ curl --fail --silent "$url/v1/admin/ready" >/dev/null || { } echo "CortexDB $(curl --silent "$url/v1/admin/health") at $url" -TINYMEMORY_LIVE_CORTEXDB_URL="$url" cargo test -p tinymemory-cortex --test live_cortexdb -- --nocapture -TINYMEMORY_LIVE_CORTEXDB_URL="$url" cargo test -p tinymemory --features documents-office --test office_live -- --nocapture +TINYMEMORY_LIVE_CORTEXDB_URL="$url" cargo test -p tinymemory-integrations --test live_cortexdb -- --nocapture +TINYMEMORY_LIVE_CORTEXDB_URL="$url" cargo test -p tinymemory-integrations --features documents-office --test office_live -- --nocapture