From 215235e8a6f61a4ea874d7da84d524f33157044d Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:33:35 +0300 Subject: [PATCH 01/24] chore: files changed integration/cortexdb/docker-compose.yml Auto-committed-on: dragonfly Co-authored-by: Medulla --- integration/cortexdb/docker-compose.yml | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/integration/cortexdb/docker-compose.yml b/integration/cortexdb/docker-compose.yml index 42ca3c2d..e90ad3ba 100644 --- a/integration/cortexdb/docker-compose.yml +++ b/integration/cortexdb/docker-compose.yml @@ -26,20 +26,23 @@ services: CORTEX_ANSWER_URL: ${CORTEX_INFERENCE_URL:-http://mock-inference:8080/v1} CORTEX_ANSWER_API_KEY: ${CORTEX_INFERENCE_KEY:-tinymemory-test} CORTEX_ANSWER_MODEL: ${CORTEX_ANSWER_MODEL:-reasoning} - CORTEX_VERIFIER_URL: ${CORTEX_INFERENCE_URL:-http://mock-inference:8080/v1} + # A flag profile turns the verifier off with an empty CORTEX_VERIFIER_URL. + CORTEX_VERIFIER_URL: ${CORTEX_VERIFIER_URL-${CORTEX_INFERENCE_URL:-http://mock-inference:8080/v1}} CORTEX_VERIFIER_API_KEY: ${CORTEX_INFERENCE_KEY:-tinymemory-test} CORTEX_VERIFIER_MODEL: ${CORTEX_VERIFIER_MODEL:-max-reasoning} CORTEX_VERIFIER_MAX_TOKENS: "16384" - CORTEX_ENTITY_GRAPH: "1" - CORTEX_V1_LAYERS_AUTO: "1" - CORTEX_AUTO_ROUTE: "1" - CORTEX_CONSOLIDATION_MIN_AGE_HOURS: "0" + # The server's feature flags: the baseline, then the profile under test + # (scripts/memory-flag-sweep.sh), which only names what it changes. See + # flags/README.md. A variable in `environment` above wins over both. + env_file: + - flags/baseline.env + - ${CORTEX_FLAGS_FILE:-flags/baseline.env} ports: ["127.0.0.1:${CORTEXDB_PORT:-3141}:3141"] extra_hosts: - host.docker.internal:host-gateway volumes: - cortex-data:/data - - ./cortex.toml:/data/cortex.toml:ro + - ./${CORTEX_TOML:-cortex.toml}:/data/cortex.toml:ro depends_on: mock-inference: condition: service_healthy From bc30478050abf77858e074fedfcec7852469c7a9 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:34:04 +0300 Subject: [PATCH 02/24] chore: files changed integration/cortexdb/flags/baseline.env,integration/cortexdb/flags/bitemporal-e Auto-committed-on: dragonfly Co-authored-by: Medulla --- integration/cortexdb/flags/baseline.env | 10 ++++++++++ integration/cortexdb/flags/bitemporal-enforce.env | 5 +++++ integration/cortexdb/flags/bitemporal-off.env | 4 ++++ integration/cortexdb/flags/cost-optimized.env | 11 +++++++++++ integration/cortexdb/flags/enrich-batched.env | 5 +++++ integration/cortexdb/flags/layers-incremental.env | 4 ++++ integration/cortexdb/flags/max-recall.env | 9 +++++++++ integration/cortexdb/flags/no-auto-route.env | 4 ++++ integration/cortexdb/flags/no-graph.env | 5 +++++ integration/cortexdb/flags/no-hyde-multihop.env | 6 ++++++ integration/cortexdb/flags/no-learning-loop.env | 5 +++++ integration/cortexdb/flags/no-polarity-recheck.env | 5 +++++ integration/cortexdb/flags/no-verifier.env | 4 ++++ integration/cortexdb/flags/rerank-cohere.env | 6 ++++++ integration/cortexdb/flags/salience-high.env | 5 +++++ integration/cortexdb/flags/surprise-loose.env | 4 ++++ integration/cortexdb/flags/surprise-strict.env | 5 +++++ 17 files changed, 97 insertions(+) create mode 100644 integration/cortexdb/flags/baseline.env create mode 100644 integration/cortexdb/flags/bitemporal-enforce.env create mode 100644 integration/cortexdb/flags/bitemporal-off.env create mode 100644 integration/cortexdb/flags/cost-optimized.env create mode 100644 integration/cortexdb/flags/enrich-batched.env create mode 100644 integration/cortexdb/flags/layers-incremental.env create mode 100644 integration/cortexdb/flags/max-recall.env create mode 100644 integration/cortexdb/flags/no-auto-route.env create mode 100644 integration/cortexdb/flags/no-graph.env create mode 100644 integration/cortexdb/flags/no-hyde-multihop.env create mode 100644 integration/cortexdb/flags/no-learning-loop.env create mode 100644 integration/cortexdb/flags/no-polarity-recheck.env create mode 100644 integration/cortexdb/flags/no-verifier.env create mode 100644 integration/cortexdb/flags/rerank-cohere.env create mode 100644 integration/cortexdb/flags/salience-high.env create mode 100644 integration/cortexdb/flags/surprise-loose.env create mode 100644 integration/cortexdb/flags/surprise-strict.env diff --git a/integration/cortexdb/flags/baseline.env b/integration/cortexdb/flags/baseline.env new file mode 100644 index 00000000..f0dbd002 --- /dev/null +++ b/integration/cortexdb/flags/baseline.env @@ -0,0 +1,10 @@ +# The harness's standing configuration: graph-fused retrieval on, belief +# layers auto-built, per-question salience routing, and consolidation with +# no minimum age (the eval's events are minutes old). Every other profile +# is applied on top of this file and names only what it changes. +# targets: all +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_ENTITY_GRAPH=1 +CORTEX_V1_LAYERS_AUTO=1 +CORTEX_AUTO_ROUTE=1 +CORTEX_CONSOLIDATION_MIN_AGE_HOURS=0 diff --git a/integration/cortexdb/flags/bitemporal-enforce.env b/integration/cortexdb/flags/bitemporal-enforce.env new file mode 100644 index 00000000..7a07780b --- /dev/null +++ b/integration/cortexdb/flags/bitemporal-enforce.env @@ -0,0 +1,5 @@ +# Dates stated in the source set validity, and contradictions close or +# conflict with the earlier claim (default: shadow, detect only). +# targets: conflicts +# docs: https://cortexdb.ai/docs/api-reference/claims-conflicts +CORTEX_BITEMPORAL_MODE=enforce diff --git a/integration/cortexdb/flags/bitemporal-off.env b/integration/cortexdb/flags/bitemporal-off.env new file mode 100644 index 00000000..ac035d8c --- /dev/null +++ b/integration/cortexdb/flags/bitemporal-off.env @@ -0,0 +1,4 @@ +# No bi-temporal conflict detection at all. +# targets: conflicts +# docs: https://cortexdb.ai/docs/api-reference/claims-conflicts +CORTEX_BITEMPORAL_MODE=off diff --git a/integration/cortexdb/flags/cost-optimized.env b/integration/cortexdb/flags/cost-optimized.env new file mode 100644 index 00000000..5d24a4d3 --- /dev/null +++ b/integration/cortexdb/flags/cost-optimized.env @@ -0,0 +1,11 @@ +# Composite, after the docs' Cost-Optimized profile: every optional model +# call off (graph, HyDE, multihop, polarity recheck, verifier) and enrichment +# batched per scope. +# targets: all +# docs: https://cortexdb.ai/docs/operations/profiles +CORTEX_ENTITY_GRAPH=0 +CORTEX_HYDE_PASSAGES_MS=0 +CORTEX_MULTIHOP_QUERY_PLANNER_DISABLE=1 +CORTEX_POLARITY_RECHECK=0 +CORTEX_VERIFIER_URL= +CORTEX_ENRICHMENT_SCOPE_QUIET_MS=5000 diff --git a/integration/cortexdb/flags/enrich-batched.env b/integration/cortexdb/flags/enrich-batched.env new file mode 100644 index 00000000..df6ee475 --- /dev/null +++ b/integration/cortexdb/flags/enrich-batched.env @@ -0,0 +1,5 @@ +# Hold a busy scope's captures for 5 s so they are extracted together, in +# fewer model calls (default 0: each capture on its own). +# targets: cost, learning +# docs: https://cortexdb.ai/docs/operations/configuration +CORTEX_ENRICHMENT_SCOPE_QUIET_MS=5000 diff --git a/integration/cortexdb/flags/layers-incremental.env b/integration/cortexdb/flags/layers-incremental.env new file mode 100644 index 00000000..0fc03490 --- /dev/null +++ b/integration/cortexdb/flags/layers-incremental.env @@ -0,0 +1,4 @@ +# Rebuild and reconcile beliefs incrementally instead of from scratch. +# targets: learning, cost +# docs: https://cortexdb.ai/docs/self-hosting/upgrading +CORTEX_V1_LAYERS_INCREMENTAL=1 diff --git a/integration/cortexdb/flags/max-recall.env b/integration/cortexdb/flags/max-recall.env new file mode 100644 index 00000000..db4663ae --- /dev/null +++ b/integration/cortexdb/flags/max-recall.env @@ -0,0 +1,9 @@ +# Composite, after the docs' Max-Recall profile (without its reranker, which +# rerank-cohere covers): more HyDE passages, entity-vector seeding, assistant +# excerpts in the pack, and enforced bi-temporal validity. +# targets: all +# docs: https://cortexdb.ai/docs/operations/profiles +CORTEX_HYDE_PASSAGES_MS=3 +CORTEX_ENTITY_VECTOR_SEED_ENABLE=1 +CORTEX_CONTEXT_ASSISTANT_EXCERPTS=1 +CORTEX_BITEMPORAL_MODE=enforce diff --git a/integration/cortexdb/flags/no-auto-route.env b/integration/cortexdb/flags/no-auto-route.env new file mode 100644 index 00000000..1c8dda47 --- /dev/null +++ b/integration/cortexdb/flags/no-auto-route.env @@ -0,0 +1,4 @@ +# One salience weighting for every question, not one per question type. +# targets: accuracy +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_AUTO_ROUTE=0 diff --git a/integration/cortexdb/flags/no-graph.env b/integration/cortexdb/flags/no-graph.env new file mode 100644 index 00000000..34b1672b --- /dev/null +++ b/integration/cortexdb/flags/no-graph.env @@ -0,0 +1,5 @@ +# Graph-fused retrieval off. The docs price the graph at about 3x the write +# cost and 4x the recall cost, for multi-session recall. +# targets: accuracy, cost +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_ENTITY_GRAPH=0 diff --git a/integration/cortexdb/flags/no-hyde-multihop.env b/integration/cortexdb/flags/no-hyde-multihop.env new file mode 100644 index 00000000..7918a668 --- /dev/null +++ b/integration/cortexdb/flags/no-hyde-multihop.env @@ -0,0 +1,6 @@ +# No hypothetical-document passages and no multihop query planner: recall +# searches with the question alone. +# targets: accuracy, cost +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_HYDE_PASSAGES_MS=0 +CORTEX_MULTIHOP_QUERY_PLANNER_DISABLE=1 diff --git a/integration/cortexdb/flags/no-learning-loop.env b/integration/cortexdb/flags/no-learning-loop.env new file mode 100644 index 00000000..cd5e377c --- /dev/null +++ b/integration/cortexdb/flags/no-learning-loop.env @@ -0,0 +1,5 @@ +# Background learning off: the feedback-weight and cognitive-persist jobs, +# which fold usage back into the ranker, never run (cortex.no-learning.toml). +# targets: learning +# docs: https://cortexdb.ai/docs/operations/storage-cluster +CORTEX_TOML=cortex.no-learning.toml diff --git a/integration/cortexdb/flags/no-polarity-recheck.env b/integration/cortexdb/flags/no-polarity-recheck.env new file mode 100644 index 00000000..42376228 --- /dev/null +++ b/integration/cortexdb/flags/no-polarity-recheck.env @@ -0,0 +1,5 @@ +# Skip the extra model call that re-checks a fact's polarity (affirmed or +# negated) before it lands. +# targets: conflicts, cost +# docs: https://cortexdb.ai/docs/operations/configuration +CORTEX_POLARITY_RECHECK=0 diff --git a/integration/cortexdb/flags/no-verifier.env b/integration/cortexdb/flags/no-verifier.env new file mode 100644 index 00000000..994b165d --- /dev/null +++ b/integration/cortexdb/flags/no-verifier.env @@ -0,0 +1,4 @@ +# No verifier pass over generated answers. +# targets: cost, accuracy +# docs: https://cortexdb.ai/docs/operations/llm-answer +CORTEX_VERIFIER_URL= diff --git a/integration/cortexdb/flags/rerank-cohere.env b/integration/cortexdb/flags/rerank-cohere.env new file mode 100644 index 00000000..12ba1f15 --- /dev/null +++ b/integration/cortexdb/flags/rerank-cohere.env @@ -0,0 +1,6 @@ +# A Cohere cross-encoder reranks the fused candidates. +# targets: accuracy, cost +# requires: COHERE_API_KEY +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_RERANKER_PROVIDER=cohere +CORTEX_RERANKER_MODEL=rerank-v3.5 diff --git a/integration/cortexdb/flags/salience-high.env b/integration/cortexdb/flags/salience-high.env new file mode 100644 index 00000000..407361e0 --- /dev/null +++ b/integration/cortexdb/flags/salience-high.env @@ -0,0 +1,5 @@ +# Triple the weight salience carries in the final ranking (default 0.10), +# which should lift rare, striking events over routine ones. +# targets: surprise, accuracy +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_SALIENCE_WEIGHT=0.3 diff --git a/integration/cortexdb/flags/surprise-loose.env b/integration/cortexdb/flags/surprise-loose.env new file mode 100644 index 00000000..2d087a79 --- /dev/null +++ b/integration/cortexdb/flags/surprise-loose.env @@ -0,0 +1,4 @@ +# Consolidate nearly everything, surprising or not (default 0.5). +# targets: surprise, learning +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_CONSOLIDATION_MAX_SURPRISE=0.9 diff --git a/integration/cortexdb/flags/surprise-strict.env b/integration/cortexdb/flags/surprise-strict.env new file mode 100644 index 00000000..3dc13f2d --- /dev/null +++ b/integration/cortexdb/flags/surprise-strict.env @@ -0,0 +1,5 @@ +# Consolidate only unsurprising memories (default 0.5): a surprising event +# stays a standalone memory instead of being folded into its entity. +# targets: surprise, learning +# docs: https://cortexdb.ai/docs/operations/recall-tuning +CORTEX_CONSOLIDATION_MAX_SURPRISE=0.2 From 774718deb96c4c47e786da3bf096c0bdee89ba0f Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:34:22 +0300 Subject: [PATCH 03/24] chore: files changed integration/cortexdb/docker-compose.yml,integration/cortexdb/cortex.no-learning Auto-committed-on: dragonfly Co-authored-by: Medulla --- integration/cortexdb/cortex.no-learning.toml | 32 ++++++++++++++++++++ integration/cortexdb/docker-compose.yml | 2 ++ 2 files changed, 34 insertions(+) create mode 100644 integration/cortexdb/cortex.no-learning.toml diff --git a/integration/cortexdb/cortex.no-learning.toml b/integration/cortexdb/cortex.no-learning.toml new file mode 100644 index 00000000..d84e7b89 --- /dev/null +++ b/integration/cortexdb/cortex.no-learning.toml @@ -0,0 +1,32 @@ +# cortex.toml with background learning off, for the no-learning-loop flag +# profile: the feedback-weight and cognitive-persist jobs (which fold usage +# back into the ranker) are pushed out to once a year. Only the whole +# scheduler can be disabled, and that would stop enrichment too. +[cluster] +node_id = 1 + +[storage] +data_path = "/data/cortex" +wal_sync = true + +[engine] +vector_dimensions = 3072 +hnsw_m = 32 +hnsw_ef_construction = 500 +hnsw_ef_search = 200 +# v0.9.9's shipped TOML enum uses `None` for lossless fp32 storage. +hnsw_quantization = "None" +block_cache_bytes = 536870912 + +[network] + +[llm] + +[governance] + +[scheduler] +enabled = true +compaction_interval_secs = 300 +methylation_interval_secs = 600 +cognitive_persist_interval_secs = 31536000 +feedback_weight_interval_secs = 31536000 diff --git a/integration/cortexdb/docker-compose.yml b/integration/cortexdb/docker-compose.yml index e90ad3ba..3e496b8b 100644 --- a/integration/cortexdb/docker-compose.yml +++ b/integration/cortexdb/docker-compose.yml @@ -31,6 +31,8 @@ services: CORTEX_VERIFIER_API_KEY: ${CORTEX_INFERENCE_KEY:-tinymemory-test} CORTEX_VERIFIER_MODEL: ${CORTEX_VERIFIER_MODEL:-max-reasoning} CORTEX_VERIFIER_MAX_TOKENS: "16384" + # Only the rerank-cohere flag profile uses it. + COHERE_API_KEY: ${COHERE_API_KEY:-} # The server's feature flags: the baseline, then the profile under test # (scripts/memory-flag-sweep.sh), which only names what it changes. See # flags/README.md. A variable in `environment` above wins over both. From c0b333ee94ad07e267e0dac5dcdd782ca963fbfb Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:34:53 +0300 Subject: [PATCH 04/24] chore: files changed crates/tinymemory-integrations/examples/memory_eval/inspect.rs Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/inspect.rs | 114 ++++++++++++++++-- 1 file changed, 103 insertions(+), 11 deletions(-) diff --git a/crates/tinymemory-integrations/examples/memory_eval/inspect.rs b/crates/tinymemory-integrations/examples/memory_eval/inspect.rs index 9f1d2f1c..aeca5bbd 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/inspect.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/inspect.rs @@ -7,7 +7,9 @@ //! the models cost. That is the only way to tell whether memory captured an //! event even when a pack does not show it. -use serde::Serialize; +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; use serde_json::{Value, json}; /// The derived layers of one scope. @@ -127,6 +129,10 @@ pub(crate) struct Captured { /// Each open or resolved conflict as "kind status: subject predicate /// [values]". pub(crate) conflicts: Vec, + /// Beliefs by stance (`supported`, `contested`, …). + pub(crate) stances: BTreeMap, + /// Every belief's confidence. + pub(crate) confidences: Vec, } impl Captured { @@ -138,19 +144,40 @@ impl Captured { .chain(&self.beliefs) .any(|line| line.to_lowercase().contains(&needle)) } + + /// Adds everything `other` holds. + pub(crate) fn extend(&mut self, other: Self) { + self.facts.extend(other.facts); + self.beliefs.extend(other.beliefs); + self.conflicts.extend(other.conflicts); + for (stance, n) in other.stances { + *self.stances.entry(stance).or_default() += n; + } + self.confidences.extend(other.confidences); + } } -/// The models' usage CortexDB accounts for, as its routers price it. -#[derive(Debug, Clone, Copy, Default, Serialize)] -pub(crate) struct Usage { +/// One line of the models' usage. +#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)] +pub(crate) struct Spend { pub(crate) calls: u64, pub(crate) tokens: u64, pub(crate) cost_usd: f64, } -impl Usage { +impl Spend { + /// A usage line as `v1/admin/usage` reports it. + fn of(line: &Value) -> Self { + let count = |name: &str| line[name].as_u64().unwrap_or_default(); + Self { + calls: count("calls"), + tokens: count("tokens_total") + count("tokens_unsplit"), + cost_usd: line["cost_usd"].as_f64().unwrap_or_default(), + } + } + /// What was spent between `before` and `self`. - pub(crate) fn since(self, before: Self) -> Self { + fn since(self, before: Self) -> Self { Self { calls: self.calls.saturating_sub(before.calls), tokens: self.tokens.saturating_sub(before.tokens), @@ -159,6 +186,46 @@ impl Usage { } } +/// The models' usage CortexDB accounts for, as its routers price it: in +/// total, and by the role a model plays (extraction, enrichment, answer, +/// embedding, …). +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub(crate) struct Usage { + pub(crate) calls: u64, + pub(crate) tokens: u64, + pub(crate) cost_usd: f64, + pub(crate) by_role: BTreeMap, +} + +impl Usage { + /// What was spent between `before` and `self`. + pub(crate) fn since(&self, before: &Self) -> Self { + let total = self.total().since(before.total()); + Self { + calls: total.calls, + tokens: total.tokens, + cost_usd: total.cost_usd, + by_role: self + .by_role + .iter() + .map(|(role, spend)| { + let earlier = before.by_role.get(role).copied().unwrap_or_default(); + (role.clone(), spend.since(earlier)) + }) + .filter(|(_, spend)| spend.calls > 0 || spend.cost_usd > 0.0) + .collect(), + } + } + + fn total(&self) -> Spend { + Spend { + calls: self.calls, + tokens: self.tokens, + cost_usd: self.cost_usd, + } + } +} + /// A claim's part (`subject`, `predicate` or `object`) as text. fn part(value: Option<&Value>) -> String { value @@ -205,6 +272,11 @@ impl Inspector { } for belief in listed(&self.get("v1/beliefs", &page).await?, "beliefs") { captured.beliefs.push(claim(&belief)); + let stance = belief["stance"].as_str().unwrap_or("unknown"); + *captured.stances.entry(stance.to_string()).or_default() += 1; + if let Some(confidence) = belief["confidence"].as_f64() { + captured.confidences.push(confidence); + } } for conflict in listed(&self.get("v1/conflicts", &page).await?, "conflicts") { let values: Vec = conflict["records"] @@ -242,12 +314,32 @@ impl Inspector { /// The models' usage so far. pub(crate) async fn usage(&self) -> Result { let report = self.get("v1/admin/usage", &[]).await?; - let total = &report["total"]; + let total = Spend::of(&report["total"]); + // A map keyed by role, or a list of lines that each name theirs. + let by_role = match &report["by_role"] { + Value::Object(roles) => roles + .iter() + .map(|(role, line)| (role.clone(), Spend::of(line))) + .collect(), + Value::Array(lines) => lines + .iter() + .map(|line| { + let role = line["role"].as_str().unwrap_or("unknown"); + (role.to_string(), Spend::of(line)) + }) + .collect(), + _ => BTreeMap::new(), + }; Ok(Usage { - calls: total["calls"].as_u64().unwrap_or_default(), - tokens: total["tokens_total"].as_u64().unwrap_or_default() - + total["tokens_unsplit"].as_u64().unwrap_or_default(), - cost_usd: total["cost_usd"].as_f64().unwrap_or_default(), + calls: total.calls, + tokens: total.tokens, + cost_usd: total.cost_usd, + by_role, }) } + + /// The server's version and the capabilities it advertises. + pub(crate) async fn version(&self) -> Result { + self.get("v1/admin/version", &[]).await + } } From c84905a1bbfad311fd28e4ec0f7d10ec6a68b784 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:35:06 +0300 Subject: [PATCH 05/24] feat(memory-eval): add llm and score modules to the eval example Adds an LLM client and a scoring module to the memory_eval example so the harness can call a model and grade its answers. The scoring module computes the metrics the example reports, and the LLM module handles the request and response plumbing they depend on. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/llm.rs | 26 ++++++++++++++++--- .../examples/memory_eval/score.rs | 5 ++++ 2 files changed, 28 insertions(+), 3 deletions(-) diff --git a/crates/tinymemory-integrations/examples/memory_eval/llm.rs b/crates/tinymemory-integrations/examples/memory_eval/llm.rs index 07e345e4..bcb14765 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/llm.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/llm.rs @@ -12,6 +12,8 @@ //! - `EVAL_LLM_MODEL`: default `openai/gpt-4.1-mini`. //! //! It answers at temperature 0 and is scored like the extractive answer. +//! Each answer carries the tokens it took and, where the endpoint reports it +//! (OpenRouter does), its cost, so a run's cost covers the answers too. use serde_json::{Value, json}; @@ -20,6 +22,14 @@ const SYSTEM: &str = "You are an assistant with a long-term memory. The user's m starts with what your memory recalled, then their question. Answer the question in one \ short sentence, using only the memory. If the memory does not say, answer \"unknown\"."; +/// One answer and what it cost. +pub(crate) struct Answer { + pub(crate) text: String, + pub(crate) tokens: u64, + /// `None` when the endpoint does not price its calls. + pub(crate) cost_usd: Option, +} + /// A chat model. pub(crate) struct Llm { client: reqwest::Client, @@ -56,7 +66,7 @@ impl Llm { /// # Errors /// /// A transport failure, or an answer without text. - pub(crate) async fn answer(&self, pack: &str, question: &str) -> Result { + pub(crate) async fn answer(&self, pack: &str, question: &str) -> Result { let body = json!({ "model": self.model, "temperature": 0, @@ -64,6 +74,8 @@ impl Llm { // Flash, cannot turn it off) as well as the one-sentence answer. "max_tokens": 2000, "reasoning": { "effort": "low" }, + // OpenRouter's usage accounting: the call's cost in `usage.cost`. + "usage": { "include": true }, "messages": [ { "role": "system", "content": SYSTEM }, { "role": "user", "content": format!("{pack}\n\nQuestion: {question}") }, @@ -81,10 +93,18 @@ impl Llm { .json() .await .map_err(|error| error.to_string())?; - answer + let text = answer .pointer("/choices/0/message/content") .and_then(Value::as_str) .map(|text| text.trim().to_string()) - .ok_or_else(|| format!("no answer text in {answer}")) + .ok_or_else(|| format!("no answer text in {answer}"))?; + Ok(Answer { + text, + tokens: answer + .pointer("/usage/total_tokens") + .and_then(Value::as_u64) + .unwrap_or_default(), + cost_usd: answer.pointer("/usage/cost").and_then(Value::as_f64), + }) } } diff --git a/crates/tinymemory-integrations/examples/memory_eval/score.rs b/crates/tinymemory-integrations/examples/memory_eval/score.rs index ecca88a4..0073ef5d 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/score.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/score.rs @@ -46,6 +46,9 @@ pub(crate) struct ProbeResult { /// The `--llm` model's answer from the same pack. pub(crate) llm_answer: Option, pub(crate) llm_ok: Option, + /// What the `--llm` answer took: tokens, and dollars where priced. + pub(crate) llm_tokens: u64, + pub(crate) llm_cost_usd: Option, /// Synthesis phase on CortexDB: whether a fact or belief CortexDB /// derived holds the expected answer, whether or not the pack shows it. pub(crate) captured: Option, @@ -135,6 +138,8 @@ pub(crate) fn score( answer_ok, llm_answer: None, llm_ok: None, + llm_tokens: 0, + llm_cost_usd: None, captured: None, ms, tokens, From fc5f9a0df04cfc7a9dee21f344b1562675b8de9e Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:36:17 +0300 Subject: [PATCH 06/24] chore: files changed crates/tinymemory-integrations/examples/memory_eval/kpi.rs Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/kpi.rs | 520 ++++++++++++++++++ 1 file changed, 520 insertions(+) create mode 100644 crates/tinymemory-integrations/examples/memory_eval/kpi.rs diff --git a/crates/tinymemory-integrations/examples/memory_eval/kpi.rs b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs new file mode 100644 index 00000000..bd932058 --- /dev/null +++ b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs @@ -0,0 +1,520 @@ +//! The run's key numbers, grouped by what memory is for. +//! +//! The accuracy tables answer "did this probe pass". The KPIs answer the +//! questions a configuration is chosen on, one number each, so two runs +//! (two CortexDB flag profiles, see `compare`) can be set side by side: +//! +//! - **accuracy**: how often the pack, and an answer read from it, is right. +//! - **learning**: whether corrections and standing instructions stick +//! (the `learnings` and `learning_from_feedback` scenarios), what belief +//! building adds over raw recall, and what beliefs CortexDB holds. +//! - **surprise**: whether a break from routine surfaces (`surprise`). +//! - **conflicts**: whether disagreeing sources are flagged (`conflicts`), +//! whether conflicts are raised where there are none, and whether the +//! newest value wins a superseded one (`contradictions`). +//! - **cost**: what CortexDB's models and the `--llm` answerer spent, per +//! correct answer, and the prompt tokens a pack costs the host. +//! - **latency**: what an agent waits for. +//! +//! Unless a KPI says otherwise it is read in the synthesis phase, after the +//! belief build: the state memory settles into. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +use crate::inspect::Usage; +use crate::score::{Latency, ProbeResult, Totals}; +use crate::{ScenarioReport, Timings}; + +/// Scenarios whose probes test learning. +const LEARNING: [&str; 2] = ["learnings", "learning_from_feedback"]; + +/// Scenarios whose probes test surprise. +const SURPRISE: [&str; 1] = ["surprise"]; + +/// The conflicts a scenario plants, each named by a word its subject or +/// values hold. Scenarios not listed plant none, so a conflict raised there +/// is spurious; `contradictions` is left out of both counts, since a +/// superseded value may fairly be either closed or flagged. +const PLANTED: [(&str, &[&str]); 1] = [("conflicts", &["refund"])]; + +/// Scenarios that may hold conflicts without them counting as spurious. +const CONFLICTED: [&str; 2] = ["conflicts", "contradictions"]; + +/// What a KPI's value is. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub(crate) enum Unit { + /// A percentage, 0 to 100. + Pct, + /// A difference of percentages, in points. + Points, + /// US dollars. + Usd, + /// A count. + Count, + /// Milliseconds. + Ms, + /// A score from 0 to 1. + Score, +} + +/// Which way a KPI improves. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub(crate) enum Better { + Higher, + Lower, + /// Neither: context, not a target. + Neither, +} + +/// One key number. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub(crate) struct Kpi { + pub(crate) group: String, + pub(crate) name: String, + /// `None` when the run could not measure it (no `--llm`, no CortexDB). + pub(crate) value: Option, + pub(crate) unit: Unit, + pub(crate) better: Better, +} + +impl Kpi { + fn new(group: &str, name: &str, value: Option, unit: Unit, better: Better) -> Self { + Self { + group: group.to_string(), + name: name.to_string(), + value, + unit, + better, + } + } + + /// The value as a table cell. + pub(crate) fn cell(&self) -> String { + self.value.map_or_else(|| "–".to_string(), |v| format(v, self.unit)) + } +} + +/// `value` in `unit`, for a table. +pub(crate) fn format(value: f64, unit: Unit) -> String { + match unit { + Unit::Pct => format!("{value:.0}%"), + Unit::Points => format!("{value:+.0} pp"), + Unit::Usd => format!("${value:.3}"), + Unit::Count => format!("{value:.0}"), + Unit::Ms => format!("{value:.0} ms"), + Unit::Score => format!("{value:.2}"), + } +} + +/// `part` of `whole` as a percentage, if there is a whole. +fn pct(part: usize, whole: usize) -> Option { + (whole > 0).then(|| 100.0 * part as f64 / whole as f64) +} + +/// The probes of `phase` in the scenarios `keep` accepts. +fn probes<'a>( + reports: &'a [ScenarioReport], + phase: &'a str, + keep: impl Fn(&str) -> bool + 'a, +) -> impl Iterator + 'a { + reports + .iter() + .filter(move |report| keep(report.name)) + .flat_map(|report| &report.probes) + .filter(move |probe| probe.phase == phase) +} + +/// Every KPI of a run. `usage` is CortexDB's spend over the run, absent on +/// the reference engine. +pub(crate) fn compute( + reports: &[ScenarioReport], + usage: Option<&Usage>, + timings: &Timings, +) -> Vec { + use Better::{Higher, Lower, Neither}; + let on_cortex = usage.is_some(); + let all = |phase| Totals::of(probes(reports, phase, |_| true)); + let (recall, synthesis) = (all("recall"), all("synthesis")); + let learning = Totals::of(probes(reports, "synthesis", |n| LEARNING.contains(&n))); + let surprise = Totals::of(probes(reports, "synthesis", |n| SURPRISE.contains(&n))); + let conflicts = Totals::of(probes(reports, "synthesis", |n| n == "conflicts")); + let leaks = Totals::of(reports.iter().flat_map(|r| &r.probes)); + let mut kpis = vec![ + Kpi::new( + "accuracy", + "pack hit (recall)", + pct(recall.hits, recall.scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "accuracy", + "pack hit", + pct(synthesis.hits, synthesis.scored), + Unit::Pct, + Higher, + ), + Kpi::new("accuracy", "MRR", Some(synthesis.mrr), Unit::Score, Higher), + Kpi::new( + "accuracy", + "extractive answer", + pct(synthesis.answers_ok, synthesis.scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "accuracy", + "model answer", + pct(synthesis.llm_ok, synthesis.llm_scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "accuracy", + "captured", + pct(synthesis.captured, synthesis.captured_checked), + Unit::Pct, + Higher, + ), + Kpi::new( + "accuracy", + "leaks", + Some(leaks.leaks as f64), + Unit::Count, + Lower, + ), + ]; + + // Learning. + let mut held = crate::inspect::Captured::default(); + for report in reports { + held.extend(report.synthesis.captured.clone()); + } + let contested: usize = held + .stances + .iter() + .filter(|(stance, _)| stance.as_str() != "supported") + .map(|(_, n)| n) + .sum(); + let confidence = (!held.confidences.is_empty()) + .then(|| held.confidences.iter().sum::() / held.confidences.len() as f64); + let gain = pct(synthesis.hits, synthesis.scored) + .zip(pct(recall.hits, recall.scored)) + .map(|(after, before)| after - before); + kpis.extend([ + Kpi::new( + "learning", + "lesson in pack", + pct(learning.hits, learning.scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "learning", + "lesson answered", + pct(learning.llm_ok, learning.llm_scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "learning", + "lesson captured", + pct(learning.captured, learning.captured_checked), + Unit::Pct, + Higher, + ), + Kpi::new("learning", "synthesis gain", gain, Unit::Points, Higher), + Kpi::new( + "learning", + "beliefs built", + Some(reports.iter().map(|r| r.synthesis.built).sum::() as f64), + Unit::Count, + Neither, + ), + Kpi::new( + "learning", + "beliefs held", + on_cortex.then_some(held.beliefs.len() as f64), + Unit::Count, + Neither, + ), + Kpi::new( + "learning", + "beliefs not supported", + on_cortex.then_some(contested as f64), + Unit::Count, + Neither, + ), + Kpi::new( + "learning", + "belief confidence", + confidence, + Unit::Score, + Neither, + ), + Kpi::new( + "learning", + "facts held", + on_cortex.then_some(held.facts.len() as f64), + Unit::Count, + Neither, + ), + ]); + + // Surprise. + kpis.extend([ + Kpi::new( + "surprise", + "surprise in pack", + pct(surprise.hits, surprise.scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "surprise", + "surprise MRR", + (surprise.scored > 0).then_some(surprise.mrr), + Unit::Score, + Higher, + ), + Kpi::new( + "surprise", + "surprise answered", + pct(surprise.llm_ok, surprise.llm_scored), + Unit::Pct, + Higher, + ), + ]); + + // Conflicts. + let (mut planted, mut found, mut spurious) = (0, 0, 0); + for report in reports { + let raised = &report.synthesis.captured.conflicts; + match PLANTED.iter().find(|(name, _)| *name == report.name) { + Some((_, words)) => { + planted += words.len(); + found += words + .iter() + .filter(|word| raised.iter().any(|c| c.to_lowercase().contains(*word))) + .count(); + } + None if !CONFLICTED.contains(&report.name) => spurious += raised.len(), + None => {} + } + } + let raised: usize = reports + .iter() + .map(|r| r.synthesis.captured.conflicts.len()) + .sum(); + kpis.extend([ + Kpi::new( + "conflicts", + "planted conflicts flagged", + on_cortex.then(|| pct(found, planted)).flatten(), + Unit::Pct, + Higher, + ), + Kpi::new( + "conflicts", + "spurious conflicts", + on_cortex.then_some(spurious as f64), + Unit::Count, + Lower, + ), + Kpi::new( + "conflicts", + "conflicts raised", + on_cortex.then_some(raised as f64), + Unit::Count, + Neither, + ), + Kpi::new( + "conflicts", + "disagreement in pack", + pct(conflicts.hits, conflicts.scored), + Unit::Pct, + Higher, + ), + Kpi::new( + "conflicts", + "fresh first", + pct(synthesis.fresh_first, synthesis.contradictions), + Unit::Pct, + Higher, + ), + ]); + + // Cost. + let answered: Vec<&ProbeResult> = reports + .iter() + .flat_map(|r| &r.probes) + .filter(|p| p.llm_ok.is_some()) + .collect(); + let answer_usd = answered + .iter() + .filter_map(|p| p.llm_cost_usd) + .reduce(|a, b| a + b); + let answer_tokens: u64 = answered.iter().map(|p| p.llm_tokens).sum(); + let spent = usage.map(|u| u.cost_usd + answer_usd.unwrap_or_default()); + let correct = if synthesis.llm_scored > 0 { + synthesis.llm_ok + } else { + synthesis.answers_ok + }; + let all_probes: Vec<&ProbeResult> = reports.iter().flat_map(|r| &r.probes).collect(); + let pack_tokens = (!all_probes.is_empty()).then(|| { + all_probes.iter().map(|p| p.tokens).sum::() as f64 / all_probes.len() as f64 + }); + kpis.extend([ + Kpi::new( + "cost", + "CortexDB models", + usage.map(|u| u.cost_usd), + Unit::Usd, + Lower, + ), + Kpi::new( + "cost", + "CortexDB model calls", + usage.map(|u| u.calls as f64), + Unit::Count, + Lower, + ), + Kpi::new( + "cost", + "CortexDB model tokens", + usage.map(|u| u.tokens as f64), + Unit::Count, + Lower, + ), + Kpi::new("cost", "answerer", answer_usd, Unit::Usd, Lower), + Kpi::new( + "cost", + "answerer tokens", + (!answered.is_empty()).then_some(answer_tokens as f64), + Unit::Count, + Lower, + ), + Kpi::new( + "cost", + "per correct answer", + spent.filter(|_| correct > 0).map(|usd| usd / correct as f64), + Unit::Usd, + Lower, + ), + Kpi::new( + "cost", + "pack tokens (mean)", + pack_tokens, + Unit::Count, + Lower, + ), + ]); + if let Some(usage) = usage { + for (role, spend) in &usage.by_role { + kpis.push(Kpi::new( + "cost by role", + role, + Some(spend.cost_usd), + Unit::Usd, + Lower, + )); + } + } + + // Latency. + let samples = |keep: &dyn Fn(&str) -> bool| -> Vec { + timings + .0 + .iter() + .filter(|(step, _)| keep(step)) + .flat_map(|(_, samples)| samples.iter().copied()) + .collect() + }; + let pre_turn = Latency::of(&samples(&|step| step.starts_with("pre_turn"))); + let probe = Latency::of(&samples(&|step| step.starts_with("probe "))); + let total = |step: &str| { + timings + .0 + .get(step) + .map(|samples| samples.iter().sum::()) + }; + kpis.extend([ + Kpi::new( + "latency", + "pre_turn p50", + (pre_turn.n > 0).then_some(pre_turn.p50), + Unit::Ms, + Lower, + ), + Kpi::new( + "latency", + "pre_turn p95", + (pre_turn.n > 0).then_some(pre_turn.p95), + Unit::Ms, + Lower, + ), + Kpi::new( + "latency", + "probe p50", + (probe.n > 0).then_some(probe.p50), + Unit::Ms, + Lower, + ), + Kpi::new( + "latency", + "probe p95", + (probe.n > 0).then_some(probe.p95), + Unit::Ms, + Lower, + ), + Kpi::new( + "latency", + "enrichment drain (total)", + total("enrichment (queue drained)"), + Unit::Ms, + Lower, + ), + Kpi::new( + "latency", + "belief builds (total)", + total("synthesis (all builds)"), + Unit::Ms, + Lower, + ), + ]); + kpis +} + +/// The KPIs as a markdown section, one table per group. +pub(crate) fn print(label: &str, kpis: &[Kpi]) { + println!("\n## KPIs (`{label}`)\n"); + let mut groups: BTreeMap)> = BTreeMap::new(); + let mut order: Vec<&str> = Vec::new(); + for kpi in kpis { + let at = order + .iter() + .position(|g| *g == kpi.group) + .unwrap_or_else(|| { + order.push(&kpi.group); + order.len() - 1 + }); + groups.entry(at).or_insert((&kpi.group, Vec::new())).1.push(kpi); + } + println!("| Group | KPI | Value | Better |"); + println!("| --- | --- | --- | --- |"); + for (group, kpis) in groups.values() { + for kpi in kpis { + let better = match kpi.better { + Better::Higher => "higher", + Better::Lower => "lower", + Better::Neither => "–", + }; + println!("| {group} | {} | {} | {better} |", kpi.name, kpi.cell()); + } + } +} From a946090d25c040f24734da56f14a55c02c154e8b Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:36:30 +0300 Subject: [PATCH 07/24] chore: files changed crates/tinymemory-integrations/examples/memory_eval/kpi.rs Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/kpi.rs | 37 +++++++------------ 1 file changed, 13 insertions(+), 24 deletions(-) diff --git a/crates/tinymemory-integrations/examples/memory_eval/kpi.rs b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs index bd932058..3585953b 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/kpi.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs @@ -19,8 +19,6 @@ //! Unless a KPI says otherwise it is read in the synthesis phase, after the //! belief build: the state memory settles into. -use std::collections::BTreeMap; - use serde::{Deserialize, Serialize}; use crate::inspect::Usage; @@ -490,31 +488,22 @@ pub(crate) fn compute( kpis } -/// The KPIs as a markdown section, one table per group. +/// The KPIs as a markdown table, in the order `compute` groups them. pub(crate) fn print(label: &str, kpis: &[Kpi]) { println!("\n## KPIs (`{label}`)\n"); - let mut groups: BTreeMap)> = BTreeMap::new(); - let mut order: Vec<&str> = Vec::new(); - for kpi in kpis { - let at = order - .iter() - .position(|g| *g == kpi.group) - .unwrap_or_else(|| { - order.push(&kpi.group); - order.len() - 1 - }); - groups.entry(at).or_insert((&kpi.group, Vec::new())).1.push(kpi); - } println!("| Group | KPI | Value | Better |"); println!("| --- | --- | --- | --- |"); - for (group, kpis) in groups.values() { - for kpi in kpis { - let better = match kpi.better { - Better::Higher => "higher", - Better::Lower => "lower", - Better::Neither => "–", - }; - println!("| {group} | {} | {} | {better} |", kpi.name, kpi.cell()); - } + for kpi in kpis { + let better = match kpi.better { + Better::Higher => "higher", + Better::Lower => "lower", + Better::Neither => "–", + }; + println!( + "| {} | {} | {} | {better} |", + kpi.group, + kpi.name, + kpi.cell() + ); } } From 268a81bdb83c5f2b0a7de421ca4881ae25f9165b Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:37:13 +0300 Subject: [PATCH 08/24] chore: files changed crates/tinymemory-integrations/examples/memory_eval/compare.rs Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/compare.rs | 234 ++++++++++++++++++ 1 file changed, 234 insertions(+) create mode 100644 crates/tinymemory-integrations/examples/memory_eval/compare.rs diff --git a/crates/tinymemory-integrations/examples/memory_eval/compare.rs b/crates/tinymemory-integrations/examples/memory_eval/compare.rs new file mode 100644 index 00000000..56cbcd80 --- /dev/null +++ b/crates/tinymemory-integrations/examples/memory_eval/compare.rs @@ -0,0 +1,234 @@ +//! Comparing runs: which CortexDB flags move which KPI. +//! +//! `memory_eval compare …` reads the `--json` reports of several +//! runs, groups them by flag profile (the `profile` each records; the label +//! when there is none), averages repeats, and prints one table per KPI +//! group with every profile's delta from the baseline (the `baseline` +//! profile, else the first file). +//! +//! A delta is called a move only when it is larger than the noise: the +//! spread (max − min) the repeats of either profile show, and never less +//! than a floor for a single run (3 points for a percentage, 0.03 for a +//! score, 10% of the baseline otherwise). A move is marked `▲` when it is an +//! improvement, `▼` when it is a regression, and `~` when it is within noise. + +use std::collections::BTreeMap; + +use serde::Deserialize; + +use crate::kpi::{Better, Kpi, Unit, format}; + +type Error = Box; + +/// The parts of a run's report a comparison reads. +#[derive(Deserialize)] +struct Run { + label: String, + #[serde(default)] + profile: Option, + /// The flags the profile set on the server. + #[serde(default)] + flags: BTreeMap, + #[serde(default)] + kpis: Vec, +} + +/// Every run of one profile. +struct Profile { + name: String, + flags: BTreeMap, + runs: usize, + /// Each KPI's values across the runs, by name. + values: BTreeMap>, +} + +impl Profile { + fn mean(&self, kpi: &str) -> Option { + let values = self.values.get(kpi)?; + (!values.is_empty()).then(|| values.iter().sum::() / values.len() as f64) + } + + fn spread(&self, kpi: &str) -> f64 { + self.values.get(kpi).map_or(0.0, |values| { + let max = values.iter().copied().fold(f64::MIN, f64::max); + let min = values.iter().copied().fold(f64::MAX, f64::min); + if values.is_empty() { 0.0 } else { max - min } + }) + } +} + +/// How a profile's KPI compares with the baseline's. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum Move { + Better, + Worse, + /// Within noise. + Same, + /// Changed, but the KPI has no better direction. + Changed, +} + +/// The smallest delta a single run can call a move. +fn floor(unit: Unit, baseline: f64) -> f64 { + match unit { + Unit::Pct | Unit::Points => 3.0, + Unit::Score => 0.03, + Unit::Usd | Unit::Count | Unit::Ms => 0.1 * baseline.abs(), + } +} + +/// Whether `value` moved from `baseline` by more than `noise`, and which way. +pub(crate) fn judge(baseline: f64, value: f64, noise: f64, unit: Unit, better: Better) -> Move { + let delta = value - baseline; + if delta.abs() <= noise.max(floor(unit, baseline)) { + return Move::Same; + } + match (better, delta > 0.0) { + (Better::Higher, true) | (Better::Lower, false) => Move::Better, + (Better::Higher, false) | (Better::Lower, true) => Move::Worse, + (Better::Neither, _) => Move::Changed, + } +} + +/// `value - baseline`, in a form readable next to `value`. +fn delta(baseline: f64, value: f64, unit: Unit) -> String { + match unit { + Unit::Pct | Unit::Points => format!("{:+.0} pp", value - baseline), + Unit::Score => format!("{:+.2}", value - baseline), + _ if baseline == 0.0 => format!("{:+.0}", value - baseline), + _ => format!("{:+.0}%", 100.0 * (value - baseline) / baseline.abs()), + } +} + +/// Reads `paths` and prints the comparison. +/// +/// # Errors +/// +/// A file that cannot be read or is not a run's report. +pub(crate) fn run(paths: &[String]) -> Result<(), Error> { + if paths.is_empty() { + return Err("compare needs at least one run's --json report".into()); + } + let mut profiles: Vec = Vec::new(); + // The KPIs in the order the first run that has them lists them. + let mut order: Vec = Vec::new(); + for path in paths { + let run: Run = serde_json::from_str(&std::fs::read_to_string(path)?) + .map_err(|error| format!("{path}: {error}"))?; + let name = run.profile.clone().unwrap_or_else(|| run.label.clone()); + let at = match profiles.iter().position(|p| p.name == name) { + Some(at) => at, + None => { + profiles.push(Profile { + name, + flags: run.flags.clone(), + runs: 0, + values: BTreeMap::new(), + }); + profiles.len() - 1 + } + }; + let profile = &mut profiles[at]; + profile.runs += 1; + for kpi in run.kpis { + if !order.iter().any(|k| k.group == kpi.group && k.name == kpi.name) { + order.push(kpi.clone()); + } + if let Some(value) = kpi.value { + profile.values.entry(kpi.name).or_default().push(value); + } + } + } + let base = profiles + .iter() + .position(|p| p.name == "baseline") + .unwrap_or(0); + profiles.swap(0, base); + let baseline = &profiles[0]; + + println!("# CortexDB flag comparison\n"); + println!( + "Deltas are against `{}`. ▲ better, ▼ worse, ~ within noise (the repeats' \ + spread, at least 3 pp / 0.03 / 10%).\n", + baseline.name + ); + println!("| Profile | Runs | Flags over the baseline |"); + println!("| --- | --- | --- |"); + for profile in &profiles { + let flags: Vec = profile + .flags + .iter() + .filter(|(key, value)| baseline.flags.get(*key) != Some(*value)) + .map(|(key, value)| format!("`{key}={value}`")) + .collect(); + println!( + "| {} | {} | {} |", + profile.name, + profile.runs, + if flags.is_empty() { + "–".to_string() + } else { + flags.join(" ") + } + ); + } + + let mut moved: BTreeMap<&str, Vec> = BTreeMap::new(); + let mut groups: Vec<&str> = order.iter().map(|k| k.group.as_str()).collect(); + groups.dedup(); + for group in groups { + let kpis: Vec<&Kpi> = order.iter().filter(|k| k.group == group).collect(); + println!("\n## {group}\n"); + let names: Vec<&str> = kpis.iter().map(|k| k.name.as_str()).collect(); + println!("| Profile | {} |", names.join(" | ")); + println!("| --- |{}", " --- |".repeat(kpis.len())); + for profile in &profiles { + let mut cells = Vec::new(); + for kpi in &kpis { + let Some(value) = profile.mean(&kpi.name) else { + cells.push("–".to_string()); + continue; + }; + let shown = format(value, kpi.unit); + let compared = baseline + .mean(&kpi.name) + .filter(|_| !std::ptr::eq(profile, baseline)); + cells.push(match compared { + None => shown, + Some(base) => { + let noise = baseline.spread(&kpi.name).max(profile.spread(&kpi.name)); + let verdict = judge(base, value, noise, kpi.unit, kpi.better); + let mark = match verdict { + Move::Better => "▲", + Move::Worse => "▼", + Move::Same => "~", + Move::Changed => "Δ", + }; + if matches!(verdict, Move::Better | Move::Worse) { + moved.entry(&profile.name).or_default().push(format!( + "{mark} {} {}", + kpi.name, + delta(base, value, kpi.unit) + )); + } + format!("{shown} ({} {mark})", delta(base, value, kpi.unit)) + } + }); + } + println!("| {} | {} |", profile.name, cells.join(" | ")); + } + } + + println!("\n## What moved\n"); + for profile in profiles.iter().skip(1) { + match moved.get(profile.name.as_str()) { + Some(changes) => println!("- **{}**: {}", profile.name, changes.join(", ")), + None => println!("- **{}**: nothing beyond noise", profile.name), + } + } + Ok(()) +} + +#[cfg(test)] +#[path = "compare_tests.rs"] +mod tests; From 89406f1e98a242177742da942de4030adbd07292 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:38:13 +0300 Subject: [PATCH 09/24] chore: files changed crates/tinymemory-integrations/Cargo.toml,crates/tinymemory-integrations/exampl Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinymemory-integrations/Cargo.toml | 3 + .../examples/memory_eval/main.rs | 103 +++++++++++++++--- 2 files changed, 90 insertions(+), 16 deletions(-) diff --git a/crates/tinymemory-integrations/Cargo.toml b/crates/tinymemory-integrations/Cargo.toml index 62c4c7a3..76f1a5e0 100644 --- a/crates/tinymemory-integrations/Cargo.toml +++ b/crates/tinymemory-integrations/Cargo.toml @@ -142,6 +142,9 @@ required-features = ["cortex"] name = "memory_eval" path = "examples/memory_eval/main.rs" required-features = ["cortex", "brain"] +# Runs the eval's own unit tests (KPI arithmetic, comparison verdicts) under +# `cargo test`; the eval itself only runs through `cargo run`. +test = true [[example]] name = "cortex_agent" diff --git a/crates/tinymemory-integrations/examples/memory_eval/main.rs b/crates/tinymemory-integrations/examples/memory_eval/main.rs index a10d1231..e714edaa 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/main.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/main.rs @@ -1,12 +1,14 @@ //! An accuracy and latency eval of the agent memory lifecycle. //! -//! A scripted agent (`agent`) plays nine scenarios (`scenarios`) through the -//! real lifecycle calls: brain lookups, a restart, contradicting facts, a +//! A scripted agent (`agent`) plays twelve scenarios (`scenarios`) through +//! the real lifecycle calls: brain lookups, a restart, contradicting facts, a //! tool-heavy incident, a team handoff, compaction, tenant isolation, a -//! needle in noise, and explicit learnings. After each scenario's writes -//! settle, its probes are scored (`score`). Every scenario then runs a -//! belief build over its whole tree and is probed again, so the effect of -//! synthesis shows up as a second phase. +//! needle in noise, explicit learnings, learning from corrections, a +//! surprise, and conflicting sources. After each scenario's writes settle, +//! its probes are scored (`score`). Every scenario then runs a belief build +//! over its whole tree and is probed again, so the effect of synthesis shows +//! up as a second phase. The run ends with its KPIs (`kpi`): accuracy, +//! learning, surprise, conflicts, cost and latency. //! //! ```sh //! # Offline, against the reference engine: @@ -31,11 +33,18 @@ //! - `--llm`: also have a model answer every probe from its pack (see //! `llm`). //! +//! `memory_eval compare …` compares the reports of several runs +//! instead (see `compare`). A run records the CortexDB flag profile it ran +//! under when `CORTEX_FLAGS_FILE` names one (see +//! `integration/cortexdb/flags/` and `scripts/memory-flag-sweep.sh`). +//! //! Everything is written below roots unique to the run and forgotten at the //! end, unless `CORTEX_DB_KEEP` is set. mod agent; +mod compare; mod inspect; +mod kpi; mod llm; mod scenarios; mod score; @@ -59,7 +68,7 @@ use tinymemory_tools::{ }; use agent::{ScriptedAgent, ms}; -use inspect::{Captured, Derived, Inspector}; +use inspect::{Captured, Derived, Inspector, Usage}; use llm::Llm; use scenarios::{MAIN, Probe, Scenario, Step, Via}; use score::{Latency, ProbeResult, Totals, grade, score}; @@ -152,11 +161,43 @@ struct ScenarioReport { settle_ms: f64, synthesis: Synthesis, probes: Vec, + /// What CortexDB's models spent on this scenario. + usage: Option, +} + +/// The CortexDB flag profile a run is under: its name, and every flag it +/// sets, the baseline's included. +fn profile() -> Result<(Option, BTreeMap), Error> { + let Ok(file) = std::env::var("CORTEX_FLAGS_FILE") else { + return Ok((None, BTreeMap::new())); + }; + let file = std::path::Path::new(&file); + let mut flags = BTreeMap::new(); + for path in [file.with_file_name("baseline.env"), file.to_path_buf()] { + let Ok(text) = std::fs::read_to_string(&path) else { + continue; + }; + for line in text.lines().map(str::trim) { + if let Some((key, value)) = line.split_once('=').filter(|_| !line.starts_with('#')) { + flags.insert(key.trim().to_string(), value.trim().to_string()); + } + } + } + let name = file + .file_stem() + .and_then(|stem| stem.to_str()) + .ok_or("CORTEX_FLAGS_FILE names no file")?; + Ok((Some(name.to_string()), flags)) } #[tokio::main] async fn main() -> Result<(), Error> { + let raw: Vec = std::env::args().skip(1).collect(); + if raw.first().map(String::as_str) == Some("compare") { + return compare::run(&raw[1..]); + } let args = args()?; + let (profile, flags) = profile()?; let url = std::env::var("CORTEX_DB_URL").unwrap_or_default(); let key = std::env::var("CORTEX_DB_KEY").unwrap_or_else(|_| "tinymemory-cortex-test".into()); let (engine, inspector): (Arc, Option) = match args.engine.as_str() @@ -227,27 +268,47 @@ async fn main() -> Result<(), Error> { print_summary(&args.label, &reports, &timings); let usage = match (&eval.inspector, usage_before) { - (Some(inspector), Some(before)) => Some(inspector.usage().await?.since(before)), + (Some(inspector), Some(before)) => Some(inspector.usage().await?.since(&before)), _ => None, }; - if let Some(usage) = usage { + if let Some(usage) = &usage { println!( "\n## Model usage\n\nCortexDB's models: {} calls, {} tokens, ${:.2} as its routers \ - price them (excludes the `--llm` answers).", + price them (excludes the `--llm` answers).\n", usage.calls, usage.tokens, usage.cost_usd ); + println!("| Scenario | Calls | Tokens | Cost |"); + println!("| --- | --- | --- | --- |"); + for report in &reports { + if let Some(spent) = &report.usage { + println!( + "| {} | {} | {} | ${:.3} |", + report.name, spent.calls, spent.tokens, spent.cost_usd + ); + } + } } + let kpis = kpi::compute(&reports, usage.as_ref(), &timings); + kpi::print(&args.label, &kpis); + let server = match &eval.inspector { + Some(inspector) => inspector.version().await.ok(), + None => None, + }; if let Some(path) = &args.json { if let Some(dir) = std::path::Path::new(path).parent() { std::fs::create_dir_all(dir)?; } let out = serde_json::json!({ "label": args.label, + "profile": profile, + "flags": flags, + "server": server, "engine": engine.descriptor().id, "run": run, "scenarios": reports, "timings": timings, "usage": usage, + "kpis": kpis, }); std::fs::write(path, serde_json::to_string_pretty(&out)?)?; println!("\nwrote {path}"); @@ -296,6 +357,10 @@ impl Eval { timings: &mut Timings, ) -> Result { let (engine, run, policy) = (&self.engine, self.run, &self.policy); + let usage_before = match &self.inspector { + Some(inspector) => Some(inspector.usage().await?), + None => None, + }; let memory = |tenant: &str, agent: &str| -> Result { Ok( AgentMemory::new(engine.clone(), layout(run, scenario.name, tenant)?, agent)? @@ -442,10 +507,9 @@ impl Eval { .derived .push(inspector.derived(scope, &questions).await?); } - let captured = inspector.captured(&scopes).await?; - synthesis.captured.facts.extend(captured.facts); - synthesis.captured.beliefs.extend(captured.beliefs); - synthesis.captured.conflicts.extend(captured.conflicts); + synthesis + .captured + .extend(inspector.captured(&scopes).await?); } } let beliefs: usize = synthesis.derived.iter().map(|d| d.beliefs).sum(); @@ -489,7 +553,12 @@ impl Eval { .await?; } } + let usage = match (&self.inspector, usage_before) { + (Some(inspector), Some(before)) => Some(inspector.usage().await?.since(&before)), + _ => None, + }; Ok(ScenarioReport { + usage, name: scenario.name, about: scenario.about, writes: writes.values().sum(), @@ -600,8 +669,10 @@ impl Eval { let mut result = score(scenario.name, phase, probe, &markdown, tokens, elapsed); if let Some(llm) = llm { let answer = llm.answer(&markdown, probe.question).await?; - result.llm_ok = grade(probe, Some(&answer)); - result.llm_answer = Some(answer); + result.llm_ok = grade(probe, Some(&answer.text)); + result.llm_answer = Some(answer.text); + result.llm_tokens = answer.tokens; + result.llm_cost_usd = answer.cost_usd; } timings.add(&format!("probe {}", result.via), elapsed); Ok(result) From 022d03fbcb1c2c66eef4678b1d956776c7c53bad Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:39:04 +0300 Subject: [PATCH 10/24] chore: files changed crates/tinymemory-integrations/examples/memory_eval/kpi.rs,crates/tinymemory-in Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../examples/memory_eval/compare_tests.rs | 85 ++++++++ .../examples/memory_eval/kpi.rs | 4 + .../examples/memory_eval/kpi_tests.rs | 182 ++++++++++++++++++ 3 files changed, 271 insertions(+) create mode 100644 crates/tinymemory-integrations/examples/memory_eval/compare_tests.rs create mode 100644 crates/tinymemory-integrations/examples/memory_eval/kpi_tests.rs diff --git a/crates/tinymemory-integrations/examples/memory_eval/compare_tests.rs b/crates/tinymemory-integrations/examples/memory_eval/compare_tests.rs new file mode 100644 index 00000000..91508726 --- /dev/null +++ b/crates/tinymemory-integrations/examples/memory_eval/compare_tests.rs @@ -0,0 +1,85 @@ +//! Tests of the comparison's verdicts and of reading run reports. + +use super::*; + +#[test] +fn a_rise_in_a_higher_is_better_kpi_beyond_noise_is_better() { + assert_eq!( + judge(60.0, 70.0, 0.0, Unit::Pct, Better::Higher), + Move::Better + ); + assert_eq!( + judge(60.0, 50.0, 0.0, Unit::Pct, Better::Higher), + Move::Worse + ); +} + +#[test] +fn a_rise_in_cost_is_worse() { + assert_eq!(judge(1.0, 1.5, 0.0, Unit::Usd, Better::Lower), Move::Worse); + assert_eq!(judge(1.0, 0.5, 0.0, Unit::Usd, Better::Lower), Move::Better); +} + +#[test] +fn a_delta_within_the_repeats_spread_is_noise() { + assert_eq!( + judge(60.0, 70.0, 12.0, Unit::Pct, Better::Higher), + Move::Same + ); +} + +#[test] +fn a_single_run_needs_more_than_the_floor_to_move() { + // 3 points for a percentage, 10% of the baseline for a cost. + assert_eq!( + judge(60.0, 62.0, 0.0, Unit::Pct, Better::Higher), + Move::Same + ); + assert_eq!(judge(1.0, 1.05, 0.0, Unit::Usd, Better::Lower), Move::Same); +} + +#[test] +fn a_kpi_with_no_direction_only_changes() { + assert_eq!( + judge(10.0, 20.0, 0.0, Unit::Count, Better::Neither), + Move::Changed + ); +} + +#[test] +fn deltas_read_in_points_or_percent() { + assert_eq!(delta(60.0, 70.0, Unit::Pct), "+10 pp"); + assert_eq!(delta(2.0, 1.0, Unit::Usd), "-50%"); + assert_eq!(delta(0.0, 3.0, Unit::Count), "+3"); +} + +#[test] +fn profiles_average_their_repeats_and_spread_their_range() { + let profile = Profile { + name: "baseline".into(), + flags: BTreeMap::new(), + runs: 2, + values: [("pack hit".to_string(), vec![60.0, 70.0])].into(), + }; + assert_eq!(profile.mean("pack hit"), Some(65.0)); + assert_eq!(profile.spread("pack hit"), 10.0); + assert_eq!(profile.mean("absent"), None); + assert_eq!(profile.spread("absent"), 0.0); +} + +#[test] +fn comparing_nothing_is_an_error() { + assert!(run(&[]).is_err()); +} + +#[test] +fn a_report_that_is_not_json_names_its_file() { + let dir = std::env::temp_dir().join(format!("memory-eval-compare-{}", std::process::id())); + std::fs::create_dir_all(&dir).expect("a temp dir"); + let path = dir.join("broken.json"); + std::fs::write(&path, "not json").expect("a temp file"); + let path = path.to_string_lossy().to_string(); + let error = run(std::slice::from_ref(&path)).expect_err("not a report"); + assert!(error.to_string().contains(&path), "{error}"); + std::fs::remove_dir_all(&dir).ok(); +} diff --git a/crates/tinymemory-integrations/examples/memory_eval/kpi.rs b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs index 3585953b..e4aef153 100644 --- a/crates/tinymemory-integrations/examples/memory_eval/kpi.rs +++ b/crates/tinymemory-integrations/examples/memory_eval/kpi.rs @@ -507,3 +507,7 @@ pub(crate) fn print(label: &str, kpis: &[Kpi]) { ); } } + +#[cfg(test)] +#[path = "kpi_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/examples/memory_eval/kpi_tests.rs b/crates/tinymemory-integrations/examples/memory_eval/kpi_tests.rs new file mode 100644 index 00000000..593c2941 --- /dev/null +++ b/crates/tinymemory-integrations/examples/memory_eval/kpi_tests.rs @@ -0,0 +1,182 @@ +//! Tests of the KPI arithmetic over hand-built scenario reports. + +use super::*; +use crate::inspect::{Captured, Spend}; +use crate::score::score; +use crate::{ScenarioReport, Synthesis}; + +/// The first probe of `scenario`, scored as a hit or a miss in `phase`. +fn probed(scenario: &'static str, phase: &'static str, hit: bool) -> ProbeResult { + let probe = crate::scenarios::all() + .into_iter() + .find(|s| s.name == scenario) + .expect("a scenario of the eval") + .probes + .remove(0); + let markdown = if hit { + format!("- {}", probe.expect.join(" ")) + } else { + "- nothing relevant".to_string() + }; + score(scenario, phase, &probe, &markdown, 10, 1.0) +} + +fn report(name: &'static str, probes: Vec, conflicts: &[&str]) -> ScenarioReport { + ScenarioReport { + name, + about: "", + writes: 0, + tool_calls: 0, + settle_ms: 0.0, + synthesis: Synthesis { + captured: Captured { + conflicts: conflicts.iter().map(|c| (*c).to_string()).collect(), + ..Captured::default() + }, + ..Synthesis::default() + }, + probes, + usage: None, + } +} + +fn value(kpis: &[Kpi], name: &str) -> Option { + kpis.iter() + .find(|kpi| kpi.name == name) + .unwrap_or_else(|| panic!("no KPI named {name}")) + .value +} + +fn usage(cost_usd: f64) -> Usage { + Usage { + calls: 4, + tokens: 1000, + cost_usd, + by_role: [( + "extraction".to_string(), + Spend { + calls: 4, + tokens: 1000, + cost_usd, + }, + )] + .into(), + } +} + +#[test] +fn synthesis_gain_is_the_hit_rate_the_belief_build_adds() { + let reports = [report( + "learnings", + vec![ + probed("learnings", "recall", false), + probed("learnings", "synthesis", true), + ], + &[], + )]; + let kpis = compute(&reports, None, &Timings::default()); + assert_eq!(value(&kpis, "pack hit (recall)"), Some(0.0)); + assert_eq!(value(&kpis, "pack hit"), Some(100.0)); + assert_eq!(value(&kpis, "synthesis gain"), Some(100.0)); + assert_eq!(value(&kpis, "lesson in pack"), Some(100.0)); +} + +#[test] +fn planted_conflicts_count_as_flagged_and_others_as_spurious() { + let reports = [ + report( + "conflicts", + vec![probed("conflicts", "synthesis", true)], + &["value_conflict open: refund settles_within [five | ten]"], + ), + report( + "brain_lookup", + vec![probed("brain_lookup", "synthesis", true)], + &["value_conflict open: office floor [2 | 3]"], + ), + // A superseded value may be flagged without counting against it. + report( + "contradictions", + vec![probed("contradictions", "synthesis", true)], + &["value_conflict open: region is [us-east-1 | eu-west-2]"], + ), + ]; + let kpis = compute(&reports, Some(&usage(0.0)), &Timings::default()); + assert_eq!(value(&kpis, "planted conflicts flagged"), Some(100.0)); + assert_eq!(value(&kpis, "spurious conflicts"), Some(1.0)); + assert_eq!(value(&kpis, "conflicts raised"), Some(3.0)); +} + +#[test] +fn a_missed_planted_conflict_lowers_the_flagged_rate() { + let reports = [report( + "conflicts", + vec![probed("conflicts", "synthesis", false)], + &[], + )]; + let kpis = compute(&reports, Some(&usage(0.0)), &Timings::default()); + assert_eq!(value(&kpis, "planted conflicts flagged"), Some(0.0)); + assert_eq!(value(&kpis, "disagreement in pack"), Some(0.0)); +} + +#[test] +fn cost_per_correct_answer_prefers_the_model_answers_and_adds_their_cost() { + let mut right = probed("surprise", "synthesis", true); + right.llm_ok = Some(true); + right.llm_tokens = 300; + right.llm_cost_usd = Some(0.5); + let mut wrong = probed("surprise", "synthesis", true); + wrong.llm_ok = Some(false); + wrong.llm_tokens = 200; + wrong.llm_cost_usd = Some(0.5); + let reports = [report("surprise", vec![right, wrong], &[])]; + let kpis = compute(&reports, Some(&usage(2.0)), &Timings::default()); + assert_eq!(value(&kpis, "answerer"), Some(1.0)); + assert_eq!(value(&kpis, "answerer tokens"), Some(500.0)); + // $2 of CortexDB plus $1 of answers over one correct model answer. + assert_eq!(value(&kpis, "per correct answer"), Some(3.0)); + assert_eq!(value(&kpis, "extraction"), Some(2.0)); + assert_eq!(value(&kpis, "surprise answered"), Some(50.0)); +} + +#[test] +fn engine_only_kpis_are_unmeasured_on_the_reference_engine() { + let reports = [report( + "conflicts", + vec![probed("conflicts", "synthesis", true)], + &[], + )]; + let kpis = compute(&reports, None, &Timings::default()); + for name in [ + "planted conflicts flagged", + "spurious conflicts", + "beliefs held", + "CortexDB models", + "per correct answer", + ] { + assert_eq!(value(&kpis, name), None, "{name}"); + } + assert_eq!(value(&kpis, "model answer"), None); +} + +#[test] +fn latency_reads_the_pre_turn_and_probe_samples() { + let mut timings = Timings::default(); + for ms in [10.0, 20.0, 30.0] { + timings.add("pre_turn (log + recall)", ms); + } + timings.add("probe pre_turn", 5.0); + timings.add("probe context.md", 7.0); + let kpis = compute(&[], None, &timings); + assert_eq!(value(&kpis, "pre_turn p50"), Some(20.0)); + assert_eq!(value(&kpis, "probe p95"), Some(7.0)); + assert_eq!(value(&kpis, "pack hit"), None); +} + +#[test] +fn values_format_by_unit() { + assert_eq!(format(62.4, Unit::Pct), "62%"); + assert_eq!(format(-4.0, Unit::Points), "-4 pp"); + assert_eq!(format(0.0123, Unit::Usd), "$0.012"); + assert_eq!(format(0.456, Unit::Score), "0.46"); +} From de8549c5340c1c24453f265038c46c4b24b4313b Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Sun, 4 Oct 2026 20:39:41 +0300 Subject: [PATCH 11/24] chore: files changed scripts/memory-eval.sh,scripts/memory-flag-sweep.sh Auto-committed-on: dragonfly Co-authored-by: Medulla --- scripts/memory-eval.sh | 14 +++-- scripts/memory-flag-sweep.sh | 113 +++++++++++++++++++++++++++++++++++ 2 files changed, 123 insertions(+), 4 deletions(-) create mode 100755 scripts/memory-flag-sweep.sh diff --git a/scripts/memory-eval.sh b/scripts/memory-eval.sh index 1d082301..51df69ce 100755 --- a/scripts/memory-eval.sh +++ b/scripts/memory-eval.sh @@ -1,12 +1,17 @@ #!/usr/bin/env bash # Runs the agent memory eval (crates/tinymemory-integrations/examples/memory_eval) # against a throwaway CortexDB from integration/cortexdb/, then tears it down. -# Reports land in target/memory-eval/