Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@

## Unreleased

- Add opt-in ECO annotations to validation reports through `--annotate-eco` and the
matching Python API keyword; mappings come from the packaged LinkML schema.
- Default validation behavior and report contents remain unchanged.

- Benchmark scenario 2 (`agent_loop.py`): an agent searches PubMed, reads papers and submits a
decision with citations; bioevidence checks each submission and returns its reasons, and the agent
may revise. Five of six models cited only what they had read; Claude Haiku 4.5 misquoted in 8 of
Expand Down
5 changes: 5 additions & 0 deletions docs/ENGINEERING.md
Original file line number Diff line number Diff line change
Expand Up @@ -57,6 +57,11 @@ version; UTC validation time; findings; and per-use reason codes. The timestamp
between runs. Hashes identify inputs/configuration; they do not sign records or prove
that a source, source hash, label, or reviewer identity is authentic.

ECO annotations are opt-in. `validate_record(..., annotate_eco=True)` and
`bioevidence validate --annotate-eco` add `evidence_eco_annotations` to the report,
derived from the packaged LinkML `ExtractionMethod` meanings. Without the option, the
report shape is unchanged.

Source artifacts require a declared version, retrieval time, and frozen hash. The optional
observed hash is compared to that hash. Validation never retrieves source bytes; without
grounders it does not calculate their hashes either. Only `build` hashes local files explicitly named in a draft; it
Expand Down
7 changes: 7 additions & 0 deletions docs/STANDARDS.md
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,13 @@ eco = {name: value.meaning for name, value in view.get_enum("ExtractionMethod").
# {'deterministic_parser': 'ECO:0000313', 'manual_curation': 'ECO:0000352', ...}
```

Validation reports expose this mapping only on request so the default report contract
stays unchanged. `bioevidence validate ... --annotate-eco` (or
`validate_record(record, annotate_eco=True)`) adds `evidence_eco_annotations`, with
one entry per evidence item containing `evidence_item_id`, `extraction_method`, and
`eco_curie`. When a record is schema-invalid or carries an unknown extraction method,
`eco_curie` is `null`.

## The ClinVar case in VA-Spec terms

In the [ClinVar case](../examples/clinvar_germline/README.md), each ClinVar submission (SCV)
Expand Down
3 changes: 2 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ classifiers = [
dependencies = [
"jsonschema>=4.23,<5",
"linkml>=1.8,<2",
"linkml-runtime>=1.1.24,<2",
"pyyaml>=6.0,<7",
]

Expand Down Expand Up @@ -87,5 +88,5 @@ warn_unused_ignores = true
warn_redundant_casts = true

[[tool.mypy.overrides]]
module = ["linkml.*"]
module = ["linkml.*", "linkml_runtime.*"]
ignore_missing_imports = true
13 changes: 12 additions & 1 deletion src/bioevidence_validator/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,11 @@ def parser() -> argparse.ArgumentParser:
validate.add_argument("--output", type=Path)
validate.add_argument("--schema", type=Path, default=default_schema_path())
validate.add_argument("--profile", default="general", help="Built-in profile name or YAML file path")
validate.add_argument(
"--annotate-eco",
action="store_true",
help="Include extraction-method ECO CURIEs for evidence items in the report",
)
validate.add_argument("--snapshot-dir", type=Path,
help="Directory of source snapshots named by SHA-256; recompute source hashes from them, "
"and check cited publications if it holds a literature.json from `ground`")
Expand Down Expand Up @@ -175,7 +180,13 @@ def _run(args) -> int:
grounders.append(SourceBytesGrounder(SnapshotStore.from_directory(args.snapshot_dir)))
if (args.snapshot_dir / CATALOG).exists():
grounders.append(LiteratureGrounder.from_directory(args.snapshot_dir))
report = validate_record(record, schema_path=args.schema, profile=args.profile, grounders=grounders)
report = validate_record(
record,
schema_path=args.schema,
profile=args.profile,
grounders=grounders,
annotate_eco=args.annotate_eco,
)
rendered = json.dumps(report, indent=2) + "\n"
if args.output:
_write_json(args.output, report)
Expand Down
43 changes: 40 additions & 3 deletions src/bioevidence_validator/engine.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
from __future__ import annotations

import functools
import hashlib
import json
from collections.abc import Sequence
Expand All @@ -10,6 +11,7 @@

from jsonschema import Draft202012Validator, FormatChecker
from linkml.generators.jsonschemagen import JsonSchemaGenerator
from linkml_runtime.utils.schemaview import SchemaView

from . import __version__
from .config import load_mapping, nonblank
Expand All @@ -36,6 +38,37 @@ def default_schema_path() -> Path:
return _resource_path("schema", "bioevidence_core.yaml")


@functools.cache
def _eco_meanings(schema_path: str) -> dict[str, str]:
enum = SchemaView(schema_path).get_enum("ExtractionMethod")
if enum is None:
raise ValueError("LinkML schema is missing ExtractionMethod")
return {name: str(v.meaning) for name, v in enum.permissible_values.items() if v.meaning}


def _evidence_eco_annotations(record: Any) -> list[dict[str, Any]]:
meanings = _eco_meanings(str(default_schema_path()))
items = record.get("evidence_items") if isinstance(record, dict) else None
if not isinstance(items, list):
return []

annotations = []
for item in items:
if not isinstance(item, dict):
continue
item_id = item.get("id")
method = item.get("extraction_method")
if isinstance(item_id, str) and isinstance(method, str):
annotations.append(
{
"evidence_item_id": item_id,
"extraction_method": method,
"eco_curie": meanings.get(method),
}
)
return annotations


def generate_json_schema(schema_path: Path | None = None) -> dict[str, Any]:
schema_path = schema_path or default_schema_path()
try:
Expand Down Expand Up @@ -275,7 +308,7 @@ def __init__(self, *, profile: str | Path = "general", schema_path: Path | None
self.schema_sha256 = (self.schema_sources[0]["sha256"] if len(paths) == 1 else
sha256_bytes(json.dumps(self._schemas, sort_keys=True, separators=(",", ":")).encode()))

def validate(self, record: Any) -> dict[str, Any]:
def validate(self, record: Any, *, annotate_eco: bool = False) -> dict[str, Any]:
canonical = json.dumps(record, sort_keys=True, separators=(",", ":"), allow_nan=False).encode()
requested = record.get("requested_uses") if isinstance(record, dict) else None
uses = list(dict.fromkeys(use for use in requested if isinstance(use, str))) if isinstance(requested, list) else []
Expand Down Expand Up @@ -307,9 +340,13 @@ def validate(self, record: Any) -> dict[str, Any]:
}
if self.grounders: # reports without grounding keep their exact previous shape
report["grounding"] = [grounder.name for grounder in self.grounders]
if annotate_eco:
report["evidence_eco_annotations"] = _evidence_eco_annotations(record)
return report


def validate_record(record: Any, *, profile: str | Path = "general", schema_path: Path | None = None,
grounders: Sequence[Any] = ()) -> dict[str, Any]:
return RecordValidator(profile=profile, schema_path=schema_path, grounders=grounders).validate(record)
grounders: Sequence[Any] = (), annotate_eco: bool = False) -> dict[str, Any]:
return RecordValidator(profile=profile, schema_path=schema_path, grounders=grounders).validate(
record, annotate_eco=annotate_eco
)
31 changes: 31 additions & 0 deletions tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,37 @@ def test_cli_writes_audit_report(tmp_path):
assert len(report["input_sha256"]) == 64


def test_cli_eco_annotations_are_opt_in(tmp_path):
source = ROOT / "examples" / "general" / "curated_assertion.json"
plain_output = tmp_path / "plain.json"
annotated_output = tmp_path / "annotated.json"

assert main(["validate", str(source), "--output", str(plain_output)]) == 0
assert (
main(
[
"validate",
str(source),
"--annotate-eco",
"--output",
str(annotated_output),
]
)
== 0
)

plain = json.loads(plain_output.read_text(encoding="utf-8"))
annotated = json.loads(annotated_output.read_text(encoding="utf-8"))
assert "evidence_eco_annotations" not in plain
assert annotated["evidence_eco_annotations"] == [
{
"evidence_item_id": "bioev:item-1",
"extraction_method": "manual_curation",
"eco_curie": "ECO:0000352",
}
]


def test_cli_exit_code_distinguishes_rejection(tmp_path):
output = tmp_path / "report.json"
code = main([
Expand Down
69 changes: 68 additions & 1 deletion tests/test_standards.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,15 @@
"""Keep the machine-readable ECO meanings, the documentation and the compiled schema consistent."""
import json
import re
from pathlib import Path

from linkml_runtime.utils.schemaview import SchemaView

from bioevidence_validator.engine import default_schema_path, generate_json_schema
from bioevidence_validator.engine import (
default_schema_path,
generate_json_schema,
validate_record,
)

ROOT = Path(__file__).resolve().parents[1]
EXPECTED = {"deterministic_parser": "ECO:0000313", "manual_curation": "ECO:0000352",
Expand Down Expand Up @@ -32,3 +37,65 @@ def test_standards_doc_matches_the_schema():
def test_annotations_do_not_change_validation():
compiled = generate_json_schema()["$defs"]["ExtractionMethod"]
assert compiled == {"description": "", "enum": list(EXPECTED), "title": "ExtractionMethod", "type": "string"}


def test_eco_report_annotations_are_opt_in_and_schema_derived():
record = json.loads(
(ROOT / "examples/general/curated_assertion.json").read_text(encoding="utf-8")
)
base = record["evidence_items"][0]
record["evidence_items"] = [
{**base, "id": f"bioev:{method}", "extraction_method": method}
for method in EXPECTED
]
record["statement"]["evidence_lines"][0]["evidence_item_ids"] = [
"bioev:manual_curation"
]

plain = validate_record(record)
annotated = validate_record(record, annotate_eco=True)

assert "evidence_eco_annotations" not in plain
assert annotated["evidence_eco_annotations"] == [
{
"evidence_item_id": f"bioev:{method}",
"extraction_method": method,
"eco_curie": curie,
}
for method, curie in EXPECTED.items()
]


def test_eco_report_annotations_unknown_method_is_null():
record = {"evidence_items": [{"id": "bioev:unknown", "extraction_method": "custom_nlp_extractor"}]}
annotated = validate_record(record, annotate_eco=True)
assert annotated["evidence_eco_annotations"] == [
{
"evidence_item_id": "bioev:unknown",
"extraction_method": "custom_nlp_extractor",
"eco_curie": None,
}
]


def test_eco_meanings_caches_schema_view():
from bioevidence_validator.engine import _eco_meanings, default_schema_path

_eco_meanings.cache_clear()
path = str(default_schema_path())
res1 = _eco_meanings(path)
res2 = _eco_meanings(path)
assert res1 is res2
assert _eco_meanings.cache_info().hits >= 1


def test_eco_report_annotations_ordering_with_grounding():
class DummyGrounder:
name = "dummy_grounder"

record = {"evidence_items": [{"id": "bioev:test", "extraction_method": "manual_curation"}]}
report = validate_record(record, grounders=[DummyGrounder()], annotate_eco=True)
assert report["grounding"] == ["dummy_grounder"]
assert "evidence_eco_annotations" in report
keys = list(report.keys())
assert keys.index("grounding") < keys.index("evidence_eco_annotations")
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading