diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ef414fa..991bc79 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -38,8 +38,9 @@ jobs: - name: Install built wheel run: | python -m venv "$RUNNER_TEMP/package-venv" - "$RUNNER_TEMP/package-venv/bin/python" -m pip install dist/evidencematrix-0.1.0-py3-none-any.whl + "$RUNNER_TEMP/package-venv/bin/python" -m pip install dist/evidencematrix-0.1.1-py3-none-any.whl entitylinkage==0.1.0 "$RUNNER_TEMP/package-venv/bin/evidencematrix" validate examples/basic + PATH="$RUNNER_TEMP/package-venv/bin:$PATH" bash examples/entitylinkage-coverage/run-example.sh "$RUNNER_TEMP/entitylinkage-coverage-example" - name: CLI smoke test shell: bash run: | diff --git a/.gitignore b/.gitignore index 9219a99..c034817 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,4 @@ +.worktrees/ .venv/ .venv-*/ .pkg-tmp/ diff --git a/MANIFEST.in b/MANIFEST.in index 992b414..cd8dfca 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -1,2 +1,2 @@ include LICENSE README.md -recursive-include examples *.yaml *.md +recursive-include examples *.yaml *.md *.json *.csv *.py *.sh diff --git a/README.md b/README.md index d0d77e1..2039373 100644 --- a/README.md +++ b/README.md @@ -120,6 +120,10 @@ GitHub Actions runs lint and format checks, tests, package builds, and CLI smoke The public [TWDisaster dataset](https://github.com/KageRyo/TWDisaster) publishes positive event-source links but no expected-source eligibility matrix. The [dogfood assessment](examples/twdisaster/README.md) explains why converting absent links into gaps would invent facts and why the current release cannot provide a useful matrix of expected coverage. +## EntityLinkage integration + +The [synthetic integration example](examples/entitylinkage-coverage/README.md) demonstrates the standalone flow from external records through [EntityLinkage](https://github.com/KageRyo/EntityLinkage) review to an EvidenceMatrix coverage matrix. It preserves ambiguous and unresolved records for review and uses only explicit context plus resolved links to declare coverage. CI runs the example with the built EvidenceMatrix wheel and the pinned EntityLinkage release. + ## License EvidenceMatrix is distributed under the [Apache License 2.0](LICENSE). diff --git a/examples/entitylinkage-coverage/README.md b/examples/entitylinkage-coverage/README.md new file mode 100644 index 0000000..7df890e --- /dev/null +++ b/examples/entitylinkage-coverage/README.md @@ -0,0 +1,51 @@ +# EntityLinkage to EvidenceMatrix + +This fully synthetic example links fictional external catalog records to a small entity list, preserves the decisions that need review, and turns only resolved links into EvidenceMatrix coverage. It uses no TAG data or network source. + +## Run the complete flow + +Install the two standalone tools: + +```bash +python -m pip install 'entitylinkage==0.1.0' 'evidencematrix==0.1.1' +bash examples/entitylinkage-coverage/run-example.sh +``` + +The script creates a temporary output directory and prints its path when it finishes. Pass a path to choose the output location: + +```bash +bash examples/entitylinkage-coverage/run-example.sh ./entitylinkage-coverage-output +``` + +For each case, the script validates and links the records with EntityLinkage, adapts its versioned `linkage.json` output, then validates, builds, summarizes, and audits the generated EvidenceMatrix manifest. The audit returns exit code 1 because the example intentionally retains unresolved coverage findings; the script checks that expected CI-gate result. + +## Review before changing coverage + +`entitylinkage.yaml` produces these record decisions: + +| Record | EntityLinkage result | Meaning | +| --- | --- | --- | +| `record-a-001` | resolved to `entity-1` | Exact fictional catalog name. | +| `record-a-002` | ambiguous between `entity-2` and `entity-4` | Shared alias; it remains in `review.csv`. | +| `record-a-003` | unresolved | No declared identity candidate; it remains in `review.csv`. | +| `record-b-001` | resolved to `entity-3` | Exact fictional archive name. | +| `record-b-002` | not applicable | The record is outside the archive's declared scope. | + +The explicit coverage context is separate from the matching configuration. It declares the source inventory, two known mapping gaps, and one entity-source relationship that is outside scope. Pairs with no declaration or resolved record stay `unknown` when their source is available. A source marked `unavailable` or `not_in_release` keeps that meaning across all entities. + +`review.csv` retains ambiguous, unresolved, and record-level not-applicable decisions, including the normalized name, candidate IDs, reason codes, and rule evidence. The adapter never promotes ambiguous or unresolved candidates into coverage. `resolved-records.csv` preserves each source record behind a supported relationship. When several records resolve to the same entity and source, the manifest has one supported relationship while the sidecar keeps every record. + +## Apply a reviewed decision + +`entitylinkage-reviewed.yaml` adds a manual override for `record-a-002`, after a human review selects `entity-4`. The adapter promotes only the already-declared `entity-4` / `catalog-a` `mapping_gap` to `supported`. The `entity-2` / `catalog-a` gap remains, and the unresolved record is still unresolved. + +| EvidenceMatrix status | Before review | After review | +| --- | ---: | ---: | +| `supported` | 2 | 3 | +| `mapping_gap` | 2 | 1 | +| `not_applicable` | 1 | 1 | +| `unknown` | 3 | 3 | +| `source_unavailable` | 4 | 4 | +| `source_not_in_release` | 4 | 4 | + +Both cases contain 16 entity-source relationships. The adapter is a standalone Python standard-library script and accepts the `entitylinkage-results-v1` JSON contract; it does not add a runtime dependency between either package. diff --git a/examples/entitylinkage-coverage/adapter.py b/examples/entitylinkage-coverage/adapter.py new file mode 100644 index 0000000..72b1322 --- /dev/null +++ b/examples/entitylinkage-coverage/adapter.py @@ -0,0 +1,449 @@ +#!/usr/bin/env python3 +"""Adapt deterministic EntityLinkage results to an EvidenceMatrix manifest.""" + +from __future__ import annotations + +import argparse +import csv +import json +import sys +from collections import defaultdict +from pathlib import Path +from typing import Any + +_LINKAGE_SCHEMA = "entitylinkage-results-v1" +_LINKAGE_STATUSES = {"resolved", "ambiguous", "unresolved", "not_applicable"} +_AVAILABILITIES = {"available", "unavailable", "not_in_release", "unknown"} +_DECLARED_STATUSES = {"mapping_gap", "not_applicable"} +_REVIEW_FIELDS = ( + "record_id", + "source_id", + "status", + "normalized_record_name", + "candidate_entity_ids", + "reason_codes", + "evidence", +) +_RESOLVED_FIELDS = ("entity_id", "source_id", "record_id") + + +class AdapterError(ValueError): + """Raised when an input cannot be translated without changing its meaning.""" + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise AdapterError(f"duplicate JSON key: {key!r}") + result[key] = value + return result + + +def _read_json(path: Path, label: str) -> Any: + try: + return json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=_unique_object) + except (OSError, UnicodeError, json.JSONDecodeError, AdapterError) as error: + raise AdapterError(f"cannot read {label} {path}: {error}") from error + + +def _object(value: Any, label: str, fields: set[str]) -> dict[str, Any]: + if not isinstance(value, dict) or any(type(key) is not str for key in value): + raise AdapterError(f"{label} must be a JSON object") + unexpected = sorted(value.keys() - fields) + missing = sorted(fields - value.keys()) + if unexpected: + raise AdapterError(f"{label} has unknown field(s): {', '.join(unexpected)}") + if missing: + raise AdapterError(f"{label} is missing field(s): {', '.join(missing)}") + return value + + +def _string(value: Any, label: str) -> str: + if type(value) is not str or not value or value.strip() != value: + raise AdapterError(f"{label} must be a non-empty string without surrounding whitespace") + return value + + +def _string_list(value: Any, label: str) -> list[str]: + if not isinstance(value, list): + raise AdapterError(f"{label} must be a list") + items = [_string(item, f"{label}[{index}]") for index, item in enumerate(value)] + if len(items) != len(set(items)): + raise AdapterError(f"{label} must not contain duplicates") + return items + + +def _read_record_sources(path: Path) -> dict[str, str]: + try: + with path.open(encoding="utf-8", newline="") as stream: + reader = csv.DictReader(stream) + if reader.fieldnames != ["record_id", "source_id"]: + raise AdapterError("record-source CSV header must be exactly record_id,source_id") + mapping: dict[str, str] = {} + for line, row in enumerate(reader, start=2): + if None in row or set(row) != {"record_id", "source_id"}: + raise AdapterError( + f"record-source CSV line {line} has an invalid number of columns" + ) + record_id = _string(row["record_id"], f"record-source CSV line {line} record_id") + source_id = _string(row["source_id"], f"record-source CSV line {line} source_id") + if record_id in mapping: + raise AdapterError(f"duplicate record-to-source mapping: {record_id!r}") + mapping[record_id] = source_id + return mapping + except (OSError, UnicodeError, csv.Error) as error: + raise AdapterError(f"cannot read record-source CSV {path}: {error}") from error + + +def _load_context( + path: Path, +) -> tuple[list[str], dict[str, str], dict[tuple[str, str], dict[str, str]]]: + context = _object( + _read_json(path, "coverage context"), + "coverage context", + {"version", "entities", "sources", "coverage"}, + ) + if type(context["version"]) is not int or context["version"] != 1: + raise AdapterError("coverage context version must be the integer 1") + if not isinstance(context["entities"], list): + raise AdapterError("coverage context entities must be a list") + if not isinstance(context["sources"], list): + raise AdapterError("coverage context sources must be a list") + if not isinstance(context["coverage"], list): + raise AdapterError("coverage context coverage must be a list") + + entity_ids: list[str] = [] + for index, value in enumerate(context["entities"]): + item = _object(value, f"coverage context entities[{index}]", {"id"}) + entity_ids.append(_string(item["id"], f"coverage context entities[{index}].id")) + if len(entity_ids) != len(set(entity_ids)): + raise AdapterError("coverage context contains duplicate entity IDs") + + sources: dict[str, str] = {} + for index, value in enumerate(context["sources"]): + item = _object(value, f"coverage context sources[{index}]", {"id", "availability"}) + source_id = _string(item["id"], f"coverage context sources[{index}].id") + availability = _string( + item["availability"], f"coverage context sources[{index}].availability" + ) + if availability not in _AVAILABILITIES: + raise AdapterError( + f"source {source_id!r} has unsupported availability {availability!r}" + ) + if source_id in sources: + raise AdapterError(f"coverage context contains duplicate source ID: {source_id!r}") + sources[source_id] = availability + + declarations: dict[tuple[str, str], dict[str, str]] = {} + for index, value in enumerate(context["coverage"]): + item = _object( + value, + f"coverage context coverage[{index}]", + {"entity", "source", "status", "reason"}, + ) + entity_id = _string(item["entity"], f"coverage context coverage[{index}].entity") + source_id = _string(item["source"], f"coverage context coverage[{index}].source") + status = _string(item["status"], f"coverage context coverage[{index}].status") + reason = _string(item["reason"], f"coverage context coverage[{index}].reason") + if entity_id not in entity_ids: + raise AdapterError(f"coverage context references unknown entity: {entity_id!r}") + if source_id not in sources: + raise AdapterError(f"coverage context references unknown source: {source_id!r}") + if status not in _DECLARED_STATUSES: + raise AdapterError( + f"coverage context status must be one of: {', '.join(sorted(_DECLARED_STATUSES))}" + ) + if sources[source_id] in {"unavailable", "not_in_release"}: + raise AdapterError( + f"coverage declaration for {entity_id!r}/{source_id!r} conflicts with " + f"source availability {sources[source_id]!r}" + ) + pair = (entity_id, source_id) + if pair in declarations: + raise AdapterError( + f"duplicate coverage context relationship: {entity_id!r}/{source_id!r}" + ) + declarations[pair] = {"status": status, "reason": reason} + + return entity_ids, sources, declarations + + +def _load_linkage(path: Path, entity_ids: set[str]) -> list[dict[str, Any]]: + payload = _object( + _read_json(path, "EntityLinkage result"), + "EntityLinkage result", + {"schema_version", "tool_version", "unicode_version", "results"}, + ) + if payload["schema_version"] != _LINKAGE_SCHEMA: + raise AdapterError( + f"unsupported EntityLinkage schema_version: {payload['schema_version']!r}" + ) + _string(payload["tool_version"], "EntityLinkage tool_version") + _string(payload["unicode_version"], "EntityLinkage unicode_version") + if not isinstance(payload["results"], list): + raise AdapterError("EntityLinkage results must be a list") + + results: list[dict[str, Any]] = [] + seen_ids: set[str] = set() + fields = { + "record_id", + "status", + "entity_id", + "candidate_entity_ids", + "reason_codes", + "normalized_record_name", + "evidence", + } + for index, value in enumerate(payload["results"]): + item = _object(value, f"EntityLinkage results[{index}]", fields) + record_id = _string(item["record_id"], f"EntityLinkage results[{index}].record_id") + status = _string(item["status"], f"EntityLinkage results[{index}].status") + if status not in _LINKAGE_STATUSES: + raise AdapterError( + f"EntityLinkage record {record_id!r} has unsupported status {status!r}" + ) + if record_id in seen_ids: + raise AdapterError(f"duplicate EntityLinkage record ID: {record_id!r}") + seen_ids.add(record_id) + + entity_id = item["entity_id"] + if status == "resolved": + entity_id = _string(entity_id, f"EntityLinkage record {record_id!r} entity_id") + if entity_id not in entity_ids: + raise AdapterError( + f"EntityLinkage record {record_id!r} resolved to unknown entity {entity_id!r}" + ) + elif entity_id is not None: + raise AdapterError( + f"EntityLinkage record {record_id!r} with status {status!r} " + "must not declare entity_id" + ) + + candidates = _string_list( + item["candidate_entity_ids"], + f"EntityLinkage record {record_id!r} candidate_entity_ids", + ) + unknown_candidates = sorted(set(candidates) - entity_ids) + if unknown_candidates: + raise AdapterError( + f"EntityLinkage record {record_id!r} has unknown candidate entity IDs: " + + ", ".join(unknown_candidates) + ) + reason_codes = _string_list( + item["reason_codes"], f"EntityLinkage record {record_id!r} reason_codes" + ) + if status == "resolved" and candidates != [entity_id]: + raise AdapterError( + f"resolved EntityLinkage record {record_id!r} must list only its resolved entity " + "in candidate_entity_ids" + ) + if status == "ambiguous" and len(candidates) < 2: + raise AdapterError( + f"ambiguous EntityLinkage record {record_id!r} must list at least two candidates" + ) + if status in {"unresolved", "not_applicable"} and candidates: + raise AdapterError( + f"EntityLinkage record {record_id!r} with status {status!r} " + "must not list candidates" + ) + if status == "not_applicable" and not reason_codes: + raise AdapterError( + f"not_applicable EntityLinkage record {record_id!r} must include a reason code" + ) + normalized_name = item["normalized_record_name"] + if normalized_name is not None and type(normalized_name) is not str: + raise AdapterError( + f"EntityLinkage record {record_id!r} normalized_record_name " + "must be a string or null" + ) + if not isinstance(item["evidence"], list) or any( + not isinstance(evidence, dict) for evidence in item["evidence"] + ): + raise AdapterError( + f"EntityLinkage record {record_id!r} evidence must be a list of objects" + ) + results.append( + { + **item, + "record_id": record_id, + "status": status, + "entity_id": entity_id, + "candidate_entity_ids": candidates, + "reason_codes": reason_codes, + } + ) + return sorted(results, key=lambda item: item["record_id"]) + + +def _canonical_json(value: Any) -> str: + return json.dumps( + value, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False + ) + + +def _yaml_string(value: str) -> str: + return json.dumps(value, ensure_ascii=False) + + +def _coverage_yaml( + entity_ids: list[str], + source_availability: dict[str, str], + declarations: dict[tuple[str, str], dict[str, str]], + resolved_by_pair: dict[tuple[str, str], list[str]], +) -> str: + rows: dict[tuple[str, str], dict[str, str]] = { + pair: dict(declaration) for pair, declaration in declarations.items() + } + for pair in resolved_by_pair: + rows[pair] = {"status": "supported", "reason": ""} + + lines = ["version: 1", "", "entities:"] + lines.extend(f" - id: {_yaml_string(entity_id)}" for entity_id in sorted(entity_ids)) + lines.extend(["", "sources:"]) + for source_id, availability in sorted(source_availability.items()): + lines.extend( + [ + f" - id: {_yaml_string(source_id)}", + f" availability: {availability}", + ] + ) + lines.extend(["", "coverage:"]) + if not rows: + lines[-1] = "coverage: []" + else: + for (entity_id, source_id), declaration in sorted(rows.items()): + lines.extend( + [ + f" - entity: {_yaml_string(entity_id)}", + f" source: {_yaml_string(source_id)}", + f" status: {declaration['status']}", + ] + ) + if declaration["reason"]: + lines.append(f" reason: {_yaml_string(declaration['reason'])}") + return "\n".join(lines) + "\n" + + +def _review_csv(review_rows: list[dict[str, Any]]) -> bytes: + import io + + stream = io.StringIO(newline="") + writer = csv.DictWriter(stream, fieldnames=_REVIEW_FIELDS, lineterminator="\n") + writer.writeheader() + for item in review_rows: + writer.writerow( + { + "record_id": item["record_id"], + "source_id": item["source_id"], + "status": item["status"], + "normalized_record_name": item["normalized_record_name"] or "", + "candidate_entity_ids": _canonical_json(item["candidate_entity_ids"]), + "reason_codes": _canonical_json(item["reason_codes"]), + "evidence": _canonical_json(item["evidence"]), + } + ) + return stream.getvalue().encode("utf-8") + + +def _resolved_csv(rows: list[dict[str, str]]) -> bytes: + import io + + stream = io.StringIO(newline="") + writer = csv.DictWriter(stream, fieldnames=_RESOLVED_FIELDS, lineterminator="\n") + writer.writeheader() + writer.writerows(rows) + return stream.getvalue().encode("utf-8") + + +def adapt( + linkage_path: Path, + record_sources_path: Path, + context_path: Path, +) -> dict[str, bytes]: + entity_ids, sources, declarations = _load_context(context_path) + results = _load_linkage(linkage_path, set(entity_ids)) + record_sources = _read_record_sources(record_sources_path) + + result_ids = {item["record_id"] for item in results} + mapping_ids = set(record_sources) + missing_mappings = sorted(result_ids - mapping_ids) + unknown_mappings = sorted(mapping_ids - result_ids) + if missing_mappings: + raise AdapterError("missing source mapping for record IDs: " + ", ".join(missing_mappings)) + if unknown_mappings: + raise AdapterError( + "source mapping references unknown record IDs: " + ", ".join(unknown_mappings) + ) + + resolved_by_pair: dict[tuple[str, str], list[str]] = defaultdict(list) + resolved_rows: list[dict[str, str]] = [] + review_rows: list[dict[str, Any]] = [] + for result in results: + record_id = result["record_id"] + source_id = record_sources[record_id] + if source_id not in sources: + raise AdapterError(f"record {record_id!r} maps to unknown source ID {source_id!r}") + status = result["status"] + if status == "resolved": + entity_id = result["entity_id"] + pair = (entity_id, source_id) + resolved_by_pair[pair].append(record_id) + resolved_rows.append( + {"entity_id": entity_id, "source_id": source_id, "record_id": record_id} + ) + else: + review_rows.append({**result, "source_id": source_id}) + + for pair in resolved_by_pair: + if pair in declarations and declarations[pair]["status"] == "not_applicable": + raise AdapterError( + f"resolved evidence for {pair[0]!r}/{pair[1]!r} conflicts with not_applicable" + ) + if sources[pair[1]] in {"unavailable", "not_in_release"}: + raise AdapterError( + f"resolved evidence for {pair[0]!r}/{pair[1]!r} conflicts with " + f"source availability {sources[pair[1]]!r}" + ) + + resolved_rows.sort(key=lambda item: (item["entity_id"], item["source_id"], item["record_id"])) + review_rows.sort(key=lambda item: item["record_id"]) + return { + "coverage.yaml": _coverage_yaml(entity_ids, sources, declarations, resolved_by_pair).encode( + "utf-8" + ), + "review.csv": _review_csv(review_rows), + "resolved-records.csv": _resolved_csv(resolved_rows), + } + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--linkage", required=True, type=Path, help="EntityLinkage linkage.json") + parser.add_argument( + "--record-sources", required=True, type=Path, help="record_id-to-source_id CSV" + ) + parser.add_argument("--context", required=True, type=Path, help="coverage context JSON") + parser.add_argument( + "--output-dir", required=True, type=Path, help="directory for adapter outputs" + ) + return parser + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + try: + outputs = adapt(args.linkage, args.record_sources, args.context) + args.output_dir.mkdir(parents=True, exist_ok=True) + for name, content in outputs.items(): + temporary = args.output_dir / f".{name}.tmp" + temporary.write_bytes(content) + temporary.replace(args.output_dir / name) + except (AdapterError, OSError, ValueError) as error: + print(f"entitylinkage-coverage-adapter: error: {error}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/examples/entitylinkage-coverage/coverage-context.json b/examples/entitylinkage-coverage/coverage-context.json new file mode 100644 index 0000000..aa7ade0 --- /dev/null +++ b/examples/entitylinkage-coverage/coverage-context.json @@ -0,0 +1,35 @@ +{ + "version": 1, + "entities": [ + {"id": "entity-1"}, + {"id": "entity-2"}, + {"id": "entity-3"}, + {"id": "entity-4"} + ], + "sources": [ + {"id": "catalog-a", "availability": "available"}, + {"id": "catalog-b", "availability": "available"}, + {"id": "retired-register", "availability": "unavailable"}, + {"id": "spring-supplement", "availability": "not_in_release"} + ], + "coverage": [ + { + "entity": "entity-2", + "source": "catalog-a", + "status": "mapping_gap", + "reason": "shared_name_requires_review" + }, + { + "entity": "entity-3", + "source": "catalog-a", + "status": "not_applicable", + "reason": "archive_is_outside_catalog_scope" + }, + { + "entity": "entity-4", + "source": "catalog-a", + "status": "mapping_gap", + "reason": "shared_name_requires_review" + } + ] +} diff --git a/examples/entitylinkage-coverage/entitylinkage-reviewed.yaml b/examples/entitylinkage-coverage/entitylinkage-reviewed.yaml new file mode 100644 index 0000000..87535ef --- /dev/null +++ b/examples/entitylinkage-coverage/entitylinkage-reviewed.yaml @@ -0,0 +1,43 @@ +version: 1 + +entities: + - id: entity-1 + name: North Basin Observatory + - id: entity-2 + name: South Basin Observatory + aliases: + - Highland Field Station + - id: entity-3 + name: Central Valley Archive + - id: entity-4 + name: Ridge Survey Office + aliases: + - Highland Field Station + +records: + - id: record-a-001 + name: North Basin Observatory + - id: record-a-002 + name: Highland Field Station + - id: record-a-003 + name: Unlisted Coastal Notice + - id: record-b-001 + name: Central Valley Archive + - id: record-b-002 + applicable: false + not_applicable_reason: outside_archive_scope + +matching: + normalization: + case_fold: true + punctuation: preserve + candidate_rules: + - id: publication-name + type: normalized_name + apply_overrides: true + +overrides: + - record_id: record-a-002 + status: resolved + entity_id: entity-4 + reason: reviewed_source_catalog_entry diff --git a/examples/entitylinkage-coverage/entitylinkage.yaml b/examples/entitylinkage-coverage/entitylinkage.yaml new file mode 100644 index 0000000..ef966f6 --- /dev/null +++ b/examples/entitylinkage-coverage/entitylinkage.yaml @@ -0,0 +1,37 @@ +version: 1 + +entities: + - id: entity-1 + name: North Basin Observatory + - id: entity-2 + name: South Basin Observatory + aliases: + - Highland Field Station + - id: entity-3 + name: Central Valley Archive + - id: entity-4 + name: Ridge Survey Office + aliases: + - Highland Field Station + +records: + - id: record-a-001 + name: North Basin Observatory + - id: record-a-002 + name: Highland Field Station + - id: record-a-003 + name: Unlisted Coastal Notice + - id: record-b-001 + name: Central Valley Archive + - id: record-b-002 + applicable: false + not_applicable_reason: outside_archive_scope + +matching: + normalization: + case_fold: true + punctuation: preserve + candidate_rules: + - id: publication-name + type: normalized_name + apply_overrides: false diff --git a/examples/entitylinkage-coverage/record-sources.csv b/examples/entitylinkage-coverage/record-sources.csv new file mode 100644 index 0000000..6a2f6e1 --- /dev/null +++ b/examples/entitylinkage-coverage/record-sources.csv @@ -0,0 +1,6 @@ +record_id,source_id +record-a-001,catalog-a +record-a-002,catalog-a +record-a-003,catalog-a +record-b-001,catalog-b +record-b-002,catalog-b diff --git a/examples/entitylinkage-coverage/run-example.sh b/examples/entitylinkage-coverage/run-example.sh new file mode 100755 index 0000000..fda81f9 --- /dev/null +++ b/examples/entitylinkage-coverage/run-example.sh @@ -0,0 +1,93 @@ +#!/usr/bin/env bash +set -euo pipefail + +example_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +if [[ $# -gt 1 ]]; then + printf 'usage: %s [OUTPUT_DIR]\n' "$0" >&2 + exit 2 +fi + +if [[ $# -eq 1 ]]; then + output_dir="$1" +else + output_dir="$(mktemp -d "${TMPDIR:-/tmp}/entitylinkage-coverage.XXXXXX")" +fi +mkdir -p "$output_dir" + +run_case() { + local label="$1" + local config="$2" + local case_dir="$output_dir/$label" + local linkage_dir="$case_dir/entitylinkage" + local adapter_dir="$case_dir/adapter" + local report_dir="$case_dir/evidencematrix" + + entitylinkage validate "$config" + entitylinkage link "$config" --output "$linkage_dir" + python "$example_dir/adapter.py" \ + --linkage "$linkage_dir/linkage.json" \ + --record-sources "$example_dir/record-sources.csv" \ + --context "$example_dir/coverage-context.json" \ + --output-dir "$adapter_dir" + evidencematrix validate "$adapter_dir/coverage.yaml" + evidencematrix build "$adapter_dir/coverage.yaml" --output "$report_dir" + evidencematrix summary "$adapter_dir/coverage.yaml" --format json > "$case_dir/summary.json" + + local audit_status=0 + if evidencematrix audit "$adapter_dir/coverage.yaml" --format json > "$case_dir/audit.json"; then + audit_status=0 + else + audit_status=$? + fi + if [[ "$audit_status" -ne 1 ]]; then + printf 'expected EvidenceMatrix audit exit code 1 for %s, got %s\n' \ + "$label" "$audit_status" >&2 + return 1 + fi +} + +run_case before-review "$example_dir/entitylinkage.yaml" +run_case after-review "$example_dir/entitylinkage-reviewed.yaml" + +python - "$output_dir" <<'PY' +import csv +import json +import sys +from pathlib import Path + +root = Path(sys.argv[1]) +expected = { + "before-review": {"supported": 2, "mapping_gap": 2, "not_applicable": 1, "unknown": 3}, + "after-review": {"supported": 3, "mapping_gap": 1, "not_applicable": 1, "unknown": 3}, +} +for case, status_counts in expected.items(): + case_dir = root / case + summary = json.loads((case_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["relationship_count"] == 16, (case, summary) + for status, count in status_counts.items(): + assert summary["status_counts"][status] == count, (case, status, summary) + +before = json.loads((root / "before-review/entitylinkage/linkage.json").read_text(encoding="utf-8")) +after = json.loads((root / "after-review/entitylinkage/linkage.json").read_text(encoding="utf-8")) +before_statuses = {item["record_id"]: item["status"] for item in before["results"]} +after_statuses = {item["record_id"]: item["status"] for item in after["results"]} +assert before_statuses["record-a-002"] == "ambiguous" +assert after_statuses["record-a-002"] == "resolved" +assert after_statuses["record-a-003"] == "unresolved" +assert after_statuses["record-b-002"] == "not_applicable" + +with (root / "before-review/adapter/review.csv").open(encoding="utf-8", newline="") as stream: + before_reviews = {row["record_id"]: row["status"] for row in csv.DictReader(stream)} +with (root / "after-review/adapter/review.csv").open(encoding="utf-8", newline="") as stream: + after_reviews = {row["record_id"]: row["status"] for row in csv.DictReader(stream)} +assert before_reviews == { + "record-a-002": "ambiguous", + "record-a-003": "unresolved", + "record-b-002": "not_applicable", +} +assert after_reviews == { + "record-a-003": "unresolved", + "record-b-002": "not_applicable", +} +print(f"integration example passed; reports are in {root}") +PY diff --git a/pyproject.toml b/pyproject.toml index cb4bd9c..e22992e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "evidencematrix" -version = "0.1.0" +version = "0.1.1" description = "A deterministic source-coverage audit tool for multi-source datasets" readme = "README.md" requires-python = ">=3.11" diff --git a/src/evidencematrix/__init__.py b/src/evidencematrix/__init__.py index 7f93dde..8a9b294 100644 --- a/src/evidencematrix/__init__.py +++ b/src/evidencematrix/__init__.py @@ -15,4 +15,4 @@ "validate_document", ] -__version__ = "0.1.0" +__version__ = "0.1.1" diff --git a/tests/test_entitylinkage_adapter.py b/tests/test_entitylinkage_adapter.py new file mode 100644 index 0000000..6fdd017 --- /dev/null +++ b/tests/test_entitylinkage_adapter.py @@ -0,0 +1,324 @@ +from __future__ import annotations + +import csv +import json +import subprocess +import sys +from pathlib import Path + +from evidencematrix.matrix import build_matrix, summarize_matrix +from evidencematrix.parser import load_manifest + +EXAMPLE = Path(__file__).parents[1] / "examples" / "entitylinkage-coverage" +ADAPTER = EXAMPLE / "adapter.py" + +CONTEXT = { + "version": 1, + "entities": [{"id": f"entity-{index}"} for index in range(1, 5)], + "sources": [ + {"id": "catalog-a", "availability": "available"}, + {"id": "catalog-b", "availability": "available"}, + {"id": "retired-register", "availability": "unavailable"}, + {"id": "spring-supplement", "availability": "not_in_release"}, + ], + "coverage": [ + { + "entity": "entity-2", + "source": "catalog-a", + "status": "mapping_gap", + "reason": "shared_name_requires_review", + }, + { + "entity": "entity-3", + "source": "catalog-a", + "status": "not_applicable", + "reason": "archive_is_outside_catalog_scope", + }, + { + "entity": "entity-4", + "source": "catalog-a", + "status": "mapping_gap", + "reason": "shared_name_requires_review", + }, + ], +} + +RESULTS = [ + { + "record_id": "record-a-001", + "status": "resolved", + "entity_id": "entity-1", + "candidate_entity_ids": ["entity-1"], + "reason_codes": ["candidate_match"], + "normalized_record_name": "north basin observatory", + "evidence": [ + { + "rule_id": "publication-name", + "phase": "candidate", + "outcome": "matched", + "entity_id": "entity-1", + "details": {}, + } + ], + }, + { + "record_id": "record-a-002", + "status": "ambiguous", + "entity_id": None, + "candidate_entity_ids": ["entity-2", "entity-4"], + "reason_codes": ["multiple_candidates"], + "normalized_record_name": "highland field station", + "evidence": [], + }, + { + "record_id": "record-a-003", + "status": "unresolved", + "entity_id": None, + "candidate_entity_ids": [], + "reason_codes": ["no_identity_candidate"], + "normalized_record_name": "unlisted coastal notice", + "evidence": [], + }, + { + "record_id": "record-b-001", + "status": "resolved", + "entity_id": "entity-3", + "candidate_entity_ids": ["entity-3"], + "reason_codes": ["candidate_match"], + "normalized_record_name": "central valley archive", + "evidence": [], + }, + { + "record_id": "record-b-002", + "status": "not_applicable", + "entity_id": None, + "candidate_entity_ids": [], + "reason_codes": ["not_applicable"], + "normalized_record_name": None, + "evidence": [], + }, +] + + +def write_inputs( + root: Path, + *, + results: list[dict[str, object]] | None = None, + context: dict[str, object] | None = None, + mapping_rows: list[tuple[str, str]] | None = None, + schema_version: str = "entitylinkage-results-v1", +) -> tuple[Path, Path, Path]: + root.mkdir(parents=True, exist_ok=True) + linkage_path = root / "linkage.json" + linkage_path.write_text( + json.dumps( + { + "schema_version": schema_version, + "tool_version": "0.1.0", + "unicode_version": "16.0", + "results": RESULTS if results is None else results, + }, + ensure_ascii=False, + ) + + "\n", + encoding="utf-8", + ) + mapping_path = root / "record-sources.csv" + rows = mapping_rows or [ + ("record-a-001", "catalog-a"), + ("record-a-002", "catalog-a"), + ("record-a-003", "catalog-a"), + ("record-b-001", "catalog-b"), + ("record-b-002", "catalog-b"), + ] + with mapping_path.open("w", encoding="utf-8", newline="") as stream: + writer = csv.writer(stream, lineterminator="\n") + writer.writerow(("record_id", "source_id")) + writer.writerows(rows) + context_path = root / "context.json" + context_path.write_text( + json.dumps(CONTEXT if context is None else context, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + return linkage_path, mapping_path, context_path + + +def run_adapter( + tmp_path: Path, + *, + results: list[dict[str, object]] | None = None, + context: dict[str, object] | None = None, + mapping_rows: list[tuple[str, str]] | None = None, + schema_version: str = "entitylinkage-results-v1", +) -> tuple[subprocess.CompletedProcess[str], Path]: + inputs = write_inputs( + tmp_path / "inputs", + results=results, + context=context, + mapping_rows=mapping_rows, + schema_version=schema_version, + ) + output = tmp_path / "output" + completed = subprocess.run( + [ + sys.executable, + str(ADAPTER), + "--linkage", + str(inputs[0]), + "--record-sources", + str(inputs[1]), + "--context", + str(inputs[2]), + "--output-dir", + str(output), + ], + check=False, + capture_output=True, + text=True, + ) + return completed, output + + +def test_adapter_preserves_unknowns_and_review_rows(tmp_path: Path) -> None: + completed, output = run_adapter(tmp_path) + + assert completed.returncode == 0, completed.stderr + assert completed.stdout == "" + manifest = load_manifest(output / "coverage.yaml") + summary = summarize_matrix(manifest, build_matrix(manifest)) + assert summary.relationship_count == 16 + assert summary.status_counts == { + "supported": 2, + "metadata_gap": 0, + "mapping_gap": 2, + "source_unavailable": 4, + "source_not_in_release": 4, + "not_applicable": 1, + "unknown": 3, + } + + with (output / "review.csv").open(encoding="utf-8", newline="") as stream: + reviews = list(csv.DictReader(stream)) + assert [(row["record_id"], row["status"]) for row in reviews] == [ + ("record-a-002", "ambiguous"), + ("record-a-003", "unresolved"), + ("record-b-002", "not_applicable"), + ] + assert reviews[0]["normalized_record_name"] == "highland field station" + assert json.loads(reviews[0]["candidate_entity_ids"]) == ["entity-2", "entity-4"] + assert json.loads(reviews[0]["evidence"]) == [] + + with (output / "resolved-records.csv").open(encoding="utf-8", newline="") as stream: + resolved = list(csv.DictReader(stream)) + assert resolved == [ + {"entity_id": "entity-1", "source_id": "catalog-a", "record_id": "record-a-001"}, + {"entity_id": "entity-3", "source_id": "catalog-b", "record_id": "record-b-001"}, + ] + + +def test_resolved_evidence_promotes_only_its_declared_mapping_gap(tmp_path: Path) -> None: + results = [dict(item) for item in RESULTS] + results[1] = { + **results[1], + "status": "resolved", + "entity_id": "entity-4", + "candidate_entity_ids": ["entity-4"], + "reason_codes": ["manual_override"], + } + completed, output = run_adapter(tmp_path, results=results) + + assert completed.returncode == 0, completed.stderr + manifest = load_manifest(output / "coverage.yaml") + summary = summarize_matrix(manifest, build_matrix(manifest)) + assert summary.status_counts["supported"] == 3 + assert summary.status_counts["mapping_gap"] == 1 + assert summary.status_counts["unknown"] == 3 + + with (output / "resolved-records.csv").open(encoding="utf-8", newline="") as stream: + resolved = list(csv.DictReader(stream)) + assert any( + row == {"entity_id": "entity-4", "source_id": "catalog-a", "record_id": "record-a-002"} + for row in resolved + ) + + +def test_multiple_resolved_records_share_one_supported_relationship(tmp_path: Path) -> None: + results = [dict(item) for item in RESULTS] + results[1] = { + **results[1], + "status": "resolved", + "entity_id": "entity-1", + "candidate_entity_ids": ["entity-1"], + "reason_codes": ["manual_override"], + } + completed, output = run_adapter(tmp_path, results=results) + + assert completed.returncode == 0, completed.stderr + manifest = load_manifest(output / "coverage.yaml") + summary = summarize_matrix(manifest, build_matrix(manifest)) + assert summary.status_counts["supported"] == 2 + with (output / "resolved-records.csv").open(encoding="utf-8", newline="") as stream: + resolved = list(csv.DictReader(stream)) + assert [row["record_id"] for row in resolved if row["entity_id"] == "entity-1"] == [ + "record-a-001", + "record-a-002", + ] + + +def test_adapter_rejects_resolved_not_applicable_conflict(tmp_path: Path) -> None: + results = [dict(item) for item in RESULTS] + results[1] = { + **results[1], + "status": "resolved", + "entity_id": "entity-3", + "candidate_entity_ids": ["entity-3"], + "reason_codes": ["manual_override"], + } + completed, output = run_adapter(tmp_path, results=results) + + assert completed.returncode == 2 + assert "conflicts with not_applicable" in completed.stderr + assert not output.exists() + + +def test_adapter_rejects_missing_record_source_mapping(tmp_path: Path) -> None: + mappings = [ + ("record-a-001", "catalog-a"), + ("record-a-002", "catalog-a"), + ("record-a-003", "catalog-a"), + ("record-b-001", "catalog-b"), + ] + completed, output = run_adapter(tmp_path, mapping_rows=mappings) + + assert completed.returncode == 2 + assert "missing source mapping" in completed.stderr + assert not output.exists() + + +def test_adapter_rejects_inconsistent_resolved_candidates(tmp_path: Path) -> None: + results = [dict(item) for item in RESULTS] + results[0] = {**results[0], "candidate_entity_ids": ["entity-2"]} + completed, output = run_adapter(tmp_path, results=results) + + assert completed.returncode == 2 + assert "must list only its resolved entity" in completed.stderr + assert not output.exists() + + +def test_adapter_outputs_are_byte_deterministic(tmp_path: Path) -> None: + first, first_output = run_adapter(tmp_path / "first") + second, second_output = run_adapter(tmp_path / "second") + + assert first.returncode == second.returncode == 0 + names = {"coverage.yaml", "review.csv", "resolved-records.csv"} + assert {name: (first_output / name).read_bytes() for name in names} == { + name: (second_output / name).read_bytes() for name in names + } + + +def test_adapter_rejects_bad_inputs_without_partial_outputs(tmp_path: Path) -> None: + completed, output = run_adapter(tmp_path, schema_version="entitylinkage-results-v2") + + assert completed.returncode == 2 + assert "schema_version" in completed.stderr + assert not output.exists() diff --git a/uv.lock b/uv.lock index bfa95f1..3ac615c 100644 --- a/uv.lock +++ b/uv.lock @@ -27,7 +27,7 @@ wheels = [ [[package]] name = "evidencematrix" -version = "0.1.0" +version = "0.1.1" source = { editable = "." } dependencies = [ { name = "pyyaml" },