diff --git a/.github/workflows/clara-review.yml b/.github/workflows/clara-review.yml new file mode 100644 index 000000000..f3a8c58f5 --- /dev/null +++ b/.github/workflows/clara-review.yml @@ -0,0 +1,453 @@ +name: CLARA Review + +# Ported from cell-ontology's clara-review.yml (see issue #3778). Adjustments +# vs. the CL original: +# +# - --edit-file points at the OBO edit file (uberon-edit.obo), not an OWL +# file, and --catalog passes uberon's own ODK-managed catalog-v001.xml. +# `robot diff`/`robot convert` detect format from content, not extension, +# so OBO input works with clara_workflow's stage-1 extractor unmodified; +# the catalog is what lets ROBOT resolve owl:imports from the locally +# committed files it maps to instead of the network -- required here +# because one of uberon's import PURLs currently 404s, and it's also what +# makes definition_refs() work for OBO input (see +# Cellular-Semantics/clara_workflow#8). +# - The ROBOT wrapper below honors $ROBOT_JAVA_ARGS (set to -Xmx9G), matching +# the heap size uberon's own diff.yml/qc.yml use for robot on this +# ontology. The CL original's wrapper doesn't parameterize JVM args at +# all, which is fine for CL's much smaller edit file but risks an OOM here. +# +# CLARA_WORKFLOW_REF is pinned to a commit (not `main`) for reproducibility, +# per that repo's own adoption guidance. + +env: + CLARA_WORKFLOW_REPO: Cellular-Semantics/clara_workflow + CLARA_WORKFLOW_REF: 234ac67ca82b577c5dedf851a17e57ed2b044478 + +on: + issue_comment: + types: [created] + workflow_dispatch: + inputs: + pr: + description: PR number to review + required: true + type: string + +permissions: + contents: read + pull-requests: write + issues: write + id-token: write + +jobs: + # Gates the review job on two independent things: who triggered it, and + # what content it would review. A trusted commenter doesn't make a fork + # PR's own CLAUDE.md / ontology text trustworthy once it's checked out and + # handed to Claude, so both checks are required, not either/or. + check-authorization: + # issue_comment also fires for plain issues; only consider /clara on PRs. + if: > + github.event_name == 'workflow_dispatch' || + (github.event.issue.pull_request != null && + startsWith(github.event.comment.body, '/clara')) + runs-on: ubuntu-latest + outputs: + allowed: ${{ steps.check.outputs.allowed }} + pr_num: ${{ steps.pr.outputs.num }} + steps: + - name: Resolve PR number + id: pr + env: + INPUT_PR: ${{ inputs.pr }} + run: | + if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then + if ! [[ "$INPUT_PR" =~ ^[0-9]+$ ]]; then + echo "::error::pr input must be a plain integer, got: $INPUT_PR" + exit 1 + fi + echo "num=$INPUT_PR" >> "$GITHUB_OUTPUT" + else + echo "num=${{ github.event.issue.number }}" >> "$GITHUB_OUTPUT" + fi + + - name: Checkout repository (default branch only -- never untrusted PR content) + uses: actions/checkout@v4 + with: + fetch-depth: 1 + + - name: Check commenter is a trusted controller, and PR is not from a fork + id: check + env: + PR_NUMBER: ${{ steps.pr.outputs.num }} + uses: actions/github-script@v8 + with: + script: | + const fs = require("fs"); + + let allowedUsers = []; + try { + allowedUsers = JSON.parse(fs.readFileSync(".github/ai-controllers.json", "utf8")); + } catch (e) { + core.setFailed(`Could not read .github/ai-controllers.json: ${e}`); + return; + } + + // workflow_dispatch already requires write access, enforced by + // GitHub itself before the workflow even starts; only the + // issue_comment path (anyone who can comment on the PR) needs + // the allowlist. + let commenterAllowed = true; + if (context.eventName === "issue_comment") { + const userLogin = context.payload.comment.user.login; + commenterAllowed = allowedUsers.includes(userLogin); + if (!commenterAllowed) { + console.log(`/clara comment from non-controller: ${userLogin}`); + } + } + + const { data: pr } = await github.rest.pulls.get({ + owner: context.repo.owner, + repo: context.repo.repo, + pull_number: Number(process.env.PR_NUMBER), + }); + const isFork = pr.head.repo.full_name !== pr.base.repo.full_name; + if (isFork) { + console.log(`Refusing to review a fork PR: ${pr.head.repo.full_name}`); + } + + core.setOutput("allowed", commenterAllowed && !isFork); + + review: + needs: check-authorization + if: needs.check-authorization.outputs.allowed == 'true' + # Scoped to this job, not the workflow: check-authorization's own job-level + # `if:` already skips (never runs, never queues) for any comment that + # isn't a qualifying /clara -- so only an actual /clara run ever reaches + # this group. A workflow-level group would be computed for every comment + # on the PR regardless of content, and cancel-in-progress would cancel a + # real review over an unrelated "LGTM". + concurrency: + group: clara-review-${{ needs.check-authorization.outputs.pr_num }} + cancel-in-progress: true + runs-on: ubuntu-latest + steps: + - name: Acknowledge trigger comment + if: github.event_name == 'issue_comment' + env: + GH_TOKEN: ${{ github.token }} + run: | + gh api --method POST \ + /repos/${{ github.repository }}/issues/comments/${{ github.event.comment.id }}/reactions \ + -f content=eyes + + - name: Look up PR refs + id: refs + env: + GH_TOKEN: ${{ github.token }} + PR_NUM: ${{ needs.check-authorization.outputs.pr_num }} + run: | + data=$(gh pr view "$PR_NUM" \ + --repo "${{ github.repository }}" \ + --json baseRefOid,headRefOid,headRefName,baseRefName) + base_tip=$(echo "$data" | jq -r .baseRefOid) + head_sha=$(echo "$data" | jq -r .headRefOid) + # Use the merge-base, not the base branch's current tip: if this + # branch is behind the base, diffing against the live tip pulls in + # every base-branch change since the branch diverged as spurious + # reverse-edits. merge_base_commit is the same value `git merge-base` + # would give, from the API so no local checkout is needed yet. + merge_base=$(gh api "repos/${{ github.repository }}/compare/${base_tip}...${head_sha}" \ + --jq .merge_base_commit.sha) + echo "base=$merge_base" >> "$GITHUB_OUTPUT" + echo "base_name=$(echo "$data" | jq -r .baseRefName)" >> "$GITHUB_OUTPUT" + echo "head=$head_sha" >> "$GITHUB_OUTPUT" + echo "head_name=$(echo "$data" | jq -r .headRefName)" >> "$GITHUB_OUTPUT" + + - name: Checkout PR head with full history + uses: actions/checkout@v4 + with: + fetch-depth: 0 + ref: ${{ steps.refs.outputs.head }} + + - name: Install ROBOT + run: | + mkdir -p "$HOME/.robot" + curl -fsSL -o "$HOME/.robot/robot.jar" \ + https://github.com/ontodev/robot/releases/latest/download/robot.jar + printf '#!/bin/bash\nexec java $ROBOT_JAVA_ARGS -jar "%s/.robot/robot.jar" "$@"\n' "$HOME" \ + > "$HOME/.robot/robot" + chmod +x "$HOME/.robot/robot" + echo "$HOME/.robot" >> "$GITHUB_PATH" + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install uv + uses: astral-sh/setup-uv@v7 + + - name: Check out clara_workflow reference + uses: actions/checkout@v4 + with: + repository: ${{ env.CLARA_WORKFLOW_REPO }} + ref: ${{ env.CLARA_WORKFLOW_REF }} + path: clara_workflow_ref + + - name: Install clara_workflow + run: pip install -e ./clara_workflow_ref + + - name: Run CLARA stage 1 extractor + env: + ROBOT_JAVA_ARGS: -Xmx9G + run: | + python -m clara_workflow.stage1.extract \ + --repo . \ + --left "${{ steps.refs.outputs.base }}" \ + --right "${{ steps.refs.outputs.head }}" \ + --edit-file src/ontology/uberon-edit.obo \ + --catalog src/ontology/catalog-v001.xml \ + --output changes.json + + - name: Upload changes.json artifact + uses: actions/upload-artifact@v4 + with: + name: clara-changes-pr-${{ needs.check-authorization.outputs.pr_num }} + path: changes.json + + - name: Select CLARA routing targets + run: | + python src/scripts/clara_select_targets.py \ + --input changes.json \ + --output routing.json + + - name: Upload routing.json artifact + uses: actions/upload-artifact@v4 + with: + name: clara-routing-pr-${{ needs.check-authorization.outputs.pr_num }} + path: routing.json + + - name: Extract routed target summary + id: routing_meta + run: | + python - <<'PY' + import json + from pathlib import Path + + routing = json.loads(Path("routing.json").read_text(encoding="utf-8")) + term_ids = sorted({target["term_id"].replace(":", "_", 1) for target in routing["targets"]}) + + Path("/tmp/routing_outputs.txt").write_text( + f"target_count={len(routing['targets'])}\nterm_count={len(term_ids)}\nterms={' '.join(term_ids)}\n", + encoding="utf-8", + ) + PY + cat /tmp/routing_outputs.txt >> "$GITHUB_OUTPUT" + + - name: Install local MCP server dependencies + if: steps.routing_meta.outputs.target_count != '0' + # The Asta MCP server is remote HTTP, but artl-mcp is configured as a + # local stdio server in clara_workflow's .mcp.json and must be runnable + # on the GitHub runner. + run: | + uv tool install artl-mcp + # Ensure uv tool bin directory is in PATH so Claude's MCP subprocess can find artl-mcp + echo "$(uv tool dir --bin)" >> "$GITHUB_PATH" + + - name: Verify artl-mcp is runnable + if: steps.routing_meta.outputs.target_count != '0' + run: uv tool run artl-mcp --help 2>&1 | head -5 + + - name: Run CLARA stage 2/3 verification for routed targets + id: claude + if: steps.routing_meta.outputs.target_count != '0' + continue-on-error: true + uses: anthropics/claude-code-action@v1 + env: + ASTA_API_KEY: ${{ secrets.ASTA_API_KEY }} + with: + claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} + github_token: ${{ github.token }} + # true for issue_comment forces claude-code-action's "tag" mode + # (an interactive @mention responder, with its own scaffolding our + # prompt doesn't need) -- but a real historical cell-ontology run + # (obophenotype/cell-ontology#3745, run 35205839595) proves tag + # mode actually completes this task fine, Write included; Claude + # just follows our prompt and ignores the irrelevant scaffolding. + # An earlier "always false" change here was based on a + # misdiagnosis: the actual instant failures seen while debugging + # this were model_not_found (fixed via --model below), not a mode + # problem. + track_progress: ${{ github.event_name == 'issue_comment' }} + settings: | + {"permissions":{"allow":["mcp__Asta_semanticscholar__*","mcp__artl-mcp__*"]}} + # MCP servers come from clara_workflow's .mcp.json at the pinned ref, + # so tool wiring is versioned with agent_instructions.md, same as + # cell-ontology (obophenotype/cell-ontology#3773). + claude_args: | + --model claude-sonnet-5 + --mcp-config ${{ github.workspace }}/clara_workflow_ref/.mcp.json + --disallowedTools "Bash(git add:*),Bash(git commit:*),Bash(git push:*),Bash(git checkout:*),Bash(git switch:*),Bash(git branch:*),Bash(gh pr create:*)" + prompt: | + REPO: ${{ github.repository }} + PR NUMBER: ${{ needs.check-authorization.outputs.pr_num }} + BASE SHA: ${{ steps.refs.outputs.base }} + HEAD SHA: ${{ steps.refs.outputs.head }} + + Work in the checked out ontology repository root. + + Read and follow `clara_workflow_ref/clara_workflow/agent_instructions.md`. + + Use `routing.json` in the repository root as the verification input. + Treat `routing.json` as the stable consumer contract for routed CLARA review. + Group targets by `term_id` and process one term-group at a time. + Write one `runs//...` bundle per processed term. + + Requirements: + - Only produce the required `runs//verdicts.json`, `runs//tool_calls.jsonl`, and `runs//report.md` outputs. + - Do not create commits, push branches, or open PRs. + - Ignore any `CLAUDE.md` guidance about committing, pushing, or opening PRs for this run. + - Follow other repository guidance in `CLAUDE.md` when useful, but do not edit ontology source files. + - Do not modify `changes.json` or `routing.json`. + + - name: Upload CLARA runs artifact + if: always() + uses: actions/upload-artifact@v4 + with: + name: clara-runs-pr-${{ needs.check-authorization.outputs.pr_num }} + path: runs + if-no-files-found: warn + + - name: Build CLARA summary comment + env: + PR_NUMBER: ${{ needs.check-authorization.outputs.pr_num }} + CLAUDE_STEP_OUTCOME: ${{ steps.claude.outcome }} + TARGET_COUNT: ${{ steps.routing_meta.outputs.target_count }} + run: | + python - <<'PY' > comment.md + import json + import os + from pathlib import Path + + def load_json(path): + with open(path, encoding="utf-8") as handle: + return json.load(handle) + + def summarize_verdict(path): + data = load_json(path) + assertions = data.get("assertions", []) + core = [a for a in assertions if a.get("category") == "core"] + background = [a for a in assertions if a.get("category") == "background"] + core_fail = [a for a in core if a.get("final_verdict") == "fail"] + core_uncertain = [a for a in core if a.get("final_verdict") == "uncertain"] + if core_fail: + status = "FAIL" + elif core_uncertain: + status = "UNCERTAIN" + else: + status = "PASS" + return { + "term_id": data["term_id"], + "name": data["name"], + "status": status, + "core_total": len(core), + "core_fail": len(core_fail), + "core_uncertain": len(core_uncertain), + "background_warn": sum(1 for a in background if a.get("warn_background")), + "failed_texts": [a["text"] for a in core_fail[:3]], + } + + data = load_json("changes.json") + routing = load_json("routing.json") + + route_counts = routing["summary"]["route_counts"] + target_count = int(os.environ["TARGET_COUNT"]) + claude_outcome = os.environ.get("CLAUDE_STEP_OUTCOME", "") + runs_dir = Path("runs") + verdict_files = sorted(runs_dir.glob("*/verdicts.json")) if runs_dir.exists() else [] + verdict_summaries = [summarize_verdict(path) for path in verdict_files] + + lines = [ + f"### CLARA review for PR #{os.environ['PR_NUMBER']}", + "", + f"- Base SHA: `{data['left']['ref']}`", + f"- Head SHA: `{data['right']['ref']}`", + f"- Terms touched: `{len(data['by_term'])}`", + f"- Total extracted changes: `{len(data['changes'])}`", + f"- Reviewable changes: `{len(data['reviewable'])}`", + f"- Decomposable changes: `{len(data['decomposable'])}`", + f"- Routed validation targets: `{routing['summary']['selected_targets']}`", + f"- Routed NTR bundles: `{route_counts['ntr']}`", + f"- Routed revised-text targets: `{route_counts.get('text_revision', 0)}`", + f"- Routed ref-only-addition targets: `{route_counts.get('refs_added', 0)}`", + f"- Routed relationship targets: `{route_counts['relationship']}`", + f"- Routed synonym targets with refs: `{route_counts['synonym']}`", + ] + + if target_count == 0: + lines.extend([ + "", + "_No routed CLARA targets were found, so stage 2/3 verification did not run._", + ]) + elif claude_outcome == "failure": + lines.extend([ + "", + "_Stage 2/3 verification was invoked for routed CLARA targets, but the Claude step did not complete successfully. Check the workflow logs and uploaded artifacts._", + ]) + else: + lines.extend([ + "", + f"_Stage 2/3 verification ran for `{len(verdict_summaries)}` routed term(s)._", + ]) + + if verdict_summaries: + lines.extend([ + "", + "#### Routed verification results", + "", + ]) + for item in verdict_summaries: + lines.append( + f"- `{item['term_id']}` — {item['name']}: **{item['status']}** " + f"(core: {item['core_total']}, fail: {item['core_fail']}, " + f"uncertain: {item['core_uncertain']}, background warnings: {item['background_warn']})" + ) + for failed in item["failed_texts"]: + lines.append(f" - Failed core assertion: {failed}") + + if routing["summary"]["ignored_changes"]: + # Labels are cosmetic; unknown reasons still surface, so a new + # ignore reason can never go silently unreported. + reason_labels = { + "existing_term_text_not_yet_routed": "Existing-term text changes not yet routed", + "synonym_without_refs": "Synonym changes without refs", + "text_removal_not_reviewable": "Text removals (not justified by refs)", + "refs_added_not_searchable": "Ref-only edits whose new refs are not searchable", + } + lines.extend(["", "#### Currently ignored by routing", ""]) + for reason, count in sorted(routing["summary"]["ignored_reason_counts"].items()): + label = reason_labels.get(reason, reason.replace("_", " ").capitalize()) + lines.append(f"- {label}: `{count}`") + + lines.extend( + [ + "", + "Artifacts uploaded:", + "- `changes.json`", + "- `routing.json`", + "- `runs/`", + ] + ) + + print("\n".join(lines)) + PY + + - name: Post CLARA summary to PR + env: + GH_TOKEN: ${{ github.token }} + PR_NUM: ${{ needs.check-authorization.outputs.pr_num }} + run: | + gh pr comment "$PR_NUM" \ + --repo "${{ github.repository }}" \ + --body-file comment.md diff --git a/src/scripts/clara_select_targets.py b/src/scripts/clara_select_targets.py new file mode 100644 index 000000000..1a979028d --- /dev/null +++ b/src/scripts/clara_select_targets.py @@ -0,0 +1,423 @@ +#!/usr/bin/env python3 + +"""Select ticket-relevant CLARA validation targets from stage-1 output. + +Phase 2 is still a routing preview. This script reads `changes.json` from +`clara_workflow.stage1.extract` and emits a normalized `routing.json` payload +that identifies which downstream CLARA checks should run for the current PR. + +This file is the producer-side contract for routed CLARA review. The consumer +is `clara_workflow/clara_workflow/agent_instructions.md`, which expects: + +- processing grouped by `term_id` +- `ntr` targets with canonical prose in `textual_changes` +- `relationship` / `synonym` targets with a single `change` +- `candidate_refs` and `term_level_candidate_refs` already filtered to + searchable literature ids (`PMID:...` / `DOI:...`) + +Notes on compatibility: + +- `textual_changes` is the canonical field for decomposable prose +- `definition_changes` is still emitted as a temporary alias so the current + consumer can tolerate older payload samples during cleanup + +Current routing policy mirrors the ticket scope: + +- New terms (NTRs): route added definitions/comments plus added structural + axioms as one NTR validation bundle. +- Existing terms: route added structural axioms (`subclass`, `relationship`, + `equivalent_class`) as direct atomic checks. +- Existing terms: route text changes from stage-1 `text_deltas`, split by what + actually changed — `text_revision` for new prose, `refs_added` for prose that + is unchanged but has gained a reference. The second exists because ROBOT + reports an annotated axiom as a removed/added pair, so attaching a dbxref to + an untouched definition otherwise looks identical to a rewrite. +- Any term: route added synonym axioms only when the synonym axiom itself has + attached refs. + +Intentionally not routed yet: + +- Reviewable removals +- Synonyms without refs +""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +TEXTUAL_KINDS = frozenset({"text_def", "comment"}) +STRUCTURAL_KINDS = frozenset({"subclass", "relationship", "equivalent_class"}) +SYNONYM_KINDS = frozenset( + {"synonym_exact", "synonym_broad", "synonym_narrow", "synonym_related"} +) + + +def _stable_unique(values: list[str]) -> list[str]: + seen: set[str] = set() + ordered: list[str] = [] + for value in values: + if value not in seen: + seen.add(value) + ordered.append(value) + return ordered + + +def _is_searchable_ref(value: str) -> bool: + """Return whether a ref is usable by the downstream literature tools.""" + upper = value.upper() + return upper.startswith("PMID:") or upper.startswith("DOI:") + + +def _refs_for_changes(changes: list[dict]) -> list[str]: + """Collect unique searchable refs from staged ontology changes. + + This is the upstream filtering point for the routed-target contract. + Downstream agentic stages should treat these lists as already curated and + should not broaden them beyond formatting normalization for tool calls. + """ + refs: list[str] = [] + for change in changes: + refs.extend(ref for ref in change.get("refs", []) if _is_searchable_ref(ref)) + return _stable_unique(refs) + + +def _change_target( + *, + route: str, + validation_mode: str, + ordinal: int, + change: dict, + term_id: str, + term_label: str, + term_is_new: bool, + term_level_candidate_refs: list[str], +) -> dict: + axiom_refs = _refs_for_changes([change]) + # Logical axioms never carry dbxrefs of their own, so an empty axiom-level + # list is the normal case, not a missing-data case: scope them to the + # definition's refs instead of handing the agent an empty list. + if not axiom_refs and change["kind"] in STRUCTURAL_KINDS: + axiom_refs = list(term_level_candidate_refs) + return { + "target_id": f"{route}:{term_id}:{ordinal}", + "route": route, + "validation_mode": validation_mode, + "term_id": term_id, + "term_label": term_label, + "term_is_new": term_is_new, + "change": change, + "candidate_refs": axiom_refs, + "term_level_candidate_refs": term_level_candidate_refs, + } + + +def _text_target( + *, + route: str, + validation_mode: str, + ordinal: int, + change: dict, + term_id: str, + term_label: str, + term_level_candidate_refs: list[str], + candidate_refs: list[str] | None = None, + extra: dict | None = None, +) -> dict: + """Build a text target for an existing term. + + Prose is carried in `textual_changes` (a one-element list) so the consumer + can reuse the same decomposition path it already applies to NTR bundles. + """ + target = { + "target_id": f"{route}:{term_id}:{ordinal}", + "route": route, + "validation_mode": validation_mode, + "term_id": term_id, + "term_label": term_label, + "term_is_new": False, + "textual_changes": [change], + "candidate_refs": ( + _refs_for_changes([change]) if candidate_refs is None else candidate_refs + ), + "term_level_candidate_refs": term_level_candidate_refs, + } + if extra: + target.update(extra) + return target + + +def select_targets(payload: dict) -> dict: + """Transform stage-1 output into routed targets for CLARA verification. + + Output contract: + + - one `ntr` target per new term containing: + - `textual_changes` (canonical prose field) + - `definition_changes` (compatibility alias only) + - `relationship_changes` + - one `relationship` target per existing-term structural assertion + - one `synonym` target per synonym assertion that carries attached refs + + The resulting `routing.json` is consumed term-grouped: every target with the + same `term_id` is expected to be processed together downstream. + """ + targets: list[dict] = [] + ignored: list[dict] = [] + route_counts = { + "ntr": 0, + "text_revision": 0, + "refs_added": 0, + "relationship": 0, + "synonym": 0, + } + # Stage-1 pairs removed/added text axioms; absent on payloads produced + # before that landed, in which case existing-term text stays unrouted. + deltas_by_term: dict[str, list[dict]] = {} + for delta in payload.get("text_deltas", []): + deltas_by_term.setdefault(delta["term_id"], []).append(delta) + have_text_deltas = "text_deltas" in payload + ignored_reason_counts: dict[str, int] = {} + reviewable_terms = 0 + obsoleted_terms_skipped = 0 + + for term_id, entry in sorted(payload["by_term"].items()): + term_label = entry["term_label"] + term_is_new = bool(entry["is_new_term"]) + term_is_obsoleted = bool(entry["is_obsoleted"]) + term_is_reviewable = bool(entry["is_reviewable"]) + if term_is_reviewable: + reviewable_terms += 1 + if term_is_obsoleted: + obsoleted_terms_skipped += 1 + continue + + added_reviewable = [ + change + for change in entry["changes"] + if change["side"] == "added" and change["kind"] in (TEXTUAL_KINDS | STRUCTURAL_KINDS | SYNONYM_KINDS) + ] + # Refs on definitions touched by this PR, falling back to the refs on + # the term's definition as it stands at the head ref. The fallback is + # what makes a logical definition checkable: `robot diff` reports only + # changed axioms, so a PR that adds an EquivalentTo to an untouched term + # carries no definition axiom, yet the text definition it formalises is + # exactly what justifies it. + term_level_candidate_refs = _stable_unique( + _refs_for_changes( + [change for change in added_reviewable if change["kind"] in TEXTUAL_KINDS] + ) + + [ref for ref in entry.get("definition_refs", []) if _is_searchable_ref(ref)] + ) + + if term_is_new: + ntr_textual = [change for change in added_reviewable if change["kind"] in TEXTUAL_KINDS] + ntr_structural = [change for change in added_reviewable if change["kind"] in STRUCTURAL_KINDS] + if ntr_textual or ntr_structural: + targets.append( + { + "target_id": f"ntr:{term_id}", + "route": "ntr", + "validation_mode": "decompose_definition_and_relationships", + "term_id": term_id, + "term_label": term_label, + "term_is_new": True, + # Canonical field consumed by clara_workflow. + "textual_changes": ntr_textual, + # Temporary alias kept during producer/consumer cleanup. + "definition_changes": ntr_textual, + "relationship_changes": ntr_structural, + "candidate_refs": _refs_for_changes(ntr_textual + ntr_structural), + "term_level_candidate_refs": term_level_candidate_refs, + } + ) + route_counts["ntr"] += 1 + + if not term_is_new and have_text_deltas: + text_ordinal = 0 + refs_ordinal = 0 + for delta in deltas_by_term.get(term_id, []): + status = delta["status"] + if status == "removed": + ignored.append( + { + "term_id": term_id, + "term_label": term_label, + "reason": "text_removal_not_reviewable", + "change": delta, + } + ) + continue + + # Prefer the parsed change (it carries the predicate) and fall + # back to the delta itself if stage 1 didn't surface a match. + change = next( + ( + c + for c in added_reviewable + if c["kind"] == delta["kind"] and c["value"] == delta["value"] + ), + None, + ) + if change is None: + continue + change = {**change, "prior_value": delta.get("prior_value")} + + if status == "refs_only": + new_refs = [ + ref for ref in delta.get("refs_added", []) if _is_searchable_ref(ref) + ] + if not new_refs: + # Nothing a literature tool can check; routing it would + # only manufacture an `uncertain` verdict. + ignored.append( + { + "term_id": term_id, + "term_label": term_label, + "reason": "refs_added_not_searchable", + "change": delta, + } + ) + continue + refs_ordinal += 1 + targets.append( + _text_target( + route="refs_added", + validation_mode="validate_new_refs_against_existing_text", + ordinal=refs_ordinal, + change=change, + term_id=term_id, + term_label=term_label, + term_level_candidate_refs=term_level_candidate_refs, + # Only the newly attached refs are under review; the + # pre-existing ones were justified when they landed. + candidate_refs=new_refs, + extra={"refs_added": new_refs}, + ) + ) + route_counts["refs_added"] += 1 + continue + + text_ordinal += 1 + targets.append( + _text_target( + route="text_revision", + validation_mode="decompose_revised_text", + ordinal=text_ordinal, + change=change, + term_id=term_id, + term_label=term_label, + term_level_candidate_refs=term_level_candidate_refs, + ) + ) + route_counts["text_revision"] += 1 + + structural_ordinal = 0 + synonym_ordinal = 0 + for change in added_reviewable: + kind = change["kind"] + if kind in STRUCTURAL_KINDS: + if term_is_new: + continue + structural_ordinal += 1 + targets.append( + _change_target( + route="relationship", + # An equivalence axiom asserts necessary AND sufficient + # conditions, so it is not an atomic relationship check. + validation_mode=( + "validate_equivalent_class_axiom" + if kind == "equivalent_class" + else "validate_atomic_relationship" + ), + ordinal=structural_ordinal, + change=change, + term_id=term_id, + term_label=term_label, + term_is_new=term_is_new, + term_level_candidate_refs=term_level_candidate_refs, + ) + ) + route_counts["relationship"] += 1 + continue + + if kind in SYNONYM_KINDS: + if change.get("refs"): + synonym_ordinal += 1 + targets.append( + _change_target( + route="synonym", + validation_mode="validate_synonym_against_attached_refs", + ordinal=synonym_ordinal, + change=change, + term_id=term_id, + term_label=term_label, + term_is_new=term_is_new, + term_level_candidate_refs=term_level_candidate_refs, + ) + ) + route_counts["synonym"] += 1 + else: + ignored.append( + { + "term_id": term_id, + "term_label": term_label, + "reason": "synonym_without_refs", + "change": change, + } + ) + continue + + if kind in TEXTUAL_KINDS and not term_is_new and not have_text_deltas: + # Pre-text_deltas payload: no way to tell a rewrite from a + # ref-only edit, so leave it unrouted rather than guess. + ignored.append( + { + "term_id": term_id, + "term_label": term_label, + "reason": "existing_term_text_not_yet_routed", + "change": change, + } + ) + + for item in ignored: + reason = item["reason"] + ignored_reason_counts[reason] = ignored_reason_counts.get(reason, 0) + 1 + + return { + "source": { + "left": payload["left"], + "right": payload["right"], + }, + "summary": { + "terms_touched": len(payload["by_term"]), + "reviewable_terms": reviewable_terms, + "obsoleted_terms_skipped": obsoleted_terms_skipped, + "reviewable_changes": len(payload["reviewable"]), + "decomposable_changes": len(payload["decomposable"]), + "selected_targets": len(targets), + "route_counts": route_counts, + "ignored_changes": len(ignored), + "ignored_reason_counts": ignored_reason_counts, + }, + "targets": targets, + "ignored": ignored, + } + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--input", type=Path, required=True, help="Path to stage-1 changes.json") + parser.add_argument("--output", type=Path, required=True, help="Path to write routing.json") + args = parser.parse_args(argv) + + payload = json.loads(args.input.read_text(encoding="utf-8")) + selected = select_targets(payload) + args.output.write_text(json.dumps(selected, indent=2) + "\n", encoding="utf-8") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())