diff --git a/CHANGELOG.md b/CHANGELOG.md index ae1f01e..dd590cd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,13 @@ All notable changes to Kindex are documented here. Format follows [Keep a Change `KIN_LOG_TIMESTAMPS=1`, and `kin` prefixes its stdout/stderr lines when that variable is set. Interactive and piped output is unchanged. Re-run `kin setup-cron` to pick it up on an existing install. +- `kin export --active-only` excludes retired and superseded nodes, including + predecessors replaced by a successor outside the selected audience. + +### Fixed +- Graph exports use a stable schema per audience and deterministic node/edge + ordering. Directed relationships retain their stored direction on import; + public/org exports preserve existing privacy omissions. ## [0.45.0] - 2026-09-28 diff --git a/docs/human-guide.md b/docs/human-guide.md index ce27565..af501be 100644 --- a/docs/human-guide.md +++ b/docs/human-guide.md @@ -398,6 +398,13 @@ kin import graph.jsonl --data-dir /path/to/isolated/graph --dry-run kin import graph.jsonl --data-dir /path/to/isolated/graph ``` +Add `--active-only` to export only canonically active nodes. Superseding edges +are evaluated across the source graph even when the successor is outside the +selected audience; out-of-scope node IDs remain omitted from the export. +Records have a stable field order for each audience, and nodes and stored arcs +have deterministic ordering. Public/org exports retain their privacy projection: +they omit intent, activity, reasoning and verification-method prose. + JSON arrays, single JSON records and JSONL remain supported. These are knowledge snapshots, **not full backups** of tasks, locks, reminders, policy or runtime state. Only lifecycle keys from `extra` travel: expiry, stale-referent and supersession diff --git a/src/kindex/cli.py b/src/kindex/cli.py index 54ee91f..914aaac 100644 --- a/src/kindex/cli.py +++ b/src/kindex/cli.py @@ -2009,14 +2009,29 @@ def cmd_export(args): # A snapshot must not silently truncate at the query helper's display limit. nodes = [n for audience in audiences for n in store.all_nodes(audience=audience, limit=-1)] + from .graph_transfer import canonical_status, export_record + if getattr(args, "active_only", False): + # Decide retirement against the full graph before audience projection: + # a visible predecessor can be superseded by a hidden successor. Only + # target IDs are used for filtering; edge serialization remains scoped. + superseded_ids = { + row["to_id"] + for row in store.conn.execute( + "SELECT DISTINCT to_id FROM edges WHERE type = 'supersedes'" + ).fetchall() + } + nodes = [node for node in nodes if canonical_status(node) == "active" + and node["id"] not in superseded_ids] + # Apply PII stripping for public/org exports strip_pii = target_audience in ("public", "org") - # Strip edges that cross audience boundaries + # Export the stored arcs verbatim. Store.add_edge persists a reciprocal + # itself when an edge is bidirectional; inferring one here would invert + # directed relationships during an export/import round trip. output = [] node_ids = {n["id"] for n in nodes} - from .graph_transfer import export_record - for n in nodes: + for n in sorted(nodes, key=lambda node: node["id"]): if strip_pii: n = _strip_pii(n) output.append(export_record(n, store.edges_from(n["id"]), node_ids, public=strip_pii)) @@ -7394,6 +7409,8 @@ def build_parser() -> argparse.ArgumentParser: help="Export graph (default) or a UA-compatible code map") s.add_argument("--audience", choices=["private", "team", "org", "public"], default="team") s.add_argument("--format", choices=["json", "jsonl", "understand-anything"], default="json") + s.add_argument("--active-only", action="store_true", + help="Export only canonical active nodes, excluding superseded predecessors") s.add_argument("--directory", help="Repository root for code-map metadata") s.add_argument("--project-name", help="Project name for code-map export") s.add_argument("--output", help="Write export to this file instead of stdout") diff --git a/src/kindex/graph_transfer.py b/src/kindex/graph_transfer.py index 79aae2a..83625b2 100644 --- a/src/kindex/graph_transfer.py +++ b/src/kindex/graph_transfer.py @@ -28,6 +28,21 @@ ) +def canonical_status(node: dict) -> str: + """Normalize legacy lifecycle omissions for the portable graph contract.""" + extra = node.get("extra") + if isinstance(extra, dict) and extra.get("superseded_by"): + return "superseded" + status = node.get("status") + return status.strip() if isinstance(status, str) and status.strip() else "active" + + +def canonical_standing(node: dict) -> str: + """Normalize pre-v13 records, which did not persist an explicit standing.""" + standing = node.get("standing") + return standing.strip() if isinstance(standing, str) and standing.strip() else "unruled" + + def scrub_shared_text(value: str) -> str: """Remove contact and machine-path prose while retaining evidence URLs.""" value = re.sub(r'\S+@\S+\.\S+', '[email]', value) @@ -90,8 +105,37 @@ def scrub(value): def export_record(node: dict, edges: list[dict], visible_ids: set[str], *, public: bool) -> dict: - record = {key: node[key] for key in (*TEXT_FIELDS, *LIST_FIELDS, *CLOCK_FIELDS, - *VERIFICATION_FIELDS, "id", "weight", "referent") if key in node} + # This is a public interchange schema, not a projection of whichever + # columns an older writer happened to load. Keep every field present so + # unchanged JSONL snapshots remain byte-stable across CLI versions. + record = { + "id": node["id"], + "type": node.get("type") or "concept", + "title": node.get("title") or "", + "content": node.get("content") or "", + "intent": node.get("intent") or "", + "status": canonical_status(node), + "audience": node.get("audience") or "private", + "prov_when": node.get("prov_when") or "", + "prov_activity": node.get("prov_activity") or "", + "prov_why": node.get("prov_why") or "", + "prov_source": node.get("prov_source") or "", + "created_at": node.get("created_at") or "", + "updated_at": node.get("updated_at") or "", + "standing": canonical_standing(node), + "aka": node.get("aka") or [], + "domains": node.get("domains") or [], + "prov_who": node.get("prov_who") or [], + "valid_at": node.get("valid_at") or None, + "invalid_at": node.get("invalid_at") or None, + "asserted_at": node.get("asserted_at") or None, + "true_of": node.get("true_of") or None, + "verified_at": node.get("verified_at") or None, + "verified_by": node.get("verified_by") or None, + "prov_method": node.get("prov_method") or None, + "weight": node.get("weight", 0.5), + "referent": node.get("referent") if isinstance(node.get("referent"), dict) else None, + } for key in LIST_FIELDS: record[key] = node.get(key) or [] # Older SQLite defaults used empty text. record["extra"] = {k: v for k, v in (node.get("extra") or {}).items() if k in LIFECYCLE_KEYS} @@ -99,10 +143,14 @@ def export_record(node: dict, edges: list[dict], visible_ids: set[str], *, publi for key in ("supersedes", "superseded_by"): if record["extra"].get(key) not in visible_ids: record["extra"].pop(key, None) + visible_edges = [e for e in edges if e["to_id"] in visible_ids] + # Store.edges_from sorts by weight only. Add a total tie-break order here + # so equivalent graphs serialize identically across insertion histories. + visible_edges.sort(key=lambda edge: (-edge["weight"], edge["to_id"], edge["type"])) record["edges"] = [ {"to": e["to_id"], "type": e["type"], "weight": e["weight"], "bidirectional": False, "provenance": "" if public else e.get("provenance", "")} - for e in edges if e["to_id"] in visible_ids + for e in visible_edges ] if public: for key in ("title", "content"): diff --git a/tests/test_import_export.py b/tests/test_import_export.py index 64c17cc..b78993b 100644 --- a/tests/test_import_export.py +++ b/tests/test_import_export.py @@ -85,6 +85,76 @@ def test_export_json_audience_filter(self, tmp_path): class TestExportJSONL: + @pytest.mark.parametrize("audience", ["private", "team", "org", "public"]) + def test_schema_is_stable_for_each_audience(self, tmp_path, audience): + d = str(tmp_path) + s = Store(Config(data_dir=d)) + s.add_node("Minimal", node_id="minimal", audience="public") + s.add_node("Detailed", node_id="detailed", audience="public", + content="Public evidence", intent="Internal intent") + s.close() + result = run("export", "--audience", audience, "--format", "jsonl", data_dir=d) + assert result.returncode == 0, result.stderr + rows = [json.loads(line) for line in result.stdout.splitlines()] + expected = ( + "id", "type", "title", "content", "intent", "status", "audience", + "prov_when", "prov_activity", "prov_why", "prov_source", "created_at", + "updated_at", "standing", "aka", "domains", "prov_who", "valid_at", + "invalid_at", "asserted_at", "true_of", "verified_at", "verified_by", + "prov_method", "weight", "referent", "extra", "edges", + ) + if audience in ("org", "public"): + expected = tuple(key for key in expected + if key not in ("intent", "prov_activity", "prov_why", "prov_method")) + assert "Internal intent" not in result.stdout + assert len(rows) == 2 + assert all(tuple(row) == expected for row in rows) + + @pytest.mark.parametrize("export_format", ["json", "jsonl"]) + def test_export_edge_order_survives_storage_reordering(self, tmp_path, export_format): + d = str(tmp_path) + s = Store(Config(data_dir=d)) + for node_id in ("hub", "a", "b", "c"): + s.add_node(node_id, node_id=node_id) + arcs = [("b", "relates_to"), ("a", "relates_to"), ("a", "implements")] + for target, edge_type in arcs: + s.add_edge("hub", target, edge_type=edge_type, weight=0.5, bidirectional=False) + s.add_edge("hub", "c", weight=0.75, bidirectional=False) + before = run("export", "--audience", "private", "--format", export_format, data_dir=d) + assert before.returncode == 0, before.stderr + s.conn.execute("DELETE FROM edges WHERE from_id = 'hub' AND weight = 0.5") + for target, edge_type in reversed(arcs): + s.add_edge("hub", target, edge_type=edge_type, weight=0.5, bidirectional=False) + s.close() + after = run("export", "--audience", "private", "--format", export_format, data_dir=d) + assert after.returncode == 0, after.stderr + assert after.stdout == before.stdout + rows = json.loads(after.stdout) if export_format == "json" else [ + json.loads(line) for line in after.stdout.splitlines()] + hub = next(row for row in rows if row["id"] == "hub") + assert [(edge["to"], edge["type"]) for edge in hub["edges"]] == [ + ("c", "relates_to"), ("a", "implements"), ("a", "relates_to"), ("b", "relates_to")] + + @pytest.mark.parametrize("audience", ["team", "org", "public"]) + def test_active_only_honors_hidden_successor_without_exposing_it(self, tmp_path, audience): + d = str(tmp_path) + s = Store(Config(data_dir=d)) + s.add_node("Visible predecessor", node_id="prior", audience="public") + s.add_node("Visible active", node_id="active", audience="public") + s.add_node("Hidden successor", node_id="secret-successor", audience="private") + s.add_edge("secret-successor", "prior", edge_type="supersedes", bidirectional=False) + s.add_edge("active", "secret-successor", bidirectional=False) + s.close() + result = run("export", "--audience", audience, "--active-only", data_dir=d) + assert result.returncode == 0, result.stderr + rows = json.loads(result.stdout) + assert [row["id"] for row in rows] == ["active"] + assert rows[0]["edges"] == [] + assert "secret-successor" not in result.stdout + full = run("export", "--audience", audience, data_dir=d) + assert full.returncode == 0, full.stderr + assert [row["id"] for row in json.loads(full.stdout)] == ["active", "prior"] + def test_export_jsonl(self, tmp_path): """Export JSONL format.""" d = str(tmp_path) @@ -122,6 +192,63 @@ def test_export_jsonl_multiple_nodes(self, tmp_path): lines = [l for l in r.stdout.strip().split("\n") if l.strip()] assert len(lines) >= 3 + def test_export_uses_complete_stable_active_contract(self, tmp_path): + """JSONL rows retain their shape across stores with legacy omissions.""" + d = str(tmp_path) + run("init", data_dir=d) + s = Store(Config(data_dir=d)) + s.add_node("Legacy active", node_id="legacy", audience="private") + s.add_node("Missing status", node_id="missing", audience="private") + s.add_node("Replacement", node_id="replacement", audience="private") + s.add_node("Archived", node_id="archived", audience="private", + status="archived") + # Simulate an older persisted row: no lifecycle value, but a successor + # marker that must still win when filtering the active projection. + s.conn.execute("UPDATE nodes SET status = '', extra = ? WHERE id = 'legacy'", + (json.dumps({"superseded_by": "replacement"}),)) + s.conn.execute("UPDATE nodes SET status = '' WHERE id = 'missing'") + s.add_edge("replacement", "archived", edge_type="relates_to", weight=0.75, + bidirectional=False) + s.close() + + r = run("export", "--audience", "private", "--format", "jsonl", + "--active-only", data_dir=d) + assert r.returncode == 0, r.stderr + rows = [json.loads(line) for line in r.stdout.splitlines() if line] + assert [row["id"] for row in rows] == ["missing", "replacement"] + assert tuple(rows[0]) == ( + "id", "type", "title", "content", "intent", "status", "audience", + "prov_when", "prov_activity", "prov_why", "prov_source", "created_at", + "updated_at", "standing", "aka", "domains", "prov_who", "valid_at", + "invalid_at", "asserted_at", "true_of", "verified_at", "verified_by", + "prov_method", "weight", "referent", "extra", "edges", + ) + for row in rows: + assert row["status"] == "active" + assert row["standing"] == "unruled" + assert row["referent"] is None + assert row["asserted_at"] is None + assert row["true_of"] is None + + def test_export_preserves_stored_directed_edges(self, tmp_path): + d = str(tmp_path) + run("init", data_dir=d) + s = Store(Config(data_dir=d)) + s.add_node("A", node_id="a", audience="private") + s.add_node("B", node_id="b", audience="private") + s.add_edge("a", "b", edge_type="implements", weight=0.75, + bidirectional=False) + s.close() + + r = run("export", "--audience", "private", "--format", "json", data_dir=d) + assert r.returncode == 0, r.stderr + rows = {row["id"]: row for row in json.loads(r.stdout)} + assert rows["a"]["edges"] == [ + {"to": "b", "type": "implements", "weight": 0.75, + "bidirectional": False, "provenance": ""}, + ] + assert rows["b"]["edges"] == [] + class TestImportJSON: def test_import_json(self, tmp_path): @@ -215,6 +342,35 @@ def test_import_merge(self, tmp_path): class TestRoundtrip: + @pytest.mark.parametrize( + ("export_format", "extension"), + [("json", "json"), ("jsonl", "jsonl")], + ) + def test_cli_roundtrip_active_only_preserves_superseding_successor( + self, tmp_path, export_format, extension, + ): + """A one-way supersedes arc must not retire its successor on replay.""" + source_dir, dest_dir = str(tmp_path / "source"), str(tmp_path / "dest") + run("init", data_dir=source_dir) + source = Store(Config(data_dir=source_dir)) + source.add_node("Prior decision", node_id="prior", node_type="decision") + source.add_node("Current decision", node_id="current", node_type="decision") + source.add_edge("current", "prior", edge_type="supersedes", bidirectional=False) + source.close() + + exported = run("export", "--audience", "private", "--format", export_format, + data_dir=source_dir) + assert exported.returncode == 0, exported.stderr + transfer = tmp_path / f"graph.{extension}" + transfer.write_text(exported.stdout) + + imported = run("import", str(transfer), data_dir=dest_dir) + assert imported.returncode == 0, imported.stderr + active_only = run("export", "--audience", "private", "--format", "json", + "--active-only", data_dir=dest_dir) + assert active_only.returncode == 0, active_only.stderr + assert [node["id"] for node in json.loads(active_only.stdout)] == ["current"] + def test_roundtrip(self, tmp_path): """Export then import, verify lossless.""" d = str(tmp_path)