From 8aa29a88527513276ddbe18b4f637253dbf370d3 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Tue, 8 Sep 2026 15:22:46 -0400 Subject: [PATCH 01/13] Export Microcosm schema metadata for Orrery --- changelog.d/orrery-schema-cli.added.md | 1 + docs/graph-explorer.md | 5 + docs/shared-graph-explorer-adapter.md | 126 +++++ packages/microcosm-graph/README.md | 5 + .../src/microcosm/graph/explorer.py | 492 ++++++++++++++++++ .../microcosm-graph/tests/test_explorer.py | 465 +++++++++++++++++ 6 files changed, 1094 insertions(+) create mode 100644 changelog.d/orrery-schema-cli.added.md create mode 100644 docs/shared-graph-explorer-adapter.md create mode 100644 packages/microcosm-graph/src/microcosm/graph/explorer.py create mode 100644 packages/microcosm-graph/tests/test_explorer.py diff --git a/changelog.d/orrery-schema-cli.added.md b/changelog.d/orrery-schema-cli.added.md new file mode 100644 index 000000000..8ed3fd1ef --- /dev/null +++ b/changelog.d/orrery-schema-cli.added.md @@ -0,0 +1 @@ +Add a Microcosm schema adapter and `python -m microcosm.graph.explorer` CLI for the Orrery shared viewer, preserving field identities and exact large integers while refusing invalid or oversized input and existing output files. diff --git a/docs/graph-explorer.md b/docs/graph-explorer.md index 0f4fd9f12..caadf0044 100644 --- a/docs/graph-explorer.md +++ b/docs/graph-explorer.md @@ -1,5 +1,10 @@ # Graph explorer +For declaration and schema inspection with the shared viewer, see the +[shared graph explorer adapter](shared-graph-explorer-adapter.md). The run +renderer described below remains unchanged, including its separate cache and +gate statuses. + The graph explorer is one self-contained HTML file generated from a compiled graph and its run manifest. It contains its own CSS, JavaScript, DAG, and small charts. A reviewer can copy the file to another machine and open it directly in diff --git a/docs/shared-graph-explorer-adapter.md b/docs/shared-graph-explorer-adapter.md new file mode 100644 index 000000000..e244254f2 --- /dev/null +++ b/docs/shared-graph-explorer-adapter.md @@ -0,0 +1,126 @@ +# Shared graph explorer adapter + +`microcosm.graph.explorer` converts a `microcosm.graph.schema.v1` metadata +export into `graph-explorer/v1`, the portable contract consumed by [Orrery](https://github.com/TheAxiomFoundation/orrery), the +shared graph viewer. Microcosm owns the graph's calculation +and schema meanings. The shared package owns navigation, rendering and bundled +offline HTML. + +The adapter adds no JavaScript dependency to Microcosm. It reads no microdata, +executes no population operation, and leaves `explain_html` unchanged. + +## Open a saved schema in Orrery + +From a synced Microcosm checkout, export a saved schema with: + +```sh +uv run python -m microcosm.graph.explorer \ + --input compiled-schema.json --output graph.json --title "Microcosm US" +graph-explorer --input graph.json --output graph.html +``` + +The second command uses the separately installed, accepted shared viewer +0.3.0 CLI, whose package name is still `@axiom-foundation/graph-explorer`. +The forthcoming Orrery package rename does not change the JSON contract. +Microcosm does not install or upgrade the viewer as part of export. + +Open `graph.html` using a local static server or your existing artifact viewer. +The HTML contains its viewer assets and needs no CDN. Direct `file://` opening +is outside the current Microcosm browser acceptance. + +The Python command accepts only saved schema JSON, rejects duplicate keys and +non-finite numbers, bounds input before decoding, and creates a new output file. +It refuses an existing output, including the input path. A saved manifest or +microdata file is not a schema export. Review schema metadata before sharing it: +the entire supplied metadata is retained, and canvas filtering does not redact it. + +The same exporter is available as a Python API: + +```python +import json +from pathlib import Path + +from microcosm.graph.explorer import graph_explorer_json + +schema = json.loads(Path("compiled-schema.json").read_text(encoding="utf-8")) +Path("graph.json").write_text(graph_explorer_json(schema), encoding="utf-8") +``` + +The schema producer is being integrated separately. In a checkout containing +`microcosm.graph.schema.graph_schema`, the input can be produced directly with +`graph_schema(compiled)`. This adapter also accepts previously saved exports; +it does not require that producer to be installed. `graph_explorer_document` +returns the same document as detached Python dictionaries and lists. + +The shared package's built CLI accepts: + +```sh +graph-explorer --input graph.json --output graph.html +``` + +The shared package must supply its built viewer assets. Microcosm does not +download them or launch a build automatically. HTML export and browser +verification are distinct checks; a successfully written HTML file alone does +not establish that its embedded viewer runs. + +## What the graph contains + +The document contains every operation, source declaration, visible field version +and input binding in the supplied schema. A field identity contains its +population, entity, column, producer and nearest declaration. A pre-rewrite +input therefore remains distinct from the final value in the same population. +Population coordinates are retained in `data`; they are not replaced by a +presentation revision or containment parent. + +| Edge kind | Meaning | +| --- | --- | +| `compiled_predecessor` | An operation dependency listed in the supplied compiler metadata | +| `produced` | The provider of a versioned field value, including a structural carrier | +| `declared_read` | An operation's slice, slice mask, output mask or rewrite incumbent; role and row mask are retained | +| `structural_input` | Ancestry from a completed base population's fields into its structural successor | +| `source` | A named external input and its declared codec | +| `artifact` | A producer-to-consumer dependency with its exact alias, artifact name and nominal type/version | + +Structural ancestry does not imply unchanged values. Declared reads describe +operation-level dependencies; they do not infer a separate mathematical formula +for each output. Typed artifact declarations do not establish the existence of +runtime artifact bytes. Source declarations remain domain nodes; citation +references cannot substitute for their codec contract. + +The complete original metadata stays in `metadata.microcosm`, including compiled +owners/order/versions and declaration details such as `mass_partition`. Exact +Python integers outside JavaScript's safe range are transported as +`{"integer_literal": "..."}` before JavaScript parses the document. Large integral +floats use `{"float_literal": "..."}` so their type is not silently changed to +integer. The raw declaration digest is retained separately from presentation +revisions. + +## Scope, identity and evidence + +`complete_supplied_schema` means the complete supplied metadata snapshot. It +does not mean the complete US recipe. Entity IDs and memberships, scientific +units, period semantics, source preparation internals and implicit runtime reads +cannot be inferred when the source schema omits them. Declared ownership is +not evidence that a field contains valid materialized values. + +The converter checks JSON bounds, declaration digest consistency, references, +field/declaration agreement and exact read-role coverage. It does not recompile +or authenticate an imported dictionary. The document revision hashes the whole +supplied schema, including its compiler tables; node and edge revisions hash +their transport records before the revision field is attached. These are +metadata content digests, not execution keys or source-byte attestations. + +No execution, authorship, cache, gate or release status is inferred from schema +metadata. This first adapter supplies no runtime activities or Receipt verdicts. +The existing deterministic run viewer retains its independent cache/gate axes. +A later run adapter must bind actual run evidence and keep these statuses +separate. Receipt assessments must come from a configured host verifier and +bind to the exact exported document bytes, not merely the source declaration's +hash. Custody verification does not establish scientific correctness. + +Exports fail explicitly at their development bounds rather than silently +truncating: 256 operations, 256 sources, 20,000 presentation nodes, 100,000 edges, +32 MiB input and 64 MiB output. Prospective transport bytes are charged as +records are added. These are metadata limits, not microdata size limits or a +claim about total Python process memory. The viewer can focus or collapse a +complete document without changing the underlying export. diff --git a/packages/microcosm-graph/README.md b/packages/microcosm-graph/README.md index cb0d46142..ab0edf67d 100644 --- a/packages/microcosm-graph/README.md +++ b/packages/microcosm-graph/README.md @@ -29,6 +29,11 @@ Module map: | `executor.py` | `run_graph`: projection, patching, ownership enforcement, receipts | | `manifest.py` | `RunManifest`, `NodeReceipt`, human decision records | | `view.py` | `describe(node)`: the one-screen view | +| `explorer.py` | Pure adapter from schema metadata to the shared `graph-explorer/v1` presentation contract | The shard depends on `microcosm-frame` only. Kernels that wrap fit, calibrate, or a rules engine live in those shards and register here. + +See [the shared graph explorer adapter](../../docs/shared-graph-explorer-adapter.md) for +versioned field inspection, offline integration and evidence boundaries. The +existing deterministic `explain_html` export remains available unchanged. diff --git a/packages/microcosm-graph/src/microcosm/graph/explorer.py b/packages/microcosm-graph/src/microcosm/graph/explorer.py new file mode 100644 index 000000000..981f18d23 --- /dev/null +++ b/packages/microcosm-graph/src/microcosm/graph/explorer.py @@ -0,0 +1,492 @@ +"""Pure presentation adapter from compiler metadata to graph-explorer/v1. + +The input is a microcosm.graph.schema.v1 export, not a Frame or RunManifest. +This module checks presentation references, not compilation or authenticity. +It never imports a kernel, opens a source, or infers an execution verdict. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +import stat +from collections import Counter +from pathlib import Path + +from .canonical import canonical_json + +__all__ = ["graph_explorer_document", "graph_explorer_json"] + +_PROTOCOL = "microcosm.graph.schema.v1" +_SAFE_INTEGER = 2**53 - 1 +_MAX_INPUT_BYTES = 32 * 1024 * 1024 +_MAX_OUTPUT_BYTES = 64 * 1024 * 1024 +_MAX_ITEMS = 2_000_000 +_MAX_NODES = 20_000 +_MAX_EDGES = 100_000 +_FIELD_KEYS = ("population", "entity", "column", "producer", "declared_in") +_READ_KINDS = {"slice", "slice_mask", "output_mask", "rewrite_incumbent"} + + +def _require(condition: bool, message: str) -> None: + if not condition: + raise ValueError(f"Graph explorer: {message}.") + + +def _copy_json(value: object, *, transport: bool = False) -> object: + """Bound and detach plain JSON; reject custom objects before invoking them.""" + items, charge = 0, 0 + + def walk(child: object, depth: int) -> object: + nonlocal items, charge + items += 1 + charge += 16 + _require(items <= _MAX_ITEMS and depth <= 64, "metadata complexity") + kind = type(child) + if kind is str: + _require(len(child) <= 16_384, "string length") + charge += len(child) * 6 + result = child + elif child is None or kind is bool: + result = child + elif kind is int: + _require(child.bit_length() <= 1024, "integer size") + charge += child.bit_length() + result = ( + {"integer_literal": str(child)} + if transport and abs(child) > _SAFE_INTEGER + else child + ) + elif kind is float: + _require(math.isfinite(child), "finite numbers required") + # JS cannot distinguish a large integral float from an unsafe int. + result = ( + {"float_literal": repr(child)} + if transport and child.is_integer() and abs(child) > _SAFE_INTEGER + else child + ) + elif kind is list: + _require(len(child) <= _MAX_ITEMS - items, "array size") + result = [walk(item, depth + 1) for item in child] + elif kind is dict: + _require(len(child) <= (_MAX_ITEMS - items) // 2, "object size") + _require(all(type(key) is str for key in child), "string keys required") + result = { + walk(key, depth + 1): walk(item, depth + 1) + for key, item in child.items() + } + else: + raise ValueError("Graph explorer: plain JSON values required.") + _require(charge <= _MAX_OUTPUT_BYTES, "prospective metadata size") + return result + + return walk(value, 0) + + +def _object(value: object) -> dict: + _require(type(value) is dict, "object required") + return value + + +def _array(value: object) -> list: + _require(type(value) is list, "array required") + return value + + +def _text(value: object) -> str: + _require(type(value) is str and bool(value.strip()), "nonempty text required") + return value + + +def _index(values: object, key: str) -> dict[str, dict]: + result = {} + for item in _array(values): + name = _text(_object(item).get(key)) + _require(name not in result, f"duplicate {key}") + result[name] = item + return result + + +def _id(*parts: str) -> str: + return json.dumps(parts, ensure_ascii=False, separators=(",", ":")) + + +def _field_id(field: dict) -> str: + return _id("field", *(_text(field.get(key)) for key in _FIELD_KEYS)) + + +def _revision(value: object) -> str: + return "sha256:" + hashlib.sha256(canonical_json(value)).hexdigest() + + +def _bounded_json(value: object, limit: int) -> str: + parts, size = [], 0 + encoder = json.JSONEncoder( + ensure_ascii=False, allow_nan=False, sort_keys=True, separators=(",", ":") + ) + for part in encoder.iterencode(value): + size += len(part.encode("utf-8")) + _require(size <= limit, "serialized metadata size") + parts.append(part) + return "".join(parts) + + +def graph_explorer_document(schema: dict, *, title: str | None = None) -> dict: + """Export the entire supplied schema snapshot; never silently truncate. + + Use ``graph_schema(compiled)`` as the producer where available. Imported + dictionaries remain declarations: reference checks and content digests do + not authenticate them or establish that a compiler or kernel actually ran. + The document revision binds all supplied metadata, including compiled tables; + node/edge revisions bind their presentation records, not runtime cache keys. + + Pre-rewrite inputs have distinct field IDs in the same population. Structural + edges mean ancestry, not equality. Reads apply to an operation, without + claiming that each input mathematically determines each individual output. + """ + doc = _object(_copy_json(schema)) + _bounded_json(doc, _MAX_INPUT_BYTES) + _require(doc.get("protocol") == _PROTOCOL, "unsupported schema protocol") + country = _text(doc.get("country")) + graph = _object(doc.get("graph")) + _require(graph.get("country") == country, "country mismatch") + digest = hashlib.sha256(canonical_json(graph)).hexdigest() + _require(doc.get("graph_sha256") == digest, "declaration digest mismatch") + operations = _index(graph.get("nodes"), "id") + sources = _index(graph.get("sources"), "name") + _require(len(operations) <= 256 and len(sources) <= 256, "declaration count") + compiled = _object(doc.get("compiled")) + order = _array(compiled.get("order")) + _require(all(type(item) is str for item in order), "order IDs") + _require(len(order) == len(operations) and set(order) == set(operations), "order") + predecessors = _object(compiled.get("predecessors")) + versions = _object(compiled.get("versions")) + _require(set(predecessors) == set(operations) == set(versions), "compiled IDs") + positions = {name: index for index, name in enumerate(order)} + populations = { + name for name, op in operations.items() if op.get("structural") != "none" + } + for name, op in operations.items(): + _require( + _text(op.get("structural")) + in {"none", "create", "filter", "expand", "reweight"}, + "structural kind", + ) + _require(_text(versions[name]) in populations, "population reference") + _require( + versions[name] == (name if name in populations else op.get("population")), + "population binding", + ) + base = op.get("base") + _require(base is None or _text(base) in populations, "base reference") + parents = _array(predecessors[name]) + _require(all(type(parent) is str for parent in parents), "predecessor IDs") + _require(len(set(parents)) == len(parents), "duplicate predecessor") + _require( + all( + parent in positions and positions[parent] < positions[name] + for parent in parents + ), + "predecessor order", + ) + + nodes, edges = {}, {} + # Reserve wrapper metadata, then charge each detached transport record before + # retaining it. Counts alone would allow long repeated IDs to over-expand. + transport_doc = _copy_json(doc, transport=True) + used = len(_bounded_json(transport_doc, _MAX_OUTPUT_BYTES).encode("utf-8")) + 65_536 + + def presentation_record(record: dict) -> dict: + nonlocal used + record = _object(_copy_json(record, transport=True)) + record["revision"] = _revision(record) + encoded = _bounded_json(record, _MAX_OUTPUT_BYTES - used) + used += len(encoded.encode("utf-8")) + 1 + return record + + def add_node(node: dict) -> None: + _require(node["id"] not in nodes, "duplicate presentation node") + _require(len(nodes) < _MAX_NODES, "node limit") + nodes[node["id"]] = presentation_record(node) + + def add_edge(source: str, target: str, kind: str, data: dict | None = None) -> None: + facts = {} if data is None else data + identity = _id( + "edge", kind, source, target, _bounded_json(facts, _MAX_INPUT_BYTES) + ) + if identity in edges: + return + _require(len(edges) < _MAX_EDGES, "edge limit") + edge = { + "id": identity, + "source": source, + "target": target, + "kind": kind, + "category": "dependency", + "data": facts, + } + edges[identity] = presentation_record(edge) + + declarations = {} + for name, op in operations.items(): + add_node( + { + "id": _id("operation", name), + "label": name, + "kind": "operation", + "data": {"declaration": op, "population": versions[name]}, + } + ) + for output in _array(op.get("outputs")): + output = _object(output) + key = (name, _text(output.get("entity")), _text(output.get("column"))) + _require(key not in declarations, "duplicate Owned declaration") + # The real serializer omits rewrite=False. Normalize only this + # internal lookup; preserve the authored declaration byte-for-byte. + rewrite = output.get("rewrite", False) + _require(type(rewrite) is bool, "rewrite flag") + declarations[key] = {**output, "rewrite": rewrite} + for name, source in sources.items(): + _text(source.get("codec")) + add_node( + { + "id": _id("source", name), + "label": name, + "kind": "source", + "data": {"declaration": source}, + } + ) + + fields, visible = {}, {} + + def add_field(field: dict, *, visible_field: bool) -> str: + identity = _field_id(field) + key = (field["declared_in"], field["entity"], field["column"]) + _require(key in declarations, "field declaration reference") + owned = declarations[key] + _require( + field["population"] in populations and field["producer"] in operations, + "field reference", + ) + _require( + versions[field["producer"]] == field["population"], + "field provider population", + ) + for prop in ("dtype", "rows", "ownership", "rewrite"): + _require( + field.get(prop) == owned.get(prop), "field declaration disagreement" + ) + if identity not in fields: + fields[identity] = field + add_node( + { + "id": identity, + "label": f"{field['entity']}.{field['column']}", + "kind": "field", + "data": {**field, "visible_in_schema": visible_field}, + } + ) + add_edge(_id("operation", field["producer"]), identity, "produced") + else: + _require(fields[identity] == field, "ambiguous field value") + return identity + + for field in _array(doc.get("schema")): + field = _object(field) + coordinate = tuple(_text(field.get(key)) for key in _FIELD_KEYS[:3]) + _require(coordinate not in visible, "duplicate visible field") + visible[coordinate] = add_field(field, visible_field=True) + + # Check role coverage against declarations, without reimplementing the + # compiler's provider resolution. Preserve repeated declared roles too. + expected_reads = Counter() + for name, op in operations.items(): + if op["structural"] == "create": + continue + population = versions[name] if op["structural"] == "none" else op["base"] + for slice_ in _array(op.get("inputs")): + slice_ = _object(slice_) + entity, rows = _text(slice_.get("entity")), _text(slice_.get("rows")) + for column in _array(slice_.get("columns")): + expected_reads[ + name, population, entity, _text(column), rows, "slice" + ] += 1 + if rows != "all": + expected_reads[name, population, entity, rows, "all", "slice_mask"] += 1 + for output in _array(op.get("outputs")): + entity, column, rows = ( + _text(output.get(key)) for key in ("entity", "column", "rows") + ) + if output.get("rewrite", False): + expected_reads[ + name, population, entity, column, rows, "rewrite_incumbent" + ] += 1 + if rows != "all": + expected_reads[ + name, population, entity, rows, "all", "output_mask" + ] += 1 + reads = _array(doc.get("input_bindings")) + _require(len(reads) <= 100_000, "read count") + actual_reads = Counter( + tuple( + _text(_object(read).get(key)) + for key in ("node", "population", "entity", "column", "rows", "kind") + ) + for read in reads + ) + _require(actual_reads == expected_reads, "declared read coverage") + for read in reads: + read = _object(read) + _require( + read.get("node") in operations and read.get("kind") in _READ_KINDS, + "read reference or kind", + ) + _text(read.get("rows")) + key = tuple(_text(read.get(k)) for k in ("declared_in", "entity", "column")) + _require(key in declarations, "read declaration reference") + owned = declarations[key] + field = {key: read[key] for key in _FIELD_KEYS} + field.update( + {key: owned[key] for key in ("dtype", "rows", "ownership", "rewrite")} + ) + identity = add_field(field, visible_field=False) + # Preserve explicit role labels alongside the supplied compiled edges. + # This adapter does not infer a new compiled dependency from a read. + add_edge( + identity, + _id("operation", read["node"]), + "declared_read", + {"read_kind": read["kind"], "rows": read["rows"]}, + ) + + for name, op in operations.items(): + target = _id("operation", name) + for parent in predecessors[name]: + add_edge(_id("operation", parent), target, "compiled_predecessor") + if op.get("base") is not None: + # Every field in the completed base is carried. Values may change + # during structural materialization; this is not an identity edge. + for (population, _entity, _column), field_id in visible.items(): + if population == op["base"]: + add_edge(field_id, target, "structural_input") + for source in _array(op.get("sources")): + _require(type(source) is str and source in sources, "source reference") + add_edge(_id("source", source), target, "source") + for binding in _array(op.get("artifact_inputs", [])): + binding = _object(binding) + producer = binding.get("producer") + _require( + type(producer) is str and producer in operations, "artifact producer" + ) + outputs = _index(operations[producer].get("artifact_outputs", []), "name") + artifact = _text(binding.get("artifact")) + _require( + artifact in outputs + and outputs[artifact].get("type") == binding.get("type"), + "artifact type or output", + ) + _require(producer in predecessors[name], "artifact predecessor") + add_edge(_id("operation", producer), target, "artifact", binding) + + _require( + all( + edge["source"] in nodes and edge["target"] in nodes + for edge in edges.values() + ), + "dangling edge", + ) + result = { + "schemaVersion": "graph-explorer/v1", + "id": _id("microcosm", country), + "title": _text(title) + if title is not None + else f"{country.upper()} graph schema", + "revision": _revision(doc), + "description": "Complete supplied declaration metadata; no execution or release verdict.", + "nodes": [nodes[key] for key in sorted(nodes)], + "edges": [edges[key] for key in sorted(edges)], + "metadata": { + "adapter": "microcosm.graph.explorer.v1", + "scope": "complete_supplied_schema", + "truncated": False, + "evidence_scope": "Imported declaration metadata; not recompiled or authenticated by this adapter.", + "revision_scope": "Content digests of supplied metadata/presentation records; not runtime cache keys.", + "edge_scope": "Operation-level reads and structural ancestry; not individual-output formulas or value equality.", + "microcosm": transport_doc, + "missing_schema": [ + "entity IDs and memberships", + "units", + "period semantics", + "source preparation internals", + "implicit runtime reads", + ], + }, + } + result = _copy_json(result, transport=True) + _bounded_json(result, _MAX_OUTPUT_BYTES) + return result + + +def graph_explorer_json(schema: dict, *, title: str | None = None) -> str: + """Deterministic UTF-8-ready JSON, with exact large-number transport tags. + + Write these bytes directly for host-side snapshot hashing. HTML bundling and + Receipt verification belong to the shared explorer, not this adapter. + """ + return ( + _bounded_json(graph_explorer_document(schema, title=title), _MAX_OUTPUT_BYTES) + + "\n" + ) + + +def _unique_json_object(pairs: list[tuple[str, object]]) -> dict: + result = {} + for key, value in pairs: + _require(key not in result, "duplicate JSON key") + result[key] = value + return result + + +def _invalid_json_constant(value: str) -> None: + raise ValueError("Graph explorer: finite JSON numbers required.") + + +def main(argv: list[str] | None = None) -> int: + """Convert saved schema metadata for Orrery without loading a saved run.""" + parser = argparse.ArgumentParser(description=main.__doc__) + parser.add_argument("--input", type=Path, required=True, help="saved schema JSON") + parser.add_argument( + "--output", type=Path, required=True, help="new graph-explorer/v1 JSON file" + ) + parser.add_argument("--title", help="viewer document title") + args = parser.parse_args(argv) + try: + # Refuse streams/devices before reading; a FIFO must not block the CLI. + descriptor = os.open(args.input, os.O_RDONLY | getattr(os, "O_NONBLOCK", 0)) + with os.fdopen(descriptor, "rb") as stream: + info = os.fstat(stream.fileno()) + _require(stat.S_ISREG(info.st_mode), "input must be a regular file") + _require(info.st_size <= _MAX_INPUT_BYTES, "input byte limit") + raw = stream.read(_MAX_INPUT_BYTES + 1) + _require(len(raw) <= _MAX_INPUT_BYTES, "input byte limit") + schema = json.loads( + raw.decode("utf-8"), + object_pairs_hook=_unique_json_object, + parse_constant=_invalid_json_constant, + ) + rendered = graph_explorer_json(schema, title=args.title) + args.output.parent.mkdir(parents=True, exist_ok=True) + # Exclusive creation also prevents overwriting the source through an alias. + with args.output.open("x", encoding="utf-8", newline="\n") as stream: + stream.write(rendered) + except (OSError, ValueError, RecursionError) as error: + parser.error(str(error)) + print(f"wrote {args.output}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/packages/microcosm-graph/tests/test_explorer.py b/packages/microcosm-graph/tests/test_explorer.py new file mode 100644 index 000000000..df4c676a4 --- /dev/null +++ b/packages/microcosm-graph/tests/test_explorer.py @@ -0,0 +1,465 @@ +"""Presentation contracts over invented declaration metadata; no population run.""" + +import copy +import hashlib +import json +import os + +import pytest + +from microcosm.graph import ( + Graph, + Node, + Owned, + Ownership, + Slice, + SourceRef, + StructuralDelta, + compile_graph, + explorer, + graph_to_json, +) +from microcosm.graph.canonical import canonical_json + + +def test_cli_exports_exact_metadata_and_large_integer(tmp_path): + schema = snapshot() + schema["audit_integer"] = 2**60 + 1 + source = tmp_path / "schema.json" + output = tmp_path / "report" / "graph.json" + source.write_text(json.dumps(schema), encoding="utf-8") + assert ( + explorer.main( + ["--input", str(source), "--output", str(output), "--title", "Microcosm"] + ) + == 0 + ) + assert output.read_text() == explorer.graph_explorer_json(schema, title="Microcosm") + doc = json.loads(output.read_text()) + assert doc["metadata"]["microcosm"]["audit_integer"] == { + "integer_literal": str(2**60 + 1) + } + + +@pytest.mark.parametrize("raw", ['{"protocol":1,"protocol":2}', '{"x":NaN}', "[]"]) +def test_cli_refuses_invalid_json_without_output(tmp_path, raw): + source, output = tmp_path / "schema.json", tmp_path / "graph.json" + source.write_text(raw) + with pytest.raises(SystemExit) as error: + explorer.main(["--input", str(source), "--output", str(output)]) + assert error.value.code == 2 + assert not output.exists() + + +def test_cli_preserves_existing_output_and_input(tmp_path): + source = tmp_path / "schema.json" + raw = json.dumps(snapshot()) + source.write_text(raw) + with pytest.raises(SystemExit) as error: + explorer.main(["--input", str(source), "--output", str(source)]) + assert error.value.code == 2 + assert source.read_text() == raw + + +def test_cli_refuses_oversized_input_before_decoding(tmp_path, monkeypatch): + source, output = tmp_path / "schema.json", tmp_path / "graph.json" + source.write_bytes(b"!" * 33) + monkeypatch.setattr(explorer, "_MAX_INPUT_BYTES", 32) + with pytest.raises(SystemExit) as error: + explorer.main(["--input", str(source), "--output", str(output)]) + assert error.value.code == 2 + assert not output.exists() + + +@pytest.mark.skipif(not hasattr(os, "mkfifo"), reason="requires POSIX FIFO") +def test_cli_refuses_fifo_without_waiting_for_writer(tmp_path): + source, output = tmp_path / "schema.json", tmp_path / "graph.json" + os.mkfifo(source) + with pytest.raises(SystemExit) as error: + explorer.main(["--input", str(source), "--output", str(output)]) + assert error.value.code == 2 + assert not output.exists() + + +def snapshot(): + graph = Graph( + "invented", + sources=(SourceRef("survey", "unused@1"), SourceRef("unused", "unused@1")), + nodes=( + Node( + "base", + "unused.create@1", + structural=StructuralDelta.CREATE, + sources=("survey",), + outputs=( + Owned("person", "age", "int64"), + Owned("person", "mask", "boolean"), + Owned("person", "missing", "float64", ownership=Ownership.ABSENT), + ), + ), + Node( + "filtered", + "unused.filter@1", + structural=StructuralDelta.FILTER, + base="base", + inputs=(Slice("person", ("age",)),), + ), + Node( + "rewrite", + "unused.rewrite@1", + population="filtered", + inputs=(Slice("person", ("age", "mask"), rows="mask"),), + outputs=(Owned("person", "age", "int64", rows="mask", rewrite=True),), + ), + Node( + "consumer", + "unused.consume@1", + population="filtered", + inputs=(Slice("person", ("age",)),), + ), + ), + ) + compiled = compile_graph(graph) + raw = graph_to_json(graph) + fields = [] + for population in ("base", "filtered"): + for owned in graph.nodes[0].outputs: + rewrite = population == "filtered" and owned.column == "age" + fields.append( + { + "population": population, + "entity": owned.entity, + "column": owned.column, + "dtype": owned.dtype, + "producer": "rewrite" if rewrite else population, + "declared_in": "rewrite" if rewrite else "base", + "rows": "mask" if rewrite else "all", + "ownership": owned.ownership.value, + "rewrite": rewrite, + } + ) + + def read( + node, + column, + *, + population="filtered", + producer="filtered", + declared_in="base", + kind="slice", + rows="all", + ): + return { + "node": node, + "entity": "person", + "column": column, + "population": population, + "producer": producer, + "declared_in": declared_in, + "kind": kind, + "rows": rows, + } + + return { + "protocol": "microcosm.graph.schema.v1", + "country": "invented", + "graph_sha256": hashlib.sha256(raw.encode()).hexdigest(), + "graph": json.loads(raw), + "compiled": { + "order": list(compiled.order), + "versions": dict(compiled.versions), + "predecessors": { + key: list(value) for key, value in compiled.predecessors.items() + }, + "owners": [ + list(key) + [owner] for key, owner in sorted(compiled.owners.items()) + ], + }, + "schema": fields, + "input_bindings": [ + read("filtered", "age", population="base", producer="base"), + read("rewrite", "age", rows="mask"), + read("rewrite", "mask", rows="mask"), + read("rewrite", "mask", kind="slice_mask"), + read("rewrite", "mask", kind="output_mask"), + read("rewrite", "age", kind="rewrite_incumbent", rows="mask"), + read("consumer", "age", producer="rewrite", declared_in="rewrite"), + ], + } + + +def update_graph_digest(doc): + doc["graph_sha256"] = hashlib.sha256(canonical_json(doc["graph"])).hexdigest() + + +def test_full_snapshot_and_no_runtime_claims(): + source = snapshot() + doc = explorer.graph_explorer_document(source) + assert doc["schemaVersion"] == "graph-explorer/v1" + assert doc["metadata"]["microcosm"] == source + assert doc["metadata"]["scope"] == "complete_supplied_schema" + assert doc["metadata"]["truncated"] is False + assert len([node for node in doc["nodes"] if node["kind"] == "operation"]) == 4 + assert len([node for node in doc["nodes"] if node["kind"] == "source"]) == 2 + assert len([node for node in doc["nodes"] if node["kind"] == "field"]) == 7 + assert not {"activities", "receipts", "artifacts", "assessments"} & doc.keys() + assert all( + "statuses" not in node and "parentId" not in node for node in doc["nodes"] + ) + assert len( + [edge for edge in doc["edges"] if edge["kind"] == "compiled_predecessor"] + ) == sum(map(len, source["compiled"]["predecessors"].values())) + assert {edge["source"] for edge in doc["edges"]} | { + edge["target"] for edge in doc["edges"] + } <= {node["id"] for node in doc["nodes"]} + + +def test_rewrite_incumbent_is_distinct_and_masks_retain_roles(): + doc = explorer.graph_explorer_document(snapshot()) + ages = [ + n + for n in doc["nodes"] + if n["kind"] == "field" + and n["data"]["population"] == "filtered" + and n["data"]["column"] == "age" + ] + assert len(ages) == 2 and ages[0]["id"] != ages[1]["id"] + incumbent = next(n for n in ages if n["data"]["producer"] == "filtered") + final = next(n for n in ages if n["data"]["producer"] == "rewrite") + assert incumbent["data"]["visible_in_schema"] is False + assert final["data"]["visible_in_schema"] is True + reads = [e for e in doc["edges"] if e["kind"] == "declared_read"] + assert {e["data"]["read_kind"] for e in reads} == { + "slice", + "slice_mask", + "output_mask", + "rewrite_incumbent", + } + incoming = [ + e + for e in reads + if json.loads(e["target"])[1] == "rewrite" + and e["data"]["read_kind"] == "rewrite_incumbent" + ] + assert len(incoming) == 1 and incoming[0]["source"] == incumbent["id"] + assert incoming[0]["data"]["rows"] == "mask" + assert not any( + e["source"] == final["id"] and json.loads(e["target"])[1] == "rewrite" + for e in reads + ) + structural = [e for e in doc["edges"] if e["kind"] == "structural_input"] + assert len(structural) == 3 + assert all(json.loads(e["source"])[1] == "base" for e in structural) + + +def test_typed_artifact_edge_and_uninterpreted_metadata_survive(): + source = snapshot() + base, _, _, consumer = source["graph"]["nodes"] + type_ = {"name": "invented.summary", "schema_version": 2} + base["artifact_outputs"] = [{"name": "summary", "type": type_}] + consumer["artifact_inputs"] = [ + { + "name": "local_alias", + "producer": "base", + "artifact": "summary", + "type": type_, + } + ] + source["compiled"]["predecessors"]["consumer"].append("base") + source["graph"]["mass_partition"] = ["person", "period"] + update_graph_digest(source) + doc = explorer.graph_explorer_document(source) + edge = next(e for e in doc["edges"] if e["kind"] == "artifact") + assert edge["data"] == consumer["artifact_inputs"][0] + assert edge["category"] == "dependency" + assert doc["metadata"]["microcosm"]["graph"]["mass_partition"] == [ + "person", + "period", + ] + + +def test_deterministic_detached_json_and_exact_large_number_transport(): + source = snapshot() + source["graph"]["nodes"][0]["params"] = { + "large": 2**80 + 1, + "negative": -(2**80 + 1), + "safe": 2**53 - 1, + "float": 1e100, + "flag": True, + } + update_graph_digest(source) + before = copy.deepcopy(source) + text = explorer.graph_explorer_json(source) + assert text == explorer.graph_explorer_json(source) and text.endswith("\n") + doc = json.loads(text) + params = doc["metadata"]["microcosm"]["graph"]["nodes"][0]["params"] + assert params == { + "large": {"integer_literal": str(2**80 + 1)}, + "negative": {"integer_literal": str(-(2**80 + 1))}, + "safe": 2**53 - 1, + "float": {"float_literal": "1e+100"}, + "flag": True, + } + doc["metadata"]["microcosm"]["graph"]["nodes"].clear() + assert source == before + + +def test_document_revision_binds_compiled_metadata_not_only_graph(): + source = snapshot() + left = explorer.graph_explorer_document(source) + source["compiled"]["extra_declaration"] = "caller supplied, not verified" + right = explorer.graph_explorer_document(source) + assert left["revision"] != right["revision"] + assert ( + left["metadata"]["microcosm"]["graph_sha256"] + == right["metadata"]["microcosm"]["graph_sha256"] + ) + assert left["nodes"] == right["nodes"] + + +def test_named_mask_read_is_preserved_separately_from_compiler_predecessors(): + graph = Graph( + "invented", + sources=(SourceRef("survey", "unused@1"),), + nodes=( + Node( + "base", + "unused.create@1", + structural=StructuralDelta.CREATE, + sources=("survey",), + outputs=(Owned("person", "age", "int64"),), + ), + Node( + "a_mask", + "unused.mask@1", + population="base", + inputs=(Slice("person", ("age",)),), + outputs=(Owned("person", "mask", "boolean"),), + ), + Node( + "z_consumer", + "unused.consume@1", + population="base", + inputs=(Slice("person", ("age", "mask"), rows="mask"),), + ), + ), + ) + compiled = compile_graph(graph) + source = snapshot() + source["graph"] = json.loads(graph_to_json(graph)) + update_graph_digest(source) + source["compiled"] = { + "order": list(compiled.order), + "versions": dict(compiled.versions), + "predecessors": { + key: list(value) for key, value in compiled.predecessors.items() + }, + "owners": [list(key) + [value] for key, value in compiled.owners.items()], + } + source["schema"] = [ + { + "population": "base", + "entity": "person", + "column": column, + "dtype": dtype, + "producer": owner, + "declared_in": owner, + "rows": "all", + "ownership": "produced", + "rewrite": False, + } + for column, dtype, owner in ( + ("age", "int64", "base"), + ("mask", "boolean", "a_mask"), + ) + ] + source["input_bindings"] = [ + { + "node": node, + "population": "base", + "entity": "person", + "column": column, + "producer": owner, + "declared_in": owner, + "kind": kind, + "rows": rows, + } + for node, column, owner, kind, rows in ( + ("a_mask", "age", "base", "slice", "all"), + ("z_consumer", "age", "base", "slice", "mask"), + ("z_consumer", "mask", "a_mask", "slice", "mask"), + ("z_consumer", "mask", "a_mask", "slice_mask", "all"), + ) + ] + doc = explorer.graph_explorer_document(source) + target = json.dumps(["operation", "z_consumer"], separators=(",", ":")) + incoming = [edge for edge in doc["edges"] if edge["target"] == target] + compiled_parents = { + json.loads(edge["source"])[1] + for edge in incoming + if edge["kind"] == "compiled_predecessor" + } + assert compiled_parents == set(compiled.predecessors["z_consumer"]) + mask = next( + edge + for edge in incoming + if edge["kind"] == "declared_read" and edge["data"]["read_kind"] == "slice_mask" + ) + assert json.loads(mask["source"])[4] == "a_mask" + + +@pytest.mark.parametrize( + "mutation", + [ + lambda doc: doc.update(protocol="future"), + lambda doc: doc.update(graph_sha256="0" * 64), + lambda doc: doc["compiled"]["order"].reverse(), + lambda doc: doc["schema"].append(copy.deepcopy(doc["schema"][0])), + lambda doc: doc["schema"][0].update(dtype="invented"), + lambda doc: doc["input_bindings"][0].update(producer="unknown"), + lambda doc: doc["input_bindings"][0].update(kind="invented"), + lambda doc: doc["input_bindings"].clear(), + lambda doc: doc["input_bindings"].append( + { + **doc["input_bindings"][-1], + "column": "mask", + "producer": "filtered", + "declared_in": "base", + } + ), + ], +) +def test_invalid_references_fail_without_partial_document(mutation): + source = snapshot() + mutation(source) + with pytest.raises(ValueError): + explorer.graph_explorer_document(source) + + +def test_oversize_or_non_json_metadata_fails(monkeypatch): + source = snapshot() + source["extra"] = float("nan") + with pytest.raises(ValueError, match="finite"): + explorer.graph_explorer_document(source) + source["extra"] = source + with pytest.raises(ValueError, match="complexity"): + explorer.graph_explorer_document(source) + source = snapshot() + monkeypatch.setattr(explorer, "_MAX_NODES", 3) + with pytest.raises(ValueError, match="node limit"): + explorer.graph_explorer_document(source) + + +def test_expanded_output_budget_is_checked_during_construction(monkeypatch): + source = snapshot() + budget = len(canonical_json(source)) + 65_536 + 2_000 + monkeypatch.setattr(explorer, "_MAX_OUTPUT_BYTES", budget) + + # The input fits, but presentation records must stop while being added. + # A late-only final serialization check would visit the field loop first. + def unexpected_field(_field): + pytest.fail("expanded records reached field construction before refusal") + + monkeypatch.setattr(explorer, "_field_id", unexpected_field) + with pytest.raises(ValueError, match="serialized metadata size"): + explorer.graph_explorer_document(source) From ea0c233b55cc61b89fa7f9f4fa12c473aced217b Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 28 Sep 2026 23:58:01 -0400 Subject: [PATCH 02/13] Test explorer metadata invariants with Hypothesis (#888) --- .../microcosm-graph/tests/test_explorer.py | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/packages/microcosm-graph/tests/test_explorer.py b/packages/microcosm-graph/tests/test_explorer.py index df4c676a4..2ab7ff453 100644 --- a/packages/microcosm-graph/tests/test_explorer.py +++ b/packages/microcosm-graph/tests/test_explorer.py @@ -6,6 +6,8 @@ import os import pytest +from hypothesis import example, given, settings +from hypothesis import strategies as st from microcosm.graph import ( Graph, @@ -463,3 +465,99 @@ def unexpected_field(_field): monkeypatch.setattr(explorer, "_field_id", unexpected_field) with pytest.raises(ValueError, match="serialized metadata size"): explorer.graph_explorer_document(source) + + +_JSON_SCALARS = st.one_of( + st.none(), + st.booleans(), + st.integers(min_value=-(2**1024 - 1), max_value=2**1024 - 1), + st.floats(allow_nan=False, allow_infinity=False), + st.text(max_size=40), +) +_JSON_METADATA = st.recursive( + _JSON_SCALARS, + lambda children: st.one_of( + st.lists(children, max_size=4), + st.dictionaries(st.text(max_size=20), children, max_size=4), + ), + max_leaves=20, +) + + +@settings(deadline=None) +@given(value=_JSON_SCALARS) +@example(value=2**53 - 1) +@example(value=2**53) +@example(value=-(2**53 - 1)) +@example(value=-(2**53)) +@example(value=float(2**53 - 1)) +@example(value=float(2**53)) +@example(value=-float(2**53)) +@example(value=-0.0) +@example(value=True) +def test_metadata_transport_is_exact_and_detached(value): + """Metadata preserves exact scalar values and shares no mutable containers.""" + source = snapshot() + source["extra"] = {"nested": [value]} + before = copy.deepcopy(source) + doc = explorer.graph_explorer_document(source) + rendered = json.loads(explorer.graph_explorer_json(source)) + transported = rendered["metadata"]["microcosm"]["extra"]["nested"][0] + + if type(value) is int and abs(value) > 2**53 - 1: + assert transported == {"integer_literal": str(value)} + elif type(value) is float and value.is_integer() and abs(value) > 2**53 - 1: + assert transported == {"float_literal": repr(value)} + else: + assert type(transported) is type(value) + assert transported == value + if type(value) is float: + assert transported.hex() == value.hex() + assert doc["metadata"]["microcosm"]["extra"]["nested"] == [transported] + + exported_values = doc["metadata"]["microcosm"]["extra"]["nested"] + exported_values.append("output-side mutation") + doc["metadata"]["microcosm"]["graph"]["nodes"].clear() + assert source == before + source["extra"]["nested"].append("input-side mutation") + assert exported_values == [transported, "output-side mutation"] + + +def reversed_mapping_order(value): + if isinstance(value, dict): + return { + key: reversed_mapping_order(child) + for key, child in reversed(list(value.items())) + } + if isinstance(value, list): + return [reversed_mapping_order(child) for child in value] + return value + + +@settings(deadline=None) +@given(metadata=_JSON_METADATA) +def test_json_is_invariant_to_mapping_order(metadata): + """Equivalent JSON objects produce identical bytes and content revisions.""" + source = snapshot() + source["extra"] = metadata + assert explorer.graph_explorer_json(source) == explorer.graph_explorer_json( + reversed_mapping_order(source) + ) + + +@settings(deadline=None) +@given(metadata=_JSON_METADATA) +def test_revision_binds_generated_metadata_changes(metadata): + """Every metadata change affects the revision without changing graph identity.""" + source = snapshot() + source["compiled"]["extra_declaration"] = {"payload": metadata, "marker": False} + before = explorer.graph_explorer_document(source) + source["compiled"]["extra_declaration"]["marker"] = True + after = explorer.graph_explorer_document(source) + assert before["revision"] != after["revision"] + assert ( + before["metadata"]["microcosm"]["graph_sha256"] + == after["metadata"]["microcosm"]["graph_sha256"] + ) + assert before["nodes"] == after["nodes"] + assert before["edges"] == after["edges"] From f887e3f1773e55a2d737395632fc6fa4239d1a3d Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:48:47 +0400 Subject: [PATCH 03/13] Derive canonical graph schema from compiler output --- .../src/microcosm/graph/schema.py | 378 ++++++++++++++++++ .../engine_free/shared/test_graph_schema.py | 204 ++++++++++ test_support/microcosm_graph/__init__.py | 1 + test_support/microcosm_graph/schema.py | 102 +++++ 4 files changed, 685 insertions(+) create mode 100644 packages/microcosm-graph/src/microcosm/graph/schema.py create mode 100644 packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py create mode 100644 test_support/microcosm_graph/schema.py diff --git a/packages/microcosm-graph/src/microcosm/graph/schema.py b/packages/microcosm-graph/src/microcosm/graph/schema.py new file mode 100644 index 000000000..f917b90d9 --- /dev/null +++ b/packages/microcosm-graph/src/microcosm/graph/schema.py @@ -0,0 +1,378 @@ +"""Canonical, compiler-derived metadata for graph presentation adapters. + +The schema contains declarations and compile-time relationships only. It does +not contain Frame values, source bytes, kernel results, cache state, or runtime +receipts. ``input_bindings`` describe the declared fields an operation can +read; they do not claim that a kernel actually read every supplied value. +""" + +from __future__ import annotations + +import hashlib +import json +import math +from collections.abc import Mapping + +from .canonical import canonical_json +from .decl import ( + ROWS_ALL, + CompiledGraph, + GraphError, + Node, + Owned, + StructuralDelta, + compile_graph, +) +from .serialize import graph_from_json, graph_to_json + +__all__ = ["graph_schema", "validate_graph_schema"] + +PROTOCOL = "microcosm.graph.schema.v1" +_ROOT_KEYS = { + "protocol", + "country", + "graph_sha256", + "graph", + "compiled", + "fields", + "input_bindings", + "extensions", +} +_MAX_ITEMS = 2_000_000 +_MAX_BYTES = 32 * 1024 * 1024 +_MAX_DEPTH = 64 +_MAX_STRING = 16_384 + + +def _plain_json(value: object) -> object: + """Return a bounded detached JSON value without invoking custom objects.""" + + items = 0 + charge = 0 + + def walk(child: object, depth: int) -> object: + nonlocal items, charge + items += 1 + charge += 16 + if items > _MAX_ITEMS or depth > _MAX_DEPTH: + raise ValueError("Graph schema exceeds the metadata complexity limit.") + kind = type(child) + if child is None or kind is bool: + result = child + elif kind is str: + if len(child) > _MAX_STRING: + raise ValueError("Graph schema string exceeds the length limit.") + charge += len(child.encode("utf-8")) + result = child + elif kind is int: + if child.bit_length() > 1024: + raise ValueError("Graph schema integer exceeds the size limit.") + charge += child.bit_length() + result = child + elif kind is float: + if not math.isfinite(child): + raise ValueError("Graph schema requires finite numbers.") + result = child + elif kind is list: + result = [walk(item, depth + 1) for item in child] + elif kind is dict: + if not all(type(key) is str for key in child): + raise ValueError("Graph schema requires string object keys.") + result = { + walk(key, depth + 1): walk(item, depth + 1) + for key, item in child.items() + } + else: + raise ValueError("Graph schema accepts plain JSON values only.") + if charge > _MAX_BYTES: + raise ValueError("Graph schema exceeds the metadata size limit.") + return result + + copied = walk(value, 0) + if len(canonical_json(copied)) > _MAX_BYTES: + raise ValueError("Graph schema exceeds the serialized size limit.") + return copied + + +def _compiled_payload(compiled: CompiledGraph) -> dict[str, object]: + return { + "order": list(compiled.order), + "versions": dict(sorted(compiled.versions.items())), + "predecessors": { + node_id: list(compiled.predecessors[node_id]) + for node_id in sorted(compiled.predecessors) + }, + "owners": [ + [*coordinate, owner] + for coordinate, owner in sorted(compiled.owners.items()) + ], + } + + +def _require_compiler_result(compiled: CompiledGraph) -> CompiledGraph: + if not isinstance(compiled, CompiledGraph): + raise TypeError( + f"graph_schema expects CompiledGraph, got {type(compiled).__name__}" + ) + expected = compile_graph(compiled.graph) + if _compiled_payload(compiled) != _compiled_payload(expected): + raise GraphError("CompiledGraph metadata disagrees with compile_graph output.") + return expected + + +def _owned(node: Node, entity: str, column: str) -> Owned: + matches = [ + owned + for owned in node.outputs + if owned.entity == entity and owned.column == column + ] + if len(matches) != 1: + raise GraphError( + f"Compiler owner {node.id!r} has no unique declaration for " + f"{entity}.{column}." + ) + return matches[0] + + +def _declaration( + compiled: CompiledGraph, + by_id: Mapping[str, Node], + version: str, + entity: str, + column: str, + *, + exclude_owner: str | None = None, +) -> tuple[str, Owned]: + """Find the nearest declaration visible in a population version.""" + + current = version + while True: + owner = compiled.owners.get((current, entity, column)) + if owner is not None and owner != exclude_owner: + return owner, _owned(by_id[owner], entity, column) + holder = by_id[current] + if holder.structural is StructuralDelta.CREATE: + raise GraphError( + f"Compiler metadata has no declaration for {entity}.{column} " + f"in population {version!r}." + ) + assert holder.base is not None + current = holder.base + + +def _field( + population: str, + provider: str, + declared_in: str, + owned: Owned, +) -> dict[str, object]: + return { + "population": population, + "entity": owned.entity, + "column": owned.column, + "dtype": owned.dtype, + "provider": provider, + "declared_in": declared_in, + "rows": owned.rows, + "ownership": owned.ownership.value, + "rewrite": owned.rewrite, + } + + +def _fields( + compiled: CompiledGraph, by_id: Mapping[str, Node] +) -> list[dict[str, object]]: + coordinates: dict[str, set[tuple[str, str]]] = {} + fields: list[dict[str, object]] = [] + for population in compiled.order: + holder = by_id[population] + if holder.structural is StructuralDelta.NONE: + continue + visible = ( + set() + if holder.structural is StructuralDelta.CREATE + else set(coordinates[holder.base]) # type: ignore[index] + ) + visible.update( + (entity, column) + for version, entity, column in compiled.owners + if version == population + ) + coordinates[population] = visible + for entity, column in sorted(visible): + owner = compiled.owners.get((population, entity, column)) + declared_in, owned = _declaration( + compiled, by_id, population, entity, column + ) + fields.append( + _field( + population, + population if owner is None else owner, + declared_in, + owned, + ) + ) + return fields + + +def _binding( + compiled: CompiledGraph, + by_id: Mapping[str, Node], + node: Node, + population: str, + entity: str, + column: str, + kind: str, + rows: str, +) -> dict[str, str]: + owner = compiled.owners.get((population, entity, column)) + exclude_owner = node.id if owner == node.id else None + declared_in, _ = _declaration( + compiled, + by_id, + population, + entity, + column, + exclude_owner=exclude_owner, + ) + return { + "node": node.id, + "population": population, + "entity": entity, + "column": column, + "provider": population if exclude_owner is not None else owner or population, + "declared_in": declared_in, + "kind": kind, + "rows": rows, + } + + +def _input_bindings( + compiled: CompiledGraph, by_id: Mapping[str, Node] +) -> list[dict[str, str]]: + bindings: list[dict[str, str]] = [] + for node_id in compiled.order: + node = by_id[node_id] + if node.structural is StructuralDelta.CREATE: + continue + population = ( + compiled.versions[node.id] + if node.structural is StructuralDelta.NONE + else node.base + ) + assert population is not None + for slice_ in node.inputs: + for column in slice_.columns: + bindings.append( + _binding( + compiled, + by_id, + node, + population, + slice_.entity, + column, + "slice", + slice_.rows, + ) + ) + if slice_.rows != ROWS_ALL: + bindings.append( + _binding( + compiled, + by_id, + node, + population, + slice_.entity, + slice_.rows, + "slice_mask", + ROWS_ALL, + ) + ) + for owned in node.outputs: + if owned.rewrite: + bindings.append( + _binding( + compiled, + by_id, + node, + population, + owned.entity, + owned.column, + "rewrite_incumbent", + owned.rows, + ) + ) + if owned.rows != ROWS_ALL: + bindings.append( + _binding( + compiled, + by_id, + node, + population, + owned.entity, + owned.rows, + "output_mask", + ROWS_ALL, + ) + ) + return bindings + + +def graph_schema( + compiled: CompiledGraph, + *, + extensions: Mapping[str, object] | None = None, +) -> dict[str, object]: + """Project a validated ``CompiledGraph`` into portable static metadata. + + ``input_bindings`` and field providers are derived from the same ownership + tables used by execution. No Frame or kernel is loaded, and no runtime value + is included. + """ + + compiled = _require_compiler_result(compiled) + extension_value: object = {} if extensions is None else dict(extensions) + detached_extensions = _plain_json(extension_value) + if type(detached_extensions) is not dict: + raise ValueError("Graph schema extensions must be an object.") + graph_text = graph_to_json(compiled.graph) + graph_value = json.loads(graph_text) + by_id = {node.id: node for node in compiled.graph.nodes} + result = { + "protocol": PROTOCOL, + "country": compiled.graph.country, + "graph_sha256": hashlib.sha256(graph_text.encode("utf-8")).hexdigest(), + "graph": graph_value, + "compiled": _compiled_payload(compiled), + "fields": _fields(compiled, by_id), + "input_bindings": _input_bindings(compiled, by_id), + "extensions": detached_extensions, + } + return _plain_json(result) # type: ignore[return-value] + + +def validate_graph_schema(value: object) -> dict[str, object]: + """Recompile and return a detached canonical schema or reject it. + + Only ``extensions`` is producer-defined. Every other field must equal the + result derived from the embedded declaration by :func:`compile_graph`. + """ + + document = _plain_json(value) + if type(document) is not dict: + raise ValueError("Graph schema must be an object.") + if set(document) != _ROOT_KEYS: + raise ValueError("Graph schema root fields do not match the protocol.") + if document.get("protocol") != PROTOCOL: + raise ValueError("Graph schema protocol is unsupported.") + graph_value = document.get("graph") + if type(graph_value) is not dict: + raise ValueError("Graph schema graph must be an object.") + extensions = document.get("extensions") + if type(extensions) is not dict: + raise ValueError("Graph schema extensions must be an object.") + restored = graph_from_json(canonical_json(graph_value).decode("utf-8")) + expected = graph_schema(compile_graph(restored), extensions=extensions) + if document != expected: + raise ValueError("Graph schema core metadata disagrees with its graph.") + return expected diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py new file mode 100644 index 000000000..d29b43399 --- /dev/null +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py @@ -0,0 +1,204 @@ +"""Canonical compiler metadata for portable graph presentation.""" + +from __future__ import annotations + +import copy + +import pytest + +from microcosm.graph.schema import graph_schema, validate_graph_schema +from test_support.microcosm_graph.schema import ( + compiled_default_population_graph, + compiled_graph, +) + + +def _field(schema, population, column): + return next( + field + for field in schema["fields"] + if field["population"] == population and field["column"] == column + ) + + +def test_schema_is_derived_from_the_compiled_graph(): + compiled = compiled_graph() + schema = graph_schema(compiled) + + assert set(schema) == { + "protocol", + "country", + "graph_sha256", + "graph", + "compiled", + "fields", + "input_bindings", + "extensions", + } + assert schema["protocol"] == "microcosm.graph.schema.v1" + assert schema["country"] == "invented" + assert schema["compiled"] == { + "order": list(compiled.order), + "versions": dict(sorted(compiled.versions.items())), + "predecessors": { + node_id: list(compiled.predecessors[node_id]) + for node_id in sorted(compiled.predecessors) + }, + "owners": [ + [*coordinate, owner] + for coordinate, owner in sorted(compiled.owners.items()) + ], + } + assert validate_graph_schema(schema) == schema + + +def test_fields_record_carriage_rewrites_and_nearest_declaration(): + schema = graph_schema(compiled_graph()) + + assert _field(schema, "base", "age") == { + "population": "base", + "entity": "person", + "column": "age", + "dtype": "int64", + "provider": "base", + "declared_in": "base", + "rows": "all", + "ownership": "produced", + "rewrite": False, + } + assert _field(schema, "filtered", "age") == { + "population": "filtered", + "entity": "person", + "column": "age", + "dtype": "int64", + "provider": "rewrite", + "declared_in": "rewrite", + "rows": "keep", + "ownership": "produced", + "rewrite": True, + } + assert _field(schema, "filtered", "keep")["provider"] == "filtered" + assert _field(schema, "filtered", "keep")["declared_in"] == "base" + assert _field(schema, "expanded", "age")["provider"] == "expanded" + assert _field(schema, "expanded", "age")["declared_in"] == "rewrite" + assert _field(schema, "reweighted", "age")["provider"] == "reweighted" + assert len(schema["fields"]) == 8 + + +def test_input_bindings_preserve_each_declared_read_role(): + bindings = graph_schema(compiled_graph())["input_bindings"] + rewrite = [binding for binding in bindings if binding["node"] == "rewrite"] + + assert rewrite == [ + { + "node": "rewrite", + "population": "filtered", + "entity": "person", + "column": "age", + "provider": "filtered", + "declared_in": "base", + "kind": "slice", + "rows": "keep", + }, + { + "node": "rewrite", + "population": "filtered", + "entity": "person", + "column": "keep", + "provider": "filtered", + "declared_in": "base", + "kind": "slice", + "rows": "keep", + }, + { + "node": "rewrite", + "population": "filtered", + "entity": "person", + "column": "keep", + "provider": "filtered", + "declared_in": "base", + "kind": "slice_mask", + "rows": "all", + }, + { + "node": "rewrite", + "population": "filtered", + "entity": "person", + "column": "age", + "provider": "filtered", + "declared_in": "base", + "kind": "rewrite_incumbent", + "rows": "keep", + }, + { + "node": "rewrite", + "population": "filtered", + "entity": "person", + "column": "keep", + "provider": "filtered", + "declared_in": "base", + "kind": "output_mask", + "rows": "all", + }, + ] + assert ( + next(binding for binding in bindings if binding["node"] == "expanded")[ + "provider" + ] + == "rewrite" + ) + + +def test_omitted_population_uses_the_compiler_resolved_version(): + compiled = compiled_default_population_graph() + schema = graph_schema(compiled) + + assert schema["graph"]["nodes"][1]["population"] is None + assert schema["compiled"]["versions"]["consumer"] == "base" + assert schema["input_bindings"] == [ + { + "node": "consumer", + "population": "base", + "entity": "person", + "column": "age", + "provider": "base", + "declared_in": "base", + "kind": "slice", + "rows": "all", + } + ] + + +@pytest.mark.parametrize( + "mutation", + [ + lambda value: value.update(extra=True), + lambda value: value["compiled"]["order"].reverse(), + lambda value: value["compiled"]["versions"].update(rewrite="base"), + lambda value: value["compiled"]["predecessors"]["rewrite"].clear(), + lambda value: value["compiled"]["owners"].clear(), + lambda value: value["fields"][0].update(provider="rewrite"), + lambda value: value["input_bindings"][0].update(provider="rewrite"), + ], +) +def test_validation_recompiles_and_refuses_core_metadata_changes(mutation): + schema = graph_schema(compiled_graph()) + mutation(schema) + with pytest.raises(ValueError): + validate_graph_schema(schema) + + +def test_extensions_are_detached_bounded_json(): + extension = {"producer": {"labels": ["one", 2**80]}} + schema = graph_schema(compiled_graph(), extensions=extension) + extension["producer"]["labels"].append("changed") + assert schema["extensions"] == {"producer": {"labels": ["one", 2**80]}} + + copied = validate_graph_schema(schema) + copied["extensions"]["producer"]["labels"].append("output change") + assert schema["extensions"] == {"producer": {"labels": ["one", 2**80]}} + + invalid = copy.deepcopy(schema) + invalid["extensions"] = {"bad": float("nan")} + with pytest.raises(ValueError, match="finite"): + validate_graph_schema(invalid) diff --git a/test_support/microcosm_graph/__init__.py b/test_support/microcosm_graph/__init__.py index e69de29bb..3b4d37e91 100644 --- a/test_support/microcosm_graph/__init__.py +++ b/test_support/microcosm_graph/__init__.py @@ -0,0 +1 @@ +"""Shared fixtures for microcosm-graph tests.""" diff --git a/test_support/microcosm_graph/schema.py b/test_support/microcosm_graph/schema.py new file mode 100644 index 000000000..7235729b8 --- /dev/null +++ b/test_support/microcosm_graph/schema.py @@ -0,0 +1,102 @@ +"""Compiler-only graph fixtures for schema and Orrery adapter tests.""" + +from __future__ import annotations + +from microcosm.graph import ( + Graph, + Node, + Owned, + Slice, + SourceRef, + StructuralDelta, + WeightUpdate, + compile_graph, +) + + +def compiled_graph(): + """Return a graph covering carried, rewritten, and structural fields.""" + + graph = Graph( + "invented", + sources=(SourceRef("survey", "unused@1", "Recorded survey tables"),), + nodes=( + Node( + "base", + "unused.create@1", + structural=StructuralDelta.CREATE, + sources=("survey",), + outputs=( + Owned("person", "age", "int64"), + Owned("person", "keep", "boolean"), + ), + description="Load the declared source tables", + ), + Node( + "filtered", + "unused.filter@1", + structural=StructuralDelta.FILTER, + base="base", + inputs=(Slice("person", ("keep",)),), + description="Retain selected people", + ), + Node( + "rewrite", + "unused.rewrite@1", + population="filtered", + inputs=(Slice("person", ("age", "keep"), rows="keep"),), + outputs=( + Owned( + "person", + "age", + "int64", + rows="keep", + rewrite=True, + ), + ), + description="Replace age for selected people", + citation="Example method", + ), + Node( + "expanded", + "unused.expand@1", + structural=StructuralDelta.EXPAND, + base="filtered", + inputs=(Slice("person", ("age",)),), + ), + Node( + "reweighted", + "unused.reweight@1", + structural=StructuralDelta.REWEIGHT, + base="expanded", + weights=WeightUpdate("person", "design", "normalize sample weights"), + mass="declared", + ), + ), + ) + return compile_graph(graph) + + +def compiled_default_population_graph(): + """Return a legal graph whose ordinary node omits its sole population.""" + + return compile_graph( + Graph( + "invented", + sources=(SourceRef("survey", "unused@1"),), + nodes=( + Node( + "base", + "unused.create@1", + structural=StructuralDelta.CREATE, + sources=("survey",), + outputs=(Owned("person", "age", "int64"),), + ), + Node( + "consumer", + "unused.consume@1", + inputs=(Slice("person", ("age",)),), + ), + ), + ) + ) From d6faf620a06e515567d0825ff54c5225939774a8 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:55:56 +0400 Subject: [PATCH 04/13] Harden and rename the Orrery adapter --- .../src/microcosm/graph/explorer.py | 492 --------------- .../src/microcosm/graph/orrery.py | 403 +++++++++++++ .../engine_free/shared/test_graph_orrery.py | 213 +++++++ .../microcosm-graph/tests/test_explorer.py | 563 ------------------ test_support/microcosm_graph/schema.py | 8 + 5 files changed, 624 insertions(+), 1055 deletions(-) delete mode 100644 packages/microcosm-graph/src/microcosm/graph/explorer.py create mode 100644 packages/microcosm-graph/src/microcosm/graph/orrery.py create mode 100644 packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py delete mode 100644 packages/microcosm-graph/tests/test_explorer.py diff --git a/packages/microcosm-graph/src/microcosm/graph/explorer.py b/packages/microcosm-graph/src/microcosm/graph/explorer.py deleted file mode 100644 index 981f18d23..000000000 --- a/packages/microcosm-graph/src/microcosm/graph/explorer.py +++ /dev/null @@ -1,492 +0,0 @@ -"""Pure presentation adapter from compiler metadata to graph-explorer/v1. - -The input is a microcosm.graph.schema.v1 export, not a Frame or RunManifest. -This module checks presentation references, not compilation or authenticity. -It never imports a kernel, opens a source, or infers an execution verdict. -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import math -import os -import stat -from collections import Counter -from pathlib import Path - -from .canonical import canonical_json - -__all__ = ["graph_explorer_document", "graph_explorer_json"] - -_PROTOCOL = "microcosm.graph.schema.v1" -_SAFE_INTEGER = 2**53 - 1 -_MAX_INPUT_BYTES = 32 * 1024 * 1024 -_MAX_OUTPUT_BYTES = 64 * 1024 * 1024 -_MAX_ITEMS = 2_000_000 -_MAX_NODES = 20_000 -_MAX_EDGES = 100_000 -_FIELD_KEYS = ("population", "entity", "column", "producer", "declared_in") -_READ_KINDS = {"slice", "slice_mask", "output_mask", "rewrite_incumbent"} - - -def _require(condition: bool, message: str) -> None: - if not condition: - raise ValueError(f"Graph explorer: {message}.") - - -def _copy_json(value: object, *, transport: bool = False) -> object: - """Bound and detach plain JSON; reject custom objects before invoking them.""" - items, charge = 0, 0 - - def walk(child: object, depth: int) -> object: - nonlocal items, charge - items += 1 - charge += 16 - _require(items <= _MAX_ITEMS and depth <= 64, "metadata complexity") - kind = type(child) - if kind is str: - _require(len(child) <= 16_384, "string length") - charge += len(child) * 6 - result = child - elif child is None or kind is bool: - result = child - elif kind is int: - _require(child.bit_length() <= 1024, "integer size") - charge += child.bit_length() - result = ( - {"integer_literal": str(child)} - if transport and abs(child) > _SAFE_INTEGER - else child - ) - elif kind is float: - _require(math.isfinite(child), "finite numbers required") - # JS cannot distinguish a large integral float from an unsafe int. - result = ( - {"float_literal": repr(child)} - if transport and child.is_integer() and abs(child) > _SAFE_INTEGER - else child - ) - elif kind is list: - _require(len(child) <= _MAX_ITEMS - items, "array size") - result = [walk(item, depth + 1) for item in child] - elif kind is dict: - _require(len(child) <= (_MAX_ITEMS - items) // 2, "object size") - _require(all(type(key) is str for key in child), "string keys required") - result = { - walk(key, depth + 1): walk(item, depth + 1) - for key, item in child.items() - } - else: - raise ValueError("Graph explorer: plain JSON values required.") - _require(charge <= _MAX_OUTPUT_BYTES, "prospective metadata size") - return result - - return walk(value, 0) - - -def _object(value: object) -> dict: - _require(type(value) is dict, "object required") - return value - - -def _array(value: object) -> list: - _require(type(value) is list, "array required") - return value - - -def _text(value: object) -> str: - _require(type(value) is str and bool(value.strip()), "nonempty text required") - return value - - -def _index(values: object, key: str) -> dict[str, dict]: - result = {} - for item in _array(values): - name = _text(_object(item).get(key)) - _require(name not in result, f"duplicate {key}") - result[name] = item - return result - - -def _id(*parts: str) -> str: - return json.dumps(parts, ensure_ascii=False, separators=(",", ":")) - - -def _field_id(field: dict) -> str: - return _id("field", *(_text(field.get(key)) for key in _FIELD_KEYS)) - - -def _revision(value: object) -> str: - return "sha256:" + hashlib.sha256(canonical_json(value)).hexdigest() - - -def _bounded_json(value: object, limit: int) -> str: - parts, size = [], 0 - encoder = json.JSONEncoder( - ensure_ascii=False, allow_nan=False, sort_keys=True, separators=(",", ":") - ) - for part in encoder.iterencode(value): - size += len(part.encode("utf-8")) - _require(size <= limit, "serialized metadata size") - parts.append(part) - return "".join(parts) - - -def graph_explorer_document(schema: dict, *, title: str | None = None) -> dict: - """Export the entire supplied schema snapshot; never silently truncate. - - Use ``graph_schema(compiled)`` as the producer where available. Imported - dictionaries remain declarations: reference checks and content digests do - not authenticate them or establish that a compiler or kernel actually ran. - The document revision binds all supplied metadata, including compiled tables; - node/edge revisions bind their presentation records, not runtime cache keys. - - Pre-rewrite inputs have distinct field IDs in the same population. Structural - edges mean ancestry, not equality. Reads apply to an operation, without - claiming that each input mathematically determines each individual output. - """ - doc = _object(_copy_json(schema)) - _bounded_json(doc, _MAX_INPUT_BYTES) - _require(doc.get("protocol") == _PROTOCOL, "unsupported schema protocol") - country = _text(doc.get("country")) - graph = _object(doc.get("graph")) - _require(graph.get("country") == country, "country mismatch") - digest = hashlib.sha256(canonical_json(graph)).hexdigest() - _require(doc.get("graph_sha256") == digest, "declaration digest mismatch") - operations = _index(graph.get("nodes"), "id") - sources = _index(graph.get("sources"), "name") - _require(len(operations) <= 256 and len(sources) <= 256, "declaration count") - compiled = _object(doc.get("compiled")) - order = _array(compiled.get("order")) - _require(all(type(item) is str for item in order), "order IDs") - _require(len(order) == len(operations) and set(order) == set(operations), "order") - predecessors = _object(compiled.get("predecessors")) - versions = _object(compiled.get("versions")) - _require(set(predecessors) == set(operations) == set(versions), "compiled IDs") - positions = {name: index for index, name in enumerate(order)} - populations = { - name for name, op in operations.items() if op.get("structural") != "none" - } - for name, op in operations.items(): - _require( - _text(op.get("structural")) - in {"none", "create", "filter", "expand", "reweight"}, - "structural kind", - ) - _require(_text(versions[name]) in populations, "population reference") - _require( - versions[name] == (name if name in populations else op.get("population")), - "population binding", - ) - base = op.get("base") - _require(base is None or _text(base) in populations, "base reference") - parents = _array(predecessors[name]) - _require(all(type(parent) is str for parent in parents), "predecessor IDs") - _require(len(set(parents)) == len(parents), "duplicate predecessor") - _require( - all( - parent in positions and positions[parent] < positions[name] - for parent in parents - ), - "predecessor order", - ) - - nodes, edges = {}, {} - # Reserve wrapper metadata, then charge each detached transport record before - # retaining it. Counts alone would allow long repeated IDs to over-expand. - transport_doc = _copy_json(doc, transport=True) - used = len(_bounded_json(transport_doc, _MAX_OUTPUT_BYTES).encode("utf-8")) + 65_536 - - def presentation_record(record: dict) -> dict: - nonlocal used - record = _object(_copy_json(record, transport=True)) - record["revision"] = _revision(record) - encoded = _bounded_json(record, _MAX_OUTPUT_BYTES - used) - used += len(encoded.encode("utf-8")) + 1 - return record - - def add_node(node: dict) -> None: - _require(node["id"] not in nodes, "duplicate presentation node") - _require(len(nodes) < _MAX_NODES, "node limit") - nodes[node["id"]] = presentation_record(node) - - def add_edge(source: str, target: str, kind: str, data: dict | None = None) -> None: - facts = {} if data is None else data - identity = _id( - "edge", kind, source, target, _bounded_json(facts, _MAX_INPUT_BYTES) - ) - if identity in edges: - return - _require(len(edges) < _MAX_EDGES, "edge limit") - edge = { - "id": identity, - "source": source, - "target": target, - "kind": kind, - "category": "dependency", - "data": facts, - } - edges[identity] = presentation_record(edge) - - declarations = {} - for name, op in operations.items(): - add_node( - { - "id": _id("operation", name), - "label": name, - "kind": "operation", - "data": {"declaration": op, "population": versions[name]}, - } - ) - for output in _array(op.get("outputs")): - output = _object(output) - key = (name, _text(output.get("entity")), _text(output.get("column"))) - _require(key not in declarations, "duplicate Owned declaration") - # The real serializer omits rewrite=False. Normalize only this - # internal lookup; preserve the authored declaration byte-for-byte. - rewrite = output.get("rewrite", False) - _require(type(rewrite) is bool, "rewrite flag") - declarations[key] = {**output, "rewrite": rewrite} - for name, source in sources.items(): - _text(source.get("codec")) - add_node( - { - "id": _id("source", name), - "label": name, - "kind": "source", - "data": {"declaration": source}, - } - ) - - fields, visible = {}, {} - - def add_field(field: dict, *, visible_field: bool) -> str: - identity = _field_id(field) - key = (field["declared_in"], field["entity"], field["column"]) - _require(key in declarations, "field declaration reference") - owned = declarations[key] - _require( - field["population"] in populations and field["producer"] in operations, - "field reference", - ) - _require( - versions[field["producer"]] == field["population"], - "field provider population", - ) - for prop in ("dtype", "rows", "ownership", "rewrite"): - _require( - field.get(prop) == owned.get(prop), "field declaration disagreement" - ) - if identity not in fields: - fields[identity] = field - add_node( - { - "id": identity, - "label": f"{field['entity']}.{field['column']}", - "kind": "field", - "data": {**field, "visible_in_schema": visible_field}, - } - ) - add_edge(_id("operation", field["producer"]), identity, "produced") - else: - _require(fields[identity] == field, "ambiguous field value") - return identity - - for field in _array(doc.get("schema")): - field = _object(field) - coordinate = tuple(_text(field.get(key)) for key in _FIELD_KEYS[:3]) - _require(coordinate not in visible, "duplicate visible field") - visible[coordinate] = add_field(field, visible_field=True) - - # Check role coverage against declarations, without reimplementing the - # compiler's provider resolution. Preserve repeated declared roles too. - expected_reads = Counter() - for name, op in operations.items(): - if op["structural"] == "create": - continue - population = versions[name] if op["structural"] == "none" else op["base"] - for slice_ in _array(op.get("inputs")): - slice_ = _object(slice_) - entity, rows = _text(slice_.get("entity")), _text(slice_.get("rows")) - for column in _array(slice_.get("columns")): - expected_reads[ - name, population, entity, _text(column), rows, "slice" - ] += 1 - if rows != "all": - expected_reads[name, population, entity, rows, "all", "slice_mask"] += 1 - for output in _array(op.get("outputs")): - entity, column, rows = ( - _text(output.get(key)) for key in ("entity", "column", "rows") - ) - if output.get("rewrite", False): - expected_reads[ - name, population, entity, column, rows, "rewrite_incumbent" - ] += 1 - if rows != "all": - expected_reads[ - name, population, entity, rows, "all", "output_mask" - ] += 1 - reads = _array(doc.get("input_bindings")) - _require(len(reads) <= 100_000, "read count") - actual_reads = Counter( - tuple( - _text(_object(read).get(key)) - for key in ("node", "population", "entity", "column", "rows", "kind") - ) - for read in reads - ) - _require(actual_reads == expected_reads, "declared read coverage") - for read in reads: - read = _object(read) - _require( - read.get("node") in operations and read.get("kind") in _READ_KINDS, - "read reference or kind", - ) - _text(read.get("rows")) - key = tuple(_text(read.get(k)) for k in ("declared_in", "entity", "column")) - _require(key in declarations, "read declaration reference") - owned = declarations[key] - field = {key: read[key] for key in _FIELD_KEYS} - field.update( - {key: owned[key] for key in ("dtype", "rows", "ownership", "rewrite")} - ) - identity = add_field(field, visible_field=False) - # Preserve explicit role labels alongside the supplied compiled edges. - # This adapter does not infer a new compiled dependency from a read. - add_edge( - identity, - _id("operation", read["node"]), - "declared_read", - {"read_kind": read["kind"], "rows": read["rows"]}, - ) - - for name, op in operations.items(): - target = _id("operation", name) - for parent in predecessors[name]: - add_edge(_id("operation", parent), target, "compiled_predecessor") - if op.get("base") is not None: - # Every field in the completed base is carried. Values may change - # during structural materialization; this is not an identity edge. - for (population, _entity, _column), field_id in visible.items(): - if population == op["base"]: - add_edge(field_id, target, "structural_input") - for source in _array(op.get("sources")): - _require(type(source) is str and source in sources, "source reference") - add_edge(_id("source", source), target, "source") - for binding in _array(op.get("artifact_inputs", [])): - binding = _object(binding) - producer = binding.get("producer") - _require( - type(producer) is str and producer in operations, "artifact producer" - ) - outputs = _index(operations[producer].get("artifact_outputs", []), "name") - artifact = _text(binding.get("artifact")) - _require( - artifact in outputs - and outputs[artifact].get("type") == binding.get("type"), - "artifact type or output", - ) - _require(producer in predecessors[name], "artifact predecessor") - add_edge(_id("operation", producer), target, "artifact", binding) - - _require( - all( - edge["source"] in nodes and edge["target"] in nodes - for edge in edges.values() - ), - "dangling edge", - ) - result = { - "schemaVersion": "graph-explorer/v1", - "id": _id("microcosm", country), - "title": _text(title) - if title is not None - else f"{country.upper()} graph schema", - "revision": _revision(doc), - "description": "Complete supplied declaration metadata; no execution or release verdict.", - "nodes": [nodes[key] for key in sorted(nodes)], - "edges": [edges[key] for key in sorted(edges)], - "metadata": { - "adapter": "microcosm.graph.explorer.v1", - "scope": "complete_supplied_schema", - "truncated": False, - "evidence_scope": "Imported declaration metadata; not recompiled or authenticated by this adapter.", - "revision_scope": "Content digests of supplied metadata/presentation records; not runtime cache keys.", - "edge_scope": "Operation-level reads and structural ancestry; not individual-output formulas or value equality.", - "microcosm": transport_doc, - "missing_schema": [ - "entity IDs and memberships", - "units", - "period semantics", - "source preparation internals", - "implicit runtime reads", - ], - }, - } - result = _copy_json(result, transport=True) - _bounded_json(result, _MAX_OUTPUT_BYTES) - return result - - -def graph_explorer_json(schema: dict, *, title: str | None = None) -> str: - """Deterministic UTF-8-ready JSON, with exact large-number transport tags. - - Write these bytes directly for host-side snapshot hashing. HTML bundling and - Receipt verification belong to the shared explorer, not this adapter. - """ - return ( - _bounded_json(graph_explorer_document(schema, title=title), _MAX_OUTPUT_BYTES) - + "\n" - ) - - -def _unique_json_object(pairs: list[tuple[str, object]]) -> dict: - result = {} - for key, value in pairs: - _require(key not in result, "duplicate JSON key") - result[key] = value - return result - - -def _invalid_json_constant(value: str) -> None: - raise ValueError("Graph explorer: finite JSON numbers required.") - - -def main(argv: list[str] | None = None) -> int: - """Convert saved schema metadata for Orrery without loading a saved run.""" - parser = argparse.ArgumentParser(description=main.__doc__) - parser.add_argument("--input", type=Path, required=True, help="saved schema JSON") - parser.add_argument( - "--output", type=Path, required=True, help="new graph-explorer/v1 JSON file" - ) - parser.add_argument("--title", help="viewer document title") - args = parser.parse_args(argv) - try: - # Refuse streams/devices before reading; a FIFO must not block the CLI. - descriptor = os.open(args.input, os.O_RDONLY | getattr(os, "O_NONBLOCK", 0)) - with os.fdopen(descriptor, "rb") as stream: - info = os.fstat(stream.fileno()) - _require(stat.S_ISREG(info.st_mode), "input must be a regular file") - _require(info.st_size <= _MAX_INPUT_BYTES, "input byte limit") - raw = stream.read(_MAX_INPUT_BYTES + 1) - _require(len(raw) <= _MAX_INPUT_BYTES, "input byte limit") - schema = json.loads( - raw.decode("utf-8"), - object_pairs_hook=_unique_json_object, - parse_constant=_invalid_json_constant, - ) - rendered = graph_explorer_json(schema, title=args.title) - args.output.parent.mkdir(parents=True, exist_ok=True) - # Exclusive creation also prevents overwriting the source through an alias. - with args.output.open("x", encoding="utf-8", newline="\n") as stream: - stream.write(rendered) - except (OSError, ValueError, RecursionError) as error: - parser.error(str(error)) - print(f"wrote {args.output}") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/packages/microcosm-graph/src/microcosm/graph/orrery.py b/packages/microcosm-graph/src/microcosm/graph/orrery.py new file mode 100644 index 000000000..d245f22f2 --- /dev/null +++ b/packages/microcosm-graph/src/microcosm/graph/orrery.py @@ -0,0 +1,403 @@ +"""Pure adapter from compiler metadata to Orrery's ``graph-explorer/v1``. + +The adapter presents static declarations. It never opens a source, loads a +Frame or kernel, or infers execution, verification, or release status. +""" + +from __future__ import annotations + +import hashlib +import json +import math + +from .canonical import canonical_json +from .schema import validate_graph_schema + +__all__ = ["orrery_document_from_schema", "orrery_json_from_schema"] + +_SAFE_INTEGER = 2**53 - 1 +_MAX_OUTPUT_BYTES = 64 * 1024 * 1024 +_MAX_ITEMS = 2_000_000 +_MAX_NODES = 20_000 +_MAX_EDGES = 100_000 +_FIELD_KEYS = ("population", "entity", "column", "provider", "declared_in") + + +def _require(condition: bool, message: str) -> None: + if not condition: + raise ValueError(f"Orrery adapter: {message}.") + + +def _transport_json(value: object) -> object: + """Detach JSON and tag numeric values JavaScript cannot preserve exactly.""" + + items = 0 + charge = 0 + + def walk(child: object, depth: int) -> object: + nonlocal items, charge + items += 1 + charge += 16 + _require(items <= _MAX_ITEMS and depth <= 64, "metadata complexity") + kind = type(child) + if child is None or kind is bool: + result = child + elif kind is str: + _require(len(child) <= 16_384, "string length") + charge += len(child.encode("utf-8")) + result = child + elif kind is int: + _require(child.bit_length() <= 1024, "integer size") + charge += child.bit_length() + result = ( + {"integer_literal": str(child)} if abs(child) > _SAFE_INTEGER else child + ) + elif kind is float: + _require(math.isfinite(child), "finite numbers required") + result = ( + {"float_literal": repr(child)} + if child.is_integer() and abs(child) > _SAFE_INTEGER + else child + ) + elif kind is list: + result = [walk(item, depth + 1) for item in child] + elif kind is dict: + _require(all(type(key) is str for key in child), "string keys required") + result = { + walk(key, depth + 1): walk(item, depth + 1) + for key, item in child.items() + } + else: + raise ValueError("Orrery adapter: plain JSON values required.") + _require(charge <= _MAX_OUTPUT_BYTES, "prospective metadata size") + return result + + return walk(value, 0) + + +def _object(value: object, label: str) -> dict[str, object]: + _require(type(value) is dict, f"{label} must be an object") + return value # type: ignore[return-value] + + +def _array(value: object, label: str) -> list[object]: + _require(type(value) is list, f"{label} must be an array") + return value # type: ignore[return-value] + + +def _text(value: object, label: str) -> str: + _require(type(value) is str and bool(value.strip()), f"{label} must be text") + return value # type: ignore[return-value] + + +def _index(values: object, key: str, label: str) -> dict[str, dict[str, object]]: + result: dict[str, dict[str, object]] = {} + for item in _array(values, label): + record = _object(item, f"{label} item") + identity = _text(record.get(key), f"{label} {key}") + _require(identity not in result, f"duplicate {label} {key}") + result[identity] = record + return result + + +def _id(*parts: str) -> str: + return json.dumps(parts, ensure_ascii=False, separators=(",", ":")) + + +def _field_id(field: dict[str, object]) -> str: + return _id("field", *(_text(field.get(key), f"field {key}") for key in _FIELD_KEYS)) + + +def _revision(value: object) -> str: + return "sha256:" + hashlib.sha256(canonical_json(value)).hexdigest() + + +def _bounded_json(value: object, limit: int) -> str: + parts: list[str] = [] + size = 0 + encoder = json.JSONEncoder( + ensure_ascii=False, allow_nan=False, sort_keys=True, separators=(",", ":") + ) + for part in encoder.iterencode(value): + size += len(part.encode("utf-8")) + _require(size <= limit, "serialized metadata size") + parts.append(part) + return "".join(parts) + + +def _owned_declarations( + operations: dict[str, dict[str, object]], +) -> dict[tuple[str, str, str], dict[str, object]]: + result: dict[tuple[str, str, str], dict[str, object]] = {} + for node_id, operation in operations.items(): + for value in _array(operation.get("outputs"), f"operation {node_id} outputs"): + owned = _object(value, f"operation {node_id} output") + key = ( + node_id, + _text(owned.get("entity"), "output entity"), + _text(owned.get("column"), "output column"), + ) + _require(key not in result, "duplicate Owned declaration") + result[key] = {**owned, "rewrite": owned.get("rewrite", False)} + return result + + +def orrery_document_from_schema( + schema: object, *, title: str | None = None +) -> dict[str, object]: + """Transform compiler metadata into a complete Orrery document. + + The schema is recompiled before projection. Stable node and edge identities + are derived from declared coordinates and relationships. Revisions are + content digests, not runtime cache keys or verification claims. + """ + + document = validate_graph_schema(schema) + graph = _object(document["graph"], "graph") + compiled = _object(document["compiled"], "compiled") + operations = _index(graph.get("nodes"), "id", "operations") + sources = _index(graph.get("sources"), "name", "sources") + versions = _object(compiled.get("versions"), "compiled versions") + predecessors = _object(compiled.get("predecessors"), "compiled predecessors") + declarations = _owned_declarations(operations) + transport_document = _transport_json(document) + + nodes: dict[str, dict[str, object]] = {} + edges: dict[str, dict[str, object]] = {} + used = ( + len(_bounded_json(transport_document, _MAX_OUTPUT_BYTES).encode("utf-8")) + + 65_536 + ) + + def presentation_record(record: dict[str, object]) -> dict[str, object]: + nonlocal used + transported = _object(_transport_json(record), "presentation record") + transported["revision"] = _revision(transported) + encoded = _bounded_json(transported, _MAX_OUTPUT_BYTES - used) + used += len(encoded.encode("utf-8")) + 1 + return transported + + def add_node(node: dict[str, object]) -> None: + identity = _text(node.get("id"), "presentation node id") + _require(identity not in nodes, "duplicate presentation node") + _require(len(nodes) < _MAX_NODES, "node limit") + nodes[identity] = presentation_record(node) + + def add_edge( + source: str, + target: str, + kind: str, + category: str, + data: dict[str, object] | None = None, + ) -> None: + facts = {} if data is None else data + identity = _id( + "edge", kind, source, target, _bounded_json(facts, _MAX_OUTPUT_BYTES) + ) + if identity in edges: + return + _require(len(edges) < _MAX_EDGES, "edge limit") + edges[identity] = presentation_record( + { + "id": identity, + "source": source, + "target": target, + "kind": kind, + "category": category, + "data": facts, + } + ) + + for node_id, operation in operations.items(): + node: dict[str, object] = { + "id": _id("operation", node_id), + "label": node_id, + "kind": "operation", + "data": { + "declaration": operation, + "population": versions[node_id], + }, + } + if operation.get("description"): + node["description"] = operation["description"] + add_node(node) + for source_name, source in sources.items(): + node = { + "id": _id("source", source_name), + "label": source_name, + "kind": "source", + "data": {"declaration": source}, + } + if source.get("description"): + node["description"] = source["description"] + add_node(node) + + fields: dict[str, dict[str, object]] = {} + visible: dict[tuple[str, str, str], str] = {} + + def add_field(field: dict[str, object], *, visible_field: bool) -> str: + identity = _field_id(field) + declared_in = _text(field.get("declared_in"), "field declared_in") + entity = _text(field.get("entity"), "field entity") + column = _text(field.get("column"), "field column") + declaration = declarations[(declared_in, entity, column)] + for prop in ("dtype", "rows", "ownership", "rewrite"): + _require( + field.get(prop) == declaration.get(prop), + "field declaration disagreement", + ) + if identity not in fields: + fields[identity] = field + add_node( + { + "id": identity, + "label": f"{entity}.{column}", + "kind": "field", + "data": {**field, "visible_in_schema": visible_field}, + } + ) + add_edge( + _id("operation", _text(field.get("provider"), "field provider")), + identity, + "provided_field", + "provenance", + ) + else: + _require(fields[identity] == field, "ambiguous field value") + return identity + + for value in _array(document["fields"], "fields"): + field = _object(value, "field") + coordinate = ( + _text(field.get("population"), "field population"), + _text(field.get("entity"), "field entity"), + _text(field.get("column"), "field column"), + ) + _require(coordinate not in visible, "duplicate visible field") + visible[coordinate] = add_field(field, visible_field=True) + + for value in _array(document["input_bindings"], "input bindings"): + binding = _object(value, "input binding") + declared_in = _text(binding.get("declared_in"), "binding declared_in") + entity = _text(binding.get("entity"), "binding entity") + column = _text(binding.get("column"), "binding column") + declaration = declarations[(declared_in, entity, column)] + field = { + key: binding[key] + for key in ("population", "entity", "column", "provider", "declared_in") + } + field.update( + {key: declaration[key] for key in ("dtype", "rows", "ownership", "rewrite")} + ) + field_id = add_field(field, visible_field=False) + add_edge( + field_id, + _id("operation", _text(binding.get("node"), "binding node")), + "declared_read", + "dependency", + {"read_kind": binding["kind"], "rows": binding["rows"]}, + ) + + for node_id, operation in operations.items(): + target = _id("operation", node_id) + for parent in _array(predecessors[node_id], f"predecessors for {node_id}"): + add_edge( + _id("operation", _text(parent, "predecessor")), + target, + "compiled_predecessor", + "dependency", + ) + base = operation.get("base") + if base is not None: + base_id = _text(base, "structural base") + for (population, _entity, _column), field_id in visible.items(): + if population == base_id: + add_edge( + field_id, + target, + "structural_input", + "provenance", + ) + for source in _array(operation.get("sources"), f"sources for {node_id}"): + add_edge( + _id("source", _text(source, "source reference")), + target, + "declared_source", + "provenance", + ) + for value in _array( + operation.get("artifact_inputs", []), f"artifact inputs for {node_id}" + ): + binding = _object(value, "artifact input") + add_edge( + _id( + "operation", + _text(binding.get("producer"), "artifact producer"), + ), + target, + "artifact_input", + "dependency", + binding, + ) + + _require( + all( + edge["source"] in nodes and edge["target"] in nodes + for edge in edges.values() + ), + "dangling presentation edge", + ) + country = _text(document["country"], "country") + result: dict[str, object] = { + "schemaVersion": "graph-explorer/v1", + "id": _id("microcosm", country), + "title": _text(title, "title") + if title is not None + else f"{country.upper()} graph schema", + "revision": _revision(document), + "description": ( + "Compiler-derived declaration metadata; no execution, verification, " + "or release result." + ), + "nodes": [nodes[key] for key in sorted(nodes)], + "edges": [edges[key] for key in sorted(edges)], + "metadata": { + "adapter": "microcosm.graph.orrery.v1", + "scope": "complete_compiler_schema", + "truncated": False, + "evidence_scope": ( + "The embedded graph was recompiled; no source bytes, kernel " + "execution, or authenticity check was performed." + ), + "revision_scope": ( + "Content digests of declarations and presentation records; not " + "runtime cache keys." + ), + "edge_scope": ( + "Declared operation reads and compiler dependencies; not " + "individual-output formulas or observed runtime access." + ), + "microcosm": transport_document, + "missing_runtime_data": [ + "entity identifiers and memberships", + "field values", + "source bytes and preparation internals", + "kernel execution and implicit runtime reads", + "runtime receipts and release decisions", + ], + }, + } + transported_result = _object(_transport_json(result), "Orrery document") + _bounded_json(transported_result, _MAX_OUTPUT_BYTES) + return transported_result + + +def orrery_json_from_schema(schema: object, *, title: str | None = None) -> str: + """Return deterministic UTF-8-ready JSON for Orrery.""" + + return ( + _bounded_json( + orrery_document_from_schema(schema, title=title), _MAX_OUTPUT_BYTES + ) + + "\n" + ) diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py new file mode 100644 index 000000000..28b66cf1c --- /dev/null +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py @@ -0,0 +1,213 @@ +"""Orrery presentation over canonical compiler metadata.""" + +from __future__ import annotations + +import copy +import json + +import pytest + +from microcosm.graph import compile_graph, graph_from_json, graph_to_json +from microcosm.graph.canonical import canonical_json +from microcosm.graph.orrery import ( + orrery_document_from_schema, + orrery_json_from_schema, +) +from microcosm.graph.schema import graph_schema +from test_support.microcosm_graph.schema import ( + compiled_default_population_graph, + compiled_graph, +) + + +def _parts(identity): + return json.loads(identity) + + +def test_document_is_complete_static_metadata_with_no_runtime_claims(): + schema = graph_schema(compiled_graph()) + document = orrery_document_from_schema(schema) + + assert document["schemaVersion"] == "graph-explorer/v1" + assert document["metadata"]["microcosm"] == schema + assert document["metadata"]["scope"] == "complete_compiler_schema" + assert document["metadata"]["truncated"] is False + assert len([node for node in document["nodes"] if node["kind"] == "operation"]) == 5 + assert len([node for node in document["nodes"] if node["kind"] == "source"]) == 1 + assert ( + not { + "activities", + "receipts", + "artifacts", + "assessments", + } + & document.keys() + ) + assert all("statuses" not in node for node in document["nodes"]) + assert {edge["category"] for edge in document["edges"]} == { + "dependency", + "provenance", + } + identities = {node["id"] for node in document["nodes"]} + assert {edge["source"] for edge in document["edges"]} | { + edge["target"] for edge in document["edges"] + } <= identities + + +def test_descriptions_are_promoted_and_declarations_remain_available(): + document = orrery_document_from_schema(graph_schema(compiled_graph())) + base = next( + node + for node in document["nodes"] + if _parts(node["id"]) == ["operation", "base"] + ) + rewrite = next( + node + for node in document["nodes"] + if _parts(node["id"]) == ["operation", "rewrite"] + ) + source = next(node for node in document["nodes"] if node["kind"] == "source") + + assert base["description"] == "Load the declared source tables" + assert rewrite["description"] == "Replace age for selected people" + assert rewrite["data"]["declaration"]["citation"] == "Example method" + assert source["description"] == "Recorded survey tables" + + +def test_rewrite_incumbent_has_a_distinct_auxiliary_field(): + document = orrery_document_from_schema(graph_schema(compiled_graph())) + ages = [ + node + for node in document["nodes"] + if node["kind"] == "field" + and node["data"]["population"] == "filtered" + and node["data"]["column"] == "age" + ] + assert len(ages) == 2 + incumbent = next(node for node in ages if node["data"]["provider"] == "filtered") + final = next(node for node in ages if node["data"]["provider"] == "rewrite") + assert incumbent["data"]["visible_in_schema"] is False + assert final["data"]["visible_in_schema"] is True + + target = json.dumps(["operation", "rewrite"], separators=(",", ":")) + reads = [ + edge + for edge in document["edges"] + if edge["target"] == target and edge["kind"] == "declared_read" + ] + assert {edge["data"]["read_kind"] for edge in reads} == { + "slice", + "slice_mask", + "output_mask", + "rewrite_incumbent", + } + rewrite_read = next( + edge for edge in reads if edge["data"]["read_kind"] == "rewrite_incumbent" + ) + assert rewrite_read["source"] == incumbent["id"] + + +def test_dependencies_and_provenance_are_separate_categories(): + document = orrery_document_from_schema(graph_schema(compiled_graph())) + dependency_kinds = { + edge["kind"] for edge in document["edges"] if edge["category"] == "dependency" + } + provenance_kinds = { + edge["kind"] for edge in document["edges"] if edge["category"] == "provenance" + } + assert dependency_kinds == { + "compiled_predecessor", + "declared_read", + "artifact_input", + } + assert provenance_kinds == { + "provided_field", + "structural_input", + "declared_source", + } + + +def test_legal_omitted_population_uses_compiler_metadata(): + document = orrery_document_from_schema( + graph_schema(compiled_default_population_graph()) + ) + consumer = next( + node + for node in document["nodes"] + if _parts(node["id"]) == ["operation", "consumer"] + ) + assert consumer["data"]["declaration"]["population"] is None + assert consumer["data"]["population"] == "base" + + +def test_json_is_deterministic_detached_and_transports_large_numbers(): + schema = graph_schema(compiled_graph(), extensions={"large": 2**80, "value": 1e100}) + before = copy.deepcopy(schema) + + first = orrery_json_from_schema(schema) + second = orrery_json_from_schema(schema) + assert first == second and first.endswith("\n") + document = json.loads(first) + assert document["metadata"]["microcosm"]["extensions"] == { + "large": {"integer_literal": str(2**80)}, + "value": {"float_literal": "1e+100"}, + } + document["metadata"]["microcosm"]["graph"]["nodes"].clear() + assert schema == before + + +def test_revisions_ignore_mapping_order_but_bind_descriptions(): + schema = graph_schema(compiled_graph()) + reordered = json.loads( + json.dumps(schema, sort_keys=True), + object_pairs_hook=lambda pairs: dict(reversed(pairs)), + ) + assert orrery_json_from_schema(schema) == orrery_json_from_schema(reordered) + + edited_graph = copy.deepcopy(schema["graph"]) + edited_graph["nodes"][0]["description"] = "Changed description" + edited = graph_schema( + compile_graph(graph_from_json(canonical_json(edited_graph).decode())) + ) + assert ( + orrery_document_from_schema(schema)["revision"] + != orrery_document_from_schema(edited)["revision"] + ) + + +@pytest.mark.parametrize( + "mutation", + [ + lambda value: value.update(extra=True), + lambda value: value["compiled"]["order"].reverse(), + lambda value: value["compiled"]["owners"].clear(), + lambda value: value["fields"][0].update(provider="unknown"), + lambda value: value["input_bindings"][0].update(kind="invented"), + ], +) +def test_invalid_schema_fails_before_producing_a_partial_document(mutation): + schema = graph_schema(compiled_graph()) + mutation(schema) + with pytest.raises(ValueError): + orrery_document_from_schema(schema) + + +def test_output_complexity_and_size_limits_are_enforced(monkeypatch): + schema = graph_schema(compiled_graph()) + from microcosm.graph import orrery + + monkeypatch.setattr(orrery, "_MAX_NODES", 2) + with pytest.raises(ValueError, match="node limit"): + orrery_document_from_schema(schema) + + monkeypatch.setattr(orrery, "_MAX_NODES", 20_000) + monkeypatch.setattr(orrery, "_MAX_OUTPUT_BYTES", len(canonical_json(schema)) + 64) + with pytest.raises(ValueError, match="metadata size"): + orrery_document_from_schema(schema) + + +def test_schema_embeds_the_exact_canonical_graph_declaration(): + compiled = compiled_graph() + document = orrery_document_from_schema(graph_schema(compiled)) + embedded = document["metadata"]["microcosm"]["graph"] + assert canonical_json(embedded).decode() == graph_to_json(compiled.graph) diff --git a/packages/microcosm-graph/tests/test_explorer.py b/packages/microcosm-graph/tests/test_explorer.py deleted file mode 100644 index 2ab7ff453..000000000 --- a/packages/microcosm-graph/tests/test_explorer.py +++ /dev/null @@ -1,563 +0,0 @@ -"""Presentation contracts over invented declaration metadata; no population run.""" - -import copy -import hashlib -import json -import os - -import pytest -from hypothesis import example, given, settings -from hypothesis import strategies as st - -from microcosm.graph import ( - Graph, - Node, - Owned, - Ownership, - Slice, - SourceRef, - StructuralDelta, - compile_graph, - explorer, - graph_to_json, -) -from microcosm.graph.canonical import canonical_json - - -def test_cli_exports_exact_metadata_and_large_integer(tmp_path): - schema = snapshot() - schema["audit_integer"] = 2**60 + 1 - source = tmp_path / "schema.json" - output = tmp_path / "report" / "graph.json" - source.write_text(json.dumps(schema), encoding="utf-8") - assert ( - explorer.main( - ["--input", str(source), "--output", str(output), "--title", "Microcosm"] - ) - == 0 - ) - assert output.read_text() == explorer.graph_explorer_json(schema, title="Microcosm") - doc = json.loads(output.read_text()) - assert doc["metadata"]["microcosm"]["audit_integer"] == { - "integer_literal": str(2**60 + 1) - } - - -@pytest.mark.parametrize("raw", ['{"protocol":1,"protocol":2}', '{"x":NaN}', "[]"]) -def test_cli_refuses_invalid_json_without_output(tmp_path, raw): - source, output = tmp_path / "schema.json", tmp_path / "graph.json" - source.write_text(raw) - with pytest.raises(SystemExit) as error: - explorer.main(["--input", str(source), "--output", str(output)]) - assert error.value.code == 2 - assert not output.exists() - - -def test_cli_preserves_existing_output_and_input(tmp_path): - source = tmp_path / "schema.json" - raw = json.dumps(snapshot()) - source.write_text(raw) - with pytest.raises(SystemExit) as error: - explorer.main(["--input", str(source), "--output", str(source)]) - assert error.value.code == 2 - assert source.read_text() == raw - - -def test_cli_refuses_oversized_input_before_decoding(tmp_path, monkeypatch): - source, output = tmp_path / "schema.json", tmp_path / "graph.json" - source.write_bytes(b"!" * 33) - monkeypatch.setattr(explorer, "_MAX_INPUT_BYTES", 32) - with pytest.raises(SystemExit) as error: - explorer.main(["--input", str(source), "--output", str(output)]) - assert error.value.code == 2 - assert not output.exists() - - -@pytest.mark.skipif(not hasattr(os, "mkfifo"), reason="requires POSIX FIFO") -def test_cli_refuses_fifo_without_waiting_for_writer(tmp_path): - source, output = tmp_path / "schema.json", tmp_path / "graph.json" - os.mkfifo(source) - with pytest.raises(SystemExit) as error: - explorer.main(["--input", str(source), "--output", str(output)]) - assert error.value.code == 2 - assert not output.exists() - - -def snapshot(): - graph = Graph( - "invented", - sources=(SourceRef("survey", "unused@1"), SourceRef("unused", "unused@1")), - nodes=( - Node( - "base", - "unused.create@1", - structural=StructuralDelta.CREATE, - sources=("survey",), - outputs=( - Owned("person", "age", "int64"), - Owned("person", "mask", "boolean"), - Owned("person", "missing", "float64", ownership=Ownership.ABSENT), - ), - ), - Node( - "filtered", - "unused.filter@1", - structural=StructuralDelta.FILTER, - base="base", - inputs=(Slice("person", ("age",)),), - ), - Node( - "rewrite", - "unused.rewrite@1", - population="filtered", - inputs=(Slice("person", ("age", "mask"), rows="mask"),), - outputs=(Owned("person", "age", "int64", rows="mask", rewrite=True),), - ), - Node( - "consumer", - "unused.consume@1", - population="filtered", - inputs=(Slice("person", ("age",)),), - ), - ), - ) - compiled = compile_graph(graph) - raw = graph_to_json(graph) - fields = [] - for population in ("base", "filtered"): - for owned in graph.nodes[0].outputs: - rewrite = population == "filtered" and owned.column == "age" - fields.append( - { - "population": population, - "entity": owned.entity, - "column": owned.column, - "dtype": owned.dtype, - "producer": "rewrite" if rewrite else population, - "declared_in": "rewrite" if rewrite else "base", - "rows": "mask" if rewrite else "all", - "ownership": owned.ownership.value, - "rewrite": rewrite, - } - ) - - def read( - node, - column, - *, - population="filtered", - producer="filtered", - declared_in="base", - kind="slice", - rows="all", - ): - return { - "node": node, - "entity": "person", - "column": column, - "population": population, - "producer": producer, - "declared_in": declared_in, - "kind": kind, - "rows": rows, - } - - return { - "protocol": "microcosm.graph.schema.v1", - "country": "invented", - "graph_sha256": hashlib.sha256(raw.encode()).hexdigest(), - "graph": json.loads(raw), - "compiled": { - "order": list(compiled.order), - "versions": dict(compiled.versions), - "predecessors": { - key: list(value) for key, value in compiled.predecessors.items() - }, - "owners": [ - list(key) + [owner] for key, owner in sorted(compiled.owners.items()) - ], - }, - "schema": fields, - "input_bindings": [ - read("filtered", "age", population="base", producer="base"), - read("rewrite", "age", rows="mask"), - read("rewrite", "mask", rows="mask"), - read("rewrite", "mask", kind="slice_mask"), - read("rewrite", "mask", kind="output_mask"), - read("rewrite", "age", kind="rewrite_incumbent", rows="mask"), - read("consumer", "age", producer="rewrite", declared_in="rewrite"), - ], - } - - -def update_graph_digest(doc): - doc["graph_sha256"] = hashlib.sha256(canonical_json(doc["graph"])).hexdigest() - - -def test_full_snapshot_and_no_runtime_claims(): - source = snapshot() - doc = explorer.graph_explorer_document(source) - assert doc["schemaVersion"] == "graph-explorer/v1" - assert doc["metadata"]["microcosm"] == source - assert doc["metadata"]["scope"] == "complete_supplied_schema" - assert doc["metadata"]["truncated"] is False - assert len([node for node in doc["nodes"] if node["kind"] == "operation"]) == 4 - assert len([node for node in doc["nodes"] if node["kind"] == "source"]) == 2 - assert len([node for node in doc["nodes"] if node["kind"] == "field"]) == 7 - assert not {"activities", "receipts", "artifacts", "assessments"} & doc.keys() - assert all( - "statuses" not in node and "parentId" not in node for node in doc["nodes"] - ) - assert len( - [edge for edge in doc["edges"] if edge["kind"] == "compiled_predecessor"] - ) == sum(map(len, source["compiled"]["predecessors"].values())) - assert {edge["source"] for edge in doc["edges"]} | { - edge["target"] for edge in doc["edges"] - } <= {node["id"] for node in doc["nodes"]} - - -def test_rewrite_incumbent_is_distinct_and_masks_retain_roles(): - doc = explorer.graph_explorer_document(snapshot()) - ages = [ - n - for n in doc["nodes"] - if n["kind"] == "field" - and n["data"]["population"] == "filtered" - and n["data"]["column"] == "age" - ] - assert len(ages) == 2 and ages[0]["id"] != ages[1]["id"] - incumbent = next(n for n in ages if n["data"]["producer"] == "filtered") - final = next(n for n in ages if n["data"]["producer"] == "rewrite") - assert incumbent["data"]["visible_in_schema"] is False - assert final["data"]["visible_in_schema"] is True - reads = [e for e in doc["edges"] if e["kind"] == "declared_read"] - assert {e["data"]["read_kind"] for e in reads} == { - "slice", - "slice_mask", - "output_mask", - "rewrite_incumbent", - } - incoming = [ - e - for e in reads - if json.loads(e["target"])[1] == "rewrite" - and e["data"]["read_kind"] == "rewrite_incumbent" - ] - assert len(incoming) == 1 and incoming[0]["source"] == incumbent["id"] - assert incoming[0]["data"]["rows"] == "mask" - assert not any( - e["source"] == final["id"] and json.loads(e["target"])[1] == "rewrite" - for e in reads - ) - structural = [e for e in doc["edges"] if e["kind"] == "structural_input"] - assert len(structural) == 3 - assert all(json.loads(e["source"])[1] == "base" for e in structural) - - -def test_typed_artifact_edge_and_uninterpreted_metadata_survive(): - source = snapshot() - base, _, _, consumer = source["graph"]["nodes"] - type_ = {"name": "invented.summary", "schema_version": 2} - base["artifact_outputs"] = [{"name": "summary", "type": type_}] - consumer["artifact_inputs"] = [ - { - "name": "local_alias", - "producer": "base", - "artifact": "summary", - "type": type_, - } - ] - source["compiled"]["predecessors"]["consumer"].append("base") - source["graph"]["mass_partition"] = ["person", "period"] - update_graph_digest(source) - doc = explorer.graph_explorer_document(source) - edge = next(e for e in doc["edges"] if e["kind"] == "artifact") - assert edge["data"] == consumer["artifact_inputs"][0] - assert edge["category"] == "dependency" - assert doc["metadata"]["microcosm"]["graph"]["mass_partition"] == [ - "person", - "period", - ] - - -def test_deterministic_detached_json_and_exact_large_number_transport(): - source = snapshot() - source["graph"]["nodes"][0]["params"] = { - "large": 2**80 + 1, - "negative": -(2**80 + 1), - "safe": 2**53 - 1, - "float": 1e100, - "flag": True, - } - update_graph_digest(source) - before = copy.deepcopy(source) - text = explorer.graph_explorer_json(source) - assert text == explorer.graph_explorer_json(source) and text.endswith("\n") - doc = json.loads(text) - params = doc["metadata"]["microcosm"]["graph"]["nodes"][0]["params"] - assert params == { - "large": {"integer_literal": str(2**80 + 1)}, - "negative": {"integer_literal": str(-(2**80 + 1))}, - "safe": 2**53 - 1, - "float": {"float_literal": "1e+100"}, - "flag": True, - } - doc["metadata"]["microcosm"]["graph"]["nodes"].clear() - assert source == before - - -def test_document_revision_binds_compiled_metadata_not_only_graph(): - source = snapshot() - left = explorer.graph_explorer_document(source) - source["compiled"]["extra_declaration"] = "caller supplied, not verified" - right = explorer.graph_explorer_document(source) - assert left["revision"] != right["revision"] - assert ( - left["metadata"]["microcosm"]["graph_sha256"] - == right["metadata"]["microcosm"]["graph_sha256"] - ) - assert left["nodes"] == right["nodes"] - - -def test_named_mask_read_is_preserved_separately_from_compiler_predecessors(): - graph = Graph( - "invented", - sources=(SourceRef("survey", "unused@1"),), - nodes=( - Node( - "base", - "unused.create@1", - structural=StructuralDelta.CREATE, - sources=("survey",), - outputs=(Owned("person", "age", "int64"),), - ), - Node( - "a_mask", - "unused.mask@1", - population="base", - inputs=(Slice("person", ("age",)),), - outputs=(Owned("person", "mask", "boolean"),), - ), - Node( - "z_consumer", - "unused.consume@1", - population="base", - inputs=(Slice("person", ("age", "mask"), rows="mask"),), - ), - ), - ) - compiled = compile_graph(graph) - source = snapshot() - source["graph"] = json.loads(graph_to_json(graph)) - update_graph_digest(source) - source["compiled"] = { - "order": list(compiled.order), - "versions": dict(compiled.versions), - "predecessors": { - key: list(value) for key, value in compiled.predecessors.items() - }, - "owners": [list(key) + [value] for key, value in compiled.owners.items()], - } - source["schema"] = [ - { - "population": "base", - "entity": "person", - "column": column, - "dtype": dtype, - "producer": owner, - "declared_in": owner, - "rows": "all", - "ownership": "produced", - "rewrite": False, - } - for column, dtype, owner in ( - ("age", "int64", "base"), - ("mask", "boolean", "a_mask"), - ) - ] - source["input_bindings"] = [ - { - "node": node, - "population": "base", - "entity": "person", - "column": column, - "producer": owner, - "declared_in": owner, - "kind": kind, - "rows": rows, - } - for node, column, owner, kind, rows in ( - ("a_mask", "age", "base", "slice", "all"), - ("z_consumer", "age", "base", "slice", "mask"), - ("z_consumer", "mask", "a_mask", "slice", "mask"), - ("z_consumer", "mask", "a_mask", "slice_mask", "all"), - ) - ] - doc = explorer.graph_explorer_document(source) - target = json.dumps(["operation", "z_consumer"], separators=(",", ":")) - incoming = [edge for edge in doc["edges"] if edge["target"] == target] - compiled_parents = { - json.loads(edge["source"])[1] - for edge in incoming - if edge["kind"] == "compiled_predecessor" - } - assert compiled_parents == set(compiled.predecessors["z_consumer"]) - mask = next( - edge - for edge in incoming - if edge["kind"] == "declared_read" and edge["data"]["read_kind"] == "slice_mask" - ) - assert json.loads(mask["source"])[4] == "a_mask" - - -@pytest.mark.parametrize( - "mutation", - [ - lambda doc: doc.update(protocol="future"), - lambda doc: doc.update(graph_sha256="0" * 64), - lambda doc: doc["compiled"]["order"].reverse(), - lambda doc: doc["schema"].append(copy.deepcopy(doc["schema"][0])), - lambda doc: doc["schema"][0].update(dtype="invented"), - lambda doc: doc["input_bindings"][0].update(producer="unknown"), - lambda doc: doc["input_bindings"][0].update(kind="invented"), - lambda doc: doc["input_bindings"].clear(), - lambda doc: doc["input_bindings"].append( - { - **doc["input_bindings"][-1], - "column": "mask", - "producer": "filtered", - "declared_in": "base", - } - ), - ], -) -def test_invalid_references_fail_without_partial_document(mutation): - source = snapshot() - mutation(source) - with pytest.raises(ValueError): - explorer.graph_explorer_document(source) - - -def test_oversize_or_non_json_metadata_fails(monkeypatch): - source = snapshot() - source["extra"] = float("nan") - with pytest.raises(ValueError, match="finite"): - explorer.graph_explorer_document(source) - source["extra"] = source - with pytest.raises(ValueError, match="complexity"): - explorer.graph_explorer_document(source) - source = snapshot() - monkeypatch.setattr(explorer, "_MAX_NODES", 3) - with pytest.raises(ValueError, match="node limit"): - explorer.graph_explorer_document(source) - - -def test_expanded_output_budget_is_checked_during_construction(monkeypatch): - source = snapshot() - budget = len(canonical_json(source)) + 65_536 + 2_000 - monkeypatch.setattr(explorer, "_MAX_OUTPUT_BYTES", budget) - - # The input fits, but presentation records must stop while being added. - # A late-only final serialization check would visit the field loop first. - def unexpected_field(_field): - pytest.fail("expanded records reached field construction before refusal") - - monkeypatch.setattr(explorer, "_field_id", unexpected_field) - with pytest.raises(ValueError, match="serialized metadata size"): - explorer.graph_explorer_document(source) - - -_JSON_SCALARS = st.one_of( - st.none(), - st.booleans(), - st.integers(min_value=-(2**1024 - 1), max_value=2**1024 - 1), - st.floats(allow_nan=False, allow_infinity=False), - st.text(max_size=40), -) -_JSON_METADATA = st.recursive( - _JSON_SCALARS, - lambda children: st.one_of( - st.lists(children, max_size=4), - st.dictionaries(st.text(max_size=20), children, max_size=4), - ), - max_leaves=20, -) - - -@settings(deadline=None) -@given(value=_JSON_SCALARS) -@example(value=2**53 - 1) -@example(value=2**53) -@example(value=-(2**53 - 1)) -@example(value=-(2**53)) -@example(value=float(2**53 - 1)) -@example(value=float(2**53)) -@example(value=-float(2**53)) -@example(value=-0.0) -@example(value=True) -def test_metadata_transport_is_exact_and_detached(value): - """Metadata preserves exact scalar values and shares no mutable containers.""" - source = snapshot() - source["extra"] = {"nested": [value]} - before = copy.deepcopy(source) - doc = explorer.graph_explorer_document(source) - rendered = json.loads(explorer.graph_explorer_json(source)) - transported = rendered["metadata"]["microcosm"]["extra"]["nested"][0] - - if type(value) is int and abs(value) > 2**53 - 1: - assert transported == {"integer_literal": str(value)} - elif type(value) is float and value.is_integer() and abs(value) > 2**53 - 1: - assert transported == {"float_literal": repr(value)} - else: - assert type(transported) is type(value) - assert transported == value - if type(value) is float: - assert transported.hex() == value.hex() - assert doc["metadata"]["microcosm"]["extra"]["nested"] == [transported] - - exported_values = doc["metadata"]["microcosm"]["extra"]["nested"] - exported_values.append("output-side mutation") - doc["metadata"]["microcosm"]["graph"]["nodes"].clear() - assert source == before - source["extra"]["nested"].append("input-side mutation") - assert exported_values == [transported, "output-side mutation"] - - -def reversed_mapping_order(value): - if isinstance(value, dict): - return { - key: reversed_mapping_order(child) - for key, child in reversed(list(value.items())) - } - if isinstance(value, list): - return [reversed_mapping_order(child) for child in value] - return value - - -@settings(deadline=None) -@given(metadata=_JSON_METADATA) -def test_json_is_invariant_to_mapping_order(metadata): - """Equivalent JSON objects produce identical bytes and content revisions.""" - source = snapshot() - source["extra"] = metadata - assert explorer.graph_explorer_json(source) == explorer.graph_explorer_json( - reversed_mapping_order(source) - ) - - -@settings(deadline=None) -@given(metadata=_JSON_METADATA) -def test_revision_binds_generated_metadata_changes(metadata): - """Every metadata change affects the revision without changing graph identity.""" - source = snapshot() - source["compiled"]["extra_declaration"] = {"payload": metadata, "marker": False} - before = explorer.graph_explorer_document(source) - source["compiled"]["extra_declaration"]["marker"] = True - after = explorer.graph_explorer_document(source) - assert before["revision"] != after["revision"] - assert ( - before["metadata"]["microcosm"]["graph_sha256"] - == after["metadata"]["microcosm"]["graph_sha256"] - ) - assert before["nodes"] == after["nodes"] - assert before["edges"] == after["edges"] diff --git a/test_support/microcosm_graph/schema.py b/test_support/microcosm_graph/schema.py index 7235729b8..b68c81669 100644 --- a/test_support/microcosm_graph/schema.py +++ b/test_support/microcosm_graph/schema.py @@ -3,6 +3,9 @@ from __future__ import annotations from microcosm.graph import ( + ArtifactInput, + ArtifactOutput, + ArtifactType, Graph, Node, Owned, @@ -17,6 +20,7 @@ def compiled_graph(): """Return a graph covering carried, rewritten, and structural fields.""" + summary_type = ArtifactType("invented.summary", 1) graph = Graph( "invented", sources=(SourceRef("survey", "unused@1", "Recorded survey tables"),), @@ -30,6 +34,7 @@ def compiled_graph(): Owned("person", "age", "int64"), Owned("person", "keep", "boolean"), ), + artifact_outputs=(ArtifactOutput("summary", summary_type),), description="Load the declared source tables", ), Node( @@ -71,6 +76,9 @@ def compiled_graph(): base="expanded", weights=WeightUpdate("person", "design", "normalize sample weights"), mass="declared", + artifact_inputs=( + ArtifactInput("source_summary", "base", "summary", summary_type), + ), ), ), ) From ab1e0e4aeb934668075589f145e4c6c806bfee36 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:57:41 +0400 Subject: [PATCH 05/13] Expose direct Orrery export APIs and CLI --- .../src/microcosm/graph/__init__.py | 32 +++++ .../src/microcosm/graph/orrery.py | 111 +++++++++++++- .../engine_free/shared/test_graph_orrery.py | 135 +++++++++++++++++- 3 files changed, 274 insertions(+), 4 deletions(-) diff --git a/packages/microcosm-graph/src/microcosm/graph/__init__.py b/packages/microcosm-graph/src/microcosm/graph/__init__.py index 37227aed5..f266a4d2b 100644 --- a/packages/microcosm-graph/src/microcosm/graph/__init__.py +++ b/packages/microcosm-graph/src/microcosm/graph/__init__.py @@ -6,6 +6,7 @@ from __future__ import annotations +from collections.abc import Mapping from importlib import metadata as _metadata from .decl import ( @@ -59,6 +60,7 @@ ) from .keys import platform_fingerprint from .randomness import keyed_uniform +from .schema import graph_schema from .weight_update import WEIGHT_UPDATE_AXIS_SCHEMA, weight_update_receipt __all__ = [ @@ -127,6 +129,7 @@ "describe", "explain_html", "graph_from_json", + "graph_schema", "graph_to_json", "keyed_uniform", "load_source", @@ -134,6 +137,8 @@ "run_graph", "source_hash", "weight_update_receipt", + "orrery_document", + "orrery_json", ] _FRAME_SERIES = "0.1" @@ -153,6 +158,33 @@ def _check_frame_version() -> None: _check_frame_version() + +def orrery_document( + graph: Graph | CompiledGraph, + *, + title: str | None = None, + extensions: Mapping[str, object] | None = None, +) -> dict[str, object]: + """Compile and transform a graph into a complete Orrery document.""" + + from .orrery import orrery_document as export + + return export(graph, title=title, extensions=extensions) + + +def orrery_json( + graph: Graph | CompiledGraph, + *, + title: str | None = None, + extensions: Mapping[str, object] | None = None, +) -> str: + """Compile and return deterministic UTF-8-ready Orrery JSON.""" + + from .orrery import orrery_json as export + + return export(graph, title=title, extensions=extensions) + + from .codecs import ( # noqa: E402 - check dependency series before runtime import SOURCE_CODECS, SourceBytesCodec, diff --git a/packages/microcosm-graph/src/microcosm/graph/orrery.py b/packages/microcosm-graph/src/microcosm/graph/orrery.py index d245f22f2..a356c4ad2 100644 --- a/packages/microcosm-graph/src/microcosm/graph/orrery.py +++ b/packages/microcosm-graph/src/microcosm/graph/orrery.py @@ -6,16 +6,29 @@ from __future__ import annotations +import argparse import hashlib import json import math +import os +import stat +from collections.abc import Mapping +from pathlib import Path from .canonical import canonical_json -from .schema import validate_graph_schema +from .decl import CompiledGraph, Graph, compile_graph +from .schema import graph_schema, validate_graph_schema +from .serialize import graph_from_json -__all__ = ["orrery_document_from_schema", "orrery_json_from_schema"] +__all__ = [ + "orrery_document", + "orrery_document_from_schema", + "orrery_json", + "orrery_json_from_schema", +] _SAFE_INTEGER = 2**53 - 1 +_MAX_INPUT_BYTES = 32 * 1024 * 1024 _MAX_OUTPUT_BYTES = 64 * 1024 * 1024 _MAX_ITEMS = 2_000_000 _MAX_NODES = 20_000 @@ -401,3 +414,97 @@ def orrery_json_from_schema(schema: object, *, title: str | None = None) -> str: ) + "\n" ) + + +def _compiled(value: Graph | CompiledGraph) -> CompiledGraph: + if isinstance(value, Graph): + return compile_graph(value) + if isinstance(value, CompiledGraph): + return value + raise TypeError( + f"orrery_document expects Graph or CompiledGraph, got {type(value).__name__}" + ) + + +def orrery_document( + graph: Graph | CompiledGraph, + *, + title: str | None = None, + extensions: Mapping[str, object] | None = None, +) -> dict[str, object]: + """Compile and transform a graph into a complete Orrery document.""" + + return orrery_document_from_schema( + graph_schema(_compiled(graph), extensions=extensions), title=title + ) + + +def orrery_json( + graph: Graph | CompiledGraph, + *, + title: str | None = None, + extensions: Mapping[str, object] | None = None, +) -> str: + """Compile and return deterministic UTF-8-ready Orrery JSON.""" + + return orrery_json_from_schema( + graph_schema(_compiled(graph), extensions=extensions), title=title + ) + + +def _unique_json_object(pairs: list[tuple[str, object]]) -> dict[str, object]: + result: dict[str, object] = {} + for key, value in pairs: + _require(key not in result, "duplicate JSON key") + result[key] = value + return result + + +def _invalid_json_constant(_value: str) -> None: + raise ValueError("Orrery adapter: finite JSON numbers required.") + + +def _read_json(path: Path) -> object: + descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NONBLOCK", 0)) + with os.fdopen(descriptor, "rb") as stream: + info = os.fstat(stream.fileno()) + _require(stat.S_ISREG(info.st_mode), "input must be a regular file") + _require(info.st_size <= _MAX_INPUT_BYTES, "input byte limit") + raw = stream.read(_MAX_INPUT_BYTES + 1) + _require(len(raw) <= _MAX_INPUT_BYTES, "input byte limit") + return json.loads( + raw.decode("utf-8"), + object_pairs_hook=_unique_json_object, + parse_constant=_invalid_json_constant, + ) + + +def main(argv: list[str] | None = None) -> int: + """Create Orrery JSON from a graph declaration or compiler schema.""" + + parser = argparse.ArgumentParser(description=main.__doc__) + inputs = parser.add_mutually_exclusive_group(required=True) + inputs.add_argument("--graph", type=Path, help="canonical Graph JSON") + inputs.add_argument("--schema", type=Path, help="microcosm.graph.schema.v1 JSON") + parser.add_argument("--output", type=Path, required=True, help="new Orrery JSON") + parser.add_argument("--title", help="Orrery document title") + args = parser.parse_args(argv) + try: + source = args.graph if args.graph is not None else args.schema + value = _read_json(source) + if args.graph is not None: + graph = graph_from_json(canonical_json(value).decode("utf-8")) + rendered = orrery_json(graph, title=args.title) + else: + rendered = orrery_json_from_schema(value, title=args.title) + args.output.parent.mkdir(parents=True, exist_ok=True) + with args.output.open("x", encoding="utf-8", newline="\n") as stream: + stream.write(rendered) + except (OSError, TypeError, ValueError, RecursionError) as error: + parser.error(str(error)) + print(f"wrote {args.output}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py index 28b66cf1c..6fd2168e1 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py @@ -4,16 +4,26 @@ import copy import json +import os +import subprocess +import sys import pytest -from microcosm.graph import compile_graph, graph_from_json, graph_to_json +from microcosm.graph import ( + compile_graph, + graph_from_json, + graph_schema, + graph_to_json, + orrery_document, + orrery_json, +) from microcosm.graph.canonical import canonical_json from microcosm.graph.orrery import ( + main, orrery_document_from_schema, orrery_json_from_schema, ) -from microcosm.graph.schema import graph_schema from test_support.microcosm_graph.schema import ( compiled_default_population_graph, compiled_graph, @@ -211,3 +221,124 @@ def test_schema_embeds_the_exact_canonical_graph_declaration(): document = orrery_document_from_schema(graph_schema(compiled)) embedded = document["metadata"]["microcosm"]["graph"] assert canonical_json(embedded).decode() == graph_to_json(compiled.graph) + + +def test_direct_graph_compiled_and_saved_schema_paths_are_identical(): + compiled = compiled_graph() + schema = graph_schema(compiled, extensions={"producer": "test"}) + + expected = orrery_json_from_schema(schema, title="Invented build") + assert ( + orrery_json( + compiled, + title="Invented build", + extensions={"producer": "test"}, + ) + == expected + ) + assert ( + orrery_json( + compiled.graph, + title="Invented build", + extensions={"producer": "test"}, + ) + == expected + ) + assert orrery_document(compiled)["schemaVersion"] == "graph-explorer/v1" + + +def test_cli_requires_an_explicit_input_type_and_matches_direct_output(tmp_path): + compiled = compiled_graph() + graph_path = tmp_path / "graph.json" + schema_path = tmp_path / "schema.json" + graph_output = tmp_path / "graph-output.json" + schema_output = tmp_path / "schema-output.json" + graph_path.write_text(graph_to_json(compiled.graph)) + schema_path.write_bytes(canonical_json(graph_schema(compiled))) + + assert ( + main( + [ + "--graph", + str(graph_path), + "--output", + str(graph_output), + "--title", + "Invented build", + ] + ) + == 0 + ) + assert ( + main( + [ + "--schema", + str(schema_path), + "--output", + str(schema_output), + "--title", + "Invented build", + ] + ) + == 0 + ) + expected = orrery_json(compiled, title="Invented build") + assert graph_output.read_text() == expected + assert schema_output.read_text() == expected + + with pytest.raises(SystemExit) as error: + main( + [ + "--graph", + str(graph_path), + "--schema", + str(schema_path), + "--output", + str(tmp_path / "ambiguous.json"), + ] + ) + assert error.value.code == 2 + + +@pytest.mark.parametrize("raw", ['{"country":1,"country":2}', '{"x":NaN}', "[]"]) +def test_cli_refuses_invalid_json_without_output(tmp_path, raw): + source = tmp_path / "input.json" + output = tmp_path / "output.json" + source.write_text(raw) + with pytest.raises(SystemExit) as error: + main(["--graph", str(source), "--output", str(output)]) + assert error.value.code == 2 + assert not output.exists() + + +def test_cli_preserves_existing_output_and_input(tmp_path): + compiled = compiled_graph() + source = tmp_path / "graph.json" + raw = graph_to_json(compiled.graph) + source.write_text(raw) + with pytest.raises(SystemExit) as error: + main(["--graph", str(source), "--output", str(source)]) + assert error.value.code == 2 + assert source.read_text() == raw + + +@pytest.mark.skipif(not hasattr(os, "mkfifo"), reason="requires POSIX FIFO") +def test_cli_refuses_fifo_without_waiting_for_a_writer(tmp_path): + source = tmp_path / "graph.json" + output = tmp_path / "output.json" + os.mkfifo(source) + with pytest.raises(SystemExit) as error: + main(["--graph", str(source), "--output", str(output)]) + assert error.value.code == 2 + assert not output.exists() + + +def test_module_cli_starts_without_an_import_order_warning(): + result = subprocess.run( + [sys.executable, "-m", "microcosm.graph.orrery", "--help"], + text=True, + capture_output=True, + check=True, + ) + assert "--graph" in result.stdout + assert "RuntimeWarning" not in result.stderr From 65cf5e4c8c498b028c0201775946ce63de29811b Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:04:57 +0400 Subject: [PATCH 06/13] Verify exports against Orrery 0.6.0 --- .github/workflows/test.yml | 59 +++++ tools/check_orrery_contract.py | 58 +++++ tools/orrery-contract/.gitignore | 1 + tools/orrery-contract/package-lock.json | 319 ++++++++++++++++++++++++ tools/orrery-contract/package.json | 11 + tools/orrery-contract/verify.mjs | 13 + 6 files changed, 461 insertions(+) create mode 100644 tools/check_orrery_contract.py create mode 100644 tools/orrery-contract/.gitignore create mode 100644 tools/orrery-contract/package-lock.json create mode 100644 tools/orrery-contract/package.json create mode 100644 tools/orrery-contract/verify.mjs diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 29eb31358..1a2b2a2a9 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -143,3 +143,62 @@ jobs: python-version: "3.13" - name: Build and inspect every shard wheel run: tools/build_and_inspect_wheels.sh + + orrery-contract: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: astral-sh/setup-uv@v6 + with: + python-version: "3.13" + - uses: actions/setup-node@v7 + with: + node-version: "24" + cache: npm + cache-dependency-path: tools/orrery-contract/package-lock.json + - name: Sync workspace + run: uv sync --all-packages --locked + - name: Install the exact Orrery contract + run: npm ci --ignore-scripts --prefix tools/orrery-contract + - name: Parse a Microcosm export with Orrery + run: uv run --no-sync python tools/check_orrery_contract.py + + ci-ok: + if: always() + needs: + - select-countries + - lint + - engine-free + - engine-us + - engine-uk + - integration-uk + - wheels + - orrery-contract + runs-on: ubuntu-latest + env: + SELECT_COUNTRIES: ${{ needs.select-countries.result }} + LINT: ${{ needs.lint.result }} + ENGINE_FREE: ${{ needs.engine-free.result }} + ENGINE_US: ${{ needs.engine-us.result }} + ENGINE_UK: ${{ needs.engine-uk.result }} + INTEGRATION_UK: ${{ needs.integration-uk.result }} + WHEELS: ${{ needs.wheels.result }} + ORRERY_CONTRACT: ${{ needs.orrery-contract.result }} + steps: + - name: Require every selected test job to pass + run: | + for result in \ + "$SELECT_COUNTRIES" \ + "$LINT" \ + "$ENGINE_FREE" \ + "$ENGINE_US" \ + "$ENGINE_UK" \ + "$INTEGRATION_UK" \ + "$WHEELS" \ + "$ORRERY_CONTRACT" + do + if [ "$result" != success ] && [ "$result" != skipped ]; then + echo "A required test job completed with result: $result" >&2 + exit 1 + fi + done diff --git a/tools/check_orrery_contract.py b/tools/check_orrery_contract.py new file mode 100644 index 000000000..302ffd96e --- /dev/null +++ b/tools/check_orrery_contract.py @@ -0,0 +1,58 @@ +#!/usr/bin/env python3 +"""Pass a public Microcosm export through the pinned Orrery parser.""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + +from microcosm.graph import ( + Graph, + Node, + Owned, + Slice, + SourceRef, + StructuralDelta, + orrery_json, +) + +ROOT = Path(__file__).resolve().parents[1] +VERIFY = ROOT / "tools" / "orrery-contract" / "verify.mjs" + + +def contract_graph() -> Graph: + """Use omitted population to cover the compiler-resolved API contract.""" + + return Graph( + "contract", + sources=(SourceRef("fixture", "contract@1"),), + nodes=( + Node( + "source", + "contract.create@1", + structural=StructuralDelta.CREATE, + sources=("fixture",), + outputs=(Owned("person", "amount", "float64"),), + ), + Node( + "consumer", + "contract.consume@1", + inputs=(Slice("person", ("amount",)),), + outputs=(Owned("person", "result", "float64"),), + ), + ), + ) + + +def main() -> int: + subprocess.run( + ["node", str(VERIFY)], + input=orrery_json(contract_graph()), + text=True, + check=True, + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/orrery-contract/.gitignore b/tools/orrery-contract/.gitignore new file mode 100644 index 000000000..c2658d7d1 --- /dev/null +++ b/tools/orrery-contract/.gitignore @@ -0,0 +1 @@ +node_modules/ diff --git a/tools/orrery-contract/package-lock.json b/tools/orrery-contract/package-lock.json new file mode 100644 index 000000000..da6b11072 --- /dev/null +++ b/tools/orrery-contract/package-lock.json @@ -0,0 +1,319 @@ +{ + "name": "microcosm-orrery-contract", + "version": "0.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "microcosm-orrery-contract", + "version": "0.0.0", + "dependencies": { + "@axiom-foundation/orrery": "0.6.0", + "react": "19.2.8", + "react-dom": "19.2.8" + } + }, + "node_modules/@axiom-foundation/orrery": { + "version": "0.6.0", + "resolved": "https://registry.npmjs.org/@axiom-foundation/orrery/-/orrery-0.6.0.tgz", + "integrity": "sha512-Dva+L5cVopuZ6V+TcP8W6IxZkKD4RhWvBcIG5wr07b2KVKws/ay1GfsafIp9pFB1e9IlhxfSzo4bzzvciHyj3g==", + "license": "MIT", + "dependencies": { + "@dagrejs/dagre": "3.1.1", + "@xyflow/react": "12.11.6" + }, + "bin": { + "graph-explorer": "dist/cli.js", + "orrery": "dist/cli.js" + }, + "engines": { + "node": ">=20" + }, + "peerDependencies": { + "react": ">=18 <20", + "react-dom": ">=18 <20" + } + }, + "node_modules/@dagrejs/dagre": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/@dagrejs/dagre/-/dagre-3.1.1.tgz", + "integrity": "sha512-zroZB1dFOFiGgv4Xcrn1DckB1o4aOikPqD2NDQPV0WM//CXGcS6xiD0rNkqHmw6FEg4tabt4nxPLwgCWT+Vb2A==", + "license": "MIT", + "dependencies": { + "@dagrejs/graphlib": "4.0.5" + } + }, + "node_modules/@dagrejs/graphlib": { + "version": "4.0.5", + "resolved": "https://registry.npmjs.org/@dagrejs/graphlib/-/graphlib-4.0.5.tgz", + "integrity": "sha512-7xrBTqIts3o+PMUZX97wSc+7TUbW+/rULzGNCTP6yooNVDXbzw4Wutg/H/xOutTB/c/k0YqOAavgPh4/Zk9PFA==", + "license": "MIT" + }, + "node_modules/@types/d3-color": { + "version": "3.1.3", + "resolved": "https://registry.npmjs.org/@types/d3-color/-/d3-color-3.1.3.tgz", + "integrity": "sha512-iO90scth9WAbmgv7ogoq57O9YpKmFBbmoEoCHDB2xMBY0+/KVrqAaCDyCE16dUspeOvIxFFRI+0sEtqDqy2b4A==", + "license": "MIT" + }, + "node_modules/@types/d3-drag": { + "version": "3.0.7", + "resolved": "https://registry.npmjs.org/@types/d3-drag/-/d3-drag-3.0.7.tgz", + "integrity": "sha512-HE3jVKlzU9AaMazNufooRJ5ZpWmLIoc90A37WU2JMmeq28w1FQqCZswHZ3xR+SuxYftzHq6WU6KJHvqxKzTxxQ==", + "license": "MIT", + "dependencies": { + "@types/d3-selection": "*" + } + }, + "node_modules/@types/d3-interpolate": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@types/d3-interpolate/-/d3-interpolate-3.0.4.tgz", + "integrity": "sha512-mgLPETlrpVV1YRJIglr4Ez47g7Yxjl1lj7YKsiMCb27VJH9W8NVM6Bb9d8kkpG/uAQS5AmbA48q2IAolKKo1MA==", + "license": "MIT", + "dependencies": { + "@types/d3-color": "*" + } + }, + "node_modules/@types/d3-selection": { + "version": "3.0.12", + "resolved": "https://registry.npmjs.org/@types/d3-selection/-/d3-selection-3.0.12.tgz", + "integrity": "sha512-Qe/KWYhEiIIxGs7HrAAjMfShxKldx19SJtr5zu53f3afPsdZNz7HHtdTLXo/kqeiWNXVycI24kSnfzBYkTzpgw==", + "license": "MIT" + }, + "node_modules/@types/d3-transition": { + "version": "3.0.9", + "resolved": "https://registry.npmjs.org/@types/d3-transition/-/d3-transition-3.0.9.tgz", + "integrity": "sha512-uZS5shfxzO3rGlu0cC3bjmMFKsXv+SmZZcgp0KD22ts4uGXp5EVYGzu/0YdwZeKmddhcAccYtREJKkPfXkZuCg==", + "license": "MIT", + "dependencies": { + "@types/d3-selection": "*" + } + }, + "node_modules/@types/d3-zoom": { + "version": "3.0.9", + "resolved": "https://registry.npmjs.org/@types/d3-zoom/-/d3-zoom-3.0.9.tgz", + "integrity": "sha512-0sE1406XBYJGiqD3AusTl9ZqC//2mIXix51tbom25gDCA8ri4xnSZg28CaSE8Srl6FClqABUKfsn0qghgdepMA==", + "license": "MIT", + "dependencies": { + "@types/d3-interpolate": "*", + "@types/d3-selection": "*" + } + }, + "node_modules/@xyflow/react": { + "version": "12.11.6", + "resolved": "https://registry.npmjs.org/@xyflow/react/-/react-12.11.6.tgz", + "integrity": "sha512-9XsEJNHjatKYndszKTF/bsU7FOP9dJ6V/EQwzy3oMdtqgBuUq7BjKSwkEo+C7s4qHstHQfwwoHA3E8QfpPxZZQ==", + "license": "MIT", + "dependencies": { + "@xyflow/system": "0.0.82", + "classcat": "^5.0.3", + "zustand": "^4.4.0" + }, + "peerDependencies": { + "@types/react": ">=17", + "@types/react-dom": ">=17", + "react": ">=17", + "react-dom": ">=17" + }, + "peerDependenciesMeta": { + "@types/react": { + "optional": true + }, + "@types/react-dom": { + "optional": true + } + } + }, + "node_modules/@xyflow/system": { + "version": "0.0.82", + "resolved": "https://registry.npmjs.org/@xyflow/system/-/system-0.0.82.tgz", + "integrity": "sha512-4DKnL3CGtCGLRSmgDqaajRVgeksMXq/Yw4wPfdMfm7JvdIiWHGzVYOfFUztiATwdBXZqUU6HhehwUdbV9G23PQ==", + "license": "MIT", + "dependencies": { + "@types/d3-drag": "^3.0.7", + "@types/d3-interpolate": "^3.0.4", + "@types/d3-selection": "^3.0.10", + "@types/d3-transition": "^3.0.8", + "@types/d3-zoom": "^3.0.8", + "d3-drag": "^3.0.0", + "d3-interpolate": "^3.0.1", + "d3-selection": "^3.0.0", + "d3-zoom": "^3.0.0" + } + }, + "node_modules/classcat": { + "version": "5.0.5", + "resolved": "https://registry.npmjs.org/classcat/-/classcat-5.0.5.tgz", + "integrity": "sha512-JhZUT7JFcQy/EzW605k/ktHtncoo9vnyW/2GspNYwFlN1C/WmjuV/xtS04e9SOkL2sTdw0VAZ2UGCcQ9lR6p6w==", + "license": "MIT" + }, + "node_modules/d3-color": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/d3-color/-/d3-color-3.1.0.tgz", + "integrity": "sha512-zg/chbXyeBtMQ1LbD/WSoW2DpC3I0mpmPdW+ynRTj/x2DAWYrIY7qeZIHidozwV24m4iavr15lNwIwLxRmOxhA==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-dispatch": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-dispatch/-/d3-dispatch-3.0.1.tgz", + "integrity": "sha512-rzUyPU/S7rwUflMyLc1ETDeBj0NRuHKKAcvukozwhshr6g6c5d8zh4c2gQjY2bZ0dXeGLWc1PF174P2tVvKhfg==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-drag": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/d3-drag/-/d3-drag-3.0.0.tgz", + "integrity": "sha512-pWbUJLdETVA8lQNJecMxoXfH6x+mO2UQo8rSmZ+QqxcbyA3hfeprFgIT//HW2nlHChWeIIMwS2Fq+gEARkhTkg==", + "license": "ISC", + "dependencies": { + "d3-dispatch": "1 - 3", + "d3-selection": "3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-ease": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-ease/-/d3-ease-3.0.1.tgz", + "integrity": "sha512-wR/XK3D3XcLIZwpbvQwQ5fK+8Ykds1ip7A2Txe0yxncXSdq1L9skcG7blcedkOX+ZcgxGAmLX1FrRGbADwzi0w==", + "license": "BSD-3-Clause", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-interpolate": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-interpolate/-/d3-interpolate-3.0.1.tgz", + "integrity": "sha512-3bYs1rOD33uo8aqJfKP3JWPAibgw8Zm2+L9vBKEHJ2Rg+viTR7o5Mmv5mZcieN+FRYaAOWX5SJATX6k1PWz72g==", + "license": "ISC", + "dependencies": { + "d3-color": "1 - 3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-selection": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/d3-selection/-/d3-selection-3.0.0.tgz", + "integrity": "sha512-fmTRWbNMmsmWq6xJV8D19U/gw/bwrHfNXxrIN+HfZgnzqTHp9jOmKMhsTUjXOJnZOdZY9Q28y4yebKzqDKlxlQ==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-timer": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-timer/-/d3-timer-3.0.1.tgz", + "integrity": "sha512-ndfJ/JxxMd3nw31uyKoY2naivF+r29V+Lc0svZxe1JvvIRmi8hUsrMvdOwgS1o6uBHmiz91geQ0ylPP0aj1VUA==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-transition": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-transition/-/d3-transition-3.0.1.tgz", + "integrity": "sha512-ApKvfjsSR6tg06xrL434C0WydLr7JewBB3V+/39RMHsaXTOG0zmt/OAXeng5M5LBm0ojmxJrpomQVZ1aPvBL4w==", + "license": "ISC", + "dependencies": { + "d3-color": "1 - 3", + "d3-dispatch": "1 - 3", + "d3-ease": "1 - 3", + "d3-interpolate": "1 - 3", + "d3-timer": "1 - 3" + }, + "engines": { + "node": ">=12" + }, + "peerDependencies": { + "d3-selection": "2 - 3" + } + }, + "node_modules/d3-zoom": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/d3-zoom/-/d3-zoom-3.0.0.tgz", + "integrity": "sha512-b8AmV3kfQaqWAuacbPuNbL6vahnOJflOhexLzMMNLga62+/nh0JzvJ0aO/5a5MVgUFGS7Hu1P9P03o3fJkDCyw==", + "license": "ISC", + "dependencies": { + "d3-dispatch": "1 - 3", + "d3-drag": "2 - 3", + "d3-interpolate": "1 - 3", + "d3-selection": "2 - 3", + "d3-transition": "2 - 3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/react": { + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react/-/react-19.2.8.tgz", + "integrity": "sha512-PWaYA1L/q9u2u7xYQi+Y3L3Yfnie7XyLeaJICV1MGD6LprsBxcAqGjYyr0eY3p+QdsA+x/Irkt4Qif8D63+Sbw==", + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/react-dom": { + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.8.tgz", + "integrity": "sha512-rVprimfGBG3DR+Tq0IQG2DT5PxKth1WIGDmj5yPmlzr4YBe7uyE+Du4oVqTDXZSHGGGXRtTJEGSSePyQCMBglQ==", + "license": "MIT", + "dependencies": { + "scheduler": "^0.27.0" + }, + "peerDependencies": { + "react": "^19.2.8" + } + }, + "node_modules/scheduler": { + "version": "0.27.0", + "resolved": "https://registry.npmjs.org/scheduler/-/scheduler-0.27.0.tgz", + "integrity": "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q==", + "license": "MIT" + }, + "node_modules/use-sync-external-store": { + "version": "1.7.0", + "resolved": "https://registry.npmjs.org/use-sync-external-store/-/use-sync-external-store-1.7.0.tgz", + "integrity": "sha512-6L+EeigHMQhdaIPNIFUKwfWJSwWFQ8gJbJ2DLOs5sDIegTwR9fRxvnM3uciHKjIZhFz+KAv2emhWMRvDmMcY8A==", + "license": "MIT", + "peerDependencies": { + "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" + } + }, + "node_modules/zustand": { + "version": "4.5.7", + "resolved": "https://registry.npmjs.org/zustand/-/zustand-4.5.7.tgz", + "integrity": "sha512-CHOUy7mu3lbD6o6LJLfllpjkzhHXSBlX8B9+qPddUsIfeF5S/UZ5q0kmCsnRqT1UHFQZchNFDDzMbQsuesHWlw==", + "license": "MIT", + "dependencies": { + "use-sync-external-store": "^1.2.2" + }, + "engines": { + "node": ">=12.7.0" + }, + "peerDependencies": { + "@types/react": ">=16.8", + "immer": ">=9.0.6", + "react": ">=16.8" + }, + "peerDependenciesMeta": { + "@types/react": { + "optional": true + }, + "immer": { + "optional": true + }, + "react": { + "optional": true + } + } + } + } +} diff --git a/tools/orrery-contract/package.json b/tools/orrery-contract/package.json new file mode 100644 index 000000000..03e5094cf --- /dev/null +++ b/tools/orrery-contract/package.json @@ -0,0 +1,11 @@ +{ + "name": "microcosm-orrery-contract", + "private": true, + "version": "0.0.0", + "type": "module", + "dependencies": { + "@axiom-foundation/orrery": "0.6.0", + "react": "19.2.8", + "react-dom": "19.2.8" + } +} diff --git a/tools/orrery-contract/verify.mjs b/tools/orrery-contract/verify.mjs new file mode 100644 index 000000000..7cff4f96b --- /dev/null +++ b/tools/orrery-contract/verify.mjs @@ -0,0 +1,13 @@ +import assert from 'node:assert/strict'; +import { parseGraphDocument } from '@axiom-foundation/orrery'; + +const chunks = []; +for await (const chunk of process.stdin) chunks.push(chunk); +const document = parseGraphDocument(JSON.parse(Buffer.concat(chunks).toString('utf8'))); + +assert.equal(document.schemaVersion, 'graph-explorer/v1'); +assert.equal(document.metadata.adapter, 'microcosm.graph.orrery.v1'); +assert.ok(document.nodes.some(node => node.kind === 'operation')); +assert.ok(document.nodes.some(node => node.kind === 'field')); +assert.ok(document.edges.some(edge => edge.kind === 'declared_read')); +process.stdout.write('Orrery accepted the Microcosm graph document.\n'); From 8bd7c8be0d31f0547df8f98073a11213fcf32f4b Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:14:25 +0400 Subject: [PATCH 07/13] Document the compiler-to-Orrery contract --- .github/workflows/test.yml | 40 ------ changelog.d/orrery-schema-cli.added.md | 2 +- docs/agent-guide.md | 7 +- docs/graph-explorer.md | 4 +- docs/orrery-adapter.md | 175 +++++++++++++++++++++++++ docs/shared-graph-explorer-adapter.md | 126 ------------------ packages/microcosm-graph/README.md | 7 +- 7 files changed, 188 insertions(+), 173 deletions(-) create mode 100644 docs/orrery-adapter.md delete mode 100644 docs/shared-graph-explorer-adapter.md diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 1a2b2a2a9..9389235b3 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -162,43 +162,3 @@ jobs: run: npm ci --ignore-scripts --prefix tools/orrery-contract - name: Parse a Microcosm export with Orrery run: uv run --no-sync python tools/check_orrery_contract.py - - ci-ok: - if: always() - needs: - - select-countries - - lint - - engine-free - - engine-us - - engine-uk - - integration-uk - - wheels - - orrery-contract - runs-on: ubuntu-latest - env: - SELECT_COUNTRIES: ${{ needs.select-countries.result }} - LINT: ${{ needs.lint.result }} - ENGINE_FREE: ${{ needs.engine-free.result }} - ENGINE_US: ${{ needs.engine-us.result }} - ENGINE_UK: ${{ needs.engine-uk.result }} - INTEGRATION_UK: ${{ needs.integration-uk.result }} - WHEELS: ${{ needs.wheels.result }} - ORRERY_CONTRACT: ${{ needs.orrery-contract.result }} - steps: - - name: Require every selected test job to pass - run: | - for result in \ - "$SELECT_COUNTRIES" \ - "$LINT" \ - "$ENGINE_FREE" \ - "$ENGINE_US" \ - "$ENGINE_UK" \ - "$INTEGRATION_UK" \ - "$WHEELS" \ - "$ORRERY_CONTRACT" - do - if [ "$result" != success ] && [ "$result" != skipped ]; then - echo "A required test job completed with result: $result" >&2 - exit 1 - fi - done diff --git a/changelog.d/orrery-schema-cli.added.md b/changelog.d/orrery-schema-cli.added.md index 8ed3fd1ef..4ee151ae4 100644 --- a/changelog.d/orrery-schema-cli.added.md +++ b/changelog.d/orrery-schema-cli.added.md @@ -1 +1 @@ -Add a Microcosm schema adapter and `python -m microcosm.graph.explorer` CLI for the Orrery shared viewer, preserving field identities and exact large integers while refusing invalid or oversized input and existing output files. +Add a compiler-derived graph schema, direct Orrery export APIs, and an explicit `python -m microcosm.graph.orrery` command for graph declarations or saved schemas. Recompile imported metadata, retain field providers and declared input roles, preserve exact large numbers, and verify generated documents with Orrery 0.6.0 in CI. diff --git a/docs/agent-guide.md b/docs/agent-guide.md index 7e5a4d96b..fc5ae30cf 100644 --- a/docs/agent-guide.md +++ b/docs/agent-guide.md @@ -23,7 +23,7 @@ uv run ruff check . # lint ``` PR CI (`.github/workflows/test.yml`) has `lint`, `engine-free`, `engine-us`, -`engine-uk`, `integration-uk`, and `wheels` jobs. +`engine-uk`, `integration-uk`, `wheels`, and `orrery-contract` jobs. `tools/ci_test_plan.py` is the only authority for test-directory ownership, CI job assignment, country ownership, US timing-report categories, and changed-path country selection. Its `TEST_GROUPS` registry defines every valid @@ -63,6 +63,11 @@ Every automated test must run from `.github/workflows/test.yml`. Add a new test group to `TEST_GROUPS` before adding its executor to that workflow; do not create a separate selection system or test workflow. +The `orrery-contract` job is not a pytest group. It installs the exact public +Orrery version and locked Node dependency set under `tools/orrery-contract/`, +generates an Orrery document through Microcosm's public Python API, and requires +Orrery's public parser to accept it. It performs no browser rendering. + New commits to a PR cancel older unfinished CI runs for that same PR. Each main-push run has a unique concurrency group, so all main-push runs remain independent and can finish validating their merged changes. diff --git a/docs/graph-explorer.md b/docs/graph-explorer.md index caadf0044..29730cda8 100644 --- a/docs/graph-explorer.md +++ b/docs/graph-explorer.md @@ -1,7 +1,7 @@ # Graph explorer -For declaration and schema inspection with the shared viewer, see the -[shared graph explorer adapter](shared-graph-explorer-adapter.md). The run +For declaration and schema inspection with Orrery, see the +[Orrery adapter](orrery-adapter.md). The run renderer described below remains unchanged, including its separate cache and gate statuses. diff --git a/docs/orrery-adapter.md b/docs/orrery-adapter.md new file mode 100644 index 000000000..b8d099819 --- /dev/null +++ b/docs/orrery-adapter.md @@ -0,0 +1,175 @@ +# Orrery adapter + +`microcosm.graph.orrery` converts a Microcosm graph declaration and its +compiler-derived metadata into `graph-explorer/v1`, the JSON format accepted by +[Orrery](https://github.com/TheAxiomFoundation/orrery). Microcosm owns the +calculation and field semantics. Orrery owns graph navigation, presentation, +and self-contained HTML export. + +This conversion is separate from graph execution: + +1. `compile_graph(graph)` validates a `Graph` and derives operation order, + population versions, field owners, and operation predecessors. The executor + uses this `CompiledGraph`. +2. `graph_schema(compiled)` derives a portable static metadata artifact from + that same `CompiledGraph`. +3. `orrery_json(compiled)` converts the static metadata into Orrery input. It + does not execute the graph. +4. The separately installed Orrery command can turn that JSON into a + self-contained HTML file. + +The schema is therefore a generated artifact, not an instruction that +Microcosm executes. The direct API performs steps 2 and 3 in memory. A saved +schema lets another process repeat step 3 without importing the original graph +construction code. + +## Export from Python + +Use the public API when the `Graph` or `CompiledGraph` is already available: + +```python +from pathlib import Path + +from microcosm.graph import compile_graph, orrery_json + +compiled = compile_graph(graph) +Path("orrery.json").write_text( + orrery_json(compiled, title="Microcosm UK"), + encoding="utf-8", +) +``` + +Passing the original `Graph` to `orrery_json` produces the same bytes because +the function compiles it first. `orrery_document` returns the same result as +detached Python dictionaries and lists. + +To persist the intermediate compiler schema, serialize `graph_schema(compiled)` +with `microcosm.graph.canonical.canonical_json`. The schema has exactly these +root fields: + +```text +protocol, country, graph_sha256, graph, compiled, +fields, input_bindings, extensions +``` + +Only `extensions` is producer-defined. Before presentation, +`validate_graph_schema` restores and recompiles the embedded graph and requires +every other field to equal the newly derived result. A caller cannot change an +owner, predecessor, population version, field provider, or input binding while +retaining a valid schema. + +## Export from the command line + +The command requires an explicit input type and creates a new output file: + +```sh +uv run python -m microcosm.graph.orrery \ + --graph graph.json \ + --output orrery.json \ + --title "Microcosm UK" +``` + +For a previously saved compiler schema, use `--schema` instead of `--graph`: + +```sh +uv run python -m microcosm.graph.orrery \ + --schema compiled-schema.json \ + --output orrery.json +``` + +The command does not infer the input type. It rejects duplicate JSON keys, +non-finite numbers, streams and devices, oversized input, an invalid graph or +schema, and an existing output path. It does not install Orrery, download +assets, or execute a population operation. + +Install the exact viewer version independently, then create the HTML file: + +```sh +npm install --save-dev @axiom-foundation/orrery@0.6.0 +npx orrery --input orrery.json --output orrery.html +``` + +Orrery validates the JSON before export. Its HTML contains the viewer assets +and does not require a network connection to render. Microcosm CI also parses a +generated document with the exact `0.6.0` package and its locked transitive +dependencies. + +## Fields and input bindings + +Every field record identifies one visible value by: + +- `population`, `entity`, and `column`: its coordinate; +- `provider`: the operation whose artifact supplies that value; +- `declared_in`: the operation containing the nearest applicable `Owned` + declaration; +- `dtype`, `rows`, `ownership`, and `rewrite`: the declaration details. + +A structural operation carries all fields from its base population. For such a +carried field, `provider` is the structural operation because its output +artifact is what downstream operations read; `declared_in` still points to the +earlier `Owned` declaration. An operation that writes or rewrites a field is +the provider after that operation completes. + +`input_bindings` are also derived entirely from `Graph` and `CompiledGraph`. +They contain no Frame values and do not observe kernel behavior. Each record +connects one declared input role to the exact pre-operation field provider: + +| `kind` | Declared role | `rows` | +| --- | --- | --- | +| `slice` | A column named by a `Slice` | The slice's row selector | +| `slice_mask` | The boolean column used to select a slice's rows | `all` | +| `output_mask` | The boolean column limiting an `Owned` output | `all` | +| `rewrite_incumbent` | The prior value replaced by `Owned(rewrite=True)` | The output's row selector | + +For a rewrite, both the explicit slice and the implicit incumbent role resolve +to the value present before that operation. The final rewritten field is a +different identity. Repeated declared roles remain repeated schema records, +even when they refer to the same coordinate. + +These records describe values supplied to an operation. They do not claim that +a kernel read every supplied value at runtime, and they do not describe an +operation-specific input-writing mechanism. + +## Orrery document structure + +The adapter creates operation, source, and versioned-field nodes. Descriptions +are promoted to Orrery's visible `description` property while the complete +declaration remains in `data`. The entire compiler schema remains available at +`metadata.microcosm`. + +| Relationship | Category | Meaning | +| --- | --- | --- | +| `compiled_predecessor` | dependency | An operation dependency derived by `compile_graph` | +| `declared_read` | dependency | A slice, mask, or rewrite-incumbent input binding | +| `artifact_input` | dependency | A typed producer-to-consumer artifact declaration | +| `provided_field` | provenance | The operation that supplies a versioned field | +| `structural_input` | provenance | A base population field carried into a structural operation | +| `declared_source` | provenance | A named external input declaration | + +Structural carriage does not assert that values are unchanged. Declared reads +do not infer an individual mathematical formula for each output. Artifact and +source declarations do not establish that the referenced bytes exist or were +validated. + +Pre-rewrite values receive auxiliary field nodes when necessary. Stable IDs +include the population, coordinate, provider, and declaration source, so the +pre-rewrite value and final rewritten value cannot collide. + +## Evidence and limits + +The export contains static declarations only. It does not contain entity IDs, +memberships, field values, source bytes, kernel code, observed runtime reads, +cache results, execution receipts, verifier assessments, or release decisions. +Content revisions identify metadata bytes; they are not execution keys or +authenticity claims. + +Python integers outside JavaScript's safe range are represented as +`{"integer_literal": "..."}` before Orrery parses the document. Large integral +floats use `{"float_literal": "..."}`. Exports reject non-finite numbers, +excessive nesting, non-JSON objects, more than 20,000 presentation nodes or +100,000 relationships, input above 32 MiB, and output above 64 MiB. They fail +instead of silently omitting records. + +Runtime receipts and precise per-output value lineage could be added by a +separate runtime-aware adapter in the future. They cannot be inferred from the +static compiler schema and are intentionally absent here. diff --git a/docs/shared-graph-explorer-adapter.md b/docs/shared-graph-explorer-adapter.md deleted file mode 100644 index e244254f2..000000000 --- a/docs/shared-graph-explorer-adapter.md +++ /dev/null @@ -1,126 +0,0 @@ -# Shared graph explorer adapter - -`microcosm.graph.explorer` converts a `microcosm.graph.schema.v1` metadata -export into `graph-explorer/v1`, the portable contract consumed by [Orrery](https://github.com/TheAxiomFoundation/orrery), the -shared graph viewer. Microcosm owns the graph's calculation -and schema meanings. The shared package owns navigation, rendering and bundled -offline HTML. - -The adapter adds no JavaScript dependency to Microcosm. It reads no microdata, -executes no population operation, and leaves `explain_html` unchanged. - -## Open a saved schema in Orrery - -From a synced Microcosm checkout, export a saved schema with: - -```sh -uv run python -m microcosm.graph.explorer \ - --input compiled-schema.json --output graph.json --title "Microcosm US" -graph-explorer --input graph.json --output graph.html -``` - -The second command uses the separately installed, accepted shared viewer -0.3.0 CLI, whose package name is still `@axiom-foundation/graph-explorer`. -The forthcoming Orrery package rename does not change the JSON contract. -Microcosm does not install or upgrade the viewer as part of export. - -Open `graph.html` using a local static server or your existing artifact viewer. -The HTML contains its viewer assets and needs no CDN. Direct `file://` opening -is outside the current Microcosm browser acceptance. - -The Python command accepts only saved schema JSON, rejects duplicate keys and -non-finite numbers, bounds input before decoding, and creates a new output file. -It refuses an existing output, including the input path. A saved manifest or -microdata file is not a schema export. Review schema metadata before sharing it: -the entire supplied metadata is retained, and canvas filtering does not redact it. - -The same exporter is available as a Python API: - -```python -import json -from pathlib import Path - -from microcosm.graph.explorer import graph_explorer_json - -schema = json.loads(Path("compiled-schema.json").read_text(encoding="utf-8")) -Path("graph.json").write_text(graph_explorer_json(schema), encoding="utf-8") -``` - -The schema producer is being integrated separately. In a checkout containing -`microcosm.graph.schema.graph_schema`, the input can be produced directly with -`graph_schema(compiled)`. This adapter also accepts previously saved exports; -it does not require that producer to be installed. `graph_explorer_document` -returns the same document as detached Python dictionaries and lists. - -The shared package's built CLI accepts: - -```sh -graph-explorer --input graph.json --output graph.html -``` - -The shared package must supply its built viewer assets. Microcosm does not -download them or launch a build automatically. HTML export and browser -verification are distinct checks; a successfully written HTML file alone does -not establish that its embedded viewer runs. - -## What the graph contains - -The document contains every operation, source declaration, visible field version -and input binding in the supplied schema. A field identity contains its -population, entity, column, producer and nearest declaration. A pre-rewrite -input therefore remains distinct from the final value in the same population. -Population coordinates are retained in `data`; they are not replaced by a -presentation revision or containment parent. - -| Edge kind | Meaning | -| --- | --- | -| `compiled_predecessor` | An operation dependency listed in the supplied compiler metadata | -| `produced` | The provider of a versioned field value, including a structural carrier | -| `declared_read` | An operation's slice, slice mask, output mask or rewrite incumbent; role and row mask are retained | -| `structural_input` | Ancestry from a completed base population's fields into its structural successor | -| `source` | A named external input and its declared codec | -| `artifact` | A producer-to-consumer dependency with its exact alias, artifact name and nominal type/version | - -Structural ancestry does not imply unchanged values. Declared reads describe -operation-level dependencies; they do not infer a separate mathematical formula -for each output. Typed artifact declarations do not establish the existence of -runtime artifact bytes. Source declarations remain domain nodes; citation -references cannot substitute for their codec contract. - -The complete original metadata stays in `metadata.microcosm`, including compiled -owners/order/versions and declaration details such as `mass_partition`. Exact -Python integers outside JavaScript's safe range are transported as -`{"integer_literal": "..."}` before JavaScript parses the document. Large integral -floats use `{"float_literal": "..."}` so their type is not silently changed to -integer. The raw declaration digest is retained separately from presentation -revisions. - -## Scope, identity and evidence - -`complete_supplied_schema` means the complete supplied metadata snapshot. It -does not mean the complete US recipe. Entity IDs and memberships, scientific -units, period semantics, source preparation internals and implicit runtime reads -cannot be inferred when the source schema omits them. Declared ownership is -not evidence that a field contains valid materialized values. - -The converter checks JSON bounds, declaration digest consistency, references, -field/declaration agreement and exact read-role coverage. It does not recompile -or authenticate an imported dictionary. The document revision hashes the whole -supplied schema, including its compiler tables; node and edge revisions hash -their transport records before the revision field is attached. These are -metadata content digests, not execution keys or source-byte attestations. - -No execution, authorship, cache, gate or release status is inferred from schema -metadata. This first adapter supplies no runtime activities or Receipt verdicts. -The existing deterministic run viewer retains its independent cache/gate axes. -A later run adapter must bind actual run evidence and keep these statuses -separate. Receipt assessments must come from a configured host verifier and -bind to the exact exported document bytes, not merely the source declaration's -hash. Custody verification does not establish scientific correctness. - -Exports fail explicitly at their development bounds rather than silently -truncating: 256 operations, 256 sources, 20,000 presentation nodes, 100,000 edges, -32 MiB input and 64 MiB output. Prospective transport bytes are charged as -records are added. These are metadata limits, not microdata size limits or a -claim about total Python process memory. The viewer can focus or collapse a -complete document without changing the underlying export. diff --git a/packages/microcosm-graph/README.md b/packages/microcosm-graph/README.md index ab0edf67d..5b50fc1f2 100644 --- a/packages/microcosm-graph/README.md +++ b/packages/microcosm-graph/README.md @@ -29,11 +29,12 @@ Module map: | `executor.py` | `run_graph`: projection, patching, ownership enforcement, receipts | | `manifest.py` | `RunManifest`, `NodeReceipt`, human decision records | | `view.py` | `describe(node)`: the one-screen view | -| `explorer.py` | Pure adapter from schema metadata to the shared `graph-explorer/v1` presentation contract | +| `schema.py` | Canonical compiler-derived fields and declared input bindings for presentation adapters | +| `orrery.py` | Pure adapter from `Graph` or compiler schema to Orrery's `graph-explorer/v1` input | The shard depends on `microcosm-frame` only. Kernels that wrap fit, calibrate, or a rules engine live in those shards and register here. -See [the shared graph explorer adapter](../../docs/shared-graph-explorer-adapter.md) for -versioned field inspection, offline integration and evidence boundaries. The +See [the Orrery adapter](../../docs/orrery-adapter.md) for versioned field +inspection, offline integration, and evidence boundaries. The existing deterministic `explain_html` export remains available unchanged. From 3a423759cbe82475852d4f0f8d3edcb3ec4c3bd2 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Mon, 5 Oct 2026 17:27:47 +0400 Subject: [PATCH 08/13] Integrate Orrery check into engine-free tests --- .github/workflows/test.yml | 24 ++------ changelog.d/orrery-test-category.fixed.md | 1 + docs/agent-guide.md | 44 ++++++++------ .../engine_free/shared/test_ci_test_plan.py | 39 ++++++++++++ .../engine_free/shared/test_graph_orrery.py | 17 ++++++ tools/check_orrery_contract.py | 58 ------------------ tools/ci_test_plan.py | 59 ++++++++++++++++++- tools/run_engine_free_tests.sh | 2 + 8 files changed, 149 insertions(+), 95 deletions(-) create mode 100644 changelog.d/orrery-test-category.fixed.md delete mode 100644 tools/check_orrery_contract.py diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 9389235b3..2a363efda 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -60,6 +60,11 @@ jobs: - uses: astral-sh/setup-uv@v6 with: python-version: ${{ matrix.python-version }} + - uses: actions/setup-node@v7 + with: + node-version: "24" + cache: npm + cache-dependency-path: tools/orrery-contract/package-lock.json - name: Sync workspace without country engines run: uv sync --all-packages --locked - name: Run engine-free tests @@ -143,22 +148,3 @@ jobs: python-version: "3.13" - name: Build and inspect every shard wheel run: tools/build_and_inspect_wheels.sh - - orrery-contract: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v6 - with: - python-version: "3.13" - - uses: actions/setup-node@v7 - with: - node-version: "24" - cache: npm - cache-dependency-path: tools/orrery-contract/package-lock.json - - name: Sync workspace - run: uv sync --all-packages --locked - - name: Install the exact Orrery contract - run: npm ci --ignore-scripts --prefix tools/orrery-contract - - name: Parse a Microcosm export with Orrery - run: uv run --no-sync python tools/check_orrery_contract.py diff --git a/changelog.d/orrery-test-category.fixed.md b/changelog.d/orrery-test-category.fixed.md new file mode 100644 index 000000000..1533a5d75 --- /dev/null +++ b/changelog.d/orrery-test-category.fixed.md @@ -0,0 +1 @@ +Run Orrery parser compatibility through the registered shared engine-free test category, and reject unregistered behavioral-test jobs in the CI workflow. diff --git a/docs/agent-guide.md b/docs/agent-guide.md index fc5ae30cf..9cb56fe60 100644 --- a/docs/agent-guide.md +++ b/docs/agent-guide.md @@ -18,12 +18,14 @@ the PEP 420 namespace `microcosm.`: `frame`, `fit`, `calibrate`, `build`, uv sync --all-packages # set up the whole workspace uv sync --all-packages --locked --extra us # US engine environment uv sync --all-packages --locked --extra uk # UK engine environment -uv run pytest # engine-free tests; integration tests remain excluded +bash tools/run_engine_free_tests.sh # complete engine-free test category +uv run pytest # focused test while developing uv run ruff check . # lint ``` PR CI (`.github/workflows/test.yml`) has `lint`, `engine-free`, `engine-us`, -`engine-uk`, `integration-uk`, `wheels`, and `orrery-contract` jobs. +`engine-uk`, `integration-uk`, and `wheels` jobs, plus the +`select-countries` orchestration job. `tools/ci_test_plan.py` is the only authority for test-directory ownership, CI job assignment, country ownership, US timing-report categories, and changed-path country selection. Its `TEST_GROUPS` registry defines every valid @@ -34,13 +36,14 @@ or country mapping. Each ordinary behavioral job has a Python 3.13/3.14 matrix and reports the 25 slowest tests. The engine-free job installs no country extra, always runs every -engine-free test, and distributes files across two pytest workers with `--dist -loadfile`. Changed-file selection only controls the more resource-intensive -country jobs. A documentation-only pull request selects neither country; a -US-only or UK-only pull request selects that country; shared, mixed, unknown, -or empty changed-path sets select both. Main pushes select both without querying -the pull-request API. The UK integration job follows the same UK selection as -the ordinary UK engine job. +engine-free test, installs its locked JavaScript test dependencies, and +distributes files across two pytest workers with `--dist loadfile`. +Changed-file selection only controls the more resource-intensive country jobs. +A documentation-only pull request selects neither country; a US-only or UK-only +pull request selects that country; shared, mixed, unknown, or empty changed-path +sets select both. Main pushes select both without querying the pull-request API. +The UK integration job follows the same UK selection as the ordinary UK engine +job. The US engine job runs contract and scenario categories in small pytest processes with at most two processes active at once, then runs each complete @@ -59,14 +62,21 @@ Ordinary behavioral jobs pass `-v --tb=short --maxfail=1 --durations=25`, so eac names tests as they run, prints a concise first-failure traceback, and reports its 25 slowest tests. -Every automated test must run from `.github/workflows/test.yml`. Add a new test -group to `TEST_GROUPS` before adding its executor to that workflow; do not create -a separate selection system or test workflow. - -The `orrery-contract` job is not a pytest group. It installs the exact public -Orrery version and locked Node dependency set under `tools/orrery-contract/`, -generates an Orrery document through Microcosm's public Python API, and requires -Orrery's public parser to accept it. It performs no browser rendering. +Every behavioral or cross-language compatibility assertion must be a pytest +test in a directory registered by `TEST_GROUPS` and must run through that +category's existing workflow job. Supporting programs and data may live under +`tools/`, but they do not receive separate workflow jobs. The test-plan verifier +allows only registered test jobs and the explicitly declared orchestration, +lint, and wheel-building jobs. Add a test category to `TEST_GROUPS` before its +executor; do not create a separate selection system, test workflow, or +single-purpose behavioral test job. + +The Orrery parser compatibility assertion lives in +`packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py`. The +engine-free runner installs the exact public Orrery version and locked Node +dependency set under `tools/orrery-contract/`; the pytest test generates a +document through Microcosm's public Python API and requires Orrery's public +parser to accept it. It performs no browser rendering. New commits to a PR cancel older unfinished CI runs for that same PR. Each main-push run has a unique concurrency group, so all main-push runs diff --git a/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py b/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py index 5c6d1cd82..11bb65e84 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py @@ -128,6 +128,45 @@ def test_registry_directories_are_unique() -> None: assert len(directories) == len(set(directories)) +def test_workflow_jobs_match_registered_test_and_infrastructure_jobs() -> None: + source = ci_test_plan.WORKFLOW.read_text(encoding="utf-8") + + assert ci_test_plan.workflow_job_errors(source) == () + + +def test_workflow_rejects_a_separate_behavioral_test_job() -> None: + source = """\ +jobs: + engine-free: + steps: + - run: pytest + invented-compatibility-test: + steps: + - run: node verify.mjs +""" + + assert ( + "invented-compatibility-test: workflow job has no registered test " + "category or approved infrastructure role" + in ci_test_plan.workflow_job_errors(source) + ) + + +def test_workflow_job_parser_ignores_nested_yaml_keys() -> None: + source = """\ +jobs: + engine-free: + strategy: + matrix: + python-version: ["3.13", "3.14"] + steps: + - name: Run tests + run: pytest +""" + + assert ci_test_plan.workflow_job_names(source) == ("engine-free",) + + def test_engine_guard_detection_ignores_strings_but_finds_executable_guards() -> None: source = """ TEXT = '@pytest.mark.requires_uk' diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py index 6fd2168e1..b83bfac74 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py @@ -28,6 +28,10 @@ compiled_default_population_graph, compiled_graph, ) +from test_support.paths import paths_for + +_TEST_PATHS = paths_for("microcosm-graph") +_ORRERY_VERIFY = _TEST_PATHS.repository / "tools" / "orrery-contract" / "verify.mjs" def _parts(identity): @@ -247,6 +251,19 @@ def test_direct_graph_compiled_and_saved_schema_paths_are_identical(): assert orrery_document(compiled)["schemaVersion"] == "graph-explorer/v1" +def test_public_orrery_parser_accepts_generated_document(): + result = subprocess.run( + ["node", str(_ORRERY_VERIFY)], + input=orrery_json(compiled_graph()), + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0, result.stderr + assert result.stdout == "Orrery accepted the Microcosm graph document.\n" + + def test_cli_requires_an_explicit_input_type_and_matches_direct_output(tmp_path): compiled = compiled_graph() graph_path = tmp_path / "graph.json" diff --git a/tools/check_orrery_contract.py b/tools/check_orrery_contract.py deleted file mode 100644 index 302ffd96e..000000000 --- a/tools/check_orrery_contract.py +++ /dev/null @@ -1,58 +0,0 @@ -#!/usr/bin/env python3 -"""Pass a public Microcosm export through the pinned Orrery parser.""" - -from __future__ import annotations - -import subprocess -from pathlib import Path - -from microcosm.graph import ( - Graph, - Node, - Owned, - Slice, - SourceRef, - StructuralDelta, - orrery_json, -) - -ROOT = Path(__file__).resolve().parents[1] -VERIFY = ROOT / "tools" / "orrery-contract" / "verify.mjs" - - -def contract_graph() -> Graph: - """Use omitted population to cover the compiler-resolved API contract.""" - - return Graph( - "contract", - sources=(SourceRef("fixture", "contract@1"),), - nodes=( - Node( - "source", - "contract.create@1", - structural=StructuralDelta.CREATE, - sources=("fixture",), - outputs=(Owned("person", "amount", "float64"),), - ), - Node( - "consumer", - "contract.consume@1", - inputs=(Slice("person", ("amount",)),), - outputs=(Owned("person", "result", "float64"),), - ), - ), - ) - - -def main() -> int: - subprocess.run( - ["node", str(VERIFY)], - input=orrery_json(contract_graph()), - text=True, - check=True, - ) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/tools/ci_test_plan.py b/tools/ci_test_plan.py index 380920c8f..550846c10 100644 --- a/tools/ci_test_plan.py +++ b/tools/ci_test_plan.py @@ -7,6 +7,7 @@ import ast import json import os +import re import sys import urllib.request from collections.abc import Callable, Iterable, Mapping @@ -16,6 +17,7 @@ ROOT = Path(__file__).resolve().parents[1] PACKAGES = ROOT / "packages" +WORKFLOW = ROOT / ".github" / "workflows" / "test.yml" API_ROOT = "https://api.github.com" @@ -73,6 +75,11 @@ class TestGroup: } GROUP_BY_DIRECTORY = {spec.directory: name for name, spec in TEST_GROUPS.items()} JOBS = tuple(dict.fromkeys(spec.job for spec in TEST_GROUPS.values())) +WORKFLOW_INFRASTRUCTURE_JOBS = { + "select-countries": "selects the country-specific test categories", + "lint": "runs static checks and verifies this test plan", + "wheels": "builds and inspects distribution archives", +} DOC_PREFIXES = ("docs/",) DOC_FILES = {"LICENSE"} @@ -80,6 +87,56 @@ class TestGroup: RequestJSON = Callable[[str, str], Any] +_WORKFLOW_JOB = re.compile(r"^ ([A-Za-z0-9][A-Za-z0-9_-]*):(?:\s.*)?$") + + +def workflow_job_names(source: str) -> tuple[str, ...]: + """Return top-level job names from the repository's workflow YAML.""" + + in_jobs = False + names: list[str] = [] + for line in source.splitlines(): + if not in_jobs: + if line == "jobs:": + in_jobs = True + continue + if line and not line[0].isspace() and not line.startswith("#"): + break + match = _WORKFLOW_JOB.fullmatch(line) + if match is not None: + names.append(match.group(1)) + if not in_jobs: + raise ValueError("workflow has no top-level jobs mapping") + return tuple(names) + + +def workflow_job_errors(source: str) -> tuple[str, ...]: + """Return errors for workflow jobs outside the declared test plan.""" + + try: + names = workflow_job_names(source) + except ValueError as error: + return (str(error),) + + errors: list[str] = [] + seen: set[str] = set() + for name in names: + if name in seen: + errors.append(f"{name}: workflow job is declared more than once") + seen.add(name) + + expected = set(JOBS) | set(WORKFLOW_INFRASTRUCTURE_JOBS) + errors.extend( + f"{name}: workflow job has no registered test category or approved " + "infrastructure role" + for name in sorted(seen - expected) + ) + errors.extend( + f"{name}: registered workflow job is missing" + for name in sorted(expected - seen) + ) + return tuple(errors) + def engine_guard_lines(source: str) -> tuple[int, ...]: """Return lines containing country-engine availability checks.""" @@ -236,7 +293,7 @@ def verify() -> None: if not files: raise SystemExit("no test files found below packages/*/tests") - errors: list[str] = [] + errors = list(workflow_job_errors(WORKFLOW.read_text(encoding="utf-8"))) counts = {name: 0 for name in TEST_GROUPS} seen_directories: dict[tuple[str, str], str] = {} seen_report_categories: dict[tuple[str, str], str] = {} diff --git a/tools/run_engine_free_tests.sh b/tools/run_engine_free_tests.sh index 27281d25d..161f7479c 100755 --- a/tools/run_engine_free_tests.sh +++ b/tools/run_engine_free_tests.sh @@ -1,6 +1,8 @@ #!/usr/bin/env bash set -euo pipefail +npm ci --ignore-scripts --prefix tools/orrery-contract + files=() while IFS= read -r file; do files+=("$file"); done < <( uv run --no-sync python tools/ci_test_plan.py list-job engine-free From 52e55457493a531c7c715bac22492df4be164e44 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Tue, 6 Oct 2026 21:55:49 +0400 Subject: [PATCH 09/13] Fix Orrery schema export review issues --- changelog.d/orrery-schema-cli.added.md | 2 +- docs/agent-guide.md | 2 +- docs/orrery-adapter.md | 21 ++++++---- .../src/microcosm/graph/decl.py | 39 +++++++++++++++++++ .../src/microcosm/graph/executor.py | 31 +++------------ .../src/microcosm/graph/schema.py | 29 +++++++++++++- .../engine_free/shared/test_graph_orrery.py | 38 +++++++++++++++++- .../engine_free/shared/test_graph_schema.py | 34 +++++++++++++++- test_support/microcosm_graph/schema.py | 8 ++++ tools/orrery-contract/package-lock.json | 8 ++-- tools/orrery-contract/package.json | 2 +- tools/orrery-contract/verify.mjs | 33 +++++++++++++++- 12 files changed, 203 insertions(+), 44 deletions(-) diff --git a/changelog.d/orrery-schema-cli.added.md b/changelog.d/orrery-schema-cli.added.md index 4ee151ae4..ef390d497 100644 --- a/changelog.d/orrery-schema-cli.added.md +++ b/changelog.d/orrery-schema-cli.added.md @@ -1 +1 @@ -Add a compiler-derived graph schema, direct Orrery export APIs, and an explicit `python -m microcosm.graph.orrery` command for graph declarations or saved schemas. Recompile imported metadata, retain field providers and declared input roles, preserve exact large numbers, and verify generated documents with Orrery 0.6.0 in CI. +Add a compiler-derived graph schema, direct Orrery export APIs, and an explicit `python -m microcosm.graph.orrery` command for graph declarations or saved schemas. Recompile imported metadata, compare derived JSON types exactly, retain field providers and declared input roles including expansion-materialization claims, preserve exact large numbers, and verify generated documents with the supported public Orrery package in CI. diff --git a/docs/agent-guide.md b/docs/agent-guide.md index 9cb56fe60..0a7351b97 100644 --- a/docs/agent-guide.md +++ b/docs/agent-guide.md @@ -73,7 +73,7 @@ single-purpose behavioral test job. The Orrery parser compatibility assertion lives in `packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py`. The -engine-free runner installs the exact public Orrery version and locked Node +engine-free runner installs the supported public Orrery range and locked Node dependency set under `tools/orrery-contract/`; the pytest test generates a document through Microcosm's public Python API and requires Orrery's public parser to accept it. It performs no browser rendering. diff --git a/docs/orrery-adapter.md b/docs/orrery-adapter.md index b8d099819..80e74929b 100644 --- a/docs/orrery-adapter.md +++ b/docs/orrery-adapter.md @@ -82,17 +82,17 @@ non-finite numbers, streams and devices, oversized input, an invalid graph or schema, and an existing output path. It does not install Orrery, download assets, or execute a population operation. -Install the exact viewer version independently, then create the HTML file: +Install the viewer independently, then create the HTML file: ```sh -npm install --save-dev @axiom-foundation/orrery@0.6.0 +npm install --save-dev @axiom-foundation/orrery npx orrery --input orrery.json --output orrery.html ``` Orrery validates the JSON before export. Its HTML contains the viewer assets -and does not require a network connection to render. Microcosm CI also parses a -generated document with the exact `0.6.0` package and its locked transitive -dependencies. +and does not require a network connection to render. Microcosm CI parses a +generated document with the compatible range declared by its contract-test +package and a locked set of transitive dependencies. ## Fields and input bindings @@ -120,15 +120,22 @@ connects one declared input role to the exact pre-operation field provider: | `slice_mask` | The boolean column used to select a slice's rows | `all` | | `output_mask` | The boolean column limiting an `Owned` output | `all` | | `rewrite_incumbent` | The prior value replaced by `Owned(rewrite=True)` | The output's row selector | +| `materialized_expand_output` | A new column physically installed by an `EXPAND` operation and read by its following ownership-claim operation | The output's row selector | For a rewrite, both the explicit slice and the implicit incumbent role resolve to the value present before that operation. The final rewritten field is a different identity. Repeated declared roles remain repeated schema records, even when they refer to the same coordinate. +For `materialized_expand_output`, the input field's provider is the `EXPAND` +operation and its declaration source is the following claim operation. The +claim's final output receives a separate field identity, preserving both the +physical materialization and the ownership declaration in the presentation. + These records describe values supplied to an operation. They do not claim that -a kernel read every supplied value at runtime, and they do not describe an -operation-specific input-writing mechanism. +a kernel read every supplied value at runtime, and they do not reproduce the +executor's complete causal-writer history, which may be refined by runtime +receipts. ## Orrery document structure diff --git a/packages/microcosm-graph/src/microcosm/graph/decl.py b/packages/microcosm-graph/src/microcosm/graph/decl.py index 24c92ab7a..fc0dfad85 100644 --- a/packages/microcosm-graph/src/microcosm/graph/decl.py +++ b/packages/microcosm-graph/src/microcosm/graph/decl.py @@ -625,6 +625,45 @@ def normative(self) -> dict[str, object]: } +def materialized_expand_coordinates( + node: Node, +) -> frozenset[tuple[str, str]]: + """Return the carried EXPAND cells an ordinary node claims as outputs. + + ``materialized_expand_outputs`` is a reserved node parameter used when an + EXPAND kernel physically installs a new column before a following ordinary + node gives that column an ownership declaration. The claim node therefore + receives the existing values as inputs even though the coordinate is not a + ``Slice`` or rewrite. + """ + + raw_materialized = node.params.get("materialized_expand_outputs", ()) + if not isinstance(raw_materialized, tuple) or any( + not isinstance(value, str) or "." not in value for value in raw_materialized + ): + raise GraphError( + f"Node {node.id!r} params['materialized_expand_outputs'] must be a " + "tuple of 'entity.column' strings." + ) + materialized: set[tuple[str, str]] = set() + owned_by_coordinate = { + (output.entity, output.column): output for output in node.outputs + } + for value in raw_materialized: + entity, column = value.split(".", 1) + coordinate = (entity, column) + output = owned_by_coordinate.get(coordinate) + if output is None or output.rewrite: + raise GraphError( + f"Node {node.id!r} materialized EXPAND output {value!r} must be " + "one of its non-rewrite owned cells." + ) + materialized.add(coordinate) + if len(materialized) != len(raw_materialized): + raise GraphError(f"Node {node.id!r} repeats a materialized EXPAND output.") + return frozenset(materialized) + + @dataclass(frozen=True) class Graph: """A complete build declaration for one country. diff --git a/packages/microcosm-graph/src/microcosm/graph/executor.py b/packages/microcosm-graph/src/microcosm/graph/executor.py index a1e3377f1..1bcebd04b 100644 --- a/packages/microcosm-graph/src/microcosm/graph/executor.py +++ b/packages/microcosm-graph/src/microcosm/graph/executor.py @@ -35,10 +35,12 @@ GATE_OUTCOMES, ROWS_ALL, CompiledGraph, + GraphError, Node, Owned, Ownership, StructuralDelta, + materialized_expand_coordinates, ) from .errors import NodeRejectedError from .kernel import ( @@ -646,31 +648,10 @@ def _structural_columns(frame: Frame, entity: str) -> list[str]: def _materialized_expand_coordinates(node: Node) -> frozenset[tuple[str, str]]: """Return and validate the carried EXPAND cells an ordinary node claims.""" - raw_materialized = node.params.get("materialized_expand_outputs", ()) - if not isinstance(raw_materialized, tuple) or any( - not isinstance(value, str) or "." not in value for value in raw_materialized - ): - raise NodeRejected( - f"Node {node.id!r} params['materialized_expand_outputs'] must be a " - "tuple of 'entity.column' strings." - ) - materialized: set[tuple[str, str]] = set() - owned_by_coordinate = { - (output.entity, output.column): output for output in node.outputs - } - for value in raw_materialized: - entity, column = value.split(".", 1) - coordinate = (entity, column) - output = owned_by_coordinate.get(coordinate) - if output is None or output.rewrite: - raise NodeRejected( - f"Node {node.id!r} materialized EXPAND output {value!r} must be " - "one of its non-rewrite owned cells." - ) - materialized.add(coordinate) - if len(materialized) != len(raw_materialized): - raise NodeRejected(f"Node {node.id!r} repeats a materialized EXPAND output.") - return frozenset(materialized) + try: + return materialized_expand_coordinates(node) + except GraphError as error: + raise NodeRejected(str(error)) from error def _expand_rewrite_coordinates( diff --git a/packages/microcosm-graph/src/microcosm/graph/schema.py b/packages/microcosm-graph/src/microcosm/graph/schema.py index f917b90d9..fdab9a6c3 100644 --- a/packages/microcosm-graph/src/microcosm/graph/schema.py +++ b/packages/microcosm-graph/src/microcosm/graph/schema.py @@ -22,6 +22,7 @@ Owned, StructuralDelta, compile_graph, + materialized_expand_coordinates, ) from .serialize import graph_from_json, graph_to_json @@ -315,6 +316,32 @@ def _input_bindings( ROWS_ALL, ) ) + if "materialized_expand_outputs" in node.params: + holder = by_id[population] + if holder.structural is not StructuralDelta.EXPAND: + raise GraphError( + f"Node {node.id!r} uses materialized_expand_outputs on " + f"population {population!r}, which is not an EXPAND version." + ) + for entity, column in sorted(materialized_expand_coordinates(node)): + if compiled.owners.get((population, entity, column)) != node.id: + raise GraphError( + f"Node {node.id!r} does not uniquely own materialized EXPAND " + f"output {entity}.{column} in population {population!r}." + ) + owned = _owned(node, entity, column) + bindings.append( + { + "node": node.id, + "population": population, + "entity": entity, + "column": column, + "provider": population, + "declared_in": node.id, + "kind": "materialized_expand_output", + "rows": owned.rows, + } + ) return bindings @@ -373,6 +400,6 @@ def validate_graph_schema(value: object) -> dict[str, object]: raise ValueError("Graph schema extensions must be an object.") restored = graph_from_json(canonical_json(graph_value).decode("utf-8")) expected = graph_schema(compile_graph(restored), extensions=extensions) - if document != expected: + if canonical_json(document) != canonical_json(expected): raise ValueError("Graph schema core metadata disagrees with its graph.") return expected diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py index b83bfac74..88ac4d0cf 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_orrery.py @@ -46,7 +46,7 @@ def test_document_is_complete_static_metadata_with_no_runtime_claims(): assert document["metadata"]["microcosm"] == schema assert document["metadata"]["scope"] == "complete_compiler_schema" assert document["metadata"]["truncated"] is False - assert len([node for node in document["nodes"] if node["kind"] == "operation"]) == 5 + assert len([node for node in document["nodes"] if node["kind"] == "operation"]) == 6 assert len([node for node in document["nodes"] if node["kind"] == "source"]) == 1 assert ( not { @@ -121,6 +121,37 @@ def test_rewrite_incumbent_has_a_distinct_auxiliary_field(): assert rewrite_read["source"] == incumbent["id"] +def test_materialized_expand_output_has_a_distinct_input_field(): + document = orrery_document_from_schema(graph_schema(compiled_graph())) + clones = [ + node + for node in document["nodes"] + if node["kind"] == "field" + and node["data"]["population"] == "expanded" + and node["data"]["column"] == "clone" + ] + assert len(clones) == 2 + materialized = next( + node for node in clones if node["data"]["provider"] == "expanded" + ) + claimed = next( + node for node in clones if node["data"]["provider"] == "expanded.claim" + ) + assert materialized["data"]["declared_in"] == "expanded.claim" + assert materialized["data"]["visible_in_schema"] is False + assert claimed["data"]["visible_in_schema"] is True + + target = json.dumps(["operation", "expanded.claim"], separators=(",", ":")) + read = next( + edge + for edge in document["edges"] + if edge["target"] == target + and edge["kind"] == "declared_read" + and edge["data"]["read_kind"] == "materialized_expand_output" + ) + assert read["source"] == materialized["id"] + + def test_dependencies_and_provenance_are_separate_categories(): document = orrery_document_from_schema(graph_schema(compiled_graph())) dependency_kinds = { @@ -254,7 +285,10 @@ def test_direct_graph_compiled_and_saved_schema_paths_are_identical(): def test_public_orrery_parser_accepts_generated_document(): result = subprocess.run( ["node", str(_ORRERY_VERIFY)], - input=orrery_json(compiled_graph()), + input=orrery_json( + compiled_graph(), + extensions={"large": 2**80, "negative": -(2**80), "float": 1e100}, + ), text=True, capture_output=True, check=False, diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py index d29b43399..768c789b4 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py @@ -81,8 +81,20 @@ def test_fields_record_carriage_rewrites_and_nearest_declaration(): assert _field(schema, "filtered", "keep")["declared_in"] == "base" assert _field(schema, "expanded", "age")["provider"] == "expanded" assert _field(schema, "expanded", "age")["declared_in"] == "rewrite" + assert _field(schema, "expanded", "clone") == { + "population": "expanded", + "entity": "person", + "column": "clone", + "dtype": "int64", + "provider": "expanded.claim", + "declared_in": "expanded.claim", + "rows": "all", + "ownership": "produced", + "rewrite": False, + } assert _field(schema, "reweighted", "age")["provider"] == "reweighted" - assert len(schema["fields"]) == 8 + assert _field(schema, "reweighted", "clone")["declared_in"] == "expanded.claim" + assert len(schema["fields"]) == 10 def test_input_bindings_preserve_each_declared_read_role(): @@ -147,6 +159,18 @@ def test_input_bindings_preserve_each_declared_read_role(): ] == "rewrite" ) + assert next( + binding for binding in bindings if binding["node"] == "expanded.claim" + ) == { + "node": "expanded.claim", + "population": "expanded", + "entity": "person", + "column": "clone", + "provider": "expanded", + "declared_in": "expanded.claim", + "kind": "materialized_expand_output", + "rows": "all", + } def test_omitted_population_uses_the_compiler_resolved_version(): @@ -188,6 +212,14 @@ def test_validation_recompiles_and_refuses_core_metadata_changes(mutation): validate_graph_schema(schema) +def test_validation_rejects_equal_values_with_different_json_types(): + schema = graph_schema(compiled_graph()) + schema["fields"][0]["rewrite"] = 0 + + with pytest.raises(ValueError, match="core metadata"): + validate_graph_schema(schema) + + def test_extensions_are_detached_bounded_json(): extension = {"producer": {"labels": ["one", 2**80]}} schema = graph_schema(compiled_graph(), extensions=extension) diff --git a/test_support/microcosm_graph/schema.py b/test_support/microcosm_graph/schema.py index b68c81669..a56672b45 100644 --- a/test_support/microcosm_graph/schema.py +++ b/test_support/microcosm_graph/schema.py @@ -68,6 +68,14 @@ def compiled_graph(): structural=StructuralDelta.EXPAND, base="filtered", inputs=(Slice("person", ("age",)),), + params={"expand_cells": (("person", "clone", "int64"),)}, + ), + Node( + "expanded.claim", + "unused.claim@1", + population="expanded", + outputs=(Owned("person", "clone", "int64"),), + params={"materialized_expand_outputs": ("person.clone",)}, ), Node( "reweighted", diff --git a/tools/orrery-contract/package-lock.json b/tools/orrery-contract/package-lock.json index da6b11072..bb093cdcb 100644 --- a/tools/orrery-contract/package-lock.json +++ b/tools/orrery-contract/package-lock.json @@ -8,15 +8,15 @@ "name": "microcosm-orrery-contract", "version": "0.0.0", "dependencies": { - "@axiom-foundation/orrery": "0.6.0", + "@axiom-foundation/orrery": ">=0.6.1 <0.7.0", "react": "19.2.8", "react-dom": "19.2.8" } }, "node_modules/@axiom-foundation/orrery": { - "version": "0.6.0", - "resolved": "https://registry.npmjs.org/@axiom-foundation/orrery/-/orrery-0.6.0.tgz", - "integrity": "sha512-Dva+L5cVopuZ6V+TcP8W6IxZkKD4RhWvBcIG5wr07b2KVKws/ay1GfsafIp9pFB1e9IlhxfSzo4bzzvciHyj3g==", + "version": "0.6.1", + "resolved": "https://registry.npmjs.org/@axiom-foundation/orrery/-/orrery-0.6.1.tgz", + "integrity": "sha512-VHYAuBZ50KhSXou1ogGG5IBjG1ET3Mnr05rIwUtoUgmgFu7eg5iH9rlsLQ28KQI/Lkd6yBsdYUsE3wo7kyCkhA==", "license": "MIT", "dependencies": { "@dagrejs/dagre": "3.1.1", diff --git a/tools/orrery-contract/package.json b/tools/orrery-contract/package.json index 03e5094cf..22dec01dc 100644 --- a/tools/orrery-contract/package.json +++ b/tools/orrery-contract/package.json @@ -4,7 +4,7 @@ "version": "0.0.0", "type": "module", "dependencies": { - "@axiom-foundation/orrery": "0.6.0", + "@axiom-foundation/orrery": ">=0.6.1 <0.7.0", "react": "19.2.8", "react-dom": "19.2.8" } diff --git a/tools/orrery-contract/verify.mjs b/tools/orrery-contract/verify.mjs index 7cff4f96b..104b168bb 100644 --- a/tools/orrery-contract/verify.mjs +++ b/tools/orrery-contract/verify.mjs @@ -8,6 +8,37 @@ const document = parseGraphDocument(JSON.parse(Buffer.concat(chunks).toString('u assert.equal(document.schemaVersion, 'graph-explorer/v1'); assert.equal(document.metadata.adapter, 'microcosm.graph.orrery.v1'); assert.ok(document.nodes.some(node => node.kind === 'operation')); +assert.ok(document.nodes.some(node => node.kind === 'source')); assert.ok(document.nodes.some(node => node.kind === 'field')); -assert.ok(document.edges.some(edge => edge.kind === 'declared_read')); +assert.deepEqual( + [...new Set(document.edges.map(edge => edge.kind))].sort(), + [ + 'artifact_input', + 'compiled_predecessor', + 'declared_read', + 'declared_source', + 'provided_field', + 'structural_input', + ], +); +assert.deepEqual( + [...new Set(document.edges.map(edge => edge.category))].sort(), + ['dependency', 'provenance'], +); +assert.deepEqual(document.metadata.microcosm.extensions, { + float: { float_literal: '1e+100' }, + large: { integer_literal: '1208925819614629174706176' }, + negative: { integer_literal: '-1208925819614629174706176' }, +}); + +const materializedRead = document.edges.find( + edge => edge.kind === 'declared_read' + && edge.data?.read_kind === 'materialized_expand_output', +); +assert.ok(materializedRead); +const materializedField = document.nodes.find(node => node.id === materializedRead.source); +assert.equal(materializedField?.kind, 'field'); +assert.equal(materializedField?.data?.provider, 'expanded'); +assert.equal(materializedField?.data?.declared_in, 'expanded.claim'); +assert.equal(materializedField?.data?.visible_in_schema, false); process.stdout.write('Orrery accepted the Microcosm graph document.\n'); From d97ffa6d4ad03fdf69072805bd1eb26810f41d2c Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 01:00:53 +0400 Subject: [PATCH 10/13] Fix issues from review: validate expansion claims --- .../src/microcosm/graph/decl.py | 30 ++++++++++++++++ .../src/microcosm/graph/population.py | 34 ++++--------------- .../src/microcosm/graph/schema.py | 18 ++++++++++ .../engine_free/shared/test_graph_schema.py | 30 ++++++++++++++++ 4 files changed, 84 insertions(+), 28 deletions(-) diff --git a/packages/microcosm-graph/src/microcosm/graph/decl.py b/packages/microcosm-graph/src/microcosm/graph/decl.py index fc0dfad85..80872273d 100644 --- a/packages/microcosm-graph/src/microcosm/graph/decl.py +++ b/packages/microcosm-graph/src/microcosm/graph/decl.py @@ -664,6 +664,36 @@ def materialized_expand_coordinates( return frozenset(materialized) +def declared_expand_cells(node: Node) -> tuple[tuple[str, str, str], ...]: + """Return the normative ``(entity, column, dtype)`` EXPAND overlays.""" + + raw = node.params.get("expand_cells") + if not isinstance(raw, tuple): + raise GraphError(f"EXPAND node {node.id!r} needs tuple params['expand_cells'].") + cells: list[tuple[str, str, str]] = [] + for item in raw: + if ( + not isinstance(item, tuple) + or len(item) != 3 + or any(not isinstance(part, str) or not part for part in item) + ): + raise GraphError( + f"EXPAND node {node.id!r} has malformed expand_cells entry {item!r}." + ) + entity, column, dtype = item + if "." in entity or "." in column: + raise GraphError( + f"EXPAND node {node.id!r} params['expand_cells'] entity and " + f"column names must be dot-free; got {entity!r}, {column!r}." + ) + if dtype not in DTYPES: + raise GraphError(f"Unknown graph dtype token {dtype!r}.") + cells.append((entity, column, dtype)) + if len({(entity, column) for entity, column, _ in cells}) != len(cells): + raise GraphError(f"EXPAND node {node.id!r} repeats an expanded cell.") + return tuple(cells) + + @dataclass(frozen=True) class Graph: """A complete build declaration for one country. diff --git a/packages/microcosm-graph/src/microcosm/graph/population.py b/packages/microcosm-graph/src/microcosm/graph/population.py index 46b8f0de0..3192b9229 100644 --- a/packages/microcosm-graph/src/microcosm/graph/population.py +++ b/packages/microcosm-graph/src/microcosm/graph/population.py @@ -20,11 +20,13 @@ from .decl import ( MASS_POLICIES, ROWS_ALL, + GraphError, Node, Owned, Ownership, StructuralDelta, WeightUpdate, + declared_expand_cells, ) from .kernel import KernelResult from .store import _encode_object_scalar @@ -1184,34 +1186,10 @@ def patch( def _expand_cells(node: Node) -> tuple[tuple[str, str, str], ...]: """Return the normative ``(entity, column, dtype)`` EXPAND overlays.""" - raw = node.params.get("expand_cells") - if not isinstance(raw, tuple): - raise PopulationError( - f"EXPAND node {node.id!r} needs tuple params['expand_cells']." - ) - cells: list[tuple[str, str, str]] = [] - for item in raw: - if ( - not isinstance(item, tuple) - or len(item) != 3 - or any(not isinstance(part, str) or not part for part in item) - ): - raise PopulationError( - f"EXPAND node {node.id!r} has malformed expand_cells entry {item!r}." - ) - entity, column, dtype = item - if "." in entity or "." in column: - raise PopulationError( - f"EXPAND node {node.id!r} params['expand_cells'] entity and " - f"column names must be dot-free; got {entity!r}, {column!r}." - ) - # Reuse the declaration token validator without importing frozen - # declaration internals into this runtime convention. - dtype_for_token(dtype) - cells.append((entity, column, dtype)) - if len({(entity, column) for entity, column, _ in cells}) != len(cells): - raise PopulationError(f"EXPAND node {node.id!r} repeats an expanded cell.") - return tuple(cells) + try: + return declared_expand_cells(node) + except GraphError as error: + raise PopulationError(str(error)) from error def _expand_weight_entity(node: Node) -> str | None: diff --git a/packages/microcosm-graph/src/microcosm/graph/schema.py b/packages/microcosm-graph/src/microcosm/graph/schema.py index fdab9a6c3..f16ebdfbb 100644 --- a/packages/microcosm-graph/src/microcosm/graph/schema.py +++ b/packages/microcosm-graph/src/microcosm/graph/schema.py @@ -22,6 +22,7 @@ Owned, StructuralDelta, compile_graph, + declared_expand_cells, materialized_expand_coordinates, ) from .serialize import graph_from_json, graph_to_json @@ -316,6 +317,7 @@ def _input_bindings( ROWS_ALL, ) ) + expanded_dtypes: dict[tuple[str, str], str] = {} if "materialized_expand_outputs" in node.params: holder = by_id[population] if holder.structural is not StructuralDelta.EXPAND: @@ -323,6 +325,10 @@ def _input_bindings( f"Node {node.id!r} uses materialized_expand_outputs on " f"population {population!r}, which is not an EXPAND version." ) + expanded_dtypes = { + (entity, column): dtype + for entity, column, dtype in declared_expand_cells(holder) + } for entity, column in sorted(materialized_expand_coordinates(node)): if compiled.owners.get((population, entity, column)) != node.id: raise GraphError( @@ -330,6 +336,18 @@ def _input_bindings( f"output {entity}.{column} in population {population!r}." ) owned = _owned(node, entity, column) + expanded_dtype = expanded_dtypes.get((entity, column)) + if expanded_dtype is None: + raise GraphError( + f"EXPAND node {population!r} does not declare materialized " + f"output {entity}.{column} claimed by node {node.id!r}." + ) + if expanded_dtype != owned.dtype: + raise GraphError( + f"EXPAND node {population!r} declares materialized output " + f"{entity}.{column} as {expanded_dtype!r}, but claimant " + f"{node.id!r} owns it as {owned.dtype!r}." + ) bindings.append( { "node": node.id, diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py index 768c789b4..ecb32289d 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_schema.py @@ -3,9 +3,11 @@ from __future__ import annotations import copy +from dataclasses import replace import pytest +from microcosm.graph import GraphError, compile_graph from microcosm.graph.schema import graph_schema, validate_graph_schema from test_support.microcosm_graph.schema import ( compiled_default_population_graph, @@ -173,6 +175,34 @@ def test_input_bindings_preserve_each_declared_read_role(): } +@pytest.mark.parametrize( + ("expand_cells", "message"), + [ + ((), "does not declare materialized output person.clone"), + ( + (("person", "clone", "float64"),), + "declares materialized output person.clone as 'float64'", + ), + ], +) +def test_materialized_expand_binding_matches_the_expand_declaration( + expand_cells, message +): + graph = compiled_graph().graph + changed = replace( + graph, + nodes=tuple( + replace(node, params={**node.params, "expand_cells": expand_cells}) + if node.id == "expanded" + else node + for node in graph.nodes + ), + ) + + with pytest.raises(GraphError, match=message): + graph_schema(compile_graph(changed)) + + def test_omitted_population_uses_the_compiler_resolved_version(): compiled = compiled_default_population_graph() schema = graph_schema(compiled) From f7e9ba788b0e4f6f8e9edec8ef62e148d4304042 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 01:18:31 +0400 Subject: [PATCH 11/13] Fix issues from review: test UK Orrery export --- .../tests/engine_free/uk/test_uk_graph.py | 50 +++++++++++++++++++ 1 file changed, 50 insertions(+) diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_graph.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_graph.py index 9371ef431..76949264c 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_graph.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_graph.py @@ -1,6 +1,7 @@ """Tests split from packages/microcosm-build/tests/test_uk_graph.py.""" # ruff: noqa: F403, F405 +from microcosm.graph import graph_schema, orrery_document from test_support.microcosm_build.uk_graph import * @@ -226,6 +227,55 @@ def test_uk_graph_json_round_trip_is_canonical() -> None: assert graph_to_json(graph_from_json(serialized)) == serialized +def test_uk_spine_exports_complete_orrery_document() -> None: + compiled = compile_graph(uk_spine_graph()) + schema = graph_schema(compiled) + materialized_claims = [ + binding + for binding in schema["input_bindings"] + if binding["kind"] == "materialized_expand_output" + ] + + document = orrery_document(compiled, title="UK spine") + nodes = {node["id"]: node for node in document["nodes"]} + materialized_reads = [ + edge + for edge in document["edges"] + if edge["kind"] == "declared_read" + and edge["data"]["read_kind"] == "materialized_expand_output" + ] + expected_reads = { + ( + binding["node"], + binding["population"], + binding["entity"], + binding["column"], + binding["provider"], + binding["declared_in"], + binding["rows"], + ) + for binding in materialized_claims + } + actual_reads = { + ( + nodes[edge["target"]]["label"], + nodes[edge["source"]]["data"]["population"], + nodes[edge["source"]]["data"]["entity"], + nodes[edge["source"]]["data"]["column"], + nodes[edge["source"]]["data"]["provider"], + nodes[edge["source"]]["data"]["declared_in"], + edge["data"]["rows"], + ) + for edge in materialized_reads + } + + assert materialized_claims + assert actual_reads == expected_reads + assert document["schemaVersion"] == "graph-explorer/v1" + assert document["metadata"]["microcosm"] == schema + assert document["metadata"]["truncated"] is False + + def test_conserve_rejects_a_mass_shift_across_household_sizes() -> None: # The executor's ledger is person mass: an expansion that conserves the # weight entity's mass but shifts it between households of different size From e6279af6da31c57de0dd77bb92212ff48413e712 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 17:04:33 +0400 Subject: [PATCH 12/13] Keep expansion validation outside frozen declarations Share EXPAND parameter validation through a dedicated graph module so execution, population handling, and static schema export agree without changing the frozen declaration interface. --- .../src/microcosm/graph/decl.py | 69 ---------------- .../src/microcosm/graph/executor.py | 2 +- .../src/microcosm/graph/expansion.py | 82 +++++++++++++++++++ .../src/microcosm/graph/population.py | 2 +- .../src/microcosm/graph/schema.py | 3 +- 5 files changed, 85 insertions(+), 73 deletions(-) create mode 100644 packages/microcosm-graph/src/microcosm/graph/expansion.py diff --git a/packages/microcosm-graph/src/microcosm/graph/decl.py b/packages/microcosm-graph/src/microcosm/graph/decl.py index 80872273d..24c92ab7a 100644 --- a/packages/microcosm-graph/src/microcosm/graph/decl.py +++ b/packages/microcosm-graph/src/microcosm/graph/decl.py @@ -625,75 +625,6 @@ def normative(self) -> dict[str, object]: } -def materialized_expand_coordinates( - node: Node, -) -> frozenset[tuple[str, str]]: - """Return the carried EXPAND cells an ordinary node claims as outputs. - - ``materialized_expand_outputs`` is a reserved node parameter used when an - EXPAND kernel physically installs a new column before a following ordinary - node gives that column an ownership declaration. The claim node therefore - receives the existing values as inputs even though the coordinate is not a - ``Slice`` or rewrite. - """ - - raw_materialized = node.params.get("materialized_expand_outputs", ()) - if not isinstance(raw_materialized, tuple) or any( - not isinstance(value, str) or "." not in value for value in raw_materialized - ): - raise GraphError( - f"Node {node.id!r} params['materialized_expand_outputs'] must be a " - "tuple of 'entity.column' strings." - ) - materialized: set[tuple[str, str]] = set() - owned_by_coordinate = { - (output.entity, output.column): output for output in node.outputs - } - for value in raw_materialized: - entity, column = value.split(".", 1) - coordinate = (entity, column) - output = owned_by_coordinate.get(coordinate) - if output is None or output.rewrite: - raise GraphError( - f"Node {node.id!r} materialized EXPAND output {value!r} must be " - "one of its non-rewrite owned cells." - ) - materialized.add(coordinate) - if len(materialized) != len(raw_materialized): - raise GraphError(f"Node {node.id!r} repeats a materialized EXPAND output.") - return frozenset(materialized) - - -def declared_expand_cells(node: Node) -> tuple[tuple[str, str, str], ...]: - """Return the normative ``(entity, column, dtype)`` EXPAND overlays.""" - - raw = node.params.get("expand_cells") - if not isinstance(raw, tuple): - raise GraphError(f"EXPAND node {node.id!r} needs tuple params['expand_cells'].") - cells: list[tuple[str, str, str]] = [] - for item in raw: - if ( - not isinstance(item, tuple) - or len(item) != 3 - or any(not isinstance(part, str) or not part for part in item) - ): - raise GraphError( - f"EXPAND node {node.id!r} has malformed expand_cells entry {item!r}." - ) - entity, column, dtype = item - if "." in entity or "." in column: - raise GraphError( - f"EXPAND node {node.id!r} params['expand_cells'] entity and " - f"column names must be dot-free; got {entity!r}, {column!r}." - ) - if dtype not in DTYPES: - raise GraphError(f"Unknown graph dtype token {dtype!r}.") - cells.append((entity, column, dtype)) - if len({(entity, column) for entity, column, _ in cells}) != len(cells): - raise GraphError(f"EXPAND node {node.id!r} repeats an expanded cell.") - return tuple(cells) - - @dataclass(frozen=True) class Graph: """A complete build declaration for one country. diff --git a/packages/microcosm-graph/src/microcosm/graph/executor.py b/packages/microcosm-graph/src/microcosm/graph/executor.py index 1bcebd04b..4a94fc1ef 100644 --- a/packages/microcosm-graph/src/microcosm/graph/executor.py +++ b/packages/microcosm-graph/src/microcosm/graph/executor.py @@ -40,9 +40,9 @@ Owned, Ownership, StructuralDelta, - materialized_expand_coordinates, ) from .errors import NodeRejectedError +from .expansion import materialized_expand_coordinates from .kernel import ( ArtifactValue, Capabilities, diff --git a/packages/microcosm-graph/src/microcosm/graph/expansion.py b/packages/microcosm-graph/src/microcosm/graph/expansion.py new file mode 100644 index 000000000..cb613916d --- /dev/null +++ b/packages/microcosm-graph/src/microcosm/graph/expansion.py @@ -0,0 +1,82 @@ +"""Shared validation for the graph's EXPAND runtime conventions. + +EXPAND kernels encode structural cell overlays and following ownership claims +in normative ``Node.params`` entries. Those conventions remain outside the +frozen declaration interface while this module gives population handling, +execution, and static presentation schemas one interpretation of them. +""" + +from __future__ import annotations + +from .decl import DTYPES, GraphError, Node + +__all__ = ["declared_expand_cells", "materialized_expand_coordinates"] + + +def materialized_expand_coordinates( + node: Node, +) -> frozenset[tuple[str, str]]: + """Return the carried EXPAND cells an ordinary node claims as outputs. + + ``materialized_expand_outputs`` is a reserved node parameter used when an + EXPAND kernel physically installs a new column before a following ordinary + node gives that column an ownership declaration. The claim node therefore + receives the existing values as inputs even though the coordinate is not a + ``Slice`` or rewrite. + """ + + raw_materialized = node.params.get("materialized_expand_outputs", ()) + if not isinstance(raw_materialized, tuple) or any( + not isinstance(value, str) or "." not in value for value in raw_materialized + ): + raise GraphError( + f"Node {node.id!r} params['materialized_expand_outputs'] must be a " + "tuple of 'entity.column' strings." + ) + materialized: set[tuple[str, str]] = set() + owned_by_coordinate = { + (output.entity, output.column): output for output in node.outputs + } + for value in raw_materialized: + entity, column = value.split(".", 1) + coordinate = (entity, column) + output = owned_by_coordinate.get(coordinate) + if output is None or output.rewrite: + raise GraphError( + f"Node {node.id!r} materialized EXPAND output {value!r} must be " + "one of its non-rewrite owned cells." + ) + materialized.add(coordinate) + if len(materialized) != len(raw_materialized): + raise GraphError(f"Node {node.id!r} repeats a materialized EXPAND output.") + return frozenset(materialized) + + +def declared_expand_cells(node: Node) -> tuple[tuple[str, str, str], ...]: + """Return the normative ``(entity, column, dtype)`` EXPAND overlays.""" + + raw = node.params.get("expand_cells") + if not isinstance(raw, tuple): + raise GraphError(f"EXPAND node {node.id!r} needs tuple params['expand_cells'].") + cells: list[tuple[str, str, str]] = [] + for item in raw: + if ( + not isinstance(item, tuple) + or len(item) != 3 + or any(not isinstance(part, str) or not part for part in item) + ): + raise GraphError( + f"EXPAND node {node.id!r} has malformed expand_cells entry {item!r}." + ) + entity, column, dtype = item + if "." in entity or "." in column: + raise GraphError( + f"EXPAND node {node.id!r} params['expand_cells'] entity and " + f"column names must be dot-free; got {entity!r}, {column!r}." + ) + if dtype not in DTYPES: + raise GraphError(f"Unknown graph dtype token {dtype!r}.") + cells.append((entity, column, dtype)) + if len({(entity, column) for entity, column, _ in cells}) != len(cells): + raise GraphError(f"EXPAND node {node.id!r} repeats an expanded cell.") + return tuple(cells) diff --git a/packages/microcosm-graph/src/microcosm/graph/population.py b/packages/microcosm-graph/src/microcosm/graph/population.py index 3192b9229..260bb2fff 100644 --- a/packages/microcosm-graph/src/microcosm/graph/population.py +++ b/packages/microcosm-graph/src/microcosm/graph/population.py @@ -26,8 +26,8 @@ Ownership, StructuralDelta, WeightUpdate, - declared_expand_cells, ) +from .expansion import declared_expand_cells from .kernel import KernelResult from .store import _encode_object_scalar from .weight_update import weight_update_receipt diff --git a/packages/microcosm-graph/src/microcosm/graph/schema.py b/packages/microcosm-graph/src/microcosm/graph/schema.py index f16ebdfbb..9f9d25cfb 100644 --- a/packages/microcosm-graph/src/microcosm/graph/schema.py +++ b/packages/microcosm-graph/src/microcosm/graph/schema.py @@ -22,9 +22,8 @@ Owned, StructuralDelta, compile_graph, - declared_expand_cells, - materialized_expand_coordinates, ) +from .expansion import declared_expand_cells, materialized_expand_coordinates from .serialize import graph_from_json, graph_to_json __all__ = ["graph_schema", "validate_graph_schema"] From 40aff441dca73243c4f871cb51bb283421f75f31 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 18:36:30 +0400 Subject: [PATCH 13/13] Fix issues from review: validate descriptions and CI jobs Recognize underscore-prefixed GitHub Actions job IDs and reject whitespace-only graph descriptions at declaration construction. Record and re-lock the intentional declaration-interface amendment. --- .../graph-description-validation.fixed.md | 1 + docs/graph-acceptance.md | 10 ++++++++++ docs/graph-interface.lock | 2 +- .../engine_free/shared/test_ci_test_plan.py | 18 ++++++++++++++++++ .../src/microcosm/graph/decl.py | 11 +++++++++++ .../engine_free/shared/test_graph_decl.py | 17 +++++++++++++++++ tools/ci_test_plan.py | 2 +- 7 files changed, 59 insertions(+), 2 deletions(-) create mode 100644 changelog.d/graph-description-validation.fixed.md diff --git a/changelog.d/graph-description-validation.fixed.md b/changelog.d/graph-description-validation.fixed.md new file mode 100644 index 000000000..4908cf607 --- /dev/null +++ b/changelog.d/graph-description-validation.fixed.md @@ -0,0 +1 @@ +Reject whitespace-only graph node and source descriptions at declaration construction while retaining the empty no-description value. diff --git a/docs/graph-acceptance.md b/docs/graph-acceptance.md index 7d22f73e3..cea6ea1d0 100644 --- a/docs/graph-acceptance.md +++ b/docs/graph-acceptance.md @@ -675,6 +675,16 @@ lock unchanged: recorded the observer opt-in as amendment 25 in the meantime; amendment 26 above was 25 in the same lane. +28. **Descriptions are empty or contain non-whitespace text.** + `SourceRef.description` and `Node.description` keep the empty string as the + explicit no-description sentinel and reject every nonempty value whose + characters are all whitespace. This makes the declaration boundary enforce + the text contract shared by presentation adapters instead of allowing a + graph to compile and later produce a document that its consumer rejects. + Descriptions remain descriptive: they stay outside `Node.normative()`, so + this validation changes no graph computation key. `decl.py` is re-locked. + Adopted during the Orrery export review in #888. + Adding a normative field with a default changes the canonical projection of every node that carries it, so node keys moved with amendments 11 and diff --git a/docs/graph-interface.lock b/docs/graph-interface.lock index 6ba6ed147..7e6a88441 100644 --- a/docs/graph-interface.lock +++ b/docs/graph-interface.lock @@ -1,2 +1,2 @@ -b25ae4a62fd2777a5dffa30ecbba7d345a804b74b9d6b6e68c2a5d38f3e0a129 decl.py +6f9271abf7a2f5157b0423e0e79007b5f20c5bb5c431262baff7a104eb91d502 decl.py 24045ab2a62295874520d7940c7f7b59be9d909a3cfbf7e111a734e369b0815b kernel.py diff --git a/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py b/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py index 11bb65e84..5300e9024 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_ci_test_plan.py @@ -152,6 +152,24 @@ def test_workflow_rejects_a_separate_behavioral_test_job() -> None: ) +def test_workflow_rejects_an_underscore_prefixed_job() -> None: + source = """\ +jobs: + engine-free: + steps: + - run: pytest + _invented-compatibility-test: + steps: + - run: node verify.mjs +""" + + assert ( + "_invented-compatibility-test: workflow job has no registered test " + "category or approved infrastructure role" + in ci_test_plan.workflow_job_errors(source) + ) + + def test_workflow_job_parser_ignores_nested_yaml_keys() -> None: source = """\ jobs: diff --git a/packages/microcosm-graph/src/microcosm/graph/decl.py b/packages/microcosm-graph/src/microcosm/graph/decl.py index 24c92ab7a..5d2bf5cfe 100644 --- a/packages/microcosm-graph/src/microcosm/graph/decl.py +++ b/packages/microcosm-graph/src/microcosm/graph/decl.py @@ -154,6 +154,15 @@ def _nonempty(label: str, value: str) -> None: raise GraphError(f"{label} must be a non-empty string, got {value!r}.") +def _description(label: str, value: str) -> None: + """Require an empty sentinel or text containing a non-whitespace character.""" + + if not isinstance(value, str) or (value and not value.strip()): + raise GraphError( + f"{label} must be empty or contain non-whitespace text, got {value!r}." + ) + + def _name(label: str, value: str) -> None: """An entity or column name: non-empty and free of ``.``. @@ -182,6 +191,7 @@ class SourceRef: def __post_init__(self) -> None: _nonempty("SourceRef.name", self.name) _nonempty("SourceRef.codec", self.codec) + _description("SourceRef.description", self.description) @dataclass(frozen=True) @@ -495,6 +505,7 @@ class Node: def __post_init__(self) -> None: _nonempty("Node.id", self.id) _nonempty("Node.kernel", self.kernel) + _description("Node.description", self.description) for name, kind in ( ("artifact_inputs", ArtifactInput), ("artifact_outputs", ArtifactOutput), diff --git a/packages/microcosm-graph/tests/engine_free/shared/test_graph_decl.py b/packages/microcosm-graph/tests/engine_free/shared/test_graph_decl.py index aefff1e95..395c88d54 100644 --- a/packages/microcosm-graph/tests/engine_free/shared/test_graph_decl.py +++ b/packages/microcosm-graph/tests/engine_free/shared/test_graph_decl.py @@ -155,6 +155,23 @@ def test_descriptive_fields_are_outside_the_normative_projection() -> None: assert "description" not in node.normative() +@pytest.mark.parametrize( + ("constructor", "label"), + [ + ( + lambda: SourceRef("survey", "frame-h5", description=" \t"), + "SourceRef.description", + ), + (lambda: Node("node", "k@1", description="\n"), "Node.description"), + ], +) +def test_descriptions_must_be_empty_or_contain_non_whitespace_text( + constructor, label +) -> None: + with pytest.raises(GraphError, match=rf"{label}.*non-whitespace"): + constructor() + + def test_graph_needs_a_create_node() -> None: with pytest.raises(GraphError, match="CREATE"): compile_graph(Graph("toy", (SRC,), (_fit("fit_a", ("age",), "a"),))) diff --git a/tools/ci_test_plan.py b/tools/ci_test_plan.py index 550846c10..4f98ed9fb 100644 --- a/tools/ci_test_plan.py +++ b/tools/ci_test_plan.py @@ -87,7 +87,7 @@ class TestGroup: RequestJSON = Callable[[str, str], Any] -_WORKFLOW_JOB = re.compile(r"^ ([A-Za-z0-9][A-Za-z0-9_-]*):(?:\s.*)?$") +_WORKFLOW_JOB = re.compile(r"^ ([A-Za-z_][A-Za-z0-9_-]*):(?:\s.*)?$") def workflow_job_names(source: str) -> tuple[str, ...]: