getEdgeIds() {
+ return Collections.unmodifiableList(edgeIds == null ? Collections.emptyList() : edgeIds);
+ }
+
+ public int getHop() {
+ return hop;
+ }
+
+ public boolean isSampled() {
+ return sampled;
+ }
+
+ public boolean sameIdentityAs(GraphPathRef other) {
+ return equals(other);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphPathRef)) {
+ return false;
+ }
+ GraphPathRef that = (GraphPathRef) object;
+ return hop == that.hop && sampled == that.sampled
+ && Objects.equals(vertexIds, that.vertexIds)
+ && Objects.equals(edgeIds, that.edgeIds);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(vertexIds, edgeIds, hop, sampled);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java
new file mode 100644
index 000000000..5f5649c30
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.graph;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable graph vertex reference associated with an entity identifier. */
+public final class GraphVertexRef {
+
+ @SerializedName("label")
+ private final String label;
+ @SerializedName("vertexId")
+ private final String vertexId;
+ @SerializedName("entityId")
+ private final String entityId;
+
+ public GraphVertexRef(String label, String vertexId, String entityId) {
+ this.label = ModelValidation.required(label, "label");
+ this.vertexId = ModelValidation.required(vertexId, "vertexId");
+ this.entityId = ModelValidation.required(entityId, "entityId");
+ }
+
+ public String getLabel() {
+ return label;
+ }
+
+ public String getVertexId() {
+ return vertexId;
+ }
+
+ public String getEntityId() {
+ return entityId;
+ }
+
+ public boolean sameIdentityAs(GraphVertexRef other) {
+ return other != null && Objects.equals(label, other.label)
+ && Objects.equals(vertexId, other.vertexId)
+ && Objects.equals(entityId, other.entityId);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphVertexRef)) {
+ return false;
+ }
+ GraphVertexRef that = (GraphVertexRef) object;
+ return Objects.equals(label, that.label)
+ && Objects.equals(vertexId, that.vertexId)
+ && Objects.equals(entityId, that.entityId);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(label, vertexId, entityId);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
new file mode 100644
index 000000000..463789720
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
@@ -0,0 +1,48 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+/**
+ * Immutable value objects shared by GraphRAG ingestion, retrieval, and evaluation.
+ *
+ * IDs in these objects are opaque values supplied by callers. This package does not generate
+ * IDs. Use {@code sameIdentityAs} for identity comparisons and {@code equals} when complete value
+ * equality is required. Evidence identity uses nested references when present, and falls back to
+ * evidenceId or text when no references are available; scores and ranks are ignored.
+ *
+ * JSON uses explicit camelCase field names. Optional fields may be added only additively; unknown
+ * fields are ignored, null optional scalars are omitted, and collection fields use empty arrays or
+ * objects. Callers must use {@link RetrievalModelJson} for deserialization so required-field and
+ * range validation cannot be bypassed by reflection. Text offsets are half-open
+ * ({@code startOffset <= endOffset}); ranks are one-based, raw scores are finite, and normalized
+ * and fused scores are finite values in {@code [0, 1]}.
+ *
+ * Identity fields are documentId/dataset/datasetVersion/split/sourceUri/sourceHash for
+ * {@code SourceDocument}; chunkId/documentId/chunkIndex/startOffset/endOffset/text/policyVersion/
+ * textHash for {@code TextChunk}; entityId/canonicalName/type for {@code EntityRef}; all fields
+ * for {@code GraphVertexRef}, {@code GraphVersion}, {@code IndexVersion}, {@code SourceRef},
+ * {@code GraphPathRef}, and {@code ChannelScore}; and edgeId/label/sourceEntityId/targetEntityId
+ * for {@code GraphEdgeRef}. Evidence compares kind and nested reference identities. Collection
+ * order does not affect Evidence identity, while vertex and edge order inside one graph path does.
+ * A graph path has exactly {@code hop + 1} vertices and {@code hop} edges. Evidence kind labels
+ * the candidate payload (chunk, entity, path, or graph fact), while optional lists preserve the
+ * supporting provenance and may be empty for compatibility.
+ */
+package org.apache.geaflow.ai.retrieval.model;
+
+import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson;
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java
new file mode 100644
index 000000000..26f6003ff
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java
@@ -0,0 +1,69 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.version;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable name/version pair identifying a graph snapshot. */
+public final class GraphVersion {
+
+ @SerializedName("graphName")
+ private final String graphName;
+ @SerializedName("version")
+ private final String version;
+
+ public GraphVersion(String graphName, String version) {
+ this.graphName = ModelValidation.required(graphName, "graphName");
+ this.version = ModelValidation.required(version, "version");
+ }
+
+ public String getGraphName() {
+ return graphName;
+ }
+
+ public String getVersion() {
+ return version;
+ }
+
+ public boolean sameIdentityAs(GraphVersion other) {
+ return other != null && Objects.equals(graphName, other.graphName)
+ && Objects.equals(version, other.version);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphVersion)) {
+ return false;
+ }
+ GraphVersion that = (GraphVersion) object;
+ return Objects.equals(graphName, that.graphName)
+ && Objects.equals(version, that.version);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(graphName, version);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java
new file mode 100644
index 000000000..1e1c791b7
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.version;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable index snapshot identifier together with its source graph version. */
+public final class IndexVersion {
+
+ @SerializedName("indexName")
+ private final String indexName;
+ @SerializedName("version")
+ private final String version;
+ @SerializedName("graphVersion")
+ private final String graphVersion;
+
+ public IndexVersion(String indexName, String version, String graphVersion) {
+ this.indexName = ModelValidation.required(indexName, "indexName");
+ this.version = ModelValidation.required(version, "version");
+ this.graphVersion = ModelValidation.required(graphVersion, "graphVersion");
+ }
+
+ public String getIndexName() {
+ return indexName;
+ }
+
+ public String getVersion() {
+ return version;
+ }
+
+ public String getGraphVersion() {
+ return graphVersion;
+ }
+
+ public boolean sameIdentityAs(IndexVersion other) {
+ return other != null && Objects.equals(indexName, other.indexName)
+ && Objects.equals(version, other.version)
+ && Objects.equals(graphVersion, other.graphVersion);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof IndexVersion)) {
+ return false;
+ }
+ IndexVersion that = (IndexVersion) object;
+ return Objects.equals(indexName, that.indexName)
+ && Objects.equals(version, that.version)
+ && Objects.equals(graphVersion, that.graphVersion);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(indexName, version, graphVersion);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java
new file mode 100644
index 000000000..b4176be4a
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java
@@ -0,0 +1,126 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.validation;
+
+import java.util.ArrayList;
+import java.util.Collections;
+import java.util.Comparator;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Objects;
+
+/** Shared constructor validation and defensive-copy helpers for retrieval models. */
+public final class ModelValidation {
+
+ private ModelValidation() {
+ }
+
+ public static String required(String value, String name) {
+ Objects.requireNonNull(value, name);
+ if (value.trim().isEmpty()) {
+ throw new RetrievalModelValidationException(name + " must not be blank");
+ }
+ return value;
+ }
+
+ public static String optional(String value) {
+ return value == null ? null : value;
+ }
+
+ public static String optionalNonBlank(String value, String name) {
+ if (value == null) {
+ return null;
+ }
+ return required(value, name);
+ }
+
+ public static int nonNegative(int value, String name) {
+ if (value < 0) {
+ throw new RetrievalModelValidationException(name + " must be non-negative");
+ }
+ return value;
+ }
+
+ public static double finite(double value, String name) {
+ if (Double.isNaN(value) || Double.isInfinite(value)) {
+ throw new RetrievalModelValidationException(name + " must be finite");
+ }
+ return value;
+ }
+
+ public static Double optionalScore(Double value, String name) {
+ if (value == null) {
+ return null;
+ }
+ finite(value, name);
+ if (value < 0.0 || value > 1.0) {
+ throw new RetrievalModelValidationException(name + " must be in [0, 1]");
+ }
+ return value;
+ }
+
+ public static Integer optionalRank(Integer value, String name) {
+ if (value != null && value < 1) {
+ throw new RetrievalModelValidationException(name + " must be at least 1");
+ }
+ return value;
+ }
+
+ public static List immutableList(List values, String name) {
+ if (values == null || values.isEmpty()) {
+ return Collections.emptyList();
+ }
+ List copy = new ArrayList<>(values);
+ for (T value : copy) {
+ Objects.requireNonNull(value, name + " element");
+ }
+ return Collections.unmodifiableList(copy);
+ }
+
+ public static List sortedStrings(List values, String name) {
+ List result = immutableList(values, name);
+ if (result.isEmpty()) {
+ return result;
+ }
+ List sorted = new ArrayList<>(result.size());
+ for (String value : result) {
+ sorted.add(required(value, name));
+ }
+ sorted.sort(Comparator.naturalOrder());
+ return Collections.unmodifiableList(sorted);
+ }
+
+ public static Map sortedMap(Map values) {
+ if (values == null || values.isEmpty()) {
+ return Collections.emptyMap();
+ }
+ List keys = new ArrayList<>(values.keySet());
+ for (String key : keys) {
+ required(key, "map key");
+ }
+ keys.sort(Comparator.naturalOrder());
+ Map sorted = new LinkedHashMap<>();
+ for (String key : keys) {
+ sorted.put(key, values.get(key));
+ }
+ return Collections.unmodifiableMap(sorted);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java
new file mode 100644
index 000000000..e09c381d4
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java
@@ -0,0 +1,32 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.validation;
+
+/** Indicates invalid data at the retrieval model boundary. */
+public class RetrievalModelValidationException extends IllegalArgumentException {
+
+ public RetrievalModelValidationException(String message) {
+ super(message);
+ }
+
+ public RetrievalModelValidationException(String message, Throwable cause) {
+ super(message, cause);
+ }
+}
diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
new file mode 100644
index 000000000..494aca92c
--- /dev/null
+++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
@@ -0,0 +1,392 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model;
+
+import com.google.gson.Gson;
+import com.google.gson.JsonParseException;
+import java.io.BufferedReader;
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.InputStreamReader;
+import java.util.Arrays;
+import java.util.Collections;
+import java.util.HashMap;
+import java.util.Map;
+import java.util.stream.Collectors;
+import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson;
+import org.apache.geaflow.ai.retrieval.model.document.SourceDocument;
+import org.apache.geaflow.ai.retrieval.model.document.SourceRef;
+import org.apache.geaflow.ai.retrieval.model.document.TextChunk;
+import org.apache.geaflow.ai.retrieval.model.evidence.ChannelScore;
+import org.apache.geaflow.ai.retrieval.model.evidence.Evidence;
+import org.apache.geaflow.ai.retrieval.model.evidence.EvidenceKind;
+import org.apache.geaflow.ai.retrieval.model.graph.EntityRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphEdgeRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphVertexRef;
+import org.apache.geaflow.ai.retrieval.model.version.GraphVersion;
+import org.apache.geaflow.ai.retrieval.model.version.IndexVersion;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/**
+ * Regression coverage for retrieval domain models, validation rules, and JSON round-tripping.
+ */
+public class RetrievalDomainModelTest {
+
+ private static final Gson GSON = new Gson();
+
+ @Test
+ public void modelsAreValueObjectsAndDefensivelyCopyCollections() {
+ java.util.List aliases = Arrays.asList("Kong Fuzi", "Master Kong");
+ EntityRef entity = new EntityRef("entity-1", "Confucius", aliases,
+ "PERSON", Collections.singletonList("chunk-1"));
+
+ Assertions.assertEquals(Arrays.asList("Kong Fuzi", "Master Kong"), entity.getAliases());
+ Assertions.assertThrows(UnsupportedOperationException.class,
+ () -> entity.getAliases().add("孔子"));
+ Assertions.assertEquals(entity, new EntityRef("entity-1", "Confucius",
+ Arrays.asList("Master Kong", "Kong Fuzi"), "PERSON",
+ Collections.singletonList("chunk-1")));
+ Assertions.assertTrue(entity.sameIdentityAs(new EntityRef("entity-1", "Confucius",
+ Collections.emptyList(), "PERSON", Collections.emptyList())));
+
+ SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev",
+ "Title", "file:///doc", "hash");
+ Assertions.assertTrue(document.sameIdentityAs(new SourceDocument("doc-1", "dataset",
+ "v1", "dev", "Other title", "file:///doc", "hash")));
+
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, "text",
+ "policy-v1", "hash");
+ Assertions.assertTrue(chunk.sameIdentityAs(new TextChunk("chunk-1", "doc-1", 0,
+ 0, 4, 99, "text", "policy-v1", "hash")));
+
+ GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1");
+ Assertions.assertFalse(vertex.sameIdentityAs(new GraphVertexRef("company", "vertex-1",
+ "entity-1")));
+ GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2",
+ Collections.singletonList("chunk-1"));
+ Assertions.assertTrue(edge.sameIdentityAs(new GraphEdgeRef("edge-1", "knows",
+ "entity-1", "entity-2", Collections.singletonList("chunk-2"))));
+
+ Assertions.assertTrue(new GraphVersion("graph", "v1")
+ .sameIdentityAs(new GraphVersion("graph", "v1")));
+ Assertions.assertFalse(new IndexVersion("bm25", "v1", "graph-v1")
+ .sameIdentityAs(new IndexVersion("vector", "v1", "graph-v1")));
+ }
+
+ @Test
+ public void validatesRequiredFieldsAndRanges() {
+ Assertions.assertThrows(NullPointerException.class,
+ () -> new GraphVersion(null, "v1"));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new TextChunk("chunk-1", "doc-1", 0, 8, 2, 1, "text", null, null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new ChannelScore("bm25", 1.0, 0.5, 0));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new SourceRef("doc-1", "file:///doc", 4, 2));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new Evidence(" ", EvidenceKind.CHUNK, null, null, null, null, null, null,
+ null, null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new TextChunk("chunk-1", "doc-1", 0, 0, 1, 1, "text", " ", null));
+ }
+
+ @Test
+ public void evidenceMergesChannelsByFieldsNotScores() {
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ GraphPathRef path = new GraphPathRef(Collections.singletonList("v1"),
+ Collections.emptyList(), 0, false);
+ Map bm25 = new HashMap<>();
+ bm25.put("bm25", new ChannelScore("bm25", 2.0, 0.8, 1));
+ Map vector = new HashMap<>();
+ vector.put("vector", new ChannelScore("vector", 0.9, 0.7, 3));
+
+ Evidence first = new Evidence("evidence-1", EvidenceKind.CHUNK, "text",
+ Collections.singletonList(chunk), Collections.emptyList(),
+ Collections.singletonList(path), Collections.singletonList(source), bm25, 0.8, 1);
+ Evidence second = new Evidence("evidence-2", EvidenceKind.CHUNK, "text",
+ Collections.singletonList(chunk), Collections.emptyList(),
+ Collections.singletonList(path), Collections.singletonList(source), vector, 0.7, 3);
+
+ Assertions.assertNotEquals(first, second);
+ Assertions.assertTrue(first.sameIdentityAs(second));
+ Assertions.assertEquals(0.8, first.getStageScores().get("bm25").getNormalizedScore());
+ Assertions.assertEquals(0.7, second.getStageScores().get("vector").getNormalizedScore());
+ String json = GSON.toJson(first);
+ Assertions.assertTrue(json.contains("\"chunks\""));
+ Assertions.assertTrue(json.contains("\"stageScores\""));
+ Assertions.assertEquals(first, RetrievalModelJson.fromJson(json, Evidence.class));
+ }
+
+ @Test
+ public void evidenceIdentityUsesNestedIdentityAndIgnoresCollectionOrder() {
+ TextChunk firstChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ TextChunk secondChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 99,
+ "text", "policy-v1", "hash");
+ EntityRef firstEntity = new EntityRef("entity-1", "Name",
+ Collections.singletonList("alias-a"), "PERSON", Collections.singletonList("chunk-1"));
+ EntityRef secondEntity = new EntityRef("entity-1", "Name",
+ Collections.singletonList("alias-b"), "PERSON", Collections.singletonList("chunk-2"));
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ SourceRef secondSource = new SourceRef("doc-2", "file:///doc-2", 0, 4);
+ GraphPathRef path = new GraphPathRef(Arrays.asList("v1", "v2"),
+ Collections.singletonList("e1"), 1, false);
+ GraphPathRef secondPath = new GraphPathRef(Arrays.asList("v2", "v3"),
+ Collections.singletonList("e2"), 1, false);
+
+ Evidence first = new Evidence("e-1", EvidenceKind.ENTITY, "first",
+ Collections.singletonList(firstChunk), Collections.singletonList(firstEntity),
+ Arrays.asList(path, secondPath), Arrays.asList(source, secondSource),
+ Collections.emptyMap(), 0.1, 1);
+ Evidence second = new Evidence("e-2", EvidenceKind.ENTITY, "second",
+ Collections.singletonList(secondChunk), Collections.singletonList(secondEntity),
+ Arrays.asList(secondPath, path), Arrays.asList(secondSource, source),
+ Collections.emptyMap(), 0.9, 2);
+
+ Assertions.assertTrue(first.sameIdentityAs(second));
+ Evidence reordered = new Evidence("e-3", EvidenceKind.ENTITY, null,
+ Collections.singletonList(secondChunk), Collections.singletonList(secondEntity),
+ Arrays.asList(secondPath, path), Arrays.asList(secondSource, source),
+ Collections.emptyMap(), null, null);
+ Assertions.assertTrue(first.sameIdentityAs(reordered));
+ }
+
+ @Test
+ public void gsonRoundTripPreservesFieldsAndOptionalCompatibility() throws IOException {
+ SourceDocument document = new SourceDocument("doc-1", "hotpotqa", "v1",
+ "dev", "Title", "https://example/doc-1", "sha256");
+ String json = GSON.toJson(document);
+ SourceDocument restored = RetrievalModelJson.fromJson(json, SourceDocument.class);
+ Assertions.assertEquals(document, restored);
+ Assertions.assertEquals(readFixture("retrieval/model/source-document.json"), json);
+
+ String oldJson = "{\"documentId\":\"doc-1\",\"dataset\":\"hotpotqa\","
+ + "\"datasetVersion\":\"v1\",\"split\":\"dev\","
+ + "\"sourceUri\":\"https://example/doc-1\",\"sourceHash\":\"sha256\"}";
+ SourceDocument withoutTitle = RetrievalModelJson.fromJson(oldJson, SourceDocument.class);
+ Assertions.assertNull(withoutTitle.getTitle());
+ Assertions.assertEquals(document.getDocumentId(), withoutTitle.getDocumentId());
+
+ String oldEvidenceJson = "{\"kind\":\"CHUNK\",\"evidenceId\":\"e-1\","
+ + "\"text\":\"text\"}";
+ Evidence oldEvidence = RetrievalModelJson.fromJson(oldEvidenceJson, Evidence.class);
+ Assertions.assertNotNull(oldEvidence.getChunks());
+ Assertions.assertNotNull(oldEvidence.getEntities());
+ Assertions.assertNotNull(oldEvidence.getPaths());
+ Assertions.assertNotNull(oldEvidence.getSources());
+ Assertions.assertNotNull(oldEvidence.getStageScores());
+ Assertions.assertTrue(oldEvidence.getChunks().isEmpty());
+ Assertions.assertTrue(GSON.toJson(oldEvidence).contains("\"kind\":\"CHUNK\""));
+ }
+
+ @Test
+ public void validatedJsonRejectsMissingRequiredFieldsAndKeepsEmptyCollections() {
+ Assertions.assertThrows(JsonParseException.class,
+ () -> RetrievalModelJson.fromJson("{\"version\":\"v1\"}", GraphVersion.class));
+ Assertions.assertThrows(JsonParseException.class,
+ () -> RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\",\"evidenceId\":\" \"}",
+ Evidence.class));
+
+ Evidence empty = RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\","
+ + "\"chunks\":null,\"entities\":[],\"paths\":null,\"sources\":null,"
+ + "\"stageScores\":null}", Evidence.class);
+ String json = RetrievalModelJson.toJson(empty);
+ Assertions.assertTrue(json.contains("\"chunks\":[]"));
+ Assertions.assertTrue(json.contains("\"entities\":[]"));
+ Assertions.assertTrue(json.contains("\"paths\":[]"));
+ Assertions.assertTrue(json.contains("\"sources\":[]"));
+ Assertions.assertTrue(json.contains("\"stageScores\":{}"));
+ }
+
+ @Test
+ public void validatedJsonRejectsWrongElementAndNumericTypes() {
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[1],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[true],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[null],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"chunkId\":\"c\",\"documentId\":\"d\",\"chunkIndex\":1.9,"
+ + "\"startOffset\":0,\"endOffset\":1,\"tokenEstimate\":1,\"text\":\"x\"}",
+ TextChunk.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"channel\":\"bm25\",\"rawScore\":1,\"rank\":2.5}",
+ ChannelScore.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"vertexIds\":[\"v1\",\"v2\"],\"edgeIds\":[\"e1\"],"
+ + "\"hop\":1.5,\"sampled\":false}", GraphPathRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"kind\":\"CHUNK\",\"fusedScore\":1.1}", Evidence.class));
+ }
+
+ @Test
+ public void modelValidationUsesDedicatedExceptionAndEnforcesRanges() {
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null,
+ null, -0.01, null));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null,
+ null, 1.01, null));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new GraphPathRef(Collections.singletonList("v1"),
+ Collections.singletonList("e1"), 0, false));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new GraphPathRef(Arrays.asList("v1", "v2"),
+ Collections.emptyList(), 1, false));
+ }
+
+ @Test
+ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() {
+ SourceDocument document = RetrievalModelJson.fromJson(
+ "{\"documentId\":\"d\",\"dataset\":\"set\",\"datasetVersion\":\"v1\","
+ + "\"split\":\"dev\",\"title\":\"title\",\"sourceUri\":\"uri\","
+ + "\"sourceHash\":\"hash\",\"futureField\":true}", SourceDocument.class);
+ SourceDocument changedTitle = new SourceDocument("d", "set", "v1", "dev",
+ "other title", "uri", "hash");
+ Assertions.assertTrue(document.sameIdentityAs(changedTitle));
+ Assertions.assertNotEquals(document, changedTitle);
+SourceDocument copy = new SourceDocument("d", "set", "v1", "dev", "title", "uri", "hash");
+Assertions.assertEquals(document, copy);
+Assertions.assertEquals(document.hashCode(), copy.hashCode());
+ }
+
+ @Test
+ public void emptyEvidenceFallsBackToStableIdentityFields() {
+ Evidence first = new Evidence("e1", EvidenceKind.CHUNK, null, null, null, null,
+ null, null, null, null);
+ Evidence second = new Evidence("e2", EvidenceKind.CHUNK, "different text", null,
+ null, null, null, null, null, 1);
+ Evidence otherKind = new Evidence("e3", EvidenceKind.ENTITY, null, null, null, null,
+ null, null, null, null);
+ Assertions.assertFalse(first.sameIdentityAs(second));
+ Assertions.assertFalse(first.sameIdentityAs(otherKind));
+ Assertions.assertNotEquals(first, second);
+
+ Assertions.assertTrue(first.sameIdentityAs(new Evidence("e1", EvidenceKind.CHUNK,
+ "different text", null, null, null, null, null, null, null)));
+ Assertions.assertTrue(new Evidence(null, EvidenceKind.CHUNK, "same text", null, null,
+ null, null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK,
+ "same text", null, null, null, null, null, null, null)));
+ Assertions.assertFalse(new Evidence(null, EvidenceKind.CHUNK, null, null, null, null,
+ null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK, null,
+ null, null, null, null, null, null, null)));
+ }
+
+ @Test
+ public void constructorCopiesInputCollections() {
+ java.util.List aliases = new java.util.ArrayList<>();
+ aliases.add("alias-a");
+ EntityRef entity = new EntityRef("entity-1", "Name", aliases, "PERSON", aliases);
+ aliases.set(0, "changed");
+ Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getAliases());
+ Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getSourceChunkIds());
+ }
+
+ @Test
+ public void allModelsRoundTripThroughValidatedJson() {
+ SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev",
+ "Title", "file:///doc", "hash");
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ EntityRef entity = new EntityRef("entity-1", "Name", Collections.singletonList("N"),
+ "PERSON", Collections.singletonList("chunk-1"));
+ GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1");
+ GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2",
+ Collections.singletonList("chunk-1"));
+ GraphPathRef path = new GraphPathRef(Arrays.asList("vertex-1", "vertex-2"),
+ Collections.singletonList("edge-1"), 1, false);
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ ChannelScore score = new ChannelScore("bm25", 2.0, 0.8, 1);
+ GraphVersion graphVersion = new GraphVersion("graph", "v1");
+ IndexVersion indexVersion = new IndexVersion("bm25", "v1", "graph-v1");
+
+ Assertions.assertEquals(document, roundTrip(document, SourceDocument.class));
+ Assertions.assertEquals(chunk, roundTrip(chunk, TextChunk.class));
+ Assertions.assertEquals(entity, roundTrip(entity, EntityRef.class));
+ Assertions.assertEquals(vertex, roundTrip(vertex, GraphVertexRef.class));
+ Assertions.assertEquals(edge, roundTrip(edge, GraphEdgeRef.class));
+ Assertions.assertEquals(path, roundTrip(path, GraphPathRef.class));
+ Assertions.assertEquals(source, roundTrip(source, SourceRef.class));
+ Assertions.assertEquals(score, roundTrip(score, ChannelScore.class));
+ Assertions.assertEquals(graphVersion, roundTrip(graphVersion, GraphVersion.class));
+ Assertions.assertEquals(indexVersion, roundTrip(indexVersion, IndexVersion.class));
+ }
+
+ @Test
+ public void completeEvidenceFixtureRoundTrips() throws IOException {
+ String json = readFixture("retrieval/model/evidence.json");
+ Evidence evidence = RetrievalModelJson.fromJson(json, Evidence.class);
+ Assertions.assertEquals(evidence, RetrievalModelJson.fromJson(
+ RetrievalModelJson.toJson(evidence), Evidence.class));
+ Assertions.assertEquals(2, evidence.getStageScores().size());
+ Assertions.assertEquals(1, evidence.getPaths().get(0).getHop());
+ }
+
+ @Test
+ public void identityFixtureDocumentsFieldBasedComparisons() throws IOException {
+ Map fixture = GSON.fromJson(readFixture(
+ "retrieval/model/identity-cases.json"), Map.class);
+ Assertions.assertNotNull(fixture.get("sourceDocument"));
+ SourceDocument source = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "sourceDocument")), SourceDocument.class);
+ SourceDocument sourceVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "sourceDocumentVariant")), SourceDocument.class);
+ Assertions.assertTrue(source.sameIdentityAs(sourceVariant));
+ TextChunk chunk = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("chunk")),
+ TextChunk.class);
+ TextChunk chunkVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "chunkVariant")), TextChunk.class);
+ Assertions.assertTrue(chunk.sameIdentityAs(chunkVariant));
+ EntityRef entity = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("entity")),
+ EntityRef.class);
+ EntityRef entityVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "entityVariant")), EntityRef.class);
+ Assertions.assertTrue(entity.sameIdentityAs(entityVariant));
+ GraphEdgeRef edge = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("edge")),
+ GraphEdgeRef.class);
+ GraphEdgeRef edgeVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "edgeVariant")), GraphEdgeRef.class);
+ Assertions.assertTrue(edge.sameIdentityAs(edgeVariant));
+ }
+
+ private T roundTrip(T value, Class type) {
+ return RetrievalModelJson.fromJson(RetrievalModelJson.toJson(value), type);
+ }
+
+ private String readFixture(String name) throws IOException {
+ InputStream stream = getClass().getClassLoader().getResourceAsStream(name);
+ Assertions.assertNotNull(stream);
+ try (BufferedReader reader = new BufferedReader(new InputStreamReader(stream, "UTF-8"))) {
+ return reader.lines().collect(Collectors.joining());
+ }
+ }
+}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/evidence.json b/geaflow-ai/src/test/resources/retrieval/model/evidence.json
new file mode 100644
index 000000000..52f46d5ea
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/evidence.json
@@ -0,0 +1 @@
+{"evidenceId":"evidence-1","kind":"GRAPH_FACT","text":"Alice knows Bob.","chunks":[{"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":16,"tokenEstimate":4,"text":"Alice knows Bob.","policyVersion":"policy-v1","textHash":"sha256:chunk-1"}],"entities":[{"entityId":"entity-1","canonicalName":"Alice","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]}],"paths":[{"vertexIds":["vertex-1","vertex-2"],"edgeIds":["edge-1"],"hop":1,"sampled":false}],"sources":[{"documentId":"doc-1","sourceUri":"https://example/doc-1","startOffset":0,"endOffset":16}],"stageScores":{"bm25":{"channel":"bm25","rawScore":2.18,"normalizedScore":0.8,"rank":1},"vector":{"channel":"vector","rawScore":0.76,"normalizedScore":0.7,"rank":2}},"fusedScore":0.84,"rank":1}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json
new file mode 100644
index 000000000..e136cbf86
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json
@@ -0,0 +1,10 @@
+{
+ "sourceDocument": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"uri","sourceHash":"hash"},
+ "sourceDocumentVariant": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Renamed","sourceUri":"uri","sourceHash":"hash"},
+ "chunk": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":1,"text":"text","policyVersion":"policy-v1","textHash":"hash"},
+ "chunkVariant": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":99,"text":"text","policyVersion":"policy-v1","textHash":"hash"},
+ "entity": {"entityId":"entity-1","canonicalName":"Name","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]},
+ "entityVariant": {"entityId":"entity-1","canonicalName":"Name","aliases":["B"],"type":"PERSON","sourceChunkIds":["chunk-2"]},
+ "edge": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-1"]},
+ "edgeVariant": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-2"]}
+}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/source-document.json b/geaflow-ai/src/test/resources/retrieval/model/source-document.json
new file mode 100644
index 000000000..00183986b
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/source-document.json
@@ -0,0 +1 @@
+{"documentId":"doc-1","dataset":"hotpotqa","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"https://example/doc-1","sourceHash":"sha256"}