From 9c762fdf8cadd1317fad7c9ad894a3c4d16feaa4 Mon Sep 17 00:00:00 2001
From: aotenjou
Date: Sun, 6 Sep 2026 11:22:46 +0800
Subject: [PATCH 1/5] feat(ai): add retrieval domain models
---
.../model/document/SourceDocument.java | 119 +++++++++++
.../retrieval/model/document/SourceRef.java | 99 +++++++++
.../retrieval/model/document/TextChunk.java | 141 +++++++++++++
.../model/evidence/ChannelScore.java | 87 ++++++++
.../ai/retrieval/model/evidence/Evidence.java | 195 ++++++++++++++++++
.../model/evidence/EvidenceKind.java | 29 +++
.../ai/retrieval/model/graph/EntityRef.java | 110 ++++++++++
.../retrieval/model/graph/GraphEdgeRef.java | 111 ++++++++++
.../retrieval/model/graph/GraphPathRef.java | 103 +++++++++
.../retrieval/model/graph/GraphVertexRef.java | 78 +++++++
.../ai/retrieval/model/package-info.java | 49 +++++
.../retrieval/model/version/GraphVersion.java | 69 +++++++
.../retrieval/model/version/IndexVersion.java | 78 +++++++
13 files changed, 1268 insertions(+)
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java
new file mode 100644
index 000000000..3b221551b
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java
@@ -0,0 +1,119 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.document;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable metadata describing a source document in a retrieval corpus. */
+public final class SourceDocument {
+
+ @SerializedName("documentId")
+ private final String documentId;
+ @SerializedName("dataset")
+ private final String dataset;
+ @SerializedName("datasetVersion")
+ private final String datasetVersion;
+ @SerializedName("split")
+ private final String split;
+ @SerializedName("title")
+ private final String title;
+ @SerializedName("sourceUri")
+ private final String sourceUri;
+ @SerializedName("sourceHash")
+ private final String sourceHash;
+
+ public SourceDocument(String documentId, String dataset, String datasetVersion,
+ String split, String sourceUri, String sourceHash) {
+ this(documentId, dataset, datasetVersion, split, null, sourceUri, sourceHash);
+ }
+
+ public SourceDocument(String documentId, String dataset, String datasetVersion,
+ String split, String title, String sourceUri, String sourceHash) {
+ this.documentId = ModelValidation.required(documentId, "documentId");
+ this.dataset = ModelValidation.required(dataset, "dataset");
+ this.datasetVersion = ModelValidation.required(datasetVersion, "datasetVersion");
+ this.split = ModelValidation.required(split, "split");
+ this.title = ModelValidation.optional(title);
+ this.sourceUri = ModelValidation.required(sourceUri, "sourceUri");
+ this.sourceHash = ModelValidation.required(sourceHash, "sourceHash");
+ }
+
+ public String getDocumentId() {
+ return documentId;
+ }
+
+ public String getDataset() {
+ return dataset;
+ }
+
+ public String getDatasetVersion() {
+ return datasetVersion;
+ }
+
+ public String getSplit() {
+ return split;
+ }
+
+ public String getTitle() {
+ return title;
+ }
+
+ public String getSourceUri() {
+ return sourceUri;
+ }
+
+ public String getSourceHash() {
+ return sourceHash;
+ }
+
+ public boolean sameIdentityAs(SourceDocument other) {
+ return other != null && Objects.equals(documentId, other.documentId)
+ && Objects.equals(dataset, other.dataset)
+ && Objects.equals(datasetVersion, other.datasetVersion)
+ && Objects.equals(split, other.split)
+ && Objects.equals(sourceUri, other.sourceUri)
+ && Objects.equals(sourceHash, other.sourceHash);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof SourceDocument)) {
+ return false;
+ }
+ SourceDocument that = (SourceDocument) object;
+ return Objects.equals(documentId, that.documentId)
+ && Objects.equals(dataset, that.dataset)
+ && Objects.equals(datasetVersion, that.datasetVersion)
+ && Objects.equals(split, that.split)
+ && Objects.equals(title, that.title)
+ && Objects.equals(sourceUri, that.sourceUri)
+ && Objects.equals(sourceHash, that.sourceHash);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(documentId, dataset, datasetVersion, split, title, sourceUri, sourceHash);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java
new file mode 100644
index 000000000..3c0b758d2
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java
@@ -0,0 +1,99 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.document;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+
+/** Immutable citation pointing to a document and, optionally, a text span. */
+public final class SourceRef {
+
+ @SerializedName("documentId")
+ private final String documentId;
+ @SerializedName("sourceUri")
+ private final String sourceUri;
+ @SerializedName("startOffset")
+ private final Integer startOffset;
+ @SerializedName("endOffset")
+ private final Integer endOffset;
+
+ public SourceRef(String documentId, String sourceUri) {
+ this(documentId, sourceUri, null, null);
+ }
+
+ public SourceRef(String documentId, String sourceUri, Integer startOffset, Integer endOffset) {
+ this.documentId = ModelValidation.required(documentId, "documentId");
+ this.sourceUri = ModelValidation.required(sourceUri, "sourceUri");
+ if ((startOffset == null) != (endOffset == null)) {
+ throw new RetrievalModelValidationException("source offsets must be provided together");
+ }
+ if (startOffset != null) {
+ ModelValidation.nonNegative(startOffset, "startOffset");
+ ModelValidation.nonNegative(endOffset, "endOffset");
+ if (endOffset < startOffset) {
+ throw new RetrievalModelValidationException("endOffset must not be before startOffset");
+ }
+ }
+ this.startOffset = startOffset;
+ this.endOffset = endOffset;
+ }
+
+ public String getDocumentId() {
+ return documentId;
+ }
+
+ public String getSourceUri() {
+ return sourceUri;
+ }
+
+ public Integer getStartOffset() {
+ return startOffset;
+ }
+
+ public Integer getEndOffset() {
+ return endOffset;
+ }
+
+ public boolean sameIdentityAs(SourceRef other) {
+ return equals(other);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof SourceRef)) {
+ return false;
+ }
+ SourceRef that = (SourceRef) object;
+ return Objects.equals(documentId, that.documentId)
+ && Objects.equals(sourceUri, that.sourceUri)
+ && Objects.equals(startOffset, that.startOffset)
+ && Objects.equals(endOffset, that.endOffset);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(documentId, sourceUri, startOffset, endOffset);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java
new file mode 100644
index 000000000..b52a8d3b8
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java
@@ -0,0 +1,141 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.document;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+
+/** Immutable, position-aware text segment produced by a chunking policy. */
+public final class TextChunk {
+
+ @SerializedName("chunkId")
+ private final String chunkId;
+ @SerializedName("documentId")
+ private final String documentId;
+ @SerializedName("chunkIndex")
+ private final int chunkIndex;
+ @SerializedName("startOffset")
+ private final int startOffset;
+ @SerializedName("endOffset")
+ private final int endOffset;
+ @SerializedName("tokenEstimate")
+ private final int tokenEstimate;
+ @SerializedName("text")
+ private final String text;
+ @SerializedName("policyVersion")
+ private final String policyVersion;
+ @SerializedName("textHash")
+ private final String textHash;
+
+ public TextChunk(String chunkId, String documentId, int chunkIndex, int startOffset,
+ int endOffset, int tokenEstimate, String text) {
+ this(chunkId, documentId, chunkIndex, startOffset, endOffset, tokenEstimate, text, null, null);
+ }
+
+ public TextChunk(String chunkId, String documentId, int chunkIndex, int startOffset,
+ int endOffset, int tokenEstimate, String text,
+ String policyVersion, String textHash) {
+ this.chunkId = ModelValidation.required(chunkId, "chunkId");
+ this.documentId = ModelValidation.required(documentId, "documentId");
+ this.chunkIndex = ModelValidation.nonNegative(chunkIndex, "chunkIndex");
+ this.startOffset = ModelValidation.nonNegative(startOffset, "startOffset");
+ this.endOffset = ModelValidation.nonNegative(endOffset, "endOffset");
+ if (endOffset < startOffset) {
+ throw new RetrievalModelValidationException("endOffset must not be before startOffset");
+ }
+ this.tokenEstimate = ModelValidation.nonNegative(tokenEstimate, "tokenEstimate");
+ this.text = ModelValidation.required(text, "text");
+ this.policyVersion = ModelValidation.optionalNonBlank(policyVersion, "policyVersion");
+ this.textHash = ModelValidation.optionalNonBlank(textHash, "textHash");
+ }
+
+ public String getChunkId() {
+ return chunkId;
+ }
+
+ public String getDocumentId() {
+ return documentId;
+ }
+
+ public int getChunkIndex() {
+ return chunkIndex;
+ }
+
+ public int getStartOffset() {
+ return startOffset;
+ }
+
+ public int getEndOffset() {
+ return endOffset;
+ }
+
+ public int getTokenEstimate() {
+ return tokenEstimate;
+ }
+
+ public String getText() {
+ return text;
+ }
+
+ public String getPolicyVersion() {
+ return policyVersion;
+ }
+
+ public String getTextHash() {
+ return textHash;
+ }
+
+ public boolean sameIdentityAs(TextChunk other) {
+ return other != null && Objects.equals(chunkId, other.chunkId)
+ && Objects.equals(documentId, other.documentId)
+ && chunkIndex == other.chunkIndex
+ && startOffset == other.startOffset
+ && endOffset == other.endOffset
+ && Objects.equals(text, other.text)
+ && Objects.equals(policyVersion, other.policyVersion)
+ && Objects.equals(textHash, other.textHash);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof TextChunk)) {
+ return false;
+ }
+ TextChunk that = (TextChunk) object;
+ return chunkIndex == that.chunkIndex && startOffset == that.startOffset
+ && endOffset == that.endOffset && tokenEstimate == that.tokenEstimate
+ && Objects.equals(chunkId, that.chunkId)
+ && Objects.equals(documentId, that.documentId)
+ && Objects.equals(text, that.text)
+ && Objects.equals(policyVersion, that.policyVersion)
+ && Objects.equals(textHash, that.textHash);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(chunkId, documentId, chunkIndex, startOffset, endOffset,
+ tokenEstimate, text, policyVersion, textHash);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java
new file mode 100644
index 000000000..ddb389e83
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java
@@ -0,0 +1,87 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.evidence;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+
+/** Score and ranking metadata emitted by one retrieval channel. */
+public final class ChannelScore {
+
+ @SerializedName("channel")
+ private final String channel;
+ @SerializedName("rawScore")
+ private final double rawScore;
+ @SerializedName("normalizedScore")
+ private final Double normalizedScore;
+ @SerializedName("rank")
+ private final int rank;
+
+ public ChannelScore(String channel, double rawScore, Double normalizedScore, int rank) {
+ this.channel = ModelValidation.required(channel, "channel");
+ this.rawScore = ModelValidation.finite(rawScore, "rawScore");
+ this.normalizedScore = ModelValidation.optionalScore(normalizedScore, "normalizedScore");
+ if (rank < 1) {
+ throw new RetrievalModelValidationException("rank must be at least 1");
+ }
+ this.rank = rank;
+ }
+
+ public String getChannel() {
+ return channel;
+ }
+
+ public double getRawScore() {
+ return rawScore;
+ }
+
+ public Double getNormalizedScore() {
+ return normalizedScore;
+ }
+
+ public int getRank() {
+ return rank;
+ }
+
+ public boolean sameIdentityAs(ChannelScore other) {
+ return equals(other);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof ChannelScore)) {
+ return false;
+ }
+ ChannelScore that = (ChannelScore) object;
+ return Double.compare(rawScore, that.rawScore) == 0 && rank == that.rank
+ && Objects.equals(channel, that.channel)
+ && Objects.equals(normalizedScore, that.normalizedScore);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(channel, rawScore, normalizedScore, rank);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
new file mode 100644
index 000000000..db60cc60d
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
@@ -0,0 +1,195 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.evidence;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Collections;
+import java.util.List;
+import java.util.Map;
+import java.util.Objects;
+import java.util.function.BiPredicate;
+import org.apache.geaflow.ai.retrieval.model.document.SourceRef;
+import org.apache.geaflow.ai.retrieval.model.document.TextChunk;
+import org.apache.geaflow.ai.retrieval.model.graph.EntityRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+
+/**
+ * Immutable retrieval evidence with optional payload references and multi-channel scores.
+ *
+ * Complete value equality includes presentation, provenance, and ranking fields. Identity
+ * comparison intentionally uses only the evidence kind and nested reference identities, allowing
+ * candidates from different retrieval channels to be merged without losing their scores.
+ */
+public final class Evidence {
+
+ @SerializedName("evidenceId")
+ private final String evidenceId;
+ @SerializedName("kind")
+ private final EvidenceKind kind;
+ @SerializedName("text")
+ private final String text;
+ @SerializedName("chunks")
+ private final List chunks;
+ @SerializedName("entities")
+ private final List entities;
+ @SerializedName("paths")
+ private final List paths;
+ @SerializedName("sources")
+ private final List sources;
+ @SerializedName("stageScores")
+ private final Map stageScores;
+ @SerializedName("fusedScore")
+ private final Double fusedScore;
+ @SerializedName("rank")
+ private final Integer rank;
+
+ private Evidence() {
+ this.evidenceId = null;
+ this.kind = null;
+ this.text = null;
+ this.chunks = Collections.emptyList();
+ this.entities = Collections.emptyList();
+ this.paths = Collections.emptyList();
+ this.sources = Collections.emptyList();
+ this.stageScores = Collections.emptyMap();
+ this.fusedScore = null;
+ this.rank = null;
+ }
+
+ public Evidence(String evidenceId, EvidenceKind kind, String text, List chunks,
+ List entities, List paths, List sources,
+ Map stageScores, Double fusedScore, Integer rank) {
+ this.evidenceId = ModelValidation.optionalNonBlank(evidenceId, "evidenceId");
+ this.kind = Objects.requireNonNull(kind, "kind");
+ this.text = ModelValidation.optional(text);
+ this.chunks = ModelValidation.immutableList(chunks, "chunks");
+ this.entities = ModelValidation.immutableList(entities, "entities");
+ this.paths = ModelValidation.immutableList(paths, "paths");
+ this.sources = ModelValidation.immutableList(sources, "sources");
+ this.stageScores = ModelValidation.sortedMap(stageScores);
+ for (Map.Entry entry : this.stageScores.entrySet()) {
+ if (entry.getValue() == null) {
+ throw new NullPointerException("stage score must not be null");
+ }
+ if (!entry.getKey().equals(entry.getValue().getChannel())) {
+ throw new RetrievalModelValidationException("stage score key must match channel");
+ }
+ }
+ this.fusedScore = ModelValidation.optionalScore(fusedScore, "fusedScore");
+ this.rank = ModelValidation.optionalRank(rank, "rank");
+ }
+
+ public String getEvidenceId() {
+ return evidenceId;
+ }
+
+ public EvidenceKind getKind() {
+ return kind;
+ }
+
+ public String getText() {
+ return text;
+ }
+
+ public List getChunks() {
+ return Collections.unmodifiableList(chunks == null ? Collections.emptyList() : chunks);
+ }
+
+ public List getEntities() {
+ return Collections.unmodifiableList(entities == null ? Collections.emptyList() : entities);
+ }
+
+ public List getPaths() {
+ return Collections.unmodifiableList(paths == null ? Collections.emptyList() : paths);
+ }
+
+ public List getSources() {
+ return Collections.unmodifiableList(sources == null ? Collections.emptyList() : sources);
+ }
+
+ public Map getStageScores() {
+ return Collections.unmodifiableMap(stageScores == null
+ ? Collections.emptyMap() : stageScores);
+ }
+
+ public Double getFusedScore() {
+ return fusedScore;
+ }
+
+ public Integer getRank() {
+ return rank;
+ }
+
+ public boolean sameIdentityAs(Evidence other) {
+ return this == other || other != null && kind == other.kind
+ && sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs)
+ && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs)
+ && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs)
+ && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs);
+ }
+
+ private static boolean sameMultiset(List first, List second,
+ BiPredicate identity) {
+ if (first == null || second == null || first.size() != second.size()) {
+ return false;
+ }
+ boolean[] matched = new boolean[second.size()];
+ for (T value : first) {
+ boolean found = false;
+ for (int i = 0; i < second.size(); i++) {
+ if (!matched[i] && identity.test(value, second.get(i))) {
+ matched[i] = true;
+ found = true;
+ break;
+ }
+ }
+ if (!found) {
+ return false;
+ }
+ }
+ return true;
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof Evidence)) {
+ return false;
+ }
+ Evidence that = (Evidence) object;
+ return Objects.equals(evidenceId, that.evidenceId) && kind == that.kind
+ && Objects.equals(text, that.text) && Objects.equals(chunks, that.chunks)
+ && Objects.equals(entities, that.entities) && Objects.equals(paths, that.paths)
+ && Objects.equals(sources, that.sources)
+ && Objects.equals(stageScores, that.stageScores)
+ && Objects.equals(fusedScore, that.fusedScore)
+ && Objects.equals(rank, that.rank);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(evidenceId, kind, text, chunks, entities, paths, sources,
+ stageScores, fusedScore, rank);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java
new file mode 100644
index 000000000..359536970
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java
@@ -0,0 +1,29 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.evidence;
+
+
+/** Candidate payload categories understood by the retrieval response model. */
+public enum EvidenceKind {
+ CHUNK,
+ ENTITY,
+ GRAPH_PATH,
+ GRAPH_FACT
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java
new file mode 100644
index 000000000..94f3d26c0
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java
@@ -0,0 +1,110 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.graph;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Collections;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable graph entity reference with optional aliases and source chunk provenance. */
+public final class EntityRef {
+
+ @SerializedName("entityId")
+ private final String entityId;
+ @SerializedName("canonicalName")
+ private final String canonicalName;
+ @SerializedName("aliases")
+ private final java.util.List aliases;
+ @SerializedName("type")
+ private final String type;
+ @SerializedName("sourceChunkIds")
+ private final java.util.List sourceChunkIds;
+
+ private EntityRef() {
+ this.entityId = null;
+ this.canonicalName = null;
+ this.aliases = java.util.Collections.emptyList();
+ this.type = null;
+ this.sourceChunkIds = java.util.Collections.emptyList();
+ }
+
+ public EntityRef(String entityId, String canonicalName, String type) {
+ this(entityId, canonicalName, java.util.Collections.emptyList(), type,
+ java.util.Collections.emptyList());
+ }
+
+ public EntityRef(String entityId, String canonicalName, java.util.List aliases,
+ String type, java.util.List sourceChunkIds) {
+ this.entityId = ModelValidation.required(entityId, "entityId");
+ this.canonicalName = ModelValidation.required(canonicalName, "canonicalName");
+ this.aliases = ModelValidation.sortedStrings(aliases, "alias");
+ this.type = ModelValidation.required(type, "type");
+ this.sourceChunkIds = ModelValidation.sortedStrings(sourceChunkIds, "sourceChunkId");
+ }
+
+ public String getEntityId() {
+ return entityId;
+ }
+
+ public String getCanonicalName() {
+ return canonicalName;
+ }
+
+ public java.util.List getAliases() {
+ return Collections.unmodifiableList(aliases == null ? Collections.emptyList() : aliases);
+ }
+
+ public String getType() {
+ return type;
+ }
+
+ public java.util.List getSourceChunkIds() {
+ return Collections.unmodifiableList(sourceChunkIds == null
+ ? Collections.emptyList() : sourceChunkIds);
+ }
+
+ public boolean sameIdentityAs(EntityRef other) {
+ return other != null && Objects.equals(entityId, other.entityId)
+ && Objects.equals(canonicalName, other.canonicalName)
+ && Objects.equals(type, other.type);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof EntityRef)) {
+ return false;
+ }
+ EntityRef that = (EntityRef) object;
+ return Objects.equals(entityId, that.entityId)
+ && Objects.equals(canonicalName, that.canonicalName)
+ && Objects.equals(aliases, that.aliases)
+ && Objects.equals(type, that.type)
+ && Objects.equals(sourceChunkIds, that.sourceChunkIds);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(entityId, canonicalName, aliases, type, sourceChunkIds);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java
new file mode 100644
index 000000000..b53a69b81
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java
@@ -0,0 +1,111 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.graph;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Collections;
+import java.util.List;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable graph edge reference connecting two entity identifiers. */
+public final class GraphEdgeRef {
+
+ @SerializedName("edgeId")
+ private final String edgeId;
+ @SerializedName("label")
+ private final String label;
+ @SerializedName("sourceEntityId")
+ private final String sourceEntityId;
+ @SerializedName("targetEntityId")
+ private final String targetEntityId;
+ @SerializedName("sourceChunkIds")
+ private final List sourceChunkIds;
+
+ private GraphEdgeRef() {
+ this.edgeId = null;
+ this.label = null;
+ this.sourceEntityId = null;
+ this.targetEntityId = null;
+ this.sourceChunkIds = Collections.emptyList();
+ }
+
+ public GraphEdgeRef(String edgeId, String label, String sourceEntityId, String targetEntityId) {
+ this(edgeId, label, sourceEntityId, targetEntityId, java.util.Collections.emptyList());
+ }
+
+ public GraphEdgeRef(String edgeId, String label, String sourceEntityId,
+ String targetEntityId, List sourceChunkIds) {
+ this.edgeId = ModelValidation.required(edgeId, "edgeId");
+ this.label = ModelValidation.required(label, "label");
+ this.sourceEntityId = ModelValidation.required(sourceEntityId, "sourceEntityId");
+ this.targetEntityId = ModelValidation.required(targetEntityId, "targetEntityId");
+ this.sourceChunkIds = ModelValidation.sortedStrings(sourceChunkIds, "sourceChunkId");
+ }
+
+ public String getEdgeId() {
+ return edgeId;
+ }
+
+ public String getLabel() {
+ return label;
+ }
+
+ public String getSourceEntityId() {
+ return sourceEntityId;
+ }
+
+ public String getTargetEntityId() {
+ return targetEntityId;
+ }
+
+ public List getSourceChunkIds() {
+ return Collections.unmodifiableList(sourceChunkIds == null
+ ? Collections.emptyList() : sourceChunkIds);
+ }
+
+ public boolean sameIdentityAs(GraphEdgeRef other) {
+ return other != null && Objects.equals(edgeId, other.edgeId)
+ && Objects.equals(label, other.label)
+ && Objects.equals(sourceEntityId, other.sourceEntityId)
+ && Objects.equals(targetEntityId, other.targetEntityId);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphEdgeRef)) {
+ return false;
+ }
+ GraphEdgeRef that = (GraphEdgeRef) object;
+ return Objects.equals(edgeId, that.edgeId)
+ && Objects.equals(label, that.label)
+ && Objects.equals(sourceEntityId, that.sourceEntityId)
+ && Objects.equals(targetEntityId, that.targetEntityId)
+ && Objects.equals(sourceChunkIds, that.sourceChunkIds);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(edgeId, label, sourceEntityId, targetEntityId, sourceChunkIds);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java
new file mode 100644
index 000000000..f2c5e46ec
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java
@@ -0,0 +1,103 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.graph;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Collections;
+import java.util.List;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+
+/** Immutable ordered graph path with parallel vertex and edge identifier sequences. */
+public final class GraphPathRef {
+
+ @SerializedName("vertexIds")
+ private final List vertexIds;
+ @SerializedName("edgeIds")
+ private final List edgeIds;
+ @SerializedName("hop")
+ private final int hop;
+ @SerializedName("sampled")
+ private final boolean sampled;
+
+ private GraphPathRef() {
+ this.vertexIds = Collections.emptyList();
+ this.edgeIds = Collections.emptyList();
+ this.hop = 0;
+ this.sampled = false;
+ }
+
+ public GraphPathRef(List vertexIds, List edgeIds, int hop, boolean sampled) {
+ this.vertexIds = ModelValidation.immutableList(vertexIds, "vertexIds");
+ this.edgeIds = ModelValidation.immutableList(edgeIds, "edgeIds");
+ this.hop = ModelValidation.nonNegative(hop, "hop");
+ this.sampled = sampled;
+ if (this.vertexIds.size() != hop + 1 || this.edgeIds.size() != hop) {
+ throw new RetrievalModelValidationException(
+ "path vertex/edge counts must match hop");
+ }
+ for (String vertexId : this.vertexIds) {
+ ModelValidation.required(vertexId, "vertexId");
+ }
+ for (String edgeId : this.edgeIds) {
+ ModelValidation.required(edgeId, "edgeId");
+ }
+ }
+
+ public List getVertexIds() {
+ return Collections.unmodifiableList(vertexIds == null ? Collections.emptyList() : vertexIds);
+ }
+
+ public List getEdgeIds() {
+ return Collections.unmodifiableList(edgeIds == null ? Collections.emptyList() : edgeIds);
+ }
+
+ public int getHop() {
+ return hop;
+ }
+
+ public boolean isSampled() {
+ return sampled;
+ }
+
+ public boolean sameIdentityAs(GraphPathRef other) {
+ return equals(other);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphPathRef)) {
+ return false;
+ }
+ GraphPathRef that = (GraphPathRef) object;
+ return hop == that.hop && sampled == that.sampled
+ && Objects.equals(vertexIds, that.vertexIds)
+ && Objects.equals(edgeIds, that.edgeIds);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(vertexIds, edgeIds, hop, sampled);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java
new file mode 100644
index 000000000..5f5649c30
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.graph;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable graph vertex reference associated with an entity identifier. */
+public final class GraphVertexRef {
+
+ @SerializedName("label")
+ private final String label;
+ @SerializedName("vertexId")
+ private final String vertexId;
+ @SerializedName("entityId")
+ private final String entityId;
+
+ public GraphVertexRef(String label, String vertexId, String entityId) {
+ this.label = ModelValidation.required(label, "label");
+ this.vertexId = ModelValidation.required(vertexId, "vertexId");
+ this.entityId = ModelValidation.required(entityId, "entityId");
+ }
+
+ public String getLabel() {
+ return label;
+ }
+
+ public String getVertexId() {
+ return vertexId;
+ }
+
+ public String getEntityId() {
+ return entityId;
+ }
+
+ public boolean sameIdentityAs(GraphVertexRef other) {
+ return other != null && Objects.equals(label, other.label)
+ && Objects.equals(vertexId, other.vertexId)
+ && Objects.equals(entityId, other.entityId);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphVertexRef)) {
+ return false;
+ }
+ GraphVertexRef that = (GraphVertexRef) object;
+ return Objects.equals(label, that.label)
+ && Objects.equals(vertexId, that.vertexId)
+ && Objects.equals(entityId, that.entityId);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(label, vertexId, entityId);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
new file mode 100644
index 000000000..54dbdf420
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
@@ -0,0 +1,49 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+/**
+ * Immutable value objects shared by GraphRAG ingestion, retrieval, and evaluation.
+ *
+ * IDs in these objects are opaque values supplied by callers. This package does not generate
+ * IDs. Use {@code sameIdentityAs} for identity comparisons and {@code equals} when complete value
+ * equality is required. Evidence identity excludes evidenceId, text, scores and ranks so
+ * multi-channel candidates can be merged; its nested references are compared recursively.
+ *
+ * JSON uses explicit camelCase field names. Optional fields may be added only additively; unknown
+ * fields are ignored, null optional scalars are omitted, and collection fields use empty arrays or
+ * objects. Callers must use {@link RetrievalModelJson} for deserialization so required-field and
+ * range validation cannot be bypassed by reflection. Text offsets are half-open
+ * ({@code startOffset <= endOffset}); ranks are one-based, raw scores are finite, and normalized
+ * and fused scores are finite values in {@code [0, 1]}.
+ *
+ * Identity fields are documentId/dataset/datasetVersion/split/sourceUri/sourceHash for
+ * {@code SourceDocument}; chunkId/documentId/chunkIndex/startOffset/endOffset/text/policyVersion/
+ * textHash for {@code TextChunk}; entityId/canonicalName/type for {@code EntityRef}; all fields
+ * for {@code GraphVertexRef}, {@code GraphVersion}, {@code IndexVersion}, {@code SourceRef},
+ * {@code GraphPathRef}, and {@code ChannelScore}; and edgeId/label/sourceEntityId/targetEntityId
+ * for {@code GraphEdgeRef}. Evidence compares kind and nested reference identities. Collection
+ * order does not affect Evidence identity, while vertex and edge order inside one graph path does.
+ * A graph path has exactly {@code hop + 1} vertices and {@code hop} edges. Evidence kind labels
+ * the candidate payload (chunk, entity, path, or graph fact), while optional lists preserve the
+ * supporting provenance and may be empty for compatibility.
+ */
+package org.apache.geaflow.ai.retrieval.model;
+
+import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson;
+
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java
new file mode 100644
index 000000000..26f6003ff
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java
@@ -0,0 +1,69 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.version;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable name/version pair identifying a graph snapshot. */
+public final class GraphVersion {
+
+ @SerializedName("graphName")
+ private final String graphName;
+ @SerializedName("version")
+ private final String version;
+
+ public GraphVersion(String graphName, String version) {
+ this.graphName = ModelValidation.required(graphName, "graphName");
+ this.version = ModelValidation.required(version, "version");
+ }
+
+ public String getGraphName() {
+ return graphName;
+ }
+
+ public String getVersion() {
+ return version;
+ }
+
+ public boolean sameIdentityAs(GraphVersion other) {
+ return other != null && Objects.equals(graphName, other.graphName)
+ && Objects.equals(version, other.version);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof GraphVersion)) {
+ return false;
+ }
+ GraphVersion that = (GraphVersion) object;
+ return Objects.equals(graphName, that.graphName)
+ && Objects.equals(version, that.version);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(graphName, version);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java
new file mode 100644
index 000000000..1e1c791b7
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model.version;
+
+import com.google.gson.annotations.SerializedName;
+import java.util.Objects;
+import org.apache.geaflow.ai.retrieval.validation.ModelValidation;
+
+/** Immutable index snapshot identifier together with its source graph version. */
+public final class IndexVersion {
+
+ @SerializedName("indexName")
+ private final String indexName;
+ @SerializedName("version")
+ private final String version;
+ @SerializedName("graphVersion")
+ private final String graphVersion;
+
+ public IndexVersion(String indexName, String version, String graphVersion) {
+ this.indexName = ModelValidation.required(indexName, "indexName");
+ this.version = ModelValidation.required(version, "version");
+ this.graphVersion = ModelValidation.required(graphVersion, "graphVersion");
+ }
+
+ public String getIndexName() {
+ return indexName;
+ }
+
+ public String getVersion() {
+ return version;
+ }
+
+ public String getGraphVersion() {
+ return graphVersion;
+ }
+
+ public boolean sameIdentityAs(IndexVersion other) {
+ return other != null && Objects.equals(indexName, other.indexName)
+ && Objects.equals(version, other.version)
+ && Objects.equals(graphVersion, other.graphVersion);
+ }
+
+ @Override
+ public boolean equals(Object object) {
+ if (this == object) {
+ return true;
+ }
+ if (!(object instanceof IndexVersion)) {
+ return false;
+ }
+ IndexVersion that = (IndexVersion) object;
+ return Objects.equals(indexName, that.indexName)
+ && Objects.equals(version, that.version)
+ && Objects.equals(graphVersion, that.graphVersion);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(indexName, version, graphVersion);
+ }
+}
From 416c25b0073e6e1dc5731a0da2ef3efab2a25a1b Mon Sep 17 00:00:00 2001
From: aotenjou
Date: Sun, 6 Sep 2026 11:22:53 +0800
Subject: [PATCH 2/5] feat(ai): add retrieval model codec and validation
---
.../retrieval/codec/RetrievalModelJson.java | 283 ++++++++++++++++++
.../retrieval/validation/ModelValidation.java | 126 ++++++++
.../RetrievalModelValidationException.java | 32 ++
3 files changed, 441 insertions(+)
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java
create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java
new file mode 100644
index 000000000..dea798bf6
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java
@@ -0,0 +1,283 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.codec;
+
+import com.google.gson.Gson;
+import com.google.gson.JsonArray;
+import com.google.gson.JsonElement;
+import com.google.gson.JsonObject;
+import com.google.gson.JsonParseException;
+import com.google.gson.JsonParser;
+import java.util.ArrayList;
+import java.util.Collections;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import org.apache.geaflow.ai.retrieval.model.document.SourceDocument;
+import org.apache.geaflow.ai.retrieval.model.document.SourceRef;
+import org.apache.geaflow.ai.retrieval.model.document.TextChunk;
+import org.apache.geaflow.ai.retrieval.model.evidence.ChannelScore;
+import org.apache.geaflow.ai.retrieval.model.evidence.Evidence;
+import org.apache.geaflow.ai.retrieval.model.evidence.EvidenceKind;
+import org.apache.geaflow.ai.retrieval.model.graph.EntityRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphEdgeRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphVertexRef;
+import org.apache.geaflow.ai.retrieval.model.version.GraphVersion;
+import org.apache.geaflow.ai.retrieval.model.version.IndexVersion;
+
+/**
+ * Validated JSON boundary for retrieval domain models.
+ *
+ * The codec keeps deserialization on the public constructors so required fields, ranges, and
+ * immutable collection guarantees are applied consistently. Unknown properties are ignored for
+ * additive wire compatibility; callers should use this class instead of Gson directly.
+ */
+public final class RetrievalModelJson {
+
+ private static final Gson GSON = new Gson();
+
+ private RetrievalModelJson() {
+ }
+
+ public static String toJson(Object value) {
+ return GSON.toJson(value);
+ }
+
+ public static T fromJson(String json, Class type) {
+ if (json == null) {
+ throw new NullPointerException("json");
+ }
+ if (type == null) {
+ throw new NullPointerException("type");
+ }
+ JsonElement element = new JsonParser().parse(json);
+ if (!element.isJsonObject()) {
+ throw new JsonParseException("model JSON must be an object");
+ }
+ try {
+ Object value = parse(element.getAsJsonObject(), type);
+ return type.cast(value);
+ } catch (JsonParseException e) {
+ throw e;
+ } catch (RuntimeException e) {
+ throw new JsonParseException("invalid " + type.getSimpleName() + ": "
+ + e.getMessage(), e);
+ }
+ }
+
+ private static Object parse(JsonObject object, Class> type) {
+ if (type == SourceDocument.class) {
+ return new SourceDocument(requiredString(object, "documentId"),
+ requiredString(object, "dataset"), requiredString(object, "datasetVersion"),
+ requiredString(object, "split"), optionalString(object, "title"),
+ requiredString(object, "sourceUri"), requiredString(object, "sourceHash"));
+ } else if (type == TextChunk.class) {
+ return new TextChunk(requiredString(object, "chunkId"),
+ requiredString(object, "documentId"), requiredInt(object, "chunkIndex"),
+ requiredInt(object, "startOffset"), requiredInt(object, "endOffset"),
+ requiredInt(object, "tokenEstimate"), requiredString(object, "text"),
+ optionalString(object, "policyVersion"), optionalString(object, "textHash"));
+ } else if (type == EntityRef.class) {
+ return new EntityRef(requiredString(object, "entityId"),
+ requiredString(object, "canonicalName"), strings(object, "aliases"),
+ requiredString(object, "type"), strings(object, "sourceChunkIds"));
+ } else if (type == GraphVertexRef.class) {
+ return new GraphVertexRef(requiredString(object, "label"),
+ requiredString(object, "vertexId"), requiredString(object, "entityId"));
+ } else if (type == GraphEdgeRef.class) {
+ return new GraphEdgeRef(requiredString(object, "edgeId"),
+ requiredString(object, "label"), requiredString(object, "sourceEntityId"),
+ requiredString(object, "targetEntityId"),
+ strings(object, "sourceChunkIds"));
+ } else if (type == GraphPathRef.class) {
+ return new GraphPathRef(strings(object, "vertexIds"),
+ strings(object, "edgeIds"), requiredInt(object, "hop"),
+ requiredBoolean(object, "sampled"));
+ } else if (type == SourceRef.class) {
+ return new SourceRef(requiredString(object, "documentId"),
+ requiredString(object, "sourceUri"), optionalInt(object, "startOffset"),
+ optionalInt(object, "endOffset"));
+ } else if (type == ChannelScore.class) {
+ return new ChannelScore(requiredString(object, "channel"),
+ requiredDouble(object, "rawScore"), optionalDouble(object, "normalizedScore"),
+ requiredInt(object, "rank"));
+ } else if (type == GraphVersion.class) {
+ return new GraphVersion(requiredString(object, "graphName"),
+ requiredString(object, "version"));
+ } else if (type == IndexVersion.class) {
+ return new IndexVersion(requiredString(object, "indexName"),
+ requiredString(object, "version"), requiredString(object, "graphVersion"));
+ } else if (type == Evidence.class) {
+ return new Evidence(optionalString(object, "evidenceId"), kind(object),
+ optionalString(object, "text"), models(object, "chunks", TextChunk.class),
+ models(object, "entities", EntityRef.class), models(object, "paths", GraphPathRef.class),
+ models(object, "sources", SourceRef.class), scores(object),
+ optionalDouble(object, "fusedScore"), optionalInt(object, "rank"));
+ }
+ throw new JsonParseException("unsupported retrieval model: " + type.getName());
+ }
+
+ private static EvidenceKind kind(JsonObject object) {
+ String value = requiredString(object, "kind");
+ try {
+ return EvidenceKind.valueOf(value);
+ } catch (IllegalArgumentException e) {
+ throw new JsonParseException("unknown evidence kind: " + value);
+ }
+ }
+
+ private static Map scores(JsonObject object) {
+ JsonElement element = object.get("stageScores");
+ if (element == null || element.isJsonNull()) {
+ return Collections.emptyMap();
+ }
+ if (!element.isJsonObject()) {
+ throw new JsonParseException("stageScores must be an object");
+ }
+ Map result = new LinkedHashMap<>();
+ for (Map.Entry entry : element.getAsJsonObject().entrySet()) {
+ if (!entry.getValue().isJsonObject()) {
+ throw new JsonParseException("stage score must be an object");
+ }
+ result.put(entry.getKey(), fromElement(entry.getValue(), ChannelScore.class));
+ }
+ return result;
+ }
+
+ private static List models(JsonObject object, String name, Class type) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull()) {
+ return Collections.emptyList();
+ }
+ if (!element.isJsonArray()) {
+ throw new JsonParseException(name + " must be an array");
+ }
+ List result = new ArrayList<>();
+ for (JsonElement item : element.getAsJsonArray()) {
+ if (!item.isJsonObject()) {
+ throw new JsonParseException(name + " elements must be objects");
+ }
+ result.add(fromElement(item, type));
+ }
+ return result;
+ }
+
+ private static List strings(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull()) {
+ return Collections.emptyList();
+ }
+ if (!element.isJsonArray()) {
+ throw new JsonParseException(name + " must be an array");
+ }
+ List result = new ArrayList<>();
+ JsonArray array = element.getAsJsonArray();
+ for (JsonElement item : array) {
+ if (item.isJsonNull() || !item.isJsonPrimitive()
+ || !item.getAsJsonPrimitive().isString()) {
+ throw new JsonParseException(name + " must contain strings");
+ }
+ result.add(item.getAsString());
+ }
+ return result;
+ }
+
+ private static T fromElement(JsonElement element, Class type) {
+ return type.cast(parse(element.getAsJsonObject(), type));
+ }
+
+ private static String requiredString(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull() || !element.isJsonPrimitive()
+ || !element.getAsJsonPrimitive().isString()) {
+ throw new JsonParseException(name + " is required");
+ }
+ return element.getAsString();
+ }
+
+ private static String optionalString(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull()) {
+ return null;
+ }
+ if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isString()) {
+ throw new JsonParseException(name + " must be a string");
+ }
+ return element.getAsString();
+ }
+
+ private static int requiredInt(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull() || !element.isJsonPrimitive()
+ || !element.getAsJsonPrimitive().isNumber()) {
+ throw new JsonParseException(name + " is required");
+ }
+ int value = element.getAsInt();
+ if (element.getAsDouble() != value) {
+ throw new JsonParseException(name + " must be an integer");
+ }
+ return value;
+ }
+
+ private static Integer optionalInt(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull()) {
+ return null;
+ }
+ if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isNumber()) {
+ throw new JsonParseException(name + " must be a number");
+ }
+ int value = element.getAsInt();
+ if (element.getAsDouble() != value) {
+ throw new JsonParseException(name + " must be an integer");
+ }
+ return value;
+ }
+
+ private static double requiredDouble(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull() || !element.isJsonPrimitive()
+ || !element.getAsJsonPrimitive().isNumber()) {
+ throw new JsonParseException(name + " is required");
+ }
+ return element.getAsDouble();
+ }
+
+ private static Double optionalDouble(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull()) {
+ return null;
+ }
+ if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isNumber()) {
+ throw new JsonParseException(name + " must be a number");
+ }
+ return element.getAsDouble();
+ }
+
+ private static boolean requiredBoolean(JsonObject object, String name) {
+ JsonElement element = object.get(name);
+ if (element == null || element.isJsonNull() || !element.isJsonPrimitive()
+ || !element.getAsJsonPrimitive().isBoolean()) {
+ throw new JsonParseException(name + " is required");
+ }
+ return element.getAsBoolean();
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java
new file mode 100644
index 000000000..b4176be4a
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java
@@ -0,0 +1,126 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.validation;
+
+import java.util.ArrayList;
+import java.util.Collections;
+import java.util.Comparator;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Objects;
+
+/** Shared constructor validation and defensive-copy helpers for retrieval models. */
+public final class ModelValidation {
+
+ private ModelValidation() {
+ }
+
+ public static String required(String value, String name) {
+ Objects.requireNonNull(value, name);
+ if (value.trim().isEmpty()) {
+ throw new RetrievalModelValidationException(name + " must not be blank");
+ }
+ return value;
+ }
+
+ public static String optional(String value) {
+ return value == null ? null : value;
+ }
+
+ public static String optionalNonBlank(String value, String name) {
+ if (value == null) {
+ return null;
+ }
+ return required(value, name);
+ }
+
+ public static int nonNegative(int value, String name) {
+ if (value < 0) {
+ throw new RetrievalModelValidationException(name + " must be non-negative");
+ }
+ return value;
+ }
+
+ public static double finite(double value, String name) {
+ if (Double.isNaN(value) || Double.isInfinite(value)) {
+ throw new RetrievalModelValidationException(name + " must be finite");
+ }
+ return value;
+ }
+
+ public static Double optionalScore(Double value, String name) {
+ if (value == null) {
+ return null;
+ }
+ finite(value, name);
+ if (value < 0.0 || value > 1.0) {
+ throw new RetrievalModelValidationException(name + " must be in [0, 1]");
+ }
+ return value;
+ }
+
+ public static Integer optionalRank(Integer value, String name) {
+ if (value != null && value < 1) {
+ throw new RetrievalModelValidationException(name + " must be at least 1");
+ }
+ return value;
+ }
+
+ public static List immutableList(List values, String name) {
+ if (values == null || values.isEmpty()) {
+ return Collections.emptyList();
+ }
+ List copy = new ArrayList<>(values);
+ for (T value : copy) {
+ Objects.requireNonNull(value, name + " element");
+ }
+ return Collections.unmodifiableList(copy);
+ }
+
+ public static List sortedStrings(List values, String name) {
+ List result = immutableList(values, name);
+ if (result.isEmpty()) {
+ return result;
+ }
+ List sorted = new ArrayList<>(result.size());
+ for (String value : result) {
+ sorted.add(required(value, name));
+ }
+ sorted.sort(Comparator.naturalOrder());
+ return Collections.unmodifiableList(sorted);
+ }
+
+ public static Map sortedMap(Map values) {
+ if (values == null || values.isEmpty()) {
+ return Collections.emptyMap();
+ }
+ List keys = new ArrayList<>(values.keySet());
+ for (String key : keys) {
+ required(key, "map key");
+ }
+ keys.sort(Comparator.naturalOrder());
+ Map sorted = new LinkedHashMap<>();
+ for (String key : keys) {
+ sorted.put(key, values.get(key));
+ }
+ return Collections.unmodifiableMap(sorted);
+ }
+}
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java
new file mode 100644
index 000000000..e09c381d4
--- /dev/null
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java
@@ -0,0 +1,32 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.validation;
+
+/** Indicates invalid data at the retrieval model boundary. */
+public class RetrievalModelValidationException extends IllegalArgumentException {
+
+ public RetrievalModelValidationException(String message) {
+ super(message);
+ }
+
+ public RetrievalModelValidationException(String message, Throwable cause) {
+ super(message, cause);
+ }
+}
From d4dc437a1a0d97ec5bde95245d612c560362ebe4 Mon Sep 17 00:00:00 2001
From: aotenjou
Date: Sun, 6 Sep 2026 11:23:03 +0800
Subject: [PATCH 3/5] test(ai): cover retrieval domain models
---
.../model/RetrievalDomainModelTest.java | 381 ++++++++++++++++++
.../resources/retrieval/model/evidence.json | 1 +
.../retrieval/model/identity-cases.json | 10 +
.../retrieval/model/source-document.json | 1 +
4 files changed, 393 insertions(+)
create mode 100644 geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
create mode 100644 geaflow-ai/src/test/resources/retrieval/model/evidence.json
create mode 100644 geaflow-ai/src/test/resources/retrieval/model/identity-cases.json
create mode 100644 geaflow-ai/src/test/resources/retrieval/model/source-document.json
diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
new file mode 100644
index 000000000..1d56475ee
--- /dev/null
+++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
@@ -0,0 +1,381 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.geaflow.ai.retrieval.model;
+
+import com.google.gson.Gson;
+import com.google.gson.JsonParseException;
+import java.io.BufferedReader;
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.InputStreamReader;
+import java.util.Arrays;
+import java.util.Collections;
+import java.util.HashMap;
+import java.util.Map;
+import java.util.stream.Collectors;
+import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson;
+import org.apache.geaflow.ai.retrieval.model.document.SourceDocument;
+import org.apache.geaflow.ai.retrieval.model.document.SourceRef;
+import org.apache.geaflow.ai.retrieval.model.document.TextChunk;
+import org.apache.geaflow.ai.retrieval.model.evidence.ChannelScore;
+import org.apache.geaflow.ai.retrieval.model.evidence.Evidence;
+import org.apache.geaflow.ai.retrieval.model.evidence.EvidenceKind;
+import org.apache.geaflow.ai.retrieval.model.graph.EntityRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphEdgeRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef;
+import org.apache.geaflow.ai.retrieval.model.graph.GraphVertexRef;
+import org.apache.geaflow.ai.retrieval.model.version.GraphVersion;
+import org.apache.geaflow.ai.retrieval.model.version.IndexVersion;
+import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException;
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/**
+ * Regression coverage for retrieval domain models, validation rules, and JSON round-tripping.
+ */
+public class RetrievalDomainModelTest {
+
+ private static final Gson GSON = new Gson();
+
+ @Test
+ public void modelsAreValueObjectsAndDefensivelyCopyCollections() {
+ java.util.List aliases = Arrays.asList("Kong Fuzi", "Master Kong");
+ EntityRef entity = new EntityRef("entity-1", "Confucius", aliases,
+ "PERSON", Collections.singletonList("chunk-1"));
+
+ Assertions.assertEquals(Arrays.asList("Kong Fuzi", "Master Kong"), entity.getAliases());
+ Assertions.assertThrows(UnsupportedOperationException.class,
+ () -> entity.getAliases().add("孔子"));
+ Assertions.assertEquals(entity, new EntityRef("entity-1", "Confucius",
+ Arrays.asList("Master Kong", "Kong Fuzi"), "PERSON",
+ Collections.singletonList("chunk-1")));
+ Assertions.assertTrue(entity.sameIdentityAs(new EntityRef("entity-1", "Confucius",
+ Collections.emptyList(), "PERSON", Collections.emptyList())));
+
+ SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev",
+ "Title", "file:///doc", "hash");
+ Assertions.assertTrue(document.sameIdentityAs(new SourceDocument("doc-1", "dataset",
+ "v1", "dev", "Other title", "file:///doc", "hash")));
+
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, "text",
+ "policy-v1", "hash");
+ Assertions.assertTrue(chunk.sameIdentityAs(new TextChunk("chunk-1", "doc-1", 0,
+ 0, 4, 99, "text", "policy-v1", "hash")));
+
+ GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1");
+ Assertions.assertFalse(vertex.sameIdentityAs(new GraphVertexRef("company", "vertex-1",
+ "entity-1")));
+ GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2",
+ Collections.singletonList("chunk-1"));
+ Assertions.assertTrue(edge.sameIdentityAs(new GraphEdgeRef("edge-1", "knows",
+ "entity-1", "entity-2", Collections.singletonList("chunk-2"))));
+
+ Assertions.assertTrue(new GraphVersion("graph", "v1")
+ .sameIdentityAs(new GraphVersion("graph", "v1")));
+ Assertions.assertFalse(new IndexVersion("bm25", "v1", "graph-v1")
+ .sameIdentityAs(new IndexVersion("vector", "v1", "graph-v1")));
+ }
+
+ @Test
+ public void validatesRequiredFieldsAndRanges() {
+ Assertions.assertThrows(NullPointerException.class,
+ () -> new GraphVersion(null, "v1"));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new TextChunk("chunk-1", "doc-1", 0, 8, 2, 1, "text", null, null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new ChannelScore("bm25", 1.0, 0.5, 0));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new SourceRef("doc-1", "file:///doc", 4, 2));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new Evidence(" ", EvidenceKind.CHUNK, null, null, null, null, null, null,
+ null, null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new TextChunk("chunk-1", "doc-1", 0, 0, 1, 1, "text", " ", null));
+ }
+
+ @Test
+ public void evidenceMergesChannelsByFieldsNotScores() {
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ GraphPathRef path = new GraphPathRef(Collections.singletonList("v1"),
+ Collections.emptyList(), 0, false);
+ Map bm25 = new HashMap<>();
+ bm25.put("bm25", new ChannelScore("bm25", 2.0, 0.8, 1));
+ Map vector = new HashMap<>();
+ vector.put("vector", new ChannelScore("vector", 0.9, 0.7, 3));
+
+ Evidence first = new Evidence("evidence-1", EvidenceKind.CHUNK, "text",
+ Collections.singletonList(chunk), Collections.emptyList(),
+ Collections.singletonList(path), Collections.singletonList(source), bm25, 0.8, 1);
+ Evidence second = new Evidence("evidence-2", EvidenceKind.CHUNK, "text",
+ Collections.singletonList(chunk), Collections.emptyList(),
+ Collections.singletonList(path), Collections.singletonList(source), vector, 0.7, 3);
+
+ Assertions.assertNotEquals(first, second);
+ Assertions.assertTrue(first.sameIdentityAs(second));
+ Assertions.assertEquals(0.8, first.getStageScores().get("bm25").getNormalizedScore());
+ Assertions.assertEquals(0.7, second.getStageScores().get("vector").getNormalizedScore());
+ String json = GSON.toJson(first);
+ Assertions.assertTrue(json.contains("\"chunks\""));
+ Assertions.assertTrue(json.contains("\"stageScores\""));
+ Assertions.assertEquals(first, RetrievalModelJson.fromJson(json, Evidence.class));
+ }
+
+ @Test
+ public void evidenceIdentityUsesNestedIdentityAndIgnoresCollectionOrder() {
+ TextChunk firstChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ TextChunk secondChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 99,
+ "text", "policy-v1", "hash");
+ EntityRef firstEntity = new EntityRef("entity-1", "Name",
+ Collections.singletonList("alias-a"), "PERSON", Collections.singletonList("chunk-1"));
+ EntityRef secondEntity = new EntityRef("entity-1", "Name",
+ Collections.singletonList("alias-b"), "PERSON", Collections.singletonList("chunk-2"));
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ SourceRef secondSource = new SourceRef("doc-2", "file:///doc-2", 0, 4);
+ GraphPathRef path = new GraphPathRef(Arrays.asList("v1", "v2"),
+ Collections.singletonList("e1"), 1, false);
+ GraphPathRef secondPath = new GraphPathRef(Arrays.asList("v2", "v3"),
+ Collections.singletonList("e2"), 1, false);
+
+ Evidence first = new Evidence("e-1", EvidenceKind.ENTITY, "first",
+ Collections.singletonList(firstChunk), Collections.singletonList(firstEntity),
+ Arrays.asList(path, secondPath), Arrays.asList(source, secondSource),
+ Collections.emptyMap(), 0.1, 1);
+ Evidence second = new Evidence("e-2", EvidenceKind.ENTITY, "second",
+ Collections.singletonList(secondChunk), Collections.singletonList(secondEntity),
+ Arrays.asList(secondPath, path), Arrays.asList(secondSource, source),
+ Collections.emptyMap(), 0.9, 2);
+
+ Assertions.assertTrue(first.sameIdentityAs(second));
+ Evidence reordered = new Evidence("e-3", EvidenceKind.ENTITY, null,
+ Collections.singletonList(secondChunk), Collections.singletonList(secondEntity),
+ Arrays.asList(secondPath, path), Arrays.asList(secondSource, source),
+ Collections.emptyMap(), null, null);
+ Assertions.assertTrue(first.sameIdentityAs(reordered));
+ }
+
+ @Test
+ public void gsonRoundTripPreservesFieldsAndOptionalCompatibility() throws IOException {
+ SourceDocument document = new SourceDocument("doc-1", "hotpotqa", "v1",
+ "dev", "Title", "https://example/doc-1", "sha256");
+ String json = GSON.toJson(document);
+ SourceDocument restored = RetrievalModelJson.fromJson(json, SourceDocument.class);
+ Assertions.assertEquals(document, restored);
+ Assertions.assertEquals(readFixture("retrieval/model/source-document.json"), json);
+
+ String oldJson = "{\"documentId\":\"doc-1\",\"dataset\":\"hotpotqa\","
+ + "\"datasetVersion\":\"v1\",\"split\":\"dev\","
+ + "\"sourceUri\":\"https://example/doc-1\",\"sourceHash\":\"sha256\"}";
+ SourceDocument withoutTitle = RetrievalModelJson.fromJson(oldJson, SourceDocument.class);
+ Assertions.assertNull(withoutTitle.getTitle());
+ Assertions.assertEquals(document.getDocumentId(), withoutTitle.getDocumentId());
+
+ String oldEvidenceJson = "{\"kind\":\"CHUNK\",\"evidenceId\":\"e-1\","
+ + "\"text\":\"text\"}";
+ Evidence oldEvidence = RetrievalModelJson.fromJson(oldEvidenceJson, Evidence.class);
+ Assertions.assertNotNull(oldEvidence.getChunks());
+ Assertions.assertNotNull(oldEvidence.getEntities());
+ Assertions.assertNotNull(oldEvidence.getPaths());
+ Assertions.assertNotNull(oldEvidence.getSources());
+ Assertions.assertNotNull(oldEvidence.getStageScores());
+ Assertions.assertTrue(oldEvidence.getChunks().isEmpty());
+ Assertions.assertTrue(GSON.toJson(oldEvidence).contains("\"kind\":\"CHUNK\""));
+ }
+
+ @Test
+ public void validatedJsonRejectsMissingRequiredFieldsAndKeepsEmptyCollections() {
+ Assertions.assertThrows(JsonParseException.class,
+ () -> RetrievalModelJson.fromJson("{\"version\":\"v1\"}", GraphVersion.class));
+ Assertions.assertThrows(JsonParseException.class,
+ () -> RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\",\"evidenceId\":\" \"}",
+ Evidence.class));
+
+ Evidence empty = RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\","
+ + "\"chunks\":null,\"entities\":[],\"paths\":null,\"sources\":null,"
+ + "\"stageScores\":null}", Evidence.class);
+ String json = RetrievalModelJson.toJson(empty);
+ Assertions.assertTrue(json.contains("\"chunks\":[]"));
+ Assertions.assertTrue(json.contains("\"entities\":[]"));
+ Assertions.assertTrue(json.contains("\"paths\":[]"));
+ Assertions.assertTrue(json.contains("\"sources\":[]"));
+ Assertions.assertTrue(json.contains("\"stageScores\":{}"));
+ }
+
+ @Test
+ public void validatedJsonRejectsWrongElementAndNumericTypes() {
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[1],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[true],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"entityId\":\"e\",\"canonicalName\":\"n\","
+ + "\"aliases\":[null],\"type\":\"PERSON\",\"sourceChunkIds\":[]}",
+ EntityRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"chunkId\":\"c\",\"documentId\":\"d\",\"chunkIndex\":1.9,"
+ + "\"startOffset\":0,\"endOffset\":1,\"tokenEstimate\":1,\"text\":\"x\"}",
+ TextChunk.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"channel\":\"bm25\",\"rawScore\":1,\"rank\":2.5}",
+ ChannelScore.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"vertexIds\":[\"v1\",\"v2\"],\"edgeIds\":[\"e1\"],"
+ + "\"hop\":1.5,\"sampled\":false}", GraphPathRef.class));
+ Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson(
+ "{\"kind\":\"CHUNK\",\"fusedScore\":1.1}", Evidence.class));
+ }
+
+ @Test
+ public void modelValidationUsesDedicatedExceptionAndEnforcesRanges() {
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null,
+ null, -0.01, null));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null,
+ null, 1.01, null));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new GraphPathRef(Collections.singletonList("v1"),
+ Collections.singletonList("e1"), 0, false));
+ Assertions.assertThrows(RetrievalModelValidationException.class,
+ () -> new GraphPathRef(Arrays.asList("v1", "v2"),
+ Collections.emptyList(), 1, false));
+ }
+
+ @Test
+ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() {
+ SourceDocument document = RetrievalModelJson.fromJson(
+ "{\"documentId\":\"d\",\"dataset\":\"set\",\"datasetVersion\":\"v1\","
+ + "\"split\":\"dev\",\"title\":\"title\",\"sourceUri\":\"uri\","
+ + "\"sourceHash\":\"hash\",\"futureField\":true}", SourceDocument.class);
+ SourceDocument changedTitle = new SourceDocument("d", "set", "v1", "dev",
+ "other title", "uri", "hash");
+ Assertions.assertTrue(document.sameIdentityAs(changedTitle));
+ Assertions.assertNotEquals(document, changedTitle);
+ Assertions.assertEquals(document.hashCode(), document.hashCode());
+ }
+
+ @Test
+ public void emptyEvidenceIsOnlyIdentityEquivalentWhenStructureMatches() {
+ Evidence first = new Evidence("e1", EvidenceKind.CHUNK, null, null, null, null,
+ null, null, null, null);
+ Evidence second = new Evidence("e2", EvidenceKind.CHUNK, "different text", null,
+ null, null, null, null, null, 1);
+ Evidence otherKind = new Evidence("e3", EvidenceKind.ENTITY, null, null, null, null,
+ null, null, null, null);
+ Assertions.assertTrue(first.sameIdentityAs(second));
+ Assertions.assertFalse(first.sameIdentityAs(otherKind));
+ Assertions.assertNotEquals(first, second);
+ }
+
+ @Test
+ public void constructorCopiesInputCollections() {
+ java.util.List aliases = new java.util.ArrayList<>();
+ aliases.add("alias-a");
+ EntityRef entity = new EntityRef("entity-1", "Name", aliases, "PERSON", aliases);
+ aliases.set(0, "changed");
+ Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getAliases());
+ Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getSourceChunkIds());
+ }
+
+ @Test
+ public void allModelsRoundTripThroughValidatedJson() {
+ SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev",
+ "Title", "file:///doc", "hash");
+ TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1,
+ "text", "policy-v1", "hash");
+ EntityRef entity = new EntityRef("entity-1", "Name", Collections.singletonList("N"),
+ "PERSON", Collections.singletonList("chunk-1"));
+ GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1");
+ GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2",
+ Collections.singletonList("chunk-1"));
+ GraphPathRef path = new GraphPathRef(Arrays.asList("vertex-1", "vertex-2"),
+ Collections.singletonList("edge-1"), 1, false);
+ SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4);
+ ChannelScore score = new ChannelScore("bm25", 2.0, 0.8, 1);
+ GraphVersion graphVersion = new GraphVersion("graph", "v1");
+ IndexVersion indexVersion = new IndexVersion("bm25", "v1", "graph-v1");
+
+ Assertions.assertEquals(document, roundTrip(document, SourceDocument.class));
+ Assertions.assertEquals(chunk, roundTrip(chunk, TextChunk.class));
+ Assertions.assertEquals(entity, roundTrip(entity, EntityRef.class));
+ Assertions.assertEquals(vertex, roundTrip(vertex, GraphVertexRef.class));
+ Assertions.assertEquals(edge, roundTrip(edge, GraphEdgeRef.class));
+ Assertions.assertEquals(path, roundTrip(path, GraphPathRef.class));
+ Assertions.assertEquals(source, roundTrip(source, SourceRef.class));
+ Assertions.assertEquals(score, roundTrip(score, ChannelScore.class));
+ Assertions.assertEquals(graphVersion, roundTrip(graphVersion, GraphVersion.class));
+ Assertions.assertEquals(indexVersion, roundTrip(indexVersion, IndexVersion.class));
+ }
+
+ @Test
+ public void completeEvidenceFixtureRoundTrips() throws IOException {
+ String json = readFixture("retrieval/model/evidence.json");
+ Evidence evidence = RetrievalModelJson.fromJson(json, Evidence.class);
+ Assertions.assertEquals(evidence, RetrievalModelJson.fromJson(
+ RetrievalModelJson.toJson(evidence), Evidence.class));
+ Assertions.assertEquals(2, evidence.getStageScores().size());
+ Assertions.assertEquals(1, evidence.getPaths().get(0).getHop());
+ }
+
+ @Test
+ public void identityFixtureDocumentsFieldBasedComparisons() throws IOException {
+ Map fixture = GSON.fromJson(readFixture(
+ "retrieval/model/identity-cases.json"), Map.class);
+ Assertions.assertNotNull(fixture.get("sourceDocument"));
+ SourceDocument source = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "sourceDocument")), SourceDocument.class);
+ SourceDocument sourceVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "sourceDocumentVariant")), SourceDocument.class);
+ Assertions.assertTrue(source.sameIdentityAs(sourceVariant));
+ TextChunk chunk = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("chunk")),
+ TextChunk.class);
+ TextChunk chunkVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "chunkVariant")), TextChunk.class);
+ Assertions.assertTrue(chunk.sameIdentityAs(chunkVariant));
+ EntityRef entity = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("entity")),
+ EntityRef.class);
+ EntityRef entityVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "entityVariant")), EntityRef.class);
+ Assertions.assertTrue(entity.sameIdentityAs(entityVariant));
+ GraphEdgeRef edge = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("edge")),
+ GraphEdgeRef.class);
+ GraphEdgeRef edgeVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get(
+ "edgeVariant")), GraphEdgeRef.class);
+ Assertions.assertTrue(edge.sameIdentityAs(edgeVariant));
+ }
+
+ private T roundTrip(T value, Class type) {
+ return RetrievalModelJson.fromJson(RetrievalModelJson.toJson(value), type);
+ }
+
+ private String readFixture(String name) throws IOException {
+ InputStream stream = getClass().getClassLoader().getResourceAsStream(name);
+ Assertions.assertNotNull(stream);
+ try (BufferedReader reader = new BufferedReader(new InputStreamReader(stream, "UTF-8"))) {
+ return reader.lines().collect(Collectors.joining());
+ }
+ }
+}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/evidence.json b/geaflow-ai/src/test/resources/retrieval/model/evidence.json
new file mode 100644
index 000000000..52f46d5ea
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/evidence.json
@@ -0,0 +1 @@
+{"evidenceId":"evidence-1","kind":"GRAPH_FACT","text":"Alice knows Bob.","chunks":[{"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":16,"tokenEstimate":4,"text":"Alice knows Bob.","policyVersion":"policy-v1","textHash":"sha256:chunk-1"}],"entities":[{"entityId":"entity-1","canonicalName":"Alice","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]}],"paths":[{"vertexIds":["vertex-1","vertex-2"],"edgeIds":["edge-1"],"hop":1,"sampled":false}],"sources":[{"documentId":"doc-1","sourceUri":"https://example/doc-1","startOffset":0,"endOffset":16}],"stageScores":{"bm25":{"channel":"bm25","rawScore":2.18,"normalizedScore":0.8,"rank":1},"vector":{"channel":"vector","rawScore":0.76,"normalizedScore":0.7,"rank":2}},"fusedScore":0.84,"rank":1}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json
new file mode 100644
index 000000000..e136cbf86
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json
@@ -0,0 +1,10 @@
+{
+ "sourceDocument": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"uri","sourceHash":"hash"},
+ "sourceDocumentVariant": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Renamed","sourceUri":"uri","sourceHash":"hash"},
+ "chunk": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":1,"text":"text","policyVersion":"policy-v1","textHash":"hash"},
+ "chunkVariant": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":99,"text":"text","policyVersion":"policy-v1","textHash":"hash"},
+ "entity": {"entityId":"entity-1","canonicalName":"Name","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]},
+ "entityVariant": {"entityId":"entity-1","canonicalName":"Name","aliases":["B"],"type":"PERSON","sourceChunkIds":["chunk-2"]},
+ "edge": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-1"]},
+ "edgeVariant": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-2"]}
+}
diff --git a/geaflow-ai/src/test/resources/retrieval/model/source-document.json b/geaflow-ai/src/test/resources/retrieval/model/source-document.json
new file mode 100644
index 000000000..00183986b
--- /dev/null
+++ b/geaflow-ai/src/test/resources/retrieval/model/source-document.json
@@ -0,0 +1 @@
+{"documentId":"doc-1","dataset":"hotpotqa","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"https://example/doc-1","sourceHash":"sha256"}
From 5b74d20597014cbe14eef31088b572afa3089d5d Mon Sep 17 00:00:00 2001
From: AzrMedit0x <161927884+aotenjou@users.noreply.github.com>
Date: Wed, 9 Sep 2026 12:09:27 +0800
Subject: [PATCH 4/5] Enhance equality checks for SourceDocument
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
---
.../geaflow/ai/retrieval/model/RetrievalDomainModelTest.java | 4 +++-
1 file changed, 3 insertions(+), 1 deletion(-)
diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
index 1d56475ee..b50563f40 100644
--- a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
+++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
@@ -274,7 +274,9 @@ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() {
"other title", "uri", "hash");
Assertions.assertTrue(document.sameIdentityAs(changedTitle));
Assertions.assertNotEquals(document, changedTitle);
- Assertions.assertEquals(document.hashCode(), document.hashCode());
+SourceDocument copy = new SourceDocument("d", "set", "v1", "dev", "title", "uri", "hash");
+Assertions.assertEquals(document, copy);
+Assertions.assertEquals(document.hashCode(), copy.hashCode());
}
@Test
From ab3273063d6f6b6689baa5867c7aba32f5d5104d Mon Sep 17 00:00:00 2001
From: aotenjou
Date: Wed, 9 Sep 2026 18:46:46 +0800
Subject: [PATCH 5/5] fix(ai): preserve identity for unreferenced evidence
---
.../ai/retrieval/model/evidence/Evidence.java | 31 ++++++++++++++-----
.../ai/retrieval/model/package-info.java | 5 ++-
.../model/RetrievalDomainModelTest.java | 13 ++++++--
3 files changed, 37 insertions(+), 12 deletions(-)
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
index db60cc60d..ef2dd250c 100644
--- a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java
@@ -36,8 +36,9 @@
* Immutable retrieval evidence with optional payload references and multi-channel scores.
*
* Complete value equality includes presentation, provenance, and ranking fields. Identity
- * comparison intentionally uses only the evidence kind and nested reference identities, allowing
- * candidates from different retrieval channels to be merged without losing their scores.
+ * comparison uses the evidence kind and nested reference identities, allowing candidates from
+ * different retrieval channels to be merged without losing their scores. When no references are
+ * available, evidenceId or text provides the fallback identity.
*/
public final class Evidence {
@@ -140,11 +141,27 @@ public Integer getRank() {
}
public boolean sameIdentityAs(Evidence other) {
- return this == other || other != null && kind == other.kind
- && sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs)
- && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs)
- && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs)
- && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs);
+ if (this == other) {
+ return true;
+ }
+ if (other == null || kind != other.kind) {
+ return false;
+ }
+ if (hasReferences() || other.hasReferences()) {
+ return sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs)
+ && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs)
+ && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs)
+ && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs);
+ }
+ if (evidenceId != null && other.evidenceId != null) {
+ return evidenceId.equals(other.evidenceId);
+ }
+ return text != null && other.text != null && text.equals(other.text);
+ }
+
+ private boolean hasReferences() {
+ return !getChunks().isEmpty() || !getEntities().isEmpty()
+ || !getPaths().isEmpty() || !getSources().isEmpty();
}
private static boolean sameMultiset(List first, List second,
diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
index 54dbdf420..463789720 100644
--- a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
+++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java
@@ -22,8 +22,8 @@
*
* IDs in these objects are opaque values supplied by callers. This package does not generate
* IDs. Use {@code sameIdentityAs} for identity comparisons and {@code equals} when complete value
- * equality is required. Evidence identity excludes evidenceId, text, scores and ranks so
- * multi-channel candidates can be merged; its nested references are compared recursively.
+ * equality is required. Evidence identity uses nested references when present, and falls back to
+ * evidenceId or text when no references are available; scores and ranks are ignored.
*
* JSON uses explicit camelCase field names. Optional fields may be added only additively; unknown
* fields are ignored, null optional scalars are omitted, and collection fields use empty arrays or
@@ -46,4 +46,3 @@
package org.apache.geaflow.ai.retrieval.model;
import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson;
-
diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
index b50563f40..494aca92c 100644
--- a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
+++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java
@@ -280,16 +280,25 @@ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() {
}
@Test
- public void emptyEvidenceIsOnlyIdentityEquivalentWhenStructureMatches() {
+ public void emptyEvidenceFallsBackToStableIdentityFields() {
Evidence first = new Evidence("e1", EvidenceKind.CHUNK, null, null, null, null,
null, null, null, null);
Evidence second = new Evidence("e2", EvidenceKind.CHUNK, "different text", null,
null, null, null, null, null, 1);
Evidence otherKind = new Evidence("e3", EvidenceKind.ENTITY, null, null, null, null,
null, null, null, null);
- Assertions.assertTrue(first.sameIdentityAs(second));
+ Assertions.assertFalse(first.sameIdentityAs(second));
Assertions.assertFalse(first.sameIdentityAs(otherKind));
Assertions.assertNotEquals(first, second);
+
+ Assertions.assertTrue(first.sameIdentityAs(new Evidence("e1", EvidenceKind.CHUNK,
+ "different text", null, null, null, null, null, null, null)));
+ Assertions.assertTrue(new Evidence(null, EvidenceKind.CHUNK, "same text", null, null,
+ null, null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK,
+ "same text", null, null, null, null, null, null, null)));
+ Assertions.assertFalse(new Evidence(null, EvidenceKind.CHUNK, null, null, null, null,
+ null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK, null,
+ null, null, null, null, null, null, null)));
}
@Test