From 9c762fdf8cadd1317fad7c9ad894a3c4d16feaa4 Mon Sep 17 00:00:00 2001 From: aotenjou Date: Sun, 6 Sep 2026 11:22:46 +0800 Subject: [PATCH 1/5] feat(ai): add retrieval domain models --- .../model/document/SourceDocument.java | 119 +++++++++++ .../retrieval/model/document/SourceRef.java | 99 +++++++++ .../retrieval/model/document/TextChunk.java | 141 +++++++++++++ .../model/evidence/ChannelScore.java | 87 ++++++++ .../ai/retrieval/model/evidence/Evidence.java | 195 ++++++++++++++++++ .../model/evidence/EvidenceKind.java | 29 +++ .../ai/retrieval/model/graph/EntityRef.java | 110 ++++++++++ .../retrieval/model/graph/GraphEdgeRef.java | 111 ++++++++++ .../retrieval/model/graph/GraphPathRef.java | 103 +++++++++ .../retrieval/model/graph/GraphVertexRef.java | 78 +++++++ .../ai/retrieval/model/package-info.java | 49 +++++ .../retrieval/model/version/GraphVersion.java | 69 +++++++ .../retrieval/model/version/IndexVersion.java | 78 +++++++ 13 files changed, 1268 insertions(+) create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java new file mode 100644 index 000000000..3b221551b --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceDocument.java @@ -0,0 +1,119 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.document; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable metadata describing a source document in a retrieval corpus. */ +public final class SourceDocument { + + @SerializedName("documentId") + private final String documentId; + @SerializedName("dataset") + private final String dataset; + @SerializedName("datasetVersion") + private final String datasetVersion; + @SerializedName("split") + private final String split; + @SerializedName("title") + private final String title; + @SerializedName("sourceUri") + private final String sourceUri; + @SerializedName("sourceHash") + private final String sourceHash; + + public SourceDocument(String documentId, String dataset, String datasetVersion, + String split, String sourceUri, String sourceHash) { + this(documentId, dataset, datasetVersion, split, null, sourceUri, sourceHash); + } + + public SourceDocument(String documentId, String dataset, String datasetVersion, + String split, String title, String sourceUri, String sourceHash) { + this.documentId = ModelValidation.required(documentId, "documentId"); + this.dataset = ModelValidation.required(dataset, "dataset"); + this.datasetVersion = ModelValidation.required(datasetVersion, "datasetVersion"); + this.split = ModelValidation.required(split, "split"); + this.title = ModelValidation.optional(title); + this.sourceUri = ModelValidation.required(sourceUri, "sourceUri"); + this.sourceHash = ModelValidation.required(sourceHash, "sourceHash"); + } + + public String getDocumentId() { + return documentId; + } + + public String getDataset() { + return dataset; + } + + public String getDatasetVersion() { + return datasetVersion; + } + + public String getSplit() { + return split; + } + + public String getTitle() { + return title; + } + + public String getSourceUri() { + return sourceUri; + } + + public String getSourceHash() { + return sourceHash; + } + + public boolean sameIdentityAs(SourceDocument other) { + return other != null && Objects.equals(documentId, other.documentId) + && Objects.equals(dataset, other.dataset) + && Objects.equals(datasetVersion, other.datasetVersion) + && Objects.equals(split, other.split) + && Objects.equals(sourceUri, other.sourceUri) + && Objects.equals(sourceHash, other.sourceHash); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof SourceDocument)) { + return false; + } + SourceDocument that = (SourceDocument) object; + return Objects.equals(documentId, that.documentId) + && Objects.equals(dataset, that.dataset) + && Objects.equals(datasetVersion, that.datasetVersion) + && Objects.equals(split, that.split) + && Objects.equals(title, that.title) + && Objects.equals(sourceUri, that.sourceUri) + && Objects.equals(sourceHash, that.sourceHash); + } + + @Override + public int hashCode() { + return Objects.hash(documentId, dataset, datasetVersion, split, title, sourceUri, sourceHash); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java new file mode 100644 index 000000000..3c0b758d2 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/SourceRef.java @@ -0,0 +1,99 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.document; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; + +/** Immutable citation pointing to a document and, optionally, a text span. */ +public final class SourceRef { + + @SerializedName("documentId") + private final String documentId; + @SerializedName("sourceUri") + private final String sourceUri; + @SerializedName("startOffset") + private final Integer startOffset; + @SerializedName("endOffset") + private final Integer endOffset; + + public SourceRef(String documentId, String sourceUri) { + this(documentId, sourceUri, null, null); + } + + public SourceRef(String documentId, String sourceUri, Integer startOffset, Integer endOffset) { + this.documentId = ModelValidation.required(documentId, "documentId"); + this.sourceUri = ModelValidation.required(sourceUri, "sourceUri"); + if ((startOffset == null) != (endOffset == null)) { + throw new RetrievalModelValidationException("source offsets must be provided together"); + } + if (startOffset != null) { + ModelValidation.nonNegative(startOffset, "startOffset"); + ModelValidation.nonNegative(endOffset, "endOffset"); + if (endOffset < startOffset) { + throw new RetrievalModelValidationException("endOffset must not be before startOffset"); + } + } + this.startOffset = startOffset; + this.endOffset = endOffset; + } + + public String getDocumentId() { + return documentId; + } + + public String getSourceUri() { + return sourceUri; + } + + public Integer getStartOffset() { + return startOffset; + } + + public Integer getEndOffset() { + return endOffset; + } + + public boolean sameIdentityAs(SourceRef other) { + return equals(other); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof SourceRef)) { + return false; + } + SourceRef that = (SourceRef) object; + return Objects.equals(documentId, that.documentId) + && Objects.equals(sourceUri, that.sourceUri) + && Objects.equals(startOffset, that.startOffset) + && Objects.equals(endOffset, that.endOffset); + } + + @Override + public int hashCode() { + return Objects.hash(documentId, sourceUri, startOffset, endOffset); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java new file mode 100644 index 000000000..b52a8d3b8 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/document/TextChunk.java @@ -0,0 +1,141 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.document; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; + +/** Immutable, position-aware text segment produced by a chunking policy. */ +public final class TextChunk { + + @SerializedName("chunkId") + private final String chunkId; + @SerializedName("documentId") + private final String documentId; + @SerializedName("chunkIndex") + private final int chunkIndex; + @SerializedName("startOffset") + private final int startOffset; + @SerializedName("endOffset") + private final int endOffset; + @SerializedName("tokenEstimate") + private final int tokenEstimate; + @SerializedName("text") + private final String text; + @SerializedName("policyVersion") + private final String policyVersion; + @SerializedName("textHash") + private final String textHash; + + public TextChunk(String chunkId, String documentId, int chunkIndex, int startOffset, + int endOffset, int tokenEstimate, String text) { + this(chunkId, documentId, chunkIndex, startOffset, endOffset, tokenEstimate, text, null, null); + } + + public TextChunk(String chunkId, String documentId, int chunkIndex, int startOffset, + int endOffset, int tokenEstimate, String text, + String policyVersion, String textHash) { + this.chunkId = ModelValidation.required(chunkId, "chunkId"); + this.documentId = ModelValidation.required(documentId, "documentId"); + this.chunkIndex = ModelValidation.nonNegative(chunkIndex, "chunkIndex"); + this.startOffset = ModelValidation.nonNegative(startOffset, "startOffset"); + this.endOffset = ModelValidation.nonNegative(endOffset, "endOffset"); + if (endOffset < startOffset) { + throw new RetrievalModelValidationException("endOffset must not be before startOffset"); + } + this.tokenEstimate = ModelValidation.nonNegative(tokenEstimate, "tokenEstimate"); + this.text = ModelValidation.required(text, "text"); + this.policyVersion = ModelValidation.optionalNonBlank(policyVersion, "policyVersion"); + this.textHash = ModelValidation.optionalNonBlank(textHash, "textHash"); + } + + public String getChunkId() { + return chunkId; + } + + public String getDocumentId() { + return documentId; + } + + public int getChunkIndex() { + return chunkIndex; + } + + public int getStartOffset() { + return startOffset; + } + + public int getEndOffset() { + return endOffset; + } + + public int getTokenEstimate() { + return tokenEstimate; + } + + public String getText() { + return text; + } + + public String getPolicyVersion() { + return policyVersion; + } + + public String getTextHash() { + return textHash; + } + + public boolean sameIdentityAs(TextChunk other) { + return other != null && Objects.equals(chunkId, other.chunkId) + && Objects.equals(documentId, other.documentId) + && chunkIndex == other.chunkIndex + && startOffset == other.startOffset + && endOffset == other.endOffset + && Objects.equals(text, other.text) + && Objects.equals(policyVersion, other.policyVersion) + && Objects.equals(textHash, other.textHash); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof TextChunk)) { + return false; + } + TextChunk that = (TextChunk) object; + return chunkIndex == that.chunkIndex && startOffset == that.startOffset + && endOffset == that.endOffset && tokenEstimate == that.tokenEstimate + && Objects.equals(chunkId, that.chunkId) + && Objects.equals(documentId, that.documentId) + && Objects.equals(text, that.text) + && Objects.equals(policyVersion, that.policyVersion) + && Objects.equals(textHash, that.textHash); + } + + @Override + public int hashCode() { + return Objects.hash(chunkId, documentId, chunkIndex, startOffset, endOffset, + tokenEstimate, text, policyVersion, textHash); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java new file mode 100644 index 000000000..ddb389e83 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/ChannelScore.java @@ -0,0 +1,87 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.evidence; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; + +/** Score and ranking metadata emitted by one retrieval channel. */ +public final class ChannelScore { + + @SerializedName("channel") + private final String channel; + @SerializedName("rawScore") + private final double rawScore; + @SerializedName("normalizedScore") + private final Double normalizedScore; + @SerializedName("rank") + private final int rank; + + public ChannelScore(String channel, double rawScore, Double normalizedScore, int rank) { + this.channel = ModelValidation.required(channel, "channel"); + this.rawScore = ModelValidation.finite(rawScore, "rawScore"); + this.normalizedScore = ModelValidation.optionalScore(normalizedScore, "normalizedScore"); + if (rank < 1) { + throw new RetrievalModelValidationException("rank must be at least 1"); + } + this.rank = rank; + } + + public String getChannel() { + return channel; + } + + public double getRawScore() { + return rawScore; + } + + public Double getNormalizedScore() { + return normalizedScore; + } + + public int getRank() { + return rank; + } + + public boolean sameIdentityAs(ChannelScore other) { + return equals(other); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof ChannelScore)) { + return false; + } + ChannelScore that = (ChannelScore) object; + return Double.compare(rawScore, that.rawScore) == 0 && rank == that.rank + && Objects.equals(channel, that.channel) + && Objects.equals(normalizedScore, that.normalizedScore); + } + + @Override + public int hashCode() { + return Objects.hash(channel, rawScore, normalizedScore, rank); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java new file mode 100644 index 000000000..db60cc60d --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java @@ -0,0 +1,195 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.evidence; + +import com.google.gson.annotations.SerializedName; +import java.util.Collections; +import java.util.List; +import java.util.Map; +import java.util.Objects; +import java.util.function.BiPredicate; +import org.apache.geaflow.ai.retrieval.model.document.SourceRef; +import org.apache.geaflow.ai.retrieval.model.document.TextChunk; +import org.apache.geaflow.ai.retrieval.model.graph.EntityRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; + +/** + * Immutable retrieval evidence with optional payload references and multi-channel scores. + * + *

Complete value equality includes presentation, provenance, and ranking fields. Identity + * comparison intentionally uses only the evidence kind and nested reference identities, allowing + * candidates from different retrieval channels to be merged without losing their scores.

+ */ +public final class Evidence { + + @SerializedName("evidenceId") + private final String evidenceId; + @SerializedName("kind") + private final EvidenceKind kind; + @SerializedName("text") + private final String text; + @SerializedName("chunks") + private final List chunks; + @SerializedName("entities") + private final List entities; + @SerializedName("paths") + private final List paths; + @SerializedName("sources") + private final List sources; + @SerializedName("stageScores") + private final Map stageScores; + @SerializedName("fusedScore") + private final Double fusedScore; + @SerializedName("rank") + private final Integer rank; + + private Evidence() { + this.evidenceId = null; + this.kind = null; + this.text = null; + this.chunks = Collections.emptyList(); + this.entities = Collections.emptyList(); + this.paths = Collections.emptyList(); + this.sources = Collections.emptyList(); + this.stageScores = Collections.emptyMap(); + this.fusedScore = null; + this.rank = null; + } + + public Evidence(String evidenceId, EvidenceKind kind, String text, List chunks, + List entities, List paths, List sources, + Map stageScores, Double fusedScore, Integer rank) { + this.evidenceId = ModelValidation.optionalNonBlank(evidenceId, "evidenceId"); + this.kind = Objects.requireNonNull(kind, "kind"); + this.text = ModelValidation.optional(text); + this.chunks = ModelValidation.immutableList(chunks, "chunks"); + this.entities = ModelValidation.immutableList(entities, "entities"); + this.paths = ModelValidation.immutableList(paths, "paths"); + this.sources = ModelValidation.immutableList(sources, "sources"); + this.stageScores = ModelValidation.sortedMap(stageScores); + for (Map.Entry entry : this.stageScores.entrySet()) { + if (entry.getValue() == null) { + throw new NullPointerException("stage score must not be null"); + } + if (!entry.getKey().equals(entry.getValue().getChannel())) { + throw new RetrievalModelValidationException("stage score key must match channel"); + } + } + this.fusedScore = ModelValidation.optionalScore(fusedScore, "fusedScore"); + this.rank = ModelValidation.optionalRank(rank, "rank"); + } + + public String getEvidenceId() { + return evidenceId; + } + + public EvidenceKind getKind() { + return kind; + } + + public String getText() { + return text; + } + + public List getChunks() { + return Collections.unmodifiableList(chunks == null ? Collections.emptyList() : chunks); + } + + public List getEntities() { + return Collections.unmodifiableList(entities == null ? Collections.emptyList() : entities); + } + + public List getPaths() { + return Collections.unmodifiableList(paths == null ? Collections.emptyList() : paths); + } + + public List getSources() { + return Collections.unmodifiableList(sources == null ? Collections.emptyList() : sources); + } + + public Map getStageScores() { + return Collections.unmodifiableMap(stageScores == null + ? Collections.emptyMap() : stageScores); + } + + public Double getFusedScore() { + return fusedScore; + } + + public Integer getRank() { + return rank; + } + + public boolean sameIdentityAs(Evidence other) { + return this == other || other != null && kind == other.kind + && sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs) + && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs) + && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs) + && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs); + } + + private static boolean sameMultiset(List first, List second, + BiPredicate identity) { + if (first == null || second == null || first.size() != second.size()) { + return false; + } + boolean[] matched = new boolean[second.size()]; + for (T value : first) { + boolean found = false; + for (int i = 0; i < second.size(); i++) { + if (!matched[i] && identity.test(value, second.get(i))) { + matched[i] = true; + found = true; + break; + } + } + if (!found) { + return false; + } + } + return true; + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof Evidence)) { + return false; + } + Evidence that = (Evidence) object; + return Objects.equals(evidenceId, that.evidenceId) && kind == that.kind + && Objects.equals(text, that.text) && Objects.equals(chunks, that.chunks) + && Objects.equals(entities, that.entities) && Objects.equals(paths, that.paths) + && Objects.equals(sources, that.sources) + && Objects.equals(stageScores, that.stageScores) + && Objects.equals(fusedScore, that.fusedScore) + && Objects.equals(rank, that.rank); + } + + @Override + public int hashCode() { + return Objects.hash(evidenceId, kind, text, chunks, entities, paths, sources, + stageScores, fusedScore, rank); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java new file mode 100644 index 000000000..359536970 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/EvidenceKind.java @@ -0,0 +1,29 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.evidence; + + +/** Candidate payload categories understood by the retrieval response model. */ +public enum EvidenceKind { + CHUNK, + ENTITY, + GRAPH_PATH, + GRAPH_FACT +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java new file mode 100644 index 000000000..94f3d26c0 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/EntityRef.java @@ -0,0 +1,110 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.graph; + +import com.google.gson.annotations.SerializedName; +import java.util.Collections; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable graph entity reference with optional aliases and source chunk provenance. */ +public final class EntityRef { + + @SerializedName("entityId") + private final String entityId; + @SerializedName("canonicalName") + private final String canonicalName; + @SerializedName("aliases") + private final java.util.List aliases; + @SerializedName("type") + private final String type; + @SerializedName("sourceChunkIds") + private final java.util.List sourceChunkIds; + + private EntityRef() { + this.entityId = null; + this.canonicalName = null; + this.aliases = java.util.Collections.emptyList(); + this.type = null; + this.sourceChunkIds = java.util.Collections.emptyList(); + } + + public EntityRef(String entityId, String canonicalName, String type) { + this(entityId, canonicalName, java.util.Collections.emptyList(), type, + java.util.Collections.emptyList()); + } + + public EntityRef(String entityId, String canonicalName, java.util.List aliases, + String type, java.util.List sourceChunkIds) { + this.entityId = ModelValidation.required(entityId, "entityId"); + this.canonicalName = ModelValidation.required(canonicalName, "canonicalName"); + this.aliases = ModelValidation.sortedStrings(aliases, "alias"); + this.type = ModelValidation.required(type, "type"); + this.sourceChunkIds = ModelValidation.sortedStrings(sourceChunkIds, "sourceChunkId"); + } + + public String getEntityId() { + return entityId; + } + + public String getCanonicalName() { + return canonicalName; + } + + public java.util.List getAliases() { + return Collections.unmodifiableList(aliases == null ? Collections.emptyList() : aliases); + } + + public String getType() { + return type; + } + + public java.util.List getSourceChunkIds() { + return Collections.unmodifiableList(sourceChunkIds == null + ? Collections.emptyList() : sourceChunkIds); + } + + public boolean sameIdentityAs(EntityRef other) { + return other != null && Objects.equals(entityId, other.entityId) + && Objects.equals(canonicalName, other.canonicalName) + && Objects.equals(type, other.type); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof EntityRef)) { + return false; + } + EntityRef that = (EntityRef) object; + return Objects.equals(entityId, that.entityId) + && Objects.equals(canonicalName, that.canonicalName) + && Objects.equals(aliases, that.aliases) + && Objects.equals(type, that.type) + && Objects.equals(sourceChunkIds, that.sourceChunkIds); + } + + @Override + public int hashCode() { + return Objects.hash(entityId, canonicalName, aliases, type, sourceChunkIds); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java new file mode 100644 index 000000000..b53a69b81 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphEdgeRef.java @@ -0,0 +1,111 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.graph; + +import com.google.gson.annotations.SerializedName; +import java.util.Collections; +import java.util.List; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable graph edge reference connecting two entity identifiers. */ +public final class GraphEdgeRef { + + @SerializedName("edgeId") + private final String edgeId; + @SerializedName("label") + private final String label; + @SerializedName("sourceEntityId") + private final String sourceEntityId; + @SerializedName("targetEntityId") + private final String targetEntityId; + @SerializedName("sourceChunkIds") + private final List sourceChunkIds; + + private GraphEdgeRef() { + this.edgeId = null; + this.label = null; + this.sourceEntityId = null; + this.targetEntityId = null; + this.sourceChunkIds = Collections.emptyList(); + } + + public GraphEdgeRef(String edgeId, String label, String sourceEntityId, String targetEntityId) { + this(edgeId, label, sourceEntityId, targetEntityId, java.util.Collections.emptyList()); + } + + public GraphEdgeRef(String edgeId, String label, String sourceEntityId, + String targetEntityId, List sourceChunkIds) { + this.edgeId = ModelValidation.required(edgeId, "edgeId"); + this.label = ModelValidation.required(label, "label"); + this.sourceEntityId = ModelValidation.required(sourceEntityId, "sourceEntityId"); + this.targetEntityId = ModelValidation.required(targetEntityId, "targetEntityId"); + this.sourceChunkIds = ModelValidation.sortedStrings(sourceChunkIds, "sourceChunkId"); + } + + public String getEdgeId() { + return edgeId; + } + + public String getLabel() { + return label; + } + + public String getSourceEntityId() { + return sourceEntityId; + } + + public String getTargetEntityId() { + return targetEntityId; + } + + public List getSourceChunkIds() { + return Collections.unmodifiableList(sourceChunkIds == null + ? Collections.emptyList() : sourceChunkIds); + } + + public boolean sameIdentityAs(GraphEdgeRef other) { + return other != null && Objects.equals(edgeId, other.edgeId) + && Objects.equals(label, other.label) + && Objects.equals(sourceEntityId, other.sourceEntityId) + && Objects.equals(targetEntityId, other.targetEntityId); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof GraphEdgeRef)) { + return false; + } + GraphEdgeRef that = (GraphEdgeRef) object; + return Objects.equals(edgeId, that.edgeId) + && Objects.equals(label, that.label) + && Objects.equals(sourceEntityId, that.sourceEntityId) + && Objects.equals(targetEntityId, that.targetEntityId) + && Objects.equals(sourceChunkIds, that.sourceChunkIds); + } + + @Override + public int hashCode() { + return Objects.hash(edgeId, label, sourceEntityId, targetEntityId, sourceChunkIds); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java new file mode 100644 index 000000000..f2c5e46ec --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphPathRef.java @@ -0,0 +1,103 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.graph; + +import com.google.gson.annotations.SerializedName; +import java.util.Collections; +import java.util.List; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; + +/** Immutable ordered graph path with parallel vertex and edge identifier sequences. */ +public final class GraphPathRef { + + @SerializedName("vertexIds") + private final List vertexIds; + @SerializedName("edgeIds") + private final List edgeIds; + @SerializedName("hop") + private final int hop; + @SerializedName("sampled") + private final boolean sampled; + + private GraphPathRef() { + this.vertexIds = Collections.emptyList(); + this.edgeIds = Collections.emptyList(); + this.hop = 0; + this.sampled = false; + } + + public GraphPathRef(List vertexIds, List edgeIds, int hop, boolean sampled) { + this.vertexIds = ModelValidation.immutableList(vertexIds, "vertexIds"); + this.edgeIds = ModelValidation.immutableList(edgeIds, "edgeIds"); + this.hop = ModelValidation.nonNegative(hop, "hop"); + this.sampled = sampled; + if (this.vertexIds.size() != hop + 1 || this.edgeIds.size() != hop) { + throw new RetrievalModelValidationException( + "path vertex/edge counts must match hop"); + } + for (String vertexId : this.vertexIds) { + ModelValidation.required(vertexId, "vertexId"); + } + for (String edgeId : this.edgeIds) { + ModelValidation.required(edgeId, "edgeId"); + } + } + + public List getVertexIds() { + return Collections.unmodifiableList(vertexIds == null ? Collections.emptyList() : vertexIds); + } + + public List getEdgeIds() { + return Collections.unmodifiableList(edgeIds == null ? Collections.emptyList() : edgeIds); + } + + public int getHop() { + return hop; + } + + public boolean isSampled() { + return sampled; + } + + public boolean sameIdentityAs(GraphPathRef other) { + return equals(other); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof GraphPathRef)) { + return false; + } + GraphPathRef that = (GraphPathRef) object; + return hop == that.hop && sampled == that.sampled + && Objects.equals(vertexIds, that.vertexIds) + && Objects.equals(edgeIds, that.edgeIds); + } + + @Override + public int hashCode() { + return Objects.hash(vertexIds, edgeIds, hop, sampled); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java new file mode 100644 index 000000000..5f5649c30 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/graph/GraphVertexRef.java @@ -0,0 +1,78 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.graph; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable graph vertex reference associated with an entity identifier. */ +public final class GraphVertexRef { + + @SerializedName("label") + private final String label; + @SerializedName("vertexId") + private final String vertexId; + @SerializedName("entityId") + private final String entityId; + + public GraphVertexRef(String label, String vertexId, String entityId) { + this.label = ModelValidation.required(label, "label"); + this.vertexId = ModelValidation.required(vertexId, "vertexId"); + this.entityId = ModelValidation.required(entityId, "entityId"); + } + + public String getLabel() { + return label; + } + + public String getVertexId() { + return vertexId; + } + + public String getEntityId() { + return entityId; + } + + public boolean sameIdentityAs(GraphVertexRef other) { + return other != null && Objects.equals(label, other.label) + && Objects.equals(vertexId, other.vertexId) + && Objects.equals(entityId, other.entityId); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof GraphVertexRef)) { + return false; + } + GraphVertexRef that = (GraphVertexRef) object; + return Objects.equals(label, that.label) + && Objects.equals(vertexId, that.vertexId) + && Objects.equals(entityId, that.entityId); + } + + @Override + public int hashCode() { + return Objects.hash(label, vertexId, entityId); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java new file mode 100644 index 000000000..54dbdf420 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java @@ -0,0 +1,49 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +/** + * Immutable value objects shared by GraphRAG ingestion, retrieval, and evaluation. + * + *

IDs in these objects are opaque values supplied by callers. This package does not generate + * IDs. Use {@code sameIdentityAs} for identity comparisons and {@code equals} when complete value + * equality is required. Evidence identity excludes evidenceId, text, scores and ranks so + * multi-channel candidates can be merged; its nested references are compared recursively.

+ * + *

JSON uses explicit camelCase field names. Optional fields may be added only additively; unknown + * fields are ignored, null optional scalars are omitted, and collection fields use empty arrays or + * objects. Callers must use {@link RetrievalModelJson} for deserialization so required-field and + * range validation cannot be bypassed by reflection. Text offsets are half-open + * ({@code startOffset <= endOffset}); ranks are one-based, raw scores are finite, and normalized + * and fused scores are finite values in {@code [0, 1]}.

+ * + *

Identity fields are documentId/dataset/datasetVersion/split/sourceUri/sourceHash for + * {@code SourceDocument}; chunkId/documentId/chunkIndex/startOffset/endOffset/text/policyVersion/ + * textHash for {@code TextChunk}; entityId/canonicalName/type for {@code EntityRef}; all fields + * for {@code GraphVertexRef}, {@code GraphVersion}, {@code IndexVersion}, {@code SourceRef}, + * {@code GraphPathRef}, and {@code ChannelScore}; and edgeId/label/sourceEntityId/targetEntityId + * for {@code GraphEdgeRef}. Evidence compares kind and nested reference identities. Collection + * order does not affect Evidence identity, while vertex and edge order inside one graph path does. + * A graph path has exactly {@code hop + 1} vertices and {@code hop} edges. Evidence kind labels + * the candidate payload (chunk, entity, path, or graph fact), while optional lists preserve the + * supporting provenance and may be empty for compatibility.

+ */ +package org.apache.geaflow.ai.retrieval.model; + +import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson; + diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java new file mode 100644 index 000000000..26f6003ff --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/GraphVersion.java @@ -0,0 +1,69 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.version; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable name/version pair identifying a graph snapshot. */ +public final class GraphVersion { + + @SerializedName("graphName") + private final String graphName; + @SerializedName("version") + private final String version; + + public GraphVersion(String graphName, String version) { + this.graphName = ModelValidation.required(graphName, "graphName"); + this.version = ModelValidation.required(version, "version"); + } + + public String getGraphName() { + return graphName; + } + + public String getVersion() { + return version; + } + + public boolean sameIdentityAs(GraphVersion other) { + return other != null && Objects.equals(graphName, other.graphName) + && Objects.equals(version, other.version); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof GraphVersion)) { + return false; + } + GraphVersion that = (GraphVersion) object; + return Objects.equals(graphName, that.graphName) + && Objects.equals(version, that.version); + } + + @Override + public int hashCode() { + return Objects.hash(graphName, version); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java new file mode 100644 index 000000000..1e1c791b7 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/version/IndexVersion.java @@ -0,0 +1,78 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model.version; + +import com.google.gson.annotations.SerializedName; +import java.util.Objects; +import org.apache.geaflow.ai.retrieval.validation.ModelValidation; + +/** Immutable index snapshot identifier together with its source graph version. */ +public final class IndexVersion { + + @SerializedName("indexName") + private final String indexName; + @SerializedName("version") + private final String version; + @SerializedName("graphVersion") + private final String graphVersion; + + public IndexVersion(String indexName, String version, String graphVersion) { + this.indexName = ModelValidation.required(indexName, "indexName"); + this.version = ModelValidation.required(version, "version"); + this.graphVersion = ModelValidation.required(graphVersion, "graphVersion"); + } + + public String getIndexName() { + return indexName; + } + + public String getVersion() { + return version; + } + + public String getGraphVersion() { + return graphVersion; + } + + public boolean sameIdentityAs(IndexVersion other) { + return other != null && Objects.equals(indexName, other.indexName) + && Objects.equals(version, other.version) + && Objects.equals(graphVersion, other.graphVersion); + } + + @Override + public boolean equals(Object object) { + if (this == object) { + return true; + } + if (!(object instanceof IndexVersion)) { + return false; + } + IndexVersion that = (IndexVersion) object; + return Objects.equals(indexName, that.indexName) + && Objects.equals(version, that.version) + && Objects.equals(graphVersion, that.graphVersion); + } + + @Override + public int hashCode() { + return Objects.hash(indexName, version, graphVersion); + } +} From 416c25b0073e6e1dc5731a0da2ef3efab2a25a1b Mon Sep 17 00:00:00 2001 From: aotenjou Date: Sun, 6 Sep 2026 11:22:53 +0800 Subject: [PATCH 2/5] feat(ai): add retrieval model codec and validation --- .../retrieval/codec/RetrievalModelJson.java | 283 ++++++++++++++++++ .../retrieval/validation/ModelValidation.java | 126 ++++++++ .../RetrievalModelValidationException.java | 32 ++ 3 files changed, 441 insertions(+) create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java create mode 100644 geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java new file mode 100644 index 000000000..dea798bf6 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/codec/RetrievalModelJson.java @@ -0,0 +1,283 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.codec; + +import com.google.gson.Gson; +import com.google.gson.JsonArray; +import com.google.gson.JsonElement; +import com.google.gson.JsonObject; +import com.google.gson.JsonParseException; +import com.google.gson.JsonParser; +import java.util.ArrayList; +import java.util.Collections; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import org.apache.geaflow.ai.retrieval.model.document.SourceDocument; +import org.apache.geaflow.ai.retrieval.model.document.SourceRef; +import org.apache.geaflow.ai.retrieval.model.document.TextChunk; +import org.apache.geaflow.ai.retrieval.model.evidence.ChannelScore; +import org.apache.geaflow.ai.retrieval.model.evidence.Evidence; +import org.apache.geaflow.ai.retrieval.model.evidence.EvidenceKind; +import org.apache.geaflow.ai.retrieval.model.graph.EntityRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphEdgeRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphVertexRef; +import org.apache.geaflow.ai.retrieval.model.version.GraphVersion; +import org.apache.geaflow.ai.retrieval.model.version.IndexVersion; + +/** + * Validated JSON boundary for retrieval domain models. + * + *

The codec keeps deserialization on the public constructors so required fields, ranges, and + * immutable collection guarantees are applied consistently. Unknown properties are ignored for + * additive wire compatibility; callers should use this class instead of Gson directly.

+ */ +public final class RetrievalModelJson { + + private static final Gson GSON = new Gson(); + + private RetrievalModelJson() { + } + + public static String toJson(Object value) { + return GSON.toJson(value); + } + + public static T fromJson(String json, Class type) { + if (json == null) { + throw new NullPointerException("json"); + } + if (type == null) { + throw new NullPointerException("type"); + } + JsonElement element = new JsonParser().parse(json); + if (!element.isJsonObject()) { + throw new JsonParseException("model JSON must be an object"); + } + try { + Object value = parse(element.getAsJsonObject(), type); + return type.cast(value); + } catch (JsonParseException e) { + throw e; + } catch (RuntimeException e) { + throw new JsonParseException("invalid " + type.getSimpleName() + ": " + + e.getMessage(), e); + } + } + + private static Object parse(JsonObject object, Class type) { + if (type == SourceDocument.class) { + return new SourceDocument(requiredString(object, "documentId"), + requiredString(object, "dataset"), requiredString(object, "datasetVersion"), + requiredString(object, "split"), optionalString(object, "title"), + requiredString(object, "sourceUri"), requiredString(object, "sourceHash")); + } else if (type == TextChunk.class) { + return new TextChunk(requiredString(object, "chunkId"), + requiredString(object, "documentId"), requiredInt(object, "chunkIndex"), + requiredInt(object, "startOffset"), requiredInt(object, "endOffset"), + requiredInt(object, "tokenEstimate"), requiredString(object, "text"), + optionalString(object, "policyVersion"), optionalString(object, "textHash")); + } else if (type == EntityRef.class) { + return new EntityRef(requiredString(object, "entityId"), + requiredString(object, "canonicalName"), strings(object, "aliases"), + requiredString(object, "type"), strings(object, "sourceChunkIds")); + } else if (type == GraphVertexRef.class) { + return new GraphVertexRef(requiredString(object, "label"), + requiredString(object, "vertexId"), requiredString(object, "entityId")); + } else if (type == GraphEdgeRef.class) { + return new GraphEdgeRef(requiredString(object, "edgeId"), + requiredString(object, "label"), requiredString(object, "sourceEntityId"), + requiredString(object, "targetEntityId"), + strings(object, "sourceChunkIds")); + } else if (type == GraphPathRef.class) { + return new GraphPathRef(strings(object, "vertexIds"), + strings(object, "edgeIds"), requiredInt(object, "hop"), + requiredBoolean(object, "sampled")); + } else if (type == SourceRef.class) { + return new SourceRef(requiredString(object, "documentId"), + requiredString(object, "sourceUri"), optionalInt(object, "startOffset"), + optionalInt(object, "endOffset")); + } else if (type == ChannelScore.class) { + return new ChannelScore(requiredString(object, "channel"), + requiredDouble(object, "rawScore"), optionalDouble(object, "normalizedScore"), + requiredInt(object, "rank")); + } else if (type == GraphVersion.class) { + return new GraphVersion(requiredString(object, "graphName"), + requiredString(object, "version")); + } else if (type == IndexVersion.class) { + return new IndexVersion(requiredString(object, "indexName"), + requiredString(object, "version"), requiredString(object, "graphVersion")); + } else if (type == Evidence.class) { + return new Evidence(optionalString(object, "evidenceId"), kind(object), + optionalString(object, "text"), models(object, "chunks", TextChunk.class), + models(object, "entities", EntityRef.class), models(object, "paths", GraphPathRef.class), + models(object, "sources", SourceRef.class), scores(object), + optionalDouble(object, "fusedScore"), optionalInt(object, "rank")); + } + throw new JsonParseException("unsupported retrieval model: " + type.getName()); + } + + private static EvidenceKind kind(JsonObject object) { + String value = requiredString(object, "kind"); + try { + return EvidenceKind.valueOf(value); + } catch (IllegalArgumentException e) { + throw new JsonParseException("unknown evidence kind: " + value); + } + } + + private static Map scores(JsonObject object) { + JsonElement element = object.get("stageScores"); + if (element == null || element.isJsonNull()) { + return Collections.emptyMap(); + } + if (!element.isJsonObject()) { + throw new JsonParseException("stageScores must be an object"); + } + Map result = new LinkedHashMap<>(); + for (Map.Entry entry : element.getAsJsonObject().entrySet()) { + if (!entry.getValue().isJsonObject()) { + throw new JsonParseException("stage score must be an object"); + } + result.put(entry.getKey(), fromElement(entry.getValue(), ChannelScore.class)); + } + return result; + } + + private static List models(JsonObject object, String name, Class type) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull()) { + return Collections.emptyList(); + } + if (!element.isJsonArray()) { + throw new JsonParseException(name + " must be an array"); + } + List result = new ArrayList<>(); + for (JsonElement item : element.getAsJsonArray()) { + if (!item.isJsonObject()) { + throw new JsonParseException(name + " elements must be objects"); + } + result.add(fromElement(item, type)); + } + return result; + } + + private static List strings(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull()) { + return Collections.emptyList(); + } + if (!element.isJsonArray()) { + throw new JsonParseException(name + " must be an array"); + } + List result = new ArrayList<>(); + JsonArray array = element.getAsJsonArray(); + for (JsonElement item : array) { + if (item.isJsonNull() || !item.isJsonPrimitive() + || !item.getAsJsonPrimitive().isString()) { + throw new JsonParseException(name + " must contain strings"); + } + result.add(item.getAsString()); + } + return result; + } + + private static T fromElement(JsonElement element, Class type) { + return type.cast(parse(element.getAsJsonObject(), type)); + } + + private static String requiredString(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull() || !element.isJsonPrimitive() + || !element.getAsJsonPrimitive().isString()) { + throw new JsonParseException(name + " is required"); + } + return element.getAsString(); + } + + private static String optionalString(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull()) { + return null; + } + if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isString()) { + throw new JsonParseException(name + " must be a string"); + } + return element.getAsString(); + } + + private static int requiredInt(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull() || !element.isJsonPrimitive() + || !element.getAsJsonPrimitive().isNumber()) { + throw new JsonParseException(name + " is required"); + } + int value = element.getAsInt(); + if (element.getAsDouble() != value) { + throw new JsonParseException(name + " must be an integer"); + } + return value; + } + + private static Integer optionalInt(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull()) { + return null; + } + if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isNumber()) { + throw new JsonParseException(name + " must be a number"); + } + int value = element.getAsInt(); + if (element.getAsDouble() != value) { + throw new JsonParseException(name + " must be an integer"); + } + return value; + } + + private static double requiredDouble(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull() || !element.isJsonPrimitive() + || !element.getAsJsonPrimitive().isNumber()) { + throw new JsonParseException(name + " is required"); + } + return element.getAsDouble(); + } + + private static Double optionalDouble(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull()) { + return null; + } + if (!element.isJsonPrimitive() || !element.getAsJsonPrimitive().isNumber()) { + throw new JsonParseException(name + " must be a number"); + } + return element.getAsDouble(); + } + + private static boolean requiredBoolean(JsonObject object, String name) { + JsonElement element = object.get(name); + if (element == null || element.isJsonNull() || !element.isJsonPrimitive() + || !element.getAsJsonPrimitive().isBoolean()) { + throw new JsonParseException(name + " is required"); + } + return element.getAsBoolean(); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java new file mode 100644 index 000000000..b4176be4a --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/ModelValidation.java @@ -0,0 +1,126 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.validation; + +import java.util.ArrayList; +import java.util.Collections; +import java.util.Comparator; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.Objects; + +/** Shared constructor validation and defensive-copy helpers for retrieval models. */ +public final class ModelValidation { + + private ModelValidation() { + } + + public static String required(String value, String name) { + Objects.requireNonNull(value, name); + if (value.trim().isEmpty()) { + throw new RetrievalModelValidationException(name + " must not be blank"); + } + return value; + } + + public static String optional(String value) { + return value == null ? null : value; + } + + public static String optionalNonBlank(String value, String name) { + if (value == null) { + return null; + } + return required(value, name); + } + + public static int nonNegative(int value, String name) { + if (value < 0) { + throw new RetrievalModelValidationException(name + " must be non-negative"); + } + return value; + } + + public static double finite(double value, String name) { + if (Double.isNaN(value) || Double.isInfinite(value)) { + throw new RetrievalModelValidationException(name + " must be finite"); + } + return value; + } + + public static Double optionalScore(Double value, String name) { + if (value == null) { + return null; + } + finite(value, name); + if (value < 0.0 || value > 1.0) { + throw new RetrievalModelValidationException(name + " must be in [0, 1]"); + } + return value; + } + + public static Integer optionalRank(Integer value, String name) { + if (value != null && value < 1) { + throw new RetrievalModelValidationException(name + " must be at least 1"); + } + return value; + } + + public static List immutableList(List values, String name) { + if (values == null || values.isEmpty()) { + return Collections.emptyList(); + } + List copy = new ArrayList<>(values); + for (T value : copy) { + Objects.requireNonNull(value, name + " element"); + } + return Collections.unmodifiableList(copy); + } + + public static List sortedStrings(List values, String name) { + List result = immutableList(values, name); + if (result.isEmpty()) { + return result; + } + List sorted = new ArrayList<>(result.size()); + for (String value : result) { + sorted.add(required(value, name)); + } + sorted.sort(Comparator.naturalOrder()); + return Collections.unmodifiableList(sorted); + } + + public static Map sortedMap(Map values) { + if (values == null || values.isEmpty()) { + return Collections.emptyMap(); + } + List keys = new ArrayList<>(values.keySet()); + for (String key : keys) { + required(key, "map key"); + } + keys.sort(Comparator.naturalOrder()); + Map sorted = new LinkedHashMap<>(); + for (String key : keys) { + sorted.put(key, values.get(key)); + } + return Collections.unmodifiableMap(sorted); + } +} diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java new file mode 100644 index 000000000..e09c381d4 --- /dev/null +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/validation/RetrievalModelValidationException.java @@ -0,0 +1,32 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.validation; + +/** Indicates invalid data at the retrieval model boundary. */ +public class RetrievalModelValidationException extends IllegalArgumentException { + + public RetrievalModelValidationException(String message) { + super(message); + } + + public RetrievalModelValidationException(String message, Throwable cause) { + super(message, cause); + } +} From d4dc437a1a0d97ec5bde95245d612c560362ebe4 Mon Sep 17 00:00:00 2001 From: aotenjou Date: Sun, 6 Sep 2026 11:23:03 +0800 Subject: [PATCH 3/5] test(ai): cover retrieval domain models --- .../model/RetrievalDomainModelTest.java | 381 ++++++++++++++++++ .../resources/retrieval/model/evidence.json | 1 + .../retrieval/model/identity-cases.json | 10 + .../retrieval/model/source-document.json | 1 + 4 files changed, 393 insertions(+) create mode 100644 geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java create mode 100644 geaflow-ai/src/test/resources/retrieval/model/evidence.json create mode 100644 geaflow-ai/src/test/resources/retrieval/model/identity-cases.json create mode 100644 geaflow-ai/src/test/resources/retrieval/model/source-document.json diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java new file mode 100644 index 000000000..1d56475ee --- /dev/null +++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java @@ -0,0 +1,381 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.geaflow.ai.retrieval.model; + +import com.google.gson.Gson; +import com.google.gson.JsonParseException; +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.util.Arrays; +import java.util.Collections; +import java.util.HashMap; +import java.util.Map; +import java.util.stream.Collectors; +import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson; +import org.apache.geaflow.ai.retrieval.model.document.SourceDocument; +import org.apache.geaflow.ai.retrieval.model.document.SourceRef; +import org.apache.geaflow.ai.retrieval.model.document.TextChunk; +import org.apache.geaflow.ai.retrieval.model.evidence.ChannelScore; +import org.apache.geaflow.ai.retrieval.model.evidence.Evidence; +import org.apache.geaflow.ai.retrieval.model.evidence.EvidenceKind; +import org.apache.geaflow.ai.retrieval.model.graph.EntityRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphEdgeRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphPathRef; +import org.apache.geaflow.ai.retrieval.model.graph.GraphVertexRef; +import org.apache.geaflow.ai.retrieval.model.version.GraphVersion; +import org.apache.geaflow.ai.retrieval.model.version.IndexVersion; +import org.apache.geaflow.ai.retrieval.validation.RetrievalModelValidationException; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** + * Regression coverage for retrieval domain models, validation rules, and JSON round-tripping. + */ +public class RetrievalDomainModelTest { + + private static final Gson GSON = new Gson(); + + @Test + public void modelsAreValueObjectsAndDefensivelyCopyCollections() { + java.util.List aliases = Arrays.asList("Kong Fuzi", "Master Kong"); + EntityRef entity = new EntityRef("entity-1", "Confucius", aliases, + "PERSON", Collections.singletonList("chunk-1")); + + Assertions.assertEquals(Arrays.asList("Kong Fuzi", "Master Kong"), entity.getAliases()); + Assertions.assertThrows(UnsupportedOperationException.class, + () -> entity.getAliases().add("孔子")); + Assertions.assertEquals(entity, new EntityRef("entity-1", "Confucius", + Arrays.asList("Master Kong", "Kong Fuzi"), "PERSON", + Collections.singletonList("chunk-1"))); + Assertions.assertTrue(entity.sameIdentityAs(new EntityRef("entity-1", "Confucius", + Collections.emptyList(), "PERSON", Collections.emptyList()))); + + SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev", + "Title", "file:///doc", "hash"); + Assertions.assertTrue(document.sameIdentityAs(new SourceDocument("doc-1", "dataset", + "v1", "dev", "Other title", "file:///doc", "hash"))); + + TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, "text", + "policy-v1", "hash"); + Assertions.assertTrue(chunk.sameIdentityAs(new TextChunk("chunk-1", "doc-1", 0, + 0, 4, 99, "text", "policy-v1", "hash"))); + + GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1"); + Assertions.assertFalse(vertex.sameIdentityAs(new GraphVertexRef("company", "vertex-1", + "entity-1"))); + GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2", + Collections.singletonList("chunk-1")); + Assertions.assertTrue(edge.sameIdentityAs(new GraphEdgeRef("edge-1", "knows", + "entity-1", "entity-2", Collections.singletonList("chunk-2")))); + + Assertions.assertTrue(new GraphVersion("graph", "v1") + .sameIdentityAs(new GraphVersion("graph", "v1"))); + Assertions.assertFalse(new IndexVersion("bm25", "v1", "graph-v1") + .sameIdentityAs(new IndexVersion("vector", "v1", "graph-v1"))); + } + + @Test + public void validatesRequiredFieldsAndRanges() { + Assertions.assertThrows(NullPointerException.class, + () -> new GraphVersion(null, "v1")); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new TextChunk("chunk-1", "doc-1", 0, 8, 2, 1, "text", null, null)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new ChannelScore("bm25", 1.0, 0.5, 0)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new SourceRef("doc-1", "file:///doc", 4, 2)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new Evidence(" ", EvidenceKind.CHUNK, null, null, null, null, null, null, + null, null)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new TextChunk("chunk-1", "doc-1", 0, 0, 1, 1, "text", " ", null)); + } + + @Test + public void evidenceMergesChannelsByFieldsNotScores() { + TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, + "text", "policy-v1", "hash"); + SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4); + GraphPathRef path = new GraphPathRef(Collections.singletonList("v1"), + Collections.emptyList(), 0, false); + Map bm25 = new HashMap<>(); + bm25.put("bm25", new ChannelScore("bm25", 2.0, 0.8, 1)); + Map vector = new HashMap<>(); + vector.put("vector", new ChannelScore("vector", 0.9, 0.7, 3)); + + Evidence first = new Evidence("evidence-1", EvidenceKind.CHUNK, "text", + Collections.singletonList(chunk), Collections.emptyList(), + Collections.singletonList(path), Collections.singletonList(source), bm25, 0.8, 1); + Evidence second = new Evidence("evidence-2", EvidenceKind.CHUNK, "text", + Collections.singletonList(chunk), Collections.emptyList(), + Collections.singletonList(path), Collections.singletonList(source), vector, 0.7, 3); + + Assertions.assertNotEquals(first, second); + Assertions.assertTrue(first.sameIdentityAs(second)); + Assertions.assertEquals(0.8, first.getStageScores().get("bm25").getNormalizedScore()); + Assertions.assertEquals(0.7, second.getStageScores().get("vector").getNormalizedScore()); + String json = GSON.toJson(first); + Assertions.assertTrue(json.contains("\"chunks\"")); + Assertions.assertTrue(json.contains("\"stageScores\"")); + Assertions.assertEquals(first, RetrievalModelJson.fromJson(json, Evidence.class)); + } + + @Test + public void evidenceIdentityUsesNestedIdentityAndIgnoresCollectionOrder() { + TextChunk firstChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, + "text", "policy-v1", "hash"); + TextChunk secondChunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 99, + "text", "policy-v1", "hash"); + EntityRef firstEntity = new EntityRef("entity-1", "Name", + Collections.singletonList("alias-a"), "PERSON", Collections.singletonList("chunk-1")); + EntityRef secondEntity = new EntityRef("entity-1", "Name", + Collections.singletonList("alias-b"), "PERSON", Collections.singletonList("chunk-2")); + SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4); + SourceRef secondSource = new SourceRef("doc-2", "file:///doc-2", 0, 4); + GraphPathRef path = new GraphPathRef(Arrays.asList("v1", "v2"), + Collections.singletonList("e1"), 1, false); + GraphPathRef secondPath = new GraphPathRef(Arrays.asList("v2", "v3"), + Collections.singletonList("e2"), 1, false); + + Evidence first = new Evidence("e-1", EvidenceKind.ENTITY, "first", + Collections.singletonList(firstChunk), Collections.singletonList(firstEntity), + Arrays.asList(path, secondPath), Arrays.asList(source, secondSource), + Collections.emptyMap(), 0.1, 1); + Evidence second = new Evidence("e-2", EvidenceKind.ENTITY, "second", + Collections.singletonList(secondChunk), Collections.singletonList(secondEntity), + Arrays.asList(secondPath, path), Arrays.asList(secondSource, source), + Collections.emptyMap(), 0.9, 2); + + Assertions.assertTrue(first.sameIdentityAs(second)); + Evidence reordered = new Evidence("e-3", EvidenceKind.ENTITY, null, + Collections.singletonList(secondChunk), Collections.singletonList(secondEntity), + Arrays.asList(secondPath, path), Arrays.asList(secondSource, source), + Collections.emptyMap(), null, null); + Assertions.assertTrue(first.sameIdentityAs(reordered)); + } + + @Test + public void gsonRoundTripPreservesFieldsAndOptionalCompatibility() throws IOException { + SourceDocument document = new SourceDocument("doc-1", "hotpotqa", "v1", + "dev", "Title", "https://example/doc-1", "sha256"); + String json = GSON.toJson(document); + SourceDocument restored = RetrievalModelJson.fromJson(json, SourceDocument.class); + Assertions.assertEquals(document, restored); + Assertions.assertEquals(readFixture("retrieval/model/source-document.json"), json); + + String oldJson = "{\"documentId\":\"doc-1\",\"dataset\":\"hotpotqa\"," + + "\"datasetVersion\":\"v1\",\"split\":\"dev\"," + + "\"sourceUri\":\"https://example/doc-1\",\"sourceHash\":\"sha256\"}"; + SourceDocument withoutTitle = RetrievalModelJson.fromJson(oldJson, SourceDocument.class); + Assertions.assertNull(withoutTitle.getTitle()); + Assertions.assertEquals(document.getDocumentId(), withoutTitle.getDocumentId()); + + String oldEvidenceJson = "{\"kind\":\"CHUNK\",\"evidenceId\":\"e-1\"," + + "\"text\":\"text\"}"; + Evidence oldEvidence = RetrievalModelJson.fromJson(oldEvidenceJson, Evidence.class); + Assertions.assertNotNull(oldEvidence.getChunks()); + Assertions.assertNotNull(oldEvidence.getEntities()); + Assertions.assertNotNull(oldEvidence.getPaths()); + Assertions.assertNotNull(oldEvidence.getSources()); + Assertions.assertNotNull(oldEvidence.getStageScores()); + Assertions.assertTrue(oldEvidence.getChunks().isEmpty()); + Assertions.assertTrue(GSON.toJson(oldEvidence).contains("\"kind\":\"CHUNK\"")); + } + + @Test + public void validatedJsonRejectsMissingRequiredFieldsAndKeepsEmptyCollections() { + Assertions.assertThrows(JsonParseException.class, + () -> RetrievalModelJson.fromJson("{\"version\":\"v1\"}", GraphVersion.class)); + Assertions.assertThrows(JsonParseException.class, + () -> RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\",\"evidenceId\":\" \"}", + Evidence.class)); + + Evidence empty = RetrievalModelJson.fromJson("{\"kind\":\"CHUNK\"," + + "\"chunks\":null,\"entities\":[],\"paths\":null,\"sources\":null," + + "\"stageScores\":null}", Evidence.class); + String json = RetrievalModelJson.toJson(empty); + Assertions.assertTrue(json.contains("\"chunks\":[]")); + Assertions.assertTrue(json.contains("\"entities\":[]")); + Assertions.assertTrue(json.contains("\"paths\":[]")); + Assertions.assertTrue(json.contains("\"sources\":[]")); + Assertions.assertTrue(json.contains("\"stageScores\":{}")); + } + + @Test + public void validatedJsonRejectsWrongElementAndNumericTypes() { + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"entityId\":\"e\",\"canonicalName\":\"n\"," + + "\"aliases\":[1],\"type\":\"PERSON\",\"sourceChunkIds\":[]}", + EntityRef.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"entityId\":\"e\",\"canonicalName\":\"n\"," + + "\"aliases\":[true],\"type\":\"PERSON\",\"sourceChunkIds\":[]}", + EntityRef.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"entityId\":\"e\",\"canonicalName\":\"n\"," + + "\"aliases\":[null],\"type\":\"PERSON\",\"sourceChunkIds\":[]}", + EntityRef.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"chunkId\":\"c\",\"documentId\":\"d\",\"chunkIndex\":1.9," + + "\"startOffset\":0,\"endOffset\":1,\"tokenEstimate\":1,\"text\":\"x\"}", + TextChunk.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"channel\":\"bm25\",\"rawScore\":1,\"rank\":2.5}", + ChannelScore.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"vertexIds\":[\"v1\",\"v2\"],\"edgeIds\":[\"e1\"]," + + "\"hop\":1.5,\"sampled\":false}", GraphPathRef.class)); + Assertions.assertThrows(JsonParseException.class, () -> RetrievalModelJson.fromJson( + "{\"kind\":\"CHUNK\",\"fusedScore\":1.1}", Evidence.class)); + } + + @Test + public void modelValidationUsesDedicatedExceptionAndEnforcesRanges() { + Assertions.assertThrows(RetrievalModelValidationException.class, + () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null, + null, -0.01, null)); + Assertions.assertThrows(RetrievalModelValidationException.class, + () -> new Evidence("e", EvidenceKind.CHUNK, null, null, null, null, null, + null, 1.01, null)); + Assertions.assertThrows(RetrievalModelValidationException.class, + () -> new GraphPathRef(Collections.singletonList("v1"), + Collections.singletonList("e1"), 0, false)); + Assertions.assertThrows(RetrievalModelValidationException.class, + () -> new GraphPathRef(Arrays.asList("v1", "v2"), + Collections.emptyList(), 1, false)); + } + + @Test + public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() { + SourceDocument document = RetrievalModelJson.fromJson( + "{\"documentId\":\"d\",\"dataset\":\"set\",\"datasetVersion\":\"v1\"," + + "\"split\":\"dev\",\"title\":\"title\",\"sourceUri\":\"uri\"," + + "\"sourceHash\":\"hash\",\"futureField\":true}", SourceDocument.class); + SourceDocument changedTitle = new SourceDocument("d", "set", "v1", "dev", + "other title", "uri", "hash"); + Assertions.assertTrue(document.sameIdentityAs(changedTitle)); + Assertions.assertNotEquals(document, changedTitle); + Assertions.assertEquals(document.hashCode(), document.hashCode()); + } + + @Test + public void emptyEvidenceIsOnlyIdentityEquivalentWhenStructureMatches() { + Evidence first = new Evidence("e1", EvidenceKind.CHUNK, null, null, null, null, + null, null, null, null); + Evidence second = new Evidence("e2", EvidenceKind.CHUNK, "different text", null, + null, null, null, null, null, 1); + Evidence otherKind = new Evidence("e3", EvidenceKind.ENTITY, null, null, null, null, + null, null, null, null); + Assertions.assertTrue(first.sameIdentityAs(second)); + Assertions.assertFalse(first.sameIdentityAs(otherKind)); + Assertions.assertNotEquals(first, second); + } + + @Test + public void constructorCopiesInputCollections() { + java.util.List aliases = new java.util.ArrayList<>(); + aliases.add("alias-a"); + EntityRef entity = new EntityRef("entity-1", "Name", aliases, "PERSON", aliases); + aliases.set(0, "changed"); + Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getAliases()); + Assertions.assertEquals(Collections.singletonList("alias-a"), entity.getSourceChunkIds()); + } + + @Test + public void allModelsRoundTripThroughValidatedJson() { + SourceDocument document = new SourceDocument("doc-1", "dataset", "v1", "dev", + "Title", "file:///doc", "hash"); + TextChunk chunk = new TextChunk("chunk-1", "doc-1", 0, 0, 4, 1, + "text", "policy-v1", "hash"); + EntityRef entity = new EntityRef("entity-1", "Name", Collections.singletonList("N"), + "PERSON", Collections.singletonList("chunk-1")); + GraphVertexRef vertex = new GraphVertexRef("person", "vertex-1", "entity-1"); + GraphEdgeRef edge = new GraphEdgeRef("edge-1", "knows", "entity-1", "entity-2", + Collections.singletonList("chunk-1")); + GraphPathRef path = new GraphPathRef(Arrays.asList("vertex-1", "vertex-2"), + Collections.singletonList("edge-1"), 1, false); + SourceRef source = new SourceRef("doc-1", "file:///doc", 0, 4); + ChannelScore score = new ChannelScore("bm25", 2.0, 0.8, 1); + GraphVersion graphVersion = new GraphVersion("graph", "v1"); + IndexVersion indexVersion = new IndexVersion("bm25", "v1", "graph-v1"); + + Assertions.assertEquals(document, roundTrip(document, SourceDocument.class)); + Assertions.assertEquals(chunk, roundTrip(chunk, TextChunk.class)); + Assertions.assertEquals(entity, roundTrip(entity, EntityRef.class)); + Assertions.assertEquals(vertex, roundTrip(vertex, GraphVertexRef.class)); + Assertions.assertEquals(edge, roundTrip(edge, GraphEdgeRef.class)); + Assertions.assertEquals(path, roundTrip(path, GraphPathRef.class)); + Assertions.assertEquals(source, roundTrip(source, SourceRef.class)); + Assertions.assertEquals(score, roundTrip(score, ChannelScore.class)); + Assertions.assertEquals(graphVersion, roundTrip(graphVersion, GraphVersion.class)); + Assertions.assertEquals(indexVersion, roundTrip(indexVersion, IndexVersion.class)); + } + + @Test + public void completeEvidenceFixtureRoundTrips() throws IOException { + String json = readFixture("retrieval/model/evidence.json"); + Evidence evidence = RetrievalModelJson.fromJson(json, Evidence.class); + Assertions.assertEquals(evidence, RetrievalModelJson.fromJson( + RetrievalModelJson.toJson(evidence), Evidence.class)); + Assertions.assertEquals(2, evidence.getStageScores().size()); + Assertions.assertEquals(1, evidence.getPaths().get(0).getHop()); + } + + @Test + public void identityFixtureDocumentsFieldBasedComparisons() throws IOException { + Map fixture = GSON.fromJson(readFixture( + "retrieval/model/identity-cases.json"), Map.class); + Assertions.assertNotNull(fixture.get("sourceDocument")); + SourceDocument source = RetrievalModelJson.fromJson(GSON.toJson(fixture.get( + "sourceDocument")), SourceDocument.class); + SourceDocument sourceVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get( + "sourceDocumentVariant")), SourceDocument.class); + Assertions.assertTrue(source.sameIdentityAs(sourceVariant)); + TextChunk chunk = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("chunk")), + TextChunk.class); + TextChunk chunkVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get( + "chunkVariant")), TextChunk.class); + Assertions.assertTrue(chunk.sameIdentityAs(chunkVariant)); + EntityRef entity = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("entity")), + EntityRef.class); + EntityRef entityVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get( + "entityVariant")), EntityRef.class); + Assertions.assertTrue(entity.sameIdentityAs(entityVariant)); + GraphEdgeRef edge = RetrievalModelJson.fromJson(GSON.toJson(fixture.get("edge")), + GraphEdgeRef.class); + GraphEdgeRef edgeVariant = RetrievalModelJson.fromJson(GSON.toJson(fixture.get( + "edgeVariant")), GraphEdgeRef.class); + Assertions.assertTrue(edge.sameIdentityAs(edgeVariant)); + } + + private T roundTrip(T value, Class type) { + return RetrievalModelJson.fromJson(RetrievalModelJson.toJson(value), type); + } + + private String readFixture(String name) throws IOException { + InputStream stream = getClass().getClassLoader().getResourceAsStream(name); + Assertions.assertNotNull(stream); + try (BufferedReader reader = new BufferedReader(new InputStreamReader(stream, "UTF-8"))) { + return reader.lines().collect(Collectors.joining()); + } + } +} diff --git a/geaflow-ai/src/test/resources/retrieval/model/evidence.json b/geaflow-ai/src/test/resources/retrieval/model/evidence.json new file mode 100644 index 000000000..52f46d5ea --- /dev/null +++ b/geaflow-ai/src/test/resources/retrieval/model/evidence.json @@ -0,0 +1 @@ +{"evidenceId":"evidence-1","kind":"GRAPH_FACT","text":"Alice knows Bob.","chunks":[{"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":16,"tokenEstimate":4,"text":"Alice knows Bob.","policyVersion":"policy-v1","textHash":"sha256:chunk-1"}],"entities":[{"entityId":"entity-1","canonicalName":"Alice","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]}],"paths":[{"vertexIds":["vertex-1","vertex-2"],"edgeIds":["edge-1"],"hop":1,"sampled":false}],"sources":[{"documentId":"doc-1","sourceUri":"https://example/doc-1","startOffset":0,"endOffset":16}],"stageScores":{"bm25":{"channel":"bm25","rawScore":2.18,"normalizedScore":0.8,"rank":1},"vector":{"channel":"vector","rawScore":0.76,"normalizedScore":0.7,"rank":2}},"fusedScore":0.84,"rank":1} diff --git a/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json new file mode 100644 index 000000000..e136cbf86 --- /dev/null +++ b/geaflow-ai/src/test/resources/retrieval/model/identity-cases.json @@ -0,0 +1,10 @@ +{ + "sourceDocument": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"uri","sourceHash":"hash"}, + "sourceDocumentVariant": {"documentId":"doc-1","dataset":"set","datasetVersion":"v1","split":"dev","title":"Renamed","sourceUri":"uri","sourceHash":"hash"}, + "chunk": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":1,"text":"text","policyVersion":"policy-v1","textHash":"hash"}, + "chunkVariant": {"chunkId":"chunk-1","documentId":"doc-1","chunkIndex":0,"startOffset":0,"endOffset":4,"tokenEstimate":99,"text":"text","policyVersion":"policy-v1","textHash":"hash"}, + "entity": {"entityId":"entity-1","canonicalName":"Name","aliases":["A"],"type":"PERSON","sourceChunkIds":["chunk-1"]}, + "entityVariant": {"entityId":"entity-1","canonicalName":"Name","aliases":["B"],"type":"PERSON","sourceChunkIds":["chunk-2"]}, + "edge": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-1"]}, + "edgeVariant": {"edgeId":"edge-1","label":"knows","sourceEntityId":"entity-1","targetEntityId":"entity-2","sourceChunkIds":["chunk-2"]} +} diff --git a/geaflow-ai/src/test/resources/retrieval/model/source-document.json b/geaflow-ai/src/test/resources/retrieval/model/source-document.json new file mode 100644 index 000000000..00183986b --- /dev/null +++ b/geaflow-ai/src/test/resources/retrieval/model/source-document.json @@ -0,0 +1 @@ +{"documentId":"doc-1","dataset":"hotpotqa","datasetVersion":"v1","split":"dev","title":"Title","sourceUri":"https://example/doc-1","sourceHash":"sha256"} From 5b74d20597014cbe14eef31088b572afa3089d5d Mon Sep 17 00:00:00 2001 From: AzrMedit0x <161927884+aotenjou@users.noreply.github.com> Date: Wed, 9 Sep 2026 12:09:27 +0800 Subject: [PATCH 4/5] Enhance equality checks for SourceDocument Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- .../geaflow/ai/retrieval/model/RetrievalDomainModelTest.java | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java index 1d56475ee..b50563f40 100644 --- a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java +++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java @@ -274,7 +274,9 @@ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() { "other title", "uri", "hash"); Assertions.assertTrue(document.sameIdentityAs(changedTitle)); Assertions.assertNotEquals(document, changedTitle); - Assertions.assertEquals(document.hashCode(), document.hashCode()); +SourceDocument copy = new SourceDocument("d", "set", "v1", "dev", "title", "uri", "hash"); +Assertions.assertEquals(document, copy); +Assertions.assertEquals(document.hashCode(), copy.hashCode()); } @Test From ab3273063d6f6b6689baa5867c7aba32f5d5104d Mon Sep 17 00:00:00 2001 From: aotenjou Date: Wed, 9 Sep 2026 18:46:46 +0800 Subject: [PATCH 5/5] fix(ai): preserve identity for unreferenced evidence --- .../ai/retrieval/model/evidence/Evidence.java | 31 ++++++++++++++----- .../ai/retrieval/model/package-info.java | 5 ++- .../model/RetrievalDomainModelTest.java | 13 ++++++-- 3 files changed, 37 insertions(+), 12 deletions(-) diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java index db60cc60d..ef2dd250c 100644 --- a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/evidence/Evidence.java @@ -36,8 +36,9 @@ * Immutable retrieval evidence with optional payload references and multi-channel scores. * *

Complete value equality includes presentation, provenance, and ranking fields. Identity - * comparison intentionally uses only the evidence kind and nested reference identities, allowing - * candidates from different retrieval channels to be merged without losing their scores.

+ * comparison uses the evidence kind and nested reference identities, allowing candidates from + * different retrieval channels to be merged without losing their scores. When no references are + * available, evidenceId or text provides the fallback identity.

*/ public final class Evidence { @@ -140,11 +141,27 @@ public Integer getRank() { } public boolean sameIdentityAs(Evidence other) { - return this == other || other != null && kind == other.kind - && sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs) - && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs) - && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs) - && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs); + if (this == other) { + return true; + } + if (other == null || kind != other.kind) { + return false; + } + if (hasReferences() || other.hasReferences()) { + return sameMultiset(getChunks(), other.getChunks(), TextChunk::sameIdentityAs) + && sameMultiset(getEntities(), other.getEntities(), EntityRef::sameIdentityAs) + && sameMultiset(getPaths(), other.getPaths(), GraphPathRef::sameIdentityAs) + && sameMultiset(getSources(), other.getSources(), SourceRef::sameIdentityAs); + } + if (evidenceId != null && other.evidenceId != null) { + return evidenceId.equals(other.evidenceId); + } + return text != null && other.text != null && text.equals(other.text); + } + + private boolean hasReferences() { + return !getChunks().isEmpty() || !getEntities().isEmpty() + || !getPaths().isEmpty() || !getSources().isEmpty(); } private static boolean sameMultiset(List first, List second, diff --git a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java index 54dbdf420..463789720 100644 --- a/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java +++ b/geaflow-ai/src/main/java/org/apache/geaflow/ai/retrieval/model/package-info.java @@ -22,8 +22,8 @@ * *

IDs in these objects are opaque values supplied by callers. This package does not generate * IDs. Use {@code sameIdentityAs} for identity comparisons and {@code equals} when complete value - * equality is required. Evidence identity excludes evidenceId, text, scores and ranks so - * multi-channel candidates can be merged; its nested references are compared recursively.

+ * equality is required. Evidence identity uses nested references when present, and falls back to + * evidenceId or text when no references are available; scores and ranks are ignored.

* *

JSON uses explicit camelCase field names. Optional fields may be added only additively; unknown * fields are ignored, null optional scalars are omitted, and collection fields use empty arrays or @@ -46,4 +46,3 @@ package org.apache.geaflow.ai.retrieval.model; import org.apache.geaflow.ai.retrieval.codec.RetrievalModelJson; - diff --git a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java index b50563f40..494aca92c 100644 --- a/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java +++ b/geaflow-ai/src/test/java/org/apache/geaflow/ai/retrieval/model/RetrievalDomainModelTest.java @@ -280,16 +280,25 @@ public void unknownJsonFieldsAreIgnoredAndIdentityDiffersFromValueEquality() { } @Test - public void emptyEvidenceIsOnlyIdentityEquivalentWhenStructureMatches() { + public void emptyEvidenceFallsBackToStableIdentityFields() { Evidence first = new Evidence("e1", EvidenceKind.CHUNK, null, null, null, null, null, null, null, null); Evidence second = new Evidence("e2", EvidenceKind.CHUNK, "different text", null, null, null, null, null, null, 1); Evidence otherKind = new Evidence("e3", EvidenceKind.ENTITY, null, null, null, null, null, null, null, null); - Assertions.assertTrue(first.sameIdentityAs(second)); + Assertions.assertFalse(first.sameIdentityAs(second)); Assertions.assertFalse(first.sameIdentityAs(otherKind)); Assertions.assertNotEquals(first, second); + + Assertions.assertTrue(first.sameIdentityAs(new Evidence("e1", EvidenceKind.CHUNK, + "different text", null, null, null, null, null, null, null))); + Assertions.assertTrue(new Evidence(null, EvidenceKind.CHUNK, "same text", null, null, + null, null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK, + "same text", null, null, null, null, null, null, null))); + Assertions.assertFalse(new Evidence(null, EvidenceKind.CHUNK, null, null, null, null, + null, null, null, null).sameIdentityAs(new Evidence(null, EvidenceKind.CHUNK, null, + null, null, null, null, null, null, null))); } @Test