From 520de4a5ced1b2b19236a82e4fc9bdfda6f3f657 Mon Sep 17 00:00:00 2001
From: lanyue-llk <270302213+lanyue-llk@users.noreply.github.com>
Date: Wed, 23 Sep 2026 20:40:43 +0800
Subject: [PATCH] Fix TEDS ordered tree distance over explicit nodes
---
tests/test_teds.py | 29 ++++++++-
webmainbench/metrics/teds_metrics.py | 97 ++++------------------------
2 files changed, 37 insertions(+), 89 deletions(-)
diff --git a/tests/test_teds.py b/tests/test_teds.py
index 93bada4..cbd4b40 100644
--- a/tests/test_teds.py
+++ b/tests/test_teds.py
@@ -53,6 +53,29 @@ def test_teds_identical_tables(self):
self.assertTrue(result.success)
self.assertEqual(result.score, 1.0)
+ def test_teds_missing_child_changes_score(self):
+ reference = "
"
+ predicted = ""
+ result = self.teds_metric.calculate(
+ predicted=predicted,
+ groundtruth=reference,
+ table_edit_result=self.valid_table_edit_result,
+ )
+ self.assertTrue(result.success)
+ self.assertLess(result.score, 1.0)
+ self.assertEqual(result.details['edit_distance'], 2)
+
+ def test_teds_same_shape_changed_cell_text(self):
+ reference = ""
+ predicted = ""
+ result = self.teds_metric.calculate(
+ predicted=predicted,
+ groundtruth=reference,
+ table_edit_result=self.valid_table_edit_result,
+ )
+ self.assertTrue(result.success)
+ self.assertLess(result.score, 1.0)
+
def test_teds_different_tables(self):
"""Test completely different tables"""
pred = ""
@@ -219,7 +242,7 @@ def test_teds_structure_same_content_different(self):
groundtruth=gt,
table_edit_result=self.valid_table_edit_result
)
- self.assertAlmostEqual(result.score, 0.96, places=6)
+ self.assertAlmostEqual(result.score, 0.8, places=6)
class TestTEDSAdvanced(unittest.TestCase):
@@ -325,7 +348,7 @@ def test_teds_content_similarity(self):
table_edit_result=self.valid_table_edit_result
)
- self.assertAlmostEqual(result.score, 0.931818, places=6)
+ self.assertAlmostEqual(result.score, 0.6, places=6)
class TestStructureTEDS(unittest.TestCase):
"""Structure-only TEDS tests"""
@@ -507,4 +530,4 @@ def run_all_teds_tests():
print("\nAll TEDS tests passed!")
else:
print("\nSome TEDS tests failed!")
- sys.exit(1)
\ No newline at end of file
+ sys.exit(1)
diff --git a/webmainbench/metrics/teds_metrics.py b/webmainbench/metrics/teds_metrics.py
index 9a67e0b..84917ba 100644
--- a/webmainbench/metrics/teds_metrics.py
+++ b/webmainbench/metrics/teds_metrics.py
@@ -1,46 +1,4 @@
-"""
-TEDS (Tree-Edit Distance based Similarity) metrics for WebMainBench.
-
-I. Core algorithm upgrade: more accurate and efficient tree edit distance calculation
-Replaced custom simplified DP algorithm with professional APTED library.
-v1 issue: Custom DP algorithm only supported basic edit operations; inaccurate for nested tables
-(multi-level headers, merged cells) and slow for complex tables (DP matrix expansion).
-v2 improvement: Uses apted library (dedicated to ordered tree edit distance), strictly follows
-academic-grade algorithm, accurately identifies child order, nesting, and complex differences.
-5-10x speed improvement for tables under 100 nodes, resolves v1 misclassification of complex tables.
-Added algorithm failure fallback mechanism.
-v1 issue: Algorithm exceptions (e.g. excessive nesting) returned errors and interrupted evaluation.
-v2 improvement: When apted fails, falls back to “node count difference” (e.g. predicted 5 nodes,
-actual 3 nodes, distance=2), ensuring batch evaluation is not interrupted.
-
-II. Text difference: from binary to quantified scoring
-Introduced Levenshtein text edit distance.
-v1 issue: Text must be identical for nodes to be equal (e.g. “Product A” vs “Product A” with
-whitespace difference were considered unequal); text difference cost fixed at 1.0.
-v2 improvement: Uses rapidfuzz.distance.Levenshtein to quantify text differences,
-normalizing them to 0-1 cost range.
-
-III. Edge case handling: greatly improved robustness
-Empty input correction.
-v1 issue: Empty strings were forced into (invalid empty table),
-violating the semantics of “empty input = no table”, distorting score calculation.
-v2 improvement: Empty strings return empty directly; _parse_html_table recognizes empty tables,
-avoiding invalid HTML structures.
-Node serialization standardization.
-v1 issue: Node info stored in dicts without unified format, prone to key-value parsing errors.
-v2 improvement: Added _to_bracket_notation to convert nodes to apted-compatible bracket notation
-(e.g. table(tr(th:Product))), eliminating parsing format discrepancies.
-
-IV. Overall value improvement
-Accuracy: TEDS scores for complex tables (nested, merged cells) better reflect true structural
-differences; quantified text differences make results more objective.
-Efficiency: APTED optimized algorithm greatly improves speed for complex tables, supporting
-larger-scale batch evaluation.
-Robustness: Empty input handling correction and algorithm failure fallback ensure evaluation
-pipeline is not interrupted, adaptable to more edge cases.
-Flexibility: Quantified text differences support evaluation needs for OCR recognition errors
-and minor format deviations.
-"""
+"""TEDS table similarity using ordered tree edit distance over explicit nodes."""
from typing import Dict, Any, List, Optional
import re
@@ -51,6 +9,9 @@
class TableConfig(Config):
+ def children(self, node):
+ return node.get('children', [])
+
def delete(self, node):
return 1
@@ -73,6 +34,9 @@ def rename(self, node1, node2):
def _parse_node(self, node_str):
"""Parse node string in 'tag:text' or 'tag' format."""
+ if isinstance(node_str, dict):
+ text = node_str.get('text', '')
+ return node_str['tag'], text.replace('(', '[').replace(')', ']').replace(',', ';')
if ':' in node_str:
tag, text = node_str.split(':', 1)
return tag, text
@@ -252,48 +216,9 @@ def _element_to_tree(self, element) -> Dict:
def _tree_edit_distance(self, tree1: Dict, tree2: Dict) -> float:
"""Compute tree edit distance using APTED."""
- try:
- # Convert to APTED bracket notation
- t1 = self._to_bracket_notation(tree1)
- t2 = self._to_bracket_notation(tree2)
-
- # Compute edit distance using APTED
- apted = APTED(t1, t2, self.config_apted)
- edit_distance = apted.compute_edit_distance()
-
- return float(edit_distance)
- except Exception as e:
- # If APTED fails, fall back to simple node count difference
- print(f"APTED calculation failed: {e}, falling back to simple distance")
- nodes1 = self._count_nodes(tree1)
- nodes2 = self._count_nodes(tree2)
- return abs(nodes1 - nodes2)
-
- def _to_bracket_notation(self, node: Dict) -> str:
- """Convert dict tree to APTED bracket notation."""
- # Build node label
- tag = node['tag']
- text = node.get('text', '')
-
- # In structure-only mode, ignore text content
- if self.structure_only:
- label = tag
- else:
- # In full mode, include text content
- if text:
- # Escape special characters to avoid APTED parsing errors
- safe_text = text.replace('(', '[').replace(')', ']').replace(',', ';')
- label = f"{tag}:{safe_text}"
- else:
- label = tag
-
- # If no children, return the label
- if not node.get('children'):
- return label
-
- # Recursively process children
- children_str = ",".join([self._to_bracket_notation(c) for c in node['children']])
- return f"{label}({children_str})"
+ # APTED needs actual child nodes. A serialized label is treated as a
+ # leaf by Config.children, so it cannot measure any nested difference.
+ return float(APTED(tree1, tree2, self.config_apted).compute_edit_distance())
def _count_nodes(self, tree: Dict) -> int:
if tree is None:
@@ -308,4 +233,4 @@ def _setup(self) -> None:
super()._setup()
self.structure_only = True
self.name = "s_teds"
- self.description = "Structure-only Tree-Edit Distance based Similarity (S-TEDS)"
\ No newline at end of file
+ self.description = "Structure-only Tree-Edit Distance based Similarity (S-TEDS)"