From 520de4a5ced1b2b19236a82e4fc9bdfda6f3f657 Mon Sep 17 00:00:00 2001 From: lanyue-llk <270302213+lanyue-llk@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:40:43 +0800 Subject: [PATCH] Fix TEDS ordered tree distance over explicit nodes --- tests/test_teds.py | 29 ++++++++- webmainbench/metrics/teds_metrics.py | 97 ++++------------------------ 2 files changed, 37 insertions(+), 89 deletions(-) diff --git a/tests/test_teds.py b/tests/test_teds.py index 93bada4..cbd4b40 100644 --- a/tests/test_teds.py +++ b/tests/test_teds.py @@ -53,6 +53,29 @@ def test_teds_identical_tables(self): self.assertTrue(result.success) self.assertEqual(result.score, 1.0) + def test_teds_missing_child_changes_score(self): + reference = "
AB
" + predicted = "
A
" + result = self.teds_metric.calculate( + predicted=predicted, + groundtruth=reference, + table_edit_result=self.valid_table_edit_result, + ) + self.assertTrue(result.success) + self.assertLess(result.score, 1.0) + self.assertEqual(result.details['edit_distance'], 2) + + def test_teds_same_shape_changed_cell_text(self): + reference = "
Alpha
" + predicted = "
Beta
" + result = self.teds_metric.calculate( + predicted=predicted, + groundtruth=reference, + table_edit_result=self.valid_table_edit_result, + ) + self.assertTrue(result.success) + self.assertLess(result.score, 1.0) + def test_teds_different_tables(self): """Test completely different tables""" pred = "
1
" @@ -219,7 +242,7 @@ def test_teds_structure_same_content_different(self): groundtruth=gt, table_edit_result=self.valid_table_edit_result ) - self.assertAlmostEqual(result.score, 0.96, places=6) + self.assertAlmostEqual(result.score, 0.8, places=6) class TestTEDSAdvanced(unittest.TestCase): @@ -325,7 +348,7 @@ def test_teds_content_similarity(self): table_edit_result=self.valid_table_edit_result ) - self.assertAlmostEqual(result.score, 0.931818, places=6) + self.assertAlmostEqual(result.score, 0.6, places=6) class TestStructureTEDS(unittest.TestCase): """Structure-only TEDS tests""" @@ -507,4 +530,4 @@ def run_all_teds_tests(): print("\nAll TEDS tests passed!") else: print("\nSome TEDS tests failed!") - sys.exit(1) \ No newline at end of file + sys.exit(1) diff --git a/webmainbench/metrics/teds_metrics.py b/webmainbench/metrics/teds_metrics.py index 9a67e0b..84917ba 100644 --- a/webmainbench/metrics/teds_metrics.py +++ b/webmainbench/metrics/teds_metrics.py @@ -1,46 +1,4 @@ -""" -TEDS (Tree-Edit Distance based Similarity) metrics for WebMainBench. - -I. Core algorithm upgrade: more accurate and efficient tree edit distance calculation -Replaced custom simplified DP algorithm with professional APTED library. -v1 issue: Custom DP algorithm only supported basic edit operations; inaccurate for nested tables -(multi-level headers, merged cells) and slow for complex tables (DP matrix expansion). -v2 improvement: Uses apted library (dedicated to ordered tree edit distance), strictly follows -academic-grade algorithm, accurately identifies child order, nesting, and complex differences. -5-10x speed improvement for tables under 100 nodes, resolves v1 misclassification of complex tables. -Added algorithm failure fallback mechanism. -v1 issue: Algorithm exceptions (e.g. excessive nesting) returned errors and interrupted evaluation. -v2 improvement: When apted fails, falls back to “node count difference” (e.g. predicted 5 nodes, -actual 3 nodes, distance=2), ensuring batch evaluation is not interrupted. - -II. Text difference: from binary to quantified scoring -Introduced Levenshtein text edit distance. -v1 issue: Text must be identical for nodes to be equal (e.g. “Product A” vs “Product A” with -whitespace difference were considered unequal); text difference cost fixed at 1.0. -v2 improvement: Uses rapidfuzz.distance.Levenshtein to quantify text differences, -normalizing them to 0-1 cost range. - -III. Edge case handling: greatly improved robustness -Empty input correction. -v1 issue: Empty strings were forced into
(invalid empty table), -violating the semantics of “empty input = no table”, distorting score calculation. -v2 improvement: Empty strings return empty directly; _parse_html_table recognizes empty tables, -avoiding invalid HTML structures. -Node serialization standardization. -v1 issue: Node info stored in dicts without unified format, prone to key-value parsing errors. -v2 improvement: Added _to_bracket_notation to convert nodes to apted-compatible bracket notation -(e.g. table(tr(th:Product))), eliminating parsing format discrepancies. - -IV. Overall value improvement -Accuracy: TEDS scores for complex tables (nested, merged cells) better reflect true structural -differences; quantified text differences make results more objective. -Efficiency: APTED optimized algorithm greatly improves speed for complex tables, supporting -larger-scale batch evaluation. -Robustness: Empty input handling correction and algorithm failure fallback ensure evaluation -pipeline is not interrupted, adaptable to more edge cases. -Flexibility: Quantified text differences support evaluation needs for OCR recognition errors -and minor format deviations. -""" +"""TEDS table similarity using ordered tree edit distance over explicit nodes.""" from typing import Dict, Any, List, Optional import re @@ -51,6 +9,9 @@ class TableConfig(Config): + def children(self, node): + return node.get('children', []) + def delete(self, node): return 1 @@ -73,6 +34,9 @@ def rename(self, node1, node2): def _parse_node(self, node_str): """Parse node string in 'tag:text' or 'tag' format.""" + if isinstance(node_str, dict): + text = node_str.get('text', '') + return node_str['tag'], text.replace('(', '[').replace(')', ']').replace(',', ';') if ':' in node_str: tag, text = node_str.split(':', 1) return tag, text @@ -252,48 +216,9 @@ def _element_to_tree(self, element) -> Dict: def _tree_edit_distance(self, tree1: Dict, tree2: Dict) -> float: """Compute tree edit distance using APTED.""" - try: - # Convert to APTED bracket notation - t1 = self._to_bracket_notation(tree1) - t2 = self._to_bracket_notation(tree2) - - # Compute edit distance using APTED - apted = APTED(t1, t2, self.config_apted) - edit_distance = apted.compute_edit_distance() - - return float(edit_distance) - except Exception as e: - # If APTED fails, fall back to simple node count difference - print(f"APTED calculation failed: {e}, falling back to simple distance") - nodes1 = self._count_nodes(tree1) - nodes2 = self._count_nodes(tree2) - return abs(nodes1 - nodes2) - - def _to_bracket_notation(self, node: Dict) -> str: - """Convert dict tree to APTED bracket notation.""" - # Build node label - tag = node['tag'] - text = node.get('text', '') - - # In structure-only mode, ignore text content - if self.structure_only: - label = tag - else: - # In full mode, include text content - if text: - # Escape special characters to avoid APTED parsing errors - safe_text = text.replace('(', '[').replace(')', ']').replace(',', ';') - label = f"{tag}:{safe_text}" - else: - label = tag - - # If no children, return the label - if not node.get('children'): - return label - - # Recursively process children - children_str = ",".join([self._to_bracket_notation(c) for c in node['children']]) - return f"{label}({children_str})" + # APTED needs actual child nodes. A serialized label is treated as a + # leaf by Config.children, so it cannot measure any nested difference. + return float(APTED(tree1, tree2, self.config_apted).compute_edit_distance()) def _count_nodes(self, tree: Dict) -> int: if tree is None: @@ -308,4 +233,4 @@ def _setup(self) -> None: super()._setup() self.structure_only = True self.name = "s_teds" - self.description = "Structure-only Tree-Edit Distance based Similarity (S-TEDS)" \ No newline at end of file + self.description = "Structure-only Tree-Edit Distance based Similarity (S-TEDS)"