diff --git a/graphtage/levenshtein.py b/graphtage/levenshtein.py index 29f6aa0..d920e9f 100644 --- a/graphtage/levenshtein.py +++ b/graphtage/levenshtein.py @@ -117,7 +117,10 @@ def __init__( """ self.penalty: int = insert_remove_penalty # Optimization: See if the sequences trivially share a common prefix or suffix. - # If so, this will quadratically reduce the size of the Levenshtein matrix + # If so, this will quadratically reduce the size of the Levenshtein matrix. + # Stripping the prefix is output-visible as well as faster: it forces the leading elements to be matched + # diagonally, which selects a different, equally optimal alignment for about 10% of small-alphabet inputs. + # See test_shared_prefix_biases_the_alignment. self.shared_prefix: list[tuple[TreeNode, TreeNode]] = [] for fn, tn in zip(from_seq, to_seq, strict=False): if fn == tn: diff --git a/test/test_json.py b/test/test_json.py index ef82f12..121d19b 100644 --- a/test/test_json.py +++ b/test/test_json.py @@ -90,3 +90,52 @@ def test_joined_output_separates_inserted_and_removed_items(self): '{"a": [1, ~~2~~, 3], ++"b": 4++}', render(b'{"a": [1, 2, 3]}', b'{"a": [1, 3], "b": 4}', join_lists=True, join_dict_items=True) ) + + +class TestRenderedDiffs(TestCase): + """Pins the rendered diff of documents whose output the string edit script decides. + + Which characters come out marked removed and inserted, and in which order, is decided by the candidate + ordering in :meth:`graphtage.levenshtein.EditDistance._best_match` and by the shared-prefix strip in its + constructor. ``test/test_levenshtein.py`` pins both of those edit by edit; these snapshots pin what a user + sees, so a change to either shows up in a normal test run. + + """ + + def test_dict_with_every_key_and_value_changed(self): + """Pins the diff of a dict in which no key and no value survives unchanged.""" + self.assertEqual( + '{\n "source++s++": "/usr/l~~ocal~~++ib++",\n "target++s++": "/opt/l~~ocal~~++ib++"\n}', + render( + b'{"source": "/usr/local", "target": "/opt/local"}', + b'{"sources": "/usr/lib", "targets": "/opt/lib"}' + ) + ) + + def test_list_of_similar_strings(self): + """Pins the diff of a list whose items are matched to each other by string edit distance.""" + self.assertEqual( + '[\n "re++a++d",\n "gre~~e~~n",\n "blue++s++"\n]', + render(b'["red", "green", "blue"]', b'["read", "gren", "blues"]') + ) + + def test_nested_dict_with_a_changed_string_and_number(self): + """Pins the diff of a string nested two levels deep alongside a replaced number.""" + self.assertEqual( + '{\n "server": {\n "path": "/usr/l~~ocal~~++ib++/bin",\n "retries": 3 -> 4\n }\n}', + render( + b'{"server": {"path": "/usr/local/bin", "retries": 3}}', + b'{"server": {"path": "/usr/lib/bin", "retries": 4}}' + ) + ) + + def test_nested_dict_with_an_appended_list_item(self): + """Pins the indentation and the insertion marker of a list nested inside two dicts.""" + self.assertEqual( + '{\n "outer": {\n "inner": {\n "leaf": 1 -> 2\n },\n' + ' "sibling": [\n 1,\n 2,\n ++3++\n ]\n }\n}', + render( + b'{"outer": {"inner": {"leaf": 1}, "sibling": [1, 2]}}', + b'{"outer": {"inner": {"leaf": 2}, "sibling": [1, 2, 3]}}' + ) + ) diff --git a/test/test_levenshtein.py b/test/test_levenshtein.py index 36a1e3c..06f2b1c 100644 --- a/test/test_levenshtein.py +++ b/test/test_levenshtein.py @@ -11,6 +11,102 @@ SMALL_ALPHABET: str = 'abcd' +def render_script(distance: EditDistance) -> list[str]: + """Renders the edit script that a string :class:`EditDistance` reconstructs. + + Each edit becomes one token: ``x>y`` for a match of ``x`` against ``y`` (a substitution when the two + differ), ``+y`` for an insertion, and ``-x`` for a removal. + + Args: + distance: the edit whose script to render. + + Returns: + list[str]: One token per edit, in the order the edits are emitted. + + """ + script: list[str] = [] + for edit in distance.edits(): + if isinstance(edit, Match): + script.append(f"{edit.from_node.object}>{edit.to_node.object}") + elif isinstance(edit, Remove): + script.append(f"-{edit.from_node.object}") + elif isinstance(edit, Insert): + script.append(f"+{edit.from_node.object}") + else: + raise AssertionError(f"Unexpected edit type {edit.__class__.__name__} in a string edit script") + return script + + +def edit_script(from_str: str, to_str: str) -> list[str]: + """Renders the edit script that :func:`graphtage.string_edit_distance` reconstructs for two strings. + + Args: + from_str: the string to match from. + to_str: the string to match to. + + Returns: + list[str]: The tokens described by :func:`render_script`. + + """ + return render_script(string_edit_distance(from_str, to_str)) + + +def replay_script(script: list[str]) -> tuple[str, str, int]: + """Replays a script rendered by :func:`edit_script`. + + Args: + script: the tokens to replay. + + Returns: + tuple[str, str, int]: The string the script matches from, the string it matches to, and its total cost. + + """ + from_str, to_str, cost = '', '', 0 + for token in script: + if token[0] == '+': + to_str += token[1] + cost += 1 + elif token[0] == '-': + from_str += token[1] + cost += 1 + else: + from_str += token[0] + to_str += token[2] + cost += int(token[0] != token[2]) + return from_str, to_str, cost + + +def optimal_alignment(from_str: str, to_str: str) -> tuple[int, int]: + """Scores the best alignment of two strings, independently of :class:`graphtage.EditDistance`. + + Alignments are ordered the way :meth:`graphtage.levenshtein.EditDistance._best_match` orders them: by total + edit cost first, then by the number of edits. A substitution, an insertion, and a removal each cost one. + + Args: + from_str: the string to align from. + to_str: the string to align to. + + Returns: + tuple[int, int]: The cost of the best alignment and the number of edits in it. + + """ + rows, cols = len(from_str) + 1, len(to_str) + 1 + best: list[list[tuple[int, int]]] = [[(0, 0)] * cols for _ in range(rows)] + for row in range(1, rows): + best[row][0] = (row, row) + for col in range(1, cols): + best[0][col] = (col, col) + for row in range(1, rows): + for col in range(1, cols): + diagonal, up, left = best[row - 1][col - 1], best[row - 1][col], best[row][col - 1] + best[row][col] = min( + (diagonal[0] + int(from_str[row - 1] != to_str[col - 1]), diagonal[1] + 1), + (up[0] + 1, up[1] + 1), + (left[0] + 1, left[1] + 1), + ) + return best[-1][-1] + + class TestEditDistance(TestCase): def test_string_edit_distance_reconstruction(self): for _ in trange(200): @@ -94,3 +190,94 @@ def test_empty_string_edit_distance(self): 3, sum(1 for _ in string_edit_distance('', 'foo').edits()) ) + + def assert_edit_script(self, from_str: str, to_str: str, expected: list[str]): + """Asserts that the edit script for a pair of strings is exactly ``expected``. + + The expectation is first checked against :func:`optimal_alignment`, so a script that is merely what the + current code emits cannot be pinned as a contract. + + Args: + from_str: the string to match from. + to_str: the string to match to. + expected: the tokens that :func:`edit_script` is expected to produce. + + """ + replayed_from, replayed_to, cost = replay_script(expected) + pair = f"{from_str!r} -> {to_str!r}" + self.assertEqual(from_str, replayed_from, f"{pair}: the expected script does not match from {from_str!r}") + self.assertEqual(to_str, replayed_to, f"{pair}: the expected script does not match to {to_str!r}") + self.assertEqual( + optimal_alignment(from_str, to_str), + (cost, len(expected)), + f"{pair}: the expected script {expected!r} is not an optimal alignment" + ) + self.assertEqual(expected, edit_script(from_str, to_str), pair) + + def test_edit_script_tie_break_is_stable(self): + """Pins the edit script for pairs of strings that have more than one optimal alignment. + + :meth:`graphtage.levenshtein.EditDistance._best_match` orders candidate predecessors by accumulated cost, + then by the number of edits on the path, then by a fixed direction order: the diagonal, then the border + insertion, then the border removal. The second key is what makes one substitution beat an insertion paired + with a removal of the same cost. The direction order is what makes removals come out before insertions, + because reconstruction walks the matrix backwards. + + Nothing else in the suite checks *which* optimal alignment is chosen, only that some optimal alignment is. + These expectations fail if the candidate order in ``_best_match`` is permuted or if the path-length key is + dropped, both of which leave the total cost optimal and the rendered diff different. + + """ + self.assert_edit_script('', 'a', ['+a']) + self.assert_edit_script('foo', '', ['-f', '-o', '-o']) + self.assert_edit_script('ab', 'ba', ['a>b', 'b>a']) + self.assert_edit_script('abc', 'acb', ['a>a', 'b>c', 'c>b']) + self.assert_edit_script('aa', 'aba', ['a>a', '+b', 'a>a']) + # Two substitutions and a removal tie on cost with a removal, two matches and an insertion; the + # path-length key picks the shorter script. + self.assert_edit_script('aabc', 'bcb', ['a>b', 'a>c', 'b>b', '-c']) + # The two border directions tie on both keys; the removal has to come out before the insertion. + self.assert_edit_script('aba', 'bab', ['-a', 'b>b', 'a>a', '+b']) + # Six alignments tie on both keys, so every one of the three directions is load-bearing here. + self.assert_edit_script('caccda', 'bcddcb', ['-c', 'a>b', 'c>c', 'c>d', 'd>d', '+c', 'a>b']) + + def test_shared_prefix_biases_the_alignment(self): + """Pins the edit scripts that the shared-prefix strip in :meth:`EditDistance.__init__` decides. + + Stripping a common prefix reads as a pure optimization, but it is output-visible: it forces the leading + characters to be matched diagonally, whereas backward reconstruction otherwise reaches the origin by a + border move. Roughly 10% of small-alphabet pairs produce a different — equally optimal — script when the + strip is removed, which is why these expectations exist. + + The shared-suffix strip has no such effect, because it agrees with the diagonal-first tie-break that + reconstruction already applies at the end of the matrix. + + """ + self.assert_edit_script('a', 'aa', ['a>a', '+a']) + self.assert_edit_script('cc', 'c', ['c>c', '-c']) + self.assert_edit_script('aac', 'ab', ['a>a', '-a', 'c>b']) + self.assert_edit_script('cbcbbb', 'caa', ['c>c', '-b', '-c', '-b', 'b>a', 'b>a']) + + def test_edit_script_realizes_the_reported_cost(self): + """Checks that the reconstructed script costs what the edit reports as its distance. + + :meth:`TestEditDistance.test_string_edit_distance_is_levenshtein` checks the reported distance and + :meth:`TestEditDistance.test_string_edit_distance_reconstruction` checks that the script rebuilds both + strings, but nothing checks that the script a user sees adds up to the cost the edit reports. A cost + computed anywhere other than from the script itself passes both of the older tests. + + """ + for _ in trange(200): + str_from = ''.join(random.choices(SMALL_ALPHABET, k=random.randint(0, 10))) + str_to = ''.join(random.choices(SMALL_ALPHABET, k=random.randint(0, 10))) + distance: EditDistance = string_edit_distance(str_from, str_to) + script = render_script(distance) + replayed_from, replayed_to, cost = replay_script(script) + pair = f"{str_from!r} -> {str_to!r}" + self.assertEqual(str_from, replayed_from, pair) + self.assertEqual(str_to, replayed_to, pair) + self.assertEqual(levenshtein_distance(str_from, str_to), cost, pair) + self.assertEqual(optimal_alignment(str_from, str_to), (cost, len(script)), pair) + bounds = distance.bounds() + self.assertTrue(bounds.definitive(), f"{pair} has bounds {bounds!s}") + self.assertEqual(cost, bounds.upper_bound, pair)