From b3826957e231604cd1b456f0cfdcc90d0a44e814 Mon Sep 17 00:00:00 2001 From: Ray Date: Wed, 7 Oct 2026 17:13:35 +0800 Subject: [PATCH 1/3] fix(flash): measure word gaps from each char code's font width Flash split words such as "infe rence" and "Va riational" that the reader it was ported from keeps whole. The gap before a char was measured from the previous glyph's ink box whenever the ink reached past the glyph's advance, as the tail of a bold-italic 'f' does. The ligature band then held the pen at that ink edge, so the next letter looked like a new word. The original measures from the pen position instead: the font's width for the char code the content stream draws. The code walk already pairs each char with that code, so it now also reads the width from the font dict (/Widths, else /MissingWidth; Identity-H /W, else /DW). PDFium's GetGlyphWidth looks the width up by unicode and can land on another glyph's width, e.g. a summation sign drawn with a letter code. Type 3 fonts keep the box end, because their exact widths split headings ("Typ es") that the box end keeps whole. Chars the walk can't pair keep the old end as well. On 16 documents (3,579 pages), checked against the original's text items: 4,744 word gaps fixed, 7 worse, all in math or references where the gap sits exactly on the 0.1 em threshold. A textbook outline gets its titles back ("Monte Carlo inference"), and a paper's formula no longer becomes a heading. CPU +4% on a 1,098-page book. --- .../parser_pdfium_charlevel/char_extract.py | 8 ++++ .../parser_pdfium_charlevel/font_unicode.py | 42 +++++++++++++++++++ .../parser_pdfium_charlevel/text_normalize.py | 4 +- .../parser_pdfium_charlevel/unicode_apply.py | 39 ++++++++++++++++- tests/conftest.py | 10 ++--- tests/test_flash_extraction.py | 36 ++++++++++++++++ 6 files changed, 132 insertions(+), 7 deletions(-) diff --git a/pageindex/flash/parser_pdfium_charlevel/char_extract.py b/pageindex/flash/parser_pdfium_charlevel/char_extract.py index ffdbc7266..3a369d470 100644 --- a/pageindex/flash/parser_pdfium_charlevel/char_extract.py +++ b/pageindex/flash/parser_pdfium_charlevel/char_extract.py @@ -255,6 +255,7 @@ def _apply_type3_sizes(raw_chars: list[dict], size_by_font: dict) -> None: def _finalize_chars(raw_chars: list[dict]) -> list[dict]: """Second pass: compute glyph_w per char and emit the merged-ready dicts. The right glyph width definition depends on how PDFium reports the font's metrics: (a) Normal Type 1 fonts (fs_raw >= 1.5, scale.a ~= 1): FPDFFont_GetGlyphWidth(font, code, fs_raw) returns the advance in page units. Use as-is x matrix.a. (b) Scaled-matrix Type 3 (fs_raw < 1.5 but matrix scale >= 1.5, e.g. vector-heavy page's a scaled Type-3 subset with scale=36.49): GetGlyphWidth at fs_raw=0.19 gives font-natural-unit width; x matrix scale recovers page units. (c) Identity-matrix Type 3 (fs_raw < 1.5, matrix.a ~= 1, e.g. identity-matrix Type-3 sample an identity-matrix Type-3 font): GetGlyphWidth's output is wrong by an unknown FontMatrix factor (PDFium doesn't fold this for these fonts). Fall back to neighbor-step fallback (next_char.ox - this_char.ox within same obj). """ out: list[dict] = [] + code_ends: dict[tuple[int, int], float] = {} for key_value, candidate_item in enumerate(raw_chars): if candidate_item.get("drop"): # Folded into the previous char by _apply_font_unicode (PDFium's @@ -335,6 +336,13 @@ def _finalize_chars(raw_chars: list[dict]) -> list[dict]: reference_item["v_pen_y"] = pen_y # pen y after this glyph's advance (text extraction previous glyph transform[5]) reference_item["v_after"] = min(candidate_item["cell_top"], candidate_item["cell_bot"]) + elif "code_w" in candidate_item: + # The pen end the font's width for this char's code gives; chars + # consuming one code (a decomposed ligature) share the first one's. + key = candidate_item["code_key"] + if key not in code_ends: + code_ends[key] = candidate_item["ox"] + candidate_item["code_w"] * obj["fs_raw"] * obj["scale_x"] + reference_item["code_end"] = code_ends[key] out.append(reference_item) return out diff --git a/pageindex/flash/parser_pdfium_charlevel/font_unicode.py b/pageindex/flash/parser_pdfium_charlevel/font_unicode.py index 0721d5eca..718e36f6a 100644 --- a/pageindex/flash/parser_pdfium_charlevel/font_unicode.py +++ b/pageindex/flash/parser_pdfium_charlevel/font_unicode.py @@ -392,3 +392,45 @@ def _xref_key(number: int, other_text: str) -> tuple[str, str]: if codepoint != -1: final[font] = _from_char_code(codepoint) # amend overwrites return 1, final + + +def _font_widths(pdf_doc, xref: int): + """Per-code advance at font size 1, read from the font dict by char code: simple fonts /FirstChar + /Widths else /MissingWidth, Identity-H composites /W else /DW (1000). ``None`` where the advance comes from elsewhere (no /Widths, other CMaps), and for Type 3: its exact widths split words ("Typ es") that the ink-box end keeps whole.""" + def number(value, default): + try: + return float(value.get_object()) + except Exception: + return default + + font = pdf_doc._resolve_object(xref) + if font is None or not hasattr(font, "raw_get"): + return None + subtype = font.get("/Subtype") + if subtype == "/Type0": + if font.get("/Encoding") != "/Identity-H": + return None + descendant = font["/DescendantFonts"][0].get_object() + default = number(descendant["/DW"], 1000.0) if "/DW" in descendant else 1000.0 + table: dict[int, float] = {} + entries = descendant["/W"] if "/W" in descendant else [] + index = 0 + while index + 1 < len(entries): + start = int(number(entries[index], 0)) + second = entries[index + 1].get_object() + if isinstance(second, list): # c [w1 w2 ...] + for offset, value in enumerate(second): + table[start + offset] = number(value, default) + index += 2 + else: # c_first c_last w + value = number(entries[index + 2], default) if index + 2 < len(entries) else default + for code in range(max(start, 0), min(int(number(second, start - 1)), 0xFFFF) + 1): + table[code] = value + index += 3 + return lambda code: table.get(code, default) * 0.001 + if subtype == "/Type3" or "/Widths" not in font: + return None + first = int(number(font["/FirstChar"], 0)) if "/FirstChar" in font else 0 + descriptor = font["/FontDescriptor"] if "/FontDescriptor" in font else None + missing = number(descriptor["/MissingWidth"], 0.0) if descriptor is not None and "/MissingWidth" in descriptor else 0.0 + table = {first + offset: number(value, missing) for offset, value in enumerate(font["/Widths"])} + return lambda code: table.get(code, missing) * 0.001 diff --git a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py index 21a6f7240..1cc3acada 100644 --- a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py +++ b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py @@ -275,8 +275,10 @@ def _reverse_if_rtl(chars: str) -> str: def _read_end(mapping: dict, sign: int) -> float: - """The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = right edge = max(ox+glyph_w, ink right). * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)""" + """The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = the pen end the font's width for the char's code gives (code_end), else right edge = max(ox+glyph_w, ink right): glyph_w is looked up by unicode and can land on another glyph's width. * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)""" if sign > 0: + if "code_end" in mapping: + return mapping["code_end"] return max(mapping["ox"] + mapping["glyph_w"], mapping["right"]) return mapping["ox"] diff --git a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py index 6d777ec12..91b38539e 100644 --- a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py +++ b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py @@ -4,11 +4,12 @@ import bisect import difflib +import itertools import re from collections import Counter from .text_normalize import _is_whitespace -from .font_unicode import _font_unicode_map +from .font_unicode import _font_unicode_map, _font_widths from .code_walk import ( _char_category, _walk_codes, @@ -48,6 +49,33 @@ def targets_for(font_xref: int | None, other_numbers: tuple[int, ...]) -> list[s return [_SURROGATES.sub("\ufffd", measure_item.get((other_numbers[key_value] << 8) | other_numbers[key_value + 1]) or chr((other_numbers[key_value] << 8) | other_numbers[key_value + 1])) for key_value in range(0, len(other_numbers) - 1, 2)] + def widths_for(font_xref: int | None, other_numbers: tuple[int, ...], count: int) -> list[float | None]: + """Per-code advances at font size 1, parallel to a covered targets_for's output.""" + if font_xref is None: + return [None] * count + key = ("widths", font_xref) + if key not in map_cache: + try: + map_cache[key] = _font_widths(pdf_doc, font_xref) + except Exception: + map_cache[key] = None + lookup = map_cache[key] + if lookup is None: + return [None] * count + codes = (other_numbers if map_cache[font_xref][0] == 1 else + [(other_numbers[key_value] << 8) | other_numbers[key_value + 1] for key_value in range(0, len(other_numbers) - 1, 2)]) + return [lookup(code) for code in codes] + + walk_ids = itertools.count() + + def record(consumed: list[tuple[int, int]], widths: list[float | None]) -> None: + walk = next(walk_ids) + for char_index, target_index in consumed: + candidate_item = chars_by_index.get(char_index) + if candidate_item is not None and widths[target_index] is not None: + candidate_item["code_w"] = widths[target_index] + candidate_item["code_key"] = (walk, target_index) + def apply(patches: list[tuple[int, str]], drops: list[int], chars_by_index: dict[int, dict]) -> None: for index_value, token_value in patches: @@ -92,6 +120,7 @@ def apply(patches: list[tuple[int, str]], drops: list[int], desynced.append(object_index) continue apply(res[0], res[1], chars_by_index) + record(res[2], widths_for(font_index, encoded_text, len(target_text_items))) # Re-walk each window of desynced objects (bridging up to 2 covered, # successfully-walked objects between them) as one unit: boundary- # attribution errors cancel inside the window (the page-mode walk @@ -105,15 +134,18 @@ def _rewalk_window(window: list[int]) -> bool: (pair for state_item in window for pair in chars_by_obj.get(id(objects[state_item]), []))) text_transform: list[str] = [] owner: list[int] = [] + window_widths: list[float | None] = [] for state_item in window: target_text_items = targets_by_object_index[state_item] assert target_text_items is not None text_transform.extend(target_text_items) owner.extend([state_item] * len(target_text_items)) + window_widths.extend(widths_for(show_codes[state_item][0], show_codes[state_item][1], len(target_text_items))) def _commit(res) -> bool: if res is None: return False apply(res[0], res[1], chars_by_index) + record(res[2], window_widths) for char_index, text_index in res[2]: candidate_item = chars_by_index.get(char_index) if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]: @@ -198,6 +230,8 @@ def _object_distance_sq(target_index: int) -> float: "prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None, "next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None, }) + record([(char_value[char_index][0], target_index) for char_index, target_index in char_to_tgt.items()], + window_widths) for char_index, target_index in char_to_tgt.items(): candidate_item = chars_by_index.get(char_value[char_index][0]) if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]: @@ -298,6 +332,7 @@ def _run_window(window: list[int]) -> None: seq = [(raw_char["i"], raw_char["ch"]) for raw_char in raw_chars if not raw_char["is_gen"]] targets: list[str] = [] + page_widths: list[float | None] = [] for font_index, encoded_text, _tz in show_codes: if not encoded_text: continue @@ -305,10 +340,12 @@ def _run_window(window: list[int]) -> None: if text_state is None: return # uncovered font used on this page: no patch targets.extend(text_state) + page_widths.extend(widths_for(font_index, encoded_text, len(text_state))) res = _walk_codes(seq, targets) if res is None: return apply(res[0], res[1], chars_by_index) + record(res[2], page_widths) def _synthesize_dropped_glyphs( diff --git a/tests/conftest.py b/tests/conftest.py index b6326e46a..ff3b8d26c 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -8,10 +8,10 @@ def _llm_key(monkeypatch): monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key") -def build_pdf(page_texts): - """Build a minimal, uncompressed PDF (one Helvetica line per page, or a - page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns - the PDF file bytes.""" +def build_pdf(page_texts, font="<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"): + """Build a minimal, uncompressed PDF (one line per page, or a page's + (x, y, size, text) lines, in ``font``, Helvetica by default) whose text + PyPDF2 can extract. Returns the PDF file bytes.""" n = len(page_texts) objects = [] kids = " ".join(f"{3 + i} 0 R" for i in range(n)) @@ -31,7 +31,7 @@ def build_pdf(page_texts): parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") stream = " ".join(parts).encode() objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream)) - objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>") + objects.append(font.encode()) out = bytearray(b"%PDF-1.4\n") offsets = [] diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c2fc301df..698a0fa91 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -71,6 +71,42 @@ def test_rtl_sign_takes_a_multi_code_point_glyph(): assert _rtl_sign("") == 1 +def test_ink_past_the_advance_does_not_split_the_word(tmp_path): + """A glyph whose ink reaches past its advance (a bold-italic 'f') must not + open a space before the next letters: the gap is measured from the pen + end the font's width for the char's code gives, not from the ink.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("f") - 32] = 50 # Helvetica's 'f' ink now reaches into the 'e' + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["inference"], font=( + "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding " + f"/FirstChar 32 /LastChar 126 /Widths [{' '.join(map(str, widths))}] >>"))) + assert extract_toc(str(pdf))["page_texts"] == ["inference"] + + +def test_composite_font_widths_read_both_w_forms(): + """/W holds `c [w1 w2 ...]` and `c_first c_last w` entries, /DW covers + the rest, and a range is bounded to two-byte codes.""" + from types import SimpleNamespace + from PyPDF2.generic import ArrayObject, DictionaryObject, NameObject, NumberObject + from pageindex.flash.parser_pdfium_charlevel.font_unicode import _font_widths + + def array(*items): + return ArrayObject(NumberObject(item) if isinstance(item, int) else item for item in items) + + descendant = DictionaryObject({NameObject("/DW"): NumberObject(800), + NameObject("/W"): array(10, array(500, 600), 20, 2 ** 31, 700)}) + font = DictionaryObject({NameObject("/Subtype"): NameObject("/Type0"), + NameObject("/Encoding"): NameObject("/Identity-H"), + NameObject("/DescendantFonts"): ArrayObject([descendant])}) + widths = _font_widths(SimpleNamespace(_resolve_object=lambda xref: font), 1) + assert [widths(code) for code in (10, 11, 12, 20, 0xFFFF, 5)] == pytest.approx( + [0.5, 0.6, 0.8, 0.7, 0.7, 0.8]) + + def test_optimize_full_keyless_reports_file_errors_first(tmp_path, monkeypatch): """No credential pre-check: a bad path is a FileNotFoundError even keyless (validation runs first), and the LLM-free spellings still run From 3ab1ba1b4acdb65e73822e66ff13c2da2f3051db Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 8 Oct 2026 12:57:02 +0800 Subject: [PATCH 2/3] fix(flash): keep code widths to upright left-to-right text Rotated and mirrored text and right-to-left characters go back to the box-end measurement: a rotated pen does not move along +x, and PDFium returns RTL text in logical order, so the code walk pairs each char with its mirror's code. The latest walk now decides a char's width, so a char re-paired with a font that has no widths drops a stale one. A glyph PDFium drops after an overlapping one is re-emitted from the previous glyph's pen end instead of its ink edge, and advances by its own code width when no later glyph gives the advance. A 10-K drawn one glyph per show op read "the ef fects of". --- .../parser_pdfium_charlevel/unicode_apply.py | 31 ++++++++--- tests/conftest.py | 20 ++++--- tests/test_flash_extraction.py | 55 ++++++++++++++++++- 3 files changed, 88 insertions(+), 18 deletions(-) diff --git a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py index 91b38539e..d848eb8f1 100644 --- a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py +++ b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py @@ -8,7 +8,7 @@ import re from collections import Counter -from .text_normalize import _is_whitespace +from .text_normalize import _is_whitespace, _rtl_sign from .font_unicode import _font_unicode_map, _font_widths from .code_walk import ( _char_category, @@ -72,9 +72,19 @@ def record(consumed: list[tuple[int, int]], widths: list[float | None]) -> None: walk = next(walk_ids) for char_index, target_index in consumed: candidate_item = chars_by_index.get(char_index) - if candidate_item is not None and widths[target_index] is not None: + if candidate_item is None: + continue + # The latest walk decides. A code width only measures a pen moving + # along +x: not rotated or mirrored text, and not PDFium's + # logical-order RTL chars, which pair with their mirror's code. + if (widths[target_index] is not None + and candidate_item["obj"]["rot"] == 0 and candidate_item["obj"]["fs_raw"] > 0 + and _rtl_sign(candidate_item["ch"]) > 0): candidate_item["code_w"] = widths[target_index] candidate_item["code_key"] = (walk, target_index) + else: + candidate_item.pop("code_w", None) + candidate_item.pop("code_key", None) def apply(patches: list[tuple[int, str]], drops: list[int], chars_by_index: dict[int, dict]) -> None: @@ -145,17 +155,18 @@ def _commit(res) -> bool: if res is None: return False apply(res[0], res[1], chars_by_index) - record(res[2], window_widths) for char_index, text_index in res[2]: candidate_item = chars_by_index.get(char_index) if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]: candidate_item["obj"] = objects[owner[text_index]] + record(res[2], window_widths) # Skipped targets are glyphs PDFium never emitted; record # each with its show op and surviving stream neighbours so # _synthesize_dropped_glyphs can re-emit it (text extraction does). for text_index, pos in res[3]: synth_sites.append({ "t": text_transform[text_index], "owner": objects[owner[text_index]], + "w": window_widths[text_index], "prev_i": char_value[pos - 1][0] if pos > 0 else None, "next_i": char_value[pos][0] if pos < len(char_value) else None, }) @@ -227,15 +238,16 @@ def _object_distance_sq(target_index: int) -> float: normalized_token = mapped_tis[page_value] if page_value < len(mapped_tis) else None synth_sites.append({ "t": tts[target_index], "owner": objects[owner[target_index]], + "w": window_widths[target_index], "prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None, "next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None, }) - record([(char_value[char_index][0], target_index) for char_index, target_index in char_to_tgt.items()], - window_widths) for char_index, target_index in char_to_tgt.items(): candidate_item = chars_by_index.get(char_value[char_index][0]) if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]: candidate_item["obj"] = objects[owner[target_index]] + record([(char_value[char_index][0], target_index) for char_index, target_index in char_to_tgt.items()], + window_widths) return True if _displacement_repair(): return True @@ -351,7 +363,7 @@ def _run_window(window: list[int]) -> None: def _synthesize_dropped_glyphs( sites: list[dict], raw_chars: list[dict], chars_by_index: dict[int, dict], ) -> None: - """Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no surviving neighbour on a side, the advance is unknowable; emit zero-width there so presence and stream order are preserved without inserting a synthetic gap.""" + """Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no next survivor to measure against, the advance is the dropped codes' font widths when prev has a code width; otherwise it is unknowable, and the glyph is emitted zero-width so presence and stream order are preserved without inserting a synthetic gap.""" groups: list[list[dict]] = [] for site in sites: if (groups and groups[-1][0]["prev_i"] == site["prev_i"] @@ -368,7 +380,10 @@ def _synthesize_dropped_glyphs( count_item = len(text) if not count_item: continue - if prev is not None: + if prev is not None and "code_w" in prev: + # prev's pen end, not its ink edge (which may reach past it). + pen, baseline_y = prev["ox"] + prev["code_w"] * prev["obj"]["fs_raw"] * prev["obj"]["scale_x"], prev["oy"] + elif prev is not None: pen, baseline_y = prev["right"], prev["oy"] elif nxt is not None: pen, baseline_y = nxt["ox"], nxt["oy"] @@ -379,6 +394,8 @@ def _synthesize_dropped_glyphs( if (prev is not None and nxt is not None and abs(nxt["oy"] - baseline_y) < 0.5 and nxt["ox"] > pen): total = nxt["ox"] - pen + elif prev is not None and "code_w" in prev and all(site["w"] is not None for site in group_value): + total = sum(site["w"] for site in group_value) * owner["fs_raw"] * owner["scale_x"] adv = total / count_item # Textpage index: fractional, slotted against the owner's own chars # so the paint-order sort keys (page_order, i) place the run in diff --git a/tests/conftest.py b/tests/conftest.py index ff3b8d26c..74d986900 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -9,9 +9,10 @@ def _llm_key(monkeypatch): def build_pdf(page_texts, font="<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"): - """Build a minimal, uncompressed PDF (one line per page, or a page's - (x, y, size, text) lines, in ``font``, Helvetica by default) whose text - PyPDF2 can extract. Returns the PDF file bytes.""" + """Build a minimal, uncompressed PDF (one line per page, a page's + (x, y, size, text) lines, or a page's raw content stream as bytes, in + ``font``, Helvetica by default) whose text PyPDF2 can extract. Returns + the PDF file bytes.""" n = len(page_texts) objects = [] kids = " ".join(f"{3 + i} 0 R" for i in range(n)) @@ -25,11 +26,14 @@ def build_pdf(page_texts, font="<< /Type /Font /Subtype /Type1 /BaseFont /Helvet f"/Contents {3 + n + i} 0 R >>".encode() ) for page in page_texts: - parts = [] - for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: - safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") - parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") - stream = " ".join(parts).encode() + if isinstance(page, bytes): + stream = page + else: + parts = [] + for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: + safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") + parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") + stream = " ".join(parts).encode() objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream)) objects.append(font.encode()) diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index 698a0fa91..18ed6b782 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -71,6 +71,11 @@ def test_rtl_sign_takes_a_multi_code_point_glyph(): assert _rtl_sign("") == 1 +def _helvetica(widths, encoding="/WinAnsiEncoding"): + return ("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica " + f"/Encoding {encoding} /FirstChar 32 /LastChar 126 /Widths [{' '.join(map(str, widths))}] >>") + + def test_ink_past_the_advance_does_not_split_the_word(tmp_path): """A glyph whose ink reaches past its advance (a bold-italic 'f') must not open a space before the next letters: the gap is measured from the pen @@ -81,12 +86,56 @@ def test_ink_past_the_advance_does_not_split_the_word(tmp_path): widths = [556] * 95 widths[ord("f") - 32] = 50 # Helvetica's 'f' ink now reaches into the 'e' pdf = tmp_path / "doc.pdf" - pdf.write_bytes(build_pdf(["inference"], font=( - "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding " - f"/FirstChar 32 /LastChar 126 /Widths [{' '.join(map(str, widths))}] >>"))) + pdf.write_bytes(build_pdf(["inference"], font=_helvetica(widths))) assert extract_toc(str(pdf))["page_texts"] == ["inference"] +def test_dropped_glyph_after_ink_past_the_advance_does_not_split_the_word(tmp_path): + """Drawn one glyph per show op, the second 'f' of 'ff' overlaps the first + one's ink and PDFium drops it; the re-emitted 'f' starts at the first + one's pen end and advances by its own width.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("f") - 32] = 50 + lines, x = [], 72.0 + for ch in "the effects of": + lines.append((round(x, 3), 720, 12, ch)) + x += widths[ord(ch) - 32] * 12 / 1000 + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf([lines], font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["the effects of"] + + +def test_rotated_text_ignores_code_widths(tmp_path): + """A rotated pen moves along y, so a narrow glyph's code width must not + open gaps in rotated words.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("'") - 32] = 191 + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf([b"BT /F1 12 Tf 0 -1 1 0 300 700 Tm (O'Brien rock'n'roll) Tj ET"], + font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["O'Brien rock'n'roll"] + + +def test_rtl_chars_ignore_code_widths(tmp_path): + """PDFium returns Hebrew in logical order, so the code walk pairs each + char with its mirror's code; that width must not move the line's gaps.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("a") - 32], widths[ord("b") - 32] = 550, 150 + hebrew = "<< /Type /Encoding /BaseEncoding /WinAnsiEncoding /Differences [97 /afii57664 /afii57665] >>" + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["xyba z"], font=_helvetica(widths, hebrew))) + assert extract_toc(str(pdf))["page_texts"] == ["xy \u05d0\u05d1 z"] + + def test_composite_font_widths_read_both_w_forms(): """/W holds `c [w1 w2 ...]` and `c_first c_last w` entries, /DW covers the rest, and a range is bounded to two-byte codes.""" From 7e26c384d8d1cd89a8010de8b83048223f74c133 Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 8 Oct 2026 13:14:51 +0800 Subject: [PATCH 3/3] test(flash): pin the dropped-glyph cases without the system font Non-embedded Helvetica is drawn with the system's substitute, so whether PDFium drops the second of two overlapping one-glyph "f" objects differs between macOS and Linux. An "f" painted first at the same spot makes the drop certain on both. The advance of a dropped glyph with no later glyph is checked on _synthesize_dropped_glyphs directly. --- tests/test_flash_extraction.py | 23 ++++++++++++++++++----- 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index 18ed6b782..f3e1c2030 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -91,9 +91,9 @@ def test_ink_past_the_advance_does_not_split_the_word(tmp_path): def test_dropped_glyph_after_ink_past_the_advance_does_not_split_the_word(tmp_path): - """Drawn one glyph per show op, the second 'f' of 'ff' overlaps the first - one's ink and PDFium drops it; the re-emitted 'f' starts at the first - one's pen end and advances by its own width.""" + """An 'f' painted first where the word's second 'f' lands makes PDFium drop + that glyph; the re-emitted 'f' starts at the first one's pen end, not at + its ink edge. The 'f' PDFium keeps reads last.""" from conftest import build_pdf from pageindex.flash.main import extract_toc @@ -104,8 +104,21 @@ def test_dropped_glyph_after_ink_past_the_advance_does_not_split_the_word(tmp_pa lines.append((round(x, 3), 720, 12, ch)) x += widths[ord(ch) - 32] * 12 / 1000 pdf = tmp_path / "doc.pdf" - pdf.write_bytes(build_pdf([lines], font=_helvetica(widths))) - assert extract_toc(str(pdf))["page_texts"] == ["the effects of"] + pdf.write_bytes(build_pdf([[(lines[6][0], 720, 12, "f")] + lines], font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["the effects off"] + + +def test_dropped_glyph_with_no_next_survivor_advances_by_its_code_width(): + """With no later glyph to measure against, a re-emitted glyph starts at + the previous glyph's pen end and advances by its own code width.""" + from pageindex.flash.parser_pdfium_charlevel.unicode_apply import _synthesize_dropped_glyphs + + obj = {"fs_raw": 10.0, "scale_x": 1.0, "fs_eff": 10.0, "l": 0.0, "b": 0.0, "font_name": "F"} + prev = {"i": 0, "ch": "f", "ox": 100.0, "oy": 700.0, "right": 104.0, "code_w": 0.25, "obj": obj} + raw_chars = [prev] + _synthesize_dropped_glyphs([{"t": "f", "owner": obj, "w": 0.25, "prev_i": 0, "next_i": None}], + raw_chars, {0: prev}) + assert (raw_chars[-1]["ox"], raw_chars[-1]["w_synth"]) == (102.5, 2.5) def test_rotated_text_ignores_code_widths(tmp_path):