diff --git a/pageindex/flash/parser_pdfium_charlevel/char_extract.py b/pageindex/flash/parser_pdfium_charlevel/char_extract.py index ffdbc7266..3a369d470 100644 --- a/pageindex/flash/parser_pdfium_charlevel/char_extract.py +++ b/pageindex/flash/parser_pdfium_charlevel/char_extract.py @@ -255,6 +255,7 @@ def _apply_type3_sizes(raw_chars: list[dict], size_by_font: dict) -> None: def _finalize_chars(raw_chars: list[dict]) -> list[dict]: """Second pass: compute glyph_w per char and emit the merged-ready dicts. The right glyph width definition depends on how PDFium reports the font's metrics: (a) Normal Type 1 fonts (fs_raw >= 1.5, scale.a ~= 1): FPDFFont_GetGlyphWidth(font, code, fs_raw) returns the advance in page units. Use as-is x matrix.a. (b) Scaled-matrix Type 3 (fs_raw < 1.5 but matrix scale >= 1.5, e.g. vector-heavy page's a scaled Type-3 subset with scale=36.49): GetGlyphWidth at fs_raw=0.19 gives font-natural-unit width; x matrix scale recovers page units. (c) Identity-matrix Type 3 (fs_raw < 1.5, matrix.a ~= 1, e.g. identity-matrix Type-3 sample an identity-matrix Type-3 font): GetGlyphWidth's output is wrong by an unknown FontMatrix factor (PDFium doesn't fold this for these fonts). Fall back to neighbor-step fallback (next_char.ox - this_char.ox within same obj). """ out: list[dict] = [] + code_ends: dict[tuple[int, int], float] = {} for key_value, candidate_item in enumerate(raw_chars): if candidate_item.get("drop"): # Folded into the previous char by _apply_font_unicode (PDFium's @@ -335,6 +336,13 @@ def _finalize_chars(raw_chars: list[dict]) -> list[dict]: reference_item["v_pen_y"] = pen_y # pen y after this glyph's advance (text extraction previous glyph transform[5]) reference_item["v_after"] = min(candidate_item["cell_top"], candidate_item["cell_bot"]) + elif "code_w" in candidate_item: + # The pen end the font's width for this char's code gives; chars + # consuming one code (a decomposed ligature) share the first one's. + key = candidate_item["code_key"] + if key not in code_ends: + code_ends[key] = candidate_item["ox"] + candidate_item["code_w"] * obj["fs_raw"] * obj["scale_x"] + reference_item["code_end"] = code_ends[key] out.append(reference_item) return out diff --git a/pageindex/flash/parser_pdfium_charlevel/font_unicode.py b/pageindex/flash/parser_pdfium_charlevel/font_unicode.py index 0721d5eca..718e36f6a 100644 --- a/pageindex/flash/parser_pdfium_charlevel/font_unicode.py +++ b/pageindex/flash/parser_pdfium_charlevel/font_unicode.py @@ -392,3 +392,45 @@ def _xref_key(number: int, other_text: str) -> tuple[str, str]: if codepoint != -1: final[font] = _from_char_code(codepoint) # amend overwrites return 1, final + + +def _font_widths(pdf_doc, xref: int): + """Per-code advance at font size 1, read from the font dict by char code: simple fonts /FirstChar + /Widths else /MissingWidth, Identity-H composites /W else /DW (1000). ``None`` where the advance comes from elsewhere (no /Widths, other CMaps), and for Type 3: its exact widths split words ("Typ es") that the ink-box end keeps whole.""" + def number(value, default): + try: + return float(value.get_object()) + except Exception: + return default + + font = pdf_doc._resolve_object(xref) + if font is None or not hasattr(font, "raw_get"): + return None + subtype = font.get("/Subtype") + if subtype == "/Type0": + if font.get("/Encoding") != "/Identity-H": + return None + descendant = font["/DescendantFonts"][0].get_object() + default = number(descendant["/DW"], 1000.0) if "/DW" in descendant else 1000.0 + table: dict[int, float] = {} + entries = descendant["/W"] if "/W" in descendant else [] + index = 0 + while index + 1 < len(entries): + start = int(number(entries[index], 0)) + second = entries[index + 1].get_object() + if isinstance(second, list): # c [w1 w2 ...] + for offset, value in enumerate(second): + table[start + offset] = number(value, default) + index += 2 + else: # c_first c_last w + value = number(entries[index + 2], default) if index + 2 < len(entries) else default + for code in range(max(start, 0), min(int(number(second, start - 1)), 0xFFFF) + 1): + table[code] = value + index += 3 + return lambda code: table.get(code, default) * 0.001 + if subtype == "/Type3" or "/Widths" not in font: + return None + first = int(number(font["/FirstChar"], 0)) if "/FirstChar" in font else 0 + descriptor = font["/FontDescriptor"] if "/FontDescriptor" in font else None + missing = number(descriptor["/MissingWidth"], 0.0) if descriptor is not None and "/MissingWidth" in descriptor else 0.0 + table = {first + offset: number(value, missing) for offset, value in enumerate(font["/Widths"])} + return lambda code: table.get(code, missing) * 0.001 diff --git a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py index 21a6f7240..1cc3acada 100644 --- a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py +++ b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py @@ -275,8 +275,10 @@ def _reverse_if_rtl(chars: str) -> str: def _read_end(mapping: dict, sign: int) -> float: - """The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = right edge = max(ox+glyph_w, ink right). * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)""" + """The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = the pen end the font's width for the char's code gives (code_end), else right edge = max(ox+glyph_w, ink right): glyph_w is looked up by unicode and can land on another glyph's width. * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)""" if sign > 0: + if "code_end" in mapping: + return mapping["code_end"] return max(mapping["ox"] + mapping["glyph_w"], mapping["right"]) return mapping["ox"] diff --git a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py index 6d777ec12..d848eb8f1 100644 --- a/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py +++ b/pageindex/flash/parser_pdfium_charlevel/unicode_apply.py @@ -4,11 +4,12 @@ import bisect import difflib +import itertools import re from collections import Counter -from .text_normalize import _is_whitespace -from .font_unicode import _font_unicode_map +from .text_normalize import _is_whitespace, _rtl_sign +from .font_unicode import _font_unicode_map, _font_widths from .code_walk import ( _char_category, _walk_codes, @@ -48,6 +49,43 @@ def targets_for(font_xref: int | None, other_numbers: tuple[int, ...]) -> list[s return [_SURROGATES.sub("\ufffd", measure_item.get((other_numbers[key_value] << 8) | other_numbers[key_value + 1]) or chr((other_numbers[key_value] << 8) | other_numbers[key_value + 1])) for key_value in range(0, len(other_numbers) - 1, 2)] + def widths_for(font_xref: int | None, other_numbers: tuple[int, ...], count: int) -> list[float | None]: + """Per-code advances at font size 1, parallel to a covered targets_for's output.""" + if font_xref is None: + return [None] * count + key = ("widths", font_xref) + if key not in map_cache: + try: + map_cache[key] = _font_widths(pdf_doc, font_xref) + except Exception: + map_cache[key] = None + lookup = map_cache[key] + if lookup is None: + return [None] * count + codes = (other_numbers if map_cache[font_xref][0] == 1 else + [(other_numbers[key_value] << 8) | other_numbers[key_value + 1] for key_value in range(0, len(other_numbers) - 1, 2)]) + return [lookup(code) for code in codes] + + walk_ids = itertools.count() + + def record(consumed: list[tuple[int, int]], widths: list[float | None]) -> None: + walk = next(walk_ids) + for char_index, target_index in consumed: + candidate_item = chars_by_index.get(char_index) + if candidate_item is None: + continue + # The latest walk decides. A code width only measures a pen moving + # along +x: not rotated or mirrored text, and not PDFium's + # logical-order RTL chars, which pair with their mirror's code. + if (widths[target_index] is not None + and candidate_item["obj"]["rot"] == 0 and candidate_item["obj"]["fs_raw"] > 0 + and _rtl_sign(candidate_item["ch"]) > 0): + candidate_item["code_w"] = widths[target_index] + candidate_item["code_key"] = (walk, target_index) + else: + candidate_item.pop("code_w", None) + candidate_item.pop("code_key", None) + def apply(patches: list[tuple[int, str]], drops: list[int], chars_by_index: dict[int, dict]) -> None: for index_value, token_value in patches: @@ -92,6 +130,7 @@ def apply(patches: list[tuple[int, str]], drops: list[int], desynced.append(object_index) continue apply(res[0], res[1], chars_by_index) + record(res[2], widths_for(font_index, encoded_text, len(target_text_items))) # Re-walk each window of desynced objects (bridging up to 2 covered, # successfully-walked objects between them) as one unit: boundary- # attribution errors cancel inside the window (the page-mode walk @@ -105,11 +144,13 @@ def _rewalk_window(window: list[int]) -> bool: (pair for state_item in window for pair in chars_by_obj.get(id(objects[state_item]), []))) text_transform: list[str] = [] owner: list[int] = [] + window_widths: list[float | None] = [] for state_item in window: target_text_items = targets_by_object_index[state_item] assert target_text_items is not None text_transform.extend(target_text_items) owner.extend([state_item] * len(target_text_items)) + window_widths.extend(widths_for(show_codes[state_item][0], show_codes[state_item][1], len(target_text_items))) def _commit(res) -> bool: if res is None: return False @@ -118,12 +159,14 @@ def _commit(res) -> bool: candidate_item = chars_by_index.get(char_index) if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]: candidate_item["obj"] = objects[owner[text_index]] + record(res[2], window_widths) # Skipped targets are glyphs PDFium never emitted; record # each with its show op and surviving stream neighbours so # _synthesize_dropped_glyphs can re-emit it (text extraction does). for text_index, pos in res[3]: synth_sites.append({ "t": text_transform[text_index], "owner": objects[owner[text_index]], + "w": window_widths[text_index], "prev_i": char_value[pos - 1][0] if pos > 0 else None, "next_i": char_value[pos][0] if pos < len(char_value) else None, }) @@ -195,6 +238,7 @@ def _object_distance_sq(target_index: int) -> float: normalized_token = mapped_tis[page_value] if page_value < len(mapped_tis) else None synth_sites.append({ "t": tts[target_index], "owner": objects[owner[target_index]], + "w": window_widths[target_index], "prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None, "next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None, }) @@ -202,6 +246,8 @@ def _object_distance_sq(target_index: int) -> float: candidate_item = chars_by_index.get(char_value[char_index][0]) if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]: candidate_item["obj"] = objects[owner[target_index]] + record([(char_value[char_index][0], target_index) for char_index, target_index in char_to_tgt.items()], + window_widths) return True if _displacement_repair(): return True @@ -298,6 +344,7 @@ def _run_window(window: list[int]) -> None: seq = [(raw_char["i"], raw_char["ch"]) for raw_char in raw_chars if not raw_char["is_gen"]] targets: list[str] = [] + page_widths: list[float | None] = [] for font_index, encoded_text, _tz in show_codes: if not encoded_text: continue @@ -305,16 +352,18 @@ def _run_window(window: list[int]) -> None: if text_state is None: return # uncovered font used on this page: no patch targets.extend(text_state) + page_widths.extend(widths_for(font_index, encoded_text, len(text_state))) res = _walk_codes(seq, targets) if res is None: return apply(res[0], res[1], chars_by_index) + record(res[2], page_widths) def _synthesize_dropped_glyphs( sites: list[dict], raw_chars: list[dict], chars_by_index: dict[int, dict], ) -> None: - """Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no surviving neighbour on a side, the advance is unknowable; emit zero-width there so presence and stream order are preserved without inserting a synthetic gap.""" + """Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no next survivor to measure against, the advance is the dropped codes' font widths when prev has a code width; otherwise it is unknowable, and the glyph is emitted zero-width so presence and stream order are preserved without inserting a synthetic gap.""" groups: list[list[dict]] = [] for site in sites: if (groups and groups[-1][0]["prev_i"] == site["prev_i"] @@ -331,7 +380,10 @@ def _synthesize_dropped_glyphs( count_item = len(text) if not count_item: continue - if prev is not None: + if prev is not None and "code_w" in prev: + # prev's pen end, not its ink edge (which may reach past it). + pen, baseline_y = prev["ox"] + prev["code_w"] * prev["obj"]["fs_raw"] * prev["obj"]["scale_x"], prev["oy"] + elif prev is not None: pen, baseline_y = prev["right"], prev["oy"] elif nxt is not None: pen, baseline_y = nxt["ox"], nxt["oy"] @@ -342,6 +394,8 @@ def _synthesize_dropped_glyphs( if (prev is not None and nxt is not None and abs(nxt["oy"] - baseline_y) < 0.5 and nxt["ox"] > pen): total = nxt["ox"] - pen + elif prev is not None and "code_w" in prev and all(site["w"] is not None for site in group_value): + total = sum(site["w"] for site in group_value) * owner["fs_raw"] * owner["scale_x"] adv = total / count_item # Textpage index: fractional, slotted against the owner's own chars # so the paint-order sort keys (page_order, i) place the run in diff --git a/tests/conftest.py b/tests/conftest.py index b6326e46a..74d986900 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -8,9 +8,10 @@ def _llm_key(monkeypatch): monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key") -def build_pdf(page_texts): - """Build a minimal, uncompressed PDF (one Helvetica line per page, or a - page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns +def build_pdf(page_texts, font="<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"): + """Build a minimal, uncompressed PDF (one line per page, a page's + (x, y, size, text) lines, or a page's raw content stream as bytes, in + ``font``, Helvetica by default) whose text PyPDF2 can extract. Returns the PDF file bytes.""" n = len(page_texts) objects = [] @@ -25,13 +26,16 @@ def build_pdf(page_texts): f"/Contents {3 + n + i} 0 R >>".encode() ) for page in page_texts: - parts = [] - for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: - safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") - parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") - stream = " ".join(parts).encode() + if isinstance(page, bytes): + stream = page + else: + parts = [] + for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: + safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") + parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") + stream = " ".join(parts).encode() objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream)) - objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>") + objects.append(font.encode()) out = bytearray(b"%PDF-1.4\n") offsets = [] diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c2fc301df..f3e1c2030 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -71,6 +71,104 @@ def test_rtl_sign_takes_a_multi_code_point_glyph(): assert _rtl_sign("") == 1 +def _helvetica(widths, encoding="/WinAnsiEncoding"): + return ("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica " + f"/Encoding {encoding} /FirstChar 32 /LastChar 126 /Widths [{' '.join(map(str, widths))}] >>") + + +def test_ink_past_the_advance_does_not_split_the_word(tmp_path): + """A glyph whose ink reaches past its advance (a bold-italic 'f') must not + open a space before the next letters: the gap is measured from the pen + end the font's width for the char's code gives, not from the ink.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("f") - 32] = 50 # Helvetica's 'f' ink now reaches into the 'e' + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["inference"], font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["inference"] + + +def test_dropped_glyph_after_ink_past_the_advance_does_not_split_the_word(tmp_path): + """An 'f' painted first where the word's second 'f' lands makes PDFium drop + that glyph; the re-emitted 'f' starts at the first one's pen end, not at + its ink edge. The 'f' PDFium keeps reads last.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("f") - 32] = 50 + lines, x = [], 72.0 + for ch in "the effects of": + lines.append((round(x, 3), 720, 12, ch)) + x += widths[ord(ch) - 32] * 12 / 1000 + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf([[(lines[6][0], 720, 12, "f")] + lines], font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["the effects off"] + + +def test_dropped_glyph_with_no_next_survivor_advances_by_its_code_width(): + """With no later glyph to measure against, a re-emitted glyph starts at + the previous glyph's pen end and advances by its own code width.""" + from pageindex.flash.parser_pdfium_charlevel.unicode_apply import _synthesize_dropped_glyphs + + obj = {"fs_raw": 10.0, "scale_x": 1.0, "fs_eff": 10.0, "l": 0.0, "b": 0.0, "font_name": "F"} + prev = {"i": 0, "ch": "f", "ox": 100.0, "oy": 700.0, "right": 104.0, "code_w": 0.25, "obj": obj} + raw_chars = [prev] + _synthesize_dropped_glyphs([{"t": "f", "owner": obj, "w": 0.25, "prev_i": 0, "next_i": None}], + raw_chars, {0: prev}) + assert (raw_chars[-1]["ox"], raw_chars[-1]["w_synth"]) == (102.5, 2.5) + + +def test_rotated_text_ignores_code_widths(tmp_path): + """A rotated pen moves along y, so a narrow glyph's code width must not + open gaps in rotated words.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("'") - 32] = 191 + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf([b"BT /F1 12 Tf 0 -1 1 0 300 700 Tm (O'Brien rock'n'roll) Tj ET"], + font=_helvetica(widths))) + assert extract_toc(str(pdf))["page_texts"] == ["O'Brien rock'n'roll"] + + +def test_rtl_chars_ignore_code_widths(tmp_path): + """PDFium returns Hebrew in logical order, so the code walk pairs each + char with its mirror's code; that width must not move the line's gaps.""" + from conftest import build_pdf + from pageindex.flash.main import extract_toc + + widths = [556] * 95 + widths[ord("a") - 32], widths[ord("b") - 32] = 550, 150 + hebrew = "<< /Type /Encoding /BaseEncoding /WinAnsiEncoding /Differences [97 /afii57664 /afii57665] >>" + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["xyba z"], font=_helvetica(widths, hebrew))) + assert extract_toc(str(pdf))["page_texts"] == ["xy \u05d0\u05d1 z"] + + +def test_composite_font_widths_read_both_w_forms(): + """/W holds `c [w1 w2 ...]` and `c_first c_last w` entries, /DW covers + the rest, and a range is bounded to two-byte codes.""" + from types import SimpleNamespace + from PyPDF2.generic import ArrayObject, DictionaryObject, NameObject, NumberObject + from pageindex.flash.parser_pdfium_charlevel.font_unicode import _font_widths + + def array(*items): + return ArrayObject(NumberObject(item) if isinstance(item, int) else item for item in items) + + descendant = DictionaryObject({NameObject("/DW"): NumberObject(800), + NameObject("/W"): array(10, array(500, 600), 20, 2 ** 31, 700)}) + font = DictionaryObject({NameObject("/Subtype"): NameObject("/Type0"), + NameObject("/Encoding"): NameObject("/Identity-H"), + NameObject("/DescendantFonts"): ArrayObject([descendant])}) + widths = _font_widths(SimpleNamespace(_resolve_object=lambda xref: font), 1) + assert [widths(code) for code in (10, 11, 12, 20, 0xFFFF, 5)] == pytest.approx( + [0.5, 0.6, 0.8, 0.7, 0.7, 0.8]) + + def test_optimize_full_keyless_reports_file_errors_first(tmp_path, monkeypatch): """No credential pre-check: a bad path is a FileNotFoundError even keyless (validation runs first), and the LLM-free spellings still run