Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions pageindex/flash/parser_pdfium_charlevel/char_extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -255,6 +255,7 @@ def _apply_type3_sizes(raw_chars: list[dict], size_by_font: dict) -> None:
def _finalize_chars(raw_chars: list[dict]) -> list[dict]:
"""Second pass: compute glyph_w per char and emit the merged-ready dicts. The right glyph width definition depends on how PDFium reports the font's metrics: (a) Normal Type 1 fonts (fs_raw >= 1.5, scale.a ~= 1): FPDFFont_GetGlyphWidth(font, code, fs_raw) returns the advance in page units. Use as-is x matrix.a. (b) Scaled-matrix Type 3 (fs_raw < 1.5 but matrix scale >= 1.5, e.g. vector-heavy page's a scaled Type-3 subset with scale=36.49): GetGlyphWidth at fs_raw=0.19 gives font-natural-unit width; x matrix scale recovers page units. (c) Identity-matrix Type 3 (fs_raw < 1.5, matrix.a ~= 1, e.g. identity-matrix Type-3 sample an identity-matrix Type-3 font): GetGlyphWidth's output is wrong by an unknown FontMatrix factor (PDFium doesn't fold this for these fonts). Fall back to neighbor-step fallback (next_char.ox - this_char.ox within same obj). """
out: list[dict] = []
code_ends: dict[tuple[int, int], float] = {}
for key_value, candidate_item in enumerate(raw_chars):
if candidate_item.get("drop"):
# Folded into the previous char by _apply_font_unicode (PDFium's
Expand Down Expand Up @@ -335,6 +336,13 @@ def _finalize_chars(raw_chars: list[dict]) -> list[dict]:
reference_item["v_pen_y"] = pen_y
# pen y after this glyph's advance (text extraction previous glyph transform[5])
reference_item["v_after"] = min(candidate_item["cell_top"], candidate_item["cell_bot"])
elif "code_w" in candidate_item:
# The pen end the font's width for this char's code gives; chars
# consuming one code (a decomposed ligature) share the first one's.
key = candidate_item["code_key"]
if key not in code_ends:
code_ends[key] = candidate_item["ox"] + candidate_item["code_w"] * obj["fs_raw"] * obj["scale_x"]
reference_item["code_end"] = code_ends[key]
out.append(reference_item)
return out

Expand Down
42 changes: 42 additions & 0 deletions pageindex/flash/parser_pdfium_charlevel/font_unicode.py
Original file line number Diff line number Diff line change
Expand Up @@ -392,3 +392,45 @@ def _xref_key(number: int, other_text: str) -> tuple[str, str]:
if codepoint != -1:
final[font] = _from_char_code(codepoint) # amend overwrites
return 1, final


def _font_widths(pdf_doc, xref: int):
"""Per-code advance at font size 1, read from the font dict by char code: simple fonts /FirstChar + /Widths else /MissingWidth, Identity-H composites /W else /DW (1000). ``None`` where the advance comes from elsewhere (no /Widths, other CMaps), and for Type 3: its exact widths split words ("Typ es") that the ink-box end keeps whole."""
def number(value, default):
try:
return float(value.get_object())
except Exception:
return default

font = pdf_doc._resolve_object(xref)
if font is None or not hasattr(font, "raw_get"):
return None
subtype = font.get("/Subtype")
if subtype == "/Type0":
if font.get("/Encoding") != "/Identity-H":
return None
descendant = font["/DescendantFonts"][0].get_object()
default = number(descendant["/DW"], 1000.0) if "/DW" in descendant else 1000.0
table: dict[int, float] = {}
entries = descendant["/W"] if "/W" in descendant else []
index = 0
while index + 1 < len(entries):
start = int(number(entries[index], 0))
second = entries[index + 1].get_object()
if isinstance(second, list): # c [w1 w2 ...]
for offset, value in enumerate(second):
table[start + offset] = number(value, default)
index += 2
else: # c_first c_last w
value = number(entries[index + 2], default) if index + 2 < len(entries) else default
for code in range(max(start, 0), min(int(number(second, start - 1)), 0xFFFF) + 1):
table[code] = value
index += 3
return lambda code: table.get(code, default) * 0.001
if subtype == "/Type3" or "/Widths" not in font:
return None
first = int(number(font["/FirstChar"], 0)) if "/FirstChar" in font else 0
descriptor = font["/FontDescriptor"] if "/FontDescriptor" in font else None
missing = number(descriptor["/MissingWidth"], 0.0) if descriptor is not None and "/MissingWidth" in descriptor else 0.0
table = {first + offset: number(value, missing) for offset, value in enumerate(font["/Widths"])}
return lambda code: table.get(code, missing) * 0.001
4 changes: 3 additions & 1 deletion pageindex/flash/parser_pdfium_charlevel/text_normalize.py
Original file line number Diff line number Diff line change
Expand Up @@ -275,8 +275,10 @@ def _reverse_if_rtl(chars: str) -> str:


def _read_end(mapping: dict, sign: int) -> float:
"""The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = right edge = max(ox+glyph_w, ink right). * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)"""
"""The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = the pen end the font's width for the char's code gives (code_end), else right edge = max(ox+glyph_w, ink right): glyph_w is looked up by unicode and can land on another glyph's width. * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)"""
if sign > 0:
if "code_end" in mapping:
return mapping["code_end"]
return max(mapping["ox"] + mapping["glyph_w"], mapping["right"])
return mapping["ox"]

Expand Down
62 changes: 58 additions & 4 deletions pageindex/flash/parser_pdfium_charlevel/unicode_apply.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,11 +4,12 @@

import bisect
import difflib
import itertools
import re
from collections import Counter

from .text_normalize import _is_whitespace
from .font_unicode import _font_unicode_map
from .text_normalize import _is_whitespace, _rtl_sign
from .font_unicode import _font_unicode_map, _font_widths
from .code_walk import (
_char_category,
_walk_codes,
Expand Down Expand Up @@ -48,6 +49,43 @@ def targets_for(font_xref: int | None, other_numbers: tuple[int, ...]) -> list[s
return [_SURROGATES.sub("\ufffd", measure_item.get((other_numbers[key_value] << 8) | other_numbers[key_value + 1]) or chr((other_numbers[key_value] << 8) | other_numbers[key_value + 1]))
for key_value in range(0, len(other_numbers) - 1, 2)]

def widths_for(font_xref: int | None, other_numbers: tuple[int, ...], count: int) -> list[float | None]:
"""Per-code advances at font size 1, parallel to a covered targets_for's output."""
if font_xref is None:
return [None] * count
key = ("widths", font_xref)
if key not in map_cache:
try:
map_cache[key] = _font_widths(pdf_doc, font_xref)
except Exception:
map_cache[key] = None
lookup = map_cache[key]
if lookup is None:
return [None] * count
codes = (other_numbers if map_cache[font_xref][0] == 1 else
[(other_numbers[key_value] << 8) | other_numbers[key_value + 1] for key_value in range(0, len(other_numbers) - 1, 2)])
return [lookup(code) for code in codes]

walk_ids = itertools.count()

def record(consumed: list[tuple[int, int]], widths: list[float | None]) -> None:
walk = next(walk_ids)
for char_index, target_index in consumed:
candidate_item = chars_by_index.get(char_index)
if candidate_item is None:
continue
# The latest walk decides. A code width only measures a pen moving
# along +x: not rotated or mirrored text, and not PDFium's
# logical-order RTL chars, which pair with their mirror's code.
if (widths[target_index] is not None
and candidate_item["obj"]["rot"] == 0 and candidate_item["obj"]["fs_raw"] > 0
and _rtl_sign(candidate_item["ch"]) > 0):
candidate_item["code_w"] = widths[target_index]
candidate_item["code_key"] = (walk, target_index)
else:
candidate_item.pop("code_w", None)
candidate_item.pop("code_key", None)

def apply(patches: list[tuple[int, str]], drops: list[int],
chars_by_index: dict[int, dict]) -> None:
for index_value, token_value in patches:
Expand Down Expand Up @@ -92,6 +130,7 @@ def apply(patches: list[tuple[int, str]], drops: list[int],
desynced.append(object_index)
continue
apply(res[0], res[1], chars_by_index)
record(res[2], widths_for(font_index, encoded_text, len(target_text_items)))
# Re-walk each window of desynced objects (bridging up to 2 covered,
# successfully-walked objects between them) as one unit: boundary-
# attribution errors cancel inside the window (the page-mode walk
Expand All @@ -105,11 +144,13 @@ def _rewalk_window(window: list[int]) -> bool:
(pair for state_item in window for pair in chars_by_obj.get(id(objects[state_item]), [])))
text_transform: list[str] = []
owner: list[int] = []
window_widths: list[float | None] = []
for state_item in window:
target_text_items = targets_by_object_index[state_item]
assert target_text_items is not None
text_transform.extend(target_text_items)
owner.extend([state_item] * len(target_text_items))
window_widths.extend(widths_for(show_codes[state_item][0], show_codes[state_item][1], len(target_text_items)))
def _commit(res) -> bool:
if res is None:
return False
Expand All @@ -118,12 +159,14 @@ def _commit(res) -> bool:
candidate_item = chars_by_index.get(char_index)
if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]:
candidate_item["obj"] = objects[owner[text_index]]
record(res[2], window_widths)
# Skipped targets are glyphs PDFium never emitted; record
# each with its show op and surviving stream neighbours so
# _synthesize_dropped_glyphs can re-emit it (text extraction does).
for text_index, pos in res[3]:
synth_sites.append({
"t": text_transform[text_index], "owner": objects[owner[text_index]],
"w": window_widths[text_index],
"prev_i": char_value[pos - 1][0] if pos > 0 else None,
"next_i": char_value[pos][0] if pos < len(char_value) else None,
})
Expand Down Expand Up @@ -195,13 +238,16 @@ def _object_distance_sq(target_index: int) -> float:
normalized_token = mapped_tis[page_value] if page_value < len(mapped_tis) else None
synth_sites.append({
"t": tts[target_index], "owner": objects[owner[target_index]],
"w": window_widths[target_index],
"prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None,
"next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None,
})
for char_index, target_index in char_to_tgt.items():
candidate_item = chars_by_index.get(char_value[char_index][0])
if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]:
candidate_item["obj"] = objects[owner[target_index]]
record([(char_value[char_index][0], target_index) for char_index, target_index in char_to_tgt.items()],
window_widths)
return True
if _displacement_repair():
return True
Expand Down Expand Up @@ -298,23 +344,26 @@ def _run_window(window: list[int]) -> None:
seq = [(raw_char["i"], raw_char["ch"])
for raw_char in raw_chars if not raw_char["is_gen"]]
targets: list[str] = []
page_widths: list[float | None] = []
for font_index, encoded_text, _tz in show_codes:
if not encoded_text:
continue
text_state = targets_for(font_index, encoded_text)
if text_state is None:
return # uncovered font used on this page: no patch
targets.extend(text_state)
page_widths.extend(widths_for(font_index, encoded_text, len(text_state)))
res = _walk_codes(seq, targets)
if res is None:
return
apply(res[0], res[1], chars_by_index)
record(res[2], page_widths)


def _synthesize_dropped_glyphs(
sites: list[dict], raw_chars: list[dict], chars_by_index: dict[int, dict],
) -> None:
"""Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no surviving neighbour on a side, the advance is unknowable; emit zero-width there so presence and stream order are preserved without inserting a synthetic gap."""
"""Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no next survivor to measure against, the advance is the dropped codes' font widths when prev has a code width; otherwise it is unknowable, and the glyph is emitted zero-width so presence and stream order are preserved without inserting a synthetic gap."""
groups: list[list[dict]] = []
for site in sites:
if (groups and groups[-1][0]["prev_i"] == site["prev_i"]
Expand All @@ -331,7 +380,10 @@ def _synthesize_dropped_glyphs(
count_item = len(text)
if not count_item:
continue
if prev is not None:
if prev is not None and "code_w" in prev:
# prev's pen end, not its ink edge (which may reach past it).
pen, baseline_y = prev["ox"] + prev["code_w"] * prev["obj"]["fs_raw"] * prev["obj"]["scale_x"], prev["oy"]
elif prev is not None:
pen, baseline_y = prev["right"], prev["oy"]
elif nxt is not None:
pen, baseline_y = nxt["ox"], nxt["oy"]
Expand All @@ -342,6 +394,8 @@ def _synthesize_dropped_glyphs(
if (prev is not None and nxt is not None
and abs(nxt["oy"] - baseline_y) < 0.5 and nxt["ox"] > pen):
total = nxt["ox"] - pen
elif prev is not None and "code_w" in prev and all(site["w"] is not None for site in group_value):
total = sum(site["w"] for site in group_value) * owner["fs_raw"] * owner["scale_x"]
adv = total / count_item
# Textpage index: fractional, slotted against the owner's own chars
# so the paint-order sort keys (page_order, i) place the run in
Expand Down
22 changes: 13 additions & 9 deletions tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,9 +8,10 @@ def _llm_key(monkeypatch):
monkeypatch.setenv("ANTHROPIC_API_KEY", "test-key")


def build_pdf(page_texts):
"""Build a minimal, uncompressed PDF (one Helvetica line per page, or a
page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns
def build_pdf(page_texts, font="<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"):
"""Build a minimal, uncompressed PDF (one line per page, a page's
(x, y, size, text) lines, or a page's raw content stream as bytes, in
``font``, Helvetica by default) whose text PyPDF2 can extract. Returns
the PDF file bytes."""
n = len(page_texts)
objects = []
Expand All @@ -25,13 +26,16 @@ def build_pdf(page_texts):
f"/Contents {3 + n + i} 0 R >>".encode()
)
for page in page_texts:
parts = []
for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page:
safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)")
parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET")
stream = " ".join(parts).encode()
if isinstance(page, bytes):
stream = page
else:
parts = []
for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page:
safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)")
parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET")
stream = " ".join(parts).encode()
objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream))
objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>")
objects.append(font.encode())

out = bytearray(b"%PDF-1.4\n")
offsets = []
Expand Down
Loading
Loading