Skip to content
Merged
9 changes: 3 additions & 6 deletions pageindex/agent_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -920,12 +920,9 @@ def _get_document_structure(client, doc_name: str,
waited and entry.get("status") != "failed")

try:
raw_tree = getattr(getattr(client, "_api", None), "raw_tree", None)
tree = raw_tree(entry["id"]) if raw_tree is not None else None
if tree is None:
# _format_structure strips text anyway — don't download it.
tree = client.get_tree(entry["id"], node_summary=True,
include_text=False).get("result")
# _format_structure strips text anyway — don't download it.
tree = client.get_tree(entry["id"], node_summary=True,
include_text=False).get("result")
except PageIndexAPIError as exc:
return _failure(
f"Failed to retrieve document structure: {exc}",
Expand Down
6 changes: 1 addition & 5 deletions pageindex/local_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -223,7 +223,7 @@ def _index_standard(self, file_path: str, page_texts: list[str]) -> tuple[list,
"summary_model": self._summary_model,
"if_add_node_id": "yes",
"if_add_node_summary": "yes",
"if_add_node_text": "yes",
"if_add_node_text": "no",
"if_add_doc_description": "yes",
})
result = page_index_main(file_path, opt, logger=logger, page_list=page_list)
Expand Down Expand Up @@ -269,10 +269,6 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list:
add_node_text(structure, pdf_pages)
return structure

def raw_tree(self, doc_id: str) -> list | None:
"""Stored tree verbatim, every key kept."""
return self._store.get_tree(doc_id)

def get_tree(self, doc_id: str, node_summary: bool = False,
include_text: bool = True) -> dict[str, Any]:
meta = self._require_doc(doc_id, "Failed to get tree result")
Expand Down
10 changes: 7 additions & 3 deletions pageindex/page_index_classic.py
Original file line number Diff line number Diff line change
Expand Up @@ -1165,13 +1165,17 @@ async def meta_processor(page_list, mode=None, toc_content=None, toc_page_list=N
raise Exception('Processing failed')


async def process_large_node_recursively(node, page_list, opt=None, logger=None):
async def process_large_node_recursively(node, page_list, opt=None, logger=None, split=None):
# split: the pages last split above this node; the model can rebuild the
# node over them, e.g. when another heading sits above its own
node_page_list = page_list[node['start_index']-1:node['end_index']]
token_num = sum([page[1] for page in node_page_list])

if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node
and token_num >= opt.max_token_num_each_node):
and token_num >= opt.max_token_num_each_node
and (node['start_index'], node['end_index']) != split):
print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num)
split = (node['start_index'], node['end_index'])

node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger)
node_toc_tree = await check_title_appearance_in_start_concurrent(node_toc_tree, page_list, model=opt.model, logger=logger)
Expand All @@ -1190,7 +1194,7 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None)

if 'nodes' in node and node['nodes']:
tasks = [
process_large_node_recursively(child_node, page_list, opt, logger=logger)
process_large_node_recursively(child_node, page_list, opt, logger=logger, split=split)
for child_node in node['nodes']
]
await asyncio.gather(*tasks)
Expand Down
136 changes: 100 additions & 36 deletions pageindex/tree_optimize.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@

trigger: S(v) > TRIGGER_PAGES (cost control on generation, not the rule)
collapse_cost = S(v)
expand_cost = R(v) + max(S_residual(v), max_i S(c_i))
expand_cost = R(v) + max_i S(c_i) the c_i include the intro attach_children adds
expand iff expand_cost < collapse_cost (ties keep collapsed)
expand_gain = collapse_cost - expand_cost

Expand Down Expand Up @@ -132,21 +132,11 @@ async def ask_model(model, prompt):


def load_pages(pdf_path):
"""Per-page text, and per-page lines ordered top to bottom."""
import pymupdf
doc = pymupdf.open(pdf_path)
text, lines = [], []
for page in doc:
text.append(page.get_text())
ordered = []
for block in page.get_text("dict")["blocks"]:
for line in block.get("lines", []):
content = "".join(s["text"] for s in line["spans"]).strip()
if content:
ordered.append((line["bbox"][1], content))
ordered.sort()
lines.append([c for _, c in ordered])
return text, lines
"""Per-page text, and per-page lines in reading order, as flash reads them."""
from .flash.api import _page_lines
from .flash.main import extract_toc
text = extract_toc(pdf_path, use_embedded_toc=False)["page_texts"]
return text, _page_lines(text)


# --------------------------------------------------------------------------
Expand All @@ -172,12 +162,14 @@ def is_frontier(node):

def heading_at_page_start(lines, page_no, heading):
"""Is the heading the first line on its page? No when that cannot be told
(a heading with no Latin letter or digit to match), so the page is shared."""
(a heading with no Latin letter to match, or a first line that a later line
repeats, as a running header does), so the page is shared."""
page = lines[page_no - 1]
key = normalize(heading)
if not page or not key:
if not page or not re.search("[a-z]", key):
return False
return key in normalize(page[0])
starts = [(normalize(line) + " ").startswith(key + " ") for line in page]
return starts[0] and not any(starts[1:])


def assign_ends(node, children, lines):
Expand Down Expand Up @@ -304,14 +296,13 @@ def tree_cost_via_frontier(node, routing=ROUTING_COST):
return max(d * routing + s for d, s, _ in entries) if entries else 0


def expand_cost(node, children, routing=ROUTING_COST):
"""Cost after one-step lookahead, children treated as collapsed."""
covered = set()
for child in children:
covered |= set(range(child["start_index"], child["end_index"] + 1))
residual = len(pages_of(node) - covered)
scans = [child["end_index"] - child["start_index"] + 1 for child in children]
return routing + max([residual] + scans), residual
def expand_cost(node, children, lines, routing=ROUTING_COST):
"""Cost after one-step lookahead, children treated as collapsed, priced with
the intro node attach_children would give them."""
trial = dict(node, nodes=children)
residual = S_residual(trial)
add_intro_nodes([trial], lines)
return tree_cost(trial, routing), residual


# --------------------------------------------------------------------------
Expand Down Expand Up @@ -676,6 +667,66 @@ async def propose_children(node, pages, args):
return accepted


def same_heading(a, b):
"""Whether two titles name one heading: equal once normalized, or equal but
for a leading number only one of them prints. A title with no Latin letter
is its number alone, so it never matches by that second rule."""
a, b = normalize(a), normalize(b)
bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b))
return bool(a) and (a == b or (bare_a == bare_b and bool(re.search("[a-z]", bare_a))
and (a == bare_a or b == bare_b)))


def headings(node):
"""The headings printed for a node. A same-page fusion keeps them in its
key_items: the summary pass may rewrite the fused title while expand runs."""
return node["key_items"] if node.get("_same_page") else [node["title"]]


def own_children(node, children, lines, known, ancestors, nxt):
"""The proposed children printed inside the node's own text.

Dropped: a heading that already is a node, in the tree as expand found it
(`known`, page -> headings) or made by the node's own ancestors; and on a
page the node shares, anything printed above its heading (else above the
nearest ancestor heading found there) or at and below the next node's (else
its first descendant's found there). Other branches grow concurrently, so
nothing they add is read. A heading not found on its page decides nothing.
"""
start, end = node["start_index"], subtree_end(node)
lineage = [n for a in ancestors for n in [a] + a["nodes"]]
after = []
while nxt is not None and nxt["start_index"] == end:
after.append(nxt)
nxt = (nxt.get("nodes") or [None])[0]

def found(page, matches):
page_lines = lines[page - 1] if page <= len(lines) else []
return [i for i, line in enumerate(page_lines) if matches(line)]

def is_node(page, title):
titles = known.get(page, []) + [t for n in lineage if n["start_index"] == page
for t in headings(n)]
return any(same_heading(title, t) for t in titles)

tops = (found(start, lambda line: any(same_heading(line, t) for t in headings(n)))
for n in [node] + ancestors if n["start_index"] == start)
top = next((hits for hits in tops if hits), [])
bottoms = (found(end, lambda line: any(same_heading(line, t) for t in headings(n)))
for n in after)
bottom = next((hits for hits in bottoms if hits), [])
kept = []
for child in children:
page, key = child["start_index"], normalize(child["title"])
printed = found(page, lambda line: key in normalize(line)) if key else []
if (is_node(page, child["title"])
or page == start and top and printed and printed[-1] < top[0]
or page == end and bottom and printed and printed[0] >= bottom[-1]):
continue
kept.append(child)
return kept


async def expand(structure, pages, lines, args, log, frozen):
"""One-step lookahead on every collapsed node over the trigger, recursively.

Expand All @@ -687,7 +738,7 @@ async def expand(structure, pages, lines, args, log, frozen):
changed = False
semaphore = asyncio.Semaphore(args.concurrency)

async def proposals_for(node):
async def proposals_for(node, own):
"""The model half of one node's lookahead: the empty-retry ladder and
absorbed errors run inside the task; log entries come back so a
node's entries stay contiguous under concurrency."""
Expand All @@ -704,12 +755,13 @@ async def proposals_for(node):
"decision": "error", "attempt": attempts,
"detail": f"{type(exc).__name__}: {exc}"})
continue
proposed = own(proposed)
if proposed:
llm_candidates.append((f"llm:{attempts}", proposed))
break # an empty answer is retried, not trusted
return llm_candidates, attempts, entries

async def process(node):
async def process(node, ancestors, nxt):
nonlocal changed
if not is_frontier(node) or node.get("node_id") in frozen:
return
Expand All @@ -718,10 +770,13 @@ async def process(node):
return # below the trigger, stay collapsed
note(args.progress, f" expand {node.get('node_id'):>8} S={span} "
f"pages {node['start_index']}-{subtree_end(node)} ...")
llm_candidates, attempts, entries = await proposals_for(node)

def own(children):
return own_children(node, children, lines, known, ancestors, nxt)
llm_candidates, attempts, entries = await proposals_for(node, own)
log.extend(entries)
candidates = []
cached = children_from_cache(node, args.cache, args.kinds)
cached = own(children_from_cache(node, args.cache, args.kinds))
if cached:
candidates.append(("cache", cached))
candidates.extend(llm_candidates)
Expand All @@ -737,7 +792,7 @@ async def process(node):
scored = []
for source, children in candidates:
sized = assign_ends(node, children, lines)
cost, residual = expand_cost(node, sized, args.routing)
cost, residual = expand_cost(node, sized, lines, args.routing)
scored.append({"source": source, "children": sized,
"expand_cost": cost, "S_residual": residual})
scored.sort(key=lambda s: s["expand_cost"])
Expand Down Expand Up @@ -777,15 +832,24 @@ async def process(node):
merge_same_page([node], log)
# settle after the fusion: mark_final snapshots the children, finish() rejects a later change
args.settled([node])
results = await asyncio.gather(*(process(child)
for child in node["nodes"]),
children = node["nodes"]
results = await asyncio.gather(*(process(child, [node] + ancestors, after)
for child, after in zip(children, children[1:] + [nxt])),
return_exceptions=True)
for result in results:
if isinstance(result, BaseException):
raise result

results = await asyncio.gather(*(process(node)
for node, _ in flatten(structure)),
# the tree as it stands now: expand only ever adds below a leaf it is
# processing, so the nodes around another leaf never change under it
flat = list(flatten(structure))
known, ancestry = {}, {}
for node, parent in flat:
known.setdefault(node["start_index"], []).extend(headings(node))
ancestry[id(node)] = [parent] + ancestry[id(parent)] if parent else []
results = await asyncio.gather(*(process(node, ancestry[id(node)], after)
for (node, _), after in
zip(flat, [n for n, _ in flat[1:]] + [None])),
return_exceptions=True)
for result in results:
if isinstance(result, BaseException):
Expand Down
6 changes: 3 additions & 3 deletions pageindex/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -591,9 +591,9 @@ def post_processing(structure, end_physical_index):
item['start_index'] = item.get('physical_index')
if i < len(structure) - 1:
if structure[i + 1].get('appear_start') == 'yes':
item['end_index'] = structure[i + 1]['physical_index']-1
item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']-1)
else:
item['end_index'] = structure[i + 1]['physical_index']
item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index'])
else:
item['end_index'] = end_physical_index
tree = list_to_tree(structure)
Expand Down Expand Up @@ -1011,7 +1011,7 @@ async def _ask(self, prompt, prio):
self._asked = True
async with self._gate.slot(prio):
reply = await llm_acompletion(self._model, prompt)
if reply:
if parse_summary(reply):
self._answered = True
return reply

Expand Down
14 changes: 9 additions & 5 deletions tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,9 @@ def _llm_key(monkeypatch):


def build_pdf(page_texts):
"""Build a minimal, uncompressed PDF (one Helvetica line per page) whose
text PyPDF2 can extract. Returns the PDF file bytes."""
"""Build a minimal, uncompressed PDF (one Helvetica line per page, or a
page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns
the PDF file bytes."""
n = len(page_texts)
objects = []
kids = " ".join(f"{3 + i} 0 R" for i in range(n))
Expand All @@ -23,9 +24,12 @@ def build_pdf(page_texts):
f"/Resources << /Font << /F1 {font_obj} 0 R >> >> "
f"/Contents {3 + n + i} 0 R >>".encode()
)
for text in page_texts:
safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)")
stream = f"BT /F1 12 Tf 72 720 Td ({safe}) Tj ET".encode()
for page in page_texts:
parts = []
for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page:
safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)")
parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET")
stream = " ".join(parts).encode()
objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream))
objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>")

Expand Down
12 changes: 12 additions & 0 deletions tests/test_agent_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -287,6 +287,18 @@ def test_structure_strips_text_and_orders_keys(client, store_path):
assert root["nodes"][0]["end_index"] == 1


def test_structure_shows_the_ranges_get_tree_serves(client, store_path):
# a leaf stored ending before it starts: its successor was judged to open the same page
tree = [{"title": "Doc", "node_id": "0000", "start_index": 1, "end_index": 2, "nodes": [
{"title": "A", "node_id": "0001", "start_index": 2, "end_index": 1},
{"title": "B", "node_id": "0002", "start_index": 2, "end_index": 2}]}]
seed_doc(store_path, "pi-a", "report.pdf", tree=tree)
payload, _ = run(client, "get_document_structure", doc_name="report.pdf")
served = client.get_tree("pi-a", include_text=False)["result"]
assert [(n["start_index"], n["end_index"]) for n in payload["structure"][0]["nodes"]] == [
(n["start_index"], n["end_index"]) for n in served[0]["nodes"]] == [(2, 2), (2, 2)]


def test_structure_multipart_pagination(client, store_path):
big_tree = [{
"title": f"Chapter {index}", "node_id": f"{index:04d}",
Expand Down
7 changes: 4 additions & 3 deletions tests/test_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,7 @@ def indexed_doc(local_client, sample_pdf, monkeypatch):
"""A document indexed through a stubbed standard pipeline."""
def fake_page_index_main(doc, opt=None, logger=None, page_list=None):
assert opt.if_add_node_summary == "yes"
assert opt.if_add_node_text == "yes"
assert opt.if_add_node_text == "no"
assert logger is not None
assert page_list is not None
assert all(isinstance(t, tuple) and len(t) == 2 for t in page_list)
Expand Down Expand Up @@ -2532,12 +2532,13 @@ def test_format_tree_node_keeps_key_items():

# ── retry-ladder and summary fail-loud edges (twelfth review) ──

def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch):
@pytest.mark.parametrize("reply", ["", '{"summary": ""}'])
def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch, reply):
"""Empty-content replies (content filter, spent output cap) must not
vouch for the model: a raw-text short leaf cannot carry the run when
every model reply comes back blank."""
async def blank(model, prompt):
return ""
return reply
monkeypatch.setattr(pageindex.utils, "llm_acompletion", blank)
pdf_pages = [("tiny", 1), ("beta " * 300, 300)]
structure = [{"title": "R", "start_index": 1, "end_index": 2,
Expand Down
Loading
Loading