diff --git a/pageindex/agent_tools.py b/pageindex/agent_tools.py index f18abe8de..afc14c9f7 100644 --- a/pageindex/agent_tools.py +++ b/pageindex/agent_tools.py @@ -920,12 +920,9 @@ def _get_document_structure(client, doc_name: str, waited and entry.get("status") != "failed") try: - raw_tree = getattr(getattr(client, "_api", None), "raw_tree", None) - tree = raw_tree(entry["id"]) if raw_tree is not None else None - if tree is None: - # _format_structure strips text anyway — don't download it. - tree = client.get_tree(entry["id"], node_summary=True, - include_text=False).get("result") + # _format_structure strips text anyway — don't download it. + tree = client.get_tree(entry["id"], node_summary=True, + include_text=False).get("result") except PageIndexAPIError as exc: return _failure( f"Failed to retrieve document structure: {exc}", diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 4802634b8..e01431919 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -223,7 +223,7 @@ def _index_standard(self, file_path: str, page_texts: list[str]) -> tuple[list, "summary_model": self._summary_model, "if_add_node_id": "yes", "if_add_node_summary": "yes", - "if_add_node_text": "yes", + "if_add_node_text": "no", "if_add_doc_description": "yes", }) result = page_index_main(file_path, opt, logger=logger, page_list=page_list) @@ -269,10 +269,6 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: add_node_text(structure, pdf_pages) return structure - def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim, every key kept.""" - return self._store.get_tree(doc_id) - def get_tree(self, doc_id: str, node_summary: bool = False, include_text: bool = True) -> dict[str, Any]: meta = self._require_doc(doc_id, "Failed to get tree result") diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index 7788e236b..2ff87fa25 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -1165,13 +1165,17 @@ async def meta_processor(page_list, mode=None, toc_content=None, toc_page_list=N raise Exception('Processing failed') -async def process_large_node_recursively(node, page_list, opt=None, logger=None): +async def process_large_node_recursively(node, page_list, opt=None, logger=None, split=None): + # split: the pages last split above this node; the model can rebuild the + # node over them, e.g. when another heading sits above its own node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node - and token_num >= opt.max_token_num_each_node): + and token_num >= opt.max_token_num_each_node + and (node['start_index'], node['end_index']) != split): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) + split = (node['start_index'], node['end_index']) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) node_toc_tree = await check_title_appearance_in_start_concurrent(node_toc_tree, page_list, model=opt.model, logger=logger) @@ -1190,7 +1194,7 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) if 'nodes' in node and node['nodes']: tasks = [ - process_large_node_recursively(child_node, page_list, opt, logger=logger) + process_large_node_recursively(child_node, page_list, opt, logger=logger, split=split) for child_node in node['nodes'] ] await asyncio.gather(*tasks) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index e607d5f00..64f96fc2f 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -11,7 +11,7 @@ trigger: S(v) > TRIGGER_PAGES (cost control on generation, not the rule) collapse_cost = S(v) - expand_cost = R(v) + max(S_residual(v), max_i S(c_i)) + expand_cost = R(v) + max_i S(c_i) the c_i include the intro attach_children adds expand iff expand_cost < collapse_cost (ties keep collapsed) expand_gain = collapse_cost - expand_cost @@ -132,21 +132,11 @@ async def ask_model(model, prompt): def load_pages(pdf_path): - """Per-page text, and per-page lines ordered top to bottom.""" - import pymupdf - doc = pymupdf.open(pdf_path) - text, lines = [], [] - for page in doc: - text.append(page.get_text()) - ordered = [] - for block in page.get_text("dict")["blocks"]: - for line in block.get("lines", []): - content = "".join(s["text"] for s in line["spans"]).strip() - if content: - ordered.append((line["bbox"][1], content)) - ordered.sort() - lines.append([c for _, c in ordered]) - return text, lines + """Per-page text, and per-page lines in reading order, as flash reads them.""" + from .flash.api import _page_lines + from .flash.main import extract_toc + text = extract_toc(pdf_path, use_embedded_toc=False)["page_texts"] + return text, _page_lines(text) # -------------------------------------------------------------------------- @@ -172,12 +162,14 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): """Is the heading the first line on its page? No when that cannot be told - (a heading with no Latin letter or digit to match), so the page is shared.""" + (a heading with no Latin letter to match, or a first line that a later line + repeats, as a running header does), so the page is shared.""" page = lines[page_no - 1] key = normalize(heading) - if not page or not key: + if not page or not re.search("[a-z]", key): return False - return key in normalize(page[0]) + starts = [(normalize(line) + " ").startswith(key + " ") for line in page] + return starts[0] and not any(starts[1:]) def assign_ends(node, children, lines): @@ -304,14 +296,13 @@ def tree_cost_via_frontier(node, routing=ROUTING_COST): return max(d * routing + s for d, s, _ in entries) if entries else 0 -def expand_cost(node, children, routing=ROUTING_COST): - """Cost after one-step lookahead, children treated as collapsed.""" - covered = set() - for child in children: - covered |= set(range(child["start_index"], child["end_index"] + 1)) - residual = len(pages_of(node) - covered) - scans = [child["end_index"] - child["start_index"] + 1 for child in children] - return routing + max([residual] + scans), residual +def expand_cost(node, children, lines, routing=ROUTING_COST): + """Cost after one-step lookahead, children treated as collapsed, priced with + the intro node attach_children would give them.""" + trial = dict(node, nodes=children) + residual = S_residual(trial) + add_intro_nodes([trial], lines) + return tree_cost(trial, routing), residual # -------------------------------------------------------------------------- @@ -676,6 +667,66 @@ async def propose_children(node, pages, args): return accepted +def same_heading(a, b): + """Whether two titles name one heading: equal once normalized, or equal but + for a leading number only one of them prints. A title with no Latin letter + is its number alone, so it never matches by that second rule.""" + a, b = normalize(a), normalize(b) + bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b)) + return bool(a) and (a == b or (bare_a == bare_b and bool(re.search("[a-z]", bare_a)) + and (a == bare_a or b == bare_b))) + + +def headings(node): + """The headings printed for a node. A same-page fusion keeps them in its + key_items: the summary pass may rewrite the fused title while expand runs.""" + return node["key_items"] if node.get("_same_page") else [node["title"]] + + +def own_children(node, children, lines, known, ancestors, nxt): + """The proposed children printed inside the node's own text. + + Dropped: a heading that already is a node, in the tree as expand found it + (`known`, page -> headings) or made by the node's own ancestors; and on a + page the node shares, anything printed above its heading (else above the + nearest ancestor heading found there) or at and below the next node's (else + its first descendant's found there). Other branches grow concurrently, so + nothing they add is read. A heading not found on its page decides nothing. + """ + start, end = node["start_index"], subtree_end(node) + lineage = [n for a in ancestors for n in [a] + a["nodes"]] + after = [] + while nxt is not None and nxt["start_index"] == end: + after.append(nxt) + nxt = (nxt.get("nodes") or [None])[0] + + def found(page, matches): + page_lines = lines[page - 1] if page <= len(lines) else [] + return [i for i, line in enumerate(page_lines) if matches(line)] + + def is_node(page, title): + titles = known.get(page, []) + [t for n in lineage if n["start_index"] == page + for t in headings(n)] + return any(same_heading(title, t) for t in titles) + + tops = (found(start, lambda line: any(same_heading(line, t) for t in headings(n))) + for n in [node] + ancestors if n["start_index"] == start) + top = next((hits for hits in tops if hits), []) + bottoms = (found(end, lambda line: any(same_heading(line, t) for t in headings(n))) + for n in after) + bottom = next((hits for hits in bottoms if hits), []) + kept = [] + for child in children: + page, key = child["start_index"], normalize(child["title"]) + printed = found(page, lambda line: key in normalize(line)) if key else [] + if (is_node(page, child["title"]) + or page == start and top and printed and printed[-1] < top[0] + or page == end and bottom and printed and printed[0] >= bottom[-1]): + continue + kept.append(child) + return kept + + async def expand(structure, pages, lines, args, log, frozen): """One-step lookahead on every collapsed node over the trigger, recursively. @@ -687,7 +738,7 @@ async def expand(structure, pages, lines, args, log, frozen): changed = False semaphore = asyncio.Semaphore(args.concurrency) - async def proposals_for(node): + async def proposals_for(node, own): """The model half of one node's lookahead: the empty-retry ladder and absorbed errors run inside the task; log entries come back so a node's entries stay contiguous under concurrency.""" @@ -704,12 +755,13 @@ async def proposals_for(node): "decision": "error", "attempt": attempts, "detail": f"{type(exc).__name__}: {exc}"}) continue + proposed = own(proposed) if proposed: llm_candidates.append((f"llm:{attempts}", proposed)) break # an empty answer is retried, not trusted return llm_candidates, attempts, entries - async def process(node): + async def process(node, ancestors, nxt): nonlocal changed if not is_frontier(node) or node.get("node_id") in frozen: return @@ -718,10 +770,13 @@ async def process(node): return # below the trigger, stay collapsed note(args.progress, f" expand {node.get('node_id'):>8} S={span} " f"pages {node['start_index']}-{subtree_end(node)} ...") - llm_candidates, attempts, entries = await proposals_for(node) + + def own(children): + return own_children(node, children, lines, known, ancestors, nxt) + llm_candidates, attempts, entries = await proposals_for(node, own) log.extend(entries) candidates = [] - cached = children_from_cache(node, args.cache, args.kinds) + cached = own(children_from_cache(node, args.cache, args.kinds)) if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) @@ -737,7 +792,7 @@ async def process(node): scored = [] for source, children in candidates: sized = assign_ends(node, children, lines) - cost, residual = expand_cost(node, sized, args.routing) + cost, residual = expand_cost(node, sized, lines, args.routing) scored.append({"source": source, "children": sized, "expand_cost": cost, "S_residual": residual}) scored.sort(key=lambda s: s["expand_cost"]) @@ -777,15 +832,24 @@ async def process(node): merge_same_page([node], log) # settle after the fusion: mark_final snapshots the children, finish() rejects a later change args.settled([node]) - results = await asyncio.gather(*(process(child) - for child in node["nodes"]), + children = node["nodes"] + results = await asyncio.gather(*(process(child, [node] + ancestors, after) + for child, after in zip(children, children[1:] + [nxt])), return_exceptions=True) for result in results: if isinstance(result, BaseException): raise result - results = await asyncio.gather(*(process(node) - for node, _ in flatten(structure)), + # the tree as it stands now: expand only ever adds below a leaf it is + # processing, so the nodes around another leaf never change under it + flat = list(flatten(structure)) + known, ancestry = {}, {} + for node, parent in flat: + known.setdefault(node["start_index"], []).extend(headings(node)) + ancestry[id(node)] = [parent] + ancestry[id(parent)] if parent else [] + results = await asyncio.gather(*(process(node, ancestry[id(node)], after) + for (node, _), after in + zip(flat, [n for n, _ in flat[1:]] + [None])), return_exceptions=True) for result in results: if isinstance(result, BaseException): diff --git a/pageindex/utils.py b/pageindex/utils.py index 4981a63c5..73f5c3170 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -591,9 +591,9 @@ def post_processing(structure, end_physical_index): item['start_index'] = item.get('physical_index') if i < len(structure) - 1: if structure[i + 1].get('appear_start') == 'yes': - item['end_index'] = structure[i + 1]['physical_index']-1 + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']-1) else: - item['end_index'] = structure[i + 1]['physical_index'] + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']) else: item['end_index'] = end_physical_index tree = list_to_tree(structure) @@ -1011,7 +1011,7 @@ async def _ask(self, prompt, prio): self._asked = True async with self._gate.slot(prio): reply = await llm_acompletion(self._model, prompt) - if reply: + if parse_summary(reply): self._answered = True return reply diff --git a/tests/conftest.py b/tests/conftest.py index 04af63745..b6326e46a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -9,8 +9,9 @@ def _llm_key(monkeypatch): def build_pdf(page_texts): - """Build a minimal, uncompressed PDF (one Helvetica line per page) whose - text PyPDF2 can extract. Returns the PDF file bytes.""" + """Build a minimal, uncompressed PDF (one Helvetica line per page, or a + page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns + the PDF file bytes.""" n = len(page_texts) objects = [] kids = " ".join(f"{3 + i} 0 R" for i in range(n)) @@ -23,9 +24,12 @@ def build_pdf(page_texts): f"/Resources << /Font << /F1 {font_obj} 0 R >> >> " f"/Contents {3 + n + i} 0 R >>".encode() ) - for text in page_texts: - safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") - stream = f"BT /F1 12 Tf 72 720 Td ({safe}) Tj ET".encode() + for page in page_texts: + parts = [] + for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: + safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") + parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") + stream = " ".join(parts).encode() objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream)) objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>") diff --git a/tests/test_agent_tools.py b/tests/test_agent_tools.py index 6fa04f6f3..3f1b1c823 100644 --- a/tests/test_agent_tools.py +++ b/tests/test_agent_tools.py @@ -287,6 +287,18 @@ def test_structure_strips_text_and_orders_keys(client, store_path): assert root["nodes"][0]["end_index"] == 1 +def test_structure_shows_the_ranges_get_tree_serves(client, store_path): + # a leaf stored ending before it starts: its successor was judged to open the same page + tree = [{"title": "Doc", "node_id": "0000", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 2, "end_index": 1}, + {"title": "B", "node_id": "0002", "start_index": 2, "end_index": 2}]}] + seed_doc(store_path, "pi-a", "report.pdf", tree=tree) + payload, _ = run(client, "get_document_structure", doc_name="report.pdf") + served = client.get_tree("pi-a", include_text=False)["result"] + assert [(n["start_index"], n["end_index"]) for n in payload["structure"][0]["nodes"]] == [ + (n["start_index"], n["end_index"]) for n in served[0]["nodes"]] == [(2, 2), (2, 2)] + + def test_structure_multipart_pagination(client, store_path): big_tree = [{ "title": f"Chapter {index}", "node_id": f"{index:04d}", diff --git a/tests/test_client.py b/tests/test_client.py index 814415ca6..24a70da82 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -44,7 +44,7 @@ def indexed_doc(local_client, sample_pdf, monkeypatch): """A document indexed through a stubbed standard pipeline.""" def fake_page_index_main(doc, opt=None, logger=None, page_list=None): assert opt.if_add_node_summary == "yes" - assert opt.if_add_node_text == "yes" + assert opt.if_add_node_text == "no" assert logger is not None assert page_list is not None assert all(isinstance(t, tuple) and len(t) == 2 for t in page_list) @@ -2532,12 +2532,13 @@ def test_format_tree_node_keeps_key_items(): # ── retry-ladder and summary fail-loud edges (twelfth review) ── -def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch): +@pytest.mark.parametrize("reply", ["", '{"summary": ""}']) +def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch, reply): """Empty-content replies (content filter, spent output cap) must not vouch for the model: a raw-text short leaf cannot carry the run when every model reply comes back blank.""" async def blank(model, prompt): - return "" + return reply monkeypatch.setattr(pageindex.utils, "llm_acompletion", blank) pdf_pages = [("tiny", 1), ("beta " * 300, 300)] structure = [{"title": "R", "start_index": 1, "end_index": 2, diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 36066a2d5..cee7e4593 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -2,9 +2,12 @@ splitting, and summaries every node gets, a parent's built from its children's.""" import asyncio +import copy import importlib from types import SimpleNamespace +import pytest + import pageindex.flash import pageindex.tree_optimize as tree_optimize import pageindex.utils as utils @@ -34,13 +37,15 @@ def test_intro_node_holds_the_pages_a_parent_opens_with(): ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] - assert tree_optimize.add_intro_nodes(tree, lines) == tree + assert shape(tree_optimize.add_intro_nodes(copy.deepcopy(tree), lines)) == shape(tree) # a standard parent ends where its first child starts; a split intro keeps its title split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ {"title": "Background", "start_index": 12, "end_index": 14}]}] assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) # a heading with nothing to match cannot be placed on its page, so the page is shared assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + # nor can digits alone, which may be a page number + assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") def test_expand_gives_a_split_node_its_intro(monkeypatch): @@ -74,6 +79,276 @@ async def run(): assert all(n["summary"] == "ok" for n in utils._subtree(tree)) +def test_a_running_header_or_a_word_prefix_does_not_open_the_page(): + page = ["4.1. Discriminant Functions 181", "opening words", "4.1. Discriminant Functions"] + assert not tree_optimize.heading_at_page_start([page], 1, "4.1. Discriminant Functions") + assert tree_optimize.heading_at_page_start([page[2:]], 1, "4.1. Discriminant Functions") + assert not tree_optimize.heading_at_page_start([["Filed 03/04/24", "I. Background"]], 1, "I.") + + +def test_expand_prices_a_level_with_the_intro_it_gets(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[10] = body + "\nSub Late\n" + body # mid-page: the intro shares page 11 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub Late", "page": 11}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + # intro 4-11 + Sub Late 11-12 costs 1 + 8 = 9 pages, no better than reading X's 9 + assert "nodes" not in tree[0]["nodes"][1] + + +def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + # mid-page, so Methods runs onto page 12; run in, so only the tree knows the headings below it + pages[11] = body + "\n3 Results. We report three findings.\n3.1 Data\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003", "nodes": [ + {"title": "3.1 Data", "start_index": 12, "end_index": 15, "node_id": "0004"}, + {"title": "3.2 More", "start_index": 16, "end_index": 20, "node_id": "0005"}]}]}] + + replies = [[{"title": "Results", "page": 12}, {"title": "3.1 Data", "page": 12}], + [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}, + {"title": "3.1 Data", "page": 12}]] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: # a reply the filter empties is asked again + return {"subsections": replies.pop(0)} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert not replies + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + +def test_expand_filters_cached_headings_like_proposed_ones(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + cache = {6: [{"title": "Setup", "kind": "section"}], + 12: [{"title": "3 Results", "kind": "section"}]} + + async def propose(model, prompt): + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, cache=cache)) + + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + +def test_expand_hands_each_new_child_its_ancestors_and_next_node(monkeypatch): + body = "body " * 250 + pages = [body] * 30 + pages[3] = "P\nP.a Early\n" + body + pages[6] = "P.b Later\n" + body + pages[9] = body + "\nQ\nQ.0 Overview\n" + body + pages[12] = "Q.i Intro part\n" + body + pages[16] = "Q.1 First\n" + body + pages[19] = "Q.2 Second\n" + body + pages[24] = body + "\nS\nS.1 Part\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 30, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "P", "start_index": 4, "end_index": 25, "node_id": "0002"}, + {"title": "S", "start_index": 25, "end_index": 30, "node_id": "0003"}]}] + replies = { + "P": [("Q", 10)], + # Q, made in this pass, is P's intro's next node: what follows its heading is not the intro's + "P (intro)": [("P.a Early", 4), ("P.b Later", 7), ("Q.0 Overview", 10)], + # Q's next node is P's: S + "Q": [("Q.1 First", 17), ("Q.2 Second", 20), ("S.1 Part", 25)], + # Q's intro sits under Q, an ancestor made in this pass + "Q (intro)": [("Q", 10), ("Q.0 Overview", 10), ("Q.i Intro part", 13)]} + + async def propose(model, prompt): + title = prompt.split("Section title: ")[1].split("\n")[0] + return {"subsections": [{"title": t, "page": p} for t, p in replies.get(title, [])]} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, do_relabel=False)) + + assert shape(tree) == [ + ("R", 1, 30), ("A", 1, 3), ("P", 4, 25), + ("P (intro)", 4, 10), ("P.a Early", 4, 6), ("P.b Later", 7, 10), + ("Q", 10, 25), ("Q (intro)", 10, 16), ("Q.0 Overview", 10, 12), ("Q.i Intro part", 13, 16), + ("Q.1 First", 17, 19), ("Q.2 Second", 20, 25), ("S", 25, 30)] + + +def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): + lines = [["body"] for _ in range(20)] + lines[11] = ["2.3 Limits", "3 Results", "Scope", "3.1 Data", "3.2 Results"] + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + root = {"title": "R", "start_index": 1, "end_index": 20, "nodes": [methods, results]} + known = {} + for node, _ in tree_optimize.flatten([root]): + known.setdefault(node["start_index"], []).append(node["title"]) + + def kept(node, ancestors, nxt, *titles, known=known): + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in + tree_optimize.own_children(node, children, lines, known, ancestors, nxt)] + + # on a shared last page: the next node's heading, and what is printed below it + assert kept(methods, [root], results, "2.3 Limits", "Results", "3.1 Data") == ["2.3 Limits"] + # a heading that already is a node, wherever it sits on the page and in the tree + assert kept(methods, [root], None, "2.3 Limits", "Results") == ["2.3 Limits"] + assert kept(methods, [root], None, "2.3 Limits", "3.1 Data", known={12: ["3.1 Data"]}) == [ + "2.3 Limits"] + # on a shared first page: what is printed above the node's heading; a number tells headings apart + assert kept(results, [root], None, "2.3 Limits", "Results", "3.1 Data", "3.2 Results") == [ + "3.1 Data", "3.2 Results"] + # an intro sits under its parent's heading and above the siblings expand made with it + data = {"title": "3.1 Data", "start_index": 12, "end_index": 20} + intro = {"title": "3 Results (intro)", "start_index": 12, "end_index": 12} + results["nodes"] = [intro, data] + assert kept(intro, [results, root], data, "2.3 Limits", "Results", "Scope", "3.1 Data") == ["Scope"] + # ... and knows them through its lineage when this pass made them + assert kept(intro, [results, root], None, "Results", "Scope", "3.1 Data", known={}) == ["Scope"] + # a next heading not found on the page decides nothing + assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ + "2.3 Limits", "3.2 Results"] + # ... the heading of its first descendant opening that page does + part = {"title": "Part II", "start_index": 12, "end_index": 20, "nodes": [results]} + assert kept(methods, [root], part, "2.3 Limits", "Scope") == ["2.3 Limits"] + + +def test_own_children_errs_toward_keeping_where_a_heading_repeats(): + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + + def kept(node, nxt, page, *titles): + lines = [["body"] for _ in range(20)] + lines[11] = page + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in tree_optimize.own_children(node, children, lines, {}, [], nxt)] + + # a running header repeats the next heading: the heading is its last line + assert kept(methods, results, ["3 Results", "2.4 Tail", "3 Results", "3.1 Data"], + "2.4 Tail", "3.1 Data") == ["2.4 Tail"] + # a child also mentioned above the node's heading: the child is its last line + assert kept(results, None, ["see 3.1 Data", "3 Results", "3.1 Data"], "3.1 Data") == ["3.1 Data"] + # the next heading spelled otherwise, on the very line the proposal is printed on + assert kept(methods, dict(results, title="IV. Results"), ["Tail", "IV. Results"], + "Tail", "Results") == ["Tail"] + # a same-page fusion is found by its headings, not the title the summary pass rewrites + fused = dict(results, title="Rewritten", _same_page=True, key_items=["3 Results"]) + assert kept(methods, fused, ["2.4 Tail", "3 Results", "3.1 Data"], "2.4 Tail", "3.1 Data") == [ + "2.4 Tail"] + + +def test_a_number_alone_does_not_name_a_non_latin_heading(): + # normalize keeps no CJK: "1. 概要" reads as "1", "1.1 背景" as "1 1" + lines = [["body"] for _ in range(20)] + lines[1] = ["1. 概要", "body", "1.1 背景"] + lines[11] = ["1.2 目的", "2. 方法", "2.1 データ", "2.2 手順"] + overview = {"title": "1. 概要", "start_index": 2, "end_index": 12} + method = {"title": "2. 方法", "start_index": 12, "end_index": 20} + known = {2: ["1. 概要"], 12: ["2. 方法"]} + children = [{"title": title, "start_index": page} + for title, page in [("1.1 背景", 2), ("1.2 目的", 12), ("2.1 データ", 12)]] + kept = tree_optimize.own_children(overview, children, lines, known, [], method) + # 1.1 is not "1.", 1.2 is not "2.", and "2.2" is not the line "2." is printed on + assert [c["title"] for c in kept] == ["1.1 背景", "1.2 目的"] + + +def test_a_node_sharing_its_first_page_with_its_parent_owns_only_what_follows_its_heading(): + data = {"title": "3.1 Data", "start_index": 4, "end_index": 12} + analysis = {"title": "3.2 Analysis", "start_index": 12, "end_index": 20} + results = {"title": "3 Results", "start_index": 4, "end_index": 20, "nodes": [data, analysis]} + + def kept(node, nxt, page, *children): + lines = [["body"] for _ in range(20)] + lines[3] = page + return [c["title"] for c in tree_optimize.own_children( + node, [{"title": t, "start_index": p} for t, p in children], lines, {}, [results], nxt)] + + # the parent's heading is printed above the node's: what sits between is the parent's + assert kept(data, analysis, ["3 Results", "Overview", "3.1 Data"], + ("Overview", 4), ("Data sources", 7)) == ["Data sources"] + # ... and a sibling's subsection printed above the node's heading is the sibling's + data["end_index"] = analysis["start_index"] = 4 + assert kept(analysis, None, ["3 Results", "3.1 Data", "3.1.1 Sources", "3.2 Analysis"], + ("3.1.1 Sources", 4), ("Method", 9)) == ["Method"] + + +def test_pdf_lines_read_a_two_column_page_column_by_column(tmp_path): + page = [] + for x, heading, inner in [(72, "2.3 Setup", "Limitations"), (320, "3 Conclusion", None)]: + page.append((x, 712, 11, heading)) + for row in range(6): + y = 692 - row * 14 + page.append((x, y, 11, inner) if inner and row == 3 else + (x, y, 9, "the method is applied to each")) + pdf = tmp_path / "two.pdf" + pdf.write_bytes(build_pdf([page])) + + _, lines = tree_optimize.load_pages(pdf) + + headings = [line for line in lines[0] if line in ("2.3 Setup", "Limitations", "3 Conclusion")] + assert headings == ["2.3 Setup", "Limitations", "3 Conclusion"] + + +@pytest.mark.parametrize("methods_delay", [0, 0.05]) +def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\nlead\n3.1 Data\n" + body + pages[14] = "3.2 Analysis\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + await asyncio.sleep(methods_delay) + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "3.1 Data", "page": 12}]} + if "Section title: 3 Results\n" in prompt: + await asyncio.sleep(0.05 - methods_delay) + return {"subsections": [{"title": "3.1 Data", "page": 12}, {"title": "3.2 Analysis", "page": 15}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1:]) == [ + ("Methods", 4, 12), ("Methods (intro)", 4, 5), ("Setup", 6, 12), + ("3 Results", 12, 20), ("3.1 Data", 12, 14), ("3.2 Analysis", 15, 20)] + + def test_merge_folds_an_intro_without_listing_its_title(): tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ {"title": "P (intro)", "start_index": 1, "end_index": 1}, @@ -82,6 +357,15 @@ def test_merge_folds_an_intro_without_listing_its_title(): assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] +def test_a_toc_item_whose_page_the_next_opens_still_ends_on_its_own_page(): + items = [{"structure": "1", "title": "Overview", "physical_index": 5}, + {"structure": "2", "title": "Scope", "physical_index": 5, "appear_start": "yes"}, + {"structure": "3", "title": "Terms", "physical_index": 9}, + {"structure": "4", "title": "Annex", "physical_index": 8}] # listed out of order + assert shape(utils.post_processing(items, 12)) == [ + ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 9), ("Annex", 8, 12)] + + LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") @@ -116,6 +400,27 @@ async def appear(items, *args, **kwargs): assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] +def test_pages_rebuilt_unchanged_by_a_split_are_not_split_again(monkeypatch): + calls = [] + + async def meta_processor(*args, **kwargs): + calls.append(1) + assert len(calls) < 5, "the same pages are split again and again" + # page 1 holds "Part II" right above the section's own heading + return [{"structure": "1", "title": "Part II", "physical_index": 1}, + {"structure": "1.1", "title": "Ch 3", "physical_index": 1}] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + node = {"title": "Ch 3", "start_index": 1, "end_index": 40} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 40, LARGE)) + + assert len(calls) == 1 + + def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] pdf = tmp_path / "doc.pdf" @@ -158,6 +463,40 @@ async def reply(model, prompt): assert '"summary": "whole C2"' in prompts[-1] +def test_standard_index_gives_toc_and_split_parents_intros(tmp_path, monkeypatch): + texts = [f"page {n}" for n in range(1, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + splits = {1: [{"structure": "1", "title": "P", "physical_index": 1}, # P's long opening + {"structure": "2", "title": "Background", "physical_index": 8}], + 20: [{"structure": "1", "title": "C2", "physical_index": 20}, # the large leaf C2 + {"structure": "2", "title": "C2.a", "physical_index": 25}, + {"structure": "3", "title": "C2.b", "physical_index": 32}]} + + async def meta_processor(page_list, mode=None, start_index=1, **kwargs): + if len(page_list) == len(texts): + return [{"structure": "1", "title": "P", "physical_index": 1}, + {"structure": "1.1", "title": "C1", "physical_index": 15}, + {"structure": "1.2", "title": "C2", "physical_index": 20}] + return splits[start_index] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "check_toc", lambda page_list, opt: {"toc_content": None}) + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_summary": "no"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 3000) for text in texts]) + + assert shape(result["structure"]) == [ + ("P", 1, 40), ("P (intro)", 1, 15), ("P (intro)", 1, 8), ("Background", 8, 15), + ("C1", 15, 20), + ("C2", 20, 40), ("C2 (intro)", 20, 25), ("C2.a", 25, 32), ("C2.b", 32, 40)] + + def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): import pageindex.flash.api as flash_api pdf = tmp_path / "doc.pdf"