From 50caf1582b06e8118afbd380f9c3b4fe5a818c71 Mon Sep 17 00:00:00 2001 From: Ray Date: Sun, 4 Oct 2026 18:07:25 +0800 Subject: [PATCH 1/7] fix: unified-tree follow-ups in flash expand and the local agent tree Flash expand priced a candidate level without the intro attach_children then adds. An intro sharing its first child's page could make the committed node cost exactly its span, so the next merge folded it back after its summary was final, and the whole index raised "dropped or changed after it was marked final" (PRML 5.1, the 2023 annual report). expand_cost now prices the level with its intro. heading_at_page_start took any first line containing the heading as the heading opening its page. A running header repeating the section title cut intros and the Preface a page early, so the opening text above the real heading was in no node. The first line must now start with the heading, no later line may start with it, and a heading with no Latin letter is never placed. An uncertain page is shared. Expand on an intro or the Preface could list the heading printed on its shared last page (the next node's) or atop it (its parent's) as a new child. A proposal whose page and title, leading number aside, match a node already in the tree is dropped. The local get_document_structure tool read the stored tree through LocalAPI.raw_tree and missed get_tree's end < start clamp. raw_tree is gone; both modes read client.get_tree. A test runs the standard tree_parser path end to end: intro insertion before and after the large-node split, and the covering pass on the default tail. --- pageindex/agent_tools.py | 9 ++-- pageindex/local_api.py | 4 -- pageindex/tree_optimize.py | 39 ++++++++++------ tests/test_agent_tools.py | 12 +++++ tests/test_tree_format.py | 91 ++++++++++++++++++++++++++++++++++++++ 5 files changed, 132 insertions(+), 23 deletions(-) diff --git a/pageindex/agent_tools.py b/pageindex/agent_tools.py index f18abe8de..afc14c9f7 100644 --- a/pageindex/agent_tools.py +++ b/pageindex/agent_tools.py @@ -920,12 +920,9 @@ def _get_document_structure(client, doc_name: str, waited and entry.get("status") != "failed") try: - raw_tree = getattr(getattr(client, "_api", None), "raw_tree", None) - tree = raw_tree(entry["id"]) if raw_tree is not None else None - if tree is None: - # _format_structure strips text anyway — don't download it. - tree = client.get_tree(entry["id"], node_summary=True, - include_text=False).get("result") + # _format_structure strips text anyway — don't download it. + tree = client.get_tree(entry["id"], node_summary=True, + include_text=False).get("result") except PageIndexAPIError as exc: return _failure( f"Failed to retrieve document structure: {exc}", diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 4802634b8..9fe63d97e 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -269,10 +269,6 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: add_node_text(structure, pdf_pages) return structure - def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim, every key kept.""" - return self._store.get_tree(doc_id) - def get_tree(self, doc_id: str, node_summary: bool = False, include_text: bool = True) -> dict[str, Any]: meta = self._require_doc(doc_id, "Failed to get tree result") diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index e607d5f00..9842fb350 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -11,7 +11,7 @@ trigger: S(v) > TRIGGER_PAGES (cost control on generation, not the rule) collapse_cost = S(v) - expand_cost = R(v) + max(S_residual(v), max_i S(c_i)) + expand_cost = R(v) + max_i S(c_i) the c_i include the intro attach_children adds expand iff expand_cost < collapse_cost (ties keep collapsed) expand_gain = collapse_cost - expand_cost @@ -172,12 +172,14 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): """Is the heading the first line on its page? No when that cannot be told - (a heading with no Latin letter or digit to match), so the page is shared.""" + (a heading with no Latin letter to match, or a first line that a later line + repeats, as a running header does), so the page is shared.""" page = lines[page_no - 1] key = normalize(heading) - if not page or not key: + if not page or not re.search("[a-z]", key): return False - return key in normalize(page[0]) + starts = [(normalize(line) + " ").startswith(key + " ") for line in page] + return starts[0] and not any(starts[1:]) def assign_ends(node, children, lines): @@ -304,14 +306,13 @@ def tree_cost_via_frontier(node, routing=ROUTING_COST): return max(d * routing + s for d, s, _ in entries) if entries else 0 -def expand_cost(node, children, routing=ROUTING_COST): - """Cost after one-step lookahead, children treated as collapsed.""" - covered = set() - for child in children: - covered |= set(range(child["start_index"], child["end_index"] + 1)) - residual = len(pages_of(node) - covered) - scans = [child["end_index"] - child["start_index"] + 1 for child in children] - return routing + max([residual] + scans), residual +def expand_cost(node, children, lines, routing=ROUTING_COST): + """Cost after one-step lookahead, children treated as collapsed, priced with + the intro node attach_children would give them.""" + trial = dict(node, nodes=children) + residual = S_residual(trial) + add_intro_nodes([trial], lines) + return tree_cost(trial, routing), residual # -------------------------------------------------------------------------- @@ -676,6 +677,13 @@ async def propose_children(node, pages, args): return accepted +def heading_key(node): + """A node's page and heading, compared without the number printed before it; + None when no text is left to compare.""" + title = re.sub(r"^(?:[0-9]+ )+", "", normalize(node["title"])) + return (node["start_index"], title) if title else None + + async def expand(structure, pages, lines, args, log, frozen): """One-step lookahead on every collapsed node over the trigger, recursively. @@ -725,6 +733,11 @@ async def process(node): if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) + # a heading the tree holds already: a neighbor's on a shared page, a parent's atop its intro + taken = {heading_key(n) for n, _ in flatten(structure)} - {None} + candidates = [(source, [c for c in children if heading_key(c) not in taken]) + for source, children in candidates] + candidates = [(source, children) for source, children in candidates if children] if not candidates: note(args.progress, f" -> no children found, kept collapsed") @@ -737,7 +750,7 @@ async def process(node): scored = [] for source, children in candidates: sized = assign_ends(node, children, lines) - cost, residual = expand_cost(node, sized, args.routing) + cost, residual = expand_cost(node, sized, lines, args.routing) scored.append({"source": source, "children": sized, "expand_cost": cost, "S_residual": residual}) scored.sort(key=lambda s: s["expand_cost"]) diff --git a/tests/test_agent_tools.py b/tests/test_agent_tools.py index 6fa04f6f3..3f1b1c823 100644 --- a/tests/test_agent_tools.py +++ b/tests/test_agent_tools.py @@ -287,6 +287,18 @@ def test_structure_strips_text_and_orders_keys(client, store_path): assert root["nodes"][0]["end_index"] == 1 +def test_structure_shows_the_ranges_get_tree_serves(client, store_path): + # a leaf stored ending before it starts: its successor was judged to open the same page + tree = [{"title": "Doc", "node_id": "0000", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 2, "end_index": 1}, + {"title": "B", "node_id": "0002", "start_index": 2, "end_index": 2}]}] + seed_doc(store_path, "pi-a", "report.pdf", tree=tree) + payload, _ = run(client, "get_document_structure", doc_name="report.pdf") + served = client.get_tree("pi-a", include_text=False)["result"] + assert [(n["start_index"], n["end_index"]) for n in payload["structure"][0]["nodes"]] == [ + (n["start_index"], n["end_index"]) for n in served[0]["nodes"]] == [(2, 2), (2, 2)] + + def test_structure_multipart_pagination(client, store_path): big_tree = [{ "title": f"Chapter {index}", "node_id": f"{index:04d}", diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 36066a2d5..c539fe048 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -74,6 +74,63 @@ async def run(): assert all(n["summary"] == "ok" for n in utils._subtree(tree)) +def test_a_running_header_or_a_word_prefix_does_not_open_the_page(): + page = ["4.1. Discriminant Functions 181", "opening words", "4.1. Discriminant Functions"] + assert not tree_optimize.heading_at_page_start([page], 1, "4.1. Discriminant Functions") + assert tree_optimize.heading_at_page_start([page[2:]], 1, "4.1. Discriminant Functions") + assert not tree_optimize.heading_at_page_start([["Filed 03/04/24", "I. Background"]], 1, "I.") + + +def test_expand_prices_a_level_with_the_intro_it_gets(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[10] = body + "\nSub Late\n" + body # mid-page: the intro shares page 11 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub Late", "page": 11}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + # intro 4-11 + Sub Late 11-12 costs 1 + 8 = 9 pages, no better than reading X's 9 + assert "nodes" not in tree[0]["nodes"][1] + + +def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\n" + body # mid-page: Methods runs onto page 12 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + def test_merge_folds_an_intro_without_listing_its_title(): tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ {"title": "P (intro)", "start_index": 1, "end_index": 1}, @@ -158,6 +215,40 @@ async def reply(model, prompt): assert '"summary": "whole C2"' in prompts[-1] +def test_standard_index_gives_toc_and_split_parents_intros(tmp_path, monkeypatch): + texts = [f"page {n}" for n in range(1, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + splits = {1: [{"structure": "1", "title": "P", "physical_index": 1}, # P's long opening + {"structure": "2", "title": "Background", "physical_index": 8}], + 20: [{"structure": "1", "title": "C2", "physical_index": 20}, # the large leaf C2 + {"structure": "2", "title": "C2.a", "physical_index": 25}, + {"structure": "3", "title": "C2.b", "physical_index": 32}]} + + async def meta_processor(page_list, mode=None, start_index=1, **kwargs): + if len(page_list) == len(texts): + return [{"structure": "1", "title": "P", "physical_index": 1}, + {"structure": "1.1", "title": "C1", "physical_index": 15}, + {"structure": "1.2", "title": "C2", "physical_index": 20}] + return splits[start_index] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "check_toc", lambda page_list, opt: {"toc_content": None}) + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_summary": "no"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 3000) for text in texts]) + + assert shape(result["structure"]) == [ + ("P", 1, 40), ("P (intro)", 1, 15), ("P (intro)", 1, 8), ("Background", 8, 15), + ("C1", 15, 20), + ("C2", 20, 40), ("C2 (intro)", 20, 25), ("C2.a", 25, 32), ("C2.b", 32, 40)] + + def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): import pageindex.flash.api as flash_api pdf = tmp_path / "doc.pdf" From ff4cfd0989e3732e43ddad390a2c3f014a04d344 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 15:16:12 +0800 Subject: [PATCH 2/7] fix: expand keeps a heading for the node it is printed under 50caf15 dropped a proposed child whose page and title, leading number aside, matched any node in the live tree. Sibling expansions run concurrently, so the result hung on reply order: when Methods' reply landed before Results', Methods kept "3.1 Discussion" from their shared page and Results lost its own. Stripping every leading number also took "3.2 Results" for its parent "3 Results" and dropped it. A proposal is now dropped when it already is a node, in the tree as expand found it or made by the node's own ancestors, which no concurrent branch can change; or when, on a page the node shares, it is printed above the node's heading or at and below the next node's. A heading not found on its page decides nothing. A leading number is ignored only when one of the two titles lacks it. The check runs before the empty-reply retry, and the tree is read once per pass instead of once per node. A summary reply counted as answered whenever it was non-empty, so a model returning {"summary": ""} everywhere stored a document of fallback summaries without the all-failed error. It now counts only when a summary parses out of it. post_processing ended a TOC item a page before its start when the next item opened the same page; the end is now at least the start. --- pageindex/tree_optimize.py | 89 ++++++++++++++++++++++++++++++-------- pageindex/utils.py | 6 +-- tests/test_client.py | 5 ++- tests/test_tree_format.py | 70 ++++++++++++++++++++++++++++++ 4 files changed, 146 insertions(+), 24 deletions(-) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index 9842fb350..5537993ed 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -677,11 +677,55 @@ async def propose_children(node, pages, args): return accepted -def heading_key(node): - """A node's page and heading, compared without the number printed before it; - None when no text is left to compare.""" - title = re.sub(r"^(?:[0-9]+ )+", "", normalize(node["title"])) - return (node["start_index"], title) if title else None +def same_heading(a, b): + """Whether two titles name one heading: equal once normalized, or equal but + for a leading number only one of them prints.""" + a, b = normalize(a), normalize(b) + bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b)) + return bool(a) and (a == b or (bare_a == bare_b and (a == bare_a or b == bare_b))) + + +def headings(node): + """The headings printed for a node. A same-page fusion keeps them in its + key_items: the summary pass may rewrite the fused title while expand runs.""" + return node["key_items"] if node.get("_same_page") else [node["title"]] + + +def own_children(node, children, lines, known, ancestors, nxt): + """The proposed children printed inside the node's own text. + + Dropped: a heading that already is a node, in the tree as expand found it + (`known`, page -> headings) or made by the node's own ancestors; and on a + page the node shares, anything printed above its heading or at and below + the next node's. Other branches grow concurrently, so nothing they add is + read. A heading not found on its page decides nothing. + """ + start, end = node["start_index"], subtree_end(node) + lineage = [n for a in ancestors for n in [a] + a["nodes"]] + above = [t for n in [node] + ancestors if n["start_index"] == start for t in headings(n)] + after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] + + def found(page, matches): + page_lines = lines[page - 1] if lines and page <= len(lines) else [] + return [i for i, line in enumerate(page_lines) if matches(line)] + + def is_node(page, title): + titles = known.get(page, []) + [t for n in lineage if n["start_index"] == page + for t in headings(n)] + return any(same_heading(title, t) for t in titles) + + top = found(start, lambda line: any(same_heading(line, t) for t in above)) + bottom = found(end, lambda line: any(same_heading(line, t) for t in after)) if after else [] + kept = [] + for child in children: + page, key = child["start_index"], normalize(child["title"]) + printed = found(page, lambda line: key in normalize(line)) if key else [] + if (is_node(page, child["title"]) + or page == start and top and printed and printed[-1] < top[0] + or page == end and bottom and printed and printed[0] >= bottom[-1]): + continue + kept.append(child) + return kept async def expand(structure, pages, lines, args, log, frozen): @@ -695,7 +739,7 @@ async def expand(structure, pages, lines, args, log, frozen): changed = False semaphore = asyncio.Semaphore(args.concurrency) - async def proposals_for(node): + async def proposals_for(node, own): """The model half of one node's lookahead: the empty-retry ladder and absorbed errors run inside the task; log entries come back so a node's entries stay contiguous under concurrency.""" @@ -704,7 +748,7 @@ async def proposals_for(node): attempts += 1 try: async with semaphore: - proposed = await propose_children(node, pages, args) + proposed = own(await propose_children(node, pages, args)) except Exception as exc: if _is_unrecoverable(exc): raise # every remaining node would fail identically @@ -717,7 +761,7 @@ async def proposals_for(node): break # an empty answer is retried, not trusted return llm_candidates, attempts, entries - async def process(node): + async def process(node, ancestors, nxt): nonlocal changed if not is_frontier(node) or node.get("node_id") in frozen: return @@ -726,18 +770,16 @@ async def process(node): return # below the trigger, stay collapsed note(args.progress, f" expand {node.get('node_id'):>8} S={span} " f"pages {node['start_index']}-{subtree_end(node)} ...") - llm_candidates, attempts, entries = await proposals_for(node) + + def own(children): + return own_children(node, children, lines, known, ancestors, nxt) + llm_candidates, attempts, entries = await proposals_for(node, own) log.extend(entries) candidates = [] - cached = children_from_cache(node, args.cache, args.kinds) + cached = own(children_from_cache(node, args.cache, args.kinds)) if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) - # a heading the tree holds already: a neighbor's on a shared page, a parent's atop its intro - taken = {heading_key(n) for n, _ in flatten(structure)} - {None} - candidates = [(source, [c for c in children if heading_key(c) not in taken]) - for source, children in candidates] - candidates = [(source, children) for source, children in candidates if children] if not candidates: note(args.progress, f" -> no children found, kept collapsed") @@ -790,15 +832,24 @@ async def process(node): merge_same_page([node], log) # settle after the fusion: mark_final snapshots the children, finish() rejects a later change args.settled([node]) - results = await asyncio.gather(*(process(child) - for child in node["nodes"]), + children = node["nodes"] + results = await asyncio.gather(*(process(child, [node] + ancestors, after) + for child, after in zip(children, children[1:] + [nxt])), return_exceptions=True) for result in results: if isinstance(result, BaseException): raise result - results = await asyncio.gather(*(process(node) - for node, _ in flatten(structure)), + # the tree as it stands now: expand only ever adds below a leaf it is + # processing, so the nodes around another leaf never change under it + flat = list(flatten(structure)) + known, ancestry = {}, {} + for node, parent in flat: + known.setdefault(node["start_index"], []).extend(headings(node)) + ancestry[id(node)] = [parent] + ancestry[id(parent)] if parent else [] + results = await asyncio.gather(*(process(node, ancestry[id(node)], after) + for (node, _), after in + zip(flat, [n for n, _ in flat[1:]] + [None])), return_exceptions=True) for result in results: if isinstance(result, BaseException): diff --git a/pageindex/utils.py b/pageindex/utils.py index 4981a63c5..73f5c3170 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -591,9 +591,9 @@ def post_processing(structure, end_physical_index): item['start_index'] = item.get('physical_index') if i < len(structure) - 1: if structure[i + 1].get('appear_start') == 'yes': - item['end_index'] = structure[i + 1]['physical_index']-1 + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']-1) else: - item['end_index'] = structure[i + 1]['physical_index'] + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']) else: item['end_index'] = end_physical_index tree = list_to_tree(structure) @@ -1011,7 +1011,7 @@ async def _ask(self, prompt, prio): self._asked = True async with self._gate.slot(prio): reply = await llm_acompletion(self._model, prompt) - if reply: + if parse_summary(reply): self._answered = True return reply diff --git a/tests/test_client.py b/tests/test_client.py index 814415ca6..6cceb98cc 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -2532,12 +2532,13 @@ def test_format_tree_node_keeps_key_items(): # ── retry-ladder and summary fail-loud edges (twelfth review) ── -def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch): +@pytest.mark.parametrize("reply", ["", '{"summary": ""}']) +def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch, reply): """Empty-content replies (content filter, spent output cap) must not vouch for the model: a raw-text short leaf cannot carry the run when every model reply comes back blank.""" async def blank(model, prompt): - return "" + return reply monkeypatch.setattr(pageindex.utils, "llm_acompletion", blank) pdf_pages = [("tiny", 1), ("beta " * 300, 300)] structure = [{"title": "R", "start_index": 1, "end_index": 2, diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index c539fe048..81dc5776d 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -5,6 +5,8 @@ import importlib from types import SimpleNamespace +import pytest + import pageindex.flash import pageindex.tree_optimize as tree_optimize import pageindex.utils as utils @@ -131,6 +133,66 @@ async def propose(model, prompt): assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] +def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): + lines = [["body"] for _ in range(20)] + lines[11] = ["2.3 Limits", "3 Results", "Scope", "3.1 Data", "3.2 Results"] + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + root = {"title": "R", "start_index": 1, "end_index": 20, "nodes": [methods, results]} + known = {} + for node, _ in tree_optimize.flatten([root]): + known.setdefault(node["start_index"], []).append(node["title"]) + + def kept(node, ancestors, nxt, *titles): + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in + tree_optimize.own_children(node, children, lines, known, ancestors, nxt)] + + # on a shared last page: the next node's heading, and what is printed below it + assert kept(methods, [root], results, "2.3 Limits", "Results", "3.1 Data") == ["2.3 Limits"] + # on a shared first page: what is printed above the node's heading; a number tells headings apart + assert kept(results, [root], None, "2.3 Limits", "Results", "3.1 Data", "3.2 Results") == [ + "3.1 Data", "3.2 Results"] + # an intro sits under its parent's heading and above the siblings expand made with it + data = {"title": "3.1 Data", "start_index": 12, "end_index": 20} + intro = {"title": "3 Results (intro)", "start_index": 12, "end_index": 12} + results["nodes"] = [intro, data] + assert kept(intro, [results, root], data, "2.3 Limits", "Results", "Scope", "3.1 Data") == ["Scope"] + # a next heading not found on the page decides nothing + assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ + "2.3 Limits", "3.2 Results"] + + +@pytest.mark.parametrize("methods_delay", [0, 0.05]) +def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\nlead\n3.1 Data\n" + body + pages[14] = "3.2 Analysis\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + await asyncio.sleep(methods_delay) + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "3.1 Data", "page": 12}]} + if "Section title: 3 Results\n" in prompt: + await asyncio.sleep(0.05 - methods_delay) + return {"subsections": [{"title": "3.1 Data", "page": 12}, {"title": "3.2 Analysis", "page": 15}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1:]) == [ + ("Methods", 4, 12), ("Methods (intro)", 4, 5), ("Setup", 6, 12), + ("3 Results", 12, 20), ("3.1 Data", 12, 14), ("3.2 Analysis", 15, 20)] + + def test_merge_folds_an_intro_without_listing_its_title(): tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ {"title": "P (intro)", "start_index": 1, "end_index": 1}, @@ -139,6 +201,14 @@ def test_merge_folds_an_intro_without_listing_its_title(): assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] +def test_a_toc_item_whose_page_the_next_opens_still_ends_on_its_own_page(): + items = [{"structure": "1", "title": "Overview", "physical_index": 5}, + {"structure": "2", "title": "Scope", "physical_index": 5, "appear_start": "yes"}, + {"structure": "3", "title": "Terms", "physical_index": 9}] + assert shape(utils.post_processing(items, 12)) == [ + ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 12)] + + LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") From cdcaf65842b594502f905cd1b752ed320d318653 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 15:20:34 +0800 Subject: [PATCH 3/7] test: pin the digit-only heading rule and intro idempotency heading_at_page_start never places a heading without a Latin letter, so a bare "2" (often a page number) shares its page; nothing tested that. The add_intro_nodes idempotency check compared the tree with itself, the same object the call returns, so it could not fail; it now compares a second pass over a copy with the first. --- tests/test_tree_format.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 81dc5776d..afd4ff84f 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -2,6 +2,7 @@ splitting, and summaries every node gets, a parent's built from its children's.""" import asyncio +import copy import importlib from types import SimpleNamespace @@ -36,13 +37,14 @@ def test_intro_node_holds_the_pages_a_parent_opens_with(): ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] - assert tree_optimize.add_intro_nodes(tree, lines) == tree + assert shape(tree_optimize.add_intro_nodes(copy.deepcopy(tree), lines)) == shape(tree) # a standard parent ends where its first child starts; a split intro keeps its title split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ {"title": "Background", "start_index": 12, "end_index": 14}]}] assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) # a heading with nothing to match cannot be placed on its page, so the page is shared assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") # or a page number def test_expand_gives_a_split_node_its_intro(monkeypatch): From c29ffa7024bcda4cc33ef97295cc1de54a479b48 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 15:33:23 +0800 Subject: [PATCH 4/7] fix: expand filters proposals outside the model-error handler The ownership filter ran inside the try that absorbs failed model calls, so a bug in it would be logged as a model error, retried, and leave the node silently unexpanded. It now runs after the call, where an error surfaces. The neighbor test now also pins the retry: a reply the filter empties is asked again. own_children drops a guard for lines expand always passes. --- pageindex/tree_optimize.py | 5 +++-- tests/test_tree_format.py | 11 ++++++++--- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index 5537993ed..6f97ca083 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -706,7 +706,7 @@ def own_children(node, children, lines, known, ancestors, nxt): after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] def found(page, matches): - page_lines = lines[page - 1] if lines and page <= len(lines) else [] + page_lines = lines[page - 1] if page <= len(lines) else [] return [i for i, line in enumerate(page_lines) if matches(line)] def is_node(page, title): @@ -748,7 +748,7 @@ async def proposals_for(node, own): attempts += 1 try: async with semaphore: - proposed = own(await propose_children(node, pages, args)) + proposed = await propose_children(node, pages, args) except Exception as exc: if _is_unrecoverable(exc): raise # every remaining node would fail identically @@ -756,6 +756,7 @@ async def proposals_for(node, own): "decision": "error", "attempt": attempts, "detail": f"{type(exc).__name__}: {exc}"}) continue + proposed = own(proposed) if proposed: llm_candidates.append((f"llm:{attempts}", proposed)) break # an empty answer is retried, not trusted diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index afd4ff84f..882587c4e 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -44,7 +44,8 @@ def test_intro_node_holds_the_pages_a_parent_opens_with(): assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) # a heading with nothing to match cannot be placed on its page, so the page is shared assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") - assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") # or a page number + # nor can digits alone, which may be a page number + assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") def test_expand_gives_a_split_node_its_intro(monkeypatch): @@ -124,14 +125,18 @@ def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatc {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + replies = [[{"title": "Results", "page": 12}], + [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}]] + async def propose(model, prompt): - if "Section title: Methods\n" in prompt: - return {"subsections": [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}]} + if "Section title: Methods\n" in prompt: # a reply the filter empties is asked again + return {"subsections": replies.pop(0)} return {"subsections": []} monkeypatch.setattr(tree_optimize, "ask_model", propose) asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + assert not replies assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] From 0be23523566d26400873272bb08a374d8a31e1f3 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 16:09:58 +0800 Subject: [PATCH 5/7] test: pin every part of expand's ownership filter A mutation pass over the filter found 13 of 21 mutants surviving the suite. Each part now has a test that fails without it: the snapshot of known nodes (Results' run-in heading leaves only the tree to recognize its subsection), the lineage of nodes this pass made, the ancestors and next node each new child is handed, filtering of cached headings, the occurrence each anchor reads (last line of a repeated next heading, last line of a child mentioned above), `>=` for a variant printed on the next heading's own line, key_items for a same-page fusion, and the clamp for a TOC item listed out of order. --- tests/test_tree_format.py | 109 +++++++++++++++++++++++++++++++++++--- 1 file changed, 102 insertions(+), 7 deletions(-) diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 882587c4e..37c61ad24 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -118,15 +118,19 @@ def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatc body = "body " * 250 pages = [body] * 20 pages[5] = "Setup\n" + body - pages[11] = body + "\n3 Results\n" + body # mid-page: Methods runs onto page 12 + # mid-page, so Methods runs onto page 12; run in, so only the tree knows the headings below it + pages[11] = body + "\n3 Results. We report three findings.\n3.1 Data\n" + body lines = [[line for line in page.splitlines() if line.strip()] for page in pages] tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, - {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003", "nodes": [ + {"title": "3.1 Data", "start_index": 12, "end_index": 15, "node_id": "0004"}, + {"title": "3.2 More", "start_index": 16, "end_index": 20, "node_id": "0005"}]}]}] - replies = [[{"title": "Results", "page": 12}], - [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}]] + replies = [[{"title": "Results", "page": 12}, {"title": "3.1 Data", "page": 12}], + [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}, + {"title": "3.1 Data", "page": 12}]] async def propose(model, prompt): if "Section title: Methods\n" in prompt: # a reply the filter empties is asked again @@ -140,6 +144,66 @@ async def propose(model, prompt): assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] +def test_expand_filters_cached_headings_like_proposed_ones(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + cache = {6: [{"title": "Setup", "kind": "section"}], + 12: [{"title": "3 Results", "kind": "section"}]} + + async def propose(model, prompt): + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, cache=cache)) + + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + +def test_expand_hands_each_new_child_its_ancestors_and_next_node(monkeypatch): + body = "body " * 250 + pages = [body] * 30 + pages[3] = "P\nP.a Early\n" + body + pages[6] = "P.b Later\n" + body + pages[9] = body + "\nQ\nQ.0 Overview\n" + body + pages[12] = "Q.i Intro part\n" + body + pages[16] = "Q.1 First\n" + body + pages[19] = "Q.2 Second\n" + body + pages[24] = body + "\nS\nS.1 Part\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 30, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "P", "start_index": 4, "end_index": 25, "node_id": "0002"}, + {"title": "S", "start_index": 25, "end_index": 30, "node_id": "0003"}]}] + replies = { + "P": [("Q", 10)], + # Q, made in this pass, is P's intro's next node: what follows its heading is not the intro's + "P (intro)": [("P.a Early", 4), ("P.b Later", 7), ("Q.0 Overview", 10)], + # Q's next node is P's: S + "Q": [("Q.1 First", 17), ("Q.2 Second", 20), ("S.1 Part", 25)], + # Q's intro sits under Q, an ancestor made in this pass + "Q (intro)": [("Q", 10), ("Q.0 Overview", 10), ("Q.i Intro part", 13)]} + + async def propose(model, prompt): + title = prompt.split("Section title: ")[1].split("\n")[0] + return {"subsections": [{"title": t, "page": p} for t, p in replies.get(title, [])]} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, do_relabel=False)) + + assert shape(tree) == [ + ("R", 1, 30), ("A", 1, 3), ("P", 4, 25), + ("P (intro)", 4, 10), ("P.a Early", 4, 6), ("P.b Later", 7, 10), + ("Q", 10, 25), ("Q (intro)", 10, 16), ("Q.0 Overview", 10, 12), ("Q.i Intro part", 13, 16), + ("Q.1 First", 17, 19), ("Q.2 Second", 20, 25), ("S", 25, 30)] + + def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): lines = [["body"] for _ in range(20)] lines[11] = ["2.3 Limits", "3 Results", "Scope", "3.1 Data", "3.2 Results"] @@ -150,13 +214,17 @@ def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): for node, _ in tree_optimize.flatten([root]): known.setdefault(node["start_index"], []).append(node["title"]) - def kept(node, ancestors, nxt, *titles): + def kept(node, ancestors, nxt, *titles, known=known): children = [{"title": title, "start_index": 12} for title in titles] return [c["title"] for c in tree_optimize.own_children(node, children, lines, known, ancestors, nxt)] # on a shared last page: the next node's heading, and what is printed below it assert kept(methods, [root], results, "2.3 Limits", "Results", "3.1 Data") == ["2.3 Limits"] + # a heading that already is a node, wherever it sits on the page and in the tree + assert kept(methods, [root], None, "2.3 Limits", "Results") == ["2.3 Limits"] + assert kept(methods, [root], None, "2.3 Limits", "3.1 Data", known={12: ["3.1 Data"]}) == [ + "2.3 Limits"] # on a shared first page: what is printed above the node's heading; a number tells headings apart assert kept(results, [root], None, "2.3 Limits", "Results", "3.1 Data", "3.2 Results") == [ "3.1 Data", "3.2 Results"] @@ -165,11 +233,37 @@ def kept(node, ancestors, nxt, *titles): intro = {"title": "3 Results (intro)", "start_index": 12, "end_index": 12} results["nodes"] = [intro, data] assert kept(intro, [results, root], data, "2.3 Limits", "Results", "Scope", "3.1 Data") == ["Scope"] + # ... and knows them through its lineage when this pass made them + assert kept(intro, [results, root], None, "Results", "Scope", "3.1 Data", known={}) == ["Scope"] # a next heading not found on the page decides nothing assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ "2.3 Limits", "3.2 Results"] +def test_own_children_errs_toward_keeping_where_a_heading_repeats(): + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + + def kept(node, nxt, page, *titles): + lines = [["body"] for _ in range(20)] + lines[11] = page + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in tree_optimize.own_children(node, children, lines, {}, [], nxt)] + + # a running header repeats the next heading: the heading is its last line + assert kept(methods, results, ["3 Results", "2.4 Tail", "3 Results", "3.1 Data"], + "2.4 Tail", "3.1 Data") == ["2.4 Tail"] + # a child also mentioned above the node's heading: the child is its last line + assert kept(results, None, ["see 3.1 Data", "3 Results", "3.1 Data"], "3.1 Data") == ["3.1 Data"] + # the next heading spelled otherwise, on the very line the proposal is printed on + assert kept(methods, dict(results, title="IV. Results"), ["Tail", "IV. Results"], + "Tail", "Results") == ["Tail"] + # a same-page fusion is found by its headings, not the title the summary pass rewrites + fused = dict(results, title="Rewritten", _same_page=True, key_items=["3 Results"]) + assert kept(methods, fused, ["2.4 Tail", "3 Results", "3.1 Data"], "2.4 Tail", "3.1 Data") == [ + "2.4 Tail"] + + @pytest.mark.parametrize("methods_delay", [0, 0.05]) def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): body = "body " * 250 @@ -211,9 +305,10 @@ def test_merge_folds_an_intro_without_listing_its_title(): def test_a_toc_item_whose_page_the_next_opens_still_ends_on_its_own_page(): items = [{"structure": "1", "title": "Overview", "physical_index": 5}, {"structure": "2", "title": "Scope", "physical_index": 5, "appear_start": "yes"}, - {"structure": "3", "title": "Terms", "physical_index": 9}] + {"structure": "3", "title": "Terms", "physical_index": 9}, + {"structure": "4", "title": "Annex", "physical_index": 8}] # listed out of order assert shape(utils.post_processing(items, 12)) == [ - ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 12)] + ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 9), ("Annex", 8, 12)] LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") From 287388331da61ba9f827222384902dc529b92549 Mon Sep 17 00:00:00 2001 From: Ray Date: Tue, 6 Oct 2026 20:21:27 +0800 Subject: [PATCH 6/7] fix: expand keeps non-Latin subsections and cuts at the node's own heading MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit normalize keeps only Latin letters and digits, so "1. 概要" reads as "1" and "1.1 背景" as "1 1". same_heading took the bare "1" for an unnumbered title and matched the pair: a CJK chapter dropped its first subsection as already a node, a "1.2" heading on the page where "2." opens was dropped the same way, and a "2.2" line passed for the next node's "2." heading, so the next node's "2.1", printed above it, stayed. Ignoring a leading number only one title prints now needs a Latin letter to compare. On a node's shared first page, the cut for "printed above the node's heading" took the first line matching the node or any ancestor that starts there. With the parent's heading printed above the node's, what sits between (the parent's opening, or a sibling's subsection) was kept as the node's child. The cut now reads the nearest heading found on the page: the node's own, else its parent's, and so on up. --- pageindex/tree_optimize.py | 18 +++++++++++------- tests/test_tree_format.py | 35 +++++++++++++++++++++++++++++++++++ 2 files changed, 46 insertions(+), 7 deletions(-) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index 6f97ca083..c5b39c080 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -679,10 +679,12 @@ async def propose_children(node, pages, args): def same_heading(a, b): """Whether two titles name one heading: equal once normalized, or equal but - for a leading number only one of them prints.""" + for a leading number only one of them prints. A title with no Latin letter + is its number alone, so it never matches by that second rule.""" a, b = normalize(a), normalize(b) bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b)) - return bool(a) and (a == b or (bare_a == bare_b and (a == bare_a or b == bare_b))) + return bool(a) and (a == b or (bare_a == bare_b and bool(re.search("[a-z]", bare_a)) + and (a == bare_a or b == bare_b))) def headings(node): @@ -696,13 +698,13 @@ def own_children(node, children, lines, known, ancestors, nxt): Dropped: a heading that already is a node, in the tree as expand found it (`known`, page -> headings) or made by the node's own ancestors; and on a - page the node shares, anything printed above its heading or at and below - the next node's. Other branches grow concurrently, so nothing they add is - read. A heading not found on its page decides nothing. + page the node shares, anything printed above its heading (else above the + nearest ancestor heading found there) or at and below the next node's. + Other branches grow concurrently, so nothing they add is read. A heading + not found on its page decides nothing. """ start, end = node["start_index"], subtree_end(node) lineage = [n for a in ancestors for n in [a] + a["nodes"]] - above = [t for n in [node] + ancestors if n["start_index"] == start for t in headings(n)] after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] def found(page, matches): @@ -714,7 +716,9 @@ def is_node(page, title): for t in headings(n)] return any(same_heading(title, t) for t in titles) - top = found(start, lambda line: any(same_heading(line, t) for t in above)) + tops = (found(start, lambda line: any(same_heading(line, t) for t in headings(n))) + for n in [node] + ancestors if n["start_index"] == start) + top = next((hits for hits in tops if hits), []) bottom = found(end, lambda line: any(same_heading(line, t) for t in after)) if after else [] kept = [] for child in children: diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 37c61ad24..d65989fba 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -264,6 +264,41 @@ def kept(node, nxt, page, *titles): "2.4 Tail"] +def test_a_number_alone_does_not_name_a_non_latin_heading(): + # normalize keeps no CJK: "1. 概要" reads as "1", "1.1 背景" as "1 1" + lines = [["body"] for _ in range(20)] + lines[1] = ["1. 概要", "body", "1.1 背景"] + lines[11] = ["1.2 目的", "2. 方法", "2.1 データ", "2.2 手順"] + overview = {"title": "1. 概要", "start_index": 2, "end_index": 12} + method = {"title": "2. 方法", "start_index": 12, "end_index": 20} + known = {2: ["1. 概要"], 12: ["2. 方法"]} + children = [{"title": title, "start_index": page} + for title, page in [("1.1 背景", 2), ("1.2 目的", 12), ("2.1 データ", 12)]] + kept = tree_optimize.own_children(overview, children, lines, known, [], method) + # 1.1 is not "1.", 1.2 is not "2.", and "2.2" is not the line "2." is printed on + assert [c["title"] for c in kept] == ["1.1 背景", "1.2 目的"] + + +def test_a_node_sharing_its_first_page_with_its_parent_owns_only_what_follows_its_heading(): + data = {"title": "3.1 Data", "start_index": 4, "end_index": 12} + analysis = {"title": "3.2 Analysis", "start_index": 12, "end_index": 20} + results = {"title": "3 Results", "start_index": 4, "end_index": 20, "nodes": [data, analysis]} + + def kept(node, nxt, page, *children): + lines = [["body"] for _ in range(20)] + lines[3] = page + return [c["title"] for c in tree_optimize.own_children( + node, [{"title": t, "start_index": p} for t, p in children], lines, {}, [results], nxt)] + + # the parent's heading is printed above the node's: what sits between is the parent's + assert kept(data, analysis, ["3 Results", "Overview", "3.1 Data"], + ("Overview", 4), ("Data sources", 7)) == ["Data sources"] + # ... and a sibling's subsection printed above the node's heading is the sibling's + data["end_index"] = analysis["start_index"] = 4 + assert kept(analysis, None, ["3 Results", "3.1 Data", "3.1.1 Sources", "3.2 Analysis"], + ("3.1.1 Sources", 4), ("Method", 9)) == ["Method"] + + @pytest.mark.parametrize("methods_delay", [0, 0.05]) def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): body = "body " * 250 From c74e1920ac7cdcae9131265db9ab0bbe8a27e11e Mon Sep 17 00:00:00 2001 From: Ray Date: Wed, 7 Oct 2026 02:06:20 +0800 Subject: [PATCH 7/7] fix: CLI expand reads pages as flash does, plus three smaller tree fixes load_pages, the reader behind optimize_tree(pdf_path) and the tree_optimize CLI, sorted each page's lines by height across columns. On a two-column page a subsection low in the left column landed after the next section's heading at the top of the right column, and expand's line cuts dropped it: 29 of 694 subsections on generated two-column papers. It now takes flash's own page text, read in layout order, so both entry points see the same lines. pymupdf's raw stream order was not enough: the 9/11 report draws a mid-page heading before the text above it. Reading a 1,100-page book takes about 30 s (5 s before), and the CLI no longer needs pymupdf. On a page a node shares with the next one, the cut anchored only on the next node's own heading and kept everything when that heading was not found (an unprinted bookmark title, a heading split by extraction). It now falls back to the heading of the next node's first descendant that starts on that page. In Murphy's ML book "23.2.1 Using the cdf" no longer hangs under "22.6.5". process_large_node_recursively could split the same pages forever when the model listed another heading above the section's own ("PART II" above "Chapter 3"): the rebuilt copy was again a large leaf. A leaf whose range equals the range last split above it now stays a leaf, as in compute. Local standard indexing asked page_index_main for node text that nothing reads any more and that the store strips before saving. It no longer builds it; the stored tree and every prompt are unchanged. build_pdf in tests/conftest.py also takes a page as (x, y, size, text) lines, so the two-column test runs without pymupdf. --- pageindex/local_api.py | 2 +- pageindex/page_index_classic.py | 10 +++++--- pageindex/tree_optimize.py | 35 ++++++++++++---------------- tests/conftest.py | 14 +++++++---- tests/test_client.py | 2 +- tests/test_tree_format.py | 41 +++++++++++++++++++++++++++++++++ 6 files changed, 74 insertions(+), 30 deletions(-) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 9fe63d97e..e01431919 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -223,7 +223,7 @@ def _index_standard(self, file_path: str, page_texts: list[str]) -> tuple[list, "summary_model": self._summary_model, "if_add_node_id": "yes", "if_add_node_summary": "yes", - "if_add_node_text": "yes", + "if_add_node_text": "no", "if_add_doc_description": "yes", }) result = page_index_main(file_path, opt, logger=logger, page_list=page_list) diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index 7788e236b..2ff87fa25 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -1165,13 +1165,17 @@ async def meta_processor(page_list, mode=None, toc_content=None, toc_page_list=N raise Exception('Processing failed') -async def process_large_node_recursively(node, page_list, opt=None, logger=None): +async def process_large_node_recursively(node, page_list, opt=None, logger=None, split=None): + # split: the pages last split above this node; the model can rebuild the + # node over them, e.g. when another heading sits above its own node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node - and token_num >= opt.max_token_num_each_node): + and token_num >= opt.max_token_num_each_node + and (node['start_index'], node['end_index']) != split): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) + split = (node['start_index'], node['end_index']) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) node_toc_tree = await check_title_appearance_in_start_concurrent(node_toc_tree, page_list, model=opt.model, logger=logger) @@ -1190,7 +1194,7 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) if 'nodes' in node and node['nodes']: tasks = [ - process_large_node_recursively(child_node, page_list, opt, logger=logger) + process_large_node_recursively(child_node, page_list, opt, logger=logger, split=split) for child_node in node['nodes'] ] await asyncio.gather(*tasks) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index c5b39c080..64f96fc2f 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -132,21 +132,11 @@ async def ask_model(model, prompt): def load_pages(pdf_path): - """Per-page text, and per-page lines ordered top to bottom.""" - import pymupdf - doc = pymupdf.open(pdf_path) - text, lines = [], [] - for page in doc: - text.append(page.get_text()) - ordered = [] - for block in page.get_text("dict")["blocks"]: - for line in block.get("lines", []): - content = "".join(s["text"] for s in line["spans"]).strip() - if content: - ordered.append((line["bbox"][1], content)) - ordered.sort() - lines.append([c for _, c in ordered]) - return text, lines + """Per-page text, and per-page lines in reading order, as flash reads them.""" + from .flash.api import _page_lines + from .flash.main import extract_toc + text = extract_toc(pdf_path, use_embedded_toc=False)["page_texts"] + return text, _page_lines(text) # -------------------------------------------------------------------------- @@ -699,13 +689,16 @@ def own_children(node, children, lines, known, ancestors, nxt): Dropped: a heading that already is a node, in the tree as expand found it (`known`, page -> headings) or made by the node's own ancestors; and on a page the node shares, anything printed above its heading (else above the - nearest ancestor heading found there) or at and below the next node's. - Other branches grow concurrently, so nothing they add is read. A heading - not found on its page decides nothing. + nearest ancestor heading found there) or at and below the next node's (else + its first descendant's found there). Other branches grow concurrently, so + nothing they add is read. A heading not found on its page decides nothing. """ start, end = node["start_index"], subtree_end(node) lineage = [n for a in ancestors for n in [a] + a["nodes"]] - after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] + after = [] + while nxt is not None and nxt["start_index"] == end: + after.append(nxt) + nxt = (nxt.get("nodes") or [None])[0] def found(page, matches): page_lines = lines[page - 1] if page <= len(lines) else [] @@ -719,7 +712,9 @@ def is_node(page, title): tops = (found(start, lambda line: any(same_heading(line, t) for t in headings(n))) for n in [node] + ancestors if n["start_index"] == start) top = next((hits for hits in tops if hits), []) - bottom = found(end, lambda line: any(same_heading(line, t) for t in after)) if after else [] + bottoms = (found(end, lambda line: any(same_heading(line, t) for t in headings(n))) + for n in after) + bottom = next((hits for hits in bottoms if hits), []) kept = [] for child in children: page, key = child["start_index"], normalize(child["title"]) diff --git a/tests/conftest.py b/tests/conftest.py index 04af63745..b6326e46a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -9,8 +9,9 @@ def _llm_key(monkeypatch): def build_pdf(page_texts): - """Build a minimal, uncompressed PDF (one Helvetica line per page) whose - text PyPDF2 can extract. Returns the PDF file bytes.""" + """Build a minimal, uncompressed PDF (one Helvetica line per page, or a + page's (x, y, size, text) lines) whose text PyPDF2 can extract. Returns + the PDF file bytes.""" n = len(page_texts) objects = [] kids = " ".join(f"{3 + i} 0 R" for i in range(n)) @@ -23,9 +24,12 @@ def build_pdf(page_texts): f"/Resources << /Font << /F1 {font_obj} 0 R >> >> " f"/Contents {3 + n + i} 0 R >>".encode() ) - for text in page_texts: - safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") - stream = f"BT /F1 12 Tf 72 720 Td ({safe}) Tj ET".encode() + for page in page_texts: + parts = [] + for x, y, size, text in [(72, 720, 12, page)] if isinstance(page, str) else page: + safe = text.replace("\\", r"\\").replace("(", r"\(").replace(")", r"\)") + parts.append(f"BT /F1 {size} Tf {x} {y} Td ({safe}) Tj ET") + stream = " ".join(parts).encode() objects.append(b"<< /Length %d >>\nstream\n%s\nendstream" % (len(stream), stream)) objects.append(b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>") diff --git a/tests/test_client.py b/tests/test_client.py index 6cceb98cc..24a70da82 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -44,7 +44,7 @@ def indexed_doc(local_client, sample_pdf, monkeypatch): """A document indexed through a stubbed standard pipeline.""" def fake_page_index_main(doc, opt=None, logger=None, page_list=None): assert opt.if_add_node_summary == "yes" - assert opt.if_add_node_text == "yes" + assert opt.if_add_node_text == "no" assert logger is not None assert page_list is not None assert all(isinstance(t, tuple) and len(t) == 2 for t in page_list) diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index d65989fba..cee7e4593 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -238,6 +238,9 @@ def kept(node, ancestors, nxt, *titles, known=known): # a next heading not found on the page decides nothing assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ "2.3 Limits", "3.2 Results"] + # ... the heading of its first descendant opening that page does + part = {"title": "Part II", "start_index": 12, "end_index": 20, "nodes": [results]} + assert kept(methods, [root], part, "2.3 Limits", "Scope") == ["2.3 Limits"] def test_own_children_errs_toward_keeping_where_a_heading_repeats(): @@ -299,6 +302,23 @@ def kept(node, nxt, page, *children): ("3.1.1 Sources", 4), ("Method", 9)) == ["Method"] +def test_pdf_lines_read_a_two_column_page_column_by_column(tmp_path): + page = [] + for x, heading, inner in [(72, "2.3 Setup", "Limitations"), (320, "3 Conclusion", None)]: + page.append((x, 712, 11, heading)) + for row in range(6): + y = 692 - row * 14 + page.append((x, y, 11, inner) if inner and row == 3 else + (x, y, 9, "the method is applied to each")) + pdf = tmp_path / "two.pdf" + pdf.write_bytes(build_pdf([page])) + + _, lines = tree_optimize.load_pages(pdf) + + headings = [line for line in lines[0] if line in ("2.3 Setup", "Limitations", "3 Conclusion")] + assert headings == ["2.3 Setup", "Limitations", "3 Conclusion"] + + @pytest.mark.parametrize("methods_delay", [0, 0.05]) def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): body = "body " * 250 @@ -380,6 +400,27 @@ async def appear(items, *args, **kwargs): assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] +def test_pages_rebuilt_unchanged_by_a_split_are_not_split_again(monkeypatch): + calls = [] + + async def meta_processor(*args, **kwargs): + calls.append(1) + assert len(calls) < 5, "the same pages are split again and again" + # page 1 holds "Part II" right above the section's own heading + return [{"structure": "1", "title": "Part II", "physical_index": 1}, + {"structure": "1.1", "title": "Ch 3", "physical_index": 1}] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + node = {"title": "Ch 3", "start_index": 1, "end_index": 40} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 40, LARGE)) + + assert len(calls) == 1 + + def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] pdf = tmp_path / "doc.pdf"