Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 34 additions & 4 deletions pageindex/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -152,7 +152,8 @@ def _needs_model(surface: str) -> PageIndexAPIError:

_LOCAL_INDEX_KEYS = ("model", "summary_model", "backend", "storage_path",
"summary_max_words", "summary_concurrency",
"use_embedded_toc", "optimize")
"summary_max_input_tokens", "summary_scope", "use_embedded_toc",
"optimize")

# Near-synonyms of "cloud" that would otherwise parse as model names —
# a silent wrong mode. They error, pointing at the real word.
Expand All @@ -178,7 +179,9 @@ def _env_cloud_key(spelling: str, inline: str = "api_key=...") -> str:
_ARG_TYPES: "dict[str, tuple[type, ...]]" = {
"model": (str,), "index_model": (str,), "summary_model": (str,),
"chat_model": (str,), "retrieve_model": (str,), "summary_max_words": (int,),
"summary_concurrency": (int,), "use_embedded_toc": (bool,), "optimize": (str,),
"summary_concurrency": (int,), "summary_max_input_tokens": (int,),
"summary_scope": (str,),
"use_embedded_toc": (bool,), "optimize": (str,),
"storage_path": (str, os.PathLike), "index_backend": (dict,),
"chat_backend": (dict,)}

Expand Down Expand Up @@ -356,7 +359,8 @@ class PageIndexClient:
environment), ``"local"``, a local index model name, or a
dict: ``{"api_key": ...}`` for cloud, ``{"model",
"summary_model", "backend", "storage_path",
"summary_max_words", "summary_concurrency", "use_embedded_toc",
"summary_max_words", "summary_concurrency",
"summary_max_input_tokens", "summary_scope", "use_embedded_toc",
"optimize"}`` for local. An
optional ``"mode"`` key (``"cloud"`` / ``"local"``) states
the side and must agree with the other keys; ``{"mode":
Expand Down Expand Up @@ -411,6 +415,18 @@ class PageIndexClient:
expand up to its own ceiling of 32. The lanes overlap, so up to
cap + min(32, cap) calls run at once. Defaults to 64. A
``mode="standard"`` submit refuses either summary knob.
summary_max_input_tokens (int, optional): Local mode only - the
context size of the indexing model. Prompts that grow with the
document stay within it: the document description is cut from
its deepest level, a flash leaf too long for one call is
summarized in parts, and an expand prompt that would overrun it
is skipped. Defaults to unbounded.
summary_scope (str, optional): Local flash mode only - ``"pages"``
summarizes a leaf from the pages of its node, ``"section"`` from
the layout blocks between its heading and the next located one,
which leaves out the end of the previous section and the start
of the next. Defaults to ``"pages"``; a ``mode="standard"``
submit refuses ``"section"``.
use_embedded_toc (bool, optional): Local mode only — whether flash
indexing consumes the PDF's embedded bookmarks when they look
trustworthy. Defaults to True.
Expand Down Expand Up @@ -467,6 +483,8 @@ def __init__(
summary_model: Optional[str] = None,
summary_max_words: Optional[int] = None,
summary_concurrency: Optional[int] = None,
summary_max_input_tokens: Optional[int] = None,
summary_scope: Optional[str] = None,
use_embedded_toc: Optional[bool] = None,
optimize: Optional[str] = None,
retrieve_model: Optional[str] = None,
Expand Down Expand Up @@ -496,6 +514,8 @@ def __init__(
("summary_model", summary_model),
("summary_max_words", summary_max_words),
("summary_concurrency", summary_concurrency),
("summary_max_input_tokens", summary_max_input_tokens),
("summary_scope", summary_scope),
("use_embedded_toc", use_embedded_toc),
("optimize", optimize),
("index_backend", index_backend),
Expand Down Expand Up @@ -582,10 +602,14 @@ def __init__(
raise PageIndexAPIError(
f"{shown} is empty — it configures nothing. Pass a "
"real value, or drop the argument.")
if name == "summary_scope" and value not in ("pages", "section"):
raise PageIndexAPIError(
f'{shown} must be "pages" or "section", got {value!r}.')
if name == "optimize" and value not in ("full", "merge", "off"):
raise PageIndexAPIError(
f'{shown} must be "full", "merge" or "off", got {value!r}.')
if (name in ("summary_max_words", "summary_concurrency")
if (name in ("summary_max_words", "summary_concurrency",
"summary_max_input_tokens")
and isinstance(value, int) and value < 1):
raise PageIndexAPIError(
f"{shown} must be a positive int, got {value!r}.")
Expand Down Expand Up @@ -660,6 +684,8 @@ def __init__(
index_backend=index_conf.get("index_backend"),
summary_max_words=index_conf.get("summary_max_words"),
summary_concurrency=index_conf.get("summary_concurrency"),
summary_max_input_tokens=index_conf.get("summary_max_input_tokens"),
summary_scope=index_conf.get("summary_scope", "pages"),
use_embedded_toc=index_conf.get("use_embedded_toc", True),
optimize=index_conf.get("optimize", "full"),
)
Expand Down Expand Up @@ -2698,6 +2724,8 @@ def __init__(
summary_model: Optional[str] = None,
summary_max_words: Optional[int] = None,
summary_concurrency: Optional[int] = None,
summary_max_input_tokens: Optional[int] = None,
summary_scope: Optional[str] = None,
use_embedded_toc: Optional[bool] = None,
optimize: Optional[str] = None,
retrieve_model: Optional[str] = None,
Expand All @@ -2711,6 +2739,8 @@ def __init__(
model=model, summary_model=summary_model,
summary_max_words=summary_max_words,
summary_concurrency=summary_concurrency,
summary_max_input_tokens=summary_max_input_tokens,
summary_scope=summary_scope,
use_embedded_toc=use_embedded_toc, optimize=optimize,
retrieve_model=retrieve_model, storage_path=storage_path,
index_backend=index_backend, chat_backend=chat_backend,
Expand Down
41 changes: 28 additions & 13 deletions pageindex/flash/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -73,14 +73,16 @@ def _validate_pdf(pdf):
return pdf


async def _summarize(structure, page_list, model, concurrency=None, max_words=None):
async def _summarize(structure, page_list, model, concurrency=None, max_words=None,
max_input_tokens=None, blocks=None):
from ..utils import summarize_tree
await summarize_tree(structure, page_list, model=model, concurrency=concurrency,
max_words=max_words)
max_words=max_words, max_input_tokens=max_input_tokens,
blocks=blocks)


async def _optimize_async(structure, page_texts, do_expand, model, on_final=None,
concurrency=None):
concurrency=None, max_input_tokens=None):
"""Merge/expand refinement after extraction, overlapped with the summaries
when `on_final` is passed; without it the caller runs them after.

Expand All @@ -92,7 +94,8 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None
lines = _page_lines(page_texts)
outcome = await optimize(structure, page_texts, lines, model=model,
do_expand=do_expand, page_count=len(page_texts),
on_final=on_final, concurrency=concurrency)
on_final=on_final, concurrency=concurrency,
max_input_tokens=max_input_tokens)
return {"merges": outcome["merges"], "expands": outcome["expands"],
"same_page_merges": outcome["same_page_merges"],
"same_page_dropped": outcome["same_page_dropped"],
Expand All @@ -107,16 +110,19 @@ def _optimize(structure, page_texts, do_expand, model, concurrency=None):


async def _optimize_and_summarize(structure, page_texts, optimize_model, summary_model,
concurrency, max_words=None):
concurrency, max_words=None, max_input_tokens=None,
blocks=None):
"""Expand and summarize on one loop: a node is summarized as soon as
expand can no longer change it, a parent once its children are done."""
from ..utils import SummaryScheduler
scheduler = SummaryScheduler(structure, [(text, 0) for text in page_texts],
model=summary_model, concurrency=concurrency,
max_words=max_words)
max_words=max_words, max_input_tokens=max_input_tokens,
blocks=blocks)
report = await _optimize_async(structure, page_texts, True, optimize_model,
on_final=scheduler.mark_final,
concurrency=concurrency)
concurrency=concurrency,
max_input_tokens=max_input_tokens)
await scheduler.finish()
return report

Expand Down Expand Up @@ -180,12 +186,16 @@ def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'")
def page_index_flash(pdf, summary=True, summary_model=None,
optimize: str | bool | None = None, optimize_expand=None,
optimize_model=None, summary_concurrency=None,
use_embedded_toc=True, summary_max_words=None) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``"<parent title> (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
use_embedded_toc=True, summary_max_words=None,
summary_max_input_tokens=None, summary_scope="pages") -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). summary_max_input_tokens: context size of the indexing model; the summary and expand prompts are kept within it (a leaf too long for one call is summarized in parts, an expand that overruns it is skipped); None leaves them unbounded. summary_scope: ``"pages"`` summarizes a leaf from the pages of its node, ``"section"`` from the layout blocks between its heading and the next located one (a leaf falls back to its pages when a section without a located heading may start inside it). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``"<parent title> (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
for name, value in (("summary_concurrency", summary_concurrency),
("summary_max_words", summary_max_words)):
("summary_max_words", summary_max_words),
("summary_max_input_tokens", summary_max_input_tokens)):
if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1):
raise ValueError(f"{name} must be a positive int, got {value!r}")
if summary_scope not in ("pages", "section"):
raise ValueError(f'summary_scope must be "pages" or "section", got {summary_scope!r}')
if optimize_expand is not None:
import warnings
warnings.warn(
Expand All @@ -202,7 +212,9 @@ def page_index_flash(pdf, summary=True, summary_model=None,
elif optimize not in ("full", "merge"):
raise ValueError(
f"optimize must be 'full', 'merge', or False, got {optimize!r}")
result = extract_toc(_validate_pdf(pdf), use_embedded_toc=use_embedded_toc)
result = extract_toc(_validate_pdf(pdf), use_embedded_toc=use_embedded_toc,
**({"with_blocks": True} if summary_scope == "section" else {}))
blocks = result.pop("block_texts", None)
structure = result.get("structure", [])
if not structure:
# the layout yields no hierarchy; the pages themselves are the tree
Expand Down Expand Up @@ -230,7 +242,8 @@ def page_index_flash(pdf, summary=True, summary_model=None,
result["optimize"] = asyncio.run(_optimize_and_summarize(
structure, pages, optimize_model=optimize_model or summary_model,
summary_model=summary_model, concurrency=summary_concurrency,
max_words=summary_max_words))
max_words=summary_max_words, max_input_tokens=summary_max_input_tokens,
blocks=blocks))
return result
if optimize and structure:
result["optimize"] = _optimize(structure, pages, do_expand,
Expand All @@ -241,7 +254,9 @@ def page_index_flash(pdf, summary=True, summary_model=None,
page_list = [(text, 0) for text in pages]
asyncio.run(_summarize(structure, page_list, summary_model,
concurrency=summary_concurrency,
max_words=summary_max_words))
max_words=summary_max_words,
max_input_tokens=summary_max_input_tokens,
blocks=blocks))
elif structure:
from ..utils import strip_internal_keys
strip_internal_keys(structure) # summarize_tree does this on its way out
Expand Down
Loading