tiammomo f4e740bc49
fix(uploads): bound document outline and preview text (#5323)
* fix(uploads): bound document outline and preview text

Bound each document outline title to 200 characters and each fallback preview to 2000 characters across its lines, including omission markers.

Refs #5322

Signed-off-by: tiammomo <26957354+tiammomo@users.noreply.github.com>

* test(uploads): cover exact preview budget and restore title guidance

Signed-off-by: tiammomo <26957354+tiammomo@users.noreply.github.com>

---------

Signed-off-by: tiammomo <26957354+tiammomo@users.noreply.github.com>
2026-09-12 14:28:15 +08:00

216 lines
9.3 KiB
Python

"""Shared document outline extraction.
Extracted from ``file_conversion.py`` and ``uploads_middleware.py`` so both
the middleware and the ``list_uploaded_files`` tool can use the same code.
"""
from __future__ import annotations
import logging
import re
from pathlib import Path
logger = logging.getLogger(__name__)
# Regex for bold structural headings produced by pymupdf4llm when it can't
# promote bold text to a Markdown # heading (common in SEC filings).
#
# Chinese headings (第三节...) are already captured as standard # headings
# by pymupdf4llm, so they don't need this pattern.
_BOLD_HEADING_RE = re.compile(r"^\*\*((ITEM|PART|SECTION|SCHEDULE|EXHIBIT|APPENDIX|ANNEX|CHAPTER)\b[A-Z0-9 .,\-]*)\*\*\s*$")
# Regex for split-bold headings produced by pymupdf4llm when a heading spans
# multiple text spans in the PDF (e.g. section number and title are separate spans).
# Matches lines like: **1** **Introduction** or **3.2** **Multi-Head Attention**
# Requirements:
# 1. Entire line consists only of **...** blocks separated by whitespace (no prose)
# 2. First block is a section number (digits and dots, e.g. "1", "3.2", "A.1")
# 3. Second block must not be purely numeric/punctuation — excludes financial table
# headers like **2023** **2022** **2021** while allowing non-ASCII titles such as
# **1** **概述** or accented words (negative lookahead instead of [A-Za-z])
# 4. At most two additional blocks (four total) with [^*]+ (no * inside) to keep
# the regex linear and avoid ReDoS on attacker-controlled content
_SPLIT_BOLD_HEADING_RE = re.compile(r"^\*\*[\dA-Z][\d\.]*\*\*\s+\*\*(?!\d[\d\s.,\-–—/:()%]*\*\*)[^*]+\*\*(?:\s+\*\*[^*]+\*\*){0,2}\s*$")
# Maximum number of outline entries injected into the agent context.
# Keeps prompt size bounded even for very long documents.
MAX_OUTLINE_ENTRIES = 50
_OUTLINE_PREVIEW_LINES = 5
_OUTLINE_TITLE_MAX_CHARS = 200
_OUTLINE_PREVIEW_MAX_CHARS = 2000
_TRUNCATION_MARKER = "… (truncated)"
# Root-level Markdown fences allow up to three leading spaces. The rest of
# the line is an info string when opening, or whitespace only when closing.
_CODE_FENCE_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$")
# ATX headings require 1-6 hashes and a space/tab separator (or end of line).
# Match the original indentation so indented code cannot become a heading.
_ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]+(.*))?$")
def _strip_atx_closing_hashes(raw: str) -> str:
"""Remove a whitespace-separated terminal hash run in linear time."""
trimmed = raw.rstrip(" \t")
prefix = trimmed.rstrip("#")
if len(prefix) < len(trimmed) and (not prefix or prefix[-1] in " \t"):
return prefix.rstrip(" \t")
return trimmed
def _clean_bold_title(raw: str) -> str:
"""Normalise a title string that may contain pymupdf4llm bold artefacts.
pymupdf4llm sometimes emits adjacent bold spans as ``**A** **B**`` instead
of a single ``**A B**`` block. This helper merges those fragments and then
strips the outermost ``**...**`` wrapper so the caller gets plain text.
Examples::
"**Overview**""Overview"
"**UNITED STATES** **SECURITIES**""UNITED STATES SECURITIES"
"plain text""plain text" (unchanged)
"""
# Merge adjacent bold spans: "** **" → " "
merged = re.sub(r"\*\*\s*\*\*", " ", raw).strip()
# Strip outermost **...** if the whole string is wrapped
if m := re.fullmatch(r"\*\*(.+?)\*\*", merged, re.DOTALL):
return m.group(1).strip()
return merged
def _truncate_outline_text(text: str, max_chars: int) -> str:
"""Keep the omission marker inside the summary's character budget."""
if len(text) <= max_chars:
return text
if max_chars <= len(_TRUNCATION_MARKER):
return ""[:max_chars]
return text[: max_chars - len(_TRUNCATION_MARKER)].rstrip() + _TRUNCATION_MARKER
def extract_outline(md_path: Path) -> list[dict]:
"""Extract document outline (headings) from a Markdown file.
Recognises three heading styles produced by pymupdf4llm:
1. Standard ATX headings: up to three spaces, then 1-6 '#' characters
followed by a space/tab or end of line. Optional closing hashes are removed.
Inline ``**...**`` wrappers and adjacent bold spans (``** **``) are
cleaned so the title is plain text.
2. Bold-only structural headings: ``**ITEM 1. BUSINESS**``, ``**PART II**``,
etc. SEC filings use bold+caps for section headings with the same font
size as body text, so pymupdf4llm cannot promote them to # headings.
3. Split-bold headings: ``**1** **Introduction**``, ``**3.2** **Attention**``.
pymupdf4llm emits these when the section number and title text are
separate spans in the underlying PDF (common in academic papers).
Args:
md_path: Path to the .md file.
Returns:
List of dicts with keys: title (str, at most 200 characters), line (int, 1-based).
When the outline is truncated at MAX_OUTLINE_ENTRIES, a sentinel entry
``{"truncated": True}`` is appended as the last element so callers can
render a "showing first N headings" hint without re-scanning the file.
Returns an empty list if the file cannot be read or has no headings.
"""
outline: list[dict] = []
fence_char = ""
fence_length = 0
try:
with md_path.open(encoding="utf-8") as f:
for lineno, line in enumerate(f, 1):
fence = _CODE_FENCE_RE.match(line.rstrip("\r\n"))
if fence_char:
if fence:
marker, suffix = fence.groups()
if marker[0] == fence_char and len(marker) >= fence_length and not suffix.strip(" \t"):
fence_char = ""
continue
if fence:
marker, info = fence.groups()
# Backtick info strings cannot contain backticks; tilde
# info strings have no such restriction.
if marker[0] == "~" or "`" not in info:
fence_char = marker[0]
fence_length = len(marker)
continue
stripped = line.strip()
if not stripped:
continue
# Style 1: standard Markdown heading
if m := _ATX_HEADING_RE.fullmatch(line.rstrip("\r\n")):
title = _clean_bold_title(_strip_atx_closing_hashes(m.group(1) or "").strip())
if title:
outline.append({"title": _truncate_outline_text(title, _OUTLINE_TITLE_MAX_CHARS), "line": lineno})
# Style 2: single bold block with SEC structural keyword
elif m := _BOLD_HEADING_RE.match(stripped):
title = m.group(1).strip()
if title:
outline.append({"title": _truncate_outline_text(title, _OUTLINE_TITLE_MAX_CHARS), "line": lineno})
# Style 3: split-bold heading — **<num>** **<title>**
# Regex already enforces max 4 blocks and non-numeric second block.
elif _SPLIT_BOLD_HEADING_RE.match(stripped):
title = " ".join(re.findall(r"\*\*([^*]+)\*\*", stripped))
if title:
outline.append({"title": _truncate_outline_text(title, _OUTLINE_TITLE_MAX_CHARS), "line": lineno})
if len(outline) > MAX_OUTLINE_ENTRIES:
outline.pop()
outline.append({"truncated": True})
break
except Exception:
return []
return outline
def extract_outline_for_file(file_path: Path) -> tuple[list[dict], list[str]]:
"""Return the document outline and fallback preview for *file_path*.
Looks for a sibling ``<stem>.md`` file produced by the upload conversion
pipeline.
Returns:
(outline, preview) where:
- outline: list of ``{title, line}`` dicts (plus optional sentinel).
Empty when no headings are found or no .md exists.
- preview: first few non-empty lines of the .md, used as a content
anchor when outline is empty, capped at 2000 characters across all lines.
Empty when outline is non-empty (no fallback needed).
"""
md_path = file_path.with_suffix(".md")
if not md_path.is_file():
return [], []
outline = extract_outline(md_path)
if outline:
logger.debug("Extracted %d outline entries from %s", len(outline), file_path.name)
return outline, []
# outline is empty — read the first few non-empty lines as a content preview
preview: list[str] = []
remaining_chars = _OUTLINE_PREVIEW_MAX_CHARS
try:
with md_path.open(encoding="utf-8") as f:
for line in f:
stripped = line.strip()
if stripped:
text = _truncate_outline_text(stripped, remaining_chars)
preview.append(text)
remaining_chars -= len(text)
if len(stripped) > len(text):
break
if len(preview) >= _OUTLINE_PREVIEW_LINES or remaining_chars == 0:
break
except Exception:
logger.debug("Failed to read preview lines from %s", md_path, exc_info=True)
return [], preview