Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
88 changes: 86 additions & 2 deletions src/_table_headers.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,8 +26,9 @@
PyMuPDF table header detection and HTML serialization (opt-in extension).

Pure text-grid module (no pymupdf import): the header-region rules operate on a
row-major ``[[cell text]]`` grid, and the serializer turns a tagged placement
grid into an HTML ``<table>``. Used only by find_tables(refine=True) (via
row-major ``[[cell text]]`` grid, extend_header_leaf_labels completes that region
from a placement grid's spans, and the serializer turns a tagged placement grid
into an HTML ``<table>``. Used only by find_tables(refine=True) (via
pymupdf.table) and Table.to_html(); never runs on the default detection path.
"""
from __future__ import annotations
Expand Down Expand Up @@ -925,6 +926,89 @@ def find_header_region(rows: list[list[str]]) -> HeaderRegion:
)


# --- Leaf labels under spanning header cells: placement grid -> depth --------
# Digits with only currency, sign, grouping, decimal, percent or date
# punctuation read as a value; a single letter makes the cell a label.
_PLAIN_NUMBER_RE = re.compile(r"^[\s$€£(),.%\-–—+0-9/:]+$")


def _plain_number(text: str) -> bool:
return bool(_PLAIN_NUMBER_RE.match(text)) and any(char.isdigit() for char in text)


def _span_slots(grid) -> tuple[list[tuple[int, int, int, int, str]], int]:
"""Place a ragged span grid on its slots, as an HTML renderer does.

Returns one ``(row0, row1, col0, col1, text)`` entry per cell (end-exclusive
slots, whitespace-collapsed text) and the grid's column count."""
occupied: set[tuple[int, int]] = set()
slots = []
for row0, row in enumerate(grid):
col0 = 0
for cell in row:
while (row0, col0) in occupied:
col0 += 1
row1, col1 = row0 + max(1, cell.rowspan), col0 + max(1, cell.colspan)
occupied.update((r, c) for r in range(row0, row1) for c in range(col0, col1))
slots.append((row0, row1, col0, col1, collapse_cell_ws(cell.text or "")))
col0 = col1
return slots, max((col for _, col in occupied), default=-1) + 1


def _row_shape(slots, row: int, ncols: int) -> tuple[str, ...]:
"""Per column, whether ``row`` shows nothing, a plain number or text there."""
shape = [""] * ncols
for row0, row1, col0, col1, text in slots:
if row0 <= row < row1 and text:
kind = "number" if _plain_number(text) else "text"
for col in range(col0, min(col1, ncols)):
shape[col] = kind
return tuple(shape)


def extend_header_leaf_labels(grid, top_header_rows: int) -> int:
"""Extend the header over the row naming the columns of a spanning header cell.

A header cell spanning more than one column but not all of them, with no
single-column label under any of its columns, leaves those columns unnamed.
The row below the header joins it when it supplies the names: a non-empty
single-column cell under every column of such a span, no plain number under
a column that is already labeled, not one full-width cell, and not the same
blank/number/text shape per column as the row after it (that repeat marks
the first record of a regular body). Repeats for nested spans; never shrinks
the header, leaves a header of 0 rows alone and never takes the last row, so
the table always keeps a body row.

``grid`` is a row-major placement grid duck-typed like ``render_table_html``
input (``text`` / ``colspan`` / ``rowspan``). Only its text and spans are
read, so the result depends on nothing but the grid itself.
"""
slots, ncols = _span_slots(grid)
depth = top_header_rows
if not 0 < depth < len(grid) or ncols < 2:
return depth
while depth < len(grid) - 1: # a header never takes the whole table
header = [slot for slot in slots if slot[1] <= depth and slot[4]]
labeled = {col0 for _, _, col0, col1, _ in header if col1 - col0 == 1}
spans = [
range(col0, col1)
for _, _, col0, col1, _ in header
if 1 < col1 - col0 < ncols and labeled.isdisjoint(range(col0, col1))
]
row = [(col0, col1, text) for row0, _, col0, col1, text in slots if row0 == depth and text]
if not spans or not row or (len(row) == 1 and row[0][1] - row[0][0] == ncols):
break
leaves = {col0 for col0, col1, _ in row if col1 - col0 == 1}
if not any(leaves.issuperset(span) for span in spans):
break
if any(_plain_number(text) and not labeled.isdisjoint(range(col0, col1)) for col0, col1, text in row):
break
if _row_shape(slots, depth, ncols) == _row_shape(slots, depth + 1, ncols):
break
depth += 1
return depth


# --- HTML serialization: tagged placement grid -> <table> --------------------
def collapse_cell_ws(text: str) -> str:
"""Whitespace-collapse a cell's text (runs of whitespace/newlines -> one space)."""
Expand Down
78 changes: 75 additions & 3 deletions src/_table_refine.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,8 @@

import itertools
import re
from bisect import bisect_left, bisect_right
from math import isfinite
import pymupdf


Expand All @@ -51,6 +53,70 @@
_REFINE_LINE_GAP = 3.0 # center-y gap (points) that groups body words into lines


# --- word center index: reduce the per-cell word scan to a sorted band -------
class _WordCenterIndex:
"""Two sorted word-center axes over a read-only page word list.

``candidates(rect)`` returns a superset of the words whose center lies in
``rect``, so every consumer still applies its own membership, blank-text and
claiming rules. The original word indices are preserved, including order and
duplicates, because they are what identifies a word to its owning cell.
"""

def __init__(self, words):
xs, ys = [], []
for i, (x0, y0, x1, y1, _text) in enumerate(words):
x, y = (x0 + x1) * 0.5, (y0 + y1) * 0.5
if not (isfinite(x) and isfinite(y)):
raise ValueError("nonfinite word center")
xs.append((x, i))
ys.append((y, i))
self.xs, self.ys = sorted(xs), sorted(ys)

def candidates(self, rect):
x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1)
n = len(self.xs)
if not all(map(isfinite, (x0, y0, x1, y1))):
return range(n) # preserve the consumers' predicates for NaN / Inf
if x0 > x1 or y0 > y1:
return []
xl, xr = bisect_left(self.xs, (x0, -1)), bisect_right(self.xs, (x1, n))
yl, yr = bisect_left(self.ys, (y0, -1)), bisect_right(self.ys, (y1, n))
# Scan the narrower band: fewer candidates for the same result.
entries = self.xs[xl:xr] if xr - xl <= yr - yl else self.ys[yl:yr]
return sorted(i for _center, i in entries)


def _refine_word_index(page, words):
"""The page word list's center index, cached on the page object.

Keyed on the word list's identity: _refine_page_words returns one stable list
per page and a fresh extraction produces a new list, so a changed text state
builds a new index. In-place edits of a cached word list are not supported.
Returns None for word lists the index cannot represent (see candidates()).
"""
cached = getattr(page, "_table_word_index_cache", None)
if cached is not None and cached[0] is words:
return cached[1]
try:
index = _WordCenterIndex(words)
except (TypeError, ValueError, OverflowError):
index = None # nonstandard inputs keep the original full scan
try:
setattr(page, "_table_word_index_cache", (words, index))
except Exception:
pass
return index


def _refine_word_candidates(page, words, rect):
"""(index, word) pairs that may lie in rect -- a superset of the members."""
index = _refine_word_index(page, words) if page is not None else None
if index is None:
return enumerate(words)
return ((i, words[i]) for i in index.candidates(rect))


# --- word selection: center-point membership + rotated-span substitution -----
def _refine_rawdict_spans(page):
"""Flattened rawdict text spans carrying line-direction metadata.
Expand Down Expand Up @@ -155,7 +221,9 @@ def _refine_words_in_rect(page, rect):
x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1)
return [
(wx0, wy0, wx1, wy1, text)
for wx0, wy0, wx1, wy1, text in _refine_page_words(page)
for _, (wx0, wy0, wx1, wy1, text) in _refine_word_candidates(
page, _refine_page_words(page), rect
)
if _refine_word_in_rect(wx0, wy0, wx1, wy1, x0, y0, x1, y1)
]

Expand Down Expand Up @@ -202,9 +270,11 @@ def _refine_table_rect(cells, table_bbox):


def _refine_raw_shaded_rects(page, table_rect, *, min_dim):
from pymupdf.table import _get_table_drawings

out = []
page_width = float(page.rect.width)
for drawing in page.get_drawings():
for drawing in _get_table_drawings(page):
if _refine_is_white(drawing.get("fill")):
continue
for item in drawing.get("items", []):
Expand Down Expand Up @@ -253,9 +323,11 @@ def _refine_cluster(values, *, tolerance):


def _refine_border_lines(page, table_rect):
from pymupdf.table import _get_table_drawings

xs = set()
ys = set()
for drawing in page.get_drawings():
for drawing in _get_table_drawings(page):
stroked = drawing.get("type") in ("s", "fs")
for item in drawing.get("items", []):
kind = item[0]
Expand Down
13 changes: 8 additions & 5 deletions src/_table_spans.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@
from pymupdf._table_refine import (
_refine_is_vertical_or_rotated,
_refine_page_words,
_refine_word_candidates,
)


Expand Down Expand Up @@ -292,13 +293,15 @@ def _span_word_line_tuple(word):
return (float(y0), float(x0), float(y1), str(text))


def _span_select_words_in_rect(page_words, rect):
def _span_select_words_in_rect(page_words, rect, *, page=None):
"""(index, word) pairs whose center lies in rect, index into ``page_words``.

The index is what lets resolve_spans claim each page word for exactly one
placement (an earlier cell's word is not re-claimed by a later one)."""
placement (an earlier cell's word is not re-claimed by a later one). ``page``
only supplies the cached word-center index, which narrows the scan without
changing the result."""
selected = []
for index, word in enumerate(page_words):
for index, word in _refine_word_candidates(page, page_words, rect):
wx0, wy0, wx1, wy1, text = word
if not str(text).strip():
continue
Expand All @@ -321,7 +324,7 @@ def _span_claim_text_in_rect(page, rect, page_words, claimed_words):
"""Text of rect's words, skipping words already claimed and claiming the rest."""
selected = [
(index, word)
for index, word in _span_select_words_in_rect(page_words, rect)
for index, word in _span_select_words_in_rect(page_words, rect, page=page)
if index not in claimed_words
]
for index, _ in selected:
Expand Down Expand Up @@ -486,7 +489,7 @@ def _span_reject_colspan_mismatch_merge(*, row_idx, cols, base, body_start):
def _span_cell_texts_for_entries(page, entries, start, end, page_words):
texts = []
for entry in entries[start : end + 1]:
words = _span_select_words_in_rect(page_words, entry)
words = _span_select_words_in_rect(page_words, entry, page=page)
texts.append(_span_words_text_for_rect(page, entry, words))
return texts

Expand Down
Loading
Loading