From 01df5c4cc5cf22a739e25cb6036c516d43baf318 Mon Sep 17 00:00:00 2001 From: "youchang.kim" Date: Tue, 22 Sep 2026 21:18:18 +0900 Subject: [PATCH 1/2] table: forward virtual rules in union, keep batched thin rules, gate picture candidates, split stacked tables find_tables() changes, all behind the existing opt-in keywords: - union=True forwards add_lines / add_boxes into the nested line finder; they were silently dropped before. - lines_strict: thin rectangles batched into one large fill-only path are kept as rule material when they are line-like and longer than the minimum edge length (some producers draw every grid rule as a filled rectangle inside a single path). The path's own bbox contributes no envelope, so dot screens and glyph fragments do not become tables. - union admission runs on the finder's cell groups (TableFinder(cell_group_filter=)): a candidate lying mostly inside a layout "picture" group is admitted only with table-shaped evidence (two-dimensional text support, at least two rows and two columns each at least half populated, mostly rectangular cell coverage), so a chart's gridlines are not turned into a table while a sparse matrix table is. - refine=True splits a grid whose confirmed leading header row recurs verbatim into one table per stack, reusing the placement grid already built for the header/body boundary. - Table.header is computed on first access (same value, restored extraction flags); drawings are read once per find_tables() call and shared with the refine helpers; word-in-rect selection uses a per-page centre index. Default find_tables() output is byte-identical on the test corpus (default, lines_strict and refine=True paths). Based on the review round by Minsu Kim. --- src/_table_refine.py | 78 ++++++++- src/_table_spans.py | 13 +- src/_table_union.py | 271 ++++++++++++++++++++++++++--- src/table.py | 384 ++++++++++++++++++++++++++++++++++------- tests/test_tables.py | 400 +++++++++++++++++++++++++++++++++++++++++++ 5 files changed, 1046 insertions(+), 100 deletions(-) diff --git a/src/_table_refine.py b/src/_table_refine.py index dc8984cac..7f2ce66fd 100644 --- a/src/_table_refine.py +++ b/src/_table_refine.py @@ -33,6 +33,8 @@ import itertools import re +from bisect import bisect_left, bisect_right +from math import isfinite import pymupdf @@ -51,6 +53,70 @@ _REFINE_LINE_GAP = 3.0 # center-y gap (points) that groups body words into lines +# --- word center index: reduce the per-cell word scan to a sorted band ------- +class _WordCenterIndex: + """Two sorted word-center axes over a read-only page word list. + + ``candidates(rect)`` returns a superset of the words whose center lies in + ``rect``, so every consumer still applies its own membership, blank-text and + claiming rules. The original word indices are preserved, including order and + duplicates, because they are what identifies a word to its owning cell. + """ + + def __init__(self, words): + xs, ys = [], [] + for i, (x0, y0, x1, y1, _text) in enumerate(words): + x, y = (x0 + x1) * 0.5, (y0 + y1) * 0.5 + if not (isfinite(x) and isfinite(y)): + raise ValueError("nonfinite word center") + xs.append((x, i)) + ys.append((y, i)) + self.xs, self.ys = sorted(xs), sorted(ys) + + def candidates(self, rect): + x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) + n = len(self.xs) + if not all(map(isfinite, (x0, y0, x1, y1))): + return range(n) # preserve the consumers' predicates for NaN / Inf + if x0 > x1 or y0 > y1: + return [] + xl, xr = bisect_left(self.xs, (x0, -1)), bisect_right(self.xs, (x1, n)) + yl, yr = bisect_left(self.ys, (y0, -1)), bisect_right(self.ys, (y1, n)) + # Scan the narrower band: fewer candidates for the same result. + entries = self.xs[xl:xr] if xr - xl <= yr - yl else self.ys[yl:yr] + return sorted(i for _center, i in entries) + + +def _refine_word_index(page, words): + """The page word list's center index, cached on the page object. + + Keyed on the word list's identity: _refine_page_words returns one stable list + per page and a fresh extraction produces a new list, so a changed text state + builds a new index. In-place edits of a cached word list are not supported. + Returns None for word lists the index cannot represent (see candidates()). + """ + cached = getattr(page, "_table_word_index_cache", None) + if cached is not None and cached[0] is words: + return cached[1] + try: + index = _WordCenterIndex(words) + except (TypeError, ValueError, OverflowError): + index = None # nonstandard inputs keep the original full scan + try: + setattr(page, "_table_word_index_cache", (words, index)) + except Exception: + pass + return index + + +def _refine_word_candidates(page, words, rect): + """(index, word) pairs that may lie in rect -- a superset of the members.""" + index = _refine_word_index(page, words) if page is not None else None + if index is None: + return enumerate(words) + return ((i, words[i]) for i in index.candidates(rect)) + + # --- word selection: center-point membership + rotated-span substitution ----- def _refine_rawdict_spans(page): """Flattened rawdict text spans carrying line-direction metadata. @@ -155,7 +221,9 @@ def _refine_words_in_rect(page, rect): x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) return [ (wx0, wy0, wx1, wy1, text) - for wx0, wy0, wx1, wy1, text in _refine_page_words(page) + for _, (wx0, wy0, wx1, wy1, text) in _refine_word_candidates( + page, _refine_page_words(page), rect + ) if _refine_word_in_rect(wx0, wy0, wx1, wy1, x0, y0, x1, y1) ] @@ -202,9 +270,11 @@ def _refine_table_rect(cells, table_bbox): def _refine_raw_shaded_rects(page, table_rect, *, min_dim): + from pymupdf.table import _get_table_drawings + out = [] page_width = float(page.rect.width) - for drawing in page.get_drawings(): + for drawing in _get_table_drawings(page): if _refine_is_white(drawing.get("fill")): continue for item in drawing.get("items", []): @@ -253,9 +323,11 @@ def _refine_cluster(values, *, tolerance): def _refine_border_lines(page, table_rect): + from pymupdf.table import _get_table_drawings + xs = set() ys = set() - for drawing in page.get_drawings(): + for drawing in _get_table_drawings(page): stroked = drawing.get("type") in ("s", "fs") for item in drawing.get("items", []): kind = item[0] diff --git a/src/_table_spans.py b/src/_table_spans.py index ff0d89d40..a15dacc26 100644 --- a/src/_table_spans.py +++ b/src/_table_spans.py @@ -36,6 +36,7 @@ from pymupdf._table_refine import ( _refine_is_vertical_or_rotated, _refine_page_words, + _refine_word_candidates, ) @@ -292,13 +293,15 @@ def _span_word_line_tuple(word): return (float(y0), float(x0), float(y1), str(text)) -def _span_select_words_in_rect(page_words, rect): +def _span_select_words_in_rect(page_words, rect, *, page=None): """(index, word) pairs whose center lies in rect, index into ``page_words``. The index is what lets resolve_spans claim each page word for exactly one - placement (an earlier cell's word is not re-claimed by a later one).""" + placement (an earlier cell's word is not re-claimed by a later one). ``page`` + only supplies the cached word-center index, which narrows the scan without + changing the result.""" selected = [] - for index, word in enumerate(page_words): + for index, word in _refine_word_candidates(page, page_words, rect): wx0, wy0, wx1, wy1, text = word if not str(text).strip(): continue @@ -321,7 +324,7 @@ def _span_claim_text_in_rect(page, rect, page_words, claimed_words): """Text of rect's words, skipping words already claimed and claiming the rest.""" selected = [ (index, word) - for index, word in _span_select_words_in_rect(page_words, rect) + for index, word in _span_select_words_in_rect(page_words, rect, page=page) if index not in claimed_words ] for index, _ in selected: @@ -486,7 +489,7 @@ def _span_reject_colspan_mismatch_merge(*, row_idx, cols, base, body_start): def _span_cell_texts_for_entries(page, entries, start, end, page_words): texts = [] for entry in entries[start : end + 1]: - words = _span_select_words_in_rect(page_words, entry) + words = _span_select_words_in_rect(page_words, entry, page=page) texts.append(_span_words_text_for_rect(page, entry, words)) return texts diff --git a/src/_table_union.py b/src/_table_union.py index d7a4d00e0..bd3a31003 100644 --- a/src/_table_union.py +++ b/src/_table_union.py @@ -28,9 +28,12 @@ TableFinder and _iou come from pymupdf.table; find_tables is imported lazily. """ +from bisect import bisect_left +from collections import namedtuple + import pymupdf -from pymupdf.table import CHARS, EDGES, Table, TableFinder, _iou +from pymupdf.table import CHARS, EDGES, Table, TableFinder, _cells_to_rows, _iou # --------------------------------------------------------------------------- @@ -53,6 +56,12 @@ _UNION_GRID_REF_SPAN_MULT_THRESHOLD = 3.0 # max horizontally-separated span groups per cell _UNION_OWNER_CONTAINMENT = 0.85 # min containment for a split candidate's owner _UNION_OWNER_AMBIGUOUS_OVERLAP = 0.25 # overlap above which an unowned candidate is suppressed +_UNION_PICTURE_CONTAINMENT = 0.8 # area fraction that puts a candidate inside a picture + +# How a candidate grid's text is laid out; see _union_grid_content_support. +_UnionContent = namedtuple( + "_UnionContent", "supported dense_rows dense_columns concrete slots" +) def _layout_table_grids(page): @@ -88,35 +97,83 @@ def _layout_table_grids(page): return grids -def _union_line_candidates(page): +def _union_line_candidates(page, *, add_lines=None, add_boxes=None): """Line-based table candidates for the union stage as ``(bbox, grid)`` pairs. Runs a nested find_tables (strategy=_UNION_STRATEGY, use_layout=False) and - keeps each detected table's bbox and row-major cell grid (Table.rows, None - for a gap), deduped by rounded bbox. Returns ``(candidates, finder)``; the - finder is reused as the returned TableFinder shell. + keeps each detected table's bbox and row-major cell grid (None for a gap), + deduped by rounded bbox. Caller-supplied virtual lines / boxes are forwarded + to that nested finder so they take part in the same union decisions as + PDF-native vector rules. Admission (see ``admit``) runs on the finder's cell + groups, before a Table is built for any of them. Returns ``(candidates, + finder)``; the finder is reused as the returned TableFinder shell. """ # Imported here, not at module top, to break the import cycle: this # module is itself imported lazily by table.find_tables (union path). from pymupdf.table import find_tables - finder = find_tables(page, strategy=_UNION_STRATEGY, use_layout=False) candidates = [] - seen = set() - for tab in (getattr(finder, "tables", None) or []): - try: - bbox = pymupdf.Rect(tab.bbox) - except (ValueError, TypeError): - continue - if bbox.is_empty: - continue - grid = [[cell for cell in row.cells] for row in (tab.rows or [])] - if not grid: - continue - key = tuple(round(value) for value in bbox) - if key in seen: - continue - seen.add(key) - candidates.append((bbox, grid)) + + def admit(live_page, groups): + """Keep the cell groups that may become tables, in detection order. + + A candidate is admitted as before: the line grid is evidence in its own + right and the layout model can miss a table entirely. The one exception is + a candidate sitting inside a region the layout model called a picture, + which has to show table-shaped text instead. That cheap geometric test + therefore gates the character scan, and the scan itself is built once for + the whole call. + """ + kept = [] + seen = set() + points = None + for cells in groups: + bbox = pymupdf.Rect( + min(cell[0] for cell in cells), + min(cell[1] for cell in cells), + max(cell[2] for cell in cells), + max(cell[3] for cell in cells), + ) + if bbox.is_empty: + continue + grid = [row.cells for row in _cells_to_rows(cells)] + if not grid: + continue + key = tuple(round(value) for value in bbox) + if key in seen: + continue + if _union_candidate_inside_picture(live_page, bbox): + if points is None: + points = _union_char_midpoints() + content = _union_grid_content_support(grid, points) + # A chart labels one row and one column -- its axes -- and its + # partial gridlines leave most slots without a cell. A table has + # a header row and a label column that are each at least half + # full, and cell coverage that is mostly rectangular. + if not ( + content.supported + and content.dense_rows >= 2 + and content.dense_columns >= 2 + and content.concrete * 2 >= content.slots + ): + continue + # Dedup only admitted candidates: a rejected one must not shadow a + # later, differently gridded candidate with the same rounded bbox. + seen.add(key) + kept.append(cells) + candidates.append((bbox, grid)) + return kept + + finder = find_tables( + page, + strategy=_UNION_STRATEGY, + use_layout=False, + add_lines=add_lines, + add_boxes=add_boxes, + _cell_group_filter=admit, + ) + if finder is None: + # Nested detection failure must not leak partially admitted candidates. + candidates.clear() return candidates, finder @@ -165,6 +222,161 @@ def _union_find_owner(candidate_bbox, existing_bboxes): return best_owner, ambiguous +def _union_char_midpoints(): + """The page's non-blank character midpoints as ``(v_mid, h_mid)``, y-sorted. + + One pass over CHARS replaces the per-cell character scan of the admission + tests, which read midpoints only. Sorting by the vertical midpoint lets each + grid row take its own characters as one slice. + """ + points = [] + for char in CHARS: + if not str(char.get("text") or "").strip(): + continue + try: + h_mid = (float(char["x0"]) + float(char["x1"])) / 2.0 + v_mid = (float(char["top"]) + float(char["bottom"])) / 2.0 + except (KeyError, TypeError, ValueError): + continue + points.append((v_mid, h_mid)) + points.sort() + return points + + +def _union_grid_content_support(grid, points): + """How a candidate line grid's text is distributed, in one scan. + + ``supported`` means at least two rows with text in at least two columns and + at least two columns with text in at least two rows -- two being the smallest + non-trivial count in each dimension, not a fitted one. A populated cell counts + for every column its x-range covers, so a grid whose missing horizontal rules + leave a spanning header above one record row is supported just like the same + table with all its rules. + + ``dense_rows`` / ``dense_columns`` count the rows and columns whose own + concrete cells are at least half populated, and ``concrete`` / ``slots`` are + the concrete cells and the grid's total slots. Those describe the *shape* of + the text rather than its amount, which is what separates a chart from a + sparse table (see _union_candidate_inside_picture). + + Character membership follows Table.extract()'s half-open cell rule. ``None`` + grid slots are span/gap placeholders, not empty text cells. The decision is + candidate-local: page area, layout ownership, file identity and virtual-line + provenance are not inputs. + """ + columns = sorted({float(cell[0]) for row in grid for cell in row if cell is not None}) + width = max((len(row) for row in grid), default=0) + column_populated = [0] * width + column_concrete = [0] * width + row_columns = [] + row_counts = [] + concrete = 0 + slots = 0 + for row in grid: + slots += len(row) + covered = set() + row_populated = 0 + rects = [(index, pymupdf.Rect(cell)) for index, cell in enumerate(row) if cell is not None] + rects = [(index, rect) for index, rect in rects if not rect.is_empty] + concrete += len(rects) + if rects: + band = _union_row_points(rects, points) + for index, rect in rects: + column_concrete[index] += 1 + if not _union_rect_has_point(rect, band): + continue + row_populated += 1 + column_populated[index] += 1 + spanned = [ + column + for column, x in enumerate(columns) + if float(rect.x0) <= x < float(rect.x1) + ] + covered.update(spanned or (index,)) + row_columns.append(covered) + row_counts.append((row_populated, len(rects))) + supported = sum(len(covered) >= 2 for covered in row_columns) >= 2 and sum( + sum(column in covered for covered in row_columns) >= 2 + for column in range(len(columns)) + ) >= 2 + return _UnionContent( + supported=supported, + dense_rows=sum(total > 0 and filled * 2 >= total for filled, total in row_counts), + dense_columns=sum( + total > 0 and filled * 2 >= total + for filled, total in zip(column_populated, column_concrete) + ), + concrete=concrete, + slots=slots, + ) + + +def _union_row_points(rects, points): + """The y-sorted midpoints falling in one grid row's vertical band.""" + y0 = min(float(rect.y0) for _index, rect in rects) + y1 = max(float(rect.y1) for _index, rect in rects) + return points[bisect_left(points, (y0,)) : bisect_left(points, (y1,))] + + +def _union_rect_has_point(rect, band): + """Whether any midpoint of the row band lies in rect (half-open, as extract).""" + x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) + for v_mid, h_mid in band: + if x0 <= h_mid < x1 and y0 <= v_mid < y1: + return True + return False + + +def _union_layout_groups(page): + """The page's layout groups as ``(class_name, rect)``, skipping unusable ones.""" + groups = [] + for group in (page.layout_information or []): + if not isinstance(group, dict): + continue + group_bbox = group.get("group_bbox") + if not group_bbox: + continue + try: + rect = pymupdf.Rect(group_bbox[:4]) + except (TypeError, ValueError): + continue + if rect.is_empty: + continue + groups.append((group.get("class_name"), rect)) + return groups + + +def _union_candidate_inside_picture(page, candidate_bbox): + """Whether a candidate sits inside a layout picture group without covering it. + + The layout model classified that region as a picture, so a line grid found + almost entirely inside one is as likely to be a chart's gridlines as a table, + and overriding the model's verdict needs table-shaped text. The shape, not the + amount, is what tells them apart: a chart's text sits along its two axes, so + exactly one row and one column are well filled and its partial gridlines leave + most of the grid without a cell at all, while a table has a header row and a + label column that are each at least half full over mostly complete cell + coverage -- which a sparse matrix of scattered marks still satisfies. See the + admission rule in _union_line_candidates. + + A candidate larger than the picture contains it instead -- a bordered table + with an image in one of its cells -- and keeps the ordinary rule. + """ + candidate_area = _union_rect_area(candidate_bbox) + if candidate_area <= 0: + return False + threshold = _UNION_PICTURE_CONTAINMENT * candidate_area + for class_name, rect in _union_layout_groups(page): + if class_name != "picture": + continue + if _union_intersection_area(candidate_bbox, rect) < threshold: + continue + if candidate_area > _union_rect_area(rect): + continue # the candidate is the larger region: it holds the picture + return True + return False + + def _union_text_span_rects(page): """Non-empty page text-span rects, cached on the page. @@ -325,18 +537,21 @@ def _union_replace_append(existing, candidates, *, page, grid_ref, grid_ref_iou, return entries -def _find_tables_union(page): +def _find_tables_union(page, *, add_lines=None, add_boxes=None): """Detect a page's tables by fusing layout grids with line-based candidates. Ensures the raw layout (computed only when page.layout_information is None, - like the official use_layout path), reads primary grids, detects candidates, - applies grid-ref / split / append, and returns a TableFinder whose .tables - carry the fused grids in contractual order (grid-ref tables keep their - explicit layout bbox).""" + like the official use_layout path), reads primary grids, detects candidates + (forwarding the caller's virtual lines / boxes to the nested finder), applies + grid-ref / split / append, and returns a TableFinder whose .tables carry the + fused grids in contractual order (grid-ref tables keep their explicit layout + bbox).""" if page.layout_information is None: page.get_layout(return_raw=True) primaries = _layout_table_grids(page) - candidates, finder = _union_line_candidates(page) + candidates, finder = _union_line_candidates( + page, add_lines=add_lines, add_boxes=add_boxes + ) entries = _union_replace_append( primaries, candidates, diff --git a/src/table.py b/src/table.py index 7b8e2c19c..991714306 100644 --- a/src/table.py +++ b/src/table.py @@ -123,6 +123,44 @@ _CHARS_VAR = ContextVar("pymupdf_table_chars", default=None) +def _get_table_drawings(page, *, native=False): + """The page's vector drawings, shared within one find_tables() call. + + One call reads them up to ``1 + 2T`` times -- once to build edges, and once + per refined table for shaded rows and for border lines -- and each read walks + the whole content stream. find_tables() therefore puts an empty cache on the + page for the duration of the call and removes it again, so nothing outlives + the call: a page whose content stream is rewritten between calls cannot be + served a stale extraction. Calling the public refine helpers directly, with + no find_tables() around them, extracts afresh exactly as before. + + ``native=True`` returns per-path copies of the mutated members: make_edges + extends path rects, normalizes rectangle items and appends a closing line, so + the other consumers must not see its edits. Point and Quad values stay shared + because make_edges only reads them. + """ + cache = getattr(page, "_table_drawings_cache", None) + if cache is None: + paths = page.get_drawings() + else: + if not cache: + cache.append(page.get_drawings()) + paths = cache[0] + if not native: + return paths + return [ + dict( + path, + rect=pymupdf.Rect(path["rect"]), + items=[ + (item[0], pymupdf.Rect(item[1]), *item[2:]) if item[0] == "re" else item + for item in path["items"] + ], + ) + for path in paths + ] + + class _TableStateList: """List-like proxy for per-call table extraction state.""" @@ -1607,6 +1645,25 @@ def __init__(self, bbox, cells, names, above): self.external = above +def _cells_to_rows(cells): + """Canonical row order and None gap slots for a flat cell list. + + Each row holds one slot per distinct cell x0 on the table, so a gap and a + span both appear as None. Shared by Table.rows and the union stage, which + needs the same grid before a Table exists.""" + _sorted = sorted(cells, key=itemgetter(1, 0)) + xs = list(sorted(set(map(itemgetter(0), cells)))) + rows = [] + for y, row_cells in itertools.groupby(_sorted, itemgetter(1)): + xdict = {cell[0]: cell for cell in row_cells} + row = TableRow([xdict.get(x) for x in xs]) + rows.append(row) + return rows + + +_NO_HEADER_YET = object() # Table.header has not been computed yet + + class Table: def __init__(self, page, cells, bbox=None): self.page = page @@ -1617,7 +1674,14 @@ def __init__(self, page, cells, bbox=None): # cells. Set only for a union grid-ref table, whose reported region # (its layout box) is decoupled from its replacement cell grid. self._bbox = bbox - self.header = self._get_header() # PyMuPDF extension + self._header = _NO_HEADER_YET + # The header rules read the page through the two global text-extraction + # toggles that were in force when this table was detected, so a header + # computed on first access has to put them back. Both are plain getters. + self._header_flags = ( + bool(pymupdf.TOOLS.set_small_glyph_heights()), + bool(pymupdf.TOOLS.unset_quad_corrections()), + ) # Filled by find_tables(refine=True): placements is a row-major grid of # tagged SpanCell colspan/rowspan placements (None otherwise); header_rows # is the leading header-row count, section_rows the section-label rows. @@ -1637,16 +1701,36 @@ def bbox(self): max(map(itemgetter(3), c)), ) + @property + def header(self): # PyMuPDF extension + """The identified table header, computed on first access. + + Identifying it renders the top row twice, so tables whose header is never + read -- every table serialized through placements, and every candidate a + later stage discards -- must not pay for it at construction time. The + header rules re-read the page, so the two global text-extraction toggles + of the detecting call are restored for the duration (see __init__): the + result is then the one an eager computation would have produced.""" + if self._header is _NO_HEADER_YET: + small, quads = self._header_flags + old_small = bool(pymupdf.TOOLS.set_small_glyph_heights()) + old_quads = bool(pymupdf.TOOLS.unset_quad_corrections()) + pymupdf.TOOLS.set_small_glyph_heights(small) + pymupdf.TOOLS.unset_quad_corrections(quads) + try: + self._header = self._get_header() + finally: + pymupdf.TOOLS.set_small_glyph_heights(old_small) + pymupdf.TOOLS.unset_quad_corrections(old_quads) + return self._header + + @header.setter + def header(self, value): + self._header = value + @property def rows(self) -> list: - _sorted = sorted(self.cells, key=itemgetter(1, 0)) - xs = list(sorted(set(map(itemgetter(0), self.cells)))) - rows = [] - for y, row_cells in itertools.groupby(_sorted, itemgetter(1)): - xdict = {cell[0]: cell for cell in row_cells} - row = TableRow([xdict.get(x) for x in xs]) - rows.append(row) - return rows + return _cells_to_rows(self.cells) @property def row_count(self) -> int: # PyMuPDF extension @@ -1869,9 +1953,12 @@ def row_has_bold(bbox): Returns True if any spans are bold else False. """ + # This table's own character snapshot, so a lazily computed header + # reads the same characters an eager one would have. + chars = self._chars if self._chars is not None else CHARS return any( c["bold"] - for c in CHARS + for c in chars if rect_in_rect((c["x0"], c["y0"], c["x1"], c["y1"]), bbox) ) @@ -2153,7 +2240,7 @@ class TableFinder: https://github.com/tabulapdf/tabula-extractor/issues/16 """ - def __init__(self, page, settings=None): + def __init__(self, page, settings=None, *, cell_group_filter=None): self.page = weakref.proxy(page) self.textpage = None self.settings = TableSettings.resolve(settings) @@ -2164,10 +2251,12 @@ def __init__(self, page, settings=None): self.settings.intersection_y_tolerance, ) self.cells = intersections_to_cells(self.intersections) - self.tables = [ - Table(self.page, cell_group) - for cell_group in cells_to_tables(self.page, self.cells) - ] + groups = cells_to_tables(self.page, self.cells) + if cell_group_filter is not None: + # Internal hook: the union stage decides admission on the cell groups, + # so a rejected candidate never becomes a Table at all. + groups = cell_group_filter(self.page, groups) + self.tables = [Table(self.page, group) for group in groups] def get_edges(self) -> list: settings = self.settings @@ -2364,6 +2453,26 @@ def make_chars(page, clip=None): # We are ignoring Bézier curves completely and are converting everything # else to lines. # ------------------------------------------------------------------------ +def _is_line_like(rect, min_length): + """Whether a rectangle is thin enough to be a simulated line.""" + return (rect.width <= min_length and rect.width < rect.height) or ( + rect.height <= min_length and rect.height < rect.width + ) + + +def _is_batched_rule(rect, min_length): + """Whether a rectangle inside a large fill path is a simulated line. + + A path's aggregate bbox is no evidence about its individual items, so a thin + rectangle batched into a large fill can still be a grid rule. A thin *and + short* one cannot: dot screens and vector-drawn glyph fragments consist of + rectangles that are small in both directions, and a few hundred of them snap + and join into one long spurious edge. A rule reaches at least the minimum + edge length along its length, which is the same bound filter_edges() applies + to the finished edges.""" + return _is_line_like(rect, min_length) and max(rect.width, rect.height) > min_length + + def make_edges(page, clip=None, tset=None, paths=None, add_lines=None, add_boxes=None): edges = EDGES._list() # bind once: avoid per-append proxy overhead below snap_x = tset.snap_x_tolerance @@ -2420,24 +2529,42 @@ def are_neighbors(r1, r2): def clean_graphics(npaths=None): """Detect and join rectangles of "connected" vector graphics.""" if npaths is None: - allpaths = page.get_drawings() + allpaths = _get_table_drawings(page, native=True) else: # accept passed-in vector graphics allpaths = npaths[:] # paths relevant for table detection paths = [] + bbox_paths = [] # paths that may contribute an enveloping bbox for p in allpaths: - # If only looking at lines, we ignore fill-only paths, - # except simulated lines (i.e. small width or height). + # If only looking at lines, we ignore fill-only paths, except + # simulated lines (i.e. small width or height) -- including the ones + # batched inside a large path. Some producers put hundreds of thin + # grid rules into a single fill path, whose aggregate bbox is large + # although every relevant item in it is a simulated line. A path's + # bbox is not evidence about its individual items, so drop the path + # and keep those items (see _is_batched_rule). They contribute no + # enveloping bbox: bbox_paths holds only the paths whose own rect is + # real graphics. if ( lines_strict and p["type"] == "f" and p["rect"].width > snap_x and p["rect"].height > snap_y ): + line_items = [ + item + for item in p["items"] + if item[0] == "re" and _is_batched_rule(item[1].normalize(), min_length) + ] + if line_items: + line_path = p.copy() + line_path["items"] = line_items + paths.append(line_path) continue paths.append(p) + bbox_paths.append(p) # start with all vector graphics rectangles - prects = sorted(set([p["rect"] for p in paths]), key=lambda r: (r.y1, r.x0)) + prects = sorted(set([p["rect"] for p in bbox_paths]), key=lambda r: (r.y1, r.x0)) new_rects = [] # the final list of joined rectangles # ---------------------------------------------------------------- # Strategy: Join rectangles that "almost touch" each other. @@ -2548,23 +2675,15 @@ def make_line(p, p1, p2, clip): # the ones that simulate a line rect = i[1].normalize() # normalize the rectangle - if ( - rect.width <= min_length and rect.width < rect.height - ): # simulates a vertical line - x = abs(rect.x1 + rect.x0) / 2 # take middle value for x - p1 = pymupdf.Point(x, rect.y0) - p2 = pymupdf.Point(x, rect.y1) - line_dict = make_line(p, p1, p2, clip) - if line_dict: - edges.append(line_to_edge(line_dict)) - continue - - if ( - rect.height <= min_length and rect.height < rect.width - ): # simulates a horizontal line - y = abs(rect.y1 + rect.y0) / 2 # take middle value for y - p1 = pymupdf.Point(rect.x0, y) - p2 = pymupdf.Point(rect.x1, y) + if _is_line_like(rect, min_length): # simulates a single line + if rect.width < rect.height: # a vertical one + x = abs(rect.x1 + rect.x0) / 2 # take middle value for x + p1 = pymupdf.Point(x, rect.y0) + p2 = pymupdf.Point(x, rect.y1) + else: # a horizontal one + y = abs(rect.y1 + rect.y0) / 2 # take middle value for y + p1 = pymupdf.Point(rect.x0, y) + p2 = pymupdf.Point(rect.x1, y) line_dict = make_line(p, p1, p2, clip) if line_dict: edges.append(line_to_edge(line_dict)) @@ -2754,7 +2873,7 @@ def _refine_flat_placement_grid(page, cells, col_count): rect = pymupdf.Rect(cell) line_words = [ _span_word_line_tuple(word) - for _, word in _span_select_words_in_rect(page_words, rect) + for _, word in _span_select_words_in_rect(page_words, rect, page=page) ] out.append( SpanCell( @@ -2797,16 +2916,91 @@ def _refine_tag_grid(grid, top_header_rows): return grid +def _refine_model_grid(page, cells): + """The merge-preserved model grid plus the header counts derived from it. + + Returns ``(grid, header_rows, body_start)``. ``grid`` is None when span + resolution or the header rules fail. ``header_rows`` is the header finder's + own count, where 0 means it found no header row at all; ``body_start`` is + that count clamped to [1, rows], which is what the row splitter needs. One + resolve_spans pass serves both the header/body boundary and the repeated- + header split check, so the split costs no extra pass. + """ + try: + grid = _refine_placement_or_flat_grid(page, cells) + region = find_header_region(_refine_placements_text_grid(grid)) + except Exception: + return None, 0, 1 + header_rows = int(region.top_header_rows) + return grid, header_rows, (max(1, min(header_rows, len(cells))) if cells else 0) + + def _refine_body_start_row(page, cells): """Header/body boundary: resolve the merge-preserved placement grid once and ask the header finder how many leading rows are header, clamped to [1, rows].""" - try: - model_grid = _refine_placement_or_flat_grid(page, cells) - region = find_header_region(_refine_placements_text_grid(model_grid)) - except Exception: - return 1 - raw = region.top_header_rows - return max(1, min(int(raw), len(cells))) if cells else 0 + grid, _header_rows, body_start = _refine_model_grid(page, cells) + return 1 if grid is None else body_start + + +def _refine_repeated_leading_header_cuts(rows, header_rows): + """Rows at which an exactly repeated leading header splits a stacked table. + + Several tables printed one under another with no gap are detected as one + grid, and the giveaway is that their shared header row recurs verbatim. The + signature is the first row that carries at least two non-empty cells in a + grid at least two columns wide and is not purely numeric or punctuation; it + must also be a leading row -- inside the header region the header rules found, + or within the first three rows, because that finder is conservative and a + title row and a units row often precede the real header. A repeat further down + is data, not a header. Repeats are matched on case- and whitespace-normalized + text only: no fuzzy matching. + + A header signature also carries the grid's column structure, so it must hold + at least half as many placements as the widest row: a row of a few spanning + cells is a group header -- a sub-header such as "From- / To-" under an + eight-column header, which recurs inside one table -- and not the header. + + A repeat is accepted as a cut only if the result really looks like a stack of + tables rather than a table that happens to repeat a row: every segment must + hold at least one row below its own signature row. That single structural + invariant rejects both adjacent duplicate header rows (one header, not a cut) + and a trailing repeat with no records under it, and it needs no row-count + constant. Each cut row opens its segment by construction; the first segment + may carry a title row above the signature. Returns the cut row indices, or () + for no split. + """ + normalized = [ + tuple(collapse_cell_ws(text).casefold() for text in row) for row in rows + ] + if not normalized: + return () + width = max(len(row) for row in normalized) + if width < 2: + return () + # Every candidate signature is a leading row, so the scan is bounded; the + # first such row that actually recurs is the one that splits. + for index in range(min(max(int(header_rows), 3), len(normalized))): + signature = normalized[index] + if len(signature) * 2 < width: + continue # a few spanning cells: a group header, not the header + nonempty = [text for text in signature if text] + if len(nonempty) < 2: + continue + if not any(char.isalpha() for text in nonempty for char in text): + continue # a repeated all-numeric row is data, not a header + repeats = [ + row_index + for row_index in range(index + 1, len(normalized)) + if normalized[row_index] == signature + ] + if not repeats: + continue + signature_rows = [index, *repeats] + segment_ends = [*repeats, len(normalized)] + if all(end - start >= 2 for start, end in zip(signature_rows, segment_ends)): + return tuple(repeats) + return () + return () def _refine_build_placements(page, working, body_start): @@ -2820,6 +3014,58 @@ def _refine_build_placements(page, working, body_start): return tagged, region +def _refine_grid_tables(page, tab): + """Refine one detected table's grid and yield the resulting tagged tables. + + Order: structural split (shaded rows + under-segmented columns), the model + grid that carries both the header/body boundary and the repeated-header split + decision, then per segment the body-row split, the final merged-cell + placement grid and the td/th tagging. A grid whose leading header row recurs + verbatim is several tables printed under one another and yields one table per + segment; that is the ordinary case of one segment, which yields exactly the + table the unsplit path produced. + """ + grid = _refine_cells_to_grid(tab.cells) + # The reported bbox (a union grid-ref table's layout box, else the cells' + # union) bounds the shaded-rectangle search. + working = refine_grid_structure(page, grid, table_bbox=tab.bbox) + model_grid, header_rows, body_start = _refine_model_grid(page, working) + if model_grid is None: + body_start = 1 + cuts = () + else: + # The model grid has one row per row of `working`, so its cut indices + # index `working` directly. + cuts = _refine_repeated_leading_header_cuts( + _refine_placements_text_grid(model_grid), header_rows + ) + if cuts: + boundaries = (0, *cuts, len(working)) + segments = [working[start:end] for start, end in zip(boundaries, boundaries[1:])] + else: + segments = [working] + was_split = len(segments) > 1 + for segment in segments: + # A split segment has its own header region; without a split this is the + # boundary the model grid already produced. + start = _refine_body_start_row(page, segment) if was_split else body_start + segment = refine_grid_rows(page, segment, header_row_count=start) + flat = _refine_grid_to_cells(segment) + if not flat and was_split: + continue # an all-gap segment produces no table of its own + # Preserve an explicit reported-bbox override (union grid-ref tables): the + # refined grid must not change the reported region. A split does change + # it, so each segment reports its own cells. + new_tab = Table(page, flat, bbox=None if was_split else tab._bbox) if flat else tab + # Build the tagged model on `segment` directly, not on the re-gridded + # new_tab.cells, so placements match the refined grid. + placements, region = _refine_build_placements(page, segment, start) + new_tab.placements = placements + new_tab.header_rows = region.top_header_rows + new_tab.section_rows = region.section_header_rows + yield new_tab + + def find_tables( page, clip=None, @@ -2849,6 +3095,7 @@ def find_tables( use_layout: bool = True, # gate line-based tables by layout table boxes union: bool = False, # opt-in: fuse layout grids with line-based candidates refine: bool = False, # opt-in: refine each detected table's cell grid + _cell_group_filter=None, # internal: union admission, before Table construction ): """Detect and extract tables on a page. @@ -2874,7 +3121,9 @@ def find_tables( the default path), the header meta as Table.header_rows/section_rows, and Table.to_html() serializes it. Off by default. A grid whose column count the span resolution cannot preserve falls back to a flat one-cell-per-slot - placement grid. + placement grid. A refined grid whose leading header row recurs verbatim is + split into one table per recurrence: that is several tables printed under one + another which the line grid detected as one. """ pymupdf._warn_layout_once() _CHARS_VAR.set([]) @@ -2886,6 +3135,16 @@ def find_tables( else: old_xref, old_rot, old_mediabox = None, None, None + # Share one drawings extraction for the duration of this call (see + # _get_table_drawings). A nested call -- the union path runs one -- finds the + # cache already there and must neither replace nor remove it. + owns_drawings = getattr(page, "_table_drawings_cache", None) is None + if owns_drawings: + try: + page._table_drawings_cache = [] + except Exception: + owns_drawings = False + if snap_x_tolerance is None: snap_x_tolerance = UNSET if snap_y_tolerance is None: @@ -2934,7 +3193,7 @@ def find_tables( # the nested finder. Imported here, not at module top, to avoid an # import cycle: _table_union imports find_tables back from this module. from pymupdf._table_union import _find_tables_union - tbf = _find_tables_union(page) + tbf = _find_tables_union(page, add_lines=add_lines, add_boxes=add_boxes) TEXTPAGE = tbf.textpage else: boxes = [] @@ -2965,7 +3224,7 @@ def find_tables( add_boxes=add_boxes, ) # create lines and curves - tbf = TableFinder(page, settings=tset) + tbf = TableFinder(page, settings=tset, cell_group_filter=_cell_group_filter) tbf.textpage = TEXTPAGE # store textpage for later use if boxes: # only keep Finder tables that match a layout box @@ -2996,28 +3255,25 @@ def find_tables( # as .header_rows/.section_rows. refined_tables = [] for tab in tbf.tables: - grid = _refine_cells_to_grid(tab.cells) - # The reported bbox (a union grid-ref table's layout box, else - # the cells' union) bounds the shaded-rectangle search. - working = refine_grid_structure(page, grid, table_bbox=tab.bbox) - body_start = _refine_body_start_row(page, working) - working = refine_grid_rows(page, working, header_row_count=body_start) - flat = _refine_grid_to_cells(working) - # Preserve an explicit reported-bbox override (union grid-ref - # tables): the refined grid must not change the reported region. - new_tab = Table(page, flat, bbox=tab._bbox) if flat else tab - # Build the tagged model on `working` directly, not on the - # re-gridded new_tab.cells, so placements match the refined grid. - placements, region = _refine_build_placements(page, working, body_start) - new_tab.placements = placements - new_tab.header_rows = region.top_header_rows - new_tab.section_rows = region.section_header_rows - refined_tables.append(new_tab) + refined_tables.extend(_refine_grid_tables(page, tab)) tbf.tables = refined_tables + if old_xref is not None: + # The header rules read the page itself -- a pixmap of the top row + # and the text above the table -- in the detected cells' coordinate + # system. On a derotated page that system disappears when the finally + # block restores the rotation, so resolve the header now. Pages that + # were not rotated keep the lazy header. + for table in tbf.tables: + table._header = table._get_header() except Exception as e: pymupdf.message("find_tables: exception occurred: %s" % str(e)) return None finally: + if owns_drawings: # before the rotation reset rebinds `page` + try: + del page._table_drawings_cache + except Exception: + pass pymupdf.TOOLS.set_small_glyph_heights(old_small) if old_xref is not None: page = page_rotation_reset(page, old_xref, old_rot, old_mediabox) diff --git a/tests/test_tables.py b/tests/test_tables.py index 9fbeec387..5ca26c359 100644 --- a/tests/test_tables.py +++ b/tests/test_tables.py @@ -227,6 +227,115 @@ def test_strict_lines(): assert tab2.col_count < tab1.col_count +def test_strict_lines_accepts_thin_rects_batched_in_large_fill_path(): + """Line-like rect items survive a large fill-only path's bbox filter.""" + doc = pymupdf.open() + page = doc.new_page(width=240, height=180) + shape = page.new_shape() + x_values = (40, 120, 200) + y_values = (30, 90, 150) + for x in x_values: + shape.draw_rect(pymupdf.Rect(x - 0.4, y_values[0], x + 0.4, y_values[-1])) + for y in y_values: + shape.draw_rect(pymupdf.Rect(x_values[0], y - 0.4, x_values[-1], y + 0.4)) + shape.finish(color=None, fill=(0, 0, 0)) + shape.commit() + for row in range(2): + for col in range(2): + page.insert_text( + (x_values[col] + 10, y_values[row] + 25), + f"r{row}c{col}", + ) + + paths = page.get_drawings() + assert len(paths) == 1 + assert paths[0]["type"] == "f" + assert paths[0]["rect"].width > 3 and paths[0]["rect"].height > 3 + + tables = page.find_tables(strategy="lines_strict", use_layout=False).tables + assert len(tables) == 1 + assert tables[0].row_count == 2 + assert tables[0].col_count == 2 + doc.close() + + +def test_strict_lines_ignores_a_dot_screen_batched_in_a_large_fill_path(): + """Short thin rects in a large fill path are a dot screen, not grid rules. + + This page batches 196 rects of 1.1 x 0.9 pt into one fill path: a dotted + shading, not a grid. Each is thin enough to look like a simulated line, and a + few hundred of them at one y snap and join into a single long edge, which + turns an empty region into a table. Only rects long enough to be a rule are + kept (see _is_batched_rule). + """ + filename = os.path.join(scriptdir, "resources", "test_3594.pdf") + doc = pymupdf.open(filename) + page = doc[14] + try: + paths = [ + p + for p in page.get_drawings() + if p["type"] == "f" and p["rect"].width > 3 and p["rect"].height > 3 + ] + assert any(len(p["items"]) > 100 for p in paths) + assert page.find_tables(strategy="lines_strict", use_layout=False).tables == [] + finally: + doc.close() + + +def test_get_table_drawings_is_shared_only_inside_one_call(): + """The drawings cache lives for one find_tables() call; copies stay isolated.""" + from pymupdf.table import _get_table_drawings + + doc = pymupdf.open() + page = doc.new_page(width=200, height=200) + page.draw_rect(pymupdf.Rect(20, 20, 120, 60)) + try: + # Without the cache every consumer extracts afresh, exactly as before. + assert _get_table_drawings(page) is not _get_table_drawings(page) + + page._table_drawings_cache = [] # what find_tables() installs + first = _get_table_drawings(page) + assert first is _get_table_drawings(page) + assert first == page.get_drawings() + rect = pymupdf.Rect(first[0]["rect"]) + item_count = len(first[0]["items"]) + + native = _get_table_drawings(page, native=True) + assert native is not first + native[0]["rect"] |= pymupdf.Point(180, 180) + native[0]["items"].append(("l", pymupdf.Point(0, 0), pymupdf.Point(1, 1))) + # The edge builder's in-place edits must not reach the other consumers. + cached = _get_table_drawings(page) + assert pymupdf.Rect(cached[0]["rect"]) == rect + assert len(cached[0]["items"]) == item_count + assert pymupdf.Rect(native[0]["rect"]) != rect # the copy really changed + + del page._table_drawings_cache + page.find_tables(use_layout=False) # must not leave the cache behind + assert not hasattr(page, "_table_drawings_cache") + finally: + doc.close() + + +def test_table_header_is_computed_on_first_access(): + """Table.header is lazy but keeps its value, so to_markdown still works.""" + doc = pymupdf.open(filename) + page = doc[0] + try: + tab = page.find_tables().tables[0] + assert tab._header is pymupdf.table._NO_HEADER_YET + header = tab.header + assert header is tab.header # computed once + assert header.external is False + assert header.names == tab.extract()[0] + assert tab.to_markdown().startswith("|") + tab.header = None # assignable, as before + assert tab.header is None + finally: + doc.close() + + def test_add_lines(): """Test new parameter add_lines for table recognition.""" if platform.python_implementation() == 'GraalVM': @@ -676,6 +785,123 @@ def test_find_tables_refine_splits_rows_default_unchanged(): doc.close() +def test_refine_repeated_leading_header_cuts_exact_signature(): + """A leading header row that recurs verbatim splits stacked tables.""" + rows = [ + ["First table", "", ""], + ["Effective Date", "BI", "PD"], + ["2024-01-01", "1", "2"], + ["2024-02-01", "3", "4"], + ["2024-03-01", "5", "6"], + ["Second table", "", ""], + [" effective date ", "bi", "PD"], + ["2023-01-01", "7", "8"], + ["2023-02-01", "9", "10"], + ["2023-03-01", "11", "12"], + ] + cuts = pymupdf.table._refine_repeated_leading_header_cuts + assert cuts(rows, 2) == (6,) + # The header finder is conservative: a title row above the header can leave it + # reporting one header row, and the signature is still a leading row. + assert cuts(rows, 1) == (6,) + assert cuts(rows, 0) == (6,) + # A two-column stack splits too: the rule reads the grid's own width. + assert cuts([row[:2] for row in rows], 2) == (6,) + + +def test_refine_repeated_leading_header_cuts_rejects_body_and_short_segments(): + """Repeated body data and adjacent duplicate headers leave the table intact.""" + repeated_body = [ + ["Report", "", ""], + ["Name", "Value", "Note"], + ["a", "1", "x"], + ["same", "record", "value"], + ["b", "2", "y"], + ["c", "3", "z"], + ["d", "4", "w"], + ["same", "record", "value"], + ["e", "5", "q"], + ["f", "6", "r"], + ] + adjacent_duplicate = [ + ["Name", "Value", "Note"], + ["a", "1", "x"], + ["b", "2", "y"], + ["c", "3", "z"], + ["Name", "Value", "Note"], + ["Name", "Value", "Note"], + ["d", "4", "w"], + ["e", "5", "q"], + ] + # A two-row header, as on IRS form 940-B: the narrow sub-header under the + # eight-column header recurs inside the one table. + spanning_subheader = [ + ["State", "Reporting No.", "Payroll", "Rate period", "Rate", "Paid", "Due", "Total"], + ["From-", "To-"], + ["", "", "", "", "", "", "", ""], + ["State agency"], + ["Fax", "Mail to", "Other"], + ["State", "Rate", "Taxable", "Rate", "Paid", "Due", "Total"], + ["From-", "To-"], + ["", "", "", "", "", "", "", ""], + ] + cuts = pymupdf.table._refine_repeated_leading_header_cuts + # The repeated row is body data: it is not a leading row. + assert cuts(repeated_body, 2) == () + # Two spanning cells do not carry an eight-column header's structure. + assert cuts(spanning_subheader, 1) == () + # A repeated all-numeric leading row is data too. + assert cuts([["1", "2"], ["a", "b"], ["1", "2"], ["c", "d"]], 1) == () + # Two header rows in a row are one header, not a cut. + assert cuts(adjacent_duplicate, 1) == () + # A trailing repeat with no record under it is not a segment. + assert cuts([["H", "H2"], ["a", "b"], ["H", "H2"]], 1) == () + + +def test_find_tables_refine_splits_a_repeated_leading_header(): + """Two tables printed under one another are one grid; refine=True splits it. + + Nothing about a repeated header is union-specific, so the split applies + whenever refine=True. The default path still reports the single grid. + + *** PyMuPDF extension. *** + """ + texts = [ + ["Date", "BI", "PD"], + ["2024-01", "1", "2"], + ["2024-02", "3", "4"], + ["Date", "BI", "PD"], + ["2023-01", "7", "8"], + ["2023-02", "9", "10"], + ] + doc = pymupdf.open() + page = doc.new_page(width=400, height=400) + x_values = (60, 160, 230, 300) + y0, row_height = 60, 22 + for row in range(len(texts) + 1): + y = y0 + row * row_height + page.draw_line((x_values[0], y), (x_values[-1], y)) + for x in x_values: + page.draw_line((x, y0), (x, y0 + len(texts) * row_height)) + for row, values in enumerate(texts): + for column, value in enumerate(values): + page.insert_text( + (x_values[column] + 4, y0 + row * row_height + 15), value + ) + try: + default = page.find_tables(use_layout=False).tables + assert len(default) == 1 + assert default[0].row_count == 6 + + refined = page.find_tables(use_layout=False, refine=True).tables + assert len(refined) == 2 + assert [t.extract() for t in refined] == [texts[:3], texts[3:]] + # Each segment reports its own region, not the parent's. + assert refined[0].bbox[3] <= refined[1].bbox[1] + 1 + finally: + doc.close() + + def _make_merged_header_page(): """A page whose line grid detects a header cell that spans both body columns. @@ -939,3 +1165,177 @@ def test_find_tables_union_no_layout_degrades_to_line_candidates(): finally: pymupdf._get_layout = original_get_layout_fn doc.close() + + +def _make_bordered_grid(page, bbox, texts): + """Draw a uniformly divided grid with the same shape as ``texts``.""" + x0, y0, x1, y1 = bbox + row_count = len(texts) + col_count = len(texts[0]) + row_height = (y1 - y0) / row_count + col_width = (x1 - x0) / col_count + for row in range(row_count + 1): + y = y0 + row * row_height + page.draw_line((x0, y), (x1, y)) + for column in range(col_count + 1): + x = x0 + column * col_width + page.draw_line((x, y0), (x, y1)) + for row, values in enumerate(texts): + for column, value in enumerate(values): + if value: + page.insert_text( + ( + x0 + column * col_width + 5, + y0 + row * row_height + min(14, row_height - 3), + ), + value, + ) + + +def test_find_tables_union_forwards_virtual_lines_to_candidates(): + """Virtual raster-style rules reach the nested line finder in union mode.""" + doc = pymupdf.open() + page = doc.new_page(width=400, height=400) + for row, y in enumerate((100, 140)): + for col, x in enumerate((80, 180)): + page.insert_text((x + 8, y + 24), f"r{row}c{col}") + page.layout_information = [] + lines = [ + ((80, y), (280, y)) for y in (100, 140, 180) + ] + [ + ((x, 100), (x, 180)) for x in (80, 180, 280) + ] + try: + tables = page.find_tables( + use_layout=True, + union=True, + add_lines=lines, + ).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 2) + assert tables[0].extract()[1][1] == "r1c1" + finally: + doc.close() + + +def test_find_tables_union_rejects_multiline_single_row_panel(): + """A one-row partition around a whole picture is not emitted as a table.""" + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid( + page, + (80, 80, 380, 240), + [["left\naxis\nlabels", "right\naxis\nlabels"]], + ) + page.layout_information = [ + {"class_name": "picture", "group_bbox": [80.0, 80.0, 380.0, 240.0]}, + ] + try: + assert page.find_tables(use_layout=True, union=True).tables == [] + finally: + doc.close() + + +def test_find_tables_union_inside_picture_needs_table_shaped_text(): + """Inside a picture region a grid is admitted only on table-shaped text. + + The Layout model called the region a picture, so a line grid found almost + entirely inside one is as likely to be a chart's rules as a table. A chart + labels its two axes, so its text sits in exactly one row and one column; + a table -- even a sparse matrix of scattered marks -- has a header row and a + label column that are each at least half full. Outside a picture nothing + changes. + """ + picture = [{"class_name": "picture", "group_bbox": [120.0, 80.0, 440.0, 360.0]}] + # Chart shape: only the first row and the first column carry text. + axes = [ + ["", "x0", "x1", "x2"], + ["y0", "", "", ""], + ["y1", "", "", ""], + ["y2", "", "", ""], + ] + dense = [["a0", "b0", "c0"], ["a1", "b1", "c1"]] + # Sparse matrix: header row, label column and a few scattered marks. + matrix = [ + ["use", "z1", "z2", "z3", "z4", "z5"], + ["r1", "P", "P", "", "", ""], + ["r2", "P", "P", "", "", ""], + ["r3", "", "", "", "", ""], + ["r4", "", "", "", "", ""], + ["r5", "", "", "", "", ""], + ] + + def find(texts, bbox, layout): + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid(page, bbox, texts) + page.layout_information = list(layout) + try: + return page.find_tables(use_layout=True, union=True).tables + finally: + doc.close() + + assert find(axes, (160, 160, 400, 300), picture) == [] + assert len(find(axes, (160, 160, 400, 300), [])) == 1 # only the picture gates it + + tables = find(dense, (160, 180, 380, 240), picture) + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 3) + assert tables[0].extract()[1][2] == "c1" + + tables = find(matrix, (140, 140, 420, 320), picture) + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (6, 6) + assert tables[0].extract()[1][1] == "P" + + +def test_find_tables_union_keeps_a_table_holding_a_picture_in_one_cell(): + """A table larger than the picture contains it and keeps the ordinary rule.""" + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + # A 200x200 grid whose upper-left cell is the picture; the right column has text. + page.draw_rect((100, 100, 300, 300)) + page.draw_line((100, 200), (300, 200)) + page.draw_line((200, 100), (200, 300)) + page.insert_text((210, 130), "right top") + page.insert_text((210, 230), "right bottom") + page.insert_text((110, 230), "left bottom") + page.layout_information = [ + {"class_name": "picture", "group_bbox": [100.0, 100.0, 200.0, 200.0]}, + ] + try: + tables = page.find_tables(use_layout=True, union=True).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 2) + assert tables[0].extract()[0][1] == "right top" + finally: + doc.close() + + +def test_find_tables_union_keeps_one_coherent_table_group_without_content_support(): + """One Layout table group is retained; ownership is not a rejection gate.""" + import types + + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid( + page, + (80, 80, 380, 240), + [["left\naxis\nlabels", "right\naxis\nlabels"]], + ) + page.layout_information = [ + { + "class_name": "table", + "group_bbox": [80.0, 80.0, 380.0, 240.0], + "table_grid": types.SimpleNamespace( + h_lines=[50.0, 100.0], + v_lines=[150.0], + ), + } + ] + try: + tables = page.find_tables(use_layout=True, union=True).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (1, 2) + finally: + doc.close() From 4572a91f46ed5e99e908e2f4895c2fe9d0fd766c Mon Sep 17 00:00:00 2001 From: "youchang.kim" Date: Thu, 24 Sep 2026 00:50:02 +0900 Subject: [PATCH 2/2] table: refine=True completes the header under a spanning header cell find_tables(refine=True) decides the header rows from the placement grid's text alone, so a header cell spanning several (but not all) columns can end the header one row early and leave the row that names those columns in the body. After the existing header rules, the header now extends over the next row when that row supplies a single-column label under every column of such a span, puts no plain number under a column that is already labeled, is not one full-width cell, and does not repeat the blank/number/text shape of the row after it (the first record of a regular body). Nested spans repeat the step; the header never shrinks. The rule reads only the placement grid's text and spans and runs where the refined placements are tagged, so it can change only the th/td tags, to_html() and Table.header_rows of refined tables. The default, lines_strict and refine=False paths do not reach it. Based on the header rules by Minsu Kim. --- src/_table_headers.py | 88 +++++++++++++++++++++++++++- src/table.py | 9 ++- tests/test_tables.py | 132 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 226 insertions(+), 3 deletions(-) diff --git a/src/_table_headers.py b/src/_table_headers.py index 00ae860ca..0c9f785ad 100644 --- a/src/_table_headers.py +++ b/src/_table_headers.py @@ -26,8 +26,9 @@ PyMuPDF table header detection and HTML serialization (opt-in extension). Pure text-grid module (no pymupdf import): the header-region rules operate on a -row-major ``[[cell text]]`` grid, and the serializer turns a tagged placement -grid into an HTML ````. Used only by find_tables(refine=True) (via +row-major ``[[cell text]]`` grid, extend_header_leaf_labels completes that region +from a placement grid's spans, and the serializer turns a tagged placement grid +into an HTML ``
``. Used only by find_tables(refine=True) (via pymupdf.table) and Table.to_html(); never runs on the default detection path. """ from __future__ import annotations @@ -925,6 +926,89 @@ def find_header_region(rows: list[list[str]]) -> HeaderRegion: ) +# --- Leaf labels under spanning header cells: placement grid -> depth -------- +# Digits with only currency, sign, grouping, decimal, percent or date +# punctuation read as a value; a single letter makes the cell a label. +_PLAIN_NUMBER_RE = re.compile(r"^[\s$€£(),.%\-–—+0-9/:]+$") + + +def _plain_number(text: str) -> bool: + return bool(_PLAIN_NUMBER_RE.match(text)) and any(char.isdigit() for char in text) + + +def _span_slots(grid) -> tuple[list[tuple[int, int, int, int, str]], int]: + """Place a ragged span grid on its slots, as an HTML renderer does. + + Returns one ``(row0, row1, col0, col1, text)`` entry per cell (end-exclusive + slots, whitespace-collapsed text) and the grid's column count.""" + occupied: set[tuple[int, int]] = set() + slots = [] + for row0, row in enumerate(grid): + col0 = 0 + for cell in row: + while (row0, col0) in occupied: + col0 += 1 + row1, col1 = row0 + max(1, cell.rowspan), col0 + max(1, cell.colspan) + occupied.update((r, c) for r in range(row0, row1) for c in range(col0, col1)) + slots.append((row0, row1, col0, col1, collapse_cell_ws(cell.text or ""))) + col0 = col1 + return slots, max((col for _, col in occupied), default=-1) + 1 + + +def _row_shape(slots, row: int, ncols: int) -> tuple[str, ...]: + """Per column, whether ``row`` shows nothing, a plain number or text there.""" + shape = [""] * ncols + for row0, row1, col0, col1, text in slots: + if row0 <= row < row1 and text: + kind = "number" if _plain_number(text) else "text" + for col in range(col0, min(col1, ncols)): + shape[col] = kind + return tuple(shape) + + +def extend_header_leaf_labels(grid, top_header_rows: int) -> int: + """Extend the header over the row naming the columns of a spanning header cell. + + A header cell spanning more than one column but not all of them, with no + single-column label under any of its columns, leaves those columns unnamed. + The row below the header joins it when it supplies the names: a non-empty + single-column cell under every column of such a span, no plain number under + a column that is already labeled, not one full-width cell, and not the same + blank/number/text shape per column as the row after it (that repeat marks + the first record of a regular body). Repeats for nested spans; never shrinks + the header, leaves a header of 0 rows alone and never takes the last row, so + the table always keeps a body row. + + ``grid`` is a row-major placement grid duck-typed like ``render_table_html`` + input (``text`` / ``colspan`` / ``rowspan``). Only its text and spans are + read, so the result depends on nothing but the grid itself. + """ + slots, ncols = _span_slots(grid) + depth = top_header_rows + if not 0 < depth < len(grid) or ncols < 2: + return depth + while depth < len(grid) - 1: # a header never takes the whole table + header = [slot for slot in slots if slot[1] <= depth and slot[4]] + labeled = {col0 for _, _, col0, col1, _ in header if col1 - col0 == 1} + spans = [ + range(col0, col1) + for _, _, col0, col1, _ in header + if 1 < col1 - col0 < ncols and labeled.isdisjoint(range(col0, col1)) + ] + row = [(col0, col1, text) for row0, _, col0, col1, text in slots if row0 == depth and text] + if not spans or not row or (len(row) == 1 and row[0][1] - row[0][0] == ncols): + break + leaves = {col0 for col0, col1, _ in row if col1 - col0 == 1} + if not any(leaves.issuperset(span) for span in spans): + break + if any(_plain_number(text) and not labeled.isdisjoint(range(col0, col1)) for col0, col1, text in row): + break + if _row_shape(slots, depth, ncols) == _row_shape(slots, depth + 1, ncols): + break + depth += 1 + return depth + + # --- HTML serialization: tagged placement grid ->
-------------------- def collapse_cell_ws(text: str) -> str: """Whitespace-collapse a cell's text (runs of whitespace/newlines -> one space).""" diff --git a/src/table.py b/src/table.py index 991714306..e86e174c5 100644 --- a/src/table.py +++ b/src/table.py @@ -104,6 +104,8 @@ ) # Header semantics and HTML serialization live in the _table_headers sibling. from pymupdf._table_headers import ( + HeaderRegion, + extend_header_leaf_labels, find_header_region, collapse_cell_ws, render_table_html, @@ -3005,11 +3007,16 @@ def _refine_repeated_leading_header_cuts(rows, header_rows): def _refine_build_placements(page, working, body_start): """Resolve the final placement grid (strict colspan, header boundary known), - run header rules on its own text grid, tag cells -> (tagged grid, region).""" + run header rules on its own text grid, complete the header under spanning + header cells from the grid's spans, tag cells -> (tagged grid, region).""" grid = _refine_placement_or_flat_grid( page, working, strict_colspan=True, header_row_count=body_start ) region = find_header_region(_refine_placements_text_grid(grid)) + # The text-grid rules cannot see spans. A row added here holds at least two + # labels, so it is never a section row and the section rows stand. + depth = extend_header_leaf_labels(grid, region.top_header_rows) + region = HeaderRegion(depth, region.section_header_rows) tagged = _refine_tag_grid(grid, region.top_header_rows) return tagged, region diff --git a/tests/test_tables.py b/tests/test_tables.py index 5ca26c359..c3623a989 100644 --- a/tests/test_tables.py +++ b/tests/test_tables.py @@ -1078,6 +1078,138 @@ def cell(text, colspan=1, rowspan=1, tag="td"): ) +def _span_grid(rows): + """A placement grid from rows of cell text or ``(text, colspan, rowspan)``.""" + from pymupdf.table import SpanCell + + return [ + [SpanCell(None, *((cell, 1, 1) if isinstance(cell, str) else cell)) for cell in row] + for row in rows + ] + + +def test_find_tables_refine_tags_leaf_labels_under_a_spanning_header(): + """The row naming the columns of a header cell that spans some of them joins + the header: find_tables(refine=True) tags it th. The default result is + unchanged. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + texts = [ + ["Product", "Availability", None], + ["", "Online", "In store"], + ["Laptop", "Yes", "No"], + ["Phone", "No", "Yes"], + ["Tablet", "Yes", "Yes"], + ] + doc = pymupdf.open() + page = doc.new_page(width=400, height=300) + x_values = (60, 160, 240, 320) + y0, row_height = 60, 20 + y1 = y0 + len(texts) * row_height + for row in range(len(texts) + 1): + y = y0 + row * row_height + page.draw_line((x_values[0], y), (x_values[-1], y)) + for x in (60, 160, 320): + page.draw_line((x, y0), (x, y1)) + page.draw_line((240, y0 + row_height), (240, y1)) # no divider in the header row + for row, values in enumerate(texts): + for column, value in enumerate(values): + if value: + page.insert_text((x_values[column] + 4, y0 + row * row_height + 14), value) + try: + default = page.find_tables(use_layout=False).tables[0] + assert (default.placements, default.header_rows) == (None, 0) + assert default.extract() == texts + + t = page.find_tables(use_layout=False, refine=True).tables[0] + assert t.header_rows == 2 + assert t.to_html() == ( + "
" + '' + "" + "" + "" + "" + "
ProductAvailability
OnlineIn store
LaptopYesNo
PhoneNoYes
TabletYesYes
" + ) + finally: + doc.close() + + # A stub label spanning both header rows; an already named span is left alone. + grid = _span_grid([ + [("Product", 1, 2), ("Units sold", 2, 1)], + ["Online", "In store"], + ["Laptop", "120", "45"], + ["Phone", "300", "80"], + ]) + assert extend_header_leaf_labels(grid, 1) == 2 + assert extend_header_leaf_labels(grid, 2) == 2 + + +def test_header_leaf_labels_keep_a_numeric_body_row(): + """A record under a spanning header cell stays in the body: its blank/number/ + text shape repeats in the next row, or it puts a number under a column that + already has a label. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + repeated_shape = _span_grid([ + ["Region", ("Revenue", 2, 1)], + ["East", "10", "12"], + ["West", "11", "13"], + ]) + assert extend_header_leaf_labels(repeated_shape, 1) == 1 + number_under_label = _span_grid([ + ["Year", ("Sales", 2, 1)], + ["2023", "North", "South"], + ["2024", "10", "12"], + ]) + assert extend_header_leaf_labels(number_under_label, 1) == 1 + + +def test_header_leaf_labels_need_a_spanning_header_cell(): + """Without a header cell spanning more than one column but not all of them, + the header is left alone even when the next row reads like labels. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + single_cells = _span_grid([ + ["Product", "Online", "In store"], + ["", "units", "units"], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(single_cells, 1) == 1 + # A cell spanning every column is a title, not a group of columns. + full_width_title = _span_grid([ + [("Units sold", 3, 1)], + ["Product", "Online", "In store"], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(full_width_title, 1) == 1 + + +def test_header_leaf_labels_never_take_the_last_row(): + """A header never takes the whole table: the single record row under a + spanning header cell stays in the body. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + single_record = _span_grid([ + ["Product", ("Units sold", 2, 1)], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(single_record, 1) == 1 + + def _make_bordered_table(page, x0, y0, texts): """Draw a bordered 2x2 table (cells 100 wide, 20 tall) at (x0, y0), with the 2x2 ``texts`` grid inserted into its cells; returns nothing (mutates page)."""