diff --git a/src/_table_headers.py b/src/_table_headers.py
index 00ae860ca..0c9f785ad 100644
--- a/src/_table_headers.py
+++ b/src/_table_headers.py
@@ -26,8 +26,9 @@
PyMuPDF table header detection and HTML serialization (opt-in extension).
Pure text-grid module (no pymupdf import): the header-region rules operate on a
-row-major ``[[cell text]]`` grid, and the serializer turns a tagged placement
-grid into an HTML ``
``. Used only by find_tables(refine=True) (via
+row-major ``[[cell text]]`` grid, extend_header_leaf_labels completes that region
+from a placement grid's spans, and the serializer turns a tagged placement grid
+into an HTML ````. Used only by find_tables(refine=True) (via
pymupdf.table) and Table.to_html(); never runs on the default detection path.
"""
from __future__ import annotations
@@ -925,6 +926,89 @@ def find_header_region(rows: list[list[str]]) -> HeaderRegion:
)
+# --- Leaf labels under spanning header cells: placement grid -> depth --------
+# Digits with only currency, sign, grouping, decimal, percent or date
+# punctuation read as a value; a single letter makes the cell a label.
+_PLAIN_NUMBER_RE = re.compile(r"^[\s$€£(),.%\-–—+0-9/:]+$")
+
+
+def _plain_number(text: str) -> bool:
+ return bool(_PLAIN_NUMBER_RE.match(text)) and any(char.isdigit() for char in text)
+
+
+def _span_slots(grid) -> tuple[list[tuple[int, int, int, int, str]], int]:
+ """Place a ragged span grid on its slots, as an HTML renderer does.
+
+ Returns one ``(row0, row1, col0, col1, text)`` entry per cell (end-exclusive
+ slots, whitespace-collapsed text) and the grid's column count."""
+ occupied: set[tuple[int, int]] = set()
+ slots = []
+ for row0, row in enumerate(grid):
+ col0 = 0
+ for cell in row:
+ while (row0, col0) in occupied:
+ col0 += 1
+ row1, col1 = row0 + max(1, cell.rowspan), col0 + max(1, cell.colspan)
+ occupied.update((r, c) for r in range(row0, row1) for c in range(col0, col1))
+ slots.append((row0, row1, col0, col1, collapse_cell_ws(cell.text or "")))
+ col0 = col1
+ return slots, max((col for _, col in occupied), default=-1) + 1
+
+
+def _row_shape(slots, row: int, ncols: int) -> tuple[str, ...]:
+ """Per column, whether ``row`` shows nothing, a plain number or text there."""
+ shape = [""] * ncols
+ for row0, row1, col0, col1, text in slots:
+ if row0 <= row < row1 and text:
+ kind = "number" if _plain_number(text) else "text"
+ for col in range(col0, min(col1, ncols)):
+ shape[col] = kind
+ return tuple(shape)
+
+
+def extend_header_leaf_labels(grid, top_header_rows: int) -> int:
+ """Extend the header over the row naming the columns of a spanning header cell.
+
+ A header cell spanning more than one column but not all of them, with no
+ single-column label under any of its columns, leaves those columns unnamed.
+ The row below the header joins it when it supplies the names: a non-empty
+ single-column cell under every column of such a span, no plain number under
+ a column that is already labeled, not one full-width cell, and not the same
+ blank/number/text shape per column as the row after it (that repeat marks
+ the first record of a regular body). Repeats for nested spans; never shrinks
+ the header, leaves a header of 0 rows alone and never takes the last row, so
+ the table always keeps a body row.
+
+ ``grid`` is a row-major placement grid duck-typed like ``render_table_html``
+ input (``text`` / ``colspan`` / ``rowspan``). Only its text and spans are
+ read, so the result depends on nothing but the grid itself.
+ """
+ slots, ncols = _span_slots(grid)
+ depth = top_header_rows
+ if not 0 < depth < len(grid) or ncols < 2:
+ return depth
+ while depth < len(grid) - 1: # a header never takes the whole table
+ header = [slot for slot in slots if slot[1] <= depth and slot[4]]
+ labeled = {col0 for _, _, col0, col1, _ in header if col1 - col0 == 1}
+ spans = [
+ range(col0, col1)
+ for _, _, col0, col1, _ in header
+ if 1 < col1 - col0 < ncols and labeled.isdisjoint(range(col0, col1))
+ ]
+ row = [(col0, col1, text) for row0, _, col0, col1, text in slots if row0 == depth and text]
+ if not spans or not row or (len(row) == 1 and row[0][1] - row[0][0] == ncols):
+ break
+ leaves = {col0 for col0, col1, _ in row if col1 - col0 == 1}
+ if not any(leaves.issuperset(span) for span in spans):
+ break
+ if any(_plain_number(text) and not labeled.isdisjoint(range(col0, col1)) for col0, col1, text in row):
+ break
+ if _row_shape(slots, depth, ncols) == _row_shape(slots, depth + 1, ncols):
+ break
+ depth += 1
+ return depth
+
+
# --- HTML serialization: tagged placement grid -> --------------------
def collapse_cell_ws(text: str) -> str:
"""Whitespace-collapse a cell's text (runs of whitespace/newlines -> one space)."""
diff --git a/src/_table_refine.py b/src/_table_refine.py
index dc8984cac..7f2ce66fd 100644
--- a/src/_table_refine.py
+++ b/src/_table_refine.py
@@ -33,6 +33,8 @@
import itertools
import re
+from bisect import bisect_left, bisect_right
+from math import isfinite
import pymupdf
@@ -51,6 +53,70 @@
_REFINE_LINE_GAP = 3.0 # center-y gap (points) that groups body words into lines
+# --- word center index: reduce the per-cell word scan to a sorted band -------
+class _WordCenterIndex:
+ """Two sorted word-center axes over a read-only page word list.
+
+ ``candidates(rect)`` returns a superset of the words whose center lies in
+ ``rect``, so every consumer still applies its own membership, blank-text and
+ claiming rules. The original word indices are preserved, including order and
+ duplicates, because they are what identifies a word to its owning cell.
+ """
+
+ def __init__(self, words):
+ xs, ys = [], []
+ for i, (x0, y0, x1, y1, _text) in enumerate(words):
+ x, y = (x0 + x1) * 0.5, (y0 + y1) * 0.5
+ if not (isfinite(x) and isfinite(y)):
+ raise ValueError("nonfinite word center")
+ xs.append((x, i))
+ ys.append((y, i))
+ self.xs, self.ys = sorted(xs), sorted(ys)
+
+ def candidates(self, rect):
+ x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1)
+ n = len(self.xs)
+ if not all(map(isfinite, (x0, y0, x1, y1))):
+ return range(n) # preserve the consumers' predicates for NaN / Inf
+ if x0 > x1 or y0 > y1:
+ return []
+ xl, xr = bisect_left(self.xs, (x0, -1)), bisect_right(self.xs, (x1, n))
+ yl, yr = bisect_left(self.ys, (y0, -1)), bisect_right(self.ys, (y1, n))
+ # Scan the narrower band: fewer candidates for the same result.
+ entries = self.xs[xl:xr] if xr - xl <= yr - yl else self.ys[yl:yr]
+ return sorted(i for _center, i in entries)
+
+
+def _refine_word_index(page, words):
+ """The page word list's center index, cached on the page object.
+
+ Keyed on the word list's identity: _refine_page_words returns one stable list
+ per page and a fresh extraction produces a new list, so a changed text state
+ builds a new index. In-place edits of a cached word list are not supported.
+ Returns None for word lists the index cannot represent (see candidates()).
+ """
+ cached = getattr(page, "_table_word_index_cache", None)
+ if cached is not None and cached[0] is words:
+ return cached[1]
+ try:
+ index = _WordCenterIndex(words)
+ except (TypeError, ValueError, OverflowError):
+ index = None # nonstandard inputs keep the original full scan
+ try:
+ setattr(page, "_table_word_index_cache", (words, index))
+ except Exception:
+ pass
+ return index
+
+
+def _refine_word_candidates(page, words, rect):
+ """(index, word) pairs that may lie in rect -- a superset of the members."""
+ index = _refine_word_index(page, words) if page is not None else None
+ if index is None:
+ return enumerate(words)
+ return ((i, words[i]) for i in index.candidates(rect))
+
+
# --- word selection: center-point membership + rotated-span substitution -----
def _refine_rawdict_spans(page):
"""Flattened rawdict text spans carrying line-direction metadata.
@@ -155,7 +221,9 @@ def _refine_words_in_rect(page, rect):
x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1)
return [
(wx0, wy0, wx1, wy1, text)
- for wx0, wy0, wx1, wy1, text in _refine_page_words(page)
+ for _, (wx0, wy0, wx1, wy1, text) in _refine_word_candidates(
+ page, _refine_page_words(page), rect
+ )
if _refine_word_in_rect(wx0, wy0, wx1, wy1, x0, y0, x1, y1)
]
@@ -202,9 +270,11 @@ def _refine_table_rect(cells, table_bbox):
def _refine_raw_shaded_rects(page, table_rect, *, min_dim):
+ from pymupdf.table import _get_table_drawings
+
out = []
page_width = float(page.rect.width)
- for drawing in page.get_drawings():
+ for drawing in _get_table_drawings(page):
if _refine_is_white(drawing.get("fill")):
continue
for item in drawing.get("items", []):
@@ -253,9 +323,11 @@ def _refine_cluster(values, *, tolerance):
def _refine_border_lines(page, table_rect):
+ from pymupdf.table import _get_table_drawings
+
xs = set()
ys = set()
- for drawing in page.get_drawings():
+ for drawing in _get_table_drawings(page):
stroked = drawing.get("type") in ("s", "fs")
for item in drawing.get("items", []):
kind = item[0]
diff --git a/src/_table_spans.py b/src/_table_spans.py
index ff0d89d40..a15dacc26 100644
--- a/src/_table_spans.py
+++ b/src/_table_spans.py
@@ -36,6 +36,7 @@
from pymupdf._table_refine import (
_refine_is_vertical_or_rotated,
_refine_page_words,
+ _refine_word_candidates,
)
@@ -292,13 +293,15 @@ def _span_word_line_tuple(word):
return (float(y0), float(x0), float(y1), str(text))
-def _span_select_words_in_rect(page_words, rect):
+def _span_select_words_in_rect(page_words, rect, *, page=None):
"""(index, word) pairs whose center lies in rect, index into ``page_words``.
The index is what lets resolve_spans claim each page word for exactly one
- placement (an earlier cell's word is not re-claimed by a later one)."""
+ placement (an earlier cell's word is not re-claimed by a later one). ``page``
+ only supplies the cached word-center index, which narrows the scan without
+ changing the result."""
selected = []
- for index, word in enumerate(page_words):
+ for index, word in _refine_word_candidates(page, page_words, rect):
wx0, wy0, wx1, wy1, text = word
if not str(text).strip():
continue
@@ -321,7 +324,7 @@ def _span_claim_text_in_rect(page, rect, page_words, claimed_words):
"""Text of rect's words, skipping words already claimed and claiming the rest."""
selected = [
(index, word)
- for index, word in _span_select_words_in_rect(page_words, rect)
+ for index, word in _span_select_words_in_rect(page_words, rect, page=page)
if index not in claimed_words
]
for index, _ in selected:
@@ -486,7 +489,7 @@ def _span_reject_colspan_mismatch_merge(*, row_idx, cols, base, body_start):
def _span_cell_texts_for_entries(page, entries, start, end, page_words):
texts = []
for entry in entries[start : end + 1]:
- words = _span_select_words_in_rect(page_words, entry)
+ words = _span_select_words_in_rect(page_words, entry, page=page)
texts.append(_span_words_text_for_rect(page, entry, words))
return texts
diff --git a/src/_table_union.py b/src/_table_union.py
index d7a4d00e0..bd3a31003 100644
--- a/src/_table_union.py
+++ b/src/_table_union.py
@@ -28,9 +28,12 @@
TableFinder and _iou come from pymupdf.table; find_tables is imported lazily.
"""
+from bisect import bisect_left
+from collections import namedtuple
+
import pymupdf
-from pymupdf.table import CHARS, EDGES, Table, TableFinder, _iou
+from pymupdf.table import CHARS, EDGES, Table, TableFinder, _cells_to_rows, _iou
# ---------------------------------------------------------------------------
@@ -53,6 +56,12 @@
_UNION_GRID_REF_SPAN_MULT_THRESHOLD = 3.0 # max horizontally-separated span groups per cell
_UNION_OWNER_CONTAINMENT = 0.85 # min containment for a split candidate's owner
_UNION_OWNER_AMBIGUOUS_OVERLAP = 0.25 # overlap above which an unowned candidate is suppressed
+_UNION_PICTURE_CONTAINMENT = 0.8 # area fraction that puts a candidate inside a picture
+
+# How a candidate grid's text is laid out; see _union_grid_content_support.
+_UnionContent = namedtuple(
+ "_UnionContent", "supported dense_rows dense_columns concrete slots"
+)
def _layout_table_grids(page):
@@ -88,35 +97,83 @@ def _layout_table_grids(page):
return grids
-def _union_line_candidates(page):
+def _union_line_candidates(page, *, add_lines=None, add_boxes=None):
"""Line-based table candidates for the union stage as ``(bbox, grid)`` pairs.
Runs a nested find_tables (strategy=_UNION_STRATEGY, use_layout=False) and
- keeps each detected table's bbox and row-major cell grid (Table.rows, None
- for a gap), deduped by rounded bbox. Returns ``(candidates, finder)``; the
- finder is reused as the returned TableFinder shell.
+ keeps each detected table's bbox and row-major cell grid (None for a gap),
+ deduped by rounded bbox. Caller-supplied virtual lines / boxes are forwarded
+ to that nested finder so they take part in the same union decisions as
+ PDF-native vector rules. Admission (see ``admit``) runs on the finder's cell
+ groups, before a Table is built for any of them. Returns ``(candidates,
+ finder)``; the finder is reused as the returned TableFinder shell.
"""
# Imported here, not at module top, to break the import cycle: this
# module is itself imported lazily by table.find_tables (union path).
from pymupdf.table import find_tables
- finder = find_tables(page, strategy=_UNION_STRATEGY, use_layout=False)
candidates = []
- seen = set()
- for tab in (getattr(finder, "tables", None) or []):
- try:
- bbox = pymupdf.Rect(tab.bbox)
- except (ValueError, TypeError):
- continue
- if bbox.is_empty:
- continue
- grid = [[cell for cell in row.cells] for row in (tab.rows or [])]
- if not grid:
- continue
- key = tuple(round(value) for value in bbox)
- if key in seen:
- continue
- seen.add(key)
- candidates.append((bbox, grid))
+
+ def admit(live_page, groups):
+ """Keep the cell groups that may become tables, in detection order.
+
+ A candidate is admitted as before: the line grid is evidence in its own
+ right and the layout model can miss a table entirely. The one exception is
+ a candidate sitting inside a region the layout model called a picture,
+ which has to show table-shaped text instead. That cheap geometric test
+ therefore gates the character scan, and the scan itself is built once for
+ the whole call.
+ """
+ kept = []
+ seen = set()
+ points = None
+ for cells in groups:
+ bbox = pymupdf.Rect(
+ min(cell[0] for cell in cells),
+ min(cell[1] for cell in cells),
+ max(cell[2] for cell in cells),
+ max(cell[3] for cell in cells),
+ )
+ if bbox.is_empty:
+ continue
+ grid = [row.cells for row in _cells_to_rows(cells)]
+ if not grid:
+ continue
+ key = tuple(round(value) for value in bbox)
+ if key in seen:
+ continue
+ if _union_candidate_inside_picture(live_page, bbox):
+ if points is None:
+ points = _union_char_midpoints()
+ content = _union_grid_content_support(grid, points)
+ # A chart labels one row and one column -- its axes -- and its
+ # partial gridlines leave most slots without a cell. A table has
+ # a header row and a label column that are each at least half
+ # full, and cell coverage that is mostly rectangular.
+ if not (
+ content.supported
+ and content.dense_rows >= 2
+ and content.dense_columns >= 2
+ and content.concrete * 2 >= content.slots
+ ):
+ continue
+ # Dedup only admitted candidates: a rejected one must not shadow a
+ # later, differently gridded candidate with the same rounded bbox.
+ seen.add(key)
+ kept.append(cells)
+ candidates.append((bbox, grid))
+ return kept
+
+ finder = find_tables(
+ page,
+ strategy=_UNION_STRATEGY,
+ use_layout=False,
+ add_lines=add_lines,
+ add_boxes=add_boxes,
+ _cell_group_filter=admit,
+ )
+ if finder is None:
+ # Nested detection failure must not leak partially admitted candidates.
+ candidates.clear()
return candidates, finder
@@ -165,6 +222,161 @@ def _union_find_owner(candidate_bbox, existing_bboxes):
return best_owner, ambiguous
+def _union_char_midpoints():
+ """The page's non-blank character midpoints as ``(v_mid, h_mid)``, y-sorted.
+
+ One pass over CHARS replaces the per-cell character scan of the admission
+ tests, which read midpoints only. Sorting by the vertical midpoint lets each
+ grid row take its own characters as one slice.
+ """
+ points = []
+ for char in CHARS:
+ if not str(char.get("text") or "").strip():
+ continue
+ try:
+ h_mid = (float(char["x0"]) + float(char["x1"])) / 2.0
+ v_mid = (float(char["top"]) + float(char["bottom"])) / 2.0
+ except (KeyError, TypeError, ValueError):
+ continue
+ points.append((v_mid, h_mid))
+ points.sort()
+ return points
+
+
+def _union_grid_content_support(grid, points):
+ """How a candidate line grid's text is distributed, in one scan.
+
+ ``supported`` means at least two rows with text in at least two columns and
+ at least two columns with text in at least two rows -- two being the smallest
+ non-trivial count in each dimension, not a fitted one. A populated cell counts
+ for every column its x-range covers, so a grid whose missing horizontal rules
+ leave a spanning header above one record row is supported just like the same
+ table with all its rules.
+
+ ``dense_rows`` / ``dense_columns`` count the rows and columns whose own
+ concrete cells are at least half populated, and ``concrete`` / ``slots`` are
+ the concrete cells and the grid's total slots. Those describe the *shape* of
+ the text rather than its amount, which is what separates a chart from a
+ sparse table (see _union_candidate_inside_picture).
+
+ Character membership follows Table.extract()'s half-open cell rule. ``None``
+ grid slots are span/gap placeholders, not empty text cells. The decision is
+ candidate-local: page area, layout ownership, file identity and virtual-line
+ provenance are not inputs.
+ """
+ columns = sorted({float(cell[0]) for row in grid for cell in row if cell is not None})
+ width = max((len(row) for row in grid), default=0)
+ column_populated = [0] * width
+ column_concrete = [0] * width
+ row_columns = []
+ row_counts = []
+ concrete = 0
+ slots = 0
+ for row in grid:
+ slots += len(row)
+ covered = set()
+ row_populated = 0
+ rects = [(index, pymupdf.Rect(cell)) for index, cell in enumerate(row) if cell is not None]
+ rects = [(index, rect) for index, rect in rects if not rect.is_empty]
+ concrete += len(rects)
+ if rects:
+ band = _union_row_points(rects, points)
+ for index, rect in rects:
+ column_concrete[index] += 1
+ if not _union_rect_has_point(rect, band):
+ continue
+ row_populated += 1
+ column_populated[index] += 1
+ spanned = [
+ column
+ for column, x in enumerate(columns)
+ if float(rect.x0) <= x < float(rect.x1)
+ ]
+ covered.update(spanned or (index,))
+ row_columns.append(covered)
+ row_counts.append((row_populated, len(rects)))
+ supported = sum(len(covered) >= 2 for covered in row_columns) >= 2 and sum(
+ sum(column in covered for covered in row_columns) >= 2
+ for column in range(len(columns))
+ ) >= 2
+ return _UnionContent(
+ supported=supported,
+ dense_rows=sum(total > 0 and filled * 2 >= total for filled, total in row_counts),
+ dense_columns=sum(
+ total > 0 and filled * 2 >= total
+ for filled, total in zip(column_populated, column_concrete)
+ ),
+ concrete=concrete,
+ slots=slots,
+ )
+
+
+def _union_row_points(rects, points):
+ """The y-sorted midpoints falling in one grid row's vertical band."""
+ y0 = min(float(rect.y0) for _index, rect in rects)
+ y1 = max(float(rect.y1) for _index, rect in rects)
+ return points[bisect_left(points, (y0,)) : bisect_left(points, (y1,))]
+
+
+def _union_rect_has_point(rect, band):
+ """Whether any midpoint of the row band lies in rect (half-open, as extract)."""
+ x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1)
+ for v_mid, h_mid in band:
+ if x0 <= h_mid < x1 and y0 <= v_mid < y1:
+ return True
+ return False
+
+
+def _union_layout_groups(page):
+ """The page's layout groups as ``(class_name, rect)``, skipping unusable ones."""
+ groups = []
+ for group in (page.layout_information or []):
+ if not isinstance(group, dict):
+ continue
+ group_bbox = group.get("group_bbox")
+ if not group_bbox:
+ continue
+ try:
+ rect = pymupdf.Rect(group_bbox[:4])
+ except (TypeError, ValueError):
+ continue
+ if rect.is_empty:
+ continue
+ groups.append((group.get("class_name"), rect))
+ return groups
+
+
+def _union_candidate_inside_picture(page, candidate_bbox):
+ """Whether a candidate sits inside a layout picture group without covering it.
+
+ The layout model classified that region as a picture, so a line grid found
+ almost entirely inside one is as likely to be a chart's gridlines as a table,
+ and overriding the model's verdict needs table-shaped text. The shape, not the
+ amount, is what tells them apart: a chart's text sits along its two axes, so
+ exactly one row and one column are well filled and its partial gridlines leave
+ most of the grid without a cell at all, while a table has a header row and a
+ label column that are each at least half full over mostly complete cell
+ coverage -- which a sparse matrix of scattered marks still satisfies. See the
+ admission rule in _union_line_candidates.
+
+ A candidate larger than the picture contains it instead -- a bordered table
+ with an image in one of its cells -- and keeps the ordinary rule.
+ """
+ candidate_area = _union_rect_area(candidate_bbox)
+ if candidate_area <= 0:
+ return False
+ threshold = _UNION_PICTURE_CONTAINMENT * candidate_area
+ for class_name, rect in _union_layout_groups(page):
+ if class_name != "picture":
+ continue
+ if _union_intersection_area(candidate_bbox, rect) < threshold:
+ continue
+ if candidate_area > _union_rect_area(rect):
+ continue # the candidate is the larger region: it holds the picture
+ return True
+ return False
+
+
def _union_text_span_rects(page):
"""Non-empty page text-span rects, cached on the page.
@@ -325,18 +537,21 @@ def _union_replace_append(existing, candidates, *, page, grid_ref, grid_ref_iou,
return entries
-def _find_tables_union(page):
+def _find_tables_union(page, *, add_lines=None, add_boxes=None):
"""Detect a page's tables by fusing layout grids with line-based candidates.
Ensures the raw layout (computed only when page.layout_information is None,
- like the official use_layout path), reads primary grids, detects candidates,
- applies grid-ref / split / append, and returns a TableFinder whose .tables
- carry the fused grids in contractual order (grid-ref tables keep their
- explicit layout bbox)."""
+ like the official use_layout path), reads primary grids, detects candidates
+ (forwarding the caller's virtual lines / boxes to the nested finder), applies
+ grid-ref / split / append, and returns a TableFinder whose .tables carry the
+ fused grids in contractual order (grid-ref tables keep their explicit layout
+ bbox)."""
if page.layout_information is None:
page.get_layout(return_raw=True)
primaries = _layout_table_grids(page)
- candidates, finder = _union_line_candidates(page)
+ candidates, finder = _union_line_candidates(
+ page, add_lines=add_lines, add_boxes=add_boxes
+ )
entries = _union_replace_append(
primaries,
candidates,
diff --git a/src/table.py b/src/table.py
index 7b8e2c19c..e86e174c5 100644
--- a/src/table.py
+++ b/src/table.py
@@ -104,6 +104,8 @@
)
# Header semantics and HTML serialization live in the _table_headers sibling.
from pymupdf._table_headers import (
+ HeaderRegion,
+ extend_header_leaf_labels,
find_header_region,
collapse_cell_ws,
render_table_html,
@@ -123,6 +125,44 @@
_CHARS_VAR = ContextVar("pymupdf_table_chars", default=None)
+def _get_table_drawings(page, *, native=False):
+ """The page's vector drawings, shared within one find_tables() call.
+
+ One call reads them up to ``1 + 2T`` times -- once to build edges, and once
+ per refined table for shaded rows and for border lines -- and each read walks
+ the whole content stream. find_tables() therefore puts an empty cache on the
+ page for the duration of the call and removes it again, so nothing outlives
+ the call: a page whose content stream is rewritten between calls cannot be
+ served a stale extraction. Calling the public refine helpers directly, with
+ no find_tables() around them, extracts afresh exactly as before.
+
+ ``native=True`` returns per-path copies of the mutated members: make_edges
+ extends path rects, normalizes rectangle items and appends a closing line, so
+ the other consumers must not see its edits. Point and Quad values stay shared
+ because make_edges only reads them.
+ """
+ cache = getattr(page, "_table_drawings_cache", None)
+ if cache is None:
+ paths = page.get_drawings()
+ else:
+ if not cache:
+ cache.append(page.get_drawings())
+ paths = cache[0]
+ if not native:
+ return paths
+ return [
+ dict(
+ path,
+ rect=pymupdf.Rect(path["rect"]),
+ items=[
+ (item[0], pymupdf.Rect(item[1]), *item[2:]) if item[0] == "re" else item
+ for item in path["items"]
+ ],
+ )
+ for path in paths
+ ]
+
+
class _TableStateList:
"""List-like proxy for per-call table extraction state."""
@@ -1607,6 +1647,25 @@ def __init__(self, bbox, cells, names, above):
self.external = above
+def _cells_to_rows(cells):
+ """Canonical row order and None gap slots for a flat cell list.
+
+ Each row holds one slot per distinct cell x0 on the table, so a gap and a
+ span both appear as None. Shared by Table.rows and the union stage, which
+ needs the same grid before a Table exists."""
+ _sorted = sorted(cells, key=itemgetter(1, 0))
+ xs = list(sorted(set(map(itemgetter(0), cells))))
+ rows = []
+ for y, row_cells in itertools.groupby(_sorted, itemgetter(1)):
+ xdict = {cell[0]: cell for cell in row_cells}
+ row = TableRow([xdict.get(x) for x in xs])
+ rows.append(row)
+ return rows
+
+
+_NO_HEADER_YET = object() # Table.header has not been computed yet
+
+
class Table:
def __init__(self, page, cells, bbox=None):
self.page = page
@@ -1617,7 +1676,14 @@ def __init__(self, page, cells, bbox=None):
# cells. Set only for a union grid-ref table, whose reported region
# (its layout box) is decoupled from its replacement cell grid.
self._bbox = bbox
- self.header = self._get_header() # PyMuPDF extension
+ self._header = _NO_HEADER_YET
+ # The header rules read the page through the two global text-extraction
+ # toggles that were in force when this table was detected, so a header
+ # computed on first access has to put them back. Both are plain getters.
+ self._header_flags = (
+ bool(pymupdf.TOOLS.set_small_glyph_heights()),
+ bool(pymupdf.TOOLS.unset_quad_corrections()),
+ )
# Filled by find_tables(refine=True): placements is a row-major grid of
# tagged SpanCell colspan/rowspan placements (None otherwise); header_rows
# is the leading header-row count, section_rows the section-label rows.
@@ -1637,16 +1703,36 @@ def bbox(self):
max(map(itemgetter(3), c)),
)
+ @property
+ def header(self): # PyMuPDF extension
+ """The identified table header, computed on first access.
+
+ Identifying it renders the top row twice, so tables whose header is never
+ read -- every table serialized through placements, and every candidate a
+ later stage discards -- must not pay for it at construction time. The
+ header rules re-read the page, so the two global text-extraction toggles
+ of the detecting call are restored for the duration (see __init__): the
+ result is then the one an eager computation would have produced."""
+ if self._header is _NO_HEADER_YET:
+ small, quads = self._header_flags
+ old_small = bool(pymupdf.TOOLS.set_small_glyph_heights())
+ old_quads = bool(pymupdf.TOOLS.unset_quad_corrections())
+ pymupdf.TOOLS.set_small_glyph_heights(small)
+ pymupdf.TOOLS.unset_quad_corrections(quads)
+ try:
+ self._header = self._get_header()
+ finally:
+ pymupdf.TOOLS.set_small_glyph_heights(old_small)
+ pymupdf.TOOLS.unset_quad_corrections(old_quads)
+ return self._header
+
+ @header.setter
+ def header(self, value):
+ self._header = value
+
@property
def rows(self) -> list:
- _sorted = sorted(self.cells, key=itemgetter(1, 0))
- xs = list(sorted(set(map(itemgetter(0), self.cells))))
- rows = []
- for y, row_cells in itertools.groupby(_sorted, itemgetter(1)):
- xdict = {cell[0]: cell for cell in row_cells}
- row = TableRow([xdict.get(x) for x in xs])
- rows.append(row)
- return rows
+ return _cells_to_rows(self.cells)
@property
def row_count(self) -> int: # PyMuPDF extension
@@ -1869,9 +1955,12 @@ def row_has_bold(bbox):
Returns True if any spans are bold else False.
"""
+ # This table's own character snapshot, so a lazily computed header
+ # reads the same characters an eager one would have.
+ chars = self._chars if self._chars is not None else CHARS
return any(
c["bold"]
- for c in CHARS
+ for c in chars
if rect_in_rect((c["x0"], c["y0"], c["x1"], c["y1"]), bbox)
)
@@ -2153,7 +2242,7 @@ class TableFinder:
https://github.com/tabulapdf/tabula-extractor/issues/16
"""
- def __init__(self, page, settings=None):
+ def __init__(self, page, settings=None, *, cell_group_filter=None):
self.page = weakref.proxy(page)
self.textpage = None
self.settings = TableSettings.resolve(settings)
@@ -2164,10 +2253,12 @@ def __init__(self, page, settings=None):
self.settings.intersection_y_tolerance,
)
self.cells = intersections_to_cells(self.intersections)
- self.tables = [
- Table(self.page, cell_group)
- for cell_group in cells_to_tables(self.page, self.cells)
- ]
+ groups = cells_to_tables(self.page, self.cells)
+ if cell_group_filter is not None:
+ # Internal hook: the union stage decides admission on the cell groups,
+ # so a rejected candidate never becomes a Table at all.
+ groups = cell_group_filter(self.page, groups)
+ self.tables = [Table(self.page, group) for group in groups]
def get_edges(self) -> list:
settings = self.settings
@@ -2364,6 +2455,26 @@ def make_chars(page, clip=None):
# We are ignoring Bézier curves completely and are converting everything
# else to lines.
# ------------------------------------------------------------------------
+def _is_line_like(rect, min_length):
+ """Whether a rectangle is thin enough to be a simulated line."""
+ return (rect.width <= min_length and rect.width < rect.height) or (
+ rect.height <= min_length and rect.height < rect.width
+ )
+
+
+def _is_batched_rule(rect, min_length):
+ """Whether a rectangle inside a large fill path is a simulated line.
+
+ A path's aggregate bbox is no evidence about its individual items, so a thin
+ rectangle batched into a large fill can still be a grid rule. A thin *and
+ short* one cannot: dot screens and vector-drawn glyph fragments consist of
+ rectangles that are small in both directions, and a few hundred of them snap
+ and join into one long spurious edge. A rule reaches at least the minimum
+ edge length along its length, which is the same bound filter_edges() applies
+ to the finished edges."""
+ return _is_line_like(rect, min_length) and max(rect.width, rect.height) > min_length
+
+
def make_edges(page, clip=None, tset=None, paths=None, add_lines=None, add_boxes=None):
edges = EDGES._list() # bind once: avoid per-append proxy overhead below
snap_x = tset.snap_x_tolerance
@@ -2420,24 +2531,42 @@ def are_neighbors(r1, r2):
def clean_graphics(npaths=None):
"""Detect and join rectangles of "connected" vector graphics."""
if npaths is None:
- allpaths = page.get_drawings()
+ allpaths = _get_table_drawings(page, native=True)
else: # accept passed-in vector graphics
allpaths = npaths[:] # paths relevant for table detection
paths = []
+ bbox_paths = [] # paths that may contribute an enveloping bbox
for p in allpaths:
- # If only looking at lines, we ignore fill-only paths,
- # except simulated lines (i.e. small width or height).
+ # If only looking at lines, we ignore fill-only paths, except
+ # simulated lines (i.e. small width or height) -- including the ones
+ # batched inside a large path. Some producers put hundreds of thin
+ # grid rules into a single fill path, whose aggregate bbox is large
+ # although every relevant item in it is a simulated line. A path's
+ # bbox is not evidence about its individual items, so drop the path
+ # and keep those items (see _is_batched_rule). They contribute no
+ # enveloping bbox: bbox_paths holds only the paths whose own rect is
+ # real graphics.
if (
lines_strict
and p["type"] == "f"
and p["rect"].width > snap_x
and p["rect"].height > snap_y
):
+ line_items = [
+ item
+ for item in p["items"]
+ if item[0] == "re" and _is_batched_rule(item[1].normalize(), min_length)
+ ]
+ if line_items:
+ line_path = p.copy()
+ line_path["items"] = line_items
+ paths.append(line_path)
continue
paths.append(p)
+ bbox_paths.append(p)
# start with all vector graphics rectangles
- prects = sorted(set([p["rect"] for p in paths]), key=lambda r: (r.y1, r.x0))
+ prects = sorted(set([p["rect"] for p in bbox_paths]), key=lambda r: (r.y1, r.x0))
new_rects = [] # the final list of joined rectangles
# ----------------------------------------------------------------
# Strategy: Join rectangles that "almost touch" each other.
@@ -2548,23 +2677,15 @@ def make_line(p, p1, p2, clip):
# the ones that simulate a line
rect = i[1].normalize() # normalize the rectangle
- if (
- rect.width <= min_length and rect.width < rect.height
- ): # simulates a vertical line
- x = abs(rect.x1 + rect.x0) / 2 # take middle value for x
- p1 = pymupdf.Point(x, rect.y0)
- p2 = pymupdf.Point(x, rect.y1)
- line_dict = make_line(p, p1, p2, clip)
- if line_dict:
- edges.append(line_to_edge(line_dict))
- continue
-
- if (
- rect.height <= min_length and rect.height < rect.width
- ): # simulates a horizontal line
- y = abs(rect.y1 + rect.y0) / 2 # take middle value for y
- p1 = pymupdf.Point(rect.x0, y)
- p2 = pymupdf.Point(rect.x1, y)
+ if _is_line_like(rect, min_length): # simulates a single line
+ if rect.width < rect.height: # a vertical one
+ x = abs(rect.x1 + rect.x0) / 2 # take middle value for x
+ p1 = pymupdf.Point(x, rect.y0)
+ p2 = pymupdf.Point(x, rect.y1)
+ else: # a horizontal one
+ y = abs(rect.y1 + rect.y0) / 2 # take middle value for y
+ p1 = pymupdf.Point(rect.x0, y)
+ p2 = pymupdf.Point(rect.x1, y)
line_dict = make_line(p, p1, p2, clip)
if line_dict:
edges.append(line_to_edge(line_dict))
@@ -2754,7 +2875,7 @@ def _refine_flat_placement_grid(page, cells, col_count):
rect = pymupdf.Rect(cell)
line_words = [
_span_word_line_tuple(word)
- for _, word in _span_select_words_in_rect(page_words, rect)
+ for _, word in _span_select_words_in_rect(page_words, rect, page=page)
]
out.append(
SpanCell(
@@ -2797,29 +2918,161 @@ def _refine_tag_grid(grid, top_header_rows):
return grid
+def _refine_model_grid(page, cells):
+ """The merge-preserved model grid plus the header counts derived from it.
+
+ Returns ``(grid, header_rows, body_start)``. ``grid`` is None when span
+ resolution or the header rules fail. ``header_rows`` is the header finder's
+ own count, where 0 means it found no header row at all; ``body_start`` is
+ that count clamped to [1, rows], which is what the row splitter needs. One
+ resolve_spans pass serves both the header/body boundary and the repeated-
+ header split check, so the split costs no extra pass.
+ """
+ try:
+ grid = _refine_placement_or_flat_grid(page, cells)
+ region = find_header_region(_refine_placements_text_grid(grid))
+ except Exception:
+ return None, 0, 1
+ header_rows = int(region.top_header_rows)
+ return grid, header_rows, (max(1, min(header_rows, len(cells))) if cells else 0)
+
+
def _refine_body_start_row(page, cells):
"""Header/body boundary: resolve the merge-preserved placement grid once and
ask the header finder how many leading rows are header, clamped to [1, rows]."""
- try:
- model_grid = _refine_placement_or_flat_grid(page, cells)
- region = find_header_region(_refine_placements_text_grid(model_grid))
- except Exception:
- return 1
- raw = region.top_header_rows
- return max(1, min(int(raw), len(cells))) if cells else 0
+ grid, _header_rows, body_start = _refine_model_grid(page, cells)
+ return 1 if grid is None else body_start
+
+
+def _refine_repeated_leading_header_cuts(rows, header_rows):
+ """Rows at which an exactly repeated leading header splits a stacked table.
+
+ Several tables printed one under another with no gap are detected as one
+ grid, and the giveaway is that their shared header row recurs verbatim. The
+ signature is the first row that carries at least two non-empty cells in a
+ grid at least two columns wide and is not purely numeric or punctuation; it
+ must also be a leading row -- inside the header region the header rules found,
+ or within the first three rows, because that finder is conservative and a
+ title row and a units row often precede the real header. A repeat further down
+ is data, not a header. Repeats are matched on case- and whitespace-normalized
+ text only: no fuzzy matching.
+
+ A header signature also carries the grid's column structure, so it must hold
+ at least half as many placements as the widest row: a row of a few spanning
+ cells is a group header -- a sub-header such as "From- / To-" under an
+ eight-column header, which recurs inside one table -- and not the header.
+
+ A repeat is accepted as a cut only if the result really looks like a stack of
+ tables rather than a table that happens to repeat a row: every segment must
+ hold at least one row below its own signature row. That single structural
+ invariant rejects both adjacent duplicate header rows (one header, not a cut)
+ and a trailing repeat with no records under it, and it needs no row-count
+ constant. Each cut row opens its segment by construction; the first segment
+ may carry a title row above the signature. Returns the cut row indices, or ()
+ for no split.
+ """
+ normalized = [
+ tuple(collapse_cell_ws(text).casefold() for text in row) for row in rows
+ ]
+ if not normalized:
+ return ()
+ width = max(len(row) for row in normalized)
+ if width < 2:
+ return ()
+ # Every candidate signature is a leading row, so the scan is bounded; the
+ # first such row that actually recurs is the one that splits.
+ for index in range(min(max(int(header_rows), 3), len(normalized))):
+ signature = normalized[index]
+ if len(signature) * 2 < width:
+ continue # a few spanning cells: a group header, not the header
+ nonempty = [text for text in signature if text]
+ if len(nonempty) < 2:
+ continue
+ if not any(char.isalpha() for text in nonempty for char in text):
+ continue # a repeated all-numeric row is data, not a header
+ repeats = [
+ row_index
+ for row_index in range(index + 1, len(normalized))
+ if normalized[row_index] == signature
+ ]
+ if not repeats:
+ continue
+ signature_rows = [index, *repeats]
+ segment_ends = [*repeats, len(normalized)]
+ if all(end - start >= 2 for start, end in zip(signature_rows, segment_ends)):
+ return tuple(repeats)
+ return ()
+ return ()
def _refine_build_placements(page, working, body_start):
"""Resolve the final placement grid (strict colspan, header boundary known),
- run header rules on its own text grid, tag cells -> (tagged grid, region)."""
+ run header rules on its own text grid, complete the header under spanning
+ header cells from the grid's spans, tag cells -> (tagged grid, region)."""
grid = _refine_placement_or_flat_grid(
page, working, strict_colspan=True, header_row_count=body_start
)
region = find_header_region(_refine_placements_text_grid(grid))
+ # The text-grid rules cannot see spans. A row added here holds at least two
+ # labels, so it is never a section row and the section rows stand.
+ depth = extend_header_leaf_labels(grid, region.top_header_rows)
+ region = HeaderRegion(depth, region.section_header_rows)
tagged = _refine_tag_grid(grid, region.top_header_rows)
return tagged, region
+def _refine_grid_tables(page, tab):
+ """Refine one detected table's grid and yield the resulting tagged tables.
+
+ Order: structural split (shaded rows + under-segmented columns), the model
+ grid that carries both the header/body boundary and the repeated-header split
+ decision, then per segment the body-row split, the final merged-cell
+ placement grid and the td/th tagging. A grid whose leading header row recurs
+ verbatim is several tables printed under one another and yields one table per
+ segment; that is the ordinary case of one segment, which yields exactly the
+ table the unsplit path produced.
+ """
+ grid = _refine_cells_to_grid(tab.cells)
+ # The reported bbox (a union grid-ref table's layout box, else the cells'
+ # union) bounds the shaded-rectangle search.
+ working = refine_grid_structure(page, grid, table_bbox=tab.bbox)
+ model_grid, header_rows, body_start = _refine_model_grid(page, working)
+ if model_grid is None:
+ body_start = 1
+ cuts = ()
+ else:
+ # The model grid has one row per row of `working`, so its cut indices
+ # index `working` directly.
+ cuts = _refine_repeated_leading_header_cuts(
+ _refine_placements_text_grid(model_grid), header_rows
+ )
+ if cuts:
+ boundaries = (0, *cuts, len(working))
+ segments = [working[start:end] for start, end in zip(boundaries, boundaries[1:])]
+ else:
+ segments = [working]
+ was_split = len(segments) > 1
+ for segment in segments:
+ # A split segment has its own header region; without a split this is the
+ # boundary the model grid already produced.
+ start = _refine_body_start_row(page, segment) if was_split else body_start
+ segment = refine_grid_rows(page, segment, header_row_count=start)
+ flat = _refine_grid_to_cells(segment)
+ if not flat and was_split:
+ continue # an all-gap segment produces no table of its own
+ # Preserve an explicit reported-bbox override (union grid-ref tables): the
+ # refined grid must not change the reported region. A split does change
+ # it, so each segment reports its own cells.
+ new_tab = Table(page, flat, bbox=None if was_split else tab._bbox) if flat else tab
+ # Build the tagged model on `segment` directly, not on the re-gridded
+ # new_tab.cells, so placements match the refined grid.
+ placements, region = _refine_build_placements(page, segment, start)
+ new_tab.placements = placements
+ new_tab.header_rows = region.top_header_rows
+ new_tab.section_rows = region.section_header_rows
+ yield new_tab
+
+
def find_tables(
page,
clip=None,
@@ -2849,6 +3102,7 @@ def find_tables(
use_layout: bool = True, # gate line-based tables by layout table boxes
union: bool = False, # opt-in: fuse layout grids with line-based candidates
refine: bool = False, # opt-in: refine each detected table's cell grid
+ _cell_group_filter=None, # internal: union admission, before Table construction
):
"""Detect and extract tables on a page.
@@ -2874,7 +3128,9 @@ def find_tables(
the default path), the header meta as Table.header_rows/section_rows, and
Table.to_html() serializes it. Off by default. A grid whose column count the
span resolution cannot preserve falls back to a flat one-cell-per-slot
- placement grid.
+ placement grid. A refined grid whose leading header row recurs verbatim is
+ split into one table per recurrence: that is several tables printed under one
+ another which the line grid detected as one.
"""
pymupdf._warn_layout_once()
_CHARS_VAR.set([])
@@ -2886,6 +3142,16 @@ def find_tables(
else:
old_xref, old_rot, old_mediabox = None, None, None
+ # Share one drawings extraction for the duration of this call (see
+ # _get_table_drawings). A nested call -- the union path runs one -- finds the
+ # cache already there and must neither replace nor remove it.
+ owns_drawings = getattr(page, "_table_drawings_cache", None) is None
+ if owns_drawings:
+ try:
+ page._table_drawings_cache = []
+ except Exception:
+ owns_drawings = False
+
if snap_x_tolerance is None:
snap_x_tolerance = UNSET
if snap_y_tolerance is None:
@@ -2934,7 +3200,7 @@ def find_tables(
# the nested finder. Imported here, not at module top, to avoid an
# import cycle: _table_union imports find_tables back from this module.
from pymupdf._table_union import _find_tables_union
- tbf = _find_tables_union(page)
+ tbf = _find_tables_union(page, add_lines=add_lines, add_boxes=add_boxes)
TEXTPAGE = tbf.textpage
else:
boxes = []
@@ -2965,7 +3231,7 @@ def find_tables(
add_boxes=add_boxes,
) # create lines and curves
- tbf = TableFinder(page, settings=tset)
+ tbf = TableFinder(page, settings=tset, cell_group_filter=_cell_group_filter)
tbf.textpage = TEXTPAGE # store textpage for later use
if boxes:
# only keep Finder tables that match a layout box
@@ -2996,28 +3262,25 @@ def find_tables(
# as .header_rows/.section_rows.
refined_tables = []
for tab in tbf.tables:
- grid = _refine_cells_to_grid(tab.cells)
- # The reported bbox (a union grid-ref table's layout box, else
- # the cells' union) bounds the shaded-rectangle search.
- working = refine_grid_structure(page, grid, table_bbox=tab.bbox)
- body_start = _refine_body_start_row(page, working)
- working = refine_grid_rows(page, working, header_row_count=body_start)
- flat = _refine_grid_to_cells(working)
- # Preserve an explicit reported-bbox override (union grid-ref
- # tables): the refined grid must not change the reported region.
- new_tab = Table(page, flat, bbox=tab._bbox) if flat else tab
- # Build the tagged model on `working` directly, not on the
- # re-gridded new_tab.cells, so placements match the refined grid.
- placements, region = _refine_build_placements(page, working, body_start)
- new_tab.placements = placements
- new_tab.header_rows = region.top_header_rows
- new_tab.section_rows = region.section_header_rows
- refined_tables.append(new_tab)
+ refined_tables.extend(_refine_grid_tables(page, tab))
tbf.tables = refined_tables
+ if old_xref is not None:
+ # The header rules read the page itself -- a pixmap of the top row
+ # and the text above the table -- in the detected cells' coordinate
+ # system. On a derotated page that system disappears when the finally
+ # block restores the rotation, so resolve the header now. Pages that
+ # were not rotated keep the lazy header.
+ for table in tbf.tables:
+ table._header = table._get_header()
except Exception as e:
pymupdf.message("find_tables: exception occurred: %s" % str(e))
return None
finally:
+ if owns_drawings: # before the rotation reset rebinds `page`
+ try:
+ del page._table_drawings_cache
+ except Exception:
+ pass
pymupdf.TOOLS.set_small_glyph_heights(old_small)
if old_xref is not None:
page = page_rotation_reset(page, old_xref, old_rot, old_mediabox)
diff --git a/tests/test_tables.py b/tests/test_tables.py
index 9fbeec387..c3623a989 100644
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -227,6 +227,115 @@ def test_strict_lines():
assert tab2.col_count < tab1.col_count
+def test_strict_lines_accepts_thin_rects_batched_in_large_fill_path():
+ """Line-like rect items survive a large fill-only path's bbox filter."""
+ doc = pymupdf.open()
+ page = doc.new_page(width=240, height=180)
+ shape = page.new_shape()
+ x_values = (40, 120, 200)
+ y_values = (30, 90, 150)
+ for x in x_values:
+ shape.draw_rect(pymupdf.Rect(x - 0.4, y_values[0], x + 0.4, y_values[-1]))
+ for y in y_values:
+ shape.draw_rect(pymupdf.Rect(x_values[0], y - 0.4, x_values[-1], y + 0.4))
+ shape.finish(color=None, fill=(0, 0, 0))
+ shape.commit()
+ for row in range(2):
+ for col in range(2):
+ page.insert_text(
+ (x_values[col] + 10, y_values[row] + 25),
+ f"r{row}c{col}",
+ )
+
+ paths = page.get_drawings()
+ assert len(paths) == 1
+ assert paths[0]["type"] == "f"
+ assert paths[0]["rect"].width > 3 and paths[0]["rect"].height > 3
+
+ tables = page.find_tables(strategy="lines_strict", use_layout=False).tables
+ assert len(tables) == 1
+ assert tables[0].row_count == 2
+ assert tables[0].col_count == 2
+ doc.close()
+
+
+def test_strict_lines_ignores_a_dot_screen_batched_in_a_large_fill_path():
+ """Short thin rects in a large fill path are a dot screen, not grid rules.
+
+ This page batches 196 rects of 1.1 x 0.9 pt into one fill path: a dotted
+ shading, not a grid. Each is thin enough to look like a simulated line, and a
+ few hundred of them at one y snap and join into a single long edge, which
+ turns an empty region into a table. Only rects long enough to be a rule are
+ kept (see _is_batched_rule).
+ """
+ filename = os.path.join(scriptdir, "resources", "test_3594.pdf")
+ doc = pymupdf.open(filename)
+ page = doc[14]
+ try:
+ paths = [
+ p
+ for p in page.get_drawings()
+ if p["type"] == "f" and p["rect"].width > 3 and p["rect"].height > 3
+ ]
+ assert any(len(p["items"]) > 100 for p in paths)
+ assert page.find_tables(strategy="lines_strict", use_layout=False).tables == []
+ finally:
+ doc.close()
+
+
+def test_get_table_drawings_is_shared_only_inside_one_call():
+ """The drawings cache lives for one find_tables() call; copies stay isolated."""
+ from pymupdf.table import _get_table_drawings
+
+ doc = pymupdf.open()
+ page = doc.new_page(width=200, height=200)
+ page.draw_rect(pymupdf.Rect(20, 20, 120, 60))
+ try:
+ # Without the cache every consumer extracts afresh, exactly as before.
+ assert _get_table_drawings(page) is not _get_table_drawings(page)
+
+ page._table_drawings_cache = [] # what find_tables() installs
+ first = _get_table_drawings(page)
+ assert first is _get_table_drawings(page)
+ assert first == page.get_drawings()
+ rect = pymupdf.Rect(first[0]["rect"])
+ item_count = len(first[0]["items"])
+
+ native = _get_table_drawings(page, native=True)
+ assert native is not first
+ native[0]["rect"] |= pymupdf.Point(180, 180)
+ native[0]["items"].append(("l", pymupdf.Point(0, 0), pymupdf.Point(1, 1)))
+ # The edge builder's in-place edits must not reach the other consumers.
+ cached = _get_table_drawings(page)
+ assert pymupdf.Rect(cached[0]["rect"]) == rect
+ assert len(cached[0]["items"]) == item_count
+ assert pymupdf.Rect(native[0]["rect"]) != rect # the copy really changed
+
+ del page._table_drawings_cache
+ page.find_tables(use_layout=False) # must not leave the cache behind
+ assert not hasattr(page, "_table_drawings_cache")
+ finally:
+ doc.close()
+
+
+def test_table_header_is_computed_on_first_access():
+ """Table.header is lazy but keeps its value, so to_markdown still works."""
+ doc = pymupdf.open(filename)
+ page = doc[0]
+ try:
+ tab = page.find_tables().tables[0]
+ assert tab._header is pymupdf.table._NO_HEADER_YET
+ header = tab.header
+ assert header is tab.header # computed once
+ assert header.external is False
+ assert header.names == tab.extract()[0]
+ assert tab.to_markdown().startswith("|")
+ tab.header = None # assignable, as before
+ assert tab.header is None
+ finally:
+ doc.close()
+
+
def test_add_lines():
"""Test new parameter add_lines for table recognition."""
if platform.python_implementation() == 'GraalVM':
@@ -676,6 +785,123 @@ def test_find_tables_refine_splits_rows_default_unchanged():
doc.close()
+def test_refine_repeated_leading_header_cuts_exact_signature():
+ """A leading header row that recurs verbatim splits stacked tables."""
+ rows = [
+ ["First table", "", ""],
+ ["Effective Date", "BI", "PD"],
+ ["2024-01-01", "1", "2"],
+ ["2024-02-01", "3", "4"],
+ ["2024-03-01", "5", "6"],
+ ["Second table", "", ""],
+ [" effective date ", "bi", "PD"],
+ ["2023-01-01", "7", "8"],
+ ["2023-02-01", "9", "10"],
+ ["2023-03-01", "11", "12"],
+ ]
+ cuts = pymupdf.table._refine_repeated_leading_header_cuts
+ assert cuts(rows, 2) == (6,)
+ # The header finder is conservative: a title row above the header can leave it
+ # reporting one header row, and the signature is still a leading row.
+ assert cuts(rows, 1) == (6,)
+ assert cuts(rows, 0) == (6,)
+ # A two-column stack splits too: the rule reads the grid's own width.
+ assert cuts([row[:2] for row in rows], 2) == (6,)
+
+
+def test_refine_repeated_leading_header_cuts_rejects_body_and_short_segments():
+ """Repeated body data and adjacent duplicate headers leave the table intact."""
+ repeated_body = [
+ ["Report", "", ""],
+ ["Name", "Value", "Note"],
+ ["a", "1", "x"],
+ ["same", "record", "value"],
+ ["b", "2", "y"],
+ ["c", "3", "z"],
+ ["d", "4", "w"],
+ ["same", "record", "value"],
+ ["e", "5", "q"],
+ ["f", "6", "r"],
+ ]
+ adjacent_duplicate = [
+ ["Name", "Value", "Note"],
+ ["a", "1", "x"],
+ ["b", "2", "y"],
+ ["c", "3", "z"],
+ ["Name", "Value", "Note"],
+ ["Name", "Value", "Note"],
+ ["d", "4", "w"],
+ ["e", "5", "q"],
+ ]
+ # A two-row header, as on IRS form 940-B: the narrow sub-header under the
+ # eight-column header recurs inside the one table.
+ spanning_subheader = [
+ ["State", "Reporting No.", "Payroll", "Rate period", "Rate", "Paid", "Due", "Total"],
+ ["From-", "To-"],
+ ["", "", "", "", "", "", "", ""],
+ ["State agency"],
+ ["Fax", "Mail to", "Other"],
+ ["State", "Rate", "Taxable", "Rate", "Paid", "Due", "Total"],
+ ["From-", "To-"],
+ ["", "", "", "", "", "", "", ""],
+ ]
+ cuts = pymupdf.table._refine_repeated_leading_header_cuts
+ # The repeated row is body data: it is not a leading row.
+ assert cuts(repeated_body, 2) == ()
+ # Two spanning cells do not carry an eight-column header's structure.
+ assert cuts(spanning_subheader, 1) == ()
+ # A repeated all-numeric leading row is data too.
+ assert cuts([["1", "2"], ["a", "b"], ["1", "2"], ["c", "d"]], 1) == ()
+ # Two header rows in a row are one header, not a cut.
+ assert cuts(adjacent_duplicate, 1) == ()
+ # A trailing repeat with no record under it is not a segment.
+ assert cuts([["H", "H2"], ["a", "b"], ["H", "H2"]], 1) == ()
+
+
+def test_find_tables_refine_splits_a_repeated_leading_header():
+ """Two tables printed under one another are one grid; refine=True splits it.
+
+ Nothing about a repeated header is union-specific, so the split applies
+ whenever refine=True. The default path still reports the single grid.
+
+ *** PyMuPDF extension. ***
+ """
+ texts = [
+ ["Date", "BI", "PD"],
+ ["2024-01", "1", "2"],
+ ["2024-02", "3", "4"],
+ ["Date", "BI", "PD"],
+ ["2023-01", "7", "8"],
+ ["2023-02", "9", "10"],
+ ]
+ doc = pymupdf.open()
+ page = doc.new_page(width=400, height=400)
+ x_values = (60, 160, 230, 300)
+ y0, row_height = 60, 22
+ for row in range(len(texts) + 1):
+ y = y0 + row * row_height
+ page.draw_line((x_values[0], y), (x_values[-1], y))
+ for x in x_values:
+ page.draw_line((x, y0), (x, y0 + len(texts) * row_height))
+ for row, values in enumerate(texts):
+ for column, value in enumerate(values):
+ page.insert_text(
+ (x_values[column] + 4, y0 + row * row_height + 15), value
+ )
+ try:
+ default = page.find_tables(use_layout=False).tables
+ assert len(default) == 1
+ assert default[0].row_count == 6
+
+ refined = page.find_tables(use_layout=False, refine=True).tables
+ assert len(refined) == 2
+ assert [t.extract() for t in refined] == [texts[:3], texts[3:]]
+ # Each segment reports its own region, not the parent's.
+ assert refined[0].bbox[3] <= refined[1].bbox[1] + 1
+ finally:
+ doc.close()
+
+
def _make_merged_header_page():
"""A page whose line grid detects a header cell that spans both body columns.
@@ -852,6 +1078,138 @@ def cell(text, colspan=1, rowspan=1, tag="td"):
)
+def _span_grid(rows):
+ """A placement grid from rows of cell text or ``(text, colspan, rowspan)``."""
+ from pymupdf.table import SpanCell
+
+ return [
+ [SpanCell(None, *((cell, 1, 1) if isinstance(cell, str) else cell)) for cell in row]
+ for row in rows
+ ]
+
+
+def test_find_tables_refine_tags_leaf_labels_under_a_spanning_header():
+ """The row naming the columns of a header cell that spans some of them joins
+ the header: find_tables(refine=True) tags it th. The default result is
+ unchanged.
+
+ *** PyMuPDF extension (opt-in header rules). ***
+ """
+ from pymupdf._table_headers import extend_header_leaf_labels
+
+ texts = [
+ ["Product", "Availability", None],
+ ["", "Online", "In store"],
+ ["Laptop", "Yes", "No"],
+ ["Phone", "No", "Yes"],
+ ["Tablet", "Yes", "Yes"],
+ ]
+ doc = pymupdf.open()
+ page = doc.new_page(width=400, height=300)
+ x_values = (60, 160, 240, 320)
+ y0, row_height = 60, 20
+ y1 = y0 + len(texts) * row_height
+ for row in range(len(texts) + 1):
+ y = y0 + row * row_height
+ page.draw_line((x_values[0], y), (x_values[-1], y))
+ for x in (60, 160, 320):
+ page.draw_line((x, y0), (x, y1))
+ page.draw_line((240, y0 + row_height), (240, y1)) # no divider in the header row
+ for row, values in enumerate(texts):
+ for column, value in enumerate(values):
+ if value:
+ page.insert_text((x_values[column] + 4, y0 + row * row_height + 14), value)
+ try:
+ default = page.find_tables(use_layout=False).tables[0]
+ assert (default.placements, default.header_rows) == (None, 0)
+ assert default.extract() == texts
+
+ t = page.find_tables(use_layout=False, refine=True).tables[0]
+ assert t.header_rows == 2
+ assert t.to_html() == (
+ ""
+ '| Product | Availability |
'
+ " | Online | In store |
"
+ "| Laptop | Yes | No |
"
+ "| Phone | No | Yes |
"
+ "| Tablet | Yes | Yes |
"
+ "
"
+ )
+ finally:
+ doc.close()
+
+ # A stub label spanning both header rows; an already named span is left alone.
+ grid = _span_grid([
+ [("Product", 1, 2), ("Units sold", 2, 1)],
+ ["Online", "In store"],
+ ["Laptop", "120", "45"],
+ ["Phone", "300", "80"],
+ ])
+ assert extend_header_leaf_labels(grid, 1) == 2
+ assert extend_header_leaf_labels(grid, 2) == 2
+
+
+def test_header_leaf_labels_keep_a_numeric_body_row():
+ """A record under a spanning header cell stays in the body: its blank/number/
+ text shape repeats in the next row, or it puts a number under a column that
+ already has a label.
+
+ *** PyMuPDF extension (opt-in header rules). ***
+ """
+ from pymupdf._table_headers import extend_header_leaf_labels
+
+ repeated_shape = _span_grid([
+ ["Region", ("Revenue", 2, 1)],
+ ["East", "10", "12"],
+ ["West", "11", "13"],
+ ])
+ assert extend_header_leaf_labels(repeated_shape, 1) == 1
+ number_under_label = _span_grid([
+ ["Year", ("Sales", 2, 1)],
+ ["2023", "North", "South"],
+ ["2024", "10", "12"],
+ ])
+ assert extend_header_leaf_labels(number_under_label, 1) == 1
+
+
+def test_header_leaf_labels_need_a_spanning_header_cell():
+ """Without a header cell spanning more than one column but not all of them,
+ the header is left alone even when the next row reads like labels.
+
+ *** PyMuPDF extension (opt-in header rules). ***
+ """
+ from pymupdf._table_headers import extend_header_leaf_labels
+
+ single_cells = _span_grid([
+ ["Product", "Online", "In store"],
+ ["", "units", "units"],
+ ["Laptop", "120", "45"],
+ ])
+ assert extend_header_leaf_labels(single_cells, 1) == 1
+ # A cell spanning every column is a title, not a group of columns.
+ full_width_title = _span_grid([
+ [("Units sold", 3, 1)],
+ ["Product", "Online", "In store"],
+ ["Laptop", "120", "45"],
+ ])
+ assert extend_header_leaf_labels(full_width_title, 1) == 1
+
+
+def test_header_leaf_labels_never_take_the_last_row():
+ """A header never takes the whole table: the single record row under a
+ spanning header cell stays in the body.
+
+ *** PyMuPDF extension (opt-in header rules). ***
+ """
+ from pymupdf._table_headers import extend_header_leaf_labels
+
+ single_record = _span_grid([
+ ["Product", ("Units sold", 2, 1)],
+ ["Laptop", "120", "45"],
+ ])
+ assert extend_header_leaf_labels(single_record, 1) == 1
+
+
def _make_bordered_table(page, x0, y0, texts):
"""Draw a bordered 2x2 table (cells 100 wide, 20 tall) at (x0, y0), with the
2x2 ``texts`` grid inserted into its cells; returns nothing (mutates page)."""
@@ -939,3 +1297,177 @@ def test_find_tables_union_no_layout_degrades_to_line_candidates():
finally:
pymupdf._get_layout = original_get_layout_fn
doc.close()
+
+
+def _make_bordered_grid(page, bbox, texts):
+ """Draw a uniformly divided grid with the same shape as ``texts``."""
+ x0, y0, x1, y1 = bbox
+ row_count = len(texts)
+ col_count = len(texts[0])
+ row_height = (y1 - y0) / row_count
+ col_width = (x1 - x0) / col_count
+ for row in range(row_count + 1):
+ y = y0 + row * row_height
+ page.draw_line((x0, y), (x1, y))
+ for column in range(col_count + 1):
+ x = x0 + column * col_width
+ page.draw_line((x, y0), (x, y1))
+ for row, values in enumerate(texts):
+ for column, value in enumerate(values):
+ if value:
+ page.insert_text(
+ (
+ x0 + column * col_width + 5,
+ y0 + row * row_height + min(14, row_height - 3),
+ ),
+ value,
+ )
+
+
+def test_find_tables_union_forwards_virtual_lines_to_candidates():
+ """Virtual raster-style rules reach the nested line finder in union mode."""
+ doc = pymupdf.open()
+ page = doc.new_page(width=400, height=400)
+ for row, y in enumerate((100, 140)):
+ for col, x in enumerate((80, 180)):
+ page.insert_text((x + 8, y + 24), f"r{row}c{col}")
+ page.layout_information = []
+ lines = [
+ ((80, y), (280, y)) for y in (100, 140, 180)
+ ] + [
+ ((x, 100), (x, 180)) for x in (80, 180, 280)
+ ]
+ try:
+ tables = page.find_tables(
+ use_layout=True,
+ union=True,
+ add_lines=lines,
+ ).tables
+ assert len(tables) == 1
+ assert (tables[0].row_count, tables[0].col_count) == (2, 2)
+ assert tables[0].extract()[1][1] == "r1c1"
+ finally:
+ doc.close()
+
+
+def test_find_tables_union_rejects_multiline_single_row_panel():
+ """A one-row partition around a whole picture is not emitted as a table."""
+ doc = pymupdf.open()
+ page = doc.new_page(width=500, height=500)
+ _make_bordered_grid(
+ page,
+ (80, 80, 380, 240),
+ [["left\naxis\nlabels", "right\naxis\nlabels"]],
+ )
+ page.layout_information = [
+ {"class_name": "picture", "group_bbox": [80.0, 80.0, 380.0, 240.0]},
+ ]
+ try:
+ assert page.find_tables(use_layout=True, union=True).tables == []
+ finally:
+ doc.close()
+
+
+def test_find_tables_union_inside_picture_needs_table_shaped_text():
+ """Inside a picture region a grid is admitted only on table-shaped text.
+
+ The Layout model called the region a picture, so a line grid found almost
+ entirely inside one is as likely to be a chart's rules as a table. A chart
+ labels its two axes, so its text sits in exactly one row and one column;
+ a table -- even a sparse matrix of scattered marks -- has a header row and a
+ label column that are each at least half full. Outside a picture nothing
+ changes.
+ """
+ picture = [{"class_name": "picture", "group_bbox": [120.0, 80.0, 440.0, 360.0]}]
+ # Chart shape: only the first row and the first column carry text.
+ axes = [
+ ["", "x0", "x1", "x2"],
+ ["y0", "", "", ""],
+ ["y1", "", "", ""],
+ ["y2", "", "", ""],
+ ]
+ dense = [["a0", "b0", "c0"], ["a1", "b1", "c1"]]
+ # Sparse matrix: header row, label column and a few scattered marks.
+ matrix = [
+ ["use", "z1", "z2", "z3", "z4", "z5"],
+ ["r1", "P", "P", "", "", ""],
+ ["r2", "P", "P", "", "", ""],
+ ["r3", "", "", "", "", ""],
+ ["r4", "", "", "", "", ""],
+ ["r5", "", "", "", "", ""],
+ ]
+
+ def find(texts, bbox, layout):
+ doc = pymupdf.open()
+ page = doc.new_page(width=500, height=500)
+ _make_bordered_grid(page, bbox, texts)
+ page.layout_information = list(layout)
+ try:
+ return page.find_tables(use_layout=True, union=True).tables
+ finally:
+ doc.close()
+
+ assert find(axes, (160, 160, 400, 300), picture) == []
+ assert len(find(axes, (160, 160, 400, 300), [])) == 1 # only the picture gates it
+
+ tables = find(dense, (160, 180, 380, 240), picture)
+ assert len(tables) == 1
+ assert (tables[0].row_count, tables[0].col_count) == (2, 3)
+ assert tables[0].extract()[1][2] == "c1"
+
+ tables = find(matrix, (140, 140, 420, 320), picture)
+ assert len(tables) == 1
+ assert (tables[0].row_count, tables[0].col_count) == (6, 6)
+ assert tables[0].extract()[1][1] == "P"
+
+
+def test_find_tables_union_keeps_a_table_holding_a_picture_in_one_cell():
+ """A table larger than the picture contains it and keeps the ordinary rule."""
+ doc = pymupdf.open()
+ page = doc.new_page(width=500, height=500)
+ # A 200x200 grid whose upper-left cell is the picture; the right column has text.
+ page.draw_rect((100, 100, 300, 300))
+ page.draw_line((100, 200), (300, 200))
+ page.draw_line((200, 100), (200, 300))
+ page.insert_text((210, 130), "right top")
+ page.insert_text((210, 230), "right bottom")
+ page.insert_text((110, 230), "left bottom")
+ page.layout_information = [
+ {"class_name": "picture", "group_bbox": [100.0, 100.0, 200.0, 200.0]},
+ ]
+ try:
+ tables = page.find_tables(use_layout=True, union=True).tables
+ assert len(tables) == 1
+ assert (tables[0].row_count, tables[0].col_count) == (2, 2)
+ assert tables[0].extract()[0][1] == "right top"
+ finally:
+ doc.close()
+
+
+def test_find_tables_union_keeps_one_coherent_table_group_without_content_support():
+ """One Layout table group is retained; ownership is not a rejection gate."""
+ import types
+
+ doc = pymupdf.open()
+ page = doc.new_page(width=500, height=500)
+ _make_bordered_grid(
+ page,
+ (80, 80, 380, 240),
+ [["left\naxis\nlabels", "right\naxis\nlabels"]],
+ )
+ page.layout_information = [
+ {
+ "class_name": "table",
+ "group_bbox": [80.0, 80.0, 380.0, 240.0],
+ "table_grid": types.SimpleNamespace(
+ h_lines=[50.0, 100.0],
+ v_lines=[150.0],
+ ),
+ }
+ ]
+ try:
+ tables = page.find_tables(use_layout=True, union=True).tables
+ assert len(tables) == 1
+ assert (tables[0].row_count, tables[0].col_count) == (1, 2)
+ finally:
+ doc.close()