diff --git a/src/_table_headers.py b/src/_table_headers.py index 00ae860ca..0c9f785ad 100644 --- a/src/_table_headers.py +++ b/src/_table_headers.py @@ -26,8 +26,9 @@ PyMuPDF table header detection and HTML serialization (opt-in extension). Pure text-grid module (no pymupdf import): the header-region rules operate on a -row-major ``[[cell text]]`` grid, and the serializer turns a tagged placement -grid into an HTML ````. Used only by find_tables(refine=True) (via +row-major ``[[cell text]]`` grid, extend_header_leaf_labels completes that region +from a placement grid's spans, and the serializer turns a tagged placement grid +into an HTML ``
``. Used only by find_tables(refine=True) (via pymupdf.table) and Table.to_html(); never runs on the default detection path. """ from __future__ import annotations @@ -925,6 +926,89 @@ def find_header_region(rows: list[list[str]]) -> HeaderRegion: ) +# --- Leaf labels under spanning header cells: placement grid -> depth -------- +# Digits with only currency, sign, grouping, decimal, percent or date +# punctuation read as a value; a single letter makes the cell a label. +_PLAIN_NUMBER_RE = re.compile(r"^[\s$€£(),.%\-–—+0-9/:]+$") + + +def _plain_number(text: str) -> bool: + return bool(_PLAIN_NUMBER_RE.match(text)) and any(char.isdigit() for char in text) + + +def _span_slots(grid) -> tuple[list[tuple[int, int, int, int, str]], int]: + """Place a ragged span grid on its slots, as an HTML renderer does. + + Returns one ``(row0, row1, col0, col1, text)`` entry per cell (end-exclusive + slots, whitespace-collapsed text) and the grid's column count.""" + occupied: set[tuple[int, int]] = set() + slots = [] + for row0, row in enumerate(grid): + col0 = 0 + for cell in row: + while (row0, col0) in occupied: + col0 += 1 + row1, col1 = row0 + max(1, cell.rowspan), col0 + max(1, cell.colspan) + occupied.update((r, c) for r in range(row0, row1) for c in range(col0, col1)) + slots.append((row0, row1, col0, col1, collapse_cell_ws(cell.text or ""))) + col0 = col1 + return slots, max((col for _, col in occupied), default=-1) + 1 + + +def _row_shape(slots, row: int, ncols: int) -> tuple[str, ...]: + """Per column, whether ``row`` shows nothing, a plain number or text there.""" + shape = [""] * ncols + for row0, row1, col0, col1, text in slots: + if row0 <= row < row1 and text: + kind = "number" if _plain_number(text) else "text" + for col in range(col0, min(col1, ncols)): + shape[col] = kind + return tuple(shape) + + +def extend_header_leaf_labels(grid, top_header_rows: int) -> int: + """Extend the header over the row naming the columns of a spanning header cell. + + A header cell spanning more than one column but not all of them, with no + single-column label under any of its columns, leaves those columns unnamed. + The row below the header joins it when it supplies the names: a non-empty + single-column cell under every column of such a span, no plain number under + a column that is already labeled, not one full-width cell, and not the same + blank/number/text shape per column as the row after it (that repeat marks + the first record of a regular body). Repeats for nested spans; never shrinks + the header, leaves a header of 0 rows alone and never takes the last row, so + the table always keeps a body row. + + ``grid`` is a row-major placement grid duck-typed like ``render_table_html`` + input (``text`` / ``colspan`` / ``rowspan``). Only its text and spans are + read, so the result depends on nothing but the grid itself. + """ + slots, ncols = _span_slots(grid) + depth = top_header_rows + if not 0 < depth < len(grid) or ncols < 2: + return depth + while depth < len(grid) - 1: # a header never takes the whole table + header = [slot for slot in slots if slot[1] <= depth and slot[4]] + labeled = {col0 for _, _, col0, col1, _ in header if col1 - col0 == 1} + spans = [ + range(col0, col1) + for _, _, col0, col1, _ in header + if 1 < col1 - col0 < ncols and labeled.isdisjoint(range(col0, col1)) + ] + row = [(col0, col1, text) for row0, _, col0, col1, text in slots if row0 == depth and text] + if not spans or not row or (len(row) == 1 and row[0][1] - row[0][0] == ncols): + break + leaves = {col0 for col0, col1, _ in row if col1 - col0 == 1} + if not any(leaves.issuperset(span) for span in spans): + break + if any(_plain_number(text) and not labeled.isdisjoint(range(col0, col1)) for col0, col1, text in row): + break + if _row_shape(slots, depth, ncols) == _row_shape(slots, depth + 1, ncols): + break + depth += 1 + return depth + + # --- HTML serialization: tagged placement grid ->
-------------------- def collapse_cell_ws(text: str) -> str: """Whitespace-collapse a cell's text (runs of whitespace/newlines -> one space).""" diff --git a/src/_table_refine.py b/src/_table_refine.py index dc8984cac..7f2ce66fd 100644 --- a/src/_table_refine.py +++ b/src/_table_refine.py @@ -33,6 +33,8 @@ import itertools import re +from bisect import bisect_left, bisect_right +from math import isfinite import pymupdf @@ -51,6 +53,70 @@ _REFINE_LINE_GAP = 3.0 # center-y gap (points) that groups body words into lines +# --- word center index: reduce the per-cell word scan to a sorted band ------- +class _WordCenterIndex: + """Two sorted word-center axes over a read-only page word list. + + ``candidates(rect)`` returns a superset of the words whose center lies in + ``rect``, so every consumer still applies its own membership, blank-text and + claiming rules. The original word indices are preserved, including order and + duplicates, because they are what identifies a word to its owning cell. + """ + + def __init__(self, words): + xs, ys = [], [] + for i, (x0, y0, x1, y1, _text) in enumerate(words): + x, y = (x0 + x1) * 0.5, (y0 + y1) * 0.5 + if not (isfinite(x) and isfinite(y)): + raise ValueError("nonfinite word center") + xs.append((x, i)) + ys.append((y, i)) + self.xs, self.ys = sorted(xs), sorted(ys) + + def candidates(self, rect): + x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) + n = len(self.xs) + if not all(map(isfinite, (x0, y0, x1, y1))): + return range(n) # preserve the consumers' predicates for NaN / Inf + if x0 > x1 or y0 > y1: + return [] + xl, xr = bisect_left(self.xs, (x0, -1)), bisect_right(self.xs, (x1, n)) + yl, yr = bisect_left(self.ys, (y0, -1)), bisect_right(self.ys, (y1, n)) + # Scan the narrower band: fewer candidates for the same result. + entries = self.xs[xl:xr] if xr - xl <= yr - yl else self.ys[yl:yr] + return sorted(i for _center, i in entries) + + +def _refine_word_index(page, words): + """The page word list's center index, cached on the page object. + + Keyed on the word list's identity: _refine_page_words returns one stable list + per page and a fresh extraction produces a new list, so a changed text state + builds a new index. In-place edits of a cached word list are not supported. + Returns None for word lists the index cannot represent (see candidates()). + """ + cached = getattr(page, "_table_word_index_cache", None) + if cached is not None and cached[0] is words: + return cached[1] + try: + index = _WordCenterIndex(words) + except (TypeError, ValueError, OverflowError): + index = None # nonstandard inputs keep the original full scan + try: + setattr(page, "_table_word_index_cache", (words, index)) + except Exception: + pass + return index + + +def _refine_word_candidates(page, words, rect): + """(index, word) pairs that may lie in rect -- a superset of the members.""" + index = _refine_word_index(page, words) if page is not None else None + if index is None: + return enumerate(words) + return ((i, words[i]) for i in index.candidates(rect)) + + # --- word selection: center-point membership + rotated-span substitution ----- def _refine_rawdict_spans(page): """Flattened rawdict text spans carrying line-direction metadata. @@ -155,7 +221,9 @@ def _refine_words_in_rect(page, rect): x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) return [ (wx0, wy0, wx1, wy1, text) - for wx0, wy0, wx1, wy1, text in _refine_page_words(page) + for _, (wx0, wy0, wx1, wy1, text) in _refine_word_candidates( + page, _refine_page_words(page), rect + ) if _refine_word_in_rect(wx0, wy0, wx1, wy1, x0, y0, x1, y1) ] @@ -202,9 +270,11 @@ def _refine_table_rect(cells, table_bbox): def _refine_raw_shaded_rects(page, table_rect, *, min_dim): + from pymupdf.table import _get_table_drawings + out = [] page_width = float(page.rect.width) - for drawing in page.get_drawings(): + for drawing in _get_table_drawings(page): if _refine_is_white(drawing.get("fill")): continue for item in drawing.get("items", []): @@ -253,9 +323,11 @@ def _refine_cluster(values, *, tolerance): def _refine_border_lines(page, table_rect): + from pymupdf.table import _get_table_drawings + xs = set() ys = set() - for drawing in page.get_drawings(): + for drawing in _get_table_drawings(page): stroked = drawing.get("type") in ("s", "fs") for item in drawing.get("items", []): kind = item[0] diff --git a/src/_table_spans.py b/src/_table_spans.py index ff0d89d40..a15dacc26 100644 --- a/src/_table_spans.py +++ b/src/_table_spans.py @@ -36,6 +36,7 @@ from pymupdf._table_refine import ( _refine_is_vertical_or_rotated, _refine_page_words, + _refine_word_candidates, ) @@ -292,13 +293,15 @@ def _span_word_line_tuple(word): return (float(y0), float(x0), float(y1), str(text)) -def _span_select_words_in_rect(page_words, rect): +def _span_select_words_in_rect(page_words, rect, *, page=None): """(index, word) pairs whose center lies in rect, index into ``page_words``. The index is what lets resolve_spans claim each page word for exactly one - placement (an earlier cell's word is not re-claimed by a later one).""" + placement (an earlier cell's word is not re-claimed by a later one). ``page`` + only supplies the cached word-center index, which narrows the scan without + changing the result.""" selected = [] - for index, word in enumerate(page_words): + for index, word in _refine_word_candidates(page, page_words, rect): wx0, wy0, wx1, wy1, text = word if not str(text).strip(): continue @@ -321,7 +324,7 @@ def _span_claim_text_in_rect(page, rect, page_words, claimed_words): """Text of rect's words, skipping words already claimed and claiming the rest.""" selected = [ (index, word) - for index, word in _span_select_words_in_rect(page_words, rect) + for index, word in _span_select_words_in_rect(page_words, rect, page=page) if index not in claimed_words ] for index, _ in selected: @@ -486,7 +489,7 @@ def _span_reject_colspan_mismatch_merge(*, row_idx, cols, base, body_start): def _span_cell_texts_for_entries(page, entries, start, end, page_words): texts = [] for entry in entries[start : end + 1]: - words = _span_select_words_in_rect(page_words, entry) + words = _span_select_words_in_rect(page_words, entry, page=page) texts.append(_span_words_text_for_rect(page, entry, words)) return texts diff --git a/src/_table_union.py b/src/_table_union.py index d7a4d00e0..bd3a31003 100644 --- a/src/_table_union.py +++ b/src/_table_union.py @@ -28,9 +28,12 @@ TableFinder and _iou come from pymupdf.table; find_tables is imported lazily. """ +from bisect import bisect_left +from collections import namedtuple + import pymupdf -from pymupdf.table import CHARS, EDGES, Table, TableFinder, _iou +from pymupdf.table import CHARS, EDGES, Table, TableFinder, _cells_to_rows, _iou # --------------------------------------------------------------------------- @@ -53,6 +56,12 @@ _UNION_GRID_REF_SPAN_MULT_THRESHOLD = 3.0 # max horizontally-separated span groups per cell _UNION_OWNER_CONTAINMENT = 0.85 # min containment for a split candidate's owner _UNION_OWNER_AMBIGUOUS_OVERLAP = 0.25 # overlap above which an unowned candidate is suppressed +_UNION_PICTURE_CONTAINMENT = 0.8 # area fraction that puts a candidate inside a picture + +# How a candidate grid's text is laid out; see _union_grid_content_support. +_UnionContent = namedtuple( + "_UnionContent", "supported dense_rows dense_columns concrete slots" +) def _layout_table_grids(page): @@ -88,35 +97,83 @@ def _layout_table_grids(page): return grids -def _union_line_candidates(page): +def _union_line_candidates(page, *, add_lines=None, add_boxes=None): """Line-based table candidates for the union stage as ``(bbox, grid)`` pairs. Runs a nested find_tables (strategy=_UNION_STRATEGY, use_layout=False) and - keeps each detected table's bbox and row-major cell grid (Table.rows, None - for a gap), deduped by rounded bbox. Returns ``(candidates, finder)``; the - finder is reused as the returned TableFinder shell. + keeps each detected table's bbox and row-major cell grid (None for a gap), + deduped by rounded bbox. Caller-supplied virtual lines / boxes are forwarded + to that nested finder so they take part in the same union decisions as + PDF-native vector rules. Admission (see ``admit``) runs on the finder's cell + groups, before a Table is built for any of them. Returns ``(candidates, + finder)``; the finder is reused as the returned TableFinder shell. """ # Imported here, not at module top, to break the import cycle: this # module is itself imported lazily by table.find_tables (union path). from pymupdf.table import find_tables - finder = find_tables(page, strategy=_UNION_STRATEGY, use_layout=False) candidates = [] - seen = set() - for tab in (getattr(finder, "tables", None) or []): - try: - bbox = pymupdf.Rect(tab.bbox) - except (ValueError, TypeError): - continue - if bbox.is_empty: - continue - grid = [[cell for cell in row.cells] for row in (tab.rows or [])] - if not grid: - continue - key = tuple(round(value) for value in bbox) - if key in seen: - continue - seen.add(key) - candidates.append((bbox, grid)) + + def admit(live_page, groups): + """Keep the cell groups that may become tables, in detection order. + + A candidate is admitted as before: the line grid is evidence in its own + right and the layout model can miss a table entirely. The one exception is + a candidate sitting inside a region the layout model called a picture, + which has to show table-shaped text instead. That cheap geometric test + therefore gates the character scan, and the scan itself is built once for + the whole call. + """ + kept = [] + seen = set() + points = None + for cells in groups: + bbox = pymupdf.Rect( + min(cell[0] for cell in cells), + min(cell[1] for cell in cells), + max(cell[2] for cell in cells), + max(cell[3] for cell in cells), + ) + if bbox.is_empty: + continue + grid = [row.cells for row in _cells_to_rows(cells)] + if not grid: + continue + key = tuple(round(value) for value in bbox) + if key in seen: + continue + if _union_candidate_inside_picture(live_page, bbox): + if points is None: + points = _union_char_midpoints() + content = _union_grid_content_support(grid, points) + # A chart labels one row and one column -- its axes -- and its + # partial gridlines leave most slots without a cell. A table has + # a header row and a label column that are each at least half + # full, and cell coverage that is mostly rectangular. + if not ( + content.supported + and content.dense_rows >= 2 + and content.dense_columns >= 2 + and content.concrete * 2 >= content.slots + ): + continue + # Dedup only admitted candidates: a rejected one must not shadow a + # later, differently gridded candidate with the same rounded bbox. + seen.add(key) + kept.append(cells) + candidates.append((bbox, grid)) + return kept + + finder = find_tables( + page, + strategy=_UNION_STRATEGY, + use_layout=False, + add_lines=add_lines, + add_boxes=add_boxes, + _cell_group_filter=admit, + ) + if finder is None: + # Nested detection failure must not leak partially admitted candidates. + candidates.clear() return candidates, finder @@ -165,6 +222,161 @@ def _union_find_owner(candidate_bbox, existing_bboxes): return best_owner, ambiguous +def _union_char_midpoints(): + """The page's non-blank character midpoints as ``(v_mid, h_mid)``, y-sorted. + + One pass over CHARS replaces the per-cell character scan of the admission + tests, which read midpoints only. Sorting by the vertical midpoint lets each + grid row take its own characters as one slice. + """ + points = [] + for char in CHARS: + if not str(char.get("text") or "").strip(): + continue + try: + h_mid = (float(char["x0"]) + float(char["x1"])) / 2.0 + v_mid = (float(char["top"]) + float(char["bottom"])) / 2.0 + except (KeyError, TypeError, ValueError): + continue + points.append((v_mid, h_mid)) + points.sort() + return points + + +def _union_grid_content_support(grid, points): + """How a candidate line grid's text is distributed, in one scan. + + ``supported`` means at least two rows with text in at least two columns and + at least two columns with text in at least two rows -- two being the smallest + non-trivial count in each dimension, not a fitted one. A populated cell counts + for every column its x-range covers, so a grid whose missing horizontal rules + leave a spanning header above one record row is supported just like the same + table with all its rules. + + ``dense_rows`` / ``dense_columns`` count the rows and columns whose own + concrete cells are at least half populated, and ``concrete`` / ``slots`` are + the concrete cells and the grid's total slots. Those describe the *shape* of + the text rather than its amount, which is what separates a chart from a + sparse table (see _union_candidate_inside_picture). + + Character membership follows Table.extract()'s half-open cell rule. ``None`` + grid slots are span/gap placeholders, not empty text cells. The decision is + candidate-local: page area, layout ownership, file identity and virtual-line + provenance are not inputs. + """ + columns = sorted({float(cell[0]) for row in grid for cell in row if cell is not None}) + width = max((len(row) for row in grid), default=0) + column_populated = [0] * width + column_concrete = [0] * width + row_columns = [] + row_counts = [] + concrete = 0 + slots = 0 + for row in grid: + slots += len(row) + covered = set() + row_populated = 0 + rects = [(index, pymupdf.Rect(cell)) for index, cell in enumerate(row) if cell is not None] + rects = [(index, rect) for index, rect in rects if not rect.is_empty] + concrete += len(rects) + if rects: + band = _union_row_points(rects, points) + for index, rect in rects: + column_concrete[index] += 1 + if not _union_rect_has_point(rect, band): + continue + row_populated += 1 + column_populated[index] += 1 + spanned = [ + column + for column, x in enumerate(columns) + if float(rect.x0) <= x < float(rect.x1) + ] + covered.update(spanned or (index,)) + row_columns.append(covered) + row_counts.append((row_populated, len(rects))) + supported = sum(len(covered) >= 2 for covered in row_columns) >= 2 and sum( + sum(column in covered for covered in row_columns) >= 2 + for column in range(len(columns)) + ) >= 2 + return _UnionContent( + supported=supported, + dense_rows=sum(total > 0 and filled * 2 >= total for filled, total in row_counts), + dense_columns=sum( + total > 0 and filled * 2 >= total + for filled, total in zip(column_populated, column_concrete) + ), + concrete=concrete, + slots=slots, + ) + + +def _union_row_points(rects, points): + """The y-sorted midpoints falling in one grid row's vertical band.""" + y0 = min(float(rect.y0) for _index, rect in rects) + y1 = max(float(rect.y1) for _index, rect in rects) + return points[bisect_left(points, (y0,)) : bisect_left(points, (y1,))] + + +def _union_rect_has_point(rect, band): + """Whether any midpoint of the row band lies in rect (half-open, as extract).""" + x0, y0, x1, y1 = float(rect.x0), float(rect.y0), float(rect.x1), float(rect.y1) + for v_mid, h_mid in band: + if x0 <= h_mid < x1 and y0 <= v_mid < y1: + return True + return False + + +def _union_layout_groups(page): + """The page's layout groups as ``(class_name, rect)``, skipping unusable ones.""" + groups = [] + for group in (page.layout_information or []): + if not isinstance(group, dict): + continue + group_bbox = group.get("group_bbox") + if not group_bbox: + continue + try: + rect = pymupdf.Rect(group_bbox[:4]) + except (TypeError, ValueError): + continue + if rect.is_empty: + continue + groups.append((group.get("class_name"), rect)) + return groups + + +def _union_candidate_inside_picture(page, candidate_bbox): + """Whether a candidate sits inside a layout picture group without covering it. + + The layout model classified that region as a picture, so a line grid found + almost entirely inside one is as likely to be a chart's gridlines as a table, + and overriding the model's verdict needs table-shaped text. The shape, not the + amount, is what tells them apart: a chart's text sits along its two axes, so + exactly one row and one column are well filled and its partial gridlines leave + most of the grid without a cell at all, while a table has a header row and a + label column that are each at least half full over mostly complete cell + coverage -- which a sparse matrix of scattered marks still satisfies. See the + admission rule in _union_line_candidates. + + A candidate larger than the picture contains it instead -- a bordered table + with an image in one of its cells -- and keeps the ordinary rule. + """ + candidate_area = _union_rect_area(candidate_bbox) + if candidate_area <= 0: + return False + threshold = _UNION_PICTURE_CONTAINMENT * candidate_area + for class_name, rect in _union_layout_groups(page): + if class_name != "picture": + continue + if _union_intersection_area(candidate_bbox, rect) < threshold: + continue + if candidate_area > _union_rect_area(rect): + continue # the candidate is the larger region: it holds the picture + return True + return False + + def _union_text_span_rects(page): """Non-empty page text-span rects, cached on the page. @@ -325,18 +537,21 @@ def _union_replace_append(existing, candidates, *, page, grid_ref, grid_ref_iou, return entries -def _find_tables_union(page): +def _find_tables_union(page, *, add_lines=None, add_boxes=None): """Detect a page's tables by fusing layout grids with line-based candidates. Ensures the raw layout (computed only when page.layout_information is None, - like the official use_layout path), reads primary grids, detects candidates, - applies grid-ref / split / append, and returns a TableFinder whose .tables - carry the fused grids in contractual order (grid-ref tables keep their - explicit layout bbox).""" + like the official use_layout path), reads primary grids, detects candidates + (forwarding the caller's virtual lines / boxes to the nested finder), applies + grid-ref / split / append, and returns a TableFinder whose .tables carry the + fused grids in contractual order (grid-ref tables keep their explicit layout + bbox).""" if page.layout_information is None: page.get_layout(return_raw=True) primaries = _layout_table_grids(page) - candidates, finder = _union_line_candidates(page) + candidates, finder = _union_line_candidates( + page, add_lines=add_lines, add_boxes=add_boxes + ) entries = _union_replace_append( primaries, candidates, diff --git a/src/table.py b/src/table.py index 7b8e2c19c..e86e174c5 100644 --- a/src/table.py +++ b/src/table.py @@ -104,6 +104,8 @@ ) # Header semantics and HTML serialization live in the _table_headers sibling. from pymupdf._table_headers import ( + HeaderRegion, + extend_header_leaf_labels, find_header_region, collapse_cell_ws, render_table_html, @@ -123,6 +125,44 @@ _CHARS_VAR = ContextVar("pymupdf_table_chars", default=None) +def _get_table_drawings(page, *, native=False): + """The page's vector drawings, shared within one find_tables() call. + + One call reads them up to ``1 + 2T`` times -- once to build edges, and once + per refined table for shaded rows and for border lines -- and each read walks + the whole content stream. find_tables() therefore puts an empty cache on the + page for the duration of the call and removes it again, so nothing outlives + the call: a page whose content stream is rewritten between calls cannot be + served a stale extraction. Calling the public refine helpers directly, with + no find_tables() around them, extracts afresh exactly as before. + + ``native=True`` returns per-path copies of the mutated members: make_edges + extends path rects, normalizes rectangle items and appends a closing line, so + the other consumers must not see its edits. Point and Quad values stay shared + because make_edges only reads them. + """ + cache = getattr(page, "_table_drawings_cache", None) + if cache is None: + paths = page.get_drawings() + else: + if not cache: + cache.append(page.get_drawings()) + paths = cache[0] + if not native: + return paths + return [ + dict( + path, + rect=pymupdf.Rect(path["rect"]), + items=[ + (item[0], pymupdf.Rect(item[1]), *item[2:]) if item[0] == "re" else item + for item in path["items"] + ], + ) + for path in paths + ] + + class _TableStateList: """List-like proxy for per-call table extraction state.""" @@ -1607,6 +1647,25 @@ def __init__(self, bbox, cells, names, above): self.external = above +def _cells_to_rows(cells): + """Canonical row order and None gap slots for a flat cell list. + + Each row holds one slot per distinct cell x0 on the table, so a gap and a + span both appear as None. Shared by Table.rows and the union stage, which + needs the same grid before a Table exists.""" + _sorted = sorted(cells, key=itemgetter(1, 0)) + xs = list(sorted(set(map(itemgetter(0), cells)))) + rows = [] + for y, row_cells in itertools.groupby(_sorted, itemgetter(1)): + xdict = {cell[0]: cell for cell in row_cells} + row = TableRow([xdict.get(x) for x in xs]) + rows.append(row) + return rows + + +_NO_HEADER_YET = object() # Table.header has not been computed yet + + class Table: def __init__(self, page, cells, bbox=None): self.page = page @@ -1617,7 +1676,14 @@ def __init__(self, page, cells, bbox=None): # cells. Set only for a union grid-ref table, whose reported region # (its layout box) is decoupled from its replacement cell grid. self._bbox = bbox - self.header = self._get_header() # PyMuPDF extension + self._header = _NO_HEADER_YET + # The header rules read the page through the two global text-extraction + # toggles that were in force when this table was detected, so a header + # computed on first access has to put them back. Both are plain getters. + self._header_flags = ( + bool(pymupdf.TOOLS.set_small_glyph_heights()), + bool(pymupdf.TOOLS.unset_quad_corrections()), + ) # Filled by find_tables(refine=True): placements is a row-major grid of # tagged SpanCell colspan/rowspan placements (None otherwise); header_rows # is the leading header-row count, section_rows the section-label rows. @@ -1637,16 +1703,36 @@ def bbox(self): max(map(itemgetter(3), c)), ) + @property + def header(self): # PyMuPDF extension + """The identified table header, computed on first access. + + Identifying it renders the top row twice, so tables whose header is never + read -- every table serialized through placements, and every candidate a + later stage discards -- must not pay for it at construction time. The + header rules re-read the page, so the two global text-extraction toggles + of the detecting call are restored for the duration (see __init__): the + result is then the one an eager computation would have produced.""" + if self._header is _NO_HEADER_YET: + small, quads = self._header_flags + old_small = bool(pymupdf.TOOLS.set_small_glyph_heights()) + old_quads = bool(pymupdf.TOOLS.unset_quad_corrections()) + pymupdf.TOOLS.set_small_glyph_heights(small) + pymupdf.TOOLS.unset_quad_corrections(quads) + try: + self._header = self._get_header() + finally: + pymupdf.TOOLS.set_small_glyph_heights(old_small) + pymupdf.TOOLS.unset_quad_corrections(old_quads) + return self._header + + @header.setter + def header(self, value): + self._header = value + @property def rows(self) -> list: - _sorted = sorted(self.cells, key=itemgetter(1, 0)) - xs = list(sorted(set(map(itemgetter(0), self.cells)))) - rows = [] - for y, row_cells in itertools.groupby(_sorted, itemgetter(1)): - xdict = {cell[0]: cell for cell in row_cells} - row = TableRow([xdict.get(x) for x in xs]) - rows.append(row) - return rows + return _cells_to_rows(self.cells) @property def row_count(self) -> int: # PyMuPDF extension @@ -1869,9 +1955,12 @@ def row_has_bold(bbox): Returns True if any spans are bold else False. """ + # This table's own character snapshot, so a lazily computed header + # reads the same characters an eager one would have. + chars = self._chars if self._chars is not None else CHARS return any( c["bold"] - for c in CHARS + for c in chars if rect_in_rect((c["x0"], c["y0"], c["x1"], c["y1"]), bbox) ) @@ -2153,7 +2242,7 @@ class TableFinder: https://github.com/tabulapdf/tabula-extractor/issues/16 """ - def __init__(self, page, settings=None): + def __init__(self, page, settings=None, *, cell_group_filter=None): self.page = weakref.proxy(page) self.textpage = None self.settings = TableSettings.resolve(settings) @@ -2164,10 +2253,12 @@ def __init__(self, page, settings=None): self.settings.intersection_y_tolerance, ) self.cells = intersections_to_cells(self.intersections) - self.tables = [ - Table(self.page, cell_group) - for cell_group in cells_to_tables(self.page, self.cells) - ] + groups = cells_to_tables(self.page, self.cells) + if cell_group_filter is not None: + # Internal hook: the union stage decides admission on the cell groups, + # so a rejected candidate never becomes a Table at all. + groups = cell_group_filter(self.page, groups) + self.tables = [Table(self.page, group) for group in groups] def get_edges(self) -> list: settings = self.settings @@ -2364,6 +2455,26 @@ def make_chars(page, clip=None): # We are ignoring Bézier curves completely and are converting everything # else to lines. # ------------------------------------------------------------------------ +def _is_line_like(rect, min_length): + """Whether a rectangle is thin enough to be a simulated line.""" + return (rect.width <= min_length and rect.width < rect.height) or ( + rect.height <= min_length and rect.height < rect.width + ) + + +def _is_batched_rule(rect, min_length): + """Whether a rectangle inside a large fill path is a simulated line. + + A path's aggregate bbox is no evidence about its individual items, so a thin + rectangle batched into a large fill can still be a grid rule. A thin *and + short* one cannot: dot screens and vector-drawn glyph fragments consist of + rectangles that are small in both directions, and a few hundred of them snap + and join into one long spurious edge. A rule reaches at least the minimum + edge length along its length, which is the same bound filter_edges() applies + to the finished edges.""" + return _is_line_like(rect, min_length) and max(rect.width, rect.height) > min_length + + def make_edges(page, clip=None, tset=None, paths=None, add_lines=None, add_boxes=None): edges = EDGES._list() # bind once: avoid per-append proxy overhead below snap_x = tset.snap_x_tolerance @@ -2420,24 +2531,42 @@ def are_neighbors(r1, r2): def clean_graphics(npaths=None): """Detect and join rectangles of "connected" vector graphics.""" if npaths is None: - allpaths = page.get_drawings() + allpaths = _get_table_drawings(page, native=True) else: # accept passed-in vector graphics allpaths = npaths[:] # paths relevant for table detection paths = [] + bbox_paths = [] # paths that may contribute an enveloping bbox for p in allpaths: - # If only looking at lines, we ignore fill-only paths, - # except simulated lines (i.e. small width or height). + # If only looking at lines, we ignore fill-only paths, except + # simulated lines (i.e. small width or height) -- including the ones + # batched inside a large path. Some producers put hundreds of thin + # grid rules into a single fill path, whose aggregate bbox is large + # although every relevant item in it is a simulated line. A path's + # bbox is not evidence about its individual items, so drop the path + # and keep those items (see _is_batched_rule). They contribute no + # enveloping bbox: bbox_paths holds only the paths whose own rect is + # real graphics. if ( lines_strict and p["type"] == "f" and p["rect"].width > snap_x and p["rect"].height > snap_y ): + line_items = [ + item + for item in p["items"] + if item[0] == "re" and _is_batched_rule(item[1].normalize(), min_length) + ] + if line_items: + line_path = p.copy() + line_path["items"] = line_items + paths.append(line_path) continue paths.append(p) + bbox_paths.append(p) # start with all vector graphics rectangles - prects = sorted(set([p["rect"] for p in paths]), key=lambda r: (r.y1, r.x0)) + prects = sorted(set([p["rect"] for p in bbox_paths]), key=lambda r: (r.y1, r.x0)) new_rects = [] # the final list of joined rectangles # ---------------------------------------------------------------- # Strategy: Join rectangles that "almost touch" each other. @@ -2548,23 +2677,15 @@ def make_line(p, p1, p2, clip): # the ones that simulate a line rect = i[1].normalize() # normalize the rectangle - if ( - rect.width <= min_length and rect.width < rect.height - ): # simulates a vertical line - x = abs(rect.x1 + rect.x0) / 2 # take middle value for x - p1 = pymupdf.Point(x, rect.y0) - p2 = pymupdf.Point(x, rect.y1) - line_dict = make_line(p, p1, p2, clip) - if line_dict: - edges.append(line_to_edge(line_dict)) - continue - - if ( - rect.height <= min_length and rect.height < rect.width - ): # simulates a horizontal line - y = abs(rect.y1 + rect.y0) / 2 # take middle value for y - p1 = pymupdf.Point(rect.x0, y) - p2 = pymupdf.Point(rect.x1, y) + if _is_line_like(rect, min_length): # simulates a single line + if rect.width < rect.height: # a vertical one + x = abs(rect.x1 + rect.x0) / 2 # take middle value for x + p1 = pymupdf.Point(x, rect.y0) + p2 = pymupdf.Point(x, rect.y1) + else: # a horizontal one + y = abs(rect.y1 + rect.y0) / 2 # take middle value for y + p1 = pymupdf.Point(rect.x0, y) + p2 = pymupdf.Point(rect.x1, y) line_dict = make_line(p, p1, p2, clip) if line_dict: edges.append(line_to_edge(line_dict)) @@ -2754,7 +2875,7 @@ def _refine_flat_placement_grid(page, cells, col_count): rect = pymupdf.Rect(cell) line_words = [ _span_word_line_tuple(word) - for _, word in _span_select_words_in_rect(page_words, rect) + for _, word in _span_select_words_in_rect(page_words, rect, page=page) ] out.append( SpanCell( @@ -2797,29 +2918,161 @@ def _refine_tag_grid(grid, top_header_rows): return grid +def _refine_model_grid(page, cells): + """The merge-preserved model grid plus the header counts derived from it. + + Returns ``(grid, header_rows, body_start)``. ``grid`` is None when span + resolution or the header rules fail. ``header_rows`` is the header finder's + own count, where 0 means it found no header row at all; ``body_start`` is + that count clamped to [1, rows], which is what the row splitter needs. One + resolve_spans pass serves both the header/body boundary and the repeated- + header split check, so the split costs no extra pass. + """ + try: + grid = _refine_placement_or_flat_grid(page, cells) + region = find_header_region(_refine_placements_text_grid(grid)) + except Exception: + return None, 0, 1 + header_rows = int(region.top_header_rows) + return grid, header_rows, (max(1, min(header_rows, len(cells))) if cells else 0) + + def _refine_body_start_row(page, cells): """Header/body boundary: resolve the merge-preserved placement grid once and ask the header finder how many leading rows are header, clamped to [1, rows].""" - try: - model_grid = _refine_placement_or_flat_grid(page, cells) - region = find_header_region(_refine_placements_text_grid(model_grid)) - except Exception: - return 1 - raw = region.top_header_rows - return max(1, min(int(raw), len(cells))) if cells else 0 + grid, _header_rows, body_start = _refine_model_grid(page, cells) + return 1 if grid is None else body_start + + +def _refine_repeated_leading_header_cuts(rows, header_rows): + """Rows at which an exactly repeated leading header splits a stacked table. + + Several tables printed one under another with no gap are detected as one + grid, and the giveaway is that their shared header row recurs verbatim. The + signature is the first row that carries at least two non-empty cells in a + grid at least two columns wide and is not purely numeric or punctuation; it + must also be a leading row -- inside the header region the header rules found, + or within the first three rows, because that finder is conservative and a + title row and a units row often precede the real header. A repeat further down + is data, not a header. Repeats are matched on case- and whitespace-normalized + text only: no fuzzy matching. + + A header signature also carries the grid's column structure, so it must hold + at least half as many placements as the widest row: a row of a few spanning + cells is a group header -- a sub-header such as "From- / To-" under an + eight-column header, which recurs inside one table -- and not the header. + + A repeat is accepted as a cut only if the result really looks like a stack of + tables rather than a table that happens to repeat a row: every segment must + hold at least one row below its own signature row. That single structural + invariant rejects both adjacent duplicate header rows (one header, not a cut) + and a trailing repeat with no records under it, and it needs no row-count + constant. Each cut row opens its segment by construction; the first segment + may carry a title row above the signature. Returns the cut row indices, or () + for no split. + """ + normalized = [ + tuple(collapse_cell_ws(text).casefold() for text in row) for row in rows + ] + if not normalized: + return () + width = max(len(row) for row in normalized) + if width < 2: + return () + # Every candidate signature is a leading row, so the scan is bounded; the + # first such row that actually recurs is the one that splits. + for index in range(min(max(int(header_rows), 3), len(normalized))): + signature = normalized[index] + if len(signature) * 2 < width: + continue # a few spanning cells: a group header, not the header + nonempty = [text for text in signature if text] + if len(nonempty) < 2: + continue + if not any(char.isalpha() for text in nonempty for char in text): + continue # a repeated all-numeric row is data, not a header + repeats = [ + row_index + for row_index in range(index + 1, len(normalized)) + if normalized[row_index] == signature + ] + if not repeats: + continue + signature_rows = [index, *repeats] + segment_ends = [*repeats, len(normalized)] + if all(end - start >= 2 for start, end in zip(signature_rows, segment_ends)): + return tuple(repeats) + return () + return () def _refine_build_placements(page, working, body_start): """Resolve the final placement grid (strict colspan, header boundary known), - run header rules on its own text grid, tag cells -> (tagged grid, region).""" + run header rules on its own text grid, complete the header under spanning + header cells from the grid's spans, tag cells -> (tagged grid, region).""" grid = _refine_placement_or_flat_grid( page, working, strict_colspan=True, header_row_count=body_start ) region = find_header_region(_refine_placements_text_grid(grid)) + # The text-grid rules cannot see spans. A row added here holds at least two + # labels, so it is never a section row and the section rows stand. + depth = extend_header_leaf_labels(grid, region.top_header_rows) + region = HeaderRegion(depth, region.section_header_rows) tagged = _refine_tag_grid(grid, region.top_header_rows) return tagged, region +def _refine_grid_tables(page, tab): + """Refine one detected table's grid and yield the resulting tagged tables. + + Order: structural split (shaded rows + under-segmented columns), the model + grid that carries both the header/body boundary and the repeated-header split + decision, then per segment the body-row split, the final merged-cell + placement grid and the td/th tagging. A grid whose leading header row recurs + verbatim is several tables printed under one another and yields one table per + segment; that is the ordinary case of one segment, which yields exactly the + table the unsplit path produced. + """ + grid = _refine_cells_to_grid(tab.cells) + # The reported bbox (a union grid-ref table's layout box, else the cells' + # union) bounds the shaded-rectangle search. + working = refine_grid_structure(page, grid, table_bbox=tab.bbox) + model_grid, header_rows, body_start = _refine_model_grid(page, working) + if model_grid is None: + body_start = 1 + cuts = () + else: + # The model grid has one row per row of `working`, so its cut indices + # index `working` directly. + cuts = _refine_repeated_leading_header_cuts( + _refine_placements_text_grid(model_grid), header_rows + ) + if cuts: + boundaries = (0, *cuts, len(working)) + segments = [working[start:end] for start, end in zip(boundaries, boundaries[1:])] + else: + segments = [working] + was_split = len(segments) > 1 + for segment in segments: + # A split segment has its own header region; without a split this is the + # boundary the model grid already produced. + start = _refine_body_start_row(page, segment) if was_split else body_start + segment = refine_grid_rows(page, segment, header_row_count=start) + flat = _refine_grid_to_cells(segment) + if not flat and was_split: + continue # an all-gap segment produces no table of its own + # Preserve an explicit reported-bbox override (union grid-ref tables): the + # refined grid must not change the reported region. A split does change + # it, so each segment reports its own cells. + new_tab = Table(page, flat, bbox=None if was_split else tab._bbox) if flat else tab + # Build the tagged model on `segment` directly, not on the re-gridded + # new_tab.cells, so placements match the refined grid. + placements, region = _refine_build_placements(page, segment, start) + new_tab.placements = placements + new_tab.header_rows = region.top_header_rows + new_tab.section_rows = region.section_header_rows + yield new_tab + + def find_tables( page, clip=None, @@ -2849,6 +3102,7 @@ def find_tables( use_layout: bool = True, # gate line-based tables by layout table boxes union: bool = False, # opt-in: fuse layout grids with line-based candidates refine: bool = False, # opt-in: refine each detected table's cell grid + _cell_group_filter=None, # internal: union admission, before Table construction ): """Detect and extract tables on a page. @@ -2874,7 +3128,9 @@ def find_tables( the default path), the header meta as Table.header_rows/section_rows, and Table.to_html() serializes it. Off by default. A grid whose column count the span resolution cannot preserve falls back to a flat one-cell-per-slot - placement grid. + placement grid. A refined grid whose leading header row recurs verbatim is + split into one table per recurrence: that is several tables printed under one + another which the line grid detected as one. """ pymupdf._warn_layout_once() _CHARS_VAR.set([]) @@ -2886,6 +3142,16 @@ def find_tables( else: old_xref, old_rot, old_mediabox = None, None, None + # Share one drawings extraction for the duration of this call (see + # _get_table_drawings). A nested call -- the union path runs one -- finds the + # cache already there and must neither replace nor remove it. + owns_drawings = getattr(page, "_table_drawings_cache", None) is None + if owns_drawings: + try: + page._table_drawings_cache = [] + except Exception: + owns_drawings = False + if snap_x_tolerance is None: snap_x_tolerance = UNSET if snap_y_tolerance is None: @@ -2934,7 +3200,7 @@ def find_tables( # the nested finder. Imported here, not at module top, to avoid an # import cycle: _table_union imports find_tables back from this module. from pymupdf._table_union import _find_tables_union - tbf = _find_tables_union(page) + tbf = _find_tables_union(page, add_lines=add_lines, add_boxes=add_boxes) TEXTPAGE = tbf.textpage else: boxes = [] @@ -2965,7 +3231,7 @@ def find_tables( add_boxes=add_boxes, ) # create lines and curves - tbf = TableFinder(page, settings=tset) + tbf = TableFinder(page, settings=tset, cell_group_filter=_cell_group_filter) tbf.textpage = TEXTPAGE # store textpage for later use if boxes: # only keep Finder tables that match a layout box @@ -2996,28 +3262,25 @@ def find_tables( # as .header_rows/.section_rows. refined_tables = [] for tab in tbf.tables: - grid = _refine_cells_to_grid(tab.cells) - # The reported bbox (a union grid-ref table's layout box, else - # the cells' union) bounds the shaded-rectangle search. - working = refine_grid_structure(page, grid, table_bbox=tab.bbox) - body_start = _refine_body_start_row(page, working) - working = refine_grid_rows(page, working, header_row_count=body_start) - flat = _refine_grid_to_cells(working) - # Preserve an explicit reported-bbox override (union grid-ref - # tables): the refined grid must not change the reported region. - new_tab = Table(page, flat, bbox=tab._bbox) if flat else tab - # Build the tagged model on `working` directly, not on the - # re-gridded new_tab.cells, so placements match the refined grid. - placements, region = _refine_build_placements(page, working, body_start) - new_tab.placements = placements - new_tab.header_rows = region.top_header_rows - new_tab.section_rows = region.section_header_rows - refined_tables.append(new_tab) + refined_tables.extend(_refine_grid_tables(page, tab)) tbf.tables = refined_tables + if old_xref is not None: + # The header rules read the page itself -- a pixmap of the top row + # and the text above the table -- in the detected cells' coordinate + # system. On a derotated page that system disappears when the finally + # block restores the rotation, so resolve the header now. Pages that + # were not rotated keep the lazy header. + for table in tbf.tables: + table._header = table._get_header() except Exception as e: pymupdf.message("find_tables: exception occurred: %s" % str(e)) return None finally: + if owns_drawings: # before the rotation reset rebinds `page` + try: + del page._table_drawings_cache + except Exception: + pass pymupdf.TOOLS.set_small_glyph_heights(old_small) if old_xref is not None: page = page_rotation_reset(page, old_xref, old_rot, old_mediabox) diff --git a/tests/test_tables.py b/tests/test_tables.py index 9fbeec387..c3623a989 100644 --- a/tests/test_tables.py +++ b/tests/test_tables.py @@ -227,6 +227,115 @@ def test_strict_lines(): assert tab2.col_count < tab1.col_count +def test_strict_lines_accepts_thin_rects_batched_in_large_fill_path(): + """Line-like rect items survive a large fill-only path's bbox filter.""" + doc = pymupdf.open() + page = doc.new_page(width=240, height=180) + shape = page.new_shape() + x_values = (40, 120, 200) + y_values = (30, 90, 150) + for x in x_values: + shape.draw_rect(pymupdf.Rect(x - 0.4, y_values[0], x + 0.4, y_values[-1])) + for y in y_values: + shape.draw_rect(pymupdf.Rect(x_values[0], y - 0.4, x_values[-1], y + 0.4)) + shape.finish(color=None, fill=(0, 0, 0)) + shape.commit() + for row in range(2): + for col in range(2): + page.insert_text( + (x_values[col] + 10, y_values[row] + 25), + f"r{row}c{col}", + ) + + paths = page.get_drawings() + assert len(paths) == 1 + assert paths[0]["type"] == "f" + assert paths[0]["rect"].width > 3 and paths[0]["rect"].height > 3 + + tables = page.find_tables(strategy="lines_strict", use_layout=False).tables + assert len(tables) == 1 + assert tables[0].row_count == 2 + assert tables[0].col_count == 2 + doc.close() + + +def test_strict_lines_ignores_a_dot_screen_batched_in_a_large_fill_path(): + """Short thin rects in a large fill path are a dot screen, not grid rules. + + This page batches 196 rects of 1.1 x 0.9 pt into one fill path: a dotted + shading, not a grid. Each is thin enough to look like a simulated line, and a + few hundred of them at one y snap and join into a single long edge, which + turns an empty region into a table. Only rects long enough to be a rule are + kept (see _is_batched_rule). + """ + filename = os.path.join(scriptdir, "resources", "test_3594.pdf") + doc = pymupdf.open(filename) + page = doc[14] + try: + paths = [ + p + for p in page.get_drawings() + if p["type"] == "f" and p["rect"].width > 3 and p["rect"].height > 3 + ] + assert any(len(p["items"]) > 100 for p in paths) + assert page.find_tables(strategy="lines_strict", use_layout=False).tables == [] + finally: + doc.close() + + +def test_get_table_drawings_is_shared_only_inside_one_call(): + """The drawings cache lives for one find_tables() call; copies stay isolated.""" + from pymupdf.table import _get_table_drawings + + doc = pymupdf.open() + page = doc.new_page(width=200, height=200) + page.draw_rect(pymupdf.Rect(20, 20, 120, 60)) + try: + # Without the cache every consumer extracts afresh, exactly as before. + assert _get_table_drawings(page) is not _get_table_drawings(page) + + page._table_drawings_cache = [] # what find_tables() installs + first = _get_table_drawings(page) + assert first is _get_table_drawings(page) + assert first == page.get_drawings() + rect = pymupdf.Rect(first[0]["rect"]) + item_count = len(first[0]["items"]) + + native = _get_table_drawings(page, native=True) + assert native is not first + native[0]["rect"] |= pymupdf.Point(180, 180) + native[0]["items"].append(("l", pymupdf.Point(0, 0), pymupdf.Point(1, 1))) + # The edge builder's in-place edits must not reach the other consumers. + cached = _get_table_drawings(page) + assert pymupdf.Rect(cached[0]["rect"]) == rect + assert len(cached[0]["items"]) == item_count + assert pymupdf.Rect(native[0]["rect"]) != rect # the copy really changed + + del page._table_drawings_cache + page.find_tables(use_layout=False) # must not leave the cache behind + assert not hasattr(page, "_table_drawings_cache") + finally: + doc.close() + + +def test_table_header_is_computed_on_first_access(): + """Table.header is lazy but keeps its value, so to_markdown still works.""" + doc = pymupdf.open(filename) + page = doc[0] + try: + tab = page.find_tables().tables[0] + assert tab._header is pymupdf.table._NO_HEADER_YET + header = tab.header + assert header is tab.header # computed once + assert header.external is False + assert header.names == tab.extract()[0] + assert tab.to_markdown().startswith("|") + tab.header = None # assignable, as before + assert tab.header is None + finally: + doc.close() + + def test_add_lines(): """Test new parameter add_lines for table recognition.""" if platform.python_implementation() == 'GraalVM': @@ -676,6 +785,123 @@ def test_find_tables_refine_splits_rows_default_unchanged(): doc.close() +def test_refine_repeated_leading_header_cuts_exact_signature(): + """A leading header row that recurs verbatim splits stacked tables.""" + rows = [ + ["First table", "", ""], + ["Effective Date", "BI", "PD"], + ["2024-01-01", "1", "2"], + ["2024-02-01", "3", "4"], + ["2024-03-01", "5", "6"], + ["Second table", "", ""], + [" effective date ", "bi", "PD"], + ["2023-01-01", "7", "8"], + ["2023-02-01", "9", "10"], + ["2023-03-01", "11", "12"], + ] + cuts = pymupdf.table._refine_repeated_leading_header_cuts + assert cuts(rows, 2) == (6,) + # The header finder is conservative: a title row above the header can leave it + # reporting one header row, and the signature is still a leading row. + assert cuts(rows, 1) == (6,) + assert cuts(rows, 0) == (6,) + # A two-column stack splits too: the rule reads the grid's own width. + assert cuts([row[:2] for row in rows], 2) == (6,) + + +def test_refine_repeated_leading_header_cuts_rejects_body_and_short_segments(): + """Repeated body data and adjacent duplicate headers leave the table intact.""" + repeated_body = [ + ["Report", "", ""], + ["Name", "Value", "Note"], + ["a", "1", "x"], + ["same", "record", "value"], + ["b", "2", "y"], + ["c", "3", "z"], + ["d", "4", "w"], + ["same", "record", "value"], + ["e", "5", "q"], + ["f", "6", "r"], + ] + adjacent_duplicate = [ + ["Name", "Value", "Note"], + ["a", "1", "x"], + ["b", "2", "y"], + ["c", "3", "z"], + ["Name", "Value", "Note"], + ["Name", "Value", "Note"], + ["d", "4", "w"], + ["e", "5", "q"], + ] + # A two-row header, as on IRS form 940-B: the narrow sub-header under the + # eight-column header recurs inside the one table. + spanning_subheader = [ + ["State", "Reporting No.", "Payroll", "Rate period", "Rate", "Paid", "Due", "Total"], + ["From-", "To-"], + ["", "", "", "", "", "", "", ""], + ["State agency"], + ["Fax", "Mail to", "Other"], + ["State", "Rate", "Taxable", "Rate", "Paid", "Due", "Total"], + ["From-", "To-"], + ["", "", "", "", "", "", "", ""], + ] + cuts = pymupdf.table._refine_repeated_leading_header_cuts + # The repeated row is body data: it is not a leading row. + assert cuts(repeated_body, 2) == () + # Two spanning cells do not carry an eight-column header's structure. + assert cuts(spanning_subheader, 1) == () + # A repeated all-numeric leading row is data too. + assert cuts([["1", "2"], ["a", "b"], ["1", "2"], ["c", "d"]], 1) == () + # Two header rows in a row are one header, not a cut. + assert cuts(adjacent_duplicate, 1) == () + # A trailing repeat with no record under it is not a segment. + assert cuts([["H", "H2"], ["a", "b"], ["H", "H2"]], 1) == () + + +def test_find_tables_refine_splits_a_repeated_leading_header(): + """Two tables printed under one another are one grid; refine=True splits it. + + Nothing about a repeated header is union-specific, so the split applies + whenever refine=True. The default path still reports the single grid. + + *** PyMuPDF extension. *** + """ + texts = [ + ["Date", "BI", "PD"], + ["2024-01", "1", "2"], + ["2024-02", "3", "4"], + ["Date", "BI", "PD"], + ["2023-01", "7", "8"], + ["2023-02", "9", "10"], + ] + doc = pymupdf.open() + page = doc.new_page(width=400, height=400) + x_values = (60, 160, 230, 300) + y0, row_height = 60, 22 + for row in range(len(texts) + 1): + y = y0 + row * row_height + page.draw_line((x_values[0], y), (x_values[-1], y)) + for x in x_values: + page.draw_line((x, y0), (x, y0 + len(texts) * row_height)) + for row, values in enumerate(texts): + for column, value in enumerate(values): + page.insert_text( + (x_values[column] + 4, y0 + row * row_height + 15), value + ) + try: + default = page.find_tables(use_layout=False).tables + assert len(default) == 1 + assert default[0].row_count == 6 + + refined = page.find_tables(use_layout=False, refine=True).tables + assert len(refined) == 2 + assert [t.extract() for t in refined] == [texts[:3], texts[3:]] + # Each segment reports its own region, not the parent's. + assert refined[0].bbox[3] <= refined[1].bbox[1] + 1 + finally: + doc.close() + + def _make_merged_header_page(): """A page whose line grid detects a header cell that spans both body columns. @@ -852,6 +1078,138 @@ def cell(text, colspan=1, rowspan=1, tag="td"): ) +def _span_grid(rows): + """A placement grid from rows of cell text or ``(text, colspan, rowspan)``.""" + from pymupdf.table import SpanCell + + return [ + [SpanCell(None, *((cell, 1, 1) if isinstance(cell, str) else cell)) for cell in row] + for row in rows + ] + + +def test_find_tables_refine_tags_leaf_labels_under_a_spanning_header(): + """The row naming the columns of a header cell that spans some of them joins + the header: find_tables(refine=True) tags it th. The default result is + unchanged. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + texts = [ + ["Product", "Availability", None], + ["", "Online", "In store"], + ["Laptop", "Yes", "No"], + ["Phone", "No", "Yes"], + ["Tablet", "Yes", "Yes"], + ] + doc = pymupdf.open() + page = doc.new_page(width=400, height=300) + x_values = (60, 160, 240, 320) + y0, row_height = 60, 20 + y1 = y0 + len(texts) * row_height + for row in range(len(texts) + 1): + y = y0 + row * row_height + page.draw_line((x_values[0], y), (x_values[-1], y)) + for x in (60, 160, 320): + page.draw_line((x, y0), (x, y1)) + page.draw_line((240, y0 + row_height), (240, y1)) # no divider in the header row + for row, values in enumerate(texts): + for column, value in enumerate(values): + if value: + page.insert_text((x_values[column] + 4, y0 + row * row_height + 14), value) + try: + default = page.find_tables(use_layout=False).tables[0] + assert (default.placements, default.header_rows) == (None, 0) + assert default.extract() == texts + + t = page.find_tables(use_layout=False, refine=True).tables[0] + assert t.header_rows == 2 + assert t.to_html() == ( + "
" + '' + "" + "" + "" + "" + "
ProductAvailability
OnlineIn store
LaptopYesNo
PhoneNoYes
TabletYesYes
" + ) + finally: + doc.close() + + # A stub label spanning both header rows; an already named span is left alone. + grid = _span_grid([ + [("Product", 1, 2), ("Units sold", 2, 1)], + ["Online", "In store"], + ["Laptop", "120", "45"], + ["Phone", "300", "80"], + ]) + assert extend_header_leaf_labels(grid, 1) == 2 + assert extend_header_leaf_labels(grid, 2) == 2 + + +def test_header_leaf_labels_keep_a_numeric_body_row(): + """A record under a spanning header cell stays in the body: its blank/number/ + text shape repeats in the next row, or it puts a number under a column that + already has a label. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + repeated_shape = _span_grid([ + ["Region", ("Revenue", 2, 1)], + ["East", "10", "12"], + ["West", "11", "13"], + ]) + assert extend_header_leaf_labels(repeated_shape, 1) == 1 + number_under_label = _span_grid([ + ["Year", ("Sales", 2, 1)], + ["2023", "North", "South"], + ["2024", "10", "12"], + ]) + assert extend_header_leaf_labels(number_under_label, 1) == 1 + + +def test_header_leaf_labels_need_a_spanning_header_cell(): + """Without a header cell spanning more than one column but not all of them, + the header is left alone even when the next row reads like labels. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + single_cells = _span_grid([ + ["Product", "Online", "In store"], + ["", "units", "units"], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(single_cells, 1) == 1 + # A cell spanning every column is a title, not a group of columns. + full_width_title = _span_grid([ + [("Units sold", 3, 1)], + ["Product", "Online", "In store"], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(full_width_title, 1) == 1 + + +def test_header_leaf_labels_never_take_the_last_row(): + """A header never takes the whole table: the single record row under a + spanning header cell stays in the body. + + *** PyMuPDF extension (opt-in header rules). *** + """ + from pymupdf._table_headers import extend_header_leaf_labels + + single_record = _span_grid([ + ["Product", ("Units sold", 2, 1)], + ["Laptop", "120", "45"], + ]) + assert extend_header_leaf_labels(single_record, 1) == 1 + + def _make_bordered_table(page, x0, y0, texts): """Draw a bordered 2x2 table (cells 100 wide, 20 tall) at (x0, y0), with the 2x2 ``texts`` grid inserted into its cells; returns nothing (mutates page).""" @@ -939,3 +1297,177 @@ def test_find_tables_union_no_layout_degrades_to_line_candidates(): finally: pymupdf._get_layout = original_get_layout_fn doc.close() + + +def _make_bordered_grid(page, bbox, texts): + """Draw a uniformly divided grid with the same shape as ``texts``.""" + x0, y0, x1, y1 = bbox + row_count = len(texts) + col_count = len(texts[0]) + row_height = (y1 - y0) / row_count + col_width = (x1 - x0) / col_count + for row in range(row_count + 1): + y = y0 + row * row_height + page.draw_line((x0, y), (x1, y)) + for column in range(col_count + 1): + x = x0 + column * col_width + page.draw_line((x, y0), (x, y1)) + for row, values in enumerate(texts): + for column, value in enumerate(values): + if value: + page.insert_text( + ( + x0 + column * col_width + 5, + y0 + row * row_height + min(14, row_height - 3), + ), + value, + ) + + +def test_find_tables_union_forwards_virtual_lines_to_candidates(): + """Virtual raster-style rules reach the nested line finder in union mode.""" + doc = pymupdf.open() + page = doc.new_page(width=400, height=400) + for row, y in enumerate((100, 140)): + for col, x in enumerate((80, 180)): + page.insert_text((x + 8, y + 24), f"r{row}c{col}") + page.layout_information = [] + lines = [ + ((80, y), (280, y)) for y in (100, 140, 180) + ] + [ + ((x, 100), (x, 180)) for x in (80, 180, 280) + ] + try: + tables = page.find_tables( + use_layout=True, + union=True, + add_lines=lines, + ).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 2) + assert tables[0].extract()[1][1] == "r1c1" + finally: + doc.close() + + +def test_find_tables_union_rejects_multiline_single_row_panel(): + """A one-row partition around a whole picture is not emitted as a table.""" + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid( + page, + (80, 80, 380, 240), + [["left\naxis\nlabels", "right\naxis\nlabels"]], + ) + page.layout_information = [ + {"class_name": "picture", "group_bbox": [80.0, 80.0, 380.0, 240.0]}, + ] + try: + assert page.find_tables(use_layout=True, union=True).tables == [] + finally: + doc.close() + + +def test_find_tables_union_inside_picture_needs_table_shaped_text(): + """Inside a picture region a grid is admitted only on table-shaped text. + + The Layout model called the region a picture, so a line grid found almost + entirely inside one is as likely to be a chart's rules as a table. A chart + labels its two axes, so its text sits in exactly one row and one column; + a table -- even a sparse matrix of scattered marks -- has a header row and a + label column that are each at least half full. Outside a picture nothing + changes. + """ + picture = [{"class_name": "picture", "group_bbox": [120.0, 80.0, 440.0, 360.0]}] + # Chart shape: only the first row and the first column carry text. + axes = [ + ["", "x0", "x1", "x2"], + ["y0", "", "", ""], + ["y1", "", "", ""], + ["y2", "", "", ""], + ] + dense = [["a0", "b0", "c0"], ["a1", "b1", "c1"]] + # Sparse matrix: header row, label column and a few scattered marks. + matrix = [ + ["use", "z1", "z2", "z3", "z4", "z5"], + ["r1", "P", "P", "", "", ""], + ["r2", "P", "P", "", "", ""], + ["r3", "", "", "", "", ""], + ["r4", "", "", "", "", ""], + ["r5", "", "", "", "", ""], + ] + + def find(texts, bbox, layout): + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid(page, bbox, texts) + page.layout_information = list(layout) + try: + return page.find_tables(use_layout=True, union=True).tables + finally: + doc.close() + + assert find(axes, (160, 160, 400, 300), picture) == [] + assert len(find(axes, (160, 160, 400, 300), [])) == 1 # only the picture gates it + + tables = find(dense, (160, 180, 380, 240), picture) + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 3) + assert tables[0].extract()[1][2] == "c1" + + tables = find(matrix, (140, 140, 420, 320), picture) + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (6, 6) + assert tables[0].extract()[1][1] == "P" + + +def test_find_tables_union_keeps_a_table_holding_a_picture_in_one_cell(): + """A table larger than the picture contains it and keeps the ordinary rule.""" + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + # A 200x200 grid whose upper-left cell is the picture; the right column has text. + page.draw_rect((100, 100, 300, 300)) + page.draw_line((100, 200), (300, 200)) + page.draw_line((200, 100), (200, 300)) + page.insert_text((210, 130), "right top") + page.insert_text((210, 230), "right bottom") + page.insert_text((110, 230), "left bottom") + page.layout_information = [ + {"class_name": "picture", "group_bbox": [100.0, 100.0, 200.0, 200.0]}, + ] + try: + tables = page.find_tables(use_layout=True, union=True).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (2, 2) + assert tables[0].extract()[0][1] == "right top" + finally: + doc.close() + + +def test_find_tables_union_keeps_one_coherent_table_group_without_content_support(): + """One Layout table group is retained; ownership is not a rejection gate.""" + import types + + doc = pymupdf.open() + page = doc.new_page(width=500, height=500) + _make_bordered_grid( + page, + (80, 80, 380, 240), + [["left\naxis\nlabels", "right\naxis\nlabels"]], + ) + page.layout_information = [ + { + "class_name": "table", + "group_bbox": [80.0, 80.0, 380.0, 240.0], + "table_grid": types.SimpleNamespace( + h_lines=[50.0, 100.0], + v_lines=[150.0], + ), + } + ] + try: + tables = page.find_tables(use_layout=True, union=True).tables + assert len(tables) == 1 + assert (tables[0].row_count, tables[0].col_count) == (1, 2) + finally: + doc.close()