Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 13 additions & 25 deletions tests/test_tables.py
Original file line number Diff line number Diff line change
Expand Up @@ -561,16 +561,14 @@ def test_table_header_is_computed_on_first_access():
doc = pymupdf.open(filename)
page = doc[0]
try:
tab = page.find_tables().tables[0]
# The second table: nothing above it can be taken for an external
# header, so its header is its top row with or without layout.
tab = page.find_tables().tables[1]
assert tab._header is pymupdf.table._NO_HEADER_YET
header = tab.header
assert header is tab.header # computed once
if util._use_layout():
assert header.external is True
print(f'Not asserting header.names == tab.extract()[0] because using layout.')
else:
assert header.external is False
assert header.names == tab.extract()[0]
assert header.external is False
assert header.names == tab.extract()[0]
assert tab.to_markdown().startswith("|")
tab.header = None # assignable, as before
assert tab.header is None
Expand Down Expand Up @@ -1357,13 +1355,15 @@ def test_find_tables_refine_splits_a_repeated_leading_header():

*** PyMuPDF extension. ***
"""
# No "-" or "." in the cells: with quad corrections off (pymupdf4llm sets
# this globally) such small glyphs are extracted as a line of their own.
texts = [
["Date", "BI", "PD"],
["2024-01", "1", "2"],
["2024-02", "3", "4"],
["2024/01", "1", "2"],
["2024/02", "3", "4"],
["Date", "BI", "PD"],
["2023-01", "7", "8"],
["2023-02", "9", "10"],
["2023/01", "7", "8"],
["2023/02", "9", "10"],
]
doc = pymupdf.open()
page = doc.new_page(width=400, height=400)
Expand All @@ -1385,20 +1385,8 @@ def test_find_tables_refine_splits_a_repeated_leading_header():
assert default[0].row_count == 6

refined = page.find_tables(use_layout=False, refine=True).tables
assert len(refined) == 2
if util._use_layout():
# Cope with regression.
texts_post = [
["Date", "BI", "PD"],
["2024 01\n-", "1", "2"],
["2024 02\n-", "3", "4"],
["Date", "BI", "PD"],
["2023 01\n-", "7", "8"],
["2023 02\n-", "9", "10"],
]
assert [t.extract() for t in refined] == [texts_post[:3], texts_post[3:]]
else:
assert [t.extract() for t in refined] == [texts[:3], texts[3:]]
assert len(refined) == 2, [t.extract() for t in refined]
assert [t.extract() for t in refined] == [texts[:3], texts[3:]]
# Each segment reports its own region, not the parent's.
assert refined[0].bbox[3] <= refined[1].bbox[1] + 1
finally:
Expand Down
Loading