From fe40317186fb02d216f72c648154be89ca36b90c Mon Sep 17 00:00:00 2001 From: "youchang.kim" Date: Mon, 28 Sep 2026 19:41:06 +0900 Subject: [PATCH] tests/test_tables.py: make the two new table tests hold with layout and pymupdf4llm imported test_table_header_is_computed_on_first_access() now uses the second table of chinese-tables.pdf. With layout, the caption above the first table is taken as an external header, which the test is not about; the second table has its header as its top row with or without layout. test_find_tables_refine_splits_a_repeated_leading_header() now writes "2024/01" instead of "2024-01". pymupdf4llm sets unset_quad_corrections(True) on import, and then "-" and "." are extracted as a line of their own. The row-count assertion prints the extracted tables if it fails. This replaces the layout-dependent expectations added in 010f0127, so both tests check the same result with or without layout and pymupdf4llm. --- tests/test_tables.py | 38 +++++++++++++------------------------- 1 file changed, 13 insertions(+), 25 deletions(-) diff --git a/tests/test_tables.py b/tests/test_tables.py index 0bd6ed6fe..df2344b15 100644 --- a/tests/test_tables.py +++ b/tests/test_tables.py @@ -561,16 +561,14 @@ def test_table_header_is_computed_on_first_access(): doc = pymupdf.open(filename) page = doc[0] try: - tab = page.find_tables().tables[0] + # The second table: nothing above it can be taken for an external + # header, so its header is its top row with or without layout. + tab = page.find_tables().tables[1] assert tab._header is pymupdf.table._NO_HEADER_YET header = tab.header assert header is tab.header # computed once - if util._use_layout(): - assert header.external is True - print(f'Not asserting header.names == tab.extract()[0] because using layout.') - else: - assert header.external is False - assert header.names == tab.extract()[0] + assert header.external is False + assert header.names == tab.extract()[0] assert tab.to_markdown().startswith("|") tab.header = None # assignable, as before assert tab.header is None @@ -1357,13 +1355,15 @@ def test_find_tables_refine_splits_a_repeated_leading_header(): *** PyMuPDF extension. *** """ + # No "-" or "." in the cells: with quad corrections off (pymupdf4llm sets + # this globally) such small glyphs are extracted as a line of their own. texts = [ ["Date", "BI", "PD"], - ["2024-01", "1", "2"], - ["2024-02", "3", "4"], + ["2024/01", "1", "2"], + ["2024/02", "3", "4"], ["Date", "BI", "PD"], - ["2023-01", "7", "8"], - ["2023-02", "9", "10"], + ["2023/01", "7", "8"], + ["2023/02", "9", "10"], ] doc = pymupdf.open() page = doc.new_page(width=400, height=400) @@ -1385,20 +1385,8 @@ def test_find_tables_refine_splits_a_repeated_leading_header(): assert default[0].row_count == 6 refined = page.find_tables(use_layout=False, refine=True).tables - assert len(refined) == 2 - if util._use_layout(): - # Cope with regression. - texts_post = [ - ["Date", "BI", "PD"], - ["2024 01\n-", "1", "2"], - ["2024 02\n-", "3", "4"], - ["Date", "BI", "PD"], - ["2023 01\n-", "7", "8"], - ["2023 02\n-", "9", "10"], - ] - assert [t.extract() for t in refined] == [texts_post[:3], texts_post[3:]] - else: - assert [t.extract() for t in refined] == [texts[:3], texts[3:]] + assert len(refined) == 2, [t.extract() for t in refined] + assert [t.extract() for t in refined] == [texts[:3], texts[3:]] # Each segment reports its own region, not the parent's. assert refined[0].bbox[3] <= refined[1].bbox[1] + 1 finally: