From 0b3b4eeaa6f9abece820be369db78d2c2634d62b Mon Sep 17 00:00:00 2001 From: stancld Date: Tue, 7 Oct 2025 13:33:03 +0200 Subject: [PATCH] Convert correctly PDFium UTF-8 integers to char/string --- pdftext/pdf/chars.py | 37 ++++++++++++++++++++++++++++++- tests/conftest.py | 4 ++++ tests/data/non_unicode_chars.pdf | Bin 0 -> 2691 bytes tests/pdf/test_chars.py | 28 +++++++++++++++++++++++ tests/test_extraction.py | 7 ++++++ 5 files changed, 75 insertions(+), 1 deletion(-) create mode 100644 tests/data/non_unicode_chars.pdf create mode 100644 tests/pdf/test_chars.py diff --git a/pdftext/pdf/chars.py b/pdftext/pdf/chars.py index 2a06a1d..475d8cc 100644 --- a/pdftext/pdf/chars.py +++ b/pdftext/pdf/chars.py @@ -7,6 +7,9 @@ from pdftext.schema import Bbox, Char, Chars, Spans, Span +MAX_UNICODE_INT = 1114111 # 0x10ffff + + def get_chars(textpage: pdfium.PdfTextPage, page_bbox: list[float], page_rotation: int, quote_loosebox=True) -> Chars: chars: Chars = [] @@ -15,7 +18,7 @@ def get_chars(textpage: pdfium.PdfTextPage, page_bbox: list[float], page_rotatio page_height = math.ceil(abs(y_end - y_start)) for i in range(textpage.count_chars()): - text = chr(pdfium_c.FPDFText_GetUnicode(textpage, i)) + text = utf8_int_to_string(pdfium_c.FPDFText_GetUnicode(textpage, i)) rotation = pdfium_c.FPDFText_GetCharAngle(textpage, i) loosebox = (rotation == 0) and (text != "'" or quote_loosebox) @@ -117,3 +120,35 @@ def word_break(): deduped.append(word) return [char for word in deduped for char in word['chars']] + + +def utf8_int_to_string(utf8_int: int) -> str: + """Decode UTF-8 integer to string. +` + PDFium's `FPDFText_GetUnicode` returns unsigned 32-bit integer. Integers ≤ 1114111 are valid + Unicode codepoint and can be converted with python in-built `chr` function. + Larger integers are UTF-8 bytes packed into integers and must be handled separately. + + Parameters + ---------- + utf8_int + Unsgined 32-bit ingeger value from FPDFText_GetUnicode that may be either a valid Unicode + codepoint, or UTF-8 bytes packed as an integer. + + Returns + ------- + The decoded character or string. + + Examples + -------- + >>> utf8_int_to_string(65) # Valid Unicode + 'A' + >>> utf8_int_to_string(15112101) # UTF-8 bytes for '日' + '日' + """ + if utf8_int <= MAX_UNICODE_INT: + return chr(utf8_int) + # Compute byte length using 8-bit ceiling + byte_length = (utf8_int.bit_length() + 7) // 8 + bytes_obj = utf8_int.to_bytes(byte_length, "big") + return bytes_obj.decode("utf-8") diff --git a/tests/conftest.py b/tests/conftest.py index dc09f49..bb92f6c 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -9,6 +9,10 @@ def pdf_path(): def pdf_path2(): return "tests/data/communication.pdf" +@pytest.fixture(scope="session") +def pdf_with_non_unicode_chars(): + return "tests/data/non_unicode_chars.pdf" + @pytest.fixture() def pdf_doc(pdf_path): doc = pdfium.PdfDocument(pdf_path) diff --git a/tests/data/non_unicode_chars.pdf b/tests/data/non_unicode_chars.pdf new file mode 100644 index 0000000000000000000000000000000000000000..36e9577e06ded9fbc542d68bf68c48f411bf039d GIT binary patch literal 2691 zcmb7G+jinM6n*zsC<$T6h1ixY84q#K;G?PB`2imv3_glMV zAfd_3npM^^l8(-0YfI-yt7p|(Uf(j5)xZDx^B>X_<1k7-^A0%>frxk_JOT~kh3;e| zf`s&wWL#9$r>Cc_DAgVE&XyaF6rE@!>^v$TKPqZXCltdbQEQYKvz|dr*;7g-MFgHy zl{#~iJsyi%hyd!L82%;_-{nNz58Tl6gC0?@{Gc4f{;S}Gj|AN7L~Ttv$@3V43;bg& z8_dAhLMRF1d!hVT_zgao9EABRM%1c^U4Y|3BJE>xN91xAvAr><)T^!MI1!_I&<#`N z)5sH%#GE_gV;<;hL~V+mA1Bc>$(6Bg0ZD5z9uGxICu@c&Bj0#x;ShKw;IUDM0PfFwz$w1UG| zVMUQeuKX%YwW>iCu7{s;${ze`akbnuDpSX9->cn71hyM@u7008-5a9*w3z3}fF5bg(oFY5)g7Z8$_M198E9Tkajs$ZM)y z2L4gKmA|aCX#S9>v&_R`eA(c4JG+p1GU?xE-Yp+@tH_cLN0!Z98gE?wD4E%U%L)z_ zvLhxAgeW-2!t?p`=xdNzI@XUZ#DdOvB-5xc)oY44oJ6jO(-cc9kE~$ua2ADbOC&fx zs;qqKh4_Z5R3Gu4*qgPLSzDj7r8dI^l@jLaMZ|;DrDPHO&rbUpj2AeupgQM7P929H zO4em8PQm~kDT?Pnr|DAR$jY#{;KN?XJpqKQC z!A$80RbTLta=>H$*ePvBYPM`YbO-Cr!bAJG*xf6HS8ewq+v@d>zwTD!y&bd*sH92)wR3Lt@38;uI#OwLs4uTZDqzzp>cnFw6g4sP9HMc2e}6)(>pcyuPj4f zs;$xOo!nBkw6+qSZFWmv3PNb-&CEksnPlz@jl<04!SMF5vQxb~TPoi~1NWj6HG9MRIkmmp zqxK+5mT&4On_HK6!NbFsV(+-(-fkZB+t=}@jqXif-Tt+0`i?kNx)G z#9SGThDWTlvQuAHoo+rcqGoCm8DB;qcaXoume2YX^k++_LYjyTRt!D-!UGgHc@h*@*61>QCii-6{ zChOxHnXco;eJi8dJ6qjwkg5Omr<)c|-W!>1z4uq3s9|%xB;x*1MCk*oUM0%i$8Lvl zZgX33-DD24Eyr*RPRFsRVRSv}lzy(!R#q3@e7*0#`SN+<`6x+9!yi5W))YlepFpxK kH5F9EFd?$cq2Ka<7jh%geBe=%ZVjcDsjRN None: + assert utf8_int_to_string(utf8_int) == expected_str \ No newline at end of file diff --git a/tests/test_extraction.py b/tests/test_extraction.py index dc8fdde..479d0d1 100644 --- a/tests/test_extraction.py +++ b/tests/test_extraction.py @@ -42,3 +42,10 @@ def test_line_joining(pdf_path2): text = plain_text_output(pdf_path2, page_range=pages).lower() assert "the axis media control viewer toolbar" in text assert "axismediacontrolviewertoolbar" not in text + + +def test_parsing_non_unicode_chars(pdf_with_non_unicode_chars): + text = plain_text_output(pdf_with_non_unicode_chars).lower() + assert "日" in text + assert "付" in text +