Skip to content
Open
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
47 changes: 45 additions & 2 deletions pdfparser/pdfoxide_parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
It is optimized for performance and multiprocessing safety.
"""

import re
from pathlib import Path
from typing import Any, Dict

Expand All @@ -13,6 +14,48 @@
from pdfparser.utils import extract_metadata, extract_summary_totals, extract_transactions


def preprocess_text(text: str) -> str:
"""
Preprocess text extracted by pdf_oxide to separate smashed columns.

pdf_oxide often extracts text in draw-order, causing columns to be merged
without spaces or newlines. This function inserts newlines to separate
fields, especially amounts and user IDs.

Args:
text: Raw text extracted from PDF page

Returns:
Processed text with inserted newlines to separate fields
"""
if not text:
return ""

# 1. Split smashed amounts (e.g. "0.001,000.00" -> "0.00\n1,000.00")
# Amounts always end in .00 in these statements
# Match .xx followed immediately by digit or comma
# Apply twice to handle chains of 3+ amounts (e.g. A.00B.00C.00)
# because re.sub doesn't handle overlapping matches in one pass
text = re.sub(r"(\.\d{2})([\d,])", r"\1\n\2", text)
text = re.sub(r"(\.\d{2})([\d,])", r"\1\n\2", text)

# 2. Split space-separated amounts to new lines
# e.g. "Description 100.00" -> "Description\n100.00"
# Matches space followed by number with decimal
text = re.sub(r" (\d[\d,]*\.\d{2})", r"\n\1", text)

# 3. Split smashed User IDs from text
# e.g. "BRImo8888049" -> "BRImo\n8888049"
# User IDs are typically 6-8 digits
text = re.sub(r"([a-zA-Z])(\d{6,8})(?=\s|$)", r"\1\n\2", text)

# 4. Split Date/Time from Description if on the same line
# e.g. "02/05/25 09:38:58 Transfer..." -> "02/05/25 09:38:58\nTransfer..."
text = re.sub(r"^(\d{2}/\d{2}/\d{2}\s+\d{2}:\d{2}:\d{2})\s+", r"\1\n", text, flags=re.MULTILINE)

return text


def parse_pdf_pdfoxide(path: str) -> Dict[str, Any]:
"""
Parse Indonesian bank statement PDF using pdf_oxide.
Expand Down Expand Up @@ -60,8 +103,6 @@ def parse_pdf_pdfoxide(path: str) -> Dict[str, Any]:

# Fallback: extract account_no from filename if not found in text
if not metadata.get("account_no"):
import re

acct_match = re.search(r"(\d{10,16})", path_obj.stem)
if acct_match:
metadata["account_no"] = acct_match.group(1)
Expand All @@ -70,6 +111,8 @@ def parse_pdf_pdfoxide(path: str) -> Dict[str, Any]:
all_text = ""
for page_num in range(page_count):
page_text = doc.extract_text(page_num) # type: ignore[attr-defined] or ""
# Preprocess the page text to fix layout issues
page_text = preprocess_text(page_text)
all_text += page_text + "\n"
Comment on lines 113 to 116

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟡 Minor

Bug: or "" is inside the comment, not in the expression.

The fallback or "" appears after the # type: ignore comment, making it ineffective. If extract_text returns None, page_text will be None.

While preprocess_text handles None via if not text, this is likely unintentional and could mask issues. The intent seems to be a fallback empty string.

🐛 Proposed fix
-            page_text = doc.extract_text(page_num)  # type: ignore[attr-defined] or ""
+            page_text = doc.extract_text(page_num) or ""  # type: ignore[attr-defined]
🤖 Prompt for AI Agents
In `@pdfparser/pdfoxide_parser.py` around lines 113 - 116, The fallback
empty-string is currently inside the comment and not applied; change the
assignment so doc.extract_text(page_num) is OR'd with "" before the type-ignore
comment (e.g., assign page_text = doc.extract_text(page_num) or ""  # type:
ignore[attr-defined]) so page_text is never None prior to calling
preprocess_text(page_text) and concatenating into all_text; update the statement
that sets page_text in pdfoxide_parser.py accordingly.

transactions = extract_transactions(all_text)

Expand Down