Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 2 additions & 5 deletions pdfparser/pymupdf_parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,13 +62,10 @@ def parse_pdf_pymupdf(path: str) -> Dict[str, Any]:
if not metadata.get("account_no"):
import re

# Match 10-16 digit number in filename, but not if it looks like part of date
# Match 10-16 digit number in filename
acct_match = re.search(r"(\d{10,16})", path_obj.stem)
if acct_match:
# Verify it's not a date-like pattern (e.g., 2024-01-15)
potential_acct = acct_match.group(1)
if not re.match(r"^\d{4}-\d{2}-\d{2}$", potential_acct):
metadata["account_no"] = potential_acct
metadata["account_no"] = acct_match.group(1)

# Extract transactions from all pages
all_text = ""
Expand Down
41 changes: 16 additions & 25 deletions pdfparser/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -308,40 +308,31 @@ def extract_transactions(text: str) -> List[Dict[str, str]]:
is_user_id = bool(_user_id_match(next_field))
is_amount = bool(_amount_match(next_field))

if is_user_id:
# Format with user ID
if is_amount:
# Format without user ID - next field is debit
user = ""
debit = next_field
i += 1
else:
# Format with user ID (or fallback to assuming it is one)
user = next_field
i += 1
# Skip to debit
while i < len(lines) and not lines[i].strip():
i += 1
debit = lines[i].strip() if i < len(lines) else ""
i += 1
while i < len(lines) and not lines[i].strip():
i += 1
credit = lines[i].strip() if i < len(lines) else ""
i += 1
while i < len(lines) and not lines[i].strip():
i += 1
balance = lines[i].strip() if i < len(lines) else ""
elif is_amount:
# Format without user ID - next field is debit
user = ""
debit = next_field

# Extract credit
while i < len(lines) and not lines[i].strip():
i += 1
while i < len(lines) and not lines[i].strip():
i += 1
credit = lines[i].strip() if i < len(lines) else ""
credit = lines[i].strip() if i < len(lines) else ""
i += 1

# Extract balance
while i < len(lines) and not lines[i].strip():
i += 1
while i < len(lines) and not lines[i].strip():
i += 1
balance = lines[i].strip() if i < len(lines) else ""
else:
# Fallback - assume user ID
user = next_field
debit = ""
credit = ""
balance = ""
balance = lines[i].strip() if i < len(lines) else ""

transaction = {
"date": date,
Expand Down
17 changes: 17 additions & 0 deletions tests/test_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,6 +110,23 @@ def test_extract_transactions_keys_are_strings(self, transaction_text):
for key in item.keys():
assert isinstance(key, str), f"Key '{key}' is not a string"

def test_extract_transactions_with_9_digit_user_id(self):
"""Verify extraction when user ID is 9 digits (regression test)."""
text = """
01/01/24 12:00:00
Transfer Test
123456789
100000.00
0.00
500000.00
"""
transactions = extract_transactions(text)
assert len(transactions) == 1
txn = transactions[0]
assert txn['user'] == "123456789"
assert txn['debit'] == "100000.00"
assert txn['balance'] == "500000.00"


class TestTransactionDatePattern:
"""Tests for transaction date regex pattern using hypothesis."""
Expand Down
Loading