Upload files to "/"
This commit is contained in:
@@ -0,0 +1,105 @@
|
||||
import camelot
|
||||
import pandas as pd
|
||||
import re
|
||||
|
||||
def extract_bank_data(file_path):
|
||||
print(f"Reading {file_path}... processing all pages.")
|
||||
|
||||
# We use 'stream' flavor for the NLB PDF layout
|
||||
tables = camelot.read_pdf(file_path, pages='1-end', flavor='stream')
|
||||
|
||||
final_transactions = []
|
||||
current_tx = None
|
||||
|
||||
# Precise Regex Patterns
|
||||
iban_pattern = r'SI56\s?\d{4}\s?\d{4}\s?\d{4}\s?\d{3}'
|
||||
amount_pattern = r'=(\d+[\.,]\d{2})'
|
||||
date_pattern = r'(\d{2}\.\d{2}\.\d{2})'
|
||||
|
||||
for table in tables:
|
||||
df = table.df
|
||||
for _, row in df.iterrows():
|
||||
cells = [str(c).strip() for c in row]
|
||||
row_text = " ".join(cells)
|
||||
|
||||
# 1. TRIGGER: A new transaction starts with an amount (=xx,xx)
|
||||
amount_match = re.search(amount_pattern, row_text)
|
||||
date_match = re.search(date_pattern, row_text)
|
||||
|
||||
if amount_match and date_match:
|
||||
# Save previous transaction
|
||||
if current_tx:
|
||||
final_transactions.append(current_tx)
|
||||
|
||||
# Assign Debit/Credit based on the column index of the '=' sign
|
||||
amt_val = amount_match.group(1)
|
||||
is_credit = any('=' in cells[i] for i in range(len(cells)) if i >= 2)
|
||||
|
||||
current_tx = {
|
||||
"value date": date_match.group(1),
|
||||
"Creditor / Debtor": cells[0].split('\n')[0], # Only take the first line of text
|
||||
"account": "",
|
||||
"payment details": "",
|
||||
"debit": "" if is_credit else amt_val,
|
||||
"credit": amt_val if is_credit else "",
|
||||
"is_collecting_details": False
|
||||
}
|
||||
|
||||
# Check if IBAN is in this row
|
||||
iban_search = re.search(iban_pattern, cells[0])
|
||||
if iban_search:
|
||||
current_tx["account"] = iban_search.group(0)
|
||||
|
||||
elif current_tx:
|
||||
# 2. CONTINUATION: Extract info from rows following the amount row
|
||||
text_line = cells[0]
|
||||
if not text_line:
|
||||
continue
|
||||
|
||||
# Check for IBAN
|
||||
iban_search = re.search(iban_pattern, text_line)
|
||||
if iban_search:
|
||||
current_tx["account"] = iban_search.group(0)
|
||||
# Once we hit the IBAN, everything following it is usually 'Payment Details'
|
||||
current_tx["is_collecting_details"] = True
|
||||
remaining = re.sub(iban_pattern, '', text_line).strip()
|
||||
# Filter out BIC/Internal Routing codes
|
||||
remaining = re.sub(r'[A-Z]{8,11}', '', remaining).strip()
|
||||
if remaining:
|
||||
current_tx["payment details"] += " " + remaining
|
||||
continue
|
||||
|
||||
# Identify if this is a Payment Detail row (Reference numbers or purpose text)
|
||||
# We filter out common address keywords (Cesta, Ulica, Trg, Postcodes)
|
||||
is_address = any(kw in text_line.upper() for kw in ["CESTA", "ULICA", "TRG", " LJUBLJANA", " LOGATEC"])
|
||||
|
||||
if current_tx["is_collecting_details"] or ("SI00" in text_line or "SI12" in text_line):
|
||||
# Filter out routing codes like BACXSI...
|
||||
clean_text = re.sub(r'[A-Z]{8,11}', '', text_line).strip()
|
||||
if clean_text:
|
||||
current_tx["payment details"] = (current_tx["payment details"] + " " + clean_text).strip()
|
||||
elif not is_address and not current_tx["account"]:
|
||||
# If it's not an address and we don't have an IBAN yet, it might be Name continuation
|
||||
if len(current_tx["Creditor / Debtor"]) < 20: # Only append if the name is very short
|
||||
current_tx["Creditor / Debtor"] += " " + text_line
|
||||
|
||||
# Append the last transaction
|
||||
if current_tx:
|
||||
final_transactions.append(current_tx)
|
||||
|
||||
# 3. Clean up and Export
|
||||
df_final = pd.DataFrame(final_transactions)
|
||||
if "is_collecting_details" in df_final.columns:
|
||||
df_final = df_final.drop(columns=["is_collecting_details"])
|
||||
|
||||
# Final text cleaning (remove double spaces, routing codes)
|
||||
df_final["payment details"] = df_final["payment details"].str.replace(r'[A-Z]{8,11}', '', regex=True)
|
||||
df_final["payment details"] = df_final["payment details"].str.replace(r'\s+', ' ', regex=True).str.strip()
|
||||
|
||||
# Export
|
||||
output_file = "bank_export_fixed_v3.csv"
|
||||
df_final.to_csv(output_file, sep=';', index=False, encoding='utf-8-sig')
|
||||
print(f"Success! Exported {len(df_final)} transactions to {output_file}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
extract_bank_data("MSDataReport (1).pdf")
|
||||
Reference in New Issue
Block a user