add nlb
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
node_modules
|
||||
*.pdf
|
||||
*.txt
|
||||
*.csv
|
||||
__pycache__
|
||||
@@ -0,0 +1 @@
|
||||
3.12
|
||||
@@ -0,0 +1,89 @@
|
||||
import csv
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
# extract transaction from izpisek pdf
|
||||
TRANSACTION_START = re.compile(r"^\d{2}\.\d{2}\.\d{2}\s")
|
||||
TRANSACTION_LINE = re.compile(
|
||||
r"^(?P<date>\d{2}\.\d{2}\.\d{2})\s+"
|
||||
r"(?P<middle>.*?)\s+"
|
||||
r"(?P<amount>[+-][\d.,]+)\s+"
|
||||
r"(?P<balance>-?[\d.,]+-?)\s*$"
|
||||
)
|
||||
CONTINUATION_INDENT = 15
|
||||
FIELD_SPLIT = re.compile(r"\s{4,}")
|
||||
ACCOUNT = re.compile(r"^SI\d{2}(?:\s?\d{4}){3}\s?\d{3}$")
|
||||
|
||||
|
||||
def pdf_to_text(pdf_path):
|
||||
result = subprocess.run(
|
||||
["pdftotext", "-layout", str(pdf_path), "-"],
|
||||
capture_output=True, text=True, check=True,
|
||||
)
|
||||
return result.stdout
|
||||
|
||||
|
||||
def parse_transaction(line):
|
||||
m = TRANSACTION_LINE.match(line.rstrip())
|
||||
account = ""
|
||||
desc_parts = []
|
||||
for f in FIELD_SPLIT.split(m.group("middle").strip()):
|
||||
if ACCOUNT.match(f):
|
||||
account = f
|
||||
else:
|
||||
desc_parts.append(f)
|
||||
balance = m.group("balance")
|
||||
if balance.endswith("-"):
|
||||
balance = "-" + balance[:-1]
|
||||
return {
|
||||
"date": m.group("date"),
|
||||
"desc": " ".join(desc_parts),
|
||||
"account": account,
|
||||
"amount": m.group("amount"),
|
||||
"balance": balance,
|
||||
}
|
||||
|
||||
|
||||
def extract_transactions(pdf_path):
|
||||
print(pdf_path)
|
||||
rows = []
|
||||
full_text = pdf_to_text(pdf_path)
|
||||
for page_text in full_text.split("\f"):
|
||||
page_start = len(rows)
|
||||
for line in page_text.split("\n"):
|
||||
if not line.strip():
|
||||
continue
|
||||
if TRANSACTION_START.match(line):
|
||||
rows.append(parse_transaction(line))
|
||||
continue
|
||||
if len(rows) <= page_start:
|
||||
continue
|
||||
indent = len(line) - len(line.lstrip(" "))
|
||||
if indent < CONTINUATION_INDENT:
|
||||
continue
|
||||
parts = FIELD_SPLIT.split(line.strip())
|
||||
rows[-1]["desc"] += " " + parts[0]
|
||||
if len(parts) > 1:
|
||||
rows[-1]["account"] = (rows[-1]["account"] + " " + parts[1]).strip()
|
||||
return rows
|
||||
|
||||
|
||||
def main():
|
||||
pdf_dir = Path(sys.argv[1])
|
||||
out_path = pdf_dir.with_suffix(".csv")
|
||||
|
||||
all_rows = []
|
||||
for pdf_path in sorted(pdf_dir.glob("*.pdf")):
|
||||
all_rows.extend(extract_transactions(pdf_path))
|
||||
|
||||
with open(out_path, "w", newline="") as f:
|
||||
writer = csv.DictWriter(f, fieldnames=["date", "desc", "account", "amount", "balance"])
|
||||
writer.writeheader()
|
||||
writer.writerows(all_rows)
|
||||
|
||||
print(f"Wrote {len(all_rows)} transactions to {out_path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,7 @@
|
||||
[project]
|
||||
name = "nlb-export"
|
||||
version = "0.1.0"
|
||||
description = "Add your description here"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
dependencies = []
|
||||
@@ -0,0 +1,52 @@
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
DATUM_IZPISKA = re.compile(r"Datum izpiska\s+(\d{2})\.(\d{2})\.(\d{4})")
|
||||
# raneme nlb-izpisek based on datum izpiska
|
||||
|
||||
def pdf_to_text(pdf_path):
|
||||
result = subprocess.run(
|
||||
["pdftotext", "-layout", str(pdf_path), "-"],
|
||||
capture_output=True, text=True, check=True,
|
||||
)
|
||||
return result.stdout
|
||||
|
||||
|
||||
def find_date(text):
|
||||
match = DATUM_IZPISKA.search(text)
|
||||
if not match:
|
||||
return None
|
||||
day, month, year = match.groups()
|
||||
return f"{year}-{month}-{day}"
|
||||
|
||||
|
||||
def unique_target(dir_path, stem, suffix, current_path):
|
||||
target = dir_path / f"{stem}{suffix}"
|
||||
n = 2
|
||||
while target.exists() and target != current_path:
|
||||
target = dir_path / f"{stem}_{n}{suffix}"
|
||||
n += 1
|
||||
return target
|
||||
|
||||
|
||||
def main():
|
||||
pdf_dir = Path(sys.argv[1])
|
||||
|
||||
for pdf_path in sorted(pdf_dir.glob("*.pdf")):
|
||||
text = pdf_to_text(pdf_path)
|
||||
date = find_date(text)
|
||||
if not date:
|
||||
print(f"skip {pdf_path.name}: no 'Datum izpiska' found")
|
||||
continue
|
||||
|
||||
target = unique_target(pdf_dir, f"izpisek_{date}", pdf_path.suffix, pdf_path)
|
||||
if target == pdf_path:
|
||||
continue
|
||||
pdf_path.rename(target)
|
||||
print(f"{pdf_path.name} -> {target.name}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Generated
+8
@@ -0,0 +1,8 @@
|
||||
version = 1
|
||||
revision = 3
|
||||
requires-python = ">=3.12"
|
||||
|
||||
[[package]]
|
||||
name = "nlb-export"
|
||||
version = "0.1.0"
|
||||
source = { virtual = "." }
|
||||
Reference in New Issue
Block a user