updating scripts

This commit is contained in:
2026-04-02 11:26:40 +01:00
parent b5c2d4593d
commit f9e935dcec
6 changed files with 70 additions and 96 deletions

View File

@@ -1,132 +1,88 @@
#!/usr/bin/env python3
import csv
from decimal import Decimal, InvalidOperation
from pathlib import Path
import re
from decimal import Decimal
from datetime import datetime
import sys
from pathlib import Path
RAW_DIR = Path("/home/zaine/master-folder/attachments/bank-statements/nationwide/raw")
CLEANED_DIR = Path("/home/zaine/master-folder/attachments/bank-statements/nationwide/cleaned")
# Only keep digits, dot, minus for money
MONEY_REGEX = re.compile(r"[^\d\.-]")
def clean_money(value: str) -> Decimal | None:
def clean_money(value: str):
if not value:
return None
value = (
value.replace("£", "")
.replace(",", "")
.strip()
)
# Remove all non-numeric characters
value = MONEY_REGEX.sub("", value)
if not value:
return None
return Decimal(value)
try:
return Decimal(value)
except InvalidOperation:
return None
def parse_date(value: str) -> str | None:
"""
Convert '01 Dec 2025''12/01/2025'
"""
value = value.strip()
if not value:
return None
try:
dt = datetime.strptime(value, "%d %b %Y")
return dt.strftime("%m/%d/%Y")
except ValueError:
return None
def parse_date(value: str):
# Input format: 27 Feb 2026
return datetime.strptime(value.strip(), "%d %b %Y").strftime("%m/%d/%Y")
def convert(in_path: Path, out_path: Path):
with open(in_path, newline="", encoding="latin-1") as f_in, \
open(out_path, "w", newline="", encoding="utf-8") as f_out:
rows_written = 0
reader = csv.reader(f_in)
# Read raw bytes and decode aggressively
with open(in_path, "rb") as f:
lines = f.read().decode("utf-8", errors="ignore").splitlines()
# Skip metadata lines until header
for row in reader:
if row and row[0] == "Date":
header = row
break
else:
raise RuntimeError("Could not find Nationwide CSV header row")
with open(out_path, "w", newline="", encoding="utf-8") as f_out:
writer = csv.writer(f_out, lineterminator="\r\n")
writer.writerow(["Date", "Payee", "Description", "Amount"])
dict_reader = csv.DictReader(f_in, fieldnames=header)
fieldnames = ["Date", "Payee", "Description", "Amount"]
writer = csv.DictWriter(f_out, fieldnames=fieldnames)
writer.writeheader()
for row in dict_reader:
raw_date = row.get("Date") or ""
date = parse_date(raw_date)
if not date:
for row in csv.reader(lines):
if not row:
continue
first = row[0].strip()
if first.startswith("Account") or first == "Date":
continue
if len(row) < 5:
continue
desc = (row.get("Description") or "").strip()
payee = desc
try:
date = parse_date(row[0])
except:
continue
paid_out = clean_money(row.get("Paid out", ""))
paid_in = clean_money(row.get("Paid in", ""))
desc = row[2].strip()
paid_out = clean_money(row[3])
paid_in = clean_money(row[4])
if paid_out is not None:
amount = -paid_out
amount = f"{-paid_out:.2f}"
elif paid_in is not None:
amount = paid_in
amount = f"{paid_in:.2f}"
else:
continue
writer.writerow({
"Date": date,
"Payee": payee,
"Description": desc,
"Amount": f"{amount:.2f}",
})
def select_file(files: list[Path]) -> Path:
print("\nSelect a Nationwide CSV to clean:\n")
for i, f in enumerate(files, start=1):
print(f"[{i}] {f.name}")
try:
choice = int(input("\nEnter number: ").strip())
return files[choice - 1]
except (ValueError, IndexError):
print("Invalid selection.")
sys.exit(1)
writer.writerow([date, desc, desc, amount])
rows_written += 1
return rows_written
def main():
if not RAW_DIR.exists():
print(f"Raw directory does not exist: {RAW_DIR}")
sys.exit(1)
CLEANED_DIR.mkdir(parents=True, exist_ok=True)
csv_files = sorted(RAW_DIR.glob("*.csv"))
files = sorted(RAW_DIR.glob("*.csv"))
if not files:
print("No CSV files found.")
return
if not csv_files:
print("No CSV files found in raw directory.")
sys.exit(0)
print("\nSelect a Nationwide CSV to clean:\n")
for i, f in enumerate(files, 1):
print(f" [{i}] {f.name}")
in_path = select_file(csv_files)
out_name = f"{in_path.stem}-cleaned.csv"
out_path = CLEANED_DIR / out_name
convert(in_path, out_path)
print(f"\n✔ Written cleaned file to:\n{out_path}\n")
choice = int(input("\nEnter number: "))
in_path = files[choice - 1]
out_path = CLEANED_DIR / f"{in_path.stem}-cleaned.csv"
n = convert(in_path, out_path)
print(f"\n{n} rows written to:\n {out_path}\n")
if __name__ == "__main__":
main()