Files
jorgecuadros-platform/migration/transform_bank.py
T
rmancinasandClaude Opus 4.8 1b79b43a54
Build and Push Images / Build jorgecuadros-web (push) Successful in 1m37s
Build and Push Images / Build jorgecuadros-api (push) Successful in 2m9s
fix(migration): make Phase B additive sync actually work + verify end-to-end
The --sync path had never been run and was broken in several ways. Fixed and
verified against the dev DB (two consecutive syncs, both exit 0, 32/32
assertions: stable PKs, manual-row preservation, changed-row updates,
legacy-delete, no child duplication, zero FK orphans; idempotent).

- policies/properties: reuse each legacy row's existing id (by provenance)
  BEFORE building child rows, so children no longer point at a discarded fresh
  uuid; rebuild legacy-owned children via scoped delete + reinsert.
- customers: replace zip(customers, refs) (mispaired almost every row) with a
  ref-grouped id remap; names now restore and no spurious customers appear.
- drop the invalid Vehicle @@unique(legacySourceTable, legacyId) — one legacy
  policy row carries up to 3 vehicles sharing a legacyId; handle via delete+reinsert.
- upsert lookup tables (policy_types, insurance_providers, type_transactions,
  adjusters) by natural name and remap child FKs instead of inserting fresh
  uuids that nothing points at.
- transactions: drop updatedAt=NOW() (no such column); guard report formatting
  on NULL legacySourceTable (manual rows). Same report guard in bank.
- add manual-safe prune (prune_empty_customers.py --sync, in SYNC_STEPS): prune
  only legacy-owned empties, never manually-added customers.

web: customer-detail mini tx list now strikes voided rows with an "(anulado)"
tag (was the last void-UI rendering gap; /estado-cuenta already handled it).

docs: RESUME.md updated — Phase B sync marked verified end-to-end, void-UI
browser pass recorded.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-24 13:37:22 -07:00

146 lines
5.3 KiB
Python

"""
Migration plan step 3 (bank register): SCOTHIA.mdb -> bank_transactions +
business_line_categories. This is the office's OWN operating checking account
("chequera"), deliberately separate from customer-facing `transactions` and
carrying no customer FK — so it can load independently of the other steps.
Sources:
- DATOS I (ingresos) -> amount = +ingreso, transferred flag, cleared=operado
- DATOS E (egresos) -> amount = -egreso (expenses negative), amountInWords
from the spelled-out "cantidad en letra"
- TABLA RAMODOS -> business_line_categories (line-of-business lookup)
Category link: DATOS E/I have no explicit FK to TABLA RAMODOS — the ramo is
inferred from the CONCEPTO text, which is a fuzzy classification, not a stored
key. So the categories are loaded but bank_transactions.categoryId is left
NULL for now; a concept->ramo classifier is a later enhancement.
Idempotent (truncate + rebuild). Run:
./.venv/bin/python transform_bank.py --env dev
"""
from __future__ import annotations
import uuid
from decimal import Decimal, InvalidOperation
from pathlib import Path
import pandas as pd
from dbenv import connect, env_arg
from sync import parse_mode
STG = Path(__file__).parent / "output" / "stg_scothia"
NULL = "∅"
def s(v):
if v is None or pd.isna(v):
return None
v = str(v).strip()
return None if v in ("", NULL, "0000-00-00") else v
def dec(v, default=None):
v = s(v)
if v is None:
return default
try:
return Decimal(v.replace(",", ""))
except (InvalidOperation, ValueError):
return default
def dt(v):
v = s(v)
if v is None:
return None
d = pd.to_datetime(v, errors="coerce")
return None if pd.isna(d) else d.to_pydatetime()
def truthy(v):
return (s(v) or "0").lower() in {"1", "-1", "true", "si", "sí", "yes"}
def load(name):
df = pd.read_parquet(STG / f"{name}.parquet").sort_values("_row_num").reset_index(drop=True)
df = df[[c for c in df.columns if c != "_legacy_source_table"]].copy()
for c in df.columns:
if c != "_row_num":
df[c] = df[c].astype("string").str.strip()
return df
def main():
env, sync_mode = parse_mode()
conn = connect(env)
print(f"[bank] target env: {env}")
c = conn.cursor()
# business_line_categories (dedup TABLA RAMODOS)
cats, seen = [], set()
for _, r in load("tabla_ramodos").iterrows():
name = s(r["ramo2"])
if name and name.upper() not in seen:
seen.add(name.upper())
cats.append((str(uuid.uuid4()), name))
rows = []
skip_date = 0
def add(r, amount, income: bool):
nonlocal skip_date
td = dt(r["fecha"])
if td is None:
skip_date += 1
return
rows.append((
str(uuid.uuid4()), td, s(r["tipo"]), s(r["num"]), s(r["concepto"]),
amount, None, # categoryId left NULL (see header)
1 if truthy(r["operado"]) else 0,
1 if (income and truthy(r["transferido"])) else 0,
s(r["notas"]),
None if income else s(r["cantidad_en_letra"]),
"DATOS I" if income else "DATOS E", str(int(r["_row_num"])),
))
for _, r in load("datos_i").iterrows():
add(r, dec(r["ingreso"], Decimal(0)), income=True)
for _, r in load("datos_e").iterrows():
add(r, -(dec(r["egreso"], Decimal(0))), income=False)
if sync_mode:
for row in rows:
c.execute("INSERT INTO bank_transactions (id,transactionDate,transactionType,reference,concept,amount,categoryId,cleared,transferred,notes,amountInWords,legacySourceTable,legacyId) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s) ON DUPLICATE KEY UPDATE transactionDate=VALUES(transactionDate),transactionType=VALUES(transactionType),reference=VALUES(reference),concept=VALUES(concept),amount=VALUES(amount),cleared=VALUES(cleared),transferred=VALUES(transferred),notes=VALUES(notes),amountInWords=VALUES(amountInWords),voidedAt=NULL", row)
else:
c.execute("SET FOREIGN_KEY_CHECKS=0")
for t in ("bank_transactions", "business_line_categories"):
c.execute(f"TRUNCATE TABLE {t}")
c.execute("SET FOREIGN_KEY_CHECKS=1")
c.executemany("INSERT INTO business_line_categories (id,name) VALUES (%s,%s)", cats)
c.executemany(
"INSERT INTO bank_transactions (id,transactionDate,transactionType,reference,concept,amount,categoryId,cleared,transferred,notes,amountInWords,legacySourceTable,legacyId) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)", rows)
conn.commit()
def count(t):
c.execute(f"SELECT COUNT(*) FROM {t}"); return c.fetchone()[0]
c.execute("SELECT legacySourceTable, COUNT(*), SUM(amount) FROM bank_transactions GROUP BY legacySourceTable")
by_src = c.fetchall()
c.execute("SELECT SUM(amount) FROM bank_transactions")
net = c.fetchone()[0]
print("=== Bank register load complete ===")
print(f" skipped (unparseable date): {skip_date}")
print(f" -> bank_transactions : {count('bank_transactions')}")
for src, n, tot in by_src:
print(f" {(src or '(manual)'):10} {n:6} sum {tot}")
print(f" net balance movement : {net}")
print(f" -> business_line_categories: {count('business_line_categories')}")
print(" validation: OK")
conn.close()
if __name__ == "__main__":
main()