diff --git a/migration/blob_extract.py b/migration/blob_extract.py index 45dc1e7..5a27c5b 100644 --- a/migration/blob_extract.py +++ b/migration/blob_extract.py @@ -33,7 +33,7 @@ from pathlib import Path import boto3 from botocore.config import Config -from dbenv import connect, load_env +from dbenv import connect, require, setting from extract import sanitize_column_name as san csv.field_size_limit(300_000_000) @@ -102,12 +102,16 @@ def main(): only = set(x.strip() for x in args.tables.split(",") if x.strip()) sources = [s for s in SOURCES if not only or s["key"] in only] - env = load_env(args.env) + # Process environment first, deploy/.env. second, with the same + # credential aliases the API uses — the "Operaciones" re-import runs this + # inside the API container, which has S3_ENDPOINT / MINIO_ROOT_* injected + # and no deploy/ directory at all. s3 = boto3.client( - "s3", endpoint_url=env["S3_ENDPOINT"], - aws_access_key_id=env["MINIO_ROOT_USER"], aws_secret_access_key=env["MINIO_ROOT_PASSWORD"], + "s3", endpoint_url=require(args.env, "S3_ENDPOINT"), + aws_access_key_id=require(args.env, "S3_ACCESS_KEY", "MINIO_ROOT_USER"), + aws_secret_access_key=require(args.env, "S3_SECRET_KEY", "MINIO_ROOT_PASSWORD"), config=Config(signature_version="s3v4"), region_name="us-east-1") - bucket = env["S3_BUCKET"] + bucket = setting(args.env, "S3_BUCKET") or "jorgecuadros-documents" conn = connect(args.env) cur = conn.cursor() diff --git a/migration/dbenv.py b/migration/dbenv.py index 25d34ca..9fcd491 100644 --- a/migration/dbenv.py +++ b/migration/dbenv.py @@ -32,29 +32,53 @@ REPO = Path(__file__).resolve().parents[1] def load_env(env: str) -> dict: + """deploy/.env. parsed to a dict, or {} when the file is absent. + + Absent is normal, not an error: the API container runs these scripts with + DATABASE_URL / S3_* injected as real environment variables and ships no + deploy/ directory. Use `setting()` / `require()` rather than this — they + layer the process environment on top, which is what actually resolves.""" f = REPO / "deploy" / f".env.{env}" if not f.exists(): - raise SystemExit( - f"missing {f} — deploy the '{env}' DB stack and write its .env first " - f"(see dbenv.py header)." - ) + return {} out = {} for line in f.read_text().splitlines(): line = line.strip() if line and not line.startswith("#") and "=" in line: k, v = line.split("=", 1) out[k] = v - if "DATABASE_URL" not in out: - raise SystemExit(f"{f} has no DATABASE_URL") return out +def setting(env: str, *keys: str): + """First non-empty value for `keys`, process environment first, then + deploy/.env.. Several keys = fallback aliases (S3_ACCESS_KEY then + MINIO_ROOT_USER, as apps/api/src/storage/storage.service.ts does).""" + fromfile = load_env(env) + for k in keys: + v = os.environ.get(k) or fromfile.get(k) + if v: + return v + return None + + +def require(env: str, *keys: str) -> str: + v = setting(env, *keys) + if not v: + raise SystemExit( + f"missing {' / '.join(keys)} — set it in the environment, or deploy the " + f"'{env}' stack and write {REPO / 'deploy' / f'.env.{env}'} " + f"(see dbenv.py header)." + ) + return v + + def database_url(env: str) -> str: """Target DB URL. A DATABASE_URL in the process environment wins over deploy/.env. — this is how the API container (which has its own DATABASE_URL and no deploy/.env files) drives a re-import against its own database.""" - return os.environ.get("DATABASE_URL") or load_env(env)["DATABASE_URL"] + return require(env, "DATABASE_URL") def connect(env: str):