Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
458e67340c | ||
|
|
b12382b436 | ||
|
|
2620559975 | ||
|
|
ed19f51a52 | ||
|
|
e85db73dbc | ||
|
|
d38bbc52ec | ||
|
|
fe761e119e | ||
|
|
4a929f7e7c | ||
|
|
66d0d071b0 |
@@ -14,7 +14,16 @@
|
||||
# ARG/ENV (APP_VERSION / GIT_SHA / BUILD_DATE) and as OCI labels, so a running
|
||||
# container can report exactly what is deployed.
|
||||
#
|
||||
# Release flow: git tag v1.2.0 && git push origin v1.2.0 -> versioned images.
|
||||
# Release flow: git tag v1.2.0 && git push origin v1.2.0 -> versioned images
|
||||
# -> deploy-on-tag.yml waits for this run to go green and then
|
||||
# dispatches deploy-galactus.yml.
|
||||
#
|
||||
# This workflow BUILDS ONLY — it never deploys. The deploy chain used to be a
|
||||
# job here, gated to tag refs, but Gitea draws every job of a workflow into the
|
||||
# run graph before it evaluates the job's `if`: a routine master build showed a
|
||||
# pending "Deploy to galactus" and looked like prod was about to be redeployed
|
||||
# off an unreleased commit. Keeping the deploy in a `on: push: tags` workflow of
|
||||
# its own makes that structurally impossible.
|
||||
|
||||
name: Build and Push Images
|
||||
|
||||
|
||||
@@ -0,0 +1,199 @@
|
||||
# Chain the PROD deploy onto a green tag build.
|
||||
#
|
||||
# This is a SEPARATE workflow, not a job in build.yml, and the trigger is the
|
||||
# whole point: `on: push: tags` cannot fire on a push to master. When this was a
|
||||
# `deploy` job inside build.yml gated by `if: startsWith(github.ref,
|
||||
# 'refs/tags/v')`, Gitea still drew "Deploy to galactus" into the job graph of
|
||||
# every ordinary master build — the `if` is not evaluated until `needs` resolve,
|
||||
# so the job sits there looking like an imminent production deploy on a commit
|
||||
# nobody released. That is indistinguishable from a real misfire, and the only
|
||||
# safe reaction is to cancel the run, which kills the images with it.
|
||||
#
|
||||
# What it does NOT do is build. build.yml already builds and pushes both images
|
||||
# from one run; this waits for that run to go green and then dispatches
|
||||
# deploy-galactus.yml, which only pulls.
|
||||
#
|
||||
# Why wait for the build run rather than just dispatching: deploy-galactus.yml
|
||||
# pulls api and web at the same tag, and a half-pushed pair is exactly the state
|
||||
# that leaves prod running one new image and one old one. The build run turning
|
||||
# green is the signal that both are in the registry.
|
||||
#
|
||||
# Why a dispatch and not a `workflow_run:` trigger, which Gitea does support as
|
||||
# of 1.24: deploy-galactus.yml reads `github.event.inputs.*` in ten places (tag,
|
||||
# scope, bootstrap, skip_migrate). Under workflow_run every one of them is the
|
||||
# empty string, so the deploy would silently run with no tag and scope != 'full'.
|
||||
# A dispatch keeps that workflow's contract intact and keeps it hand-runnable for
|
||||
# rollbacks, which is the whole point of it.
|
||||
#
|
||||
# Kill switch: set the repo variable AUTO_DEPLOY_GALACTUS to `false` to cut the
|
||||
# chain and go back to dispatching the deploy by hand. Anything else (including
|
||||
# unset) deploys.
|
||||
|
||||
name: Deploy on tag
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
name: Deploy to galactus
|
||||
runs-on: docker
|
||||
container:
|
||||
image: node:20-alpine
|
||||
steps:
|
||||
- name: Preflight — RELEASE_TOKEN
|
||||
env:
|
||||
RELEASE_TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||
run: |
|
||||
set -eu
|
||||
if [ -z "${RELEASE_TOKEN:-}" ]; then
|
||||
echo "::error::Secret RELEASE_TOKEN is not set, so this cannot wait"
|
||||
echo "::error::for the build or dispatch the deploy. Once build.yml"
|
||||
V=${GITHUB_REF#refs/tags/}
|
||||
echo "::error::is green, run 'Deploy to galactus' by hand with tag=${V#v}."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Wait for the tag build, then dispatch deploy-galactus.yml
|
||||
env:
|
||||
RELEASE_TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||
AUTO_DEPLOY: ${{ vars.AUTO_DEPLOY_GALACTUS }}
|
||||
TAG_REF: ${{ github.ref }}
|
||||
BUILD_SHA: ${{ github.sha }}
|
||||
run: |
|
||||
node -e '
|
||||
const base = `${process.env.GITHUB_SERVER_URL}/api/v1/repos/${process.env.GITHUB_REPOSITORY}`;
|
||||
const headers = { Authorization: `token ${process.env.RELEASE_TOKEN}` };
|
||||
// refs/tags/v1.2.3 — derived from github.ref rather than ref_name so
|
||||
// it does not depend on how Gitea populates GITHUB_REF_NAME.
|
||||
const tagRef = process.env.TAG_REF;
|
||||
const tag = tagRef.replace(/^refs\/tags\//, "");
|
||||
// The git tag carries the leading v; the image tag does not.
|
||||
const version = tag.replace(/^v/, "");
|
||||
const sha = process.env.BUILD_SHA;
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
const runs = async () => {
|
||||
const r = await fetch(`${base}/actions/runs?limit=50`, { headers });
|
||||
if (!r.ok) throw new Error(`runs query failed: HTTP ${r.status}`);
|
||||
return (await r.json()).workflow_runs || [];
|
||||
};
|
||||
|
||||
// The release commit and its tag are the SAME sha, and build.yml
|
||||
// skips the master run by design — so a sha match alone can latch
|
||||
// onto that skipped run and call the build green when no image was
|
||||
// ever pushed. Require the tag ref when the API reports one.
|
||||
const isTagRun = (r) => {
|
||||
const ref = r.head_branch || r.ref || "";
|
||||
return !ref || ref === tag || ref === tagRef;
|
||||
};
|
||||
|
||||
const buildRun = async () =>
|
||||
(await runs()).find(
|
||||
(r) =>
|
||||
r.head_sha === sha &&
|
||||
String(r.path || "").includes("build.yml") &&
|
||||
isTagRun(r),
|
||||
);
|
||||
|
||||
// Gitea reports a run as `status` and mirrors it into `conclusion`;
|
||||
// read whichever is populated rather than betting on one field.
|
||||
const outcome = (r) => String(r.conclusion || r.status || "").toLowerCase();
|
||||
const DONE = ["success", "failure", "cancelled", "canceled", "skipped"];
|
||||
|
||||
// Every deploy-galactus run id visible right now. A dispatch is only
|
||||
// confirmed by an id that is NOT in here — a plain "is there a deploy
|
||||
// run" check is satisfied by the PREVIOUS release run, and would
|
||||
// report success for a dispatch that never took.
|
||||
const deployRunIds = async () =>
|
||||
new Set(
|
||||
(await runs())
|
||||
.filter((r) => String(r.path || "").includes("deploy-galactus.yml"))
|
||||
.map((r) => r.id),
|
||||
);
|
||||
|
||||
(async () => {
|
||||
if (process.env.AUTO_DEPLOY === "false") {
|
||||
console.log("AUTO_DEPLOY_GALACTUS=false — not deploying.");
|
||||
console.log(`Deploy by hand with tag=${version} when ready.`);
|
||||
return;
|
||||
}
|
||||
|
||||
// ~20 min. A build is about 90s; the rest is queue time behind
|
||||
// other runs on a single runner.
|
||||
let run = null;
|
||||
for (let i = 0; i < 80; i++) {
|
||||
run = await buildRun();
|
||||
if (run && DONE.includes(outcome(run))) break;
|
||||
if (!run && i === 11) {
|
||||
// Two minutes with no run at all. The post-receive hook drops
|
||||
// runs silently when it errors (this cost v1.0.3 its images),
|
||||
// so say so rather than timing out with no explanation.
|
||||
console.log(`::warning::No build.yml run for ${tag} yet after 2 min.`);
|
||||
console.log(`::warning::If the Gitea post-receive hook is broken, dispatch`);
|
||||
console.log(`::warning::"Build and Push Images" by hand with ref=${tag}.`);
|
||||
}
|
||||
await sleep(15_000);
|
||||
}
|
||||
|
||||
if (!run) {
|
||||
console.log(`::error::No build.yml run for ${tag} (${sha}) after 20 min.`);
|
||||
console.log(`::error::Dispatch "Build and Push Images" with ref=${tag} (the`);
|
||||
console.log(`::error::tag, not master), then deploy by hand with tag=${version}.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const result = outcome(run);
|
||||
if (result !== "success") {
|
||||
console.log(`::error::build.yml for ${tag} ended as "${result}" — not deploying.`);
|
||||
console.log(`::error::Fix the build, re-run it, then deploy by hand with tag=${version}.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`build.yml for ${tag} is green (run ${run.id}). Deploying ${version}.`);
|
||||
const before = await deployRunIds();
|
||||
|
||||
// Dispatch against the TAG, not master: the deploy applies the
|
||||
// compose files under deploy/galactus/ from whatever ref it runs
|
||||
// on, and those must be the ones this release was cut with.
|
||||
const res = await fetch(
|
||||
`${base}/actions/workflows/deploy-galactus.yml/dispatches`,
|
||||
{
|
||||
method: "POST",
|
||||
headers: { ...headers, "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
ref: tagRef,
|
||||
inputs: {
|
||||
tag: version,
|
||||
scope: "app",
|
||||
bootstrap: "false",
|
||||
skip_migrate: "false",
|
||||
},
|
||||
}),
|
||||
},
|
||||
);
|
||||
if (!res.ok) {
|
||||
console.log(`::error::Dispatch returned HTTP ${res.status}: ${await res.text()}`);
|
||||
console.log(`::error::Images for ${version} are published. Run`);
|
||||
console.log(`::error::"Deploy to galactus" by hand with tag=${version}.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// A 204 only means Gitea accepted the request. Confirm a NEW run
|
||||
// exists — an accepted call that creates no run is the failure mode
|
||||
// that cost v1.0.3 its images.
|
||||
for (let i = 0; i < 3; i++) {
|
||||
await sleep(5_000);
|
||||
const fresh = [...(await deployRunIds())].filter((id) => !before.has(id));
|
||||
if (fresh.length) {
|
||||
console.log(`Deploy of ${version} to galactus is running (run ${fresh[0]}).`);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`::error::Dispatch was accepted but no deploy run appeared.`);
|
||||
console.log(`::error::Run "Deploy to galactus" by hand with tag=${version}.`);
|
||||
process.exit(1);
|
||||
})();
|
||||
'
|
||||
@@ -1,10 +1,16 @@
|
||||
# Cut a release: stamp the version across every package.json, commit, tag, push.
|
||||
#
|
||||
# This does NOT build and does NOT deploy. Pushing the `vX.Y.Z` tag is what
|
||||
# triggers build.yml, which publishes `X.Y.Z`, `X.Y`, `sha-<short>` and `latest`
|
||||
# image tags. Deploying stays a separate, deliberate act: once the build is
|
||||
# green, dispatch deploy-galactus.yml with `tag=X.Y.Z` (no leading v — the tag
|
||||
# carries the `v`, the image tag does not).
|
||||
# This does NOT build and does NOT deploy itself. Pushing the `vX.Y.Z` tag is
|
||||
# what triggers both build.yml, which publishes the `X.Y.Z`, `X.Y`,
|
||||
# `sha-<short>` and `latest` image tags, and deploy-on-tag.yml, which waits for
|
||||
# that build to go green and then dispatches deploy-galactus.yml with
|
||||
# `tag=X.Y.Z scope=app` (no leading v — the git tag carries the `v`, the image
|
||||
# tag does not). A tag is the only ref that starts either chain; pushing to
|
||||
# master builds images and stops there.
|
||||
#
|
||||
# So cutting a release DOES reach prod. To cut a version without deploying it,
|
||||
# set the repo variable AUTO_DEPLOY_GALACTUS=false first; deploy-on-tag.yml then
|
||||
# prints the manual command instead of running it.
|
||||
#
|
||||
# Why a workflow instead of three local commands: the release commit is the one
|
||||
# thing that must be identical every time, and cutting it from a laptop is how
|
||||
@@ -266,5 +272,9 @@ jobs:
|
||||
echo "Released v${VERSION}."
|
||||
echo ""
|
||||
echo "build.yml is now building git.mancinas.io/rmancinas/jorgecuadros-{api,web}:${VERSION}."
|
||||
echo "When it is green, dispatch 'Deploy to galactus' with:"
|
||||
echo " tag=${VERSION} scope=app bootstrap=false skip_migrate=false"
|
||||
echo "deploy-on-tag.yml is watching that build; when it goes green it dispatches"
|
||||
echo "'Deploy to galactus' with tag=${VERSION} scope=app bootstrap=false skip_migrate=false."
|
||||
echo ""
|
||||
echo "Watch that run. If it did not start (or AUTO_DEPLOY_GALACTUS=false),"
|
||||
echo "dispatch 'Deploy to galactus' by hand with the same inputs."
|
||||
echo "Rollback = re-dispatch it with an older tag."
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/api",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.14",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"build": "nest build",
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
import { jobProgress } from "./ops.service";
|
||||
|
||||
/** Shape run_all.py emits, with the shell trace lines it interleaves. */
|
||||
const line = (i: number, n: number, name: string) =>
|
||||
`[paso ${i}/${n}] ${name}\n+ /repo/migration/.venv/bin/python /repo/migration/${name} --env prod\n[${name}] target env: prod\n validation: OK`;
|
||||
|
||||
describe("jobProgress", () => {
|
||||
it("returns null before any step marker appears", () => {
|
||||
// The safety backup runs before run_all.py, so this is the real state for
|
||||
// the first stretch of every REIMPORT.
|
||||
expect(jobProgress("== Respaldo de seguridad previo ==\ntablas capturadas: 39", "RUNNING")).toBeNull();
|
||||
});
|
||||
|
||||
it("returns null for jobs that have no steps at all", () => {
|
||||
// BACKUP/RESTORE are a single mysqldump; a fabricated percentage would be
|
||||
// worse than none.
|
||||
expect(jobProgress("mysqldump ... done", "SUCCESS")).toBeNull();
|
||||
});
|
||||
|
||||
it("tracks the most recent marker, not the first", () => {
|
||||
const log = [line(1, 9, "transform_customers.py"), line(2, 9, "transform_properties.py")].join("\n");
|
||||
const p = jobProgress(log, "RUNNING");
|
||||
expect(p).toMatchObject({ step: 2, total: 9, name: "transform_properties.py" });
|
||||
});
|
||||
|
||||
/**
|
||||
* The point of the whole feature. While RUNNING, step i is IN PROGRESS, so
|
||||
* only i-1 are done. Counting i as complete would show 100% while the final
|
||||
* and slowest step (blob_extract) is still working.
|
||||
*/
|
||||
it("does not claim a running step is finished", () => {
|
||||
expect(jobProgress(line(1, 9, "transform_customers.py"), "RUNNING")?.percent).toBe(0);
|
||||
expect(jobProgress(line(9, 9, "blob_extract.py"), "RUNNING")?.percent).toBe(88);
|
||||
});
|
||||
|
||||
it("reaches 100 only once the job is no longer running", () => {
|
||||
expect(jobProgress(line(9, 9, "blob_extract.py"), "SUCCESS")?.percent).toBe(100);
|
||||
});
|
||||
|
||||
/** A job that died mid-way must report where it died, not 100%. */
|
||||
it("reports the failed step rather than completion", () => {
|
||||
const p = jobProgress(line(5, 9, "transform_transactions.py"), "FAILED");
|
||||
expect(p).toMatchObject({ step: 5, total: 9 });
|
||||
expect(p!.percent).toBe(55);
|
||||
});
|
||||
|
||||
it("handles the 8-step SYNC list as well as the 9-step REIMPORT one", () => {
|
||||
expect(jobProgress(line(8, 8, "transform_bank.py"), "SUCCESS")?.percent).toBe(100);
|
||||
expect(jobProgress(line(4, 8, "transform_policies.py"), "RUNNING")?.percent).toBe(37);
|
||||
});
|
||||
|
||||
/**
|
||||
* Captured verbatim from `run_all.run(..., step=8, total=9)`. This is the
|
||||
* contract between the Python and this parser; if run_all.py's format
|
||||
* changes, this fails rather than the panel silently showing no progress.
|
||||
*/
|
||||
it("parses the exact line run_all.py emits", () => {
|
||||
const real =
|
||||
"[paso 8/9] transform_bank.py\n+ /repo/migration/.venv/bin/python /repo/migration/transform_bank.py --env prod";
|
||||
expect(jobProgress(real, "RUNNING")).toMatchObject({
|
||||
step: 8,
|
||||
total: 9,
|
||||
name: "transform_bank.py",
|
||||
percent: 77,
|
||||
});
|
||||
});
|
||||
|
||||
it("ignores a malformed marker instead of reporting NaN", () => {
|
||||
expect(jobProgress("[paso 3/0] x.py", "RUNNING")).toBeNull();
|
||||
});
|
||||
|
||||
/** The marker must be at line start so log text quoting it cannot spoof it. */
|
||||
it("does not match a marker embedded mid-line", () => {
|
||||
expect(jobProgress("some output mentioning [paso 4/9] fake.py", "RUNNING")).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -212,7 +212,9 @@ export class OpsService implements OnModuleInit {
|
||||
async getJob(id: string) {
|
||||
const job = await this.prisma.opsJob.findUnique({ where: { id } });
|
||||
if (!job) throw new NotFoundException("Trabajo no encontrado.");
|
||||
return job;
|
||||
// Derived, never stored: the log is the single source of truth for how far
|
||||
// a job got, so progress cannot drift out of sync with it.
|
||||
return { ...job, progress: jobProgress(job.log, job.status) };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -398,7 +400,12 @@ export class OpsService implements OnModuleInit {
|
||||
`${PIPEFAIL}echo '== Respaldo de seguridad previo ==' && ` +
|
||||
`${this.dumpCommand(flags, db, out)} && ` +
|
||||
`echo '== Sincronización aditiva desde carpeta de ingesta ==' && ` +
|
||||
`${shq(py)} ${runAll} --env ${shq(this.migrationEnv)} --sync`;
|
||||
// --stage is not optional here. The staged Parquet lives in the image
|
||||
// at migration/output, NOT on a volume, so every redeploy wipes it and
|
||||
// a sync without --stage dies on a missing stg_*/*.parquet. Re-staging
|
||||
// is also the only thing that makes "desde carpeta de ingesta" true:
|
||||
// stale Parquet would sync the previous upload, not the current one.
|
||||
`${shq(py)} ${runAll} --env ${shq(this.migrationEnv)} --stage --sync`;
|
||||
return { cmd, resolvedParams: { safetyBackup: file } };
|
||||
}
|
||||
|
||||
@@ -514,3 +521,52 @@ export class OpsService implements OnModuleInit {
|
||||
function shq(v: string): string {
|
||||
return `'${v.replace(/'/g, `'\\''`)}'`;
|
||||
}
|
||||
|
||||
/** Progress derived from a job's log. Null when the job reports no steps. */
|
||||
export interface JobProgress {
|
||||
/** 1-based index of the step currently running (or last reached). */
|
||||
step: number;
|
||||
total: number;
|
||||
/** Script name, e.g. "transform_bank.py". */
|
||||
name: string;
|
||||
/** 0..100, floored. 100 only once the job is no longer RUNNING. */
|
||||
percent: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the "[paso i/N] name" markers migration/run_all.py emits.
|
||||
*
|
||||
* Progress is DERIVED from the log rather than tracked in a column: the log is
|
||||
* already the record of what happened, and a separate counter could disagree
|
||||
* with it — which is exactly the confusion a progress display is supposed to
|
||||
* remove. run_all.py owns the step count, so adding a step cannot desync this.
|
||||
*
|
||||
* BACKUP and RESTORE are a single mysqldump with no steps, so they return null
|
||||
* and the UI shows an indeterminate spinner. Reporting a fabricated percentage
|
||||
* for them would be worse than showing none.
|
||||
*/
|
||||
export function jobProgress(
|
||||
log: string,
|
||||
status: string,
|
||||
): JobProgress | null {
|
||||
// Last marker wins: the log grows, and the newest line is the current step.
|
||||
const matches = [...log.matchAll(/^\[paso (\d+)\/(\d+)\] (\S+)/gm)];
|
||||
const last = matches[matches.length - 1];
|
||||
if (!last) return null;
|
||||
|
||||
const step = Number(last[1]);
|
||||
const total = Number(last[2]);
|
||||
if (!Number.isFinite(step) || !Number.isFinite(total) || total <= 0) return null;
|
||||
|
||||
// While RUNNING, step i means i is IN PROGRESS, not finished — so report
|
||||
// (i-1) completed. Claiming 100% while the last step is still working is the
|
||||
// classic progress-bar lie, and here the last step (blob_extract) is also the
|
||||
// slowest, so it would sit at "100%" for the longest stretch of the job.
|
||||
const done = status === "RUNNING" ? step - 1 : step;
|
||||
return {
|
||||
step,
|
||||
total,
|
||||
name: last[3],
|
||||
percent: Math.max(0, Math.min(100, Math.floor((done / total) * 100))),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -4,6 +4,45 @@ import { promisify } from "node:util";
|
||||
|
||||
const exec = promisify(execFile);
|
||||
|
||||
/**
|
||||
* How far the SQL thread is behind the I/O thread, in source binlog bytes.
|
||||
*
|
||||
* This is a different question from `secondsBehind`, and it answers the case
|
||||
* that lag hides: while the SQL thread grinds through one huge transaction,
|
||||
* `Seconds_Behind_Source` can sit still or even read 0, but the relay backlog
|
||||
* is plainly shrinking (or not). It costs nothing extra — every field here
|
||||
* comes out of the same `SHOW REPLICA STATUS` the panel already runs.
|
||||
*
|
||||
* Both positions are coordinates in the SOURCE's binlog, so they are only
|
||||
* comparable while both threads are working on the SAME source file. When they
|
||||
* are not, the replica is whole files behind and the byte delta is meaningless
|
||||
* (positions restart at ~4 in each new file), so `backlogBytes` and `percent`
|
||||
* are null and `sameFile` says why.
|
||||
*/
|
||||
export interface ApplyProgress {
|
||||
/** Source binlog file the I/O thread is currently reading. */
|
||||
sourceLogFile: string | null;
|
||||
/** Position in `sourceLogFile` that the I/O thread has fetched up to. */
|
||||
readPos: number;
|
||||
/** Source binlog file the SQL thread is currently applying. */
|
||||
relayLogFile: string | null;
|
||||
/** Position in `relayLogFile` that the SQL thread has applied up to. */
|
||||
execPos: number;
|
||||
/** True while both threads are on the same source file. */
|
||||
sameFile: boolean;
|
||||
/** Fetched-but-not-yet-applied bytes. Null when the files differ. */
|
||||
backlogBytes: number | null;
|
||||
/**
|
||||
* `execPos / readPos` as a percentage, null when the files differ.
|
||||
*
|
||||
* Deliberately never rounded up to 100 while any backlog remains: binlog
|
||||
* positions are large, so a real backlog of a few KB is 99.99% of the file
|
||||
* and would render as "caught up" when it is not. Read `backlogBytes === 0`
|
||||
* for actually caught up.
|
||||
*/
|
||||
percent: number | null;
|
||||
}
|
||||
|
||||
export interface ReplicationStatus {
|
||||
/** false when the replica is not configured for this environment at all. */
|
||||
configured: boolean;
|
||||
@@ -17,6 +56,8 @@ export interface ReplicationStatus {
|
||||
lastIoError: string | null;
|
||||
lastSqlError: string | null;
|
||||
sourceHost: string | null;
|
||||
/** Relay-log apply progress. Null when the status output has no positions. */
|
||||
apply: ApplyProgress | null;
|
||||
/** Human-readable reason when healthy is false. */
|
||||
problem: string | null;
|
||||
checkedAt: string;
|
||||
@@ -57,6 +98,7 @@ export class ReplicationService {
|
||||
lastIoError: null,
|
||||
lastSqlError: null,
|
||||
sourceHost: null,
|
||||
apply: null,
|
||||
problem: null,
|
||||
checkedAt: now,
|
||||
};
|
||||
@@ -142,12 +184,60 @@ export class ReplicationService {
|
||||
lastIoError,
|
||||
lastSqlError,
|
||||
sourceHost: field("Source_Host"),
|
||||
// Reported, never folded into `healthy`: a non-zero backlog is the normal
|
||||
// state of a working replica for the instant between fetch and apply, so
|
||||
// alarming on it would cry wolf. It is here to answer "is it moving?"
|
||||
// when the lag counter is stuck.
|
||||
apply: applyProgress(raw),
|
||||
problem,
|
||||
checkedAt: now,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Derive relay-apply progress from `SHOW REPLICA STATUS\G` output.
|
||||
*
|
||||
* Exported for testing. Free in query terms — it re-reads four more fields from
|
||||
* the output the caller already has, with no second round trip to the replica
|
||||
* and no connection to the source.
|
||||
*
|
||||
* @returns null when either position is missing or unparseable, which is what
|
||||
* happens on a server that is not a replica at all.
|
||||
*/
|
||||
export function applyProgress(raw: string): ApplyProgress | null {
|
||||
const num = (name: string): number | null => {
|
||||
const v = replicaField(raw, name);
|
||||
if (v === null || v === "NULL") return null;
|
||||
const n = Number(v);
|
||||
return Number.isFinite(n) ? n : null;
|
||||
};
|
||||
|
||||
const readPos = num("Read_Source_Log_Pos");
|
||||
const execPos = num("Exec_Source_Log_Pos");
|
||||
if (readPos === null || execPos === null) return null;
|
||||
|
||||
const sourceLogFile = replicaField(raw, "Source_Log_File");
|
||||
const relayLogFile = replicaField(raw, "Relay_Source_Log_File");
|
||||
const sameFile =
|
||||
sourceLogFile !== null && relayLogFile !== null && sourceLogFile === relayLogFile;
|
||||
|
||||
// Clamped at 0: the SQL thread cannot be ahead of the I/O thread, but the two
|
||||
// fields are sampled independently, so a rotation racing this read can print
|
||||
// a momentarily negative delta. Zero is the honest floor, not a bug.
|
||||
const backlogBytes = sameFile ? Math.max(0, readPos - execPos) : null;
|
||||
|
||||
let percent: number | null = null;
|
||||
if (backlogBytes !== null && readPos > 0) {
|
||||
// Truncate rather than round, and hold short of 100 while bytes remain —
|
||||
// see the doc on ApplyProgress.percent.
|
||||
const p = Math.floor((execPos / readPos) * 10_000) / 100;
|
||||
percent = backlogBytes === 0 ? 100 : Math.min(p, 99.99);
|
||||
}
|
||||
|
||||
return { sourceLogFile, readPos, relayLogFile, execPos, sameFile, backlogBytes, percent };
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one field out of `SHOW REPLICA STATUS\G` output.
|
||||
*
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { replicaField } from "./replication.service";
|
||||
import { applyProgress, replicaField } from "./replication.service";
|
||||
|
||||
/**
|
||||
* Verbatim shape of `SHOW REPLICA STATUS\G` from the live replica, trimmed to
|
||||
@@ -13,6 +13,10 @@ const HEALTHY = [
|
||||
" Replica_IO_State: Waiting for source to send event",
|
||||
" Source_Host: 100.103.77.46",
|
||||
" Source_User: repl",
|
||||
" Source_Log_File: binlog.000042",
|
||||
" Read_Source_Log_Pos: 194884231",
|
||||
" Relay_Source_Log_File: binlog.000042",
|
||||
" Exec_Source_Log_Pos: 194884231",
|
||||
" Replica_IO_Running: Yes",
|
||||
" Replica_SQL_Running: Yes",
|
||||
" Replicate_Do_DB: ",
|
||||
@@ -85,3 +89,92 @@ describe("replicaField", () => {
|
||||
expect(replicaField(raw, "Last_SQL_Error")).toBe("boom");
|
||||
});
|
||||
});
|
||||
|
||||
/** Builds the four position fields the apply-progress reader cares about. */
|
||||
function positions(
|
||||
sourceFile: string,
|
||||
readPos: number | string,
|
||||
relayFile: string,
|
||||
execPos: number | string,
|
||||
): string {
|
||||
return [
|
||||
` Source_Log_File: ${sourceFile}`,
|
||||
` Read_Source_Log_Pos: ${readPos}`,
|
||||
` Relay_Source_Log_File: ${relayFile}`,
|
||||
` Exec_Source_Log_Pos: ${execPos}`,
|
||||
].join("\n");
|
||||
}
|
||||
|
||||
describe("applyProgress", () => {
|
||||
it("reports zero backlog and 100% when both positions match", () => {
|
||||
const p = applyProgress(HEALTHY)!;
|
||||
expect(p.sameFile).toBe(true);
|
||||
expect(p.sourceLogFile).toBe("binlog.000042");
|
||||
expect(p.readPos).toBe(194884231);
|
||||
expect(p.execPos).toBe(194884231);
|
||||
expect(p.backlogBytes).toBe(0);
|
||||
expect(p.percent).toBe(100);
|
||||
});
|
||||
|
||||
it("reports the byte delta when the SQL thread trails inside one file", () => {
|
||||
const p = applyProgress(positions("binlog.000042", 2_000_000, "binlog.000042", 1_500_000))!;
|
||||
expect(p.backlogBytes).toBe(500_000);
|
||||
expect(p.percent).toBe(75);
|
||||
});
|
||||
|
||||
/**
|
||||
* The reason the byte delta exists at all. `Seconds_Behind_Source` holds at 0
|
||||
* while the SQL thread is mid-transaction, so the backlog is the only field
|
||||
* that moves — and the only one that says the replica is not caught up.
|
||||
*/
|
||||
it("shows a backlog even when the lag counter reads zero", () => {
|
||||
const raw = [
|
||||
" Seconds_Behind_Source: 0",
|
||||
positions("binlog.000042", 900, "binlog.000042", 400),
|
||||
].join("\n");
|
||||
expect(replicaField(raw, "Seconds_Behind_Source")).toBe("0");
|
||||
expect(applyProgress(raw)!.backlogBytes).toBe(500);
|
||||
});
|
||||
|
||||
/**
|
||||
* Positions restart near 4 in every new binlog file, so subtracting across
|
||||
* files produces a number that is not a backlog — here it would be a large
|
||||
* NEGATIVE one, which would render as "ahead of the source".
|
||||
*/
|
||||
it("refuses to compare positions across different binlog files", () => {
|
||||
const p = applyProgress(positions("binlog.000043", 500, "binlog.000042", 194_000_000))!;
|
||||
expect(p.sameFile).toBe(false);
|
||||
expect(p.backlogBytes).toBeNull();
|
||||
expect(p.percent).toBeNull();
|
||||
expect(p.sourceLogFile).toBe("binlog.000043");
|
||||
expect(p.relayLogFile).toBe("binlog.000042");
|
||||
});
|
||||
|
||||
/**
|
||||
* Percent must not round up to 100 while bytes remain: binlog positions are
|
||||
* large, so a genuine backlog is a rounding error away from the whole file
|
||||
* and would otherwise render as "caught up" on a replica that is not.
|
||||
*/
|
||||
it("stops short of 100% while any backlog remains", () => {
|
||||
const p = applyProgress(positions("binlog.000042", 194_884_231, "binlog.000042", 194_884_230))!;
|
||||
expect(p.backlogBytes).toBe(1);
|
||||
expect(p.percent).toBe(99.99);
|
||||
});
|
||||
|
||||
/** Sampled independently, so a rotation racing the read can invert them. */
|
||||
it("clamps a momentarily negative delta to zero", () => {
|
||||
const p = applyProgress(positions("binlog.000042", 400, "binlog.000042", 500))!;
|
||||
expect(p.backlogBytes).toBe(0);
|
||||
expect(p.percent).toBe(100);
|
||||
});
|
||||
|
||||
it("returns null when the server is not a replica and prints no positions", () => {
|
||||
expect(applyProgress("")).toBeNull();
|
||||
expect(applyProgress(BROKEN)).toBeNull();
|
||||
});
|
||||
|
||||
/** A stopped thread makes MySQL print NULL, which is not a position. */
|
||||
it("returns null when a position is NULL", () => {
|
||||
expect(applyProgress(positions("binlog.000042", "NULL", "binlog.000042", 400))).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/web",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.14",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"dev": "next dev -p 4500",
|
||||
|
||||
@@ -23,6 +23,7 @@ import {
|
||||
} from "@/lib/api";
|
||||
import type { UploadProgress } from "@/lib/api";
|
||||
import type {
|
||||
ApplyProgress,
|
||||
BackupFile,
|
||||
IngestFile,
|
||||
OpsJob,
|
||||
@@ -225,6 +226,7 @@ function Operaciones() {
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
<JobProgressBar job={activeJob} />
|
||||
<pre className="ops-log">{activeJob.log || "Iniciando…"}</pre>
|
||||
</div>
|
||||
)}
|
||||
@@ -684,8 +686,58 @@ function ReplicationCard() {
|
||||
label="Retraso"
|
||||
value={status.secondsBehind === null ? "sin dato" : `${status.secondsBehind} s`}
|
||||
/>
|
||||
<KV label="Pendiente de aplicar" value={backlogLabel(status.apply)} />
|
||||
<KV label="Consultado" value={formatDateTime(status.checkedAt)} />
|
||||
</div>
|
||||
|
||||
<ApplyProgressBar apply={status.apply} />
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bytes the replica has fetched but not yet applied.
|
||||
*
|
||||
* Kept separate from the lag figure because it answers a question the lag
|
||||
* cannot: while the SQL thread chews through one big transaction, the seconds
|
||||
* counter can hold still, but this number visibly falls.
|
||||
*/
|
||||
function backlogLabel(apply: ApplyProgress | null): string {
|
||||
if (!apply) return "sin dato";
|
||||
// Different source binlog files means the replica is whole files behind and
|
||||
// the byte delta is not a delta at all — positions restart in each new file.
|
||||
if (!apply.sameFile) return "más de un archivo de binlog";
|
||||
if (apply.backlogBytes === 0) return "al día";
|
||||
return formatBytes(apply.backlogBytes);
|
||||
}
|
||||
|
||||
/**
|
||||
* Applied-vs-fetched bar. Rendered only when both threads are on the same
|
||||
* source binlog file, because that is the only case where the percentage is
|
||||
* arithmetic rather than a guess.
|
||||
*/
|
||||
function ApplyProgressBar({ apply }: { apply: ApplyProgress | null }) {
|
||||
if (!apply || !apply.sameFile || apply.percent === null) return null;
|
||||
|
||||
return (
|
||||
<div className="upload-progress" style={{ marginTop: 12 }}>
|
||||
<div
|
||||
className="progress-track"
|
||||
role="progressbar"
|
||||
aria-valuenow={apply.percent}
|
||||
aria-valuemin={0}
|
||||
aria-valuemax={100}
|
||||
aria-label="Eventos aplicados de los recibidos"
|
||||
>
|
||||
<div className="progress-fill" style={{ width: `${apply.percent}%` }} />
|
||||
</div>
|
||||
<div className="upload-progress-stats mono">
|
||||
<span>{apply.percent}% aplicado</span>
|
||||
<span>
|
||||
{apply.sourceLogFile} · {apply.execPos.toLocaleString("es-MX")} /{" "}
|
||||
{apply.readPos.toLocaleString("es-MX")}
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -699,3 +751,70 @@ function KV({ label, value }: { label: string; value: string | null | undefined
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Step progress for a running migration.
|
||||
*
|
||||
* Only REIMPORT and SYNC report steps; BACKUP and RESTORE are a single
|
||||
* mysqldump, so they render nothing here rather than a made-up bar — the
|
||||
* spinner in the heading already says "working".
|
||||
*
|
||||
* The safety backup runs before the migration, so `progress` is null for the
|
||||
* first stretch of every REIMPORT. That phase is named explicitly instead of
|
||||
* showing 0%, which would read as "stuck".
|
||||
*/
|
||||
function JobProgressBar({ job }: { job: OpsJob }) {
|
||||
const running = job.status === "RUNNING";
|
||||
const p = job.progress;
|
||||
|
||||
if (!p) {
|
||||
if (!running) return null;
|
||||
return (
|
||||
<p className="inline-form-note" style={{ marginTop: 8 }}>
|
||||
Respaldo de seguridad previo…
|
||||
</p>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div style={{ marginTop: 10, marginBottom: 4 }}>
|
||||
<div
|
||||
className="row-actions"
|
||||
style={{ justifyContent: "space-between", marginBottom: 6 }}
|
||||
>
|
||||
<span className="inline-form-note" style={{ margin: 0 }}>
|
||||
Paso {p.step} de {p.total} — {p.name}
|
||||
</span>
|
||||
<span className="inline-form-note" style={{ margin: 0 }}>
|
||||
{p.percent}%
|
||||
</span>
|
||||
</div>
|
||||
<div
|
||||
role="progressbar"
|
||||
aria-valuenow={p.percent}
|
||||
aria-valuemin={0}
|
||||
aria-valuemax={100}
|
||||
aria-label={`Paso ${p.step} de ${p.total}`}
|
||||
style={{
|
||||
height: 6,
|
||||
borderRadius: 999,
|
||||
background: "var(--line)",
|
||||
overflow: "hidden",
|
||||
}}
|
||||
>
|
||||
<div
|
||||
style={{
|
||||
width: `${p.percent}%`,
|
||||
height: "100%",
|
||||
borderRadius: 999,
|
||||
transition: "width 400ms ease",
|
||||
background:
|
||||
job.status === "FAILED"
|
||||
? "var(--negative)"
|
||||
: "var(--positive)",
|
||||
}}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -61,6 +61,14 @@ export interface UserRow {
|
||||
export type OpsJobKind = "BACKUP" | "RESTORE" | "REIMPORT" | "SYNC";
|
||||
export type OpsJobStatus = "RUNNING" | "SUCCESS" | "FAILED";
|
||||
|
||||
/** Derived from the job log by the API; null for jobs with no step markers. */
|
||||
export interface JobProgress {
|
||||
step: number;
|
||||
total: number;
|
||||
name: string;
|
||||
percent: number;
|
||||
}
|
||||
|
||||
export interface OpsJob {
|
||||
id: string;
|
||||
kind: OpsJobKind;
|
||||
@@ -70,6 +78,8 @@ export interface OpsJob {
|
||||
createdById: string | null;
|
||||
startedAt: string;
|
||||
finishedAt: string | null;
|
||||
/** Only present on getOpsJob (the polled endpoint), not on the list. */
|
||||
progress?: JobProgress | null;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -89,10 +99,32 @@ export interface ReplicationStatus {
|
||||
lastIoError: string | null;
|
||||
lastSqlError: string | null;
|
||||
sourceHost: string | null;
|
||||
apply: ApplyProgress | null;
|
||||
problem: string | null;
|
||||
checkedAt: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Relay-log apply progress, in source binlog bytes.
|
||||
*
|
||||
* Answers "is it moving?" when `secondsBehind` cannot: the lag counter sits
|
||||
* still while the SQL thread works through one large transaction, but the
|
||||
* backlog visibly shrinks. `backlogBytes === 0` is the only reading that means
|
||||
* caught up — `percent` deliberately stops at 99.99 while bytes remain.
|
||||
*
|
||||
* Null fields when the two threads are on different source binlog files
|
||||
* (`sameFile === false`), because the positions are then not comparable.
|
||||
*/
|
||||
export interface ApplyProgress {
|
||||
sourceLogFile: string | null;
|
||||
readPos: number;
|
||||
relayLogFile: string | null;
|
||||
execPos: number;
|
||||
sameFile: boolean;
|
||||
backlogBytes: number | null;
|
||||
percent: number | null;
|
||||
}
|
||||
|
||||
/** One of the four legacy Access files expected in the ingest folder. */
|
||||
export interface IngestFile {
|
||||
name: string;
|
||||
|
||||
+24
-5
@@ -15,6 +15,11 @@ Then:
|
||||
./.venv/bin/python run_all.py --env dev # data only (staging already present)
|
||||
./.venv/bin/python run_all.py --env prod --stage # re-extract from Access first, then load
|
||||
|
||||
--sync swaps the truncate+rebuild steps for the additive upsert ones. It reads
|
||||
the same staged Parquet, so it needs --stage too unless a previous run left
|
||||
migration/output populated on this machine — which is never true in a
|
||||
container, where that directory is part of the image and dies with it.
|
||||
|
||||
Reproducing dev -> prod is exactly `--env prod` (plus --stage if the staged
|
||||
Parquet isn't present on the machine running it).
|
||||
|
||||
@@ -78,7 +83,13 @@ SYNC_STEPS = [
|
||||
]
|
||||
|
||||
|
||||
def run(cmd: list[str]) -> None:
|
||||
def run(cmd: list[str], step: int | None = None, total: int | None = None) -> None:
|
||||
# The "[paso i/N] name" marker is a contract with the Operaciones screen,
|
||||
# which parses the last one to show progress. Emitting it here rather than
|
||||
# letting the UI count STEPS itself keeps the two from drifting when a step
|
||||
# is added — the number of steps is only ever stated in this file.
|
||||
if step is not None and total is not None:
|
||||
print(f"[paso {step}/{total}] {Path(cmd[1]).name}", flush=True)
|
||||
print("+ " + " ".join(cmd), flush=True)
|
||||
r = subprocess.run(cmd)
|
||||
if r.returncode:
|
||||
@@ -94,14 +105,22 @@ def main() -> None:
|
||||
help="upsert legacy rows and archive removed legacy rows; preserve manual rows")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.stage:
|
||||
run([PY, str(HERE / "load_staging.py"), "--output-dir", str(HERE / "output")])
|
||||
steps = SYNC_STEPS if args.sync else STEPS
|
||||
# Staging counts as a step when it runs: it is the slowest part of the pass
|
||||
# (mdbtools re-reads every Access file), so leaving it outside the numbering
|
||||
# would park the Operaciones progress bar at "nothing yet" for minutes.
|
||||
total = len(steps) + (1 if args.stage else 0)
|
||||
offset = 1 if args.stage else 0
|
||||
|
||||
for step in SYNC_STEPS if args.sync else STEPS:
|
||||
if args.stage:
|
||||
run([PY, str(HERE / "load_staging.py"), "--output-dir", str(HERE / "output")],
|
||||
step=1, total=total)
|
||||
|
||||
for i, step in enumerate(steps, start=1 + offset):
|
||||
cmd = [PY, str(HERE / step), "--env", args.env]
|
||||
if args.sync:
|
||||
cmd.append("--sync")
|
||||
run(cmd)
|
||||
run(cmd, step=i, total=total)
|
||||
|
||||
print(f"\n✓ migration complete for env={args.env}")
|
||||
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "jorgecuadros-platform",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.14",
|
||||
"private": true,
|
||||
"workspaces": [
|
||||
"apps/*",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/database",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.14",
|
||||
"private": true,
|
||||
"main": "generated/client/index.js",
|
||||
"types": "generated/client/index.d.ts",
|
||||
|
||||
Reference in New Issue
Block a user