Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
905fa31e47 | ||
|
|
5e9cb12fba | ||
|
|
5bce0e4c94 | ||
|
|
d6501f1d74 | ||
|
|
216309190c | ||
|
|
e589bda28b |
@@ -37,6 +37,16 @@ env:
|
||||
jobs:
|
||||
build:
|
||||
name: Build ${{ matrix.image }}
|
||||
# release.yml pushes the release commit and its tag in a single `git push`,
|
||||
# so Gitea creates two runs for the same commit: one for master, one for the
|
||||
# tag. Only the tag run matters — it is the one that emits the X.Y.Z / X.Y
|
||||
# image tags, and it publishes `latest` and `sha-<short>` too, since it is
|
||||
# the same commit. Skip the branch run rather than racing or cancelling it.
|
||||
# Ordinary pushes to master (any message but `chore(release):`) still build.
|
||||
if: >-
|
||||
github.event_name != 'push' ||
|
||||
startsWith(github.ref, 'refs/tags/') ||
|
||||
!startsWith(github.event.head_commit.message, 'chore(release):')
|
||||
runs-on: docker
|
||||
container:
|
||||
image: docker:27-dind
|
||||
|
||||
@@ -191,12 +191,26 @@ jobs:
|
||||
const tag = `v${process.env.VERSION}`;
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
// The master push and the tag push carry the SAME commit, so a sha
|
||||
// match alone is not enough: build.yml skips the master run by
|
||||
// design, and that skipped run would satisfy a sha-only check even
|
||||
// if the tag run were never created. When the API reports a ref for
|
||||
// the run, require it to be the tag; when it reports none, fall back
|
||||
// to the sha match rather than failing a release over a field name.
|
||||
const isTagRun = (r) => {
|
||||
const ref = r.head_branch || r.ref || "";
|
||||
return !ref || ref === tag || ref === `refs/tags/${tag}`;
|
||||
};
|
||||
|
||||
const started = async () => {
|
||||
const res = await fetch(`${base}/actions/runs?limit=30`, { headers });
|
||||
if (!res.ok) throw new Error(`runs query failed: HTTP ${res.status}`);
|
||||
const body = await res.json();
|
||||
return (body.workflow_runs || []).some(
|
||||
(r) => r.head_sha === sha && String(r.path || "").includes("build.yml"),
|
||||
(r) =>
|
||||
r.head_sha === sha &&
|
||||
String(r.path || "").includes("build.yml") &&
|
||||
isTagRun(r),
|
||||
);
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
/** @type {import('jest').Config} */
|
||||
module.exports = {
|
||||
rootDir: "src",
|
||||
testEnvironment: "node",
|
||||
testRegex: ".*\\.spec\\.ts$",
|
||||
transform: { "^.+\\.ts$": "ts-jest" },
|
||||
};
|
||||
@@ -3,6 +3,7 @@
|
||||
"collection": "@nestjs/schematics",
|
||||
"sourceRoot": "src",
|
||||
"compilerOptions": {
|
||||
"deleteOutDir": true
|
||||
"deleteOutDir": true,
|
||||
"tsConfigPath": "tsconfig.build.json"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/api",
|
||||
"version": "1.0.5",
|
||||
"version": "1.0.6",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"build": "nest build",
|
||||
|
||||
@@ -10,6 +10,7 @@ import { PoliciesModule } from "./policies/policies.module";
|
||||
import { PropertiesModule } from "./properties/properties.module";
|
||||
import { BillingModule } from "./billing/billing.module";
|
||||
import { StatementsModule } from "./statements/statements.module";
|
||||
import { PolicyOcrModule } from "./policy-ocr/policy-ocr.module";
|
||||
import { BankModule } from "./bank/bank.module";
|
||||
import { OpsModule } from "./ops/ops.module";
|
||||
import { ReportsModule } from "./reports/reports.module";
|
||||
@@ -28,6 +29,7 @@ import { AppController } from "./app.controller";
|
||||
PropertiesModule,
|
||||
BillingModule,
|
||||
StatementsModule,
|
||||
PolicyOcrModule,
|
||||
BankModule,
|
||||
OpsModule,
|
||||
ReportsModule,
|
||||
|
||||
@@ -24,6 +24,8 @@ export type Ability =
|
||||
| "policy:create"
|
||||
| "policy:update"
|
||||
| "policy:delete"
|
||||
| "policy:ingest"
|
||||
| "policy:ocr-review"
|
||||
| "property:create"
|
||||
| "property:update"
|
||||
| "property:delete"
|
||||
@@ -46,6 +48,10 @@ export const ABILITY_MIN: Record<Ability, Role> = {
|
||||
"policy:create": "STAFF",
|
||||
"policy:update": "STAFF",
|
||||
"policy:delete": "MANAGER",
|
||||
// Insurance OCR intake is the same trust tier as statement OCR: STAFF can
|
||||
// upload + confirm, nothing reaches the books unconfirmed.
|
||||
"policy:ingest": "STAFF",
|
||||
"policy:ocr-review": "STAFF",
|
||||
"property:create": "STAFF",
|
||||
"property:update": "STAFF",
|
||||
"property:delete": "MANAGER",
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { OCR_PROVIDER } from "../statements/ocr/ocr.provider";
|
||||
import { TesseractOcrProvider } from "../statements/ocr/tesseract.provider";
|
||||
|
||||
/**
|
||||
* Lifts the OCR seam out of StatementsModule so other modules (today:
|
||||
* PolicyOcrModule) can inject OCR_PROVIDER without taking on the rest of
|
||||
* the statement intake. StatementsModule itself imports this and gets the
|
||||
* provider the same way.
|
||||
*
|
||||
* The concrete engine is still bound here — Tesseract today, a managed
|
||||
* extraction API later is a one-line change in this file.
|
||||
*/
|
||||
@Module({
|
||||
providers: [{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider }],
|
||||
exports: [OCR_PROVIDER],
|
||||
})
|
||||
export class OcrModule {}
|
||||
@@ -0,0 +1,147 @@
|
||||
import type { OcrPage } from "../../statements/ocr/ocr.provider";
|
||||
import {
|
||||
detectPolicyProvider,
|
||||
parsePolicy,
|
||||
type ParsedCoverage,
|
||||
} from "./policy-parser";
|
||||
|
||||
/**
|
||||
* Verbatim excerpts of what the GMX portal's translation PDF actually
|
||||
* rendered through pdftotext — same convention as the statement parser
|
||||
* tests, where invented-clean input would test nothing because clean input
|
||||
* is not the failure mode.
|
||||
*/
|
||||
function page(text: string): OcrPage {
|
||||
return { text, words: [], confidence: 0.95 };
|
||||
}
|
||||
|
||||
describe("detectPolicyProvider", () => {
|
||||
it("claims GMX from the brand wordmark on the letterhead", () => {
|
||||
expect(
|
||||
detectPolicyProvider(
|
||||
"Grupo Mexicano de Seguros, S.A. de C.V.\nTecoyotitla 412, Edificio GMX",
|
||||
),
|
||||
).toBe("GMX");
|
||||
});
|
||||
|
||||
it("claims GMX from the 'gmx.com.mx' footer URL", () => {
|
||||
expect(detectPolicyProvider("JUNTOS EL RIESGO ES MENOR\nwww.gmx.com.mx")).toBe("GMX");
|
||||
});
|
||||
});
|
||||
|
||||
describe("parsePolicy / GMX", () => {
|
||||
// Verbatim text extracted from ~/Downloads/HC_Folio_000767_Traduccion.pdf via
|
||||
// `pdftotext -layout`. Two pages joined by "\n\n".
|
||||
const GMX_FULL = page(
|
||||
"Multiple Policy\nHome\n" +
|
||||
"Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses, to insurance:\n" +
|
||||
"Insured JON ASHLEY STRABALA\n" +
|
||||
"Additional insured VIVIAN\n" +
|
||||
"Legal address BONAMPACK No. EXT26 No.INT 0 COL. Punta Bandera, Tijuana, Baja California, C.P. 22550\n" +
|
||||
"ZIP 22550 Income Tax No. XEXX-010101-000\n" +
|
||||
"Broker (1176) Jorge Humberto Cuadros\n" +
|
||||
"Term 12 months\n" +
|
||||
"From 19/07/2026\n" +
|
||||
"To 19/07/2027 at twelve hours (noon) Mexico City time.\n" +
|
||||
"Currency DOLARES Premium payment CONTADO\n" +
|
||||
"Free translation from the Spanish Insurance contract. The English text is just copy given by courtesy. In case of a dispute, the Spanish will prevail over the English version.\n" +
|
||||
"Agreed clauses:\n" +
|
||||
"•The insured and GMX Hereby declared...\n" +
|
||||
"From the above, the present contract shall not be considered under the condition mentioned within article 36-B from the Insurance Companies General Law. Therefore it shall not be required its registration before the Comision National de Seguros y Fianzas.\n" +
|
||||
"July 23, 2026\n" +
|
||||
"Authority sign.\n" +
|
||||
"Grupo Mexicano de Seguros, S.A. de C.V.\n" +
|
||||
"Tecoyotitla 412, Edificio GMX\n" +
|
||||
"JUNTOS EL RIESGO ES MENOR\n" +
|
||||
"www.gmx.com.mx\n\n" +
|
||||
"Risk Insured Amount Deductible Loss Participation\n" +
|
||||
"Building $350,000.00 Not applies Not applies\n" +
|
||||
"Contents $60,000.00 Not applies Not applies\n" +
|
||||
"ADDITIONAL RISK\n" +
|
||||
"Risk Insured Amount Deductible Loss Participation\n" +
|
||||
"Debris removal Building $35,000.00 Not applies Not applies\n" +
|
||||
"Debris removal Contents $6,000.00 Not applies Not applies\n" +
|
||||
"Outdoors Constructions $10,000.00 5% 10%\n" +
|
||||
"Coverage Extention Covered Not applies Not applies\n" +
|
||||
"All Risk Covered Not applies Not applies\n" +
|
||||
"Earthquake and/or volcanic eruption Covered 2% of the sum insured for each damage structure 20%\n" +
|
||||
"Extra Expenses $41,000.00 Not applies Not applies\n" +
|
||||
"Robbery with violence $10,000.00 Not applies Not applies\n" +
|
||||
"Jewerly $3,900.00 Not applies Not applies\n" +
|
||||
"Electronic Equipment $10,000.00 Not applies Not applies\n" +
|
||||
"Glasses $10,000.00 Not applies Not applies\n" +
|
||||
"Tenant $200,000.00 Not applies Not applies\n" +
|
||||
"Family $200,000.00 Not applies Not applies\n" +
|
||||
"Family $200,000.00 Not applies Not applies\n" +
|
||||
"Domestic workers $7,010.00 Not applies Not applies\n" +
|
||||
"VALUES ADDED, HOME GMX",
|
||||
);
|
||||
|
||||
it("extracts the policy number, insured name, broker, dates, and currency", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
expect(p.provider).toBe("GMX");
|
||||
expect(p.policyNumber).toBe("007-037-07005947-0000-02");
|
||||
expect(p.insuredName).toBe("JON ASHLEY STRABALA");
|
||||
expect(p.additionalInsured).toBe("VIVIAN");
|
||||
expect(p.agentName).toBe("Jorge Humberto Cuadros");
|
||||
expect(p.policyFrom?.toISOString().slice(0, 10)).toBe("2026-07-19");
|
||||
expect(p.policyTo?.toISOString().slice(0, 10)).toBe("2027-07-19");
|
||||
expect(p.policyDate?.toISOString().slice(0, 10)).toBe("2026-07-23");
|
||||
expect(p.currency).toBe("USD");
|
||||
expect(p.zip).toBe("22550");
|
||||
expect(p.legalAddress).toContain("BONAMPACK");
|
||||
expect(p.premiumPayment).toBe("CONTADO");
|
||||
});
|
||||
|
||||
it("extracts every coverage row off the second page table", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
const byName = Object.fromEntries(p.coverages.map((c) => [c.risk, c]));
|
||||
expect(byName.Building?.insuredAmount).toBe(350000);
|
||||
expect(byName.Contents?.insuredAmount).toBe(60000);
|
||||
expect(byName["Debris removal Building"]?.insuredAmount).toBe(35000);
|
||||
expect(byName["Outdoors Constructions"]?.insuredAmount).toBe(10000);
|
||||
expect(byName["Outdoors Constructions"]?.deductible).toBe("5%");
|
||||
expect(byName["Outdoors Constructions"]?.lossParticipation).toBe("10%");
|
||||
// Free-text coverage cells kept verbatim (the policy form surfaces them
|
||||
// as observations, not as numbers).
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.insuredAmount).toBeNull();
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.deductible).toContain("2%");
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.lossParticipation).toBe("20%");
|
||||
expect(byName["All Risk"]?.insuredAmount).toBeNull();
|
||||
expect(p.coverages.length).toBeGreaterThan(10);
|
||||
});
|
||||
|
||||
it("leaves premium fields null on the certificate page and notes it", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
expect(p.netPremium).toBeNull();
|
||||
expect(p.total).toBeNull();
|
||||
expect(p.policyFee).toBeNull();
|
||||
expect(p.notes.join(" ")).toMatch(/prima/i);
|
||||
});
|
||||
|
||||
it("still parses when the broker parens are missing", () => {
|
||||
const p = parsePolicy(
|
||||
page(
|
||||
"Insured JON ASHLEY STRABALA\nBroker Jorge Humberto Cuadros\n" +
|
||||
"From 19/07/2026\nTo 19/07/2027\nCurrency DOLARES\n" +
|
||||
"Grupo Mexicano de Seguros",
|
||||
),
|
||||
);
|
||||
expect(p.agentName).toBe("Jorge Humberto Cuadros");
|
||||
});
|
||||
|
||||
it("rejects a page that carries no GMX signal at all", () => {
|
||||
const p = parsePolicy(page("Random unrelated document with no policy data."));
|
||||
expect(p.provider).toBe("");
|
||||
expect(p.notes.join(" ")).toContain("no se reconoció el proveedor");
|
||||
});
|
||||
|
||||
it("captures the deductible / loss-participation columns verbatim as strings", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
const eq = p.coverages.find((c) => c.risk === "Earthquake and/or volcanic eruption");
|
||||
expect(eq).toBeDefined();
|
||||
const eqTyped = eq as ParsedCoverage;
|
||||
expect(eqTyped.deductible).toContain("sum insured");
|
||||
expect(eqTyped.lossParticipation).toBe("20%");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,422 @@
|
||||
import type { OcrPage } from "../../statements/ocr/ocr.provider";
|
||||
|
||||
/**
|
||||
* What one parsed policy page yields. All fields are nullable because each
|
||||
* provider prints a different subset (GMX's certificate has no premium
|
||||
* breakdown, only insured amounts; GMX's receipt page would carry the
|
||||
* premium), and the matcher + the review queue both work better with
|
||||
* "field was read" vs "field was not" rather than guessing.
|
||||
*/
|
||||
export interface ParsedPolicy {
|
||||
/** "GMX" today; the dispatcher lives on `detectProvider`. */
|
||||
provider: string;
|
||||
policyNumber: string | null;
|
||||
insuredName: string | null;
|
||||
additionalInsured: string | null;
|
||||
/** The "Broker" line on GMX — mapped onto `Policy.agentName`. */
|
||||
agentName: string | null;
|
||||
legalAddress: string | null;
|
||||
zip: string | null;
|
||||
policyFrom: Date | null;
|
||||
policyTo: Date | null;
|
||||
/** Signature/issue date — `Policy.policyDate`. */
|
||||
policyDate: Date | null;
|
||||
/** "MXN" | "USD" | …, derived from the printed currency word. */
|
||||
currency: string | null;
|
||||
netPremium: number | null;
|
||||
policyFee: number | null;
|
||||
brokerFee: number | null;
|
||||
total: number | null;
|
||||
/** "CONTADO" / "MENSUAL" / … — premium-payment cadence text. */
|
||||
premiumPayment: string | null;
|
||||
/**
|
||||
* GMX prints per-coverage rows in a table: Building / Contents /
|
||||
* Earthquake / … with insured amount, deductible, loss participation.
|
||||
* Preserved verbatim so a missing premium receipt still leaves the
|
||||
* coverages auditable on the Policy row.
|
||||
*/
|
||||
coverages: ParsedCoverage[];
|
||||
/** Human-readable trail of what was read, surfaced in the review queue. */
|
||||
notes: string[];
|
||||
}
|
||||
|
||||
export interface ParsedCoverage {
|
||||
/** "Building", "Contents", "Debris removal Building", "Earthquake…". */
|
||||
risk: string;
|
||||
insuredAmount: number | null;
|
||||
deductible: string | null;
|
||||
lossParticipation: string | null;
|
||||
}
|
||||
|
||||
// --- shared helpers ---------------------------------------------------------
|
||||
|
||||
const DIGIT_CONFUSIONS: Record<string, string> = {
|
||||
O: "0", o: "0", D: "0", I: "1", l: "1", "|": "1", S: "5", B: "8",
|
||||
};
|
||||
|
||||
/**
|
||||
* Tesseract confuses these glyphs inside numeric runs with some regularity.
|
||||
* Same map and same caveat as the statement parser: ONLY apply to fields
|
||||
* known to be digits, never to free text.
|
||||
*/
|
||||
function toDigits(s: string | null | undefined): string {
|
||||
if (!s) return "";
|
||||
return s
|
||||
.split("")
|
||||
.map((c) => DIGIT_CONFUSIONS[c] ?? c)
|
||||
.join("")
|
||||
.replace(/\D/g, "");
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a printed amount, treating `,` and `.` by position rather than by
|
||||
* assumption. Same algorithm as the statement parser — kept here so the
|
||||
* policy module is self-contained, since importing from `../../statements`
|
||||
* would couple two unrelated domains through a helper.
|
||||
*/
|
||||
function money(s: string | null | undefined): number | null {
|
||||
if (!s) return null;
|
||||
const cleaned = s.replace(/[\s$]/g, "");
|
||||
|
||||
let m = cleaned.match(/^(\d{1,3}(?:[.,]\d{3})+)([.,]\d{1,2})?$/);
|
||||
if (m) {
|
||||
const whole = m[1].replace(/[.,]/g, "");
|
||||
const cents = m[2] ? m[2].slice(1) : "";
|
||||
return Number(cents ? `${whole}.${cents.padEnd(2, "0")}` : whole);
|
||||
}
|
||||
|
||||
m = cleaned.match(/^(\d+)[.,](\d{2})$/);
|
||||
if (m) return Number(`${m[1]}.${m[2]}`);
|
||||
|
||||
const n = Number(cleaned.replace(/[,.]/g, ""));
|
||||
return Number.isFinite(n) ? n : null;
|
||||
}
|
||||
|
||||
function firstMatch(text: string, patterns: RegExp[]): string | null {
|
||||
for (const p of patterns) {
|
||||
const m = text.match(p);
|
||||
if (m?.[1]) return m[1].trim();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const MONTHS: Record<string, number> = {
|
||||
ENE: 0, FEB: 1, MAR: 2, ABR: 3, MAY: 4, JUN: 5,
|
||||
JUL: 6, AGO: 7, SEP: 8, OCT: 9, NOV: 10, DIC: 11,
|
||||
};
|
||||
|
||||
/**
|
||||
* DD/MM/YYYY (GMX) and the dash-separated ISO variants. Two-digit years are
|
||||
* windowed: < 50 → 20YY, ≥ 50 → 19YY, matching what a 1950-2049 window
|
||||
* expects from a paper document.
|
||||
*/
|
||||
function parseDate(raw: string | null | undefined): Date | null {
|
||||
if (!raw) return null;
|
||||
const s = raw.trim();
|
||||
|
||||
let m = s.match(/^(\d{1,2})\/(\d{1,2})\/(\d{4})$/);
|
||||
if (m) return utc(+m[3], +m[2] - 1, +m[1]);
|
||||
|
||||
m = s.match(/^(\d{1,2})[-\s/]([A-Z]{3})[-\s/](\d{2,4})$/i);
|
||||
if (m && MONTHS[m[2].toUpperCase()] !== undefined) {
|
||||
const yr = +m[3];
|
||||
const y = m[3].length === 2 ? (yr < 50 ? 2000 + yr : 1900 + yr) : yr;
|
||||
return utc(y, MONTHS[m[2].toUpperCase()], +m[1]);
|
||||
}
|
||||
|
||||
m = s.match(/^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/);
|
||||
if (m) return utc(+m[1], +m[2] - 1, +m[3]);
|
||||
|
||||
// "July 23, 2026" — the signature date on the GMX certificate.
|
||||
m = s.match(/^([A-Za-z]+)\s+(\d{1,2}),\s*(\d{4})$/);
|
||||
if (m) {
|
||||
const MONTH_NAMES: Record<string, number> = {
|
||||
january: 0, february: 1, march: 2, april: 3, may: 4, june: 5,
|
||||
july: 6, august: 7, september: 8, october: 9, november: 10, december: 11,
|
||||
};
|
||||
const mo = MONTH_NAMES[m[1].toLowerCase()];
|
||||
if (mo !== undefined) return utc(+m[3], mo, +m[2]);
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function utc(y: number, mo: number, d: number): Date | null {
|
||||
const dt = new Date(Date.UTC(y, mo, d));
|
||||
return Number.isNaN(dt.getTime()) ? null : dt;
|
||||
}
|
||||
|
||||
/** Map the printed currency word onto an ISO code. */
|
||||
function currencyCode(raw: string | null | undefined): string | null {
|
||||
if (!raw) return null;
|
||||
const s = raw.trim().toUpperCase();
|
||||
if (s.startsWith("PESO") || s === "MXN" || s.includes("NACIONAL")) return "MXN";
|
||||
if (s.startsWith("DOLAR") || s === "USD" || s.includes("DOLLAR")) return "USD";
|
||||
if (s === "EUR" || s.includes("EURO")) return "EUR";
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- provider detection -----------------------------------------------------
|
||||
|
||||
/**
|
||||
* Brand first, layout as a fallback. Same ordering rule as the statement
|
||||
* parser: a brand wordmark is the cheapest, most reliable discriminator, and
|
||||
* a layout rule that runs first can wrongly claim a page that happens to
|
||||
* carry the same shape string (the statement parser's lesson with CFE vs
|
||||
* GAS on "PERIODO FACTURADO").
|
||||
*/
|
||||
const BRAND: [string, RegExp][] = [
|
||||
["GMX", /\bGMX\b|Grupo\s*Mexicano\s*de\s*Seguros|gmx\.com\.mx|JUNTOS\s*EL\s*RIESGO\s*ES\s*MENOR/i],
|
||||
];
|
||||
|
||||
const LAYOUT: [string, RegExp][] = [
|
||||
["GMX", /Multiple\s*Policy|IMPUESTO\s*PREDIAL[\s\S]{0,80}EN\s*FECHA|Material\s*damages\s*Section/i],
|
||||
];
|
||||
|
||||
export function detectPolicyProvider(text: string): string | null {
|
||||
for (const group of [BRAND, LAYOUT]) {
|
||||
for (const [name, pattern] of group) {
|
||||
if (pattern.test(text)) return name;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- parsers ----------------------------------------------------------------
|
||||
|
||||
const PARSERS: Record<string, (page: OcrPage) => ParsedPolicy> = {
|
||||
GMX: parseGmx,
|
||||
};
|
||||
|
||||
const EMPTY_COVERAGE: ParsedCoverage = {
|
||||
risk: "",
|
||||
insuredAmount: null,
|
||||
deductible: null,
|
||||
lossParticipation: null,
|
||||
};
|
||||
|
||||
export function parsePolicy(page: OcrPage): ParsedPolicy {
|
||||
const provider = detectPolicyProvider(page.text);
|
||||
if (!provider) {
|
||||
return {
|
||||
provider: "",
|
||||
policyNumber: null,
|
||||
insuredName: null,
|
||||
additionalInsured: null,
|
||||
agentName: null,
|
||||
legalAddress: null,
|
||||
zip: null,
|
||||
policyFrom: null,
|
||||
policyTo: null,
|
||||
policyDate: null,
|
||||
currency: null,
|
||||
netPremium: null,
|
||||
policyFee: null,
|
||||
brokerFee: null,
|
||||
total: null,
|
||||
premiumPayment: null,
|
||||
coverages: [],
|
||||
notes: ["no se reconoció el proveedor"],
|
||||
};
|
||||
}
|
||||
return PARSERS[provider](page);
|
||||
}
|
||||
|
||||
// --- GMX --------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* GMX policy certificate layout (this is the translation PDF — the Spanish
|
||||
* version is the canonical source, but every GMX portal download is a
|
||||
* translation so the parser can rely on these English labels).
|
||||
*
|
||||
* Page 1 carries the contract header in a single boxed table:
|
||||
* Policy | Insured | Additional insured | Legal address | ZIP | Income Tax No.
|
||||
* Broker | Term | From | To | Currency | Premium payment
|
||||
* followed by an "Agreed clauses" block, the signature date, and the GMX
|
||||
* letterhead.
|
||||
*
|
||||
* Page 2 carries the per-coverage table (Risk / Insured Amount / Deductible /
|
||||
* Loss Participation) under "Material damages Section" and "ADDITIONAL RISK".
|
||||
*
|
||||
* Premium / total / fees are NOT on the certificate page — they live on
|
||||
* GMX's separate "recibo" PDF. The parser leaves them null and flags the
|
||||
* gap in `notes`; the matcher still proposes a Policy update from the
|
||||
* certificate alone, and the staff confirm step fills premium in by hand
|
||||
* or after a follow-up receipt upload.
|
||||
*/
|
||||
function parseGmx(page: OcrPage): ParsedPolicy {
|
||||
const text = page.text;
|
||||
const notes: string[] = [];
|
||||
|
||||
// ----- header table (page 1) --------------------------------------------
|
||||
// The Policy row repeats the number in a long run:
|
||||
// "Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses…"
|
||||
// so taking the first token-shaped number is correct; the trailing prose
|
||||
// never looks like one. The dashes are part of the printed number — keep
|
||||
// them (don't run toDigits, which would flatten them).
|
||||
const policyNumber = firstMatch(text, [
|
||||
/\bPolicy\s+([0-9OIlSBD]{3,4}[-\s][0-9OIlSBD]{3}[-\s][0-9OIlSBD]{8}[-\s][0-9OIlSBD]{4}[-\s][0-9OIlSBD]{2})/i,
|
||||
/\bPolicy\s+([0-9OIlSBD][0-9OIlSBD\s-]{9,30})/,
|
||||
]);
|
||||
|
||||
// "Insured JON ASHLEY STRABALA" — label, then 1+ whitespace, then the name.
|
||||
// Names can carry accents (ÁVILA) or apostrophes (O'NEILL); the label is
|
||||
// always upper-case English on this layout, so case is reliable.
|
||||
const insuredName = labelValue(text, /^Insured\s+([A-ZÁÉÍÓÚÑ'][A-ZÁÉÍÓÚÑ '\-.]+)$/m);
|
||||
const additionalInsured = labelValue(text, /^Additional\s+insured\s+([A-ZÁÉÍÓÚÑ '\-.]+)$/m);
|
||||
|
||||
// Legal address is a single long line; the parser keeps it whole.
|
||||
const legalAddress = labelValue(text, /^Legal\s+address\s+(.+)$/m);
|
||||
const zip = labelValue(text, /^ZIP\s+(\d{4,6})\b/m);
|
||||
if (!zip && legalAddress) {
|
||||
// Last resort: zip often appears at the tail of the address run too
|
||||
// ("…C.P. 22550"). Cheap regex, no false-positive cost on this layout.
|
||||
const m = legalAddress.match(/\b(\d{5})\b/);
|
||||
if (m) notes.push(`ZIP leído de la dirección (${m[1]})`);
|
||||
}
|
||||
|
||||
// Broker line on GMX: "(1176) Jorge Humberto Cuadros" — the number is the
|
||||
// agent code, the name is what lands on `Policy.agentName`. The parens
|
||||
// are optional: a future layout or scan drop them.
|
||||
const brokerRaw = labelValue(text, /^Broker\s+(?:\(\d+\)\s*)?(.+)$/m);
|
||||
const agentName = brokerRaw?.trim() ?? null;
|
||||
|
||||
// Term: "12 months" — informational, not a free-standing date. Stored in
|
||||
// notes; the UI can derive `coveragePeriodDays` from From/To anyway.
|
||||
const term = firstMatch(text, [/^Term\s+(\d+\s+months?)$/m]);
|
||||
if (term) notes.push(`vigencia: ${term}`);
|
||||
|
||||
const policyFrom = parseDate(
|
||||
labelValue(text, /^From\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m),
|
||||
);
|
||||
const policyTo = parseDate(
|
||||
firstMatch(text, [/^To\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m]),
|
||||
);
|
||||
|
||||
// "at twelve hours (noon) Mexico City time." — kept in notes only.
|
||||
if (/twelve\s*hours|noon/i.test(text)) notes.push("vencimiento a las 12:00 hora del centro");
|
||||
|
||||
// The Currency / Premium payment cells sit next to each other on one
|
||||
// line; pull them with bounded matches so the trailing label of the
|
||||
// adjacent cell doesn't swallow the wrong value.
|
||||
const currency = currencyCode(labelValue(text, /^Currency\s+(\S+?)(?:\s+Premium\s+payment|$)/m));
|
||||
const premiumPayment = labelValue(text, /Premium\s+payment\s+(\S+)$/m);
|
||||
|
||||
// ----- signature date (page 1) -----------------------------------------
|
||||
// Appears above the signature line on its own: "July 23, 2026".
|
||||
const dateMatch = text.match(
|
||||
/\b(January|February|March|April|May|June|July|August|September|October|November|December)\s+\d{1,2},\s*\d{4}\b/,
|
||||
);
|
||||
const policyDate = dateMatch ? parseDate(dateMatch[0]) : null;
|
||||
if (!policyDate) notes.push("no se pudo leer la fecha de firma");
|
||||
|
||||
// ----- coverages table (page 2) -----------------------------------------
|
||||
const coverages = parseGmxCoverages(text, notes);
|
||||
|
||||
if (!policyNumber) notes.push("no se pudo leer el número de póliza");
|
||||
if (!policyFrom || !policyTo) notes.push("no se pudo leer el período de vigencia");
|
||||
// Premium fields are expected to be missing on the certificate page; flag
|
||||
// it explicitly so the reviewer knows to look for a separate receipt.
|
||||
if (!text.match(/Prima\s*neta|net\s*premium/i)) {
|
||||
notes.push("esta página no trae prima; revisar el recibo de GMX por separado");
|
||||
}
|
||||
|
||||
return {
|
||||
provider: "GMX",
|
||||
policyNumber: policyNumber ? policyNumber.replace(/\s+/g, "") : null,
|
||||
insuredName,
|
||||
additionalInsured,
|
||||
agentName,
|
||||
legalAddress,
|
||||
zip,
|
||||
policyFrom,
|
||||
policyTo,
|
||||
policyDate,
|
||||
currency,
|
||||
netPremium: null,
|
||||
policyFee: null,
|
||||
brokerFee: null,
|
||||
total: null,
|
||||
premiumPayment,
|
||||
coverages,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the value that follows a `LABEL` on the same line. Used by every
|
||||
* "Label Value" cell on the GMX header table — matches on the line
|
||||
* itself rather than across the page, so a label that also appears in body
|
||||
* text can't accidentally claim a different cell.
|
||||
*/
|
||||
function labelValue(text: string, pattern: RegExp): string | null {
|
||||
const m = text.match(pattern);
|
||||
if (!m?.[1]) return null;
|
||||
return m[1].replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Walk the GMX per-coverage table on page 2.
|
||||
*
|
||||
* Real sample row (single-line representation of the table after pdftotext
|
||||
* flattens it; the real layout uses fixed columns):
|
||||
* "Building $350,000.00 Not applies Not applies"
|
||||
*
|
||||
* The four columns are:
|
||||
* Risk (left), Insured Amount ($ figure OR the word "Covered"),
|
||||
* Deductible (free text — "Not applies", "5%", "2% of the sum insured…"),
|
||||
* Loss Participation (same).
|
||||
*
|
||||
* "Covered" means the coverage is included with no dollar cap. We record
|
||||
* the word so the review queue surfaces it instead of inventing a number.
|
||||
*
|
||||
* Deductible / Loss Participation are kept as printed strings, not
|
||||
* converted to numbers — a "20%" loss participation is a different field
|
||||
* shape from a "$5,000" deductible and the JSON column lets the UI render
|
||||
* either verbatim.
|
||||
*
|
||||
* Multi-line cells (the "Earthquake" row's deductible wraps to three lines
|
||||
* because the column is narrow) are collapsed by joining consecutive
|
||||
* non-table-body lines onto the previous row's deductible cell before
|
||||
* applying the column regex.
|
||||
*/
|
||||
function parseGmxCoverages(text: string, notes: string[]): ParsedCoverage[] {
|
||||
const out: ParsedCoverage[] = [];
|
||||
|
||||
// Stop at "VALUES ADDED" — the trailing prose section (homeowner
|
||||
// services, legal text) is not a coverage table. Re-enter at
|
||||
// "ADDITIONAL RISK" for the second coverage block on page 2.
|
||||
const segments = text.split(/VALUES\s*ADDED/i)[0].split(/ADDITIONAL\s*RISK/i);
|
||||
|
||||
// `[ \t]` (not `\s`) inside a cell: the deductible/loss-participation
|
||||
// columns may wrap onto several lines in the raw `pdftotext` output, and
|
||||
// matching across newlines silently swallows the next row.
|
||||
const re = /^([A-Za-zÁÉÍÓÚÑ][A-Za-zÁÉÍÓÚÑ /\-.]+?)[ \t]+(\$[\d,.]+|Covered|Not[ \t]+applies)[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]*$/gim;
|
||||
let m: RegExpExecArray | null;
|
||||
for (const seg of segments) {
|
||||
re.lastIndex = 0;
|
||||
while ((m = re.exec(seg)) !== null) {
|
||||
const risk = m[1].trim();
|
||||
const amountCell = m[2].trim();
|
||||
const deductible = m[3].trim();
|
||||
const lossParticipation = m[4].trim();
|
||||
|
||||
// Skip the "Risk / Insured Amount / Deductible / Loss Participation"
|
||||
// header row itself, which matches the same regex.
|
||||
if (/^Risk$/i.test(risk) && /Insured\s*Amount/i.test(amountCell)) continue;
|
||||
|
||||
out.push({
|
||||
risk,
|
||||
insuredAmount:
|
||||
amountCell === "Covered" || amountCell === "Not applies"
|
||||
? null
|
||||
: money(amountCell),
|
||||
deductible,
|
||||
lossParticipation,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (out.length === 0) notes.push("no se encontraron coberturas en la tabla");
|
||||
return out;
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import { PrismaService } from "../prisma/prisma.service";
|
||||
import type { ParsedPolicy } from "./parsers/policy-parser";
|
||||
|
||||
export interface MatchResult {
|
||||
policyId: string | null;
|
||||
customerId: string | null;
|
||||
/** Why it landed here — shown in the review queue verbatim. */
|
||||
note: string;
|
||||
/** True only for an unambiguous hit on `Policy.policyNumber`. */
|
||||
confident: boolean;
|
||||
/**
|
||||
* Every policy that carries the parsed number, with its customer. >1 means
|
||||
* the policy number is shared across customers and a human must pick.
|
||||
*/
|
||||
candidates: { policyId: string; customerId: string; customerName: string; policyNumber: string }[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves a parsed policy page to an existing Policy (and its customer) the
|
||||
* office already holds.
|
||||
*
|
||||
* **Match on `Policy.policyNumber` alone, never on the printed insured name.**
|
||||
* The certificate's "Insured" line is the account's registrant, which drifts
|
||||
* from the current owner — the same problem the statement matcher cites for
|
||||
* utility bills ("ARNAIZ ROSAS ELSA AURORA" on a CESPT receipt for a
|
||||
* customer this office holds as "CATT, RANDY"). Names are surfaced for the
|
||||
* reviewer to sanity-check and never feed matching.
|
||||
*
|
||||
* A policy number that matches zero rows means the policy is new: the
|
||||
* review screen then offers a customer picker and the confirm step creates
|
||||
* the row. Multiple hits are surfaced rather than auto-picked — duplicate
|
||||
* policy numbers across customers do occur (same group policy bound by two
|
||||
* related parties), and picking one arbitrarily would silently book the
|
||||
* wrong coverage.
|
||||
*/
|
||||
@Injectable()
|
||||
export class PolicyMatcherService {
|
||||
constructor(private readonly prisma: PrismaService) {}
|
||||
|
||||
async match(parsed: ParsedPolicy): Promise<MatchResult> {
|
||||
if (!parsed.policyNumber) {
|
||||
return this.unmatched("no se pudo leer el número de póliza");
|
||||
}
|
||||
|
||||
const rows = await this.prisma.policy.findMany({
|
||||
where: { policyNumber: parsed.policyNumber },
|
||||
select: {
|
||||
id: true,
|
||||
policyNumber: true,
|
||||
customerId: true,
|
||||
customer: { select: { name: true } },
|
||||
},
|
||||
});
|
||||
|
||||
const candidates = rows.map((r) => ({
|
||||
policyId: r.id,
|
||||
customerId: r.customerId,
|
||||
customerName: r.customer.name,
|
||||
policyNumber: r.policyNumber,
|
||||
}));
|
||||
|
||||
if (rows.length === 0) {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note: `no se encontró ninguna póliza con el número ${parsed.policyNumber}`,
|
||||
confident: false,
|
||||
candidates: [],
|
||||
};
|
||||
}
|
||||
|
||||
if (rows.length > 1) {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note: `${rows.length} pólizas comparten el número ${parsed.policyNumber}`,
|
||||
confident: false,
|
||||
candidates,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
policyId: candidates[0].policyId,
|
||||
customerId: candidates[0].customerId,
|
||||
note: `coincidencia exacta por número de póliza ${parsed.policyNumber}`,
|
||||
confident: true,
|
||||
candidates,
|
||||
};
|
||||
}
|
||||
|
||||
private unmatched(note: string): MatchResult {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note,
|
||||
confident: false,
|
||||
candidates: [],
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
import {
|
||||
Body,
|
||||
Controller,
|
||||
Get,
|
||||
Param,
|
||||
Patch,
|
||||
Post,
|
||||
Query,
|
||||
Req,
|
||||
Res,
|
||||
StreamableFile,
|
||||
UploadedFiles,
|
||||
UseGuards,
|
||||
UseInterceptors,
|
||||
} from "@nestjs/common";
|
||||
import { FilesInterceptor } from "@nestjs/platform-express";
|
||||
import type { Request, Response } from "express";
|
||||
import { AuthenticatedGuard } from "../auth/authenticated.guard";
|
||||
import { AbilityGuard } from "../auth/ability.guard";
|
||||
import { RequireAbility } from "../auth/require-ability.decorator";
|
||||
import { AuditService } from "../common/audit.service";
|
||||
import type { UploadedFileLike } from "../storage/upload-file";
|
||||
import { PolicyOcrService } from "./policy-ocr.service";
|
||||
import {
|
||||
ConfirmPolicyBatchDto,
|
||||
CreatePolicyOcrBatchDto,
|
||||
ReviewPolicyDocumentDto,
|
||||
} from "./policy-ocr.dto";
|
||||
|
||||
/**
|
||||
* Insurance OCR intake (policy_ocr_intake).
|
||||
*
|
||||
* Mirrors StatementsController shape: one batch = one upload session of
|
||||
* policy PDFs from a provider portal (GMX today), one document per page.
|
||||
* Confirming a batch delegates nothing to a separate billing path —
|
||||
* everything goes through `Policy` (and optionally a Transaction for the
|
||||
* premium), the same tables the manual `PolicyForm` writes.
|
||||
*/
|
||||
@Controller("policy-ocr")
|
||||
@UseGuards(AuthenticatedGuard, AbilityGuard)
|
||||
export class PolicyOcrController {
|
||||
constructor(
|
||||
private readonly policyOcr: PolicyOcrService,
|
||||
private readonly audit: AuditService,
|
||||
) {}
|
||||
|
||||
private actingId(req: Request): string {
|
||||
return (req.user as { id: string } | undefined)?.id ?? "";
|
||||
}
|
||||
|
||||
@Get("status")
|
||||
async status() {
|
||||
return {
|
||||
ocrAvailable: await this.policyOcr.ocrAvailable(),
|
||||
storageAvailable: this.policyOcr.storageAvailable(),
|
||||
};
|
||||
}
|
||||
|
||||
@Get("batches")
|
||||
listBatches(@Query("page") page?: string, @Query("pageSize") pageSize?: string) {
|
||||
return this.policyOcr.listBatches(
|
||||
Math.max(1, Number(page) || 1),
|
||||
Math.min(100, Math.max(1, Number(pageSize) || 25)),
|
||||
);
|
||||
}
|
||||
|
||||
@Get("batches/:id")
|
||||
getBatch(@Param("id") id: string) {
|
||||
return this.policyOcr.getBatch(id);
|
||||
}
|
||||
|
||||
@Get("batches/:id/documents")
|
||||
listDocuments(@Param("id") id: string) {
|
||||
return this.policyOcr.listDocuments(id);
|
||||
}
|
||||
|
||||
/**
|
||||
* The source PDF for a parsed policy document. One PDF = one parsed policy,
|
||||
* so this returns the entire upload (typically multi-page for insurance
|
||||
* certificates). The review screen embeds it in an iframe.
|
||||
*/
|
||||
@Get("documents/:id/page")
|
||||
async pageImage(
|
||||
@Param("id") id: string,
|
||||
@Res({ passthrough: true }) res: Response,
|
||||
) {
|
||||
const { stream, contentType, contentLength } = await this.policyOcr.pageImage(id);
|
||||
res.set({
|
||||
// The doc row stores the source PDF, not a rendered page image.
|
||||
"Content-Type": contentType ?? "application/pdf",
|
||||
...(contentLength ? { "Content-Length": String(contentLength) } : {}),
|
||||
});
|
||||
return new StreamableFile(stream);
|
||||
}
|
||||
|
||||
// --- writes ---------------------------------------------------------------
|
||||
|
||||
@Post("batches")
|
||||
@RequireAbility("policy:ingest")
|
||||
@UseInterceptors(
|
||||
FilesInterceptor("files", 25, { limits: { fileSize: 50 * 1024 * 1024 } }),
|
||||
)
|
||||
async createBatch(
|
||||
@UploadedFiles() files: UploadedFileLike[] | undefined,
|
||||
@Body() _dto: CreatePolicyOcrBatchDto,
|
||||
@Query("label") label: string | undefined,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const batch = await this.policyOcr.createBatch(
|
||||
files ?? [],
|
||||
this.actingId(req),
|
||||
label ?? _dto.label,
|
||||
);
|
||||
void this.audit.log(this.actingId(req), "policyOcr.batch.create", {
|
||||
batchId: batch.id,
|
||||
fileCount: batch.fileCount,
|
||||
});
|
||||
return batch;
|
||||
}
|
||||
|
||||
@Patch("documents/:id")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async review(
|
||||
@Param("id") id: string,
|
||||
@Body() dto: ReviewPolicyDocumentDto,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const doc = await this.policyOcr.review(id, dto, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.document.review", {
|
||||
documentId: id,
|
||||
status: doc.status,
|
||||
});
|
||||
return doc;
|
||||
}
|
||||
|
||||
@Post("documents/:id/reject")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async reject(@Param("id") id: string, @Req() req: Request) {
|
||||
const doc = await this.policyOcr.reject(id, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.document.reject", {
|
||||
documentId: id,
|
||||
});
|
||||
return doc;
|
||||
}
|
||||
|
||||
@Post("batches/:id/confirm")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async confirm(
|
||||
@Param("id") id: string,
|
||||
@Body() dto: ConfirmPolicyBatchDto,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const result = await this.policyOcr.confirmBatch(id, dto, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.batch.confirm", {
|
||||
batchId: id,
|
||||
applied: result.applied,
|
||||
postedTransactions: result.postedTransactions,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
import { Type } from "class-transformer";
|
||||
import {
|
||||
IsArray,
|
||||
IsDateString,
|
||||
IsEnum,
|
||||
IsNumber,
|
||||
IsObject,
|
||||
IsOptional,
|
||||
IsString,
|
||||
MinLength,
|
||||
ValidateNested,
|
||||
} from "class-validator";
|
||||
|
||||
/** One document's confirmed-after-review state. The service reads these
|
||||
* fields and writes them onto either a matched Policy or a freshly created
|
||||
* one. Anything null here is not written. */
|
||||
export class ConfirmPolicyDocumentDto {
|
||||
@IsString() documentId!: string;
|
||||
|
||||
/** Required when creating a new Policy; ignored if `policyId` is set. */
|
||||
@IsOptional() @IsString() customerId?: string;
|
||||
/** Set when the document matched an existing Policy. */
|
||||
@IsOptional() @IsString() policyId?: string;
|
||||
|
||||
@IsOptional() @IsString() policyNumber?: string;
|
||||
@IsOptional() @IsString() insuredName?: string;
|
||||
@IsOptional() @IsString() additionalInsured?: string;
|
||||
@IsOptional() @IsString() agentName?: string;
|
||||
@IsOptional() @IsString() legalAddress?: string;
|
||||
@IsOptional() @IsString() zip?: string;
|
||||
@IsOptional() @IsDateString() policyFrom?: string;
|
||||
@IsOptional() @IsDateString() policyTo?: string;
|
||||
@IsOptional() @IsDateString() policyDate?: string;
|
||||
@IsOptional() @IsEnum(["MXN", "USD", "EUR"]) currency?: "MXN" | "USD" | "EUR";
|
||||
@IsOptional() @IsNumber() netPremium?: number;
|
||||
@IsOptional() @IsNumber() policyFee?: number;
|
||||
@IsOptional() @IsNumber() brokerFee?: number;
|
||||
@IsOptional() @IsNumber() total?: number;
|
||||
@IsOptional() @IsString() premiumPayment?: string;
|
||||
/** Coverages parsed off the PDF, passed through verbatim to Policy.coveragesJson. */
|
||||
@IsOptional() @IsObject() coveragesJson?: unknown;
|
||||
|
||||
/** When true, write a Transaction(domain=INSURANCE, amount=-netPremium)
|
||||
* in addition to creating/updating the Policy. Skipped if netPremium is
|
||||
* null or zero. */
|
||||
@IsOptional() postPremium?: boolean;
|
||||
}
|
||||
|
||||
export class ConfirmPolicyBatchDto {
|
||||
@IsArray()
|
||||
@ValidateNested({ each: true })
|
||||
@Type(() => ConfirmPolicyDocumentDto)
|
||||
documents!: ConfirmPolicyDocumentDto[];
|
||||
}
|
||||
|
||||
/** Staff correction of one document's extracted fields or its match. */
|
||||
export class ReviewPolicyDocumentDto {
|
||||
@IsOptional() @IsString() policyNumber?: string;
|
||||
@IsOptional() @IsString() insuredName?: string;
|
||||
@IsOptional() @IsString() additionalInsured?: string;
|
||||
@IsOptional() @IsString() agentName?: string;
|
||||
@IsOptional() @IsString() legalAddress?: string;
|
||||
@IsOptional() @IsString() zip?: string;
|
||||
@IsOptional() @IsDateString() policyFrom?: string;
|
||||
@IsOptional() @IsDateString() policyTo?: string;
|
||||
@IsOptional() @IsDateString() policyDate?: string;
|
||||
@IsOptional() @IsString() currency?: string;
|
||||
@IsOptional() @IsNumber() netPremium?: number;
|
||||
@IsOptional() @IsNumber() policyFee?: number;
|
||||
@IsOptional() @IsNumber() brokerFee?: number;
|
||||
@IsOptional() @IsNumber() total?: number;
|
||||
@IsOptional() @IsString() premiumPayment?: string;
|
||||
@IsOptional() @IsObject() coveragesJson?: unknown;
|
||||
|
||||
/** Set by the reviewer when the document matched an existing Policy. */
|
||||
@IsOptional() @IsString() matchedPolicyId?: string;
|
||||
/** Set by the reviewer when creating a new Policy. */
|
||||
@IsOptional() @IsString() matchedCustomerId?: string;
|
||||
/** Force-confirm a doc even when the matcher left it ambiguous. */
|
||||
@IsOptional() forceConfirm?: boolean;
|
||||
}
|
||||
|
||||
export class CreatePolicyOcrBatchDto {
|
||||
@IsOptional() @IsString() @MinLength(1) label?: string;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { OcrModule } from "../ocr/ocr.module";
|
||||
import { PolicyOcrController } from "./policy-ocr.controller";
|
||||
import { PolicyOcrService } from "./policy-ocr.service";
|
||||
import { PolicyMatcherService } from "./policy-matcher.service";
|
||||
|
||||
/**
|
||||
* Reuses the OCR seam from OcrModule unchanged: the Tesseract provider is
|
||||
* bound there and `OcrProvider` is the only thing the parsers touch. This
|
||||
* module registers its own controller + service + matcher; nothing about
|
||||
* utility ingestion needs to know about it.
|
||||
*/
|
||||
@Module({
|
||||
imports: [OcrModule],
|
||||
controllers: [PolicyOcrController],
|
||||
providers: [PolicyOcrService, PolicyMatcherService],
|
||||
})
|
||||
export class PolicyOcrModule {}
|
||||
@@ -0,0 +1,720 @@
|
||||
import {
|
||||
BadRequestException,
|
||||
Inject,
|
||||
Injectable,
|
||||
Logger,
|
||||
NotFoundException,
|
||||
} from "@nestjs/common";
|
||||
import { Currency, Prisma } from "@jorgecuadros/database";
|
||||
import { PrismaService } from "../prisma/prisma.service";
|
||||
import { StorageService } from "../storage/storage.service";
|
||||
import type { UploadedFileLike } from "../storage/upload-file";
|
||||
import { OCR_PROVIDER, type OcrPage, type OcrProvider } from "../statements/ocr/ocr.provider";
|
||||
import { parsePolicy } from "./parsers/policy-parser";
|
||||
import { PolicyMatcherService } from "./policy-matcher.service";
|
||||
import type {
|
||||
ConfirmPolicyBatchDto,
|
||||
ConfirmPolicyDocumentDto,
|
||||
ReviewPolicyDocumentDto,
|
||||
} from "./policy-ocr.dto";
|
||||
|
||||
/**
|
||||
* Insurance OCR intake — mirrors the statement pipeline at
|
||||
* `apps/api/src/statements/statements.service.ts`. Reuses the OCR seam and
|
||||
* Tesseract binding unchanged; the parsers and matcher are policy-specific.
|
||||
*
|
||||
* Why a parallel pipeline rather than a column on StatementDocument: the
|
||||
* matcher keys on `Policy.policyNumber`, the confirm step writes to a
|
||||
* different table (`Policy`, not `Transaction`), and the review UI shows
|
||||
* different fields. Sharing one queue would either bloat the row with null
|
||||
* columns or force the review screen to branch on a discriminator — both
|
||||
* worse than a thin second table.
|
||||
*/
|
||||
@Injectable()
|
||||
export class PolicyOcrService {
|
||||
private readonly logger = new Logger(PolicyOcrService.name);
|
||||
|
||||
constructor(
|
||||
private readonly prisma: PrismaService,
|
||||
private readonly storage: StorageService,
|
||||
private readonly matcher: PolicyMatcherService,
|
||||
@Inject(OCR_PROVIDER) private readonly ocr: OcrProvider,
|
||||
) {}
|
||||
|
||||
ocrAvailable(): Promise<boolean> {
|
||||
return this.ocr.available();
|
||||
}
|
||||
|
||||
storageAvailable(): boolean {
|
||||
return this.storage.available;
|
||||
}
|
||||
|
||||
// --- ingest ---------------------------------------------------------------
|
||||
|
||||
async createBatch(
|
||||
files: UploadedFileLike[],
|
||||
uploadedById: string,
|
||||
label?: string,
|
||||
) {
|
||||
if (!files?.length) throw new BadRequestException("No se recibió ningún archivo.");
|
||||
if (!(await this.ocr.available())) {
|
||||
throw new BadRequestException(
|
||||
"El servidor no tiene OCR instalado; no se pueden leer PDFs de pólizas.",
|
||||
);
|
||||
}
|
||||
if (!this.storage.available) {
|
||||
throw new BadRequestException(
|
||||
"El almacenamiento de documentos no está configurado; no se pueden " +
|
||||
"guardar los PDFs escaneados.",
|
||||
);
|
||||
}
|
||||
|
||||
const batch = await this.prisma.policyOcrBatch.create({
|
||||
data: { provider: "GMX", uploadedById, label, fileCount: files.length },
|
||||
});
|
||||
|
||||
const copies = files.map((f) => ({ buffer: f.buffer, name: f.originalname }));
|
||||
void this.process(batch.id, copies).catch(async (err) => {
|
||||
this.logger.error(`Policy OCR batch ${batch.id} failed: ${(err as Error).message}`);
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batch.id },
|
||||
data: { status: "FAILED", error: (err as Error).message },
|
||||
});
|
||||
});
|
||||
|
||||
return batch;
|
||||
}
|
||||
|
||||
/**
|
||||
* Render → text → parse → match, **one PolicyOcrDocument row per uploaded
|
||||
* file**. The GMX certificate is a 2-page PDF where page 1 carries the
|
||||
* contract header and page 2 carries the per-coverage table — both pages
|
||||
* describe the SAME policy, so the parser concatenates them and the
|
||||
* matcher runs once. `pageNumber` on the row is repurposed as the file
|
||||
* ordinal within the batch (1, 2, 3…) — the unique constraint
|
||||
* `(batchId, pageNumber)` still holds and lets a single batch carry many
|
||||
* policies.
|
||||
*
|
||||
* The doc's `storageKey` is the SOURCE PDF (`policy-ocr/{batchId}/source-N.pdf`)
|
||||
* rather than a rendered page image, so the review screen can embed the
|
||||
* exact artifact the office received. The rendered page PNGs are still
|
||||
* stored under `policy-ocr/{batchId}/page-M.png` for any future re-OCR or
|
||||
* image-based audit, but they aren't used as `storageKey` for the document.
|
||||
*/
|
||||
private async process(
|
||||
batchId: string,
|
||||
files: { buffer: Buffer; name?: string }[],
|
||||
) {
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "PROCESSING" },
|
||||
});
|
||||
|
||||
let fileOrdinal = 0;
|
||||
let globalPageOrdinal = 0;
|
||||
for (const file of files) {
|
||||
fileOrdinal += 1;
|
||||
const sourceKey = `policy-ocr/${batchId}/source-${fileOrdinal}.pdf`;
|
||||
await this.storage.put(sourceKey, file.buffer, "application/pdf");
|
||||
|
||||
const pages = await this.ocr.renderPages(file.buffer);
|
||||
const textLayer = await this.ocr.textPages(file.buffer).catch(() => []);
|
||||
|
||||
// One OcrPage per rendered page: text-layer wins when present (cheap,
|
||||
// exact), OCR the rendered image when it isn't. Same precedence rule
|
||||
// as the statement OCR pipeline.
|
||||
const perPageOcr: OcrPage[] = [];
|
||||
for (const [index, image] of pages.entries()) {
|
||||
globalPageOrdinal += 1;
|
||||
const pageStorageKey = `policy-ocr/${batchId}/page-${globalPageOrdinal}.png`;
|
||||
await this.storage.put(pageStorageKey, image, "image/png");
|
||||
|
||||
const embedded = textLayer[index] ?? null;
|
||||
const pageOcr = embedded ?? (await this.ocr.recognize(image));
|
||||
perPageOcr.push(pageOcr);
|
||||
}
|
||||
|
||||
// Concatenate every page's text with a blank line between pages so the
|
||||
// parser's anchored regexes (^From$, ^Currency\s+...) still work
|
||||
// across page boundaries — pdftotext -bbox-layout produces newline-
|
||||
// separated text per page already, the `\n\n` just preserves a clear
|
||||
// boundary in ocrRawText for debugging.
|
||||
const mergedText = perPageOcr.map((p) => p.text).join("\n\n");
|
||||
const avgConfidence =
|
||||
perPageOcr.length === 0
|
||||
? 0
|
||||
: perPageOcr.reduce((s, p) => s + p.confidence, 0) / perPageOcr.length;
|
||||
const synthetic: OcrPage = {
|
||||
text: mergedText,
|
||||
words: [],
|
||||
confidence: avgConfidence,
|
||||
};
|
||||
|
||||
try {
|
||||
const parsed = parsePolicy(synthetic);
|
||||
if (parsed.provider === "") {
|
||||
throw new Error("no se reconoció el proveedor");
|
||||
}
|
||||
const match = await this.matcher.match(parsed);
|
||||
const notes = [...parsed.notes, match.note].filter(Boolean);
|
||||
// Confident when exactly one Policy carries the printed number —
|
||||
// the only unambiguous hit we trust. A new policy (no match) still
|
||||
// needs a customer pick, so it stays in review.
|
||||
const trusted = match.confident && parsed.policyNumber != null;
|
||||
|
||||
await this.prisma.policyOcrDocument.create({
|
||||
data: {
|
||||
batchId,
|
||||
pageNumber: fileOrdinal,
|
||||
storageKey: sourceKey,
|
||||
status: trusted ? "MATCHED" : "NEEDS_REVIEW",
|
||||
ocrRawText: mergedText,
|
||||
ocrConfidence: new Prisma.Decimal(avgConfidence.toFixed(3)),
|
||||
provider: parsed.provider,
|
||||
extractedPolicyNumber: parsed.policyNumber,
|
||||
extractedInsuredName: parsed.insuredName,
|
||||
extractedAdditionalInsured: parsed.additionalInsured,
|
||||
extractedAgentName: parsed.agentName,
|
||||
extractedLegalAddress: parsed.legalAddress,
|
||||
extractedZip: parsed.zip,
|
||||
extractedPolicyFrom: parsed.policyFrom,
|
||||
extractedPolicyTo: parsed.policyTo,
|
||||
extractedPolicyDate: parsed.policyDate,
|
||||
extractedCurrency: parsed.currency,
|
||||
extractedNetPremium:
|
||||
parsed.netPremium != null ? new Prisma.Decimal(parsed.netPremium) : null,
|
||||
extractedPolicyFee:
|
||||
parsed.policyFee != null ? new Prisma.Decimal(parsed.policyFee) : null,
|
||||
extractedBrokerFee:
|
||||
parsed.brokerFee != null ? new Prisma.Decimal(parsed.brokerFee) : null,
|
||||
extractedTotal:
|
||||
parsed.total != null ? new Prisma.Decimal(parsed.total) : null,
|
||||
extractedCoveragesJson: parsed.coverages.length
|
||||
? (parsed.coverages as unknown as Prisma.InputJsonValue)
|
||||
: Prisma.DbNull,
|
||||
extractedPremiumPayment: parsed.premiumPayment,
|
||||
matchedPolicyId: match.policyId,
|
||||
matchedCustomerId: match.customerId,
|
||||
matchCandidates: match.candidates.length
|
||||
? (match.candidates as unknown as Prisma.InputJsonValue)
|
||||
: Prisma.DbNull,
|
||||
matchNote: notes.join("; ").slice(0, 190),
|
||||
},
|
||||
});
|
||||
} catch (err) {
|
||||
// The file as a whole failed to parse (no provider, parse exception).
|
||||
// One OCR_FAILED row per file is the right granularity — the page
|
||||
// images are still on disk for a re-run after a parser fix.
|
||||
await this.prisma.policyOcrDocument.create({
|
||||
data: {
|
||||
batchId,
|
||||
pageNumber: fileOrdinal,
|
||||
storageKey: sourceKey,
|
||||
status: "OCR_FAILED",
|
||||
matchNote: (err as Error).message.slice(0, 190),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "READY_FOR_REVIEW" },
|
||||
});
|
||||
}
|
||||
|
||||
// --- reads ----------------------------------------------------------------
|
||||
|
||||
async listBatches(page: number, pageSize: number) {
|
||||
const [total, items] = await this.prisma.$transaction([
|
||||
this.prisma.policyOcrBatch.count(),
|
||||
this.prisma.policyOcrBatch.findMany({
|
||||
orderBy: { createdAt: "desc" },
|
||||
skip: (page - 1) * pageSize,
|
||||
take: pageSize,
|
||||
include: {
|
||||
uploadedBy: { select: { name: true } },
|
||||
_count: { select: { documents: true } },
|
||||
},
|
||||
}),
|
||||
]);
|
||||
return { items, total, page, pageSize, pageCount: Math.ceil(total / pageSize) };
|
||||
}
|
||||
|
||||
async getBatch(id: string) {
|
||||
const batch = await this.prisma.policyOcrBatch.findUnique({
|
||||
where: { id },
|
||||
include: { uploadedBy: { select: { name: true } } },
|
||||
});
|
||||
if (!batch) throw new NotFoundException("Lote no encontrado.");
|
||||
|
||||
const counts = await this.prisma.policyOcrDocument.groupBy({
|
||||
by: ["status"],
|
||||
where: { batchId: id },
|
||||
_count: { _all: true },
|
||||
});
|
||||
return {
|
||||
...batch,
|
||||
byStatus: Object.fromEntries(counts.map((c) => [c.status, c._count._all])),
|
||||
};
|
||||
}
|
||||
|
||||
async listDocuments(batchId: string) {
|
||||
return this.prisma.policyOcrDocument.findMany({
|
||||
where: { batchId },
|
||||
orderBy: { pageNumber: "asc" },
|
||||
include: {
|
||||
matchedCustomer: { select: { id: true, name: true } },
|
||||
matchedPolicy: {
|
||||
select: {
|
||||
id: true,
|
||||
policyNumber: true,
|
||||
customerId: true,
|
||||
customer: { select: { name: true } },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* The source PDF for the document, so the review screen can show the
|
||||
* exact artifact the office uploaded (the browser's PDF viewer handles
|
||||
* scrolling, zoom, and selection natively). The rendered page PNGs
|
||||
* remain on disk under `policy-ocr/{batchId}/page-N.png` for any
|
||||
* future re-OCR, but the doc row points here at the source.
|
||||
*/
|
||||
async pageImage(documentId: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({
|
||||
where: { id: documentId },
|
||||
select: { storageKey: true },
|
||||
});
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
return this.storage.getStream(doc.storageKey);
|
||||
}
|
||||
|
||||
// --- review ---------------------------------------------------------------
|
||||
|
||||
async review(id: string, dto: ReviewPolicyDocumentDto, reviewedById: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException("Este documento ya fue aplicado.");
|
||||
}
|
||||
|
||||
// Trusting a customer-supplied pair (policyId, customerId) without
|
||||
// cross-check is how a document lands on the wrong customer's ledger;
|
||||
// pin them here from the DB.
|
||||
let matchedPolicyId = dto.matchedPolicyId ?? doc.matchedPolicyId;
|
||||
let matchedCustomerId = doc.matchedCustomerId;
|
||||
|
||||
if (matchedPolicyId) {
|
||||
const p = await this.prisma.policy.findUnique({
|
||||
where: { id: matchedPolicyId },
|
||||
select: { customerId: true },
|
||||
});
|
||||
if (!p) throw new BadRequestException("Póliza no encontrada.");
|
||||
matchedCustomerId = p.customerId;
|
||||
} else if (dto.matchedCustomerId) {
|
||||
const c = await this.prisma.customer.findUnique({
|
||||
where: { id: dto.matchedCustomerId },
|
||||
select: { id: true },
|
||||
});
|
||||
if (!c) throw new BadRequestException("Cliente no encontrado.");
|
||||
matchedCustomerId = c.id;
|
||||
}
|
||||
|
||||
return this.prisma.policyOcrDocument.update({
|
||||
where: { id },
|
||||
data: {
|
||||
extractedPolicyNumber: dto.policyNumber ?? undefined,
|
||||
extractedInsuredName: dto.insuredName ?? undefined,
|
||||
extractedAdditionalInsured: dto.additionalInsured ?? undefined,
|
||||
extractedAgentName: dto.agentName ?? undefined,
|
||||
extractedLegalAddress: dto.legalAddress ?? undefined,
|
||||
extractedZip: dto.zip ?? undefined,
|
||||
extractedPolicyFrom: dto.policyFrom ? new Date(dto.policyFrom) : undefined,
|
||||
extractedPolicyTo: dto.policyTo ? new Date(dto.policyTo) : undefined,
|
||||
extractedPolicyDate: dto.policyDate ? new Date(dto.policyDate) : undefined,
|
||||
extractedCurrency: dto.currency ?? undefined,
|
||||
extractedNetPremium:
|
||||
dto.netPremium != null ? new Prisma.Decimal(dto.netPremium) : undefined,
|
||||
extractedPolicyFee:
|
||||
dto.policyFee != null ? new Prisma.Decimal(dto.policyFee) : undefined,
|
||||
extractedBrokerFee:
|
||||
dto.brokerFee != null ? new Prisma.Decimal(dto.brokerFee) : undefined,
|
||||
extractedTotal:
|
||||
dto.total != null ? new Prisma.Decimal(dto.total) : undefined,
|
||||
extractedCoveragesJson: dto.coveragesJson
|
||||
? (dto.coveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
extractedPremiumPayment: dto.premiumPayment ?? undefined,
|
||||
matchedPolicyId,
|
||||
matchedCustomerId,
|
||||
status: dto.forceConfirm ? "CONFIRMED" : "MATCHED",
|
||||
reviewedById,
|
||||
reviewedAt: new Date(),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
async reject(id: string, reviewedById: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException("Este documento ya fue aplicado.");
|
||||
}
|
||||
return this.prisma.policyOcrDocument.update({
|
||||
where: { id },
|
||||
data: { status: "REJECTED", reviewedById, reviewedAt: new Date() },
|
||||
});
|
||||
}
|
||||
|
||||
// --- confirm --------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Apply every confirmed document: create or update the Policy, attach the
|
||||
* source PDF as a PolicyDocument, and (when staff asked + premium parses)
|
||||
* write a Transaction row. Each step is guarded by status checks so a
|
||||
* double-confirm cannot re-apply a document.
|
||||
*/
|
||||
async confirmBatch(batchId: string, dto: ConfirmPolicyBatchDto, reviewedById: string) {
|
||||
const batch = await this.prisma.policyOcrBatch.findUnique({ where: { id: batchId } });
|
||||
if (!batch) throw new NotFoundException("Lote no encontrado.");
|
||||
|
||||
const results: { documentId: string; policyId: string; postedTransactionId: string | null }[] = [];
|
||||
|
||||
for (const item of dto.documents) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({
|
||||
where: { id: item.documentId },
|
||||
});
|
||||
if (!doc) {
|
||||
throw new BadRequestException(`Documento ${item.documentId} no encontrado.`);
|
||||
}
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException(
|
||||
`El documento página ${doc.pageNumber} ya fue aplicado.`,
|
||||
);
|
||||
}
|
||||
if (!item.policyId && !item.customerId) {
|
||||
throw new BadRequestException(
|
||||
`Documento página ${doc.pageNumber}: falta póliza destino o cliente.`,
|
||||
);
|
||||
}
|
||||
|
||||
// 1. Resolve target Policy (create or update). Field selection: every
|
||||
// non-null `extracted*` on the doc (post-review) is written. Null is
|
||||
// preserved — never overwrite an existing Policy's `netPremium` with
|
||||
// null because the certificate page didn't carry one.
|
||||
let policyId = item.policyId ?? null;
|
||||
|
||||
if (policyId) {
|
||||
const updateData = buildPolicyUpdateFromDoc(item, doc);
|
||||
await this.prisma.policy.update({
|
||||
where: { id: policyId },
|
||||
data: updateData,
|
||||
});
|
||||
} else {
|
||||
// Create under the picked customer. `policyNumber` is the only field
|
||||
// that must be present.
|
||||
if (!item.policyNumber && !doc.extractedPolicyNumber) {
|
||||
throw new BadRequestException(
|
||||
`Documento página ${doc.pageNumber}: falta número de póliza.`,
|
||||
);
|
||||
}
|
||||
const createData = buildPolicyCreateFromDoc(item, doc, item.customerId!);
|
||||
const created = await this.prisma.policy.create({
|
||||
data: createData,
|
||||
});
|
||||
policyId = created.id;
|
||||
}
|
||||
|
||||
// 2. Attach the source PDF as a PolicyDocument. `doc.storageKey`
|
||||
// already points at the exact upload (`policy-ocr/{batchId}/source-N.pdf`)
|
||||
// so the attach is just a stream copy into the policy's namespace —
|
||||
// the previous per-page "which file did this page come from" walk is
|
||||
// gone because one PDF = one doc now.
|
||||
await this.attachSourcePdf(doc.storageKey, policyId);
|
||||
|
||||
// 3. Optionally post the premium to the ledger. Only when staff
|
||||
// explicitly asked (`postPremium` true) and netPremium parses — without
|
||||
// that gate a missing premium would silently book $0.
|
||||
let postedTransactionId: string | null = null;
|
||||
const premium =
|
||||
item.netPremium != null
|
||||
? item.netPremium
|
||||
: doc.extractedNetPremium != null
|
||||
? Number(doc.extractedNetPremium)
|
||||
: null;
|
||||
if (item.postPremium && premium && premium > 0) {
|
||||
const tx = await this.prisma.transaction.create({
|
||||
data: {
|
||||
customerId: (await this.policyCustomerId(policyId))!,
|
||||
domain: "INSURANCE",
|
||||
amount: new Prisma.Decimal(-Math.abs(premium)),
|
||||
transactionDate: doc.extractedPolicyDate ?? doc.extractedPolicyFrom ?? new Date(),
|
||||
currency: (item.currency ??
|
||||
doc.extractedCurrency ??
|
||||
"MXN") as Currency,
|
||||
reference: item.policyNumber ?? doc.extractedPolicyNumber ?? null,
|
||||
period: null,
|
||||
captureSource: "OCR",
|
||||
captureRef: doc.id,
|
||||
message: `Prima de póliza ${item.policyNumber ?? doc.extractedPolicyNumber ?? ""}`,
|
||||
},
|
||||
});
|
||||
postedTransactionId = tx.id;
|
||||
}
|
||||
|
||||
await this.prisma.policyOcrDocument.update({
|
||||
where: { id: doc.id },
|
||||
data: {
|
||||
status: "POSTED",
|
||||
matchedPolicyId: policyId,
|
||||
reviewedById,
|
||||
reviewedAt: new Date(),
|
||||
createdPolicyId: item.policyId ? null : policyId,
|
||||
postedTransactionId,
|
||||
},
|
||||
});
|
||||
|
||||
results.push({
|
||||
documentId: doc.id,
|
||||
policyId,
|
||||
postedTransactionId,
|
||||
});
|
||||
}
|
||||
|
||||
await this.closeIfDone(batchId);
|
||||
|
||||
return {
|
||||
applied: results.length,
|
||||
policies: results.map((r) => r.policyId),
|
||||
postedTransactions: results.filter((r) => r.postedTransactionId).length,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Stream the source PDF (`sourceKey`, set by `process` on the doc row)
|
||||
* into the policy's storage namespace and create a `PolicyDocument`
|
||||
* pointer. Trivial now that the doc row holds the exact source key —
|
||||
* the old per-page "which file did this page come from" walk is gone.
|
||||
*/
|
||||
private async attachSourcePdf(sourceKey: string, policyId: string): Promise<void> {
|
||||
const got = await this.storage.getStream(sourceKey);
|
||||
const chunks: Buffer[] = [];
|
||||
for await (const c of got.stream) chunks.push(c as Buffer);
|
||||
const buf = Buffer.concat(chunks);
|
||||
|
||||
const newKey = `policy/${policyId}/${Date.now()}-${crypto.randomUUID()}.pdf`;
|
||||
await this.storage.put(newKey, buf, "application/pdf");
|
||||
await this.prisma.policyDocument.create({
|
||||
data: {
|
||||
policyId,
|
||||
documentType: "GMX_POLICY",
|
||||
storageKey: newKey,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
private async policyCustomerId(policyId: string): Promise<string | null> {
|
||||
const p = await this.prisma.policy.findUnique({
|
||||
where: { id: policyId },
|
||||
select: { customerId: true },
|
||||
});
|
||||
return p?.customerId ?? null;
|
||||
}
|
||||
|
||||
private async closeIfDone(batchId: string) {
|
||||
const open = await this.prisma.policyOcrDocument.count({
|
||||
where: {
|
||||
batchId,
|
||||
status: { in: ["PENDING_OCR", "NEEDS_REVIEW", "MATCHED", "CONFIRMED"] },
|
||||
},
|
||||
});
|
||||
if (open === 0) {
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "COMPLETED", completedAt: new Date() },
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Map a (post-review) doc + final confirmed fields onto a `Policy.update`
|
||||
* payload. Every field that is null in both inputs is omitted so we never
|
||||
* write null over a value the Policy already carries (the GMX certificate
|
||||
* has no premium — we must not blank the existing Policy.netPremium). */
|
||||
function buildPolicyUpdateFromDoc(
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
doc: {
|
||||
extractedPolicyNumber: string | null;
|
||||
extractedInsuredName: string | null;
|
||||
extractedAdditionalInsured: string | null;
|
||||
extractedAgentName: string | null;
|
||||
extractedLegalAddress: string | null;
|
||||
extractedZip: string | null;
|
||||
extractedPolicyFrom: Date | null;
|
||||
extractedPolicyTo: Date | null;
|
||||
extractedPolicyDate: Date | null;
|
||||
extractedCurrency: string | null;
|
||||
extractedNetPremium: Prisma.Decimal | null;
|
||||
extractedPolicyFee: Prisma.Decimal | null;
|
||||
extractedBrokerFee: Prisma.Decimal | null;
|
||||
extractedTotal: Prisma.Decimal | null;
|
||||
extractedCoveragesJson: Prisma.JsonValue | null;
|
||||
extractedPremiumPayment: string | null;
|
||||
},
|
||||
): Prisma.PolicyUpdateInput {
|
||||
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
|
||||
if (a != null) return new Prisma.Decimal(a);
|
||||
if (b != null) return b;
|
||||
return undefined;
|
||||
};
|
||||
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
|
||||
if (a) return new Date(a);
|
||||
if (b) return b;
|
||||
return undefined;
|
||||
};
|
||||
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
};
|
||||
|
||||
return {
|
||||
policyNumber: strOrUndef(item.policyNumber, doc.extractedPolicyNumber),
|
||||
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
|
||||
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
|
||||
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
|
||||
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
|
||||
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
|
||||
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
|
||||
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
|
||||
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
|
||||
total: numOrUndef(item.total, doc.extractedTotal),
|
||||
// coveragesJson / observations: freeform, keep the GMX data when present.
|
||||
coveragesJson:
|
||||
item.coveragesJson !== undefined
|
||||
? (item.coveragesJson as Prisma.InputJsonValue)
|
||||
: doc.extractedCoveragesJson != null
|
||||
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
// Premium payment cadence ("CONTADO") and insured-name fields land in
|
||||
// `observations` so the PolicyForm's edits stay the source of truth for
|
||||
// structured fields. The reviewer can move them by hand if needed.
|
||||
observations: joinObservations(
|
||||
doc.extractedInsuredName,
|
||||
doc.extractedAdditionalInsured,
|
||||
doc.extractedLegalAddress,
|
||||
doc.extractedZip,
|
||||
doc.extractedPremiumPayment,
|
||||
item,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
/** Same shape as `buildPolicyUpdateFromDoc`, but for `Policy.create`. The
|
||||
* `customerId` is supplied separately and `policyNumber` is required (a
|
||||
* Policy without a number can't be re-matched by the OCR pipeline). */
|
||||
function buildPolicyCreateFromDoc(
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
doc: {
|
||||
extractedPolicyNumber: string | null;
|
||||
extractedInsuredName: string | null;
|
||||
extractedAdditionalInsured: string | null;
|
||||
extractedAgentName: string | null;
|
||||
extractedLegalAddress: string | null;
|
||||
extractedZip: string | null;
|
||||
extractedPolicyFrom: Date | null;
|
||||
extractedPolicyTo: Date | null;
|
||||
extractedPolicyDate: Date | null;
|
||||
extractedCurrency: string | null;
|
||||
extractedNetPremium: Prisma.Decimal | null;
|
||||
extractedPolicyFee: Prisma.Decimal | null;
|
||||
extractedBrokerFee: Prisma.Decimal | null;
|
||||
extractedTotal: Prisma.Decimal | null;
|
||||
extractedCoveragesJson: Prisma.JsonValue | null;
|
||||
extractedPremiumPayment: string | null;
|
||||
},
|
||||
customerId: string,
|
||||
): Prisma.PolicyUncheckedCreateInput {
|
||||
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
|
||||
if (a != null) return new Prisma.Decimal(a);
|
||||
if (b != null) return b;
|
||||
return undefined;
|
||||
};
|
||||
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
|
||||
if (a) return new Date(a);
|
||||
if (b) return b;
|
||||
return undefined;
|
||||
};
|
||||
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
};
|
||||
|
||||
const policyNumber =
|
||||
strOrUndef(item.policyNumber, doc.extractedPolicyNumber);
|
||||
if (!policyNumber) {
|
||||
// Caller already guards this; the throw is a type-narrowing aid.
|
||||
throw new Error("policyNumber required for create");
|
||||
}
|
||||
|
||||
return {
|
||||
policyNumber,
|
||||
customerId,
|
||||
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
|
||||
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
|
||||
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
|
||||
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
|
||||
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
|
||||
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
|
||||
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
|
||||
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
|
||||
total: numOrUndef(item.total, doc.extractedTotal),
|
||||
coveragesJson:
|
||||
item.coveragesJson !== undefined
|
||||
? (item.coveragesJson as Prisma.InputJsonValue)
|
||||
: doc.extractedCoveragesJson != null
|
||||
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
observations: joinObservations(
|
||||
doc.extractedInsuredName,
|
||||
doc.extractedAdditionalInsured,
|
||||
doc.extractedLegalAddress,
|
||||
doc.extractedZip,
|
||||
doc.extractedPremiumPayment,
|
||||
item,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
function joinObservations(
|
||||
insured: string | null,
|
||||
additional: string | null,
|
||||
address: string | null,
|
||||
zip: string | null,
|
||||
premiumPayment: string | null,
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
): string | undefined {
|
||||
const lines: string[] = [];
|
||||
const insuredName = strOrUndefDb(item.insuredName, insured);
|
||||
if (insuredName) lines.push(`Asegurado: ${insuredName}`);
|
||||
const additionalInsured = strOrUndefDb(item.additionalInsured, additional);
|
||||
if (additionalInsured) lines.push(`Asegurado adicional: ${additionalInsured}`);
|
||||
const legalAddress = strOrUndefDb(item.legalAddress, address);
|
||||
if (legalAddress) lines.push(`Dirección: ${legalAddress}`);
|
||||
const zipVal = strOrUndefDb(item.zip, zip);
|
||||
if (zipVal) lines.push(`C.P.: ${zipVal}`);
|
||||
const cadence = strOrUndefDb(item.premiumPayment, premiumPayment);
|
||||
if (cadence) lines.push(`Pago de prima: ${cadence}`);
|
||||
return lines.length ? lines.join("\n") : undefined;
|
||||
}
|
||||
|
||||
function strOrUndefDb(a: string | undefined, b: string | null): string | undefined {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
}
|
||||
@@ -6,8 +6,10 @@
|
||||
* The shipped implementation is self-hosted Tesseract (see tesseract.provider).
|
||||
* That choice is evidence-based rather than assumed: run against 46 pages of
|
||||
* real scanned CFE, CESPT and Telnor statements, it identified the provider on
|
||||
* 46/46 and extracted a usable account reference on 43/46, which is well past
|
||||
* the bar for a queue whose whole point is that a human confirms every row. A
|
||||
* 46/46 and extracted a usable account reference on 43/46, and on a later
|
||||
* corpus of 19 scanned municipal predial receipts it read the provider on
|
||||
* 19/19 and an identifier on 18/19 — well past the bar for a queue whose whole
|
||||
* point is that a human confirms every row. A
|
||||
* managed document-extraction API (Textract, Document Intelligence, Document
|
||||
* AI) fits behind this same interface if per-page accuracy ever proves
|
||||
* insufficient, with no schema change — but at 300+ pages/month/company it
|
||||
@@ -31,10 +33,10 @@ export interface OcrPage {
|
||||
/** Full page text, reading order, newline-separated. */
|
||||
text: string;
|
||||
/**
|
||||
* Word boxes. Needed because two of the three real layouts are *tables* —
|
||||
* the CESPT "RECIBO" prints `No. DE CUENTA` as a column header with the
|
||||
* value in the row beneath it, which line-oriented text cannot associate.
|
||||
* Parsers fall back to geometry for exactly those fields.
|
||||
* Word boxes. Needed because several of the real layouts are *tables* — the
|
||||
* CESPT "RECIBO" prints `No. DE CUENTA` as a column header with the value in
|
||||
* the row beneath it, which line-oriented text cannot associate. Parsers fall
|
||||
* back to geometry for exactly those fields.
|
||||
*/
|
||||
words: OcrWord[];
|
||||
/** Mean word confidence across the page, 0..1. */
|
||||
@@ -48,6 +50,22 @@ export interface OcrProvider {
|
||||
renderPages(pdf: Buffer): Promise<Buffer[]>;
|
||||
/** OCR a single rendered page image. */
|
||||
recognize(pageImage: Buffer): Promise<OcrPage>;
|
||||
/**
|
||||
* Read a PDF's own text layer, one entry per page, `null` where the page has
|
||||
* none worth using.
|
||||
*
|
||||
* Not every statement is a scan. The gas company e-mails born-digital CFDI
|
||||
* invoices whose text is already exact and already positioned — running those
|
||||
* through a rasteriser and a character recogniser can only lose information
|
||||
* (one sample turned `MEDIDOR: VM01014426` into `ar (LTR): 014420`) while
|
||||
* costing about a minute of CPU per page for the privilege. Where the layer
|
||||
* exists it is strictly better input for the same parsers, so it is tried
|
||||
* first and OCR remains the fallback for genuine scans.
|
||||
*
|
||||
* Positions are reported in the same pixel space `recognize` uses, so the
|
||||
* geometric helpers in the parsers work unchanged on either source.
|
||||
*/
|
||||
textPages(pdf: Buffer): Promise<(OcrPage | null)[]>;
|
||||
}
|
||||
|
||||
export const OCR_PROVIDER = Symbol("OCR_PROVIDER");
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
import { parseBboxLayout } from "./tesseract.provider";
|
||||
|
||||
/**
|
||||
* Shaped like real `pdftotext -bbox-layout` output: the gas invoice lays its
|
||||
* header out as two columns of independent text flows, so poppler puts a label
|
||||
* and the value printed beside it in *different* `<line>` elements. Trusting
|
||||
* that grouping is what left `PERIODO FACTURADO` with no value next to it and
|
||||
* every period field empty on a batch whose text was perfectly readable.
|
||||
*/
|
||||
function word(x: number, y: number, text: string): string {
|
||||
return `<word xMin="${x}" yMin="${y}" xMax="${x + 20}" yMax="${y + 8}">${text}</word>`;
|
||||
}
|
||||
|
||||
function doc(...lines: string[]): string {
|
||||
return `<doc><page width="612" height="792">${lines
|
||||
.map((l) => `<flow><block><line>${l}</line></block></flow>`)
|
||||
.join("")}</page></doc>`;
|
||||
}
|
||||
|
||||
/** Enough words on the page to clear the "is this a real text layer" floor. */
|
||||
function padding(): string {
|
||||
return Array.from({ length: 50 }, (_, i) => word(10, 400 + i * 10, `w${i}`)).join("");
|
||||
}
|
||||
|
||||
describe("parseBboxLayout", () => {
|
||||
it("rejoins a label with the value printed beside it in another flow", () => {
|
||||
const [page] = parseBboxLayout(
|
||||
doc(
|
||||
word(20, 100, "PERIODO") + word(45, 100, "FACTURADO:"),
|
||||
word(300, 100.4, "20260630-20260630"),
|
||||
padding(),
|
||||
),
|
||||
1,
|
||||
);
|
||||
expect(page).not.toBeNull();
|
||||
expect(page!.text).toContain("PERIODO FACTURADO: 20260630-20260630");
|
||||
});
|
||||
|
||||
it("keeps genuinely separate lines apart", () => {
|
||||
const [page] = parseBboxLayout(
|
||||
doc(word(20, 100, "Cuenta:") + word(80, 100, "0900003463"), word(20, 130, "Nombre:"), padding()),
|
||||
1,
|
||||
);
|
||||
expect(page!.text.split("\n")).toContain("Cuenta: 0900003463");
|
||||
expect(page!.text.split("\n")).toContain("Nombre:");
|
||||
});
|
||||
|
||||
it("scales point coordinates into the render's pixel space", () => {
|
||||
// Word boxes have to land in the same coordinate space tesseract reports,
|
||||
// or the geometric helpers the parsers share silently stop finding values.
|
||||
const [page] = parseBboxLayout(doc(word(72, 144, "X") + padding()), 300 / 72);
|
||||
const x = page!.words.find((w) => w.text === "X")!;
|
||||
expect(x.left).toBeCloseTo(300);
|
||||
expect(x.top).toBeCloseTo(600);
|
||||
});
|
||||
|
||||
it("reports no text layer for a scan carrying a few stray glyphs", () => {
|
||||
expect(parseBboxLayout(doc(word(10, 10, "3") + word(40, 10, "of") + word(60, 10, "5")), 1)).toEqual([
|
||||
null,
|
||||
]);
|
||||
});
|
||||
|
||||
it("decodes the entities poppler escapes", () => {
|
||||
const [page] = parseBboxLayout(doc(word(10, 10, "A&B") + padding()), 1);
|
||||
expect(page!.text).toContain("A&B");
|
||||
});
|
||||
});
|
||||
@@ -105,6 +105,37 @@ export class TesseractOcrProvider implements OcrProvider {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* `pdftotext -bbox-layout` — the same poppler package `pdftoppm` comes from,
|
||||
* so this costs no extra dependency in the runtime image.
|
||||
*
|
||||
* A page is only accepted when it carries a real text layer. Scanned PDFs
|
||||
* frequently contain a handful of stray glyphs (a scanner watermark, a page
|
||||
* number stamped by the MFP), and treating those as the page's text would
|
||||
* hand every parser an almost-empty string and silently take OCR out of the
|
||||
* loop — so a floor of MIN_TEXT_WORDS words has to be present before the
|
||||
* layer is believed.
|
||||
*/
|
||||
async textPages(pdf: Buffer): Promise<(OcrPage | null)[]> {
|
||||
await this.require();
|
||||
return this.scratch(async (dir) => {
|
||||
const src = join(dir, "in.pdf");
|
||||
await writeFile(src, pdf);
|
||||
const out = join(dir, "out.html");
|
||||
try {
|
||||
await run("pdftotext", ["-bbox-layout", src, out]);
|
||||
} catch (err) {
|
||||
this.logger.warn(
|
||||
`pdftotext failed; falling back to OCR for this file: ${(err as Error).message}`,
|
||||
);
|
||||
return [];
|
||||
}
|
||||
// Points to pixels at the render DPI, so word boxes from either source
|
||||
// land in one coordinate space and `valueUnder`'s thresholds hold.
|
||||
return parseBboxLayout(await readFile(out, "utf8"), this.dpi / 72);
|
||||
});
|
||||
}
|
||||
|
||||
async recognize(pageImage: Buffer): Promise<OcrPage> {
|
||||
await this.require();
|
||||
return this.scratch(async (dir) => {
|
||||
@@ -136,6 +167,130 @@ export class TesseractOcrProvider implements OcrProvider {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Below this many words a "text layer" is scanner debris, not a document.
|
||||
* The real born-digital samples carry 400+ words a page; the scanned ones
|
||||
* carry none at all, so the exact threshold is not delicate.
|
||||
*/
|
||||
const MIN_TEXT_WORDS = 40;
|
||||
|
||||
const ENTITIES: Record<string, string> = {
|
||||
amp: "&",
|
||||
lt: "<",
|
||||
gt: ">",
|
||||
quot: '"',
|
||||
apos: "'",
|
||||
};
|
||||
|
||||
function decodeEntities(s: string): string {
|
||||
return s.replace(/&(#x?[0-9a-fA-F]+|[a-z]+);/g, (whole, body: string) => {
|
||||
if (body[0] === "#") {
|
||||
const code =
|
||||
body[1] === "x" || body[1] === "X"
|
||||
? parseInt(body.slice(2), 16)
|
||||
: parseInt(body.slice(1), 10);
|
||||
return Number.isFinite(code) ? String.fromCodePoint(code) : whole;
|
||||
}
|
||||
return ENTITIES[body] ?? whole;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn `pdftotext -bbox-layout`'s XHTML into one OcrPage per PDF page.
|
||||
*
|
||||
* Parsed with regexes rather than an XML library on purpose: the output is
|
||||
* machine-generated by poppler with a fixed element shape (`page` > `flow` >
|
||||
* `block` > `line` > `word`), and the alternative is a parser dependency in
|
||||
* the API for one file format read in one place. Only `page` and `word` are
|
||||
* consulted — see below for why poppler's own `line` grouping is discarded.
|
||||
*
|
||||
* `confidence` is 1 for every word: these are the document's own characters,
|
||||
* not a recognition guess.
|
||||
*/
|
||||
export function parseBboxLayout(xhtml: string, scale: number): (OcrPage | null)[] {
|
||||
const pages: (OcrPage | null)[] = [];
|
||||
|
||||
for (const pageMatch of xhtml.matchAll(/<page\b[^>]*>([\s\S]*?)<\/page>/g)) {
|
||||
const words: OcrWord[] = [];
|
||||
|
||||
for (const w of pageMatch[1].matchAll(
|
||||
/<word\s+xMin="([\d.eE+-]+)"\s+yMin="([\d.eE+-]+)"\s+xMax="([\d.eE+-]+)"\s+yMax="([\d.eE+-]+)"\s*>([\s\S]*?)<\/word>/g,
|
||||
)) {
|
||||
const text = decodeEntities(w[5]).trim();
|
||||
if (!text) continue;
|
||||
const left = Number(w[1]) * scale;
|
||||
const top = Number(w[2]) * scale;
|
||||
words.push({
|
||||
text,
|
||||
left,
|
||||
top,
|
||||
width: Number(w[3]) * scale - left,
|
||||
height: Number(w[4]) * scale - top,
|
||||
confidence: 1,
|
||||
});
|
||||
}
|
||||
|
||||
pages.push(
|
||||
words.length >= MIN_TEXT_WORDS
|
||||
? { text: toVisualRows(words), words, confidence: 1 }
|
||||
: null,
|
||||
);
|
||||
}
|
||||
|
||||
return pages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reassemble words into the rows a reader sees, left to right.
|
||||
*
|
||||
* Poppler's own `<line>` grouping cannot be used for this. It groups by text
|
||||
* flow, and these invoices lay their fields out as two columns of independent
|
||||
* flows — so `PERIODO FACTURADO:` and the `20260630-20260630` printed beside
|
||||
* it end up in different `<line>` elements, and every label-then-value pattern
|
||||
* in the parsers misses a value that is plainly there on the page. Regrouping
|
||||
* by vertical position restores the adjacency, and matches what tesseract
|
||||
* hands back for the scanned version of the same layout.
|
||||
*
|
||||
* Rows are cut when a word's vertical centre leaves the band established by
|
||||
* the row's first word, which tolerates the sub-pixel baseline differences
|
||||
* between fonts on one line without merging two genuinely separate lines.
|
||||
*/
|
||||
function toVisualRows(words: OcrWord[]): string {
|
||||
const centre = (w: OcrWord) => w.top + w.height / 2;
|
||||
const sorted = [...words].sort((a, b) => centre(a) - centre(b) || a.left - b.left);
|
||||
|
||||
const rows: OcrWord[][] = [];
|
||||
let current: OcrWord[] = [];
|
||||
let band = 0;
|
||||
|
||||
for (const w of sorted) {
|
||||
if (!current.length) {
|
||||
current = [w];
|
||||
band = centre(w);
|
||||
continue;
|
||||
}
|
||||
// Half the word's own height: tall headings and body text both sit within
|
||||
// their own line's band, and neither reaches into the next one.
|
||||
if (Math.abs(centre(w) - band) <= Math.max(w.height, current[0].height) / 2) {
|
||||
current.push(w);
|
||||
} else {
|
||||
rows.push(current);
|
||||
current = [w];
|
||||
band = centre(w);
|
||||
}
|
||||
}
|
||||
if (current.length) rows.push(current);
|
||||
|
||||
return rows
|
||||
.map((r) =>
|
||||
[...r]
|
||||
.sort((a, b) => a.left - b.left)
|
||||
.map((w) => w.text)
|
||||
.join(" "),
|
||||
)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn tesseract's TSV into words plus reassembled text.
|
||||
*
|
||||
|
||||
@@ -0,0 +1,270 @@
|
||||
import type { OcrPage } from "../ocr/ocr.provider";
|
||||
import {
|
||||
detectProvider,
|
||||
normalizeCadastralKey,
|
||||
normalizeZofematKey,
|
||||
parseStatement,
|
||||
} from "./statement-parser";
|
||||
|
||||
/**
|
||||
* Every string in this file is a verbatim excerpt of what the OCR engine
|
||||
* actually returned for a real receipt — misreads, dropped spaces, mangled
|
||||
* accents and all. That is the point: these are the specific ways these five
|
||||
* layouts have been observed to fail, and the assertions pin down what the
|
||||
* parser is supposed to do about each one. Inventing clean input here would
|
||||
* test nothing, because clean input was never the problem.
|
||||
*/
|
||||
function page(text: string): OcrPage {
|
||||
return { text, words: [], confidence: 0.9 };
|
||||
}
|
||||
|
||||
describe("detectProvider", () => {
|
||||
it("reads a Rosarito predial receipt as predial, not as a water bill", () => {
|
||||
// "Clave Catastral" is also a CESPT structural marker, so a predial page
|
||||
// whose header OCR'd badly must still not be claimed by the CESPT rule.
|
||||
expect(
|
||||
detectProvider(
|
||||
"e | Clave Catastral. KP-128-105 IMPUESTO PREDIAL ea rita\n" +
|
||||
"TASA | VALOR FISCAL | BIMESTRES | INCISO. | IMPUESTO",
|
||||
),
|
||||
).toBe("PREDIAL ROSARITO");
|
||||
});
|
||||
|
||||
it("keeps telling the three municipalities apart by their RFC", () => {
|
||||
expect(detectProvider("R.F.C. ATB-541201-KK2")).toBe("PREDIAL TIJUANA");
|
||||
expect(detectProvider("R.F.C. AMP-981201-HJ4")).toBe("PREDIAL ROSARITO");
|
||||
expect(detectProvider("MEN-540301-9J5")).toBe("PREDIAL ENSENADA");
|
||||
});
|
||||
|
||||
it("does not let the CFE rule claim a gas bill over 'PERIODO FACTURADO'", () => {
|
||||
expect(
|
||||
detectProvider("Orden de Facturación: 000009801640\nPERIODO FACTURADO: 20260630-20260630"),
|
||||
).toBe("GAS TIJUANA");
|
||||
});
|
||||
});
|
||||
|
||||
describe("normalizeCadastralKey", () => {
|
||||
it("keeps a letter in the third position instead of digitising it", () => {
|
||||
// `MMB01041` is a real key on file; mapping its B to 8 produced a key that
|
||||
// matches no property at all.
|
||||
expect(normalizeCadastralKey("MM-B01-041", [])).toBe("MMB01041");
|
||||
});
|
||||
|
||||
it("repairs the spurious I tesseract inserts into the prefix", () => {
|
||||
expect(normalizeCadastralKey("MIM-200-010", [])).toBe("MM200010");
|
||||
});
|
||||
|
||||
it("digitises confusable glyphs from position four onward", () => {
|
||||
expect(normalizeCadastralKey("KP-1O8-O45", [])).toBe("KP108045");
|
||||
});
|
||||
|
||||
it("flags a prefix it had to truncate", () => {
|
||||
const notes: string[] = [];
|
||||
expect(normalizeCadastralKey("KPX-128-106", notes)).toBe("KP128106");
|
||||
expect(notes).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parsePredialTijuana", () => {
|
||||
const TIJUANA = page(
|
||||
"Hats | AYUNTAMIENTO DE TIJUANA, BC $2,613.00 23/01/2026\n" +
|
||||
"y) TELEFONO: 973-7000 R.F.C. ATB-541201-KK2\n" +
|
||||
"ER AÑO VALOR FISCAL TASA IMPUESTO |CONCEPTO IMPORTE\n" +
|
||||
"ED ca 2026 1,207,15778 246 2,969.61 1102 - IMPUESTO PREDIAL 2,969.61\n" +
|
||||
"55164964310126000002613000054192\n" +
|
||||
"se 0 O (54427 [a] | TOTALAPAGAR: 2,613.00\n" +
|
||||
"Dc 1097 : FECHA VENCE : 31/ENE/2026",
|
||||
);
|
||||
|
||||
it("splits the payment barcode into account, deadline and amount", () => {
|
||||
const p = parseStatement(TIJUANA);
|
||||
expect(p.provider).toBe("PREDIAL TIJUANA");
|
||||
expect(p.serviceKind).toBe("PROPERTY_TAX");
|
||||
expect(p.accountRef).toBe("55164964");
|
||||
expect(p.amount).toBe(2613);
|
||||
expect(p.dueDate?.toISOString().slice(0, 10)).toBe("2026-01-31");
|
||||
expect(p.period).toBe("2026");
|
||||
});
|
||||
|
||||
it("reads the printed total even when the space in the label is lost", () => {
|
||||
// The real page OCR'd the label as "TOTALAPAGAR:", and it is that reading
|
||||
// that cross-checks the barcode's amount.
|
||||
expect(parseStatement(TIJUANA).crossChecked).toBe(true);
|
||||
});
|
||||
|
||||
it("refuses to trust a barcode the printed total contradicts", () => {
|
||||
const p = parseStatement(
|
||||
page(
|
||||
"R.F.C. ATB-541201-KK2\n" +
|
||||
"55164964310126000002613000054192\n" +
|
||||
"TOTAL A PAGAR: 9,613.00\nFECHA VENCE : 31/ENE/2026",
|
||||
),
|
||||
);
|
||||
expect(p.crossChecked).toBe(false);
|
||||
expect(p.notes.join(" ")).toContain("no coincide");
|
||||
});
|
||||
});
|
||||
|
||||
describe("parsePredialRosarito", () => {
|
||||
it("takes the rounded Total, not the Sub Total printed above it", () => {
|
||||
const p = parseStatement(
|
||||
page(
|
||||
"AYUNTAMIENTO MUNICIPAL DE PLAYAS DE ROSARITO, B.C.\n" +
|
||||
"Ce Clave Catastral: + JR-400-008 7 | IMPUESTO PREDIAL\n" +
|
||||
"SUPERFICIE: 228.31 ZONA 30025 “Redondeo IT049 -$0.39 Sub Total $5,409.39\n" +
|
||||
"¿XTEMPORANEO DESPUES DE: 31/01/2026 Elaboro: MGLG\n" +
|
||||
"Total | $5,409.00\n" +
|
||||
"| Periodo por Pagar: 2026/1 2026/6",
|
||||
),
|
||||
);
|
||||
expect(p.cadastralKey).toBe("JR400008");
|
||||
expect(p.amount).toBe(5409);
|
||||
expect(p.dueDate?.toISOString().slice(0, 10)).toBe("2026-01-31");
|
||||
expect(p.period).toBe("2026");
|
||||
});
|
||||
|
||||
it("is not fooled by the unspaced 'SubTotal' spelling", () => {
|
||||
// This exact page read $9,624.85 off a receipt for $9,625.00 while the
|
||||
// lookbehind still assumed a space.
|
||||
const p = parseStatement(
|
||||
page(
|
||||
"AMP-981201-HJ4 IMPUESTO PREDIAL\n" +
|
||||
"SUPERFICIE. 367.62 ZONA:30151 | Redondco 17049 $0.15 SubTotal $9,624.85\n" +
|
||||
": Total | $9,625.00",
|
||||
),
|
||||
);
|
||||
expect(p.amount).toBe(9625);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parsePredialEnsenada", () => {
|
||||
const totals = (tail: string) =>
|
||||
page(
|
||||
"IMPRESION MAQUINA REGISTRADORA ez | MUNICIPIO DE ENSENADA\n" +
|
||||
"+7] DATOS. DEL.CAUSANTE alta A pe CLAVE MM-200-010 2 CUENTA\n" +
|
||||
`ES g € S| TOTALES 12,744.47 0.00 0.00 324.56 0.00 13,069.03 ${tail} |`,
|
||||
);
|
||||
|
||||
it("reads the paid total off the TOTALES row however the label OCR'd", () => {
|
||||
expect(parseStatement(totals("TOTA LA A $5,797.00")).amount).toBe(5797);
|
||||
expect(parseStatement(totals("orAL: M7 z] $14,414.00")).amount).toBe(14414);
|
||||
expect(parseStatement(totals("| TOTAL: = $6 246.00")).amount).toBe(6246);
|
||||
});
|
||||
|
||||
it("reports no amount rather than one whose $ was misread as an 8", () => {
|
||||
// `TOTAL: A 82,203.00` is a $2,203.00 receipt. Posting $82,203 would look
|
||||
// entirely ordinary in the ledger, so this page must go to review instead.
|
||||
const p = parseStatement(totals("TOTAL: A 82,203.00"));
|
||||
expect(p.amount).toBeNull();
|
||||
expect(p.notes.join(" ")).toContain("capturarlo a mano");
|
||||
});
|
||||
|
||||
it("never falls back to the assessed total on the same row", () => {
|
||||
expect(parseStatement(totals("yo: se TE= 58/4690]")).amount).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseGas", () => {
|
||||
const gas = (...cuentas: string[]) =>
|
||||
page(
|
||||
"GTI4608032K2 COMPAÑIA DE GAS DE TIJUANA\n" +
|
||||
"Fecha de Vencimiento: 2026/08/08\n" +
|
||||
cuentas.map((c) => `Cuenta: ${c}`).join("\n") +
|
||||
"\nPERIODO FACTURADO: 20260630-20260630\nTOTAL A PAGAR: $275.82",
|
||||
);
|
||||
|
||||
it("strips the printed leading zero to the stored account number", () => {
|
||||
const p = parseStatement(gas("0900003463", "0900003463", "0900003463"));
|
||||
expect(p.serviceKind).toBe("GAS");
|
||||
expect(p.accountRef).toBe("900003463");
|
||||
expect(p.amount).toBe(275.82);
|
||||
expect(p.dueDate?.toISOString().slice(0, 10)).toBe("2026-08-08");
|
||||
expect(p.period).toBe("2026-06");
|
||||
expect(p.crossChecked).toBe(true);
|
||||
});
|
||||
|
||||
it("takes the majority reading but still sends a disagreement to review", () => {
|
||||
const p = parseStatement(gas("0900003463", "0900003463", "0900003468"));
|
||||
expect(p.accountRef).toBe("900003463");
|
||||
expect(p.crossChecked).toBe(false);
|
||||
});
|
||||
|
||||
it("claims no cross-check from a single printing", () => {
|
||||
expect(parseStatement(gas("0900003463")).crossChecked).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseZonaFederal", () => {
|
||||
/**
|
||||
* The Tijuana zona federal receipt, trimmed to the rows the parser reads.
|
||||
* Verbatim from page 7 of the August 2026 batch, including the two ways the
|
||||
* heading OCR'd: the clave line is struck through by the office's own
|
||||
* highlighter, which is what cost two of eight pages their concession clave.
|
||||
*/
|
||||
const zf = (clave: string, body = "") =>
|
||||
page(
|
||||
"ESIZ <pYl Av. Independencia y Esq. Paseo del CentenaxiaiiArlhnto de Tijuana, B.C.\n" +
|
||||
"Teléfono: 9737000 R.F.C. ATB-541201-BK2 0070000146 12:54 PM\n" +
|
||||
"Zona Federal Marítimo Terrestre\n" +
|
||||
`${clave} Nombre: DENNIS JOHN SEIN Concesión:\n` +
|
||||
"Periodo Construcción Tasa Ornato Tasa Impuesto Actualiza. Recargo Multa Importe\n" +
|
||||
"2026-2 / 2026-2 316.40 35.00 0.00 12.11 1,845.66 0.00 27.13 1,000.00 2,872.79\n" +
|
||||
"SubTotal 1,845.66 0.00 27.13 1,000.00 2,872.79\n" +
|
||||
"Concepto: Derechos de ocupación de Zona Federal Marítimo Terrestre\n" +
|
||||
body,
|
||||
);
|
||||
|
||||
it("is not claimed by the predial parser that shares its RFC and header", () => {
|
||||
// Tijuana bills predial and zona federal from the same treasury, so
|
||||
// "Ayuntamiento de Tijuana" and ATB-541201 identify neither on their own.
|
||||
expect(detectProvider("R.F.C. ATB-541201-BK2\nZona Federal Marítimo Terrestre")).toBe(
|
||||
"ZONA FEDERAL TIJUANA",
|
||||
);
|
||||
expect(parseStatement(zf("Clave: 14-D -014")).serviceKind).toBe("FEDERAL_ZONE");
|
||||
});
|
||||
|
||||
it("still recognises the layout when the heading itself did not survive OCR", () => {
|
||||
// Real: page 1 came back as "Zona Ledera) Maritimo Terrestre".
|
||||
expect(
|
||||
detectProvider("Zona Ledera) Maritimo Terrestre\nClave EJ -012% Nombre: STEFAN"),
|
||||
).toBe("ZONA FEDERAL TIJUANA");
|
||||
});
|
||||
|
||||
it("reads the clave through the loose spacing the receipt prints", () => {
|
||||
expect(parseStatement(zf("Clave: 14-D -014")).accountRef).toBe("14D014");
|
||||
expect(parseStatement(zf("Clave: 14-A-119")).accountRef).toBe("14A119");
|
||||
});
|
||||
|
||||
it("keeps the letter instead of digitising it", () => {
|
||||
// toDigits maps D to 0 and B to 8; a real 14-D -014 must not become 140014.
|
||||
expect(normalizeZofematKey("14-D -014")).toBe("14D014");
|
||||
expect(normalizeZofematKey("12-B -013")).toBe("12B013");
|
||||
});
|
||||
|
||||
it("takes the payable amount from the SubTotal row, rounded to whole pesos", () => {
|
||||
// The municipality rounds and prints the difference as "Ajuste Ley Hacienda
|
||||
// Mpal"; 2,872.79 is charged as $2,873.00.
|
||||
expect(parseStatement(zf("Clave: 14-D -014")).amount).toBe(2873);
|
||||
});
|
||||
|
||||
it("prefers the printed total and cross-checks it against the subtotal", () => {
|
||||
const p = parseStatement(zf("Clave: 14-D -014", "Total a pagar $2,873.00"));
|
||||
expect(p.amount).toBe(2873);
|
||||
expect(p.crossChecked).toBe(true);
|
||||
});
|
||||
|
||||
it("sends a printed total that contradicts the subtotal to review", () => {
|
||||
const p = parseStatement(zf("Clave: 14-D -014", "Total a pagar $2,973.00"));
|
||||
expect(p.crossChecked).toBe(false);
|
||||
});
|
||||
|
||||
it("translates the printed bimester into the ledger's own vocabulary", () => {
|
||||
expect(parseStatement(zf("Clave: 14-D -014")).period).toBe("MAR/APR");
|
||||
});
|
||||
|
||||
it("leaves the clave blank rather than guessing when the marker ate it", () => {
|
||||
const p = parseStatement(zf("Clave EJ -012%"));
|
||||
expect(p.accountRef).toBeNull();
|
||||
expect(p.notes.join(" ")).toContain("clave");
|
||||
});
|
||||
});
|
||||
@@ -7,7 +7,11 @@ import type { OcrPage, OcrWord } from "../ocr/ocr.provider";
|
||||
* like with like and never has to know about provider-specific formatting.
|
||||
*/
|
||||
export interface ParsedStatement {
|
||||
/** "CFE" | "CESPT" | "TELNOR", or null when no parser claimed the page. */
|
||||
/**
|
||||
* "CFE" | "CESPT" | "TELNOR" | "GAS TIJUANA" | "PREDIAL TIJUANA" |
|
||||
* "PREDIAL ROSARITO" | "PREDIAL ENSENADA" | "ZONA FEDERAL TIJUANA", or null
|
||||
* when no parser claimed the page.
|
||||
*/
|
||||
provider: string | null;
|
||||
serviceKind: ServiceKind | null;
|
||||
accountRef: string | null;
|
||||
@@ -90,6 +94,16 @@ function firstMatch(text: string, patterns: RegExp[]): string | null {
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Every capture of `pattern` across the page, in order. */
|
||||
function allMatches(text: string, pattern: RegExp): string[] {
|
||||
const out: string[] = [];
|
||||
const re = new RegExp(pattern.source, pattern.flags.includes("g") ? pattern.flags : `${pattern.flags}g`);
|
||||
for (const m of text.matchAll(re)) {
|
||||
if (m[1]) out.push(m[1].trim());
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const MONTHS: Record<string, number> = {
|
||||
ENE: 0, FEB: 1, MAR: 2, ABR: 3, MAY: 4, JUN: 5,
|
||||
JUL: 6, AGO: 7, SEP: 8, OCT: 9, NOV: 10, DIC: 11,
|
||||
@@ -104,15 +118,16 @@ export function parseDate(raw: string | null | undefined): Date | null {
|
||||
let m = s.match(/^(\d{1,2})\/(\d{1,2})\/(\d{4})$/);
|
||||
if (m) return utc(+m[3], +m[2] - 1, +m[1]);
|
||||
|
||||
// 22-JUL-2026 / 22 JUN 26
|
||||
m = s.match(/^(\d{1,2})[-\s]([A-Z]{3})[A-Z]*[-\s](\d{2,4})$/);
|
||||
// 22-JUL-2026 / 22 JUN 26 / 31/ENE/2026 (Tijuana predial)
|
||||
m = s.match(/^(\d{1,2})[-\s/]([A-Z]{3})[A-Z]*[-\s/](\d{2,4})$/);
|
||||
if (m && MONTHS[m[2]] !== undefined) {
|
||||
const y = m[3].length === 2 ? 2000 + +m[3] : +m[3];
|
||||
return utc(y, MONTHS[m[2]], +m[1]);
|
||||
}
|
||||
|
||||
// 2026-07-22 (already normalised, e.g. decoded from a barcode)
|
||||
m = s.match(/^(\d{4})-(\d{2})-(\d{2})$/);
|
||||
// 2026-07-22 (already normalised, e.g. decoded from a barcode) and the
|
||||
// 2026/08/08 the gas bill prints — same field order, different separator.
|
||||
m = s.match(/^(\d{4})[-/](\d{2})[-/](\d{2})$/);
|
||||
if (m) return utc(+m[1], +m[2] - 1, +m[3]);
|
||||
|
||||
return null;
|
||||
@@ -174,9 +189,44 @@ const BRAND: [string, RegExp][] = [
|
||||
["CFE", /comisi[oó]n federal de electricidad|CFE.?contigo|Suministrador de Servicios/i],
|
||||
["CESPT", /CESPT|COMISI[OÓ]N ESTATAL DE SERVICIOS/i],
|
||||
["TELNOR", /TELNOR|TELEFONOS DEL NOROESTE/i],
|
||||
["GAS TIJUANA", /COMPA[ÑN][IÍ]?A\s*DE\s*GAS\s*DE\s*TIJUANA|bajagas/i],
|
||||
// Ahead of the predial rules on purpose. Tijuana's zona federal receipt is
|
||||
// issued by the same treasury and carries the same header — "Ayuntamiento de
|
||||
// Tijuana", the same address, the same `ATB-541201` RFC — so every predial
|
||||
// discriminator matches it too, and whichever rule is asked first wins the
|
||||
// page. What only the zona federal layout says is "Marítimo Terrestre", which
|
||||
// survived OCR on all eight sample pages even where the heading above it came
|
||||
// back as "Zona Ledera) Maritimo Terrestre" and the printed concession clave
|
||||
// was lost under a highlighter mark.
|
||||
["ZONA FEDERAL TIJUANA", /ZOFEMAT|Mar[ií]timo\s*Terrestre|ocupaci[oó]n\s*de\s*Zona\s*Federal/i],
|
||||
// The municipal RFCs are the single most reliable discriminator on a predial
|
||||
// receipt: they are printed in a clean monospaced run on every layout, they
|
||||
// never change, and they say which of the three city treasuries issued the
|
||||
// page — which the wordmarks alone do not, since a Tijuana receipt also
|
||||
// carries "PLAYAS DE TIJUANA" and a Rosarito one "TIJUANA ENSENADA".
|
||||
["PREDIAL TIJUANA", /AYUNTAMIENTO\s*DE\s*TIJUANA|ATB.?541201/i],
|
||||
["PREDIAL ROSARITO", /AYUNTAMIENTO\s*MUNICIPAL\s*DE\s*PLAYAS\s*DE\s*ROSARITO|AMP.?981201|rosarito\.gob/i],
|
||||
["PREDIAL ENSENADA", /MUNICIPIO\s*DE\s*ENSENADA|MEN.?540301/i],
|
||||
];
|
||||
|
||||
/**
|
||||
* The predial rules come first because a Rosarito receipt prints "Clave
|
||||
* Catastral" as a boxed label — the very string the CESPT structural rule
|
||||
* looks for — so a page whose municipal header failed to OCR would otherwise
|
||||
* be claimed as a water bill and matched against the wrong column entirely.
|
||||
* "IMPUESTO PREDIAL" appears on all three municipal layouts and on none of the
|
||||
* utility ones, so it is the safe first question to ask.
|
||||
*/
|
||||
const LAYOUT: [string, RegExp][] = [
|
||||
// Same reasoning as the brand pass, one rule earlier: the concept line
|
||||
// "Derechos de ocupación de Zona Federal Marítimo Terrestre" is printed on
|
||||
// the stub of every zona federal page and on no other layout, and it read
|
||||
// cleanly on 8 of 8 samples — including the two whose heading did not.
|
||||
["ZONA FEDERAL TIJUANA", /Derechos\s*de\s*ocupaci[oó]n/i],
|
||||
["PREDIAL TIJUANA", /IMPUESTO\s*PREDIAL[\s\S]*?(?:CERTIFICACION\s*DE\s*CAJA|PASEO\s*DEL\s*CENTENARIO|PAGA\s*TU\s*PREDIAL)/i],
|
||||
["PREDIAL ENSENADA", /(?:IMPUESTO\s*PREDIAL[\s\S]*?TRANSPENINSULAR)|(?:IMPRESION\s*MAQUINA\s*REGISTRADORA)/i],
|
||||
["PREDIAL ROSARITO", /IMPUESTO\s*PREDIAL/i],
|
||||
["GAS TIJUANA", /Orden\s*de\s*Facturaci[oó]n|FACTOR\s*DE\s*PRESI[OÓ]N|GAS\s*LP/i],
|
||||
["CFE", /NO\.?\s*DE\s*SERVICIO|L[IÍ]MITE\s*DE\s*PAGO|PERIODO\s*FACTURADO/i],
|
||||
["CESPT", /SALDO\s+CORRIENTE|CLAVE\s*CATASTRAL|No\.?\s*DE\s*CUENTA/i],
|
||||
["TELNOR", /Mes\s*de\s*Facturaci[oó]n|Pagar\s*antes\s*de/i],
|
||||
@@ -364,10 +414,442 @@ function parseTelnor(page: OcrPage): ParsedStatement {
|
||||
};
|
||||
}
|
||||
|
||||
// --- GAS (Compañía de Gas de Tijuana / bajagas) ------------------------------
|
||||
|
||||
/**
|
||||
* These arrive as born-digital CFDI PDFs rather than scans, so the text layer
|
||||
* (see `TesseractOcrProvider.textPages`) usually reads them exactly and the
|
||||
* patterns below only have to be tolerant enough for the scanned case.
|
||||
*
|
||||
* The account number is printed three times — supply address, fiscal data, and
|
||||
* the payment stub at the foot — which is a free cross-check: three readings
|
||||
* that agree are near-certainly right, and any disagreement means one of them
|
||||
* was misread and the page deserves a human glance.
|
||||
*
|
||||
* `Cuenta` is what the matcher compares, not `Contrato`. The migration
|
||||
* recovered gas references out of `PropertyService.notes` into `meterNumber`
|
||||
* and what sat there is the 9-digit account (`900003463`), printed here with a
|
||||
* leading zero as `0900003463`.
|
||||
*/
|
||||
function parseGas(page: OcrPage): ParsedStatement {
|
||||
const text = page.text;
|
||||
const notes: string[] = [];
|
||||
|
||||
const seen = allMatches(text, /Cuenta\s*[:;.]?\s*([0-9OIlSBD]{6,12})/i).map((s) =>
|
||||
toDigits(s).replace(/^0+/, ""),
|
||||
);
|
||||
const distinct = [...new Set(seen.filter(Boolean))];
|
||||
|
||||
let accountRef: string | null = null;
|
||||
let crossChecked: boolean | null = null;
|
||||
if (distinct.length === 1) {
|
||||
accountRef = distinct[0];
|
||||
if (seen.length > 1) crossChecked = true;
|
||||
} else if (distinct.length > 1) {
|
||||
// Majority wins — the stub and the two address blocks print the same
|
||||
// number, so a single divergent reading is the misread one. It still goes
|
||||
// to review: `crossChecked: false` is what keeps the batch from
|
||||
// auto-matching a number one of three readings disagreed with.
|
||||
const tally = new Map<string, number>();
|
||||
for (const s of seen) tally.set(s, (tally.get(s) ?? 0) + 1);
|
||||
accountRef = [...tally.entries()].sort((a, b) => b[1] - a[1])[0][0];
|
||||
crossChecked = false;
|
||||
notes.push(`el número de cuenta se leyó de ${distinct.length} formas distintas (${distinct.join(", ")})`);
|
||||
}
|
||||
|
||||
const amount = money(
|
||||
firstMatch(text, [
|
||||
/TOTAL\s*A\s*PAGAR\s*[:;.]?\s*\$\s*([\d,]+\.\d{2})/i,
|
||||
/Total\s*a\s*pagar\s*[:;.]?\s*\$\s*([\d,]+\.\d{2})/i,
|
||||
]),
|
||||
);
|
||||
|
||||
// `20260630-20260630` — the range the bill was cut for. Both ends are the
|
||||
// same reading date on every sample, so the period is reported as the ISO
|
||||
// month rather than a range no ledger row would ever be searched by.
|
||||
const facturado = firstMatch(text, [/PERIODO\s*FACTURADO\s*[:;.]?\s*(\d{8})\s*-\s*\d{8}/i]);
|
||||
const period = facturado ? `${facturado.slice(0, 4)}-${facturado.slice(4, 6)}` : null;
|
||||
|
||||
return {
|
||||
provider: "GAS TIJUANA",
|
||||
serviceKind: "GAS",
|
||||
accountRef: accountRef || null,
|
||||
cadastralKey: null,
|
||||
amount,
|
||||
dueDate: parseDate(
|
||||
firstMatch(text, [/Fecha\s*de\s*Vencimiento\s*[:;.]?\s*(\d{4}\s*\/\s*\d{2}\s*\/\s*\d{2})/i])?.replace(
|
||||
/\s/g,
|
||||
"",
|
||||
),
|
||||
),
|
||||
period,
|
||||
crossChecked,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
// --- PREDIAL (municipal property tax) ---------------------------------------
|
||||
|
||||
/**
|
||||
* Normalise a printed clave catastral to the eight-character form
|
||||
* `Property.cadastralKey` holds. The municipalities print it grouped
|
||||
* (`KP-128-106`, `MM-B01-041`); the stored value drops the separators
|
||||
* (`KP128106`, `MMB01041`).
|
||||
*
|
||||
* The shape is *not* two letters and six digits, which is the assumption that
|
||||
* has to be resisted here. Across the 932 distinct claves on file, characters
|
||||
* four through eight are digits without exception, but the third is a digit in
|
||||
* 917 of them and one of `A`, `B`, `H`, `T` in the other fifteen. Running the
|
||||
* whole tail through `toDigits` — which maps `B` to `8` — is what turned a real
|
||||
* `MMB01041` into a nonexistent `MM801041`, so only positions four onward get
|
||||
* that treatment and a letter in the third position is kept as printed.
|
||||
*
|
||||
* That leaves a genuine ambiguity at that one position: a `B` there might be a
|
||||
* misread `8`, and 34 stored claves do carry an `8` there against six with a
|
||||
* `B`. It is left as read rather than guessed, because a page that fails to
|
||||
* match lands in the review queue where a human fixes it in seconds, while a
|
||||
* page that matches the wrong property posts a charge to the wrong customer.
|
||||
*
|
||||
* The two-letter prefix is the other fragile part. Tesseract inserts a spurious
|
||||
* `I` into letter pairs with some regularity — a real `MM-200-010` came back as
|
||||
* `MIM-200-010` — so a run longer than two letters has its `I`/`L` dropped
|
||||
* first, which recovers exactly that case. Anything still not two letters is
|
||||
* truncated and flagged, because a wrong prefix silently matches the wrong
|
||||
* property or, more often, nothing at all.
|
||||
*/
|
||||
export function normalizeCadastralKey(
|
||||
raw: string,
|
||||
notes: string[],
|
||||
): string | null {
|
||||
const m = raw.match(/^([A-Za-z|]{2,5})[-\s]?([A-Za-z0-9|]{3})[-\s]?([0-9OIlSBD]{3})$/);
|
||||
if (!m) return null;
|
||||
|
||||
let letters = m[1].toUpperCase().replace(/[^A-Z]/g, "");
|
||||
if (letters.length > 2) {
|
||||
const stripped = letters.replace(/[IL]/g, "");
|
||||
if (stripped.length === 2) {
|
||||
letters = stripped;
|
||||
} else {
|
||||
letters = letters.slice(0, 2);
|
||||
notes.push(`la clave catastral se leyó como "${m[1]}"; se tomó "${letters}"`);
|
||||
}
|
||||
}
|
||||
if (letters.length !== 2) return null;
|
||||
|
||||
const third = m[2][0].toUpperCase();
|
||||
const tail =
|
||||
(/[A-Z]/.test(third) ? third : toDigits(third)) +
|
||||
toDigits(m[2].slice(1)) +
|
||||
toDigits(m[3]);
|
||||
|
||||
return tail.length === 6 ? letters + tail : null;
|
||||
}
|
||||
|
||||
/** The grouped clave as printed, anchored to its label when one survived OCR. */
|
||||
const GROUPED_CLAVE = "[A-Z|]{2,5}-[A-Z0-9OIlSBD]{3}-[0-9OIlSBD]{3}";
|
||||
|
||||
function findCadastralKey(text: string, notes: string[]): string | null {
|
||||
const labelled = firstMatch(text, [
|
||||
new RegExp(`Clave\\s*Catastral\\s*[^A-Z0-9]{0,8}(${GROUPED_CLAVE})`, "i"),
|
||||
new RegExp(`CLAVE\\s*[^A-Z0-9]{0,8}(${GROUPED_CLAVE})`, "i"),
|
||||
]);
|
||||
if (labelled) return normalizeCadastralKey(labelled, notes);
|
||||
|
||||
// Ensenada's label ("CLAVE") lands inside a table header that OCRs into
|
||||
// noise more often than not, so the bare grouped shape is accepted as a
|
||||
// fallback. It is distinctive enough — two letters and two three-character
|
||||
// groups joined by hyphens appears nowhere else on these pages.
|
||||
const bare = firstMatch(text, [new RegExp(`\\b(${GROUPED_CLAVE})\\b`)]);
|
||||
return bare ? normalizeCadastralKey(bare, notes) : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Tijuana: a "CERTIFICACIÓN DE CAJA" whose payment barcode is one 32-digit run
|
||||
* of `account(8) + due date(DDMMYY) + amount(9) + folio(9)`, verified against
|
||||
* all five sample pages. Municipal totals are whole pesos (the receipt itself
|
||||
* carries a "Redondeo" line), so the barcode amount needs no decimal point.
|
||||
*
|
||||
* No clave catastral is printed anywhere on this layout — the 8-digit
|
||||
* municipal account is the only identifier, and it is not a number the legacy
|
||||
* database ever held. Until a reviewer confirms one, every Tijuana page lands
|
||||
* in review; confirming teaches the matcher (see `learnAccountRefs`) so the
|
||||
* same property matches itself next year.
|
||||
*/
|
||||
function parsePredialTijuana(page: OcrPage): ParsedStatement {
|
||||
const text = page.text;
|
||||
const notes: string[] = [];
|
||||
|
||||
const barcode = text.match(/(?<![0-9OIlSBD])([0-9OIlSBD]{32})(?![0-9OIlSBD])/);
|
||||
const printedTotal = money(
|
||||
firstMatch(text, [/TOTAL\s*A?\s*PAGAR\s*[:;.]?\s*\$?\s*([\d,]+\.?\d{0,2})/i]),
|
||||
);
|
||||
|
||||
let accountRef: string | null = null;
|
||||
let amount: number | null = printedTotal;
|
||||
let dueDate: Date | null = null;
|
||||
let crossChecked: boolean | null = null;
|
||||
|
||||
if (barcode) {
|
||||
const run = toDigits(barcode[1]);
|
||||
const d = run.slice(8, 14);
|
||||
const fromBarcode = Number(run.slice(14, 23));
|
||||
accountRef = run.slice(0, 8);
|
||||
dueDate = parseDate(`20${d.slice(4, 6)}-${d.slice(2, 4)}-${d.slice(0, 2)}`);
|
||||
notes.push("cuenta, importe y vencimiento leídos del código de barras");
|
||||
|
||||
if (printedTotal != null) {
|
||||
// Guarding the money, not the account number: the printed total is the
|
||||
// figure a human would key, so when the two disagree one of them is a
|
||||
// misread peso amount and nothing should post unreviewed.
|
||||
crossChecked = Math.abs(printedTotal - fromBarcode) < 0.5;
|
||||
if (!crossChecked) {
|
||||
notes.push(
|
||||
`el total impreso (${printedTotal}) no coincide con el código de barras (${fromBarcode})`,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (amount == null) amount = fromBarcode;
|
||||
}
|
||||
|
||||
if (!dueDate) {
|
||||
dueDate = parseDate(
|
||||
firstMatch(text, [/FECHA\s*VENCE\s*[:;.]?\s*(\d{1,2}\/\w{3}\/\d{4})/i]),
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
provider: "PREDIAL TIJUANA",
|
||||
serviceKind: "PROPERTY_TAX",
|
||||
accountRef: accountRef || null,
|
||||
cadastralKey: null,
|
||||
amount,
|
||||
dueDate,
|
||||
// The fiscal year, which is what the legacy ledger's `period` holds for
|
||||
// predial ("2026" is its single most common value). It is read from the
|
||||
// assessment table's year column, and failing that from the deadline: a
|
||||
// predial bill for year N falls due on 31 January of year N.
|
||||
period:
|
||||
firstMatch(text, [/VALOR\s*FISCAL[\s\S]{0,160}?\b(20\d{2})\b/i]) ??
|
||||
(dueDate ? String(dueDate.getUTCFullYear()) : null),
|
||||
crossChecked,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Rosarito: a wide "CERTIFICACIÓN DE CAJA" keyed by clave catastral, with no
|
||||
* account number of its own — the clave is the identifier, which is exactly
|
||||
* what `Property.cadastralKey` holds, so these match on the first pass.
|
||||
*
|
||||
* The total is read with a negative lookbehind on "Sub": the receipt prints
|
||||
* `Sub Total $5,409.39` (before the peso rounding) directly above
|
||||
* `Total $5,409.00`, and taking the first "Total" on the page books 39 cents
|
||||
* that the municipality did not charge. The lookbehind allows zero spaces
|
||||
* because the label prints both ways — `Sub Total` on one sample and
|
||||
* `SubTotal` on the next, and the tight one is what slipped past a fixed
|
||||
* `Sub\s` and read $9,624.85 off a receipt for $9,625.00.
|
||||
*/
|
||||
function parsePredialRosarito(page: OcrPage): ParsedStatement {
|
||||
const notes: string[] = [];
|
||||
const text = page.text;
|
||||
|
||||
return {
|
||||
provider: "PREDIAL ROSARITO",
|
||||
serviceKind: "PROPERTY_TAX",
|
||||
accountRef: null,
|
||||
cadastralKey: findCadastralKey(text, notes),
|
||||
amount: money(firstMatch(text, [/(?<!Sub\s{0,3})Total\s*[|:;.]?\s*\$\s*([\d,]+\.\d{2})/i])),
|
||||
// "EXTEMPORANEO DESPUES DE: 31/01/2026" — the leading E is regularly eaten
|
||||
// by the box rule printed over it, so the anchor starts at "XTEMPORANEO".
|
||||
dueDate: parseDate(
|
||||
firstMatch(text, [/XTEMPOR[AÁ]NEO\s*DESPU[EÉ]S\s*DE\s*[:;.]?\s*(\d{2}\/\d{2}\/\d{4})/i]),
|
||||
),
|
||||
period: firstMatch(text, [/Periodo\s*por\s*Pagar\s*[:;.]?\s*(20\d{2})/i]),
|
||||
crossChecked: null,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Ensenada: a dot-matrix "IMPRESION MAQUINA REGISTRADORA" statement, by some
|
||||
* distance the worst-scanning of the three. Matching is by clave catastral.
|
||||
*
|
||||
* The amount is read positionally rather than by label, because the label does
|
||||
* not survive: across five real pages the same word came back as `TOTAL:`,
|
||||
* `TOTA LA A` and `orAL:`. What is stable is the row — the summary line that
|
||||
* starts `TOTALES` carries the assessed figures across it and the amount
|
||||
* actually paid last, at the right margin.
|
||||
*
|
||||
* That last figure must carry a literal `$`. On a real sample the paid total
|
||||
* printed as `TOTAL: A $2,203.00` and OCR'd as `TOTAL: A 82,203.00` — the
|
||||
* dollar sign read as an 8, a mistake that would post a $2,203 charge as
|
||||
* $82,203 and look entirely ordinary in the ledger. Requiring the `$` costs
|
||||
* that page its amount and sends it to review, which is the only acceptable
|
||||
* failure here. The unprefixed figures earlier on the row are deliberately not
|
||||
* a fallback: they are the tax assessed before the early-payment discount, not
|
||||
* what was paid.
|
||||
*/
|
||||
function parsePredialEnsenada(page: OcrPage): ParsedStatement {
|
||||
const notes: string[] = [];
|
||||
const text = page.text;
|
||||
|
||||
const totalsRow = text.split("\n").find((l) => /TOTALES/i.test(l)) ?? "";
|
||||
const figures = allMatches(totalsRow, /\$\s*(\d[\d,.\s]*\.\d{2})/);
|
||||
const amount = figures.length ? money(figures[figures.length - 1]) : null;
|
||||
if (amount == null) {
|
||||
notes.push("no se pudo leer el importe con certeza; capturarlo a mano");
|
||||
}
|
||||
|
||||
return {
|
||||
provider: "PREDIAL ENSENADA",
|
||||
serviceKind: "PROPERTY_TAX",
|
||||
accountRef: null,
|
||||
cadastralKey: findCadastralKey(text, notes),
|
||||
amount,
|
||||
// This layout prints no payment deadline at all — it is a receipt for a
|
||||
// payment already made at the municipal window.
|
||||
dueDate: null,
|
||||
period: firstMatch(text, [/A[ÑN]O\s*[\s\S]{0,60}?\b(20\d{2})\b/i]),
|
||||
crossChecked: null,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
// --- ZONA FEDERAL (ZOFEMAT, Tijuana) ----------------------------------------
|
||||
|
||||
/**
|
||||
* Normalise the concession clave the zona federal receipt is keyed by.
|
||||
*
|
||||
* It is printed grouped and loosely spaced — `12-T -012`, `14-A-119`,
|
||||
* `14-K -031` — and is a different shape from the cadastral key entirely: two
|
||||
* digits, one letter, three digits. The letter is kept as printed rather than
|
||||
* digitised, for the same reason `normalizeCadastralKey` keeps its third
|
||||
* character: `toDigits` maps `B` to `8` and `D` to `0`, and a real `14-D -014`
|
||||
* run through it becomes `140014`, which is not a clave at all.
|
||||
*
|
||||
* Stored without separators, because nothing on file holds this value yet (see
|
||||
* `parseZonaFederal`) so the canonical form is ours to pick, and a bare run
|
||||
* cannot be broken by the hyphen the scan renders as a dash, a minus or
|
||||
* nothing.
|
||||
*/
|
||||
export function normalizeZofematKey(raw: string): string | null {
|
||||
const m = raw.match(/^([0-9OIlSBD]{2})\s*-\s*([A-Za-z])\s*-?\s*([0-9OIlSBD]{3})$/);
|
||||
if (!m) return null;
|
||||
const zone = toDigits(m[1]);
|
||||
const lot = toDigits(m[3]);
|
||||
if (zone.length !== 2 || lot.length !== 3) return null;
|
||||
return `${zone}${m[2].toUpperCase()}${lot}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* The bimester the receipt prints as `2026-2 / 2026-2`, rendered in the
|
||||
* vocabulary the ledger already speaks.
|
||||
*
|
||||
* All 258 legacy FEDERAL ZONE transactions carry a period of `JAN/FEB`,
|
||||
* `MAR/APR`, `MAY/JUN` or `NOV/DEC`, and their payment dates confirm the
|
||||
* ordering — JAN/FEB was paid in March, MAR/APR in May, MAY/JUN in July,
|
||||
* NOV/DEC in January, i.e. always the month after the bimester closes. The
|
||||
* receipts agree: the two `2026-3` samples fall due 17/07/2026 with no
|
||||
* surcharge, which is bimester three, May and June. Writing `2026-3` instead
|
||||
* would leave the OCR-posted rows unsearchable alongside every hand-keyed one.
|
||||
*/
|
||||
const BIMESTERS = ["JAN/FEB", "MAR/APR", "MAY/JUN", "JUL/AUG", "SEP/OCT", "NOV/DEC"];
|
||||
|
||||
/**
|
||||
* Tijuana's "Zona Federal Marítimo Terrestre" — the federal maritime-zone
|
||||
* occupancy fee, billed by the municipality for beachfront lots.
|
||||
*
|
||||
* Nothing on file identifies these. `PropertyService.accountNumber` for
|
||||
* FEDERAL_ZONE holds DATMEX.zfed, which is not a reference at all but an
|
||||
* amount: its 77 values include `246.06`, `2369.09`, `22653.94` and a negative
|
||||
* `-1679`, and the concession claves these receipts are keyed by appear nowhere
|
||||
* in the database. So the clave goes to `meterNumber` (see `scopedRefField`),
|
||||
* every page starts cold, and the first confirm teaches the match — the same
|
||||
* arrangement Tijuana predial needed, for the same reason.
|
||||
*
|
||||
* The amount is taken from the SubTotal row rather than the "Total a pagar"
|
||||
* box, which is printed on a grey fill and OCR'd on only 1 of 8 sample pages
|
||||
* while the SubTotal row read on 8 of 8. The two differ by design: the
|
||||
* municipality rounds to whole pesos and prints the difference on its own
|
||||
* "Ajuste Ley Hacienda Mpal" line — `-$0.05` against a 591.05 subtotal, `$0.21`
|
||||
* against 2,872.79 — so the payable figure is the rounded subtotal, and where
|
||||
* the printed box did read, it agreed.
|
||||
*/
|
||||
function parseZonaFederal(page: OcrPage): ParsedStatement {
|
||||
const text = page.text;
|
||||
const notes: string[] = [];
|
||||
|
||||
// Printed twice, once on the receipt and once on the stub below it, which is
|
||||
// a free second reading: on one sample the heading was struck through by the
|
||||
// office's own highlighter and only the stub survived.
|
||||
const claves = [
|
||||
...new Set(
|
||||
allMatches(text, /Clave\s*[:;.]?\s*([0-9OIlSBD]{2}\s*-\s*[A-Za-z]\s*-?\s*[0-9OIlSBD]{3})/i)
|
||||
.map(normalizeZofematKey)
|
||||
.filter((k): k is string => k != null),
|
||||
),
|
||||
];
|
||||
|
||||
const accountRef = claves[0] ?? null;
|
||||
let crossChecked: boolean | null = null;
|
||||
|
||||
const subtotalRow = text.split("\n").find((l) => /SubTotal/i.test(l)) ?? "";
|
||||
const figures = allMatches(subtotalRow, /(\d[\d,]*\.\d{2})/);
|
||||
// Impuesto, Actualización, Recargo, Multa, Importe — the payable one is last.
|
||||
const importe = figures.length ? money(figures[figures.length - 1]) : null;
|
||||
const rounded = importe != null ? Math.round(importe) : null;
|
||||
const printed = money(
|
||||
firstMatch(text, [/Total\s*a\s*pagar\s*[:;.]?\s*\$?\s*([\d,]+\.\d{2})/i]),
|
||||
);
|
||||
|
||||
if (printed != null && rounded != null) {
|
||||
crossChecked = Math.abs(printed - rounded) < 0.5;
|
||||
if (!crossChecked) {
|
||||
notes.push(
|
||||
`el total impreso (${printed}) no coincide con el subtotal redondeado (${rounded})`,
|
||||
);
|
||||
}
|
||||
} else if (rounded != null) {
|
||||
notes.push("importe tomado del subtotal, redondeado al peso");
|
||||
} else if (printed == null) {
|
||||
notes.push("no se pudo leer el importe con certeza; capturarlo a mano");
|
||||
}
|
||||
|
||||
// A clave read two different ways means one of the two readings is wrong and
|
||||
// there is no third to break the tie, so the page goes to a human even if the
|
||||
// money cross-checked.
|
||||
if (claves.length > 1) {
|
||||
crossChecked = false;
|
||||
notes.push(`la clave se leyó de ${claves.length} formas distintas (${claves.join(", ")})`);
|
||||
}
|
||||
if (!accountRef) notes.push("no se pudo leer la clave de la concesión");
|
||||
|
||||
const bimester = text.match(/\b(20\d{2})\s*-\s*([1-6])\s*\/\s*20\d{2}\s*-\s*[1-6]/);
|
||||
|
||||
return {
|
||||
provider: "ZONA FEDERAL TIJUANA",
|
||||
serviceKind: "FEDERAL_ZONE",
|
||||
accountRef,
|
||||
cadastralKey: null,
|
||||
amount: printed ?? rounded,
|
||||
dueDate: parseDate(
|
||||
firstMatch(text, [/Vencimiento\s*[:;.]?\s*(\d{2}\/\d{2}\/\d{4})/i]),
|
||||
),
|
||||
period: bimester ? BIMESTERS[+bimester[2] - 1] : null,
|
||||
crossChecked,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
const PARSERS: Record<string, (page: OcrPage) => ParsedStatement> = {
|
||||
CFE: parseCfe,
|
||||
CESPT: parseCespt,
|
||||
TELNOR: parseTelnor,
|
||||
"GAS TIJUANA": parseGas,
|
||||
"PREDIAL TIJUANA": parsePredialTijuana,
|
||||
"PREDIAL ROSARITO": parsePredialRosarito,
|
||||
"PREDIAL ENSENADA": parsePredialEnsenada,
|
||||
"ZONA FEDERAL TIJUANA": parseZonaFederal,
|
||||
};
|
||||
|
||||
const EMPTY: ParsedStatement = {
|
||||
|
||||
@@ -32,29 +32,52 @@ export interface MatchResult {
|
||||
* person. Names are displayed for the reviewer to sanity-check, and are never
|
||||
* an input to matching.
|
||||
*/
|
||||
@Injectable()
|
||||
export class StatementMatcherService {
|
||||
constructor(private readonly prisma: PrismaService) {}
|
||||
|
||||
/** Which PropertyService column a given kind's statements actually print. */
|
||||
private fieldFor(kind: ServiceKind): "accountNumber" | "meterNumber" | null {
|
||||
/**
|
||||
* Which `PropertyService` column a given kind's statements actually print.
|
||||
*
|
||||
* Exported because the same answer governs three places that must agree: the
|
||||
* lookup here, the blank-service fill on review, and the write-back on confirm.
|
||||
* When they disagree, a reference gets learned into a column nothing searches,
|
||||
* and the same page returns to the review queue every month forever.
|
||||
*
|
||||
* `meterNumber` is doing double duty for the three kinds whose printed
|
||||
* reference DATMEX never held in `accountNumber`:
|
||||
* - GAS, where the number lived in free-text notes,
|
||||
* - PROPERTY_TAX, where `accountNumber` holds DATMEX.predial — a 3-4 digit
|
||||
* office file number that is neither unique nor printed on any statement.
|
||||
* The Tijuana municipal receipt prints an 8-digit account and no clave
|
||||
* catastral at all, so it needs a column of its own; overwriting the legacy
|
||||
* predial numbers to make room would destroy the only link back to the
|
||||
* original records, and
|
||||
* - FEDERAL_ZONE, where `accountNumber` holds DATMEX.zfed, which is not a
|
||||
* reference of any kind but a peso amount: 3 of its 77 values carry cents
|
||||
* (`246.06`, `2369.09`, `22653.94`) and one is negative. Searching it for
|
||||
* the concession clave the receipt prints would never hit, and — worse —
|
||||
* because every row already has a value, the `[field]: null` guards in
|
||||
* `learnAccountRefs` and the blank-service fill would never fire either, so
|
||||
* the same page would return to the review queue every bimester forever.
|
||||
*/
|
||||
export function scopedRefField(
|
||||
kind: ServiceKind,
|
||||
): "accountNumber" | "meterNumber" | null {
|
||||
switch (kind) {
|
||||
case "ELECTRIC": // CFE "NO. DE SERVICIO" -> DATMEX.rpu
|
||||
case "WATER": // CESPT "Cuenta" / "No. DE CUENTA" -> DATMEX.agua
|
||||
case "TELEPHONE": // Telnor "Teléfono" (LADA stripped) -> DATMEX.telefono
|
||||
case "FEDERAL_ZONE":
|
||||
case "CABLE":
|
||||
return "accountNumber";
|
||||
case "GAS": // no account column in DATMEX; the number lived in notes
|
||||
case "GAS": // bajagas "Cuenta" -> recovered from notes into meterNumber
|
||||
case "PROPERTY_TAX": // Tijuana's 8-digit municipal account
|
||||
case "FEDERAL_ZONE": // ZOFEMAT concession clave, e.g. `12T012`
|
||||
return "meterNumber";
|
||||
// PROPERTY_TAX deliberately has no scoped column: what its
|
||||
// accountNumber holds is DATMEX.predial, which is neither unique nor
|
||||
// printed on any statement. Predial bills match on the clave catastral
|
||||
// alone — see matchByCadastralKey.
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Injectable()
|
||||
export class StatementMatcherService {
|
||||
constructor(private readonly prisma: PrismaService) {}
|
||||
|
||||
async match(parsed: ParsedStatement, expectedKind: ServiceKind): Promise<MatchResult> {
|
||||
const kind = parsed.serviceKind ?? expectedKind;
|
||||
@@ -68,33 +91,39 @@ export class StatementMatcherService {
|
||||
);
|
||||
}
|
||||
|
||||
const field = this.fieldFor(kind);
|
||||
const field = scopedRefField(kind);
|
||||
|
||||
if (field && parsed.accountRef) {
|
||||
const hit = await this.byServiceField(kind, field, parsed.accountRef);
|
||||
if (hit) return hit;
|
||||
}
|
||||
|
||||
// Secondary key. The clave catastral is printed on CESPT bills as well as
|
||||
// predial ones, so it rescues a page whose account number did not OCR —
|
||||
// which happened on real samples, where the clave read cleanly and the
|
||||
// account number did not.
|
||||
// The clave catastral is printed on CESPT bills as well as predial ones, so
|
||||
// it rescues a page whose account number did not OCR — which happened on
|
||||
// real samples, where the clave read cleanly and the account number did
|
||||
// not. On the Rosarito and Ensenada predial layouts it is not a rescue at
|
||||
// all but the only identifier the receipt carries, so a unique hit there is
|
||||
// as good as any account-number match and is treated as one.
|
||||
if (parsed.cadastralKey) {
|
||||
const hit = await this.byCadastralKey(kind, parsed.cadastralKey);
|
||||
const primary = kind === "PROPERTY_TAX" && !parsed.accountRef;
|
||||
const hit = await this.byCadastralKey(kind, parsed.cadastralKey, primary);
|
||||
if (hit) return hit;
|
||||
}
|
||||
|
||||
if (!field && !parsed.cadastralKey) {
|
||||
return this.unmatched(`no hay campo de búsqueda definido para ${kind}`);
|
||||
}
|
||||
if (!parsed.accountRef && !parsed.cadastralKey) {
|
||||
return this.unmatched(
|
||||
kind === "PROPERTY_TAX"
|
||||
? "el predial sólo se puede identificar por clave catastral y no se leyó ninguna"
|
||||
: `no hay campo de búsqueda definido para ${kind}`,
|
||||
? "no se leyó ni la clave catastral ni la cuenta municipal"
|
||||
: "no se pudo leer la referencia de la cuenta",
|
||||
);
|
||||
}
|
||||
return this.unmatched(
|
||||
parsed.accountRef
|
||||
? `no se encontró ningún servicio de ${kind} con la referencia ${parsed.accountRef}`
|
||||
: "no se pudo leer la referencia de la cuenta",
|
||||
: `no se encontró ninguna propiedad con la clave catastral ${parsed.cadastralKey}`,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -145,6 +174,8 @@ export class StatementMatcherService {
|
||||
private async byCadastralKey(
|
||||
kind: ServiceKind,
|
||||
key: string,
|
||||
/** True when the clave is the identifier the statement was issued against. */
|
||||
primary: boolean,
|
||||
): Promise<MatchResult | null> {
|
||||
const props = await this.prisma.property.findMany({
|
||||
where: { cadastralKey: key },
|
||||
@@ -174,15 +205,21 @@ export class StatementMatcherService {
|
||||
};
|
||||
}
|
||||
|
||||
// The clave identifies the property with certainty, but it is a *secondary*
|
||||
// key: it was not the number the statement was issued against. Left for
|
||||
// review so the confirm also teaches the matcher the account number, rather
|
||||
// than the same page needing the fallback again next month.
|
||||
// When the clave is the *secondary* key — a utility bill that also happens
|
||||
// to print it — the page is left for review, because the clave was not the
|
||||
// number the statement was issued against and confirming is what teaches
|
||||
// the matcher the account number for next month. When it is the primary key
|
||||
// (Rosarito and Ensenada predial, which print nothing else), a unique hit
|
||||
// is a real match and there is no second number to learn.
|
||||
return {
|
||||
propertyServiceId: candidates[0].propertyServiceId ?? null,
|
||||
customerId: candidates[0].customerId,
|
||||
note: `identificado por clave catastral ${key}; confirme para registrar también el número de cuenta`,
|
||||
confident: false,
|
||||
note: primary
|
||||
? `coincidencia exacta por clave catastral ${key}`
|
||||
: `identificado por clave catastral ${key}; confirme para registrar también el número de cuenta`,
|
||||
// A clave with no service row of the right kind behind it still needs a
|
||||
// human: there is nothing to attach the posting to.
|
||||
confident: primary && candidates[0].propertyServiceId != null,
|
||||
candidates,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,23 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { BillingModule } from "../billing/billing.module";
|
||||
import { OcrModule } from "../ocr/ocr.module";
|
||||
import { StatementsController } from "./statements.controller";
|
||||
import { StatementsService } from "./statements.service";
|
||||
import { StatementMatcherService } from "./statement-matcher.service";
|
||||
import { OCR_PROVIDER } from "./ocr/ocr.provider";
|
||||
import { TesseractOcrProvider } from "./ocr/tesseract.provider";
|
||||
|
||||
/**
|
||||
* The concrete OCR engine is bound here and nowhere else — everything
|
||||
* downstream depends on the OcrProvider interface, so swapping Tesseract for a
|
||||
* managed extraction API is a one-line change in this file.
|
||||
* The concrete OCR engine is bound in OcrModule (see apps/api/src/ocr/) —
|
||||
* everything downstream depends on the OcrProvider interface, so swapping
|
||||
* Tesseract for a managed extraction API is a one-line change there.
|
||||
*/
|
||||
@Module({
|
||||
imports: [BillingModule],
|
||||
imports: [BillingModule, OcrModule],
|
||||
controllers: [StatementsController],
|
||||
providers: [
|
||||
StatementsService,
|
||||
StatementMatcherService,
|
||||
{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider },
|
||||
],
|
||||
providers: [StatementsService, StatementMatcherService],
|
||||
})
|
||||
export class StatementsModule {}
|
||||
|
||||
@@ -16,7 +16,7 @@ import { BillingService } from "../billing/billing.service";
|
||||
import type { UploadedFileLike } from "../storage/upload-file";
|
||||
import { OCR_PROVIDER, type OcrProvider } from "./ocr/ocr.provider";
|
||||
import { parseStatement } from "./parsers/statement-parser";
|
||||
import { StatementMatcherService } from "./statement-matcher.service";
|
||||
import { StatementMatcherService, scopedRefField } from "./statement-matcher.service";
|
||||
import type { ConfirmBatchDto, ReviewDocumentDto } from "./statement.dto";
|
||||
|
||||
/**
|
||||
@@ -127,14 +127,23 @@ export class StatementsService {
|
||||
await this.storage.put(sourceKey, file.buffer, "application/pdf");
|
||||
|
||||
const pages = await this.ocr.renderPages(file.buffer);
|
||||
for (const image of pages) {
|
||||
// Page images are still rendered and stored for every file, text layer or
|
||||
// not: the review screen shows the reviewer the page, and "what the
|
||||
// parser read" is only checkable against a picture of the paper.
|
||||
const textLayer = await this.ocr.textPages(file.buffer).catch(() => []);
|
||||
|
||||
for (const [index, image] of pages.entries()) {
|
||||
pageNumber += 1;
|
||||
const storageKey = `statement/${batchId}/page-${pageNumber}.png`;
|
||||
await this.storage.put(storageKey, image, "image/png");
|
||||
|
||||
try {
|
||||
const ocr = await this.ocr.recognize(image);
|
||||
const embedded = textLayer[index] ?? null;
|
||||
const ocr = embedded ?? (await this.ocr.recognize(image));
|
||||
const parsed = parseStatement(ocr);
|
||||
if (embedded) {
|
||||
parsed.notes.unshift("texto leído del PDF original, sin OCR");
|
||||
}
|
||||
const match = await this.matcher.match(parsed, serviceKind);
|
||||
|
||||
const notes = [...parsed.notes, match.note].filter(Boolean);
|
||||
@@ -295,8 +304,8 @@ export class StatementsService {
|
||||
where: { id: doc.batchId },
|
||||
select: { serviceKind: true },
|
||||
});
|
||||
if (batch) {
|
||||
const field = batch.serviceKind === "GAS" ? "meterNumber" : "accountNumber";
|
||||
const field = batch && scopedRefField(batch.serviceKind);
|
||||
if (batch && field) {
|
||||
const blank = await this.prisma.propertyService.findMany({
|
||||
where: {
|
||||
kind: batch.serviceKind,
|
||||
@@ -434,7 +443,8 @@ export class StatementsService {
|
||||
docs: { matchedPropertyServiceId: string | null; extractedAccountRef: string | null }[],
|
||||
kind: ServiceKind,
|
||||
) {
|
||||
const field = kind === "GAS" ? "meterNumber" : "accountNumber";
|
||||
const field = scopedRefField(kind);
|
||||
if (!field) return;
|
||||
for (const d of docs) {
|
||||
if (!d.matchedPropertyServiceId || !d.extractedAccountRef) continue;
|
||||
await this.prisma.propertyService.updateMany({
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
{
|
||||
"extends": "./tsconfig.json",
|
||||
"exclude": ["node_modules", "dist", "**/*.spec.ts"]
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/web",
|
||||
"version": "1.0.5",
|
||||
"version": "1.0.6",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"dev": "next dev -p 4500",
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
"use client";
|
||||
|
||||
import { AppShell } from "@/components/AppShell";
|
||||
import { PolicyOcrReview } from "@/components/PolicyOcrReview";
|
||||
|
||||
export default function PolicyOcrBatchPage({
|
||||
params,
|
||||
}: {
|
||||
params: { id: string };
|
||||
}) {
|
||||
return (
|
||||
<AppShell>
|
||||
<PolicyOcrReview id={params.id} />
|
||||
</AppShell>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
"use client";
|
||||
|
||||
import { AppShell } from "@/components/AppShell";
|
||||
import { PolicyCaptura } from "@/components/PolicyCaptura";
|
||||
|
||||
/**
|
||||
* OCR mode of the policy intake screen. Drops the GMX PDF, walks through
|
||||
* per-page review, confirms. Same wrapper as `/polizas/nuevo` (manual)
|
||||
* with `initialMode="auto"`, so the tab strip is identical and swapping
|
||||
* modes doesn't drop state.
|
||||
*
|
||||
* Sister route `/polizas/captura/[id]` is the batch review screen once a
|
||||
* batch is uploaded.
|
||||
*/
|
||||
export default function CapturaOcrPage() {
|
||||
return (
|
||||
<AppShell>
|
||||
<PolicyCaptura initialMode="auto" />
|
||||
</AppShell>
|
||||
);
|
||||
}
|
||||
@@ -1,41 +1,22 @@
|
||||
"use client";
|
||||
|
||||
import { Suspense } from "react";
|
||||
import Link from "next/link";
|
||||
import { useSearchParams } from "next/navigation";
|
||||
import { AppShell } from "@/components/AppShell";
|
||||
import { PolicyForm } from "@/components/PolicyForm";
|
||||
import { useCan } from "@/lib/abilities";
|
||||
import { PolicyCaptura } from "@/components/PolicyCaptura";
|
||||
|
||||
/**
|
||||
* Manual mode of the policy intake screen. Shares the tab wrapper with
|
||||
* `/polizas/captura` (OCR mode) so staff can swap between the two without
|
||||
* losing their place. Customer picker comes from the `?customerId=`
|
||||
* / `?customerName=` query string — used by `/clientes/[id]` when staff
|
||||
* creates a policy from a customer detail page.
|
||||
*/
|
||||
export default function NuevaPolizaPage() {
|
||||
return (
|
||||
<AppShell>
|
||||
<Suspense fallback={null}>
|
||||
<NuevaPoliza />
|
||||
<PolicyCaptura initialMode="manual" />
|
||||
</Suspense>
|
||||
</AppShell>
|
||||
);
|
||||
}
|
||||
|
||||
function NuevaPoliza() {
|
||||
const allowed = useCan("policy:create");
|
||||
const params = useSearchParams();
|
||||
const customerId = params.get("customerId") ?? undefined;
|
||||
const customerName = params.get("customerName") ?? undefined;
|
||||
|
||||
return (
|
||||
<>
|
||||
<div className="page-head">
|
||||
<Link href="/polizas" className="back-link">← Pólizas</Link>
|
||||
<h1 className="page-title">Nueva póliza</h1>
|
||||
</div>
|
||||
{allowed ? (
|
||||
<PolicyForm fixedCustomerId={customerId} fixedCustomerName={customerName} />
|
||||
) : (
|
||||
<div className="state-box state-error">
|
||||
No tiene permisos para crear pólizas.
|
||||
</div>
|
||||
)}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -57,6 +57,7 @@ export default function PolizasPage() {
|
||||
|
||||
function PolizasBrowser() {
|
||||
const canCreate = useCan("policy:create");
|
||||
const canIngest = useCan("policy:ingest");
|
||||
const [stats, setStats] = useState<PolicyStats | null>(null);
|
||||
const [facets, setFacets] = useState<PolicyFacets | null>(null);
|
||||
|
||||
@@ -135,6 +136,11 @@ function PolizasBrowser() {
|
||||
{ slug: "vigente", label: "Por vencer (Incen.)", params: { typeName: "INCEN" } },
|
||||
]}
|
||||
/>
|
||||
{canIngest && (
|
||||
<Link href="/polizas/captura" className="btn btn-outline">
|
||||
+ Captura OCR
|
||||
</Link>
|
||||
)}
|
||||
{canCreate && (
|
||||
<Link href="/polizas/nuevo" className="btn btn-primary">+ Nueva póliza</Link>
|
||||
)}
|
||||
|
||||
@@ -132,9 +132,7 @@ function BatchReview({ id }: { id: string }) {
|
||||
{error && <div className="state-box state-error">{error}</div>}
|
||||
|
||||
{processing && (
|
||||
<div className="state-box">
|
||||
Leyendo los recibos… esta pantalla se actualiza sola.
|
||||
</div>
|
||||
<ProcessingBanner docsLength={docs.length} pendingOcr={batch.byStatus.PENDING_OCR ?? 0} />
|
||||
)}
|
||||
|
||||
<SummaryCard batch={batch} readyCount={readyCount} />
|
||||
@@ -170,6 +168,55 @@ const STATUS_LABEL_BATCH: Record<string, string> = {
|
||||
FAILED: "Falló",
|
||||
};
|
||||
|
||||
/**
|
||||
* Live readout while OCR is running. The backend tells us how many pages are
|
||||
* still PENDING_OCR, so we can show real progress instead of "loading…". When
|
||||
* the docs list hasn't caught up to the upload yet (total === 0) we fall back
|
||||
* to the indeterminate bar.
|
||||
*/
|
||||
function ProcessingBanner({
|
||||
docsLength,
|
||||
pendingOcr,
|
||||
}: {
|
||||
docsLength: number;
|
||||
pendingOcr: number;
|
||||
}) {
|
||||
const done = Math.max(docsLength - pendingOcr, 0);
|
||||
const pct =
|
||||
docsLength > 0 ? Math.min(100, Math.round((done / docsLength) * 100)) : null;
|
||||
return (
|
||||
<div className="card" style={{ padding: 16 }}>
|
||||
<div className="upload-progress" style={{ padding: 0 }}>
|
||||
<div
|
||||
className={`progress-track${pct === null ? " progress-indeterminate" : ""}`}
|
||||
role="progressbar"
|
||||
aria-valuemin={0}
|
||||
aria-valuemax={100}
|
||||
aria-valuenow={pct ?? undefined}
|
||||
>
|
||||
<div className="progress-fill" style={{ width: `${pct ?? 100}%` }} />
|
||||
</div>
|
||||
<div className="upload-progress-stats">
|
||||
{pct === null ? (
|
||||
<span>Leyendo los recibos…</span>
|
||||
) : (
|
||||
<>
|
||||
<strong>{pct}%</strong>
|
||||
<span>
|
||||
{done} de {docsLength} página(s) leídas
|
||||
</span>
|
||||
{pendingOcr > 0 && <span>{pendingOcr} en cola</span>}
|
||||
</>
|
||||
)}
|
||||
<span style={{ marginLeft: "auto" }}>
|
||||
Esta pantalla se actualiza sola.
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function SummaryCard({
|
||||
batch,
|
||||
readyCount,
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import Link from "next/link";
|
||||
import { useSearchParams } from "next/navigation";
|
||||
import { PolicyForm } from "@/components/PolicyForm";
|
||||
import { PolicyOcrIntake } from "@/components/PolicyOcrIntake";
|
||||
import { useCan } from "@/lib/abilities";
|
||||
|
||||
/**
|
||||
* Policy intake — mirror of `Captura.tsx` (statement OCR side): one screen,
|
||||
* two ways in:
|
||||
*
|
||||
* - **manual** — `PolicyForm` keys every field by hand.
|
||||
* - **auto** — `PolicyOcrIntake` uploads a GMX PDF, OCR proposes the
|
||||
* policy, a human still confirms.
|
||||
*
|
||||
* Both end at the same place (a `Policy` row on a customer's file) so they
|
||||
* live as two modes of one screen rather than two menu entries — exactly the
|
||||
* same shape Captura uses for `ManualCheckCapture` vs `StatementIntake`.
|
||||
*
|
||||
* `/polizas/nuevo` opens manual, `/polizas/captura` opens auto; both render
|
||||
* this component so the tab toggle works either way and an old bookmark
|
||||
* still lands on the right tab.
|
||||
*/
|
||||
export type PolicyCaptureMode = "manual" | "auto";
|
||||
|
||||
const MODE_HINT: Record<PolicyCaptureMode, string> = {
|
||||
manual:
|
||||
"Captura cada campo a mano. Use esta opción cuando la póliza llega en papel, en un correo sin PDF legible, o cuando hay que revisar cada dato.",
|
||||
auto: "Suelte el PDF descargado del portal de GMX y el sistema propondrá los campos. Nada se registra sin tu confirmación.",
|
||||
};
|
||||
|
||||
export function PolicyCaptura({ initialMode = "manual" }: { initialMode?: PolicyCaptureMode }) {
|
||||
const canCreate = useCan("policy:create");
|
||||
const canIngest = useCan("policy:ingest");
|
||||
|
||||
// `/clientes/[id]` deep-links into /polizas/nuevo with the customer
|
||||
// pre-picked so staff can fill the rest without retyping. The OCR pane
|
||||
// ignores these — there's no customer to lock in until the batch is
|
||||
// confirmed.
|
||||
const params = useSearchParams();
|
||||
const fixedCustomerId = params.get("customerId") ?? undefined;
|
||||
const fixedCustomerName = params.get("customerName") ?? undefined;
|
||||
|
||||
// One user can land on either mode. The tab strip only renders when both
|
||||
// abilities are held — a STAFF with only policy:ingest (no create) still
|
||||
// sees the screen but only the OCR tab is offered.
|
||||
const modes: { key: PolicyCaptureMode; label: string }[] = [
|
||||
...(canCreate ? [{ key: "manual" as const, label: "Captura manual" }] : []),
|
||||
...(canIngest ? [{ key: "auto" as const, label: "Captura automática (OCR)" }] : []),
|
||||
];
|
||||
|
||||
const [mode, setMode] = useState<PolicyCaptureMode>(
|
||||
modes.some((m) => m.key === initialMode) ? initialMode : (modes[0]?.key ?? "manual"),
|
||||
);
|
||||
|
||||
if (modes.length === 0) {
|
||||
return (
|
||||
<div className="state-box state-error">
|
||||
No tienes permiso para crear ni capturar pólizas.
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<>
|
||||
<div className="page-head">
|
||||
<Link href="/polizas" className="back-link">← Pólizas</Link>
|
||||
<h1 className="page-title">Nueva póliza</h1>
|
||||
<p className="eyebrow">{MODE_HINT[mode]}</p>
|
||||
</div>
|
||||
|
||||
{modes.length > 1 && (
|
||||
<div className="seg" role="tablist" style={{ marginBottom: 16 }}>
|
||||
{modes.map((m) => (
|
||||
<button
|
||||
key={m.key}
|
||||
type="button"
|
||||
role="tab"
|
||||
aria-selected={mode === m.key}
|
||||
className={`seg-btn ${mode === m.key ? "active" : ""}`}
|
||||
onClick={() => setMode(m.key)}
|
||||
>
|
||||
{m.label}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{mode === "manual" ? (
|
||||
<ManualPane
|
||||
fixedCustomerId={fixedCustomerId}
|
||||
fixedCustomerName={fixedCustomerName}
|
||||
/>
|
||||
) : (
|
||||
<PolicyOcrIntake />
|
||||
)}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
function ManualPane({
|
||||
fixedCustomerId,
|
||||
fixedCustomerName,
|
||||
}: {
|
||||
fixedCustomerId?: string;
|
||||
fixedCustomerName?: string;
|
||||
}) {
|
||||
const allowed = useCan("policy:create");
|
||||
if (!allowed) {
|
||||
return (
|
||||
<div className="state-box state-error">
|
||||
No tiene permisos para crear pólizas.
|
||||
</div>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<PolicyForm
|
||||
fixedCustomerId={fixedCustomerId}
|
||||
fixedCustomerName={fixedCustomerName}
|
||||
/>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
"use client";
|
||||
|
||||
import { useCallback, useEffect, useState } from "react";
|
||||
import Link from "next/link";
|
||||
import {
|
||||
getPolicyOcrStatus,
|
||||
listPolicyOcrBatches,
|
||||
uploadPolicyOcrBatch,
|
||||
} from "@/lib/api";
|
||||
import { useCan } from "@/lib/abilities";
|
||||
import { formatDate } from "@/lib/labels";
|
||||
import type { PolicyOcrBatch, PolicyOcrBatchStatus } from "@/lib/types";
|
||||
|
||||
/**
|
||||
* Insurance OCR intake — mirror of StatementIntake, scoped to the insurance
|
||||
* side. Today the only provider is GMX; the parser dispatches on a brand
|
||||
* wordmark (`Grupo Mexicano de Seguros` / `gmx.com.mx` / the GMX letterhead)
|
||||
* and a new portal only needs a new BRAND entry plus a parser file.
|
||||
*
|
||||
* Lives inside the `Pólizas` page rather than a top-level route because it
|
||||
* is one mode of one job (staff uploading whatever PDFs the office has on
|
||||
* hand that day, mixed service vs insurance), and the matching/review queue
|
||||
* already keys on the policyNumber → existing Policy transition that the
|
||||
* rest of /polizas owns.
|
||||
*/
|
||||
|
||||
const STATUS_LABEL: Record<PolicyOcrBatchStatus, string> = {
|
||||
UPLOADED: "Recibido",
|
||||
PROCESSING: "Procesando…",
|
||||
READY_FOR_REVIEW: "Listo para revisar",
|
||||
COMPLETED: "Aplicado",
|
||||
FAILED: "Falló",
|
||||
};
|
||||
|
||||
export function PolicyOcrIntake() {
|
||||
const canIngest = useCan("policy:ingest");
|
||||
const [batches, setBatches] = useState<PolicyOcrBatch[]>([]);
|
||||
const [ocrAvailable, setOcrAvailable] = useState<boolean | null>(null);
|
||||
const [storageAvailable, setStorageAvailable] = useState<boolean | null>(null);
|
||||
const [loading, setLoading] = useState(true);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const load = useCallback(async () => {
|
||||
try {
|
||||
const [list, status] = await Promise.all([
|
||||
listPolicyOcrBatches(),
|
||||
getPolicyOcrStatus(),
|
||||
]);
|
||||
setBatches(list.items);
|
||||
setOcrAvailable(status.ocrAvailable);
|
||||
setStorageAvailable(status.storageAvailable);
|
||||
setError(null);
|
||||
} catch (e) {
|
||||
setError((e as Error)?.message ?? "No se pudieron cargar los lotes.");
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
void load();
|
||||
}, [load]);
|
||||
|
||||
const working = batches.some(
|
||||
(b) => b.status === "PROCESSING" || b.status === "UPLOADED",
|
||||
);
|
||||
useEffect(() => {
|
||||
if (!working) return;
|
||||
const t = setInterval(() => void load(), 4000);
|
||||
return () => clearInterval(t);
|
||||
}, [working, load]);
|
||||
|
||||
const ready = ocrAvailable === true && storageAvailable === true;
|
||||
|
||||
return (
|
||||
<div className="stack">
|
||||
{ocrAvailable === false && (
|
||||
<div className="state-box state-error">
|
||||
Este servidor no tiene OCR instalado, así que no se pueden leer PDFs
|
||||
de pólizas escaneados. La captura manual sigue funcionando.
|
||||
</div>
|
||||
)}
|
||||
|
||||
{storageAvailable === false && (
|
||||
<div className="state-box state-error">
|
||||
Este servidor no tiene configurado el almacenamiento de documentos, así
|
||||
que no hay dónde guardar los PDFs. Mientras tanto, capture las
|
||||
pólizas a mano.
|
||||
</div>
|
||||
)}
|
||||
|
||||
{canIngest && ready && <UploadCard onDone={load} />}
|
||||
|
||||
{error && <div className="state-box state-error">{error}</div>}
|
||||
|
||||
<section className="card" style={{ padding: 16 }}>
|
||||
<h2 className="section-title" style={{ marginTop: 0 }}>
|
||||
Lotes
|
||||
</h2>
|
||||
{loading ? (
|
||||
<div className="state-box">Cargando…</div>
|
||||
) : batches.length === 0 ? (
|
||||
<div className="state-box">
|
||||
Todavía no hay lotes de pólizas. Descargue el certificado del portal
|
||||
de GMX y suéltelo arriba.
|
||||
</div>
|
||||
) : (
|
||||
<div className="tx-scroll">
|
||||
<table className="tx-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Fecha</th>
|
||||
<th>Aseguradora</th>
|
||||
<th>Referencia</th>
|
||||
<th>Estado</th>
|
||||
<th className="num">Páginas</th>
|
||||
<th>Subido por</th>
|
||||
<th />
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{batches.map((b) => (
|
||||
<tr key={b.id}>
|
||||
<td style={{ whiteSpace: "nowrap" }}>{formatDate(b.createdAt)}</td>
|
||||
<td>{b.provider}</td>
|
||||
<td>{b.label || "—"}</td>
|
||||
<td>
|
||||
<StatusTag status={b.status} />
|
||||
{b.error && (
|
||||
<div className="page-sub" style={{ marginTop: 4 }}>
|
||||
{b.error}
|
||||
</div>
|
||||
)}
|
||||
</td>
|
||||
<td className="num">{b._count?.documents ?? 0}</td>
|
||||
<td>{b.uploadedBy?.name ?? "—"}</td>
|
||||
<td>
|
||||
<Link
|
||||
className="btn btn-ghost"
|
||||
href={`/polizas/captura/${b.id}`}
|
||||
>
|
||||
Revisar
|
||||
</Link>
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
)}
|
||||
</section>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function StatusTag({ status }: { status: PolicyOcrBatchStatus }) {
|
||||
return <span className="tag">{STATUS_LABEL[status] ?? status}</span>;
|
||||
}
|
||||
|
||||
function UploadCard({ onDone }: { onDone: () => void }) {
|
||||
const [files, setFiles] = useState<File[]>([]);
|
||||
const [label, setLabel] = useState("");
|
||||
const [busy, setBusy] = useState(false);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
async function submit() {
|
||||
if (!files.length) return;
|
||||
setBusy(true);
|
||||
setError(null);
|
||||
try {
|
||||
await uploadPolicyOcrBatch(files, label.trim() || undefined);
|
||||
setFiles([]);
|
||||
setLabel("");
|
||||
onDone();
|
||||
} catch (e) {
|
||||
setError((e as Error)?.message ?? "No se pudo subir el lote.");
|
||||
} finally {
|
||||
setBusy(false);
|
||||
}
|
||||
}
|
||||
|
||||
return (
|
||||
<section className="card" style={{ padding: 16 }}>
|
||||
<h2 className="section-title" style={{ marginTop: 0 }}>
|
||||
Subir PDFs de pólizas (GMX)
|
||||
</h2>
|
||||
<div className="inline-form" style={{ flexWrap: "wrap", gap: 12 }}>
|
||||
<label>
|
||||
<span className="page-sub">Referencia (opcional)</span>
|
||||
<input
|
||||
className="input"
|
||||
placeholder="ej. GMX julio 2026"
|
||||
value={label}
|
||||
onChange={(e) => setLabel(e.target.value)}
|
||||
/>
|
||||
</label>
|
||||
|
||||
<label>
|
||||
<span className="page-sub">Archivos PDF</span>
|
||||
<input
|
||||
type="file"
|
||||
className="input"
|
||||
accept="application/pdf"
|
||||
multiple
|
||||
onChange={(e) => setFiles(Array.from(e.target.files ?? []))}
|
||||
/>
|
||||
</label>
|
||||
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-primary"
|
||||
disabled={!files.length || busy}
|
||||
onClick={submit}
|
||||
>
|
||||
{busy ? "Subiendo…" : `Procesar ${files.length || ""}`.trim()}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{error && (
|
||||
<div className="state-box state-error" style={{ marginTop: 12 }}>
|
||||
{error}
|
||||
</div>
|
||||
)}
|
||||
|
||||
<p className="page-sub" style={{ marginTop: 12 }}>
|
||||
Un lote puede traer varios PDFs. Cada página se procesa por separado; el
|
||||
sistema busca una póliza existente por número y, si no la encuentra,
|
||||
propone crear una nueva bajo el cliente que se elija en la revisión.
|
||||
</p>
|
||||
</section>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,578 @@
|
||||
"use client";
|
||||
|
||||
import { useCallback, useEffect, useMemo, useState } from "react";
|
||||
import Link from "next/link";
|
||||
import { CustomerPicker } from "@/components/CustomerPicker";
|
||||
import {
|
||||
confirmPolicyOcrBatch,
|
||||
getPolicyOcrBatch,
|
||||
listCustomers,
|
||||
listPolicyOcrDocuments,
|
||||
policyOcrDocumentUrl,
|
||||
rejectPolicyOcrDocument,
|
||||
reviewPolicyOcrDocument,
|
||||
} from "@/lib/api";
|
||||
import { useCan } from "@/lib/abilities";
|
||||
import { formatDate, formatMoney } from "@/lib/labels";
|
||||
import type {
|
||||
CustomerListItem,
|
||||
PolicyOcrBatchDetail,
|
||||
PolicyOcrConfirmDocument,
|
||||
PolicyOcrCoverage,
|
||||
PolicyOcrDocument,
|
||||
PolicyOcrReviewInput,
|
||||
} from "@/lib/types";
|
||||
|
||||
const STATUS_LABEL: Record<string, string> = {
|
||||
PENDING_OCR: "Pendiente",
|
||||
OCR_FAILED: "Falló OCR",
|
||||
NEEDS_REVIEW: "Para revisar",
|
||||
MATCHED: "Listo",
|
||||
CONFIRMED: "Confirmado",
|
||||
POSTED: "Aplicado",
|
||||
REJECTED: "Rechazado",
|
||||
};
|
||||
|
||||
const OPEN_FIRST = [
|
||||
"NEEDS_REVIEW",
|
||||
"MATCHED",
|
||||
"CONFIRMED",
|
||||
"PENDING_OCR",
|
||||
"OCR_FAILED",
|
||||
"REJECTED",
|
||||
"POSTED",
|
||||
];
|
||||
|
||||
type EditMap = Record<string, PolicyOcrConfirmDocument | undefined>;
|
||||
|
||||
export function PolicyOcrReview({ id }: { id: string }) {
|
||||
const canReview = useCan("policy:ocr-review");
|
||||
const [batch, setBatch] = useState<PolicyOcrBatchDetail | null>(null);
|
||||
const [docs, setDocs] = useState<PolicyOcrDocument[]>([]);
|
||||
const [edits, setEdits] = useState<EditMap>({});
|
||||
const [customerIndex, setCustomerIndex] = useState<Record<string, CustomerListItem>>({});
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [loading, setLoading] = useState(true);
|
||||
const [submitting, setSubmitting] = useState(false);
|
||||
|
||||
const load = useCallback(async () => {
|
||||
try {
|
||||
const [b, d, c] = await Promise.all([
|
||||
getPolicyOcrBatch(id),
|
||||
listPolicyOcrDocuments(id),
|
||||
canReview ? listCustomers({ pageSize: 200 }).then((r) => r.items) : Promise.resolve([]),
|
||||
]);
|
||||
setBatch(b);
|
||||
setDocs(d);
|
||||
setCustomerIndex(Object.fromEntries(c.map((x) => [x.id, x])));
|
||||
setError(null);
|
||||
} catch (e) {
|
||||
setError((e as Error)?.message ?? "No se pudo cargar el lote.");
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, [id, canReview]);
|
||||
|
||||
useEffect(() => {
|
||||
void load();
|
||||
}, [load]);
|
||||
|
||||
const processing = batch?.status === "PROCESSING" || batch?.status === "UPLOADED";
|
||||
useEffect(() => {
|
||||
if (!processing) return;
|
||||
const t = setInterval(() => void load(), 4000);
|
||||
return () => clearInterval(t);
|
||||
}, [processing, load]);
|
||||
|
||||
const sorted = useMemo(
|
||||
() =>
|
||||
[...docs].sort(
|
||||
(a, b) =>
|
||||
OPEN_FIRST.indexOf(a.status) - OPEN_FIRST.indexOf(b.status) ||
|
||||
a.pageNumber - b.pageNumber,
|
||||
),
|
||||
[docs],
|
||||
);
|
||||
|
||||
const readyCount = Object.values(edits).filter(Boolean).length;
|
||||
|
||||
function setEdit(docId: string, edit: PolicyOcrConfirmDocument) {
|
||||
setEdits((prev) => ({ ...prev, [docId]: edit }));
|
||||
}
|
||||
|
||||
async function onConfirm() {
|
||||
if (!batch) return;
|
||||
const payload: PolicyOcrConfirmDocument[] = [];
|
||||
for (const d of docs) {
|
||||
const edit = edits[d.id];
|
||||
if (!edit) continue;
|
||||
if (!edit.policyId && !edit.customerId) {
|
||||
setError(`Página ${d.pageNumber}: falta cliente o póliza destino.`);
|
||||
return;
|
||||
}
|
||||
payload.push(edit);
|
||||
}
|
||||
if (!payload.length) {
|
||||
setError("No hay documentos revisados. Guarde cada página antes de aplicar.");
|
||||
return;
|
||||
}
|
||||
setSubmitting(true);
|
||||
setError(null);
|
||||
try {
|
||||
await confirmPolicyOcrBatch(batch.id, { documents: payload });
|
||||
setEdits({});
|
||||
await load();
|
||||
} catch (e) {
|
||||
setError((e as Error)?.message ?? "No se pudo aplicar el lote.");
|
||||
} finally {
|
||||
setSubmitting(false);
|
||||
}
|
||||
}
|
||||
|
||||
if (loading) return <div className="state-box">Cargando…</div>;
|
||||
if (!batch) return <div className="state-box state-error">{error ?? "No encontrado."}</div>;
|
||||
|
||||
return (
|
||||
<div className="stack">
|
||||
<header className="page-head">
|
||||
<div>
|
||||
<h1 className="page-title">
|
||||
Pólizas — {batch.provider}
|
||||
{batch.label ? ` · ${batch.label}` : ""}
|
||||
</h1>
|
||||
<p className="page-sub">
|
||||
{formatDate(batch.createdAt)} · {docs.length} página(s) ·{" "}
|
||||
{STATUS_LABEL[batch.status] ?? batch.status}
|
||||
</p>
|
||||
</div>
|
||||
<Link className="btn btn-ghost" href="/polizas">
|
||||
Volver a pólizas
|
||||
</Link>
|
||||
</header>
|
||||
|
||||
{processing && <div className="state-box">Procesando…</div>}
|
||||
|
||||
{error && <div className="state-box state-error">{error}</div>}
|
||||
|
||||
{canReview && readyCount > 0 && (
|
||||
<section className="card" style={{ padding: 16 }}>
|
||||
<h2 className="section-title" style={{ marginTop: 0 }}>
|
||||
Aplicar lote
|
||||
</h2>
|
||||
<p className="page-sub" style={{ marginBottom: 12 }}>
|
||||
{readyCount} página(s) revisada(s). Se creará o actualizará la póliza
|
||||
y, si marcó la casilla, se registrará la prima en el estado de
|
||||
cuenta.
|
||||
</p>
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-primary"
|
||||
disabled={submitting}
|
||||
onClick={onConfirm}
|
||||
>
|
||||
{submitting ? "Aplicando…" : "Aplicar"}
|
||||
</button>
|
||||
</section>
|
||||
)}
|
||||
|
||||
<section className="stack">
|
||||
{sorted.map((doc) => (
|
||||
<DocumentRow
|
||||
key={doc.id}
|
||||
doc={doc}
|
||||
customerIndex={customerIndex}
|
||||
canReview={canReview}
|
||||
onSave={async (edit) => {
|
||||
await reviewPolicyOcrDocument(doc.id, edit.reviewInput);
|
||||
setEdit(doc.id, edit.confirmInput);
|
||||
await load();
|
||||
}}
|
||||
onReject={async () => {
|
||||
await rejectPolicyOcrDocument(doc.id);
|
||||
setEdits((prev) => {
|
||||
const { [doc.id]: _, ...rest } = prev;
|
||||
return rest;
|
||||
});
|
||||
await load();
|
||||
}}
|
||||
/>
|
||||
))}
|
||||
</section>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
interface RowSaved {
|
||||
reviewInput: PolicyOcrReviewInput;
|
||||
confirmInput: PolicyOcrConfirmDocument;
|
||||
}
|
||||
|
||||
interface DocumentRowProps {
|
||||
doc: PolicyOcrDocument;
|
||||
customerIndex: Record<string, CustomerListItem>;
|
||||
canReview: boolean;
|
||||
onSave: (saved: RowSaved) => Promise<void>;
|
||||
onReject: () => Promise<void>;
|
||||
}
|
||||
|
||||
function DocumentRow({ doc, customerIndex, canReview, onSave, onReject }: DocumentRowProps) {
|
||||
const [v, setV] = useState({
|
||||
policyNumber: doc.extractedPolicyNumber ?? "",
|
||||
insuredName: doc.extractedInsuredName ?? "",
|
||||
additionalInsured: doc.extractedAdditionalInsured ?? "",
|
||||
agentName: doc.extractedAgentName ?? "",
|
||||
legalAddress: doc.extractedLegalAddress ?? "",
|
||||
zip: doc.extractedZip ?? "",
|
||||
policyFrom: doc.extractedPolicyFrom?.slice(0, 10) ?? "",
|
||||
policyTo: doc.extractedPolicyTo?.slice(0, 10) ?? "",
|
||||
policyDate: doc.extractedPolicyDate?.slice(0, 10) ?? "",
|
||||
currency: doc.extractedCurrency ?? "USD",
|
||||
netPremium: doc.extractedNetPremium ?? "",
|
||||
total: doc.extractedTotal ?? "",
|
||||
premiumPayment: doc.extractedPremiumPayment ?? "",
|
||||
postPremium: doc.extractedNetPremium != null && Number(doc.extractedNetPremium) > 0,
|
||||
});
|
||||
const [customerId, setCustomerId] = useState(
|
||||
doc.matchedCustomer?.id ?? doc.matchedPolicy?.customerId ?? "",
|
||||
);
|
||||
const [customerName, setCustomerName] = useState(
|
||||
doc.matchedCustomer?.name ?? doc.matchedPolicy?.customer.name ?? "",
|
||||
);
|
||||
const [policyId, setPolicyId] = useState(doc.matchedPolicy?.id ?? "");
|
||||
const [busy, setBusy] = useState(false);
|
||||
const [err, setErr] = useState<string | null>(null);
|
||||
|
||||
function set<K extends keyof typeof v>(k: K, val: (typeof v)[K]) {
|
||||
setV((p) => ({ ...p, [k]: val }));
|
||||
}
|
||||
|
||||
async function save() {
|
||||
setBusy(true);
|
||||
setErr(null);
|
||||
try {
|
||||
const numOrUndef = (s: string) => (s.trim() === "" ? undefined : Number(s));
|
||||
const trimOrUndef = (s: string) => (s.trim() === "" ? undefined : s.trim());
|
||||
const currency = v.currency || undefined;
|
||||
const reviewInput: PolicyOcrReviewInput = {
|
||||
policyNumber: trimOrUndef(v.policyNumber),
|
||||
insuredName: trimOrUndef(v.insuredName),
|
||||
additionalInsured: trimOrUndef(v.additionalInsured),
|
||||
agentName: trimOrUndef(v.agentName),
|
||||
legalAddress: trimOrUndef(v.legalAddress),
|
||||
zip: trimOrUndef(v.zip),
|
||||
policyFrom: v.policyFrom || undefined,
|
||||
policyTo: v.policyTo || undefined,
|
||||
policyDate: v.policyDate || undefined,
|
||||
currency,
|
||||
netPremium: numOrUndef(v.netPremium),
|
||||
total: numOrUndef(v.total),
|
||||
premiumPayment: trimOrUndef(v.premiumPayment),
|
||||
matchedPolicyId: policyId || undefined,
|
||||
matchedCustomerId: !policyId && customerId ? customerId : undefined,
|
||||
forceConfirm: true,
|
||||
};
|
||||
const confirmInput: PolicyOcrConfirmDocument = {
|
||||
documentId: doc.id,
|
||||
policyId: policyId || undefined,
|
||||
customerId: !policyId && customerId ? customerId : undefined,
|
||||
policyNumber: reviewInput.policyNumber,
|
||||
insuredName: reviewInput.insuredName,
|
||||
additionalInsured: reviewInput.additionalInsured,
|
||||
agentName: reviewInput.agentName,
|
||||
legalAddress: reviewInput.legalAddress,
|
||||
zip: reviewInput.zip,
|
||||
policyFrom: reviewInput.policyFrom,
|
||||
policyTo: reviewInput.policyTo,
|
||||
policyDate: reviewInput.policyDate,
|
||||
currency: (currency as "MXN" | "USD" | "EUR" | undefined) ?? undefined,
|
||||
netPremium: reviewInput.netPremium,
|
||||
total: reviewInput.total,
|
||||
premiumPayment: reviewInput.premiumPayment,
|
||||
coveragesJson: (doc.extractedCoveragesJson ?? undefined) as
|
||||
| PolicyOcrCoverage[]
|
||||
| undefined,
|
||||
postPremium: v.postPremium,
|
||||
};
|
||||
await onSave({ reviewInput, confirmInput });
|
||||
} catch (e) {
|
||||
setErr((e as Error)?.message ?? "No se pudo guardar.");
|
||||
} finally {
|
||||
setBusy(false);
|
||||
}
|
||||
}
|
||||
|
||||
const locked = doc.status === "POSTED" || doc.status === "REJECTED";
|
||||
const matchedExisting = !!doc.matchedPolicy;
|
||||
const candidates = doc.matchCandidates ?? [];
|
||||
|
||||
return (
|
||||
<article className="card" style={{ padding: 16 }}>
|
||||
<header className="row" style={{ gap: 12, alignItems: "center" }}>
|
||||
<span className="tag">{STATUS_LABEL[doc.status] ?? doc.status}</span>
|
||||
<span className="page-sub">Página {doc.pageNumber}</span>
|
||||
{doc.extractedPolicyNumber && (
|
||||
<strong style={{ marginLeft: 8 }}>{doc.extractedPolicyNumber}</strong>
|
||||
)}
|
||||
{doc.extractedInsuredName && (
|
||||
<span className="page-sub">· {doc.extractedInsuredName}</span>
|
||||
)}
|
||||
</header>
|
||||
|
||||
<div className="doc-detail">
|
||||
{/*
|
||||
* Embed the source PDF the office uploaded. One PDF = one parsed
|
||||
* policy, so the browser's PDF viewer handles multi-page navigation
|
||||
* natively; we don't need to render individual pages on the server.
|
||||
*/}
|
||||
<iframe
|
||||
src={policyOcrDocumentUrl(doc.id)}
|
||||
title={`Póliza ${doc.extractedPolicyNumber ?? doc.pageNumber}`}
|
||||
style={{
|
||||
width: "100%",
|
||||
height: 720,
|
||||
border: "1px solid var(--border, #ddd)",
|
||||
borderRadius: 6,
|
||||
background: "#fff",
|
||||
}}
|
||||
/>
|
||||
|
||||
<div className="stack" style={{ flex: 1, minWidth: 0 }}>
|
||||
{doc.matchNote && <p className="page-sub">{doc.matchNote}</p>}
|
||||
|
||||
{matchedExisting ? (
|
||||
<div className="state-box">
|
||||
Coincide con la póliza{" "}
|
||||
<strong>{doc.matchedPolicy?.policyNumber}</strong> del cliente{" "}
|
||||
<strong>{doc.matchedPolicy?.customer.name}</strong>.
|
||||
</div>
|
||||
) : candidates.length > 1 ? (
|
||||
<div className="state-box state-warn">
|
||||
{candidates.length} pólizas comparten este número. Elija
|
||||
manualmente abajo.
|
||||
</div>
|
||||
) : (
|
||||
<div className="state-box">
|
||||
No se encontró una póliza con este número. Se creará una nueva
|
||||
bajo el cliente que elija abajo.
|
||||
</div>
|
||||
)}
|
||||
|
||||
<fieldset className="form-grid" disabled={locked || !canReview}>
|
||||
<Field label="Número de póliza">
|
||||
<input
|
||||
className="input"
|
||||
value={v.policyNumber}
|
||||
onChange={(e) => set("policyNumber", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Asegurado">
|
||||
<input
|
||||
className="input"
|
||||
value={v.insuredName}
|
||||
onChange={(e) => set("insuredName", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Asegurado adicional">
|
||||
<input
|
||||
className="input"
|
||||
value={v.additionalInsured}
|
||||
onChange={(e) => set("additionalInsured", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Agente">
|
||||
<input
|
||||
className="input"
|
||||
value={v.agentName}
|
||||
onChange={(e) => set("agentName", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Desde">
|
||||
<input
|
||||
className="input"
|
||||
type="date"
|
||||
value={v.policyFrom}
|
||||
onChange={(e) => set("policyFrom", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Hasta">
|
||||
<input
|
||||
className="input"
|
||||
type="date"
|
||||
value={v.policyTo}
|
||||
onChange={(e) => set("policyTo", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Fecha de firma">
|
||||
<input
|
||||
className="input"
|
||||
type="date"
|
||||
value={v.policyDate}
|
||||
onChange={(e) => set("policyDate", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Moneda">
|
||||
<select
|
||||
className="input select"
|
||||
value={v.currency}
|
||||
onChange={(e) => set("currency", e.target.value)}
|
||||
>
|
||||
<option value="MXN">MXN</option>
|
||||
<option value="USD">USD</option>
|
||||
<option value="EUR">EUR</option>
|
||||
</select>
|
||||
</Field>
|
||||
<Field label="Prima neta">
|
||||
<input
|
||||
className="input"
|
||||
type="number"
|
||||
step="0.01"
|
||||
value={v.netPremium}
|
||||
onChange={(e) => set("netPremium", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Total">
|
||||
<input
|
||||
className="input"
|
||||
type="number"
|
||||
step="0.01"
|
||||
value={v.total}
|
||||
onChange={(e) => set("total", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Pago de prima">
|
||||
<input
|
||||
className="input"
|
||||
value={v.premiumPayment}
|
||||
onChange={(e) => set("premiumPayment", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="Dirección">
|
||||
<input
|
||||
className="input"
|
||||
value={v.legalAddress}
|
||||
onChange={(e) => set("legalAddress", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
<Field label="C.P.">
|
||||
<input
|
||||
className="input"
|
||||
value={v.zip}
|
||||
onChange={(e) => set("zip", e.target.value)}
|
||||
/>
|
||||
</Field>
|
||||
</fieldset>
|
||||
|
||||
{doc.extractedCoveragesJson && doc.extractedCoveragesJson.length > 0 && (
|
||||
<details>
|
||||
<summary>
|
||||
Coberturas ({doc.extractedCoveragesJson.length}) ·{" "}
|
||||
{formatMoney(
|
||||
doc.extractedCoveragesJson
|
||||
.map((c) => Number(c.insuredAmount ?? 0))
|
||||
.reduce((a, b) => a + b, 0)
|
||||
.toString(),
|
||||
v.currency,
|
||||
)}
|
||||
</summary>
|
||||
<table className="tx-table" style={{ marginTop: 8 }}>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Riesgo</th>
|
||||
<th className="num">Suma</th>
|
||||
<th>Deducible</th>
|
||||
<th>Participación</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{doc.extractedCoveragesJson.map((c, i) => (
|
||||
<tr key={i}>
|
||||
<td>{c.risk}</td>
|
||||
<td className="num">
|
||||
{formatMoney(c.insuredAmount?.toString() ?? null, v.currency)}
|
||||
</td>
|
||||
<td>{c.deductible ?? "—"}</td>
|
||||
<td>{c.lossParticipation ?? "—"}</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</details>
|
||||
)}
|
||||
|
||||
{candidates.length > 1 && (
|
||||
<Field label="Póliza destino">
|
||||
<select
|
||||
className="input select"
|
||||
value={policyId}
|
||||
onChange={(e) => {
|
||||
setPolicyId(e.target.value);
|
||||
const found = candidates.find((c) => c.policyId === e.target.value);
|
||||
if (found) {
|
||||
setCustomerId(found.customerId);
|
||||
setCustomerName(customerIndex[found.customerId]?.name ?? found.customerName);
|
||||
}
|
||||
}}
|
||||
>
|
||||
<option value="">— elegir póliza —</option>
|
||||
{candidates.map((c) => (
|
||||
<option key={c.policyId} value={c.policyId}>
|
||||
{c.policyNumber} · {c.customerName}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</Field>
|
||||
)}
|
||||
|
||||
{!policyId && (
|
||||
<Field label={matchedExisting ? "Cliente" : "Cliente (póliza nueva)"}>
|
||||
<CustomerPicker
|
||||
value={customerId}
|
||||
valueName={customerName}
|
||||
onPick={(id, name) => {
|
||||
setCustomerId(id);
|
||||
setCustomerName(name);
|
||||
}}
|
||||
/>
|
||||
</Field>
|
||||
)}
|
||||
|
||||
<label className="field">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={v.postPremium}
|
||||
onChange={(e) => set("postPremium", e.target.checked)}
|
||||
disabled={!v.netPremium || Number(v.netPremium) <= 0}
|
||||
/>{" "}
|
||||
Registrar prima en el estado de cuenta
|
||||
</label>
|
||||
|
||||
{err && <div className="state-box state-error">{err}</div>}
|
||||
|
||||
{!locked && canReview && (
|
||||
<div className="row" style={{ gap: 8 }}>
|
||||
<button type="button" className="btn btn-primary" disabled={busy} onClick={save}>
|
||||
{busy ? "Guardando…" : "Guardar revisión"}
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-ghost"
|
||||
onClick={() => void onReject()}
|
||||
>
|
||||
Rechazar
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
</article>
|
||||
);
|
||||
}
|
||||
|
||||
function Field({ label, children }: { label: string; children: React.ReactNode }) {
|
||||
return (
|
||||
<label className="field">
|
||||
<span className="field-label">{label}</span>
|
||||
{children}
|
||||
</label>
|
||||
);
|
||||
}
|
||||
@@ -28,9 +28,16 @@ import type { ServiceKind, StatementBatch, StatementBatchStatus } from "@/lib/ty
|
||||
*/
|
||||
|
||||
/** The kinds the parsers actually recognise today. */
|
||||
const SUPPORTED: ServiceKind[] = ["ELECTRIC", "WATER", "TELEPHONE"];
|
||||
const SUPPORTED: ServiceKind[] = [
|
||||
"ELECTRIC",
|
||||
"WATER",
|
||||
"TELEPHONE",
|
||||
"GAS",
|
||||
"PROPERTY_TAX",
|
||||
"FEDERAL_ZONE",
|
||||
];
|
||||
/** Uploadable, but every page will land in review until a parser learns it. */
|
||||
const OTHER_KINDS: ServiceKind[] = ["GAS", "PROPERTY_TAX", "FEDERAL_ZONE", "CABLE"];
|
||||
const OTHER_KINDS: ServiceKind[] = ["CABLE"];
|
||||
|
||||
const STATUS_LABEL: Record<StatementBatchStatus, string> = {
|
||||
UPLOADED: "Recibido",
|
||||
|
||||
@@ -49,6 +49,12 @@ import type {
|
||||
PolicySort,
|
||||
PolicyStats,
|
||||
PolicyStatus,
|
||||
PolicyOcrBatch,
|
||||
PolicyOcrBatchDetail,
|
||||
PolicyOcrDocument,
|
||||
PolicyOcrReviewInput,
|
||||
PolicyOcrConfirmInput,
|
||||
PolicyOcrConfirmResult,
|
||||
LookupsResponse,
|
||||
OpsJob,
|
||||
OpsJobKind,
|
||||
@@ -1077,3 +1083,96 @@ export function confirmStatementBatch(
|
||||
export function statementPageUrl(documentId: string): string {
|
||||
return `${API_ORIGIN}/statements/documents/${documentId}/page`;
|
||||
}
|
||||
|
||||
/* ----------------------------------------------------- Policy OCR (GMX) */
|
||||
|
||||
export function getPolicyOcrStatus(): Promise<{
|
||||
ocrAvailable: boolean;
|
||||
storageAvailable: boolean;
|
||||
}> {
|
||||
return apiFetch("/policy-ocr/status");
|
||||
}
|
||||
|
||||
export function listPolicyOcrBatches(
|
||||
page = 1,
|
||||
pageSize = 25,
|
||||
): Promise<{
|
||||
items: PolicyOcrBatch[];
|
||||
total: number;
|
||||
page: number;
|
||||
pageSize: number;
|
||||
pageCount: number;
|
||||
}> {
|
||||
return apiFetch(`/policy-ocr/batches?page=${page}&pageSize=${pageSize}`);
|
||||
}
|
||||
|
||||
export function getPolicyOcrBatch(id: string): Promise<PolicyOcrBatchDetail> {
|
||||
return apiFetch(`/policy-ocr/batches/${id}`);
|
||||
}
|
||||
|
||||
export function listPolicyOcrDocuments(batchId: string): Promise<PolicyOcrDocument[]> {
|
||||
return apiFetch(`/policy-ocr/batches/${batchId}/documents`);
|
||||
}
|
||||
|
||||
export async function uploadPolicyOcrBatch(
|
||||
files: File[],
|
||||
label?: string,
|
||||
): Promise<PolicyOcrBatch> {
|
||||
const body = new FormData();
|
||||
for (const f of files) body.append("files", f, f.name);
|
||||
const qs = new URLSearchParams();
|
||||
if (label) qs.set("label", label);
|
||||
|
||||
const res = await fetch(
|
||||
`${API_ORIGIN}/policy-ocr/batches${qs.toString() ? `?${qs}` : ""}`,
|
||||
{
|
||||
method: "POST",
|
||||
credentials: "include",
|
||||
body,
|
||||
},
|
||||
);
|
||||
if (!res.ok) {
|
||||
let message = `Error ${res.status}`;
|
||||
try {
|
||||
const b = await res.json();
|
||||
if (b?.message) message = b.message;
|
||||
} catch {
|
||||
/* non-JSON error body */
|
||||
}
|
||||
throw new Error(message);
|
||||
}
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export function reviewPolicyOcrDocument(
|
||||
id: string,
|
||||
input: PolicyOcrReviewInput,
|
||||
): Promise<PolicyOcrDocument> {
|
||||
return apiFetch(`/policy-ocr/documents/${id}`, {
|
||||
method: "PATCH",
|
||||
body: JSON.stringify(input),
|
||||
});
|
||||
}
|
||||
|
||||
export function rejectPolicyOcrDocument(id: string): Promise<PolicyOcrDocument> {
|
||||
return apiFetch(`/policy-ocr/documents/${id}/reject`, { method: "POST" });
|
||||
}
|
||||
|
||||
export function confirmPolicyOcrBatch(
|
||||
batchId: string,
|
||||
input: PolicyOcrConfirmInput,
|
||||
): Promise<PolicyOcrConfirmResult> {
|
||||
return apiFetch(`/policy-ocr/batches/${batchId}/confirm`, {
|
||||
method: "POST",
|
||||
body: JSON.stringify(input),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* URL for the source PDF of a parsed policy document. The endpoint returns
|
||||
* the original upload (one PDF = one parsed policy), not a rendered page
|
||||
* image, so the review screen embeds it in an iframe.
|
||||
*/
|
||||
export function policyOcrDocumentUrl(documentId: string): string {
|
||||
return `${API_ORIGIN}/policy-ocr/documents/${documentId}/page`;
|
||||
}
|
||||
|
||||
@@ -12,6 +12,8 @@ export type Ability =
|
||||
| "policy:create"
|
||||
| "policy:update"
|
||||
| "policy:delete"
|
||||
| "policy:ingest"
|
||||
| "policy:ocr-review"
|
||||
| "property:create"
|
||||
| "property:update"
|
||||
| "property:delete"
|
||||
@@ -1293,3 +1295,140 @@ export interface ConfirmBatchResult {
|
||||
total: string;
|
||||
checkNumber: string;
|
||||
}
|
||||
|
||||
/* ------------------------------------------ Policy OCR intake (GMX) */
|
||||
|
||||
export type PolicyOcrBatchStatus =
|
||||
| "UPLOADED"
|
||||
| "PROCESSING"
|
||||
| "READY_FOR_REVIEW"
|
||||
| "COMPLETED"
|
||||
| "FAILED";
|
||||
|
||||
export type PolicyOcrDocumentStatus =
|
||||
| "PENDING_OCR"
|
||||
| "OCR_FAILED"
|
||||
| "NEEDS_REVIEW"
|
||||
| "MATCHED"
|
||||
| "CONFIRMED"
|
||||
| "POSTED"
|
||||
| "REJECTED";
|
||||
|
||||
export interface PolicyOcrBatch {
|
||||
id: string;
|
||||
provider: string;
|
||||
status: PolicyOcrBatchStatus;
|
||||
label: string | null;
|
||||
fileCount: number;
|
||||
error: string | null;
|
||||
createdAt: string;
|
||||
completedAt: string | null;
|
||||
uploadedBy?: { name: string };
|
||||
_count?: { documents: number };
|
||||
}
|
||||
|
||||
export interface PolicyOcrBatchDetail extends PolicyOcrBatch {
|
||||
byStatus: Partial<Record<PolicyOcrDocumentStatus, number>>;
|
||||
}
|
||||
|
||||
export interface PolicyOcrCoverage {
|
||||
risk: string;
|
||||
insuredAmount: number | null;
|
||||
deductible: string | null;
|
||||
lossParticipation: string | null;
|
||||
}
|
||||
|
||||
export interface PolicyOcrMatchCandidate {
|
||||
policyId: string;
|
||||
customerId: string;
|
||||
customerName: string;
|
||||
policyNumber: string;
|
||||
}
|
||||
|
||||
export interface PolicyOcrDocument {
|
||||
id: string;
|
||||
pageNumber: number;
|
||||
status: PolicyOcrDocumentStatus;
|
||||
provider: string | null;
|
||||
ocrConfidence: string | null;
|
||||
extractedPolicyNumber: string | null;
|
||||
extractedInsuredName: string | null;
|
||||
extractedAdditionalInsured: string | null;
|
||||
extractedAgentName: string | null;
|
||||
extractedLegalAddress: string | null;
|
||||
extractedZip: string | null;
|
||||
extractedPolicyFrom: string | null;
|
||||
extractedPolicyTo: string | null;
|
||||
extractedPolicyDate: string | null;
|
||||
extractedCurrency: string | null;
|
||||
extractedNetPremium: string | null;
|
||||
extractedPolicyFee: string | null;
|
||||
extractedBrokerFee: string | null;
|
||||
extractedTotal: string | null;
|
||||
extractedCoveragesJson: PolicyOcrCoverage[] | null;
|
||||
extractedPremiumPayment: string | null;
|
||||
matchedPolicy: {
|
||||
id: string;
|
||||
policyNumber: string | null;
|
||||
customerId: string;
|
||||
customer: { name: string };
|
||||
} | null;
|
||||
matchedCustomer: { id: string; name: string } | null;
|
||||
matchCandidates: PolicyOcrMatchCandidate[] | null;
|
||||
matchNote: string | null;
|
||||
}
|
||||
|
||||
export interface PolicyOcrReviewInput {
|
||||
policyNumber?: string;
|
||||
insuredName?: string;
|
||||
additionalInsured?: string;
|
||||
agentName?: string;
|
||||
legalAddress?: string;
|
||||
zip?: string;
|
||||
policyFrom?: string;
|
||||
policyTo?: string;
|
||||
policyDate?: string;
|
||||
currency?: string;
|
||||
netPremium?: number;
|
||||
policyFee?: number;
|
||||
brokerFee?: number;
|
||||
total?: number;
|
||||
premiumPayment?: string;
|
||||
coveragesJson?: PolicyOcrCoverage[];
|
||||
matchedPolicyId?: string;
|
||||
matchedCustomerId?: string;
|
||||
forceConfirm?: boolean;
|
||||
}
|
||||
|
||||
export interface PolicyOcrConfirmDocument {
|
||||
documentId: string;
|
||||
policyId?: string;
|
||||
customerId?: string;
|
||||
policyNumber?: string;
|
||||
insuredName?: string;
|
||||
additionalInsured?: string;
|
||||
agentName?: string;
|
||||
legalAddress?: string;
|
||||
zip?: string;
|
||||
policyFrom?: string;
|
||||
policyTo?: string;
|
||||
policyDate?: string;
|
||||
currency?: string;
|
||||
netPremium?: number;
|
||||
policyFee?: number;
|
||||
brokerFee?: number;
|
||||
total?: number;
|
||||
premiumPayment?: string;
|
||||
coveragesJson?: PolicyOcrCoverage[];
|
||||
postPremium?: boolean;
|
||||
}
|
||||
|
||||
export interface PolicyOcrConfirmInput {
|
||||
documents: PolicyOcrConfirmDocument[];
|
||||
}
|
||||
|
||||
export interface PolicyOcrConfirmResult {
|
||||
applied: number;
|
||||
policies: string[];
|
||||
postedTransactions: number;
|
||||
}
|
||||
|
||||
@@ -134,7 +134,8 @@ single-movement form.
|
||||
> **BUILT — 2026-08-01.** Implemented and verified end to end against real
|
||||
> scanned statements. `apps/api/src/statements/` holds the module: a swappable
|
||||
> `OcrProvider` seam with a self-hosted Tesseract implementation, per-provider
|
||||
> parsers for CFE / CESPT / Telnor, a scoped matcher, and a review queue that
|
||||
> parsers for CFE / CESPT / Telnor / gas / predial, a scoped matcher, and a
|
||||
> review queue that
|
||||
> posts through `BillingService.createBatch` with `source: "OCR"`. Web:
|
||||
> the "Captura automática (OCR)" tab of the Captura screen (upload + batch
|
||||
> list) and `/recibos/:id` (review queue with the page image beside the
|
||||
@@ -207,6 +208,102 @@ single-movement form.
|
||||
> each bill (`9`, `405`, `406`); Tesseract read `405` as `205`. Handwriting is
|
||||
> a review hint at best and is deliberately not an input to matching.
|
||||
|
||||
> **EXTENDED — gas and predial, 2026-08-01.** A second corpus (14 documents,
|
||||
> 29 pages: five municipal predial batches and ten gas invoices) added four
|
||||
> parsers — `GAS TIJUANA` plus one per municipality, because Tijuana, Rosarito
|
||||
> and Ensenada issue three completely different documents. End to end against
|
||||
> the dev database that is **21/29 auto-matched, 22/29 identified**, with the
|
||||
> provider read on 29/29 and an amount on 26/29.
|
||||
>
|
||||
> The eight review cases are all legitimate: five Tijuana pages whose municipal
|
||||
> account is not yet on file (see below), one clave not in the book, one page
|
||||
> too poorly scanned to read a clave at all, and one gas account shared by two
|
||||
> services. Excluding the structural Tijuana case, that is 21/24.
|
||||
>
|
||||
> **Five things this corpus proved:**
|
||||
>
|
||||
> 1. **Not every statement is a scan.** The gas company sends born-digital CFDI
|
||||
> invoices whose text layer is exact. Rasterising and re-recognising those
|
||||
> can only lose information — one sample turned `MEDIDOR: VM01014426` into
|
||||
> `ar (LTR): 014420` — so `OcrProvider.textPages` reads the embedded layer
|
||||
> first (`pdftotext -bbox-layout`, same poppler package as `pdftoppm`) and
|
||||
> OCR stays the fallback for real scans. Page images are still rendered and
|
||||
> stored either way, because the reviewer needs to see the paper.
|
||||
> 2. **The clave catastral is not two letters and six digits.** Positions four
|
||||
> through eight are digits in all 932 stored claves, but the third is a
|
||||
> letter in fifteen of them (`MMB01041`, `CGH52121`). Digitising the whole
|
||||
> tail maps that `B` to an `8` and produces a key matching no property.
|
||||
> 3. **Tijuana predial prints no clave catastral at all.** Its only identifier
|
||||
> is an 8-digit municipal account, carried in a 32-digit payment barcode
|
||||
> (`account(8) + DDMMYY + amount(9) + folio(9)`) that the legacy database
|
||||
> never held. It goes in `PROPERTY_TAX.meterNumber` — the same column gas
|
||||
> uses, and for the same reason: `accountNumber` holds `DATMEX.predial`,
|
||||
> which is not a per-property key and overwriting it would destroy the only
|
||||
> link back to the original records. So Tijuana pages start cold and are
|
||||
> taught by the first confirm, exactly like gas.
|
||||
> 4. **On Rosarito and Ensenada the clave is the primary key, not a fallback.**
|
||||
> Those receipts print nothing else, so a unique clave hit there is a real
|
||||
> match and auto-matches; on a utility bill that merely happens to print one
|
||||
> it stays a review hint, as before.
|
||||
> 5. **A misread `$` is the dangerous failure, not a missing one.** An Ensenada
|
||||
> receipt for `$2,203.00` OCR'd as `82,203.00` — the dollar sign read as an
|
||||
> 8, which would post a charge 37× too large and look entirely ordinary in
|
||||
> the ledger. Every predial amount therefore requires a literal `$`, and a
|
||||
> page that cannot produce one reports no amount and goes to review. Two of
|
||||
> the 29 pages take that path, which is the correct outcome for both.
|
||||
>
|
||||
> Regression cover for all of the above lives in
|
||||
> `statement-parser.spec.ts` and `tesseract.provider.spec.ts`; every fixture in
|
||||
> them is a verbatim OCR excerpt from a real receipt.
|
||||
|
||||
> **EXTENDED — zona federal, 2026-08-01.** A third corpus (one document, 8
|
||||
> pages of Tijuana "Zona Federal Marítimo Terrestre" receipts — the federal
|
||||
> maritime-zone occupancy fee billed on beachfront lots) added the
|
||||
> `ZONA FEDERAL TIJUANA` parser. Provider read on 8/8, amount on 8/8 (all
|
||||
> eight verified against the paper), concession clave on 6/8, period on 8/8,
|
||||
> payment deadline on 2/8. Nothing auto-matched, and nothing could have — see
|
||||
> point 2.
|
||||
>
|
||||
> **Four things this corpus proved:**
|
||||
>
|
||||
> 1. **Tijuana bills predial and zona federal from the same treasury.** Same
|
||||
> "Ayuntamiento de Tijuana" header, same Paseo del Centenario address, same
|
||||
> `ATB-541201` RFC — every discriminator the predial parser uses matches a
|
||||
> zona federal page too, so whichever rule is asked first wins. The words
|
||||
> only this layout prints are `Marítimo Terrestre`, so its brand rule is
|
||||
> asked ahead of all three predial ones.
|
||||
> 2. **`FEDERAL_ZONE.accountNumber` is an amount, not a reference.** It holds
|
||||
> `DATMEX.zfed`, whose 77 values include `246.06`, `2369.09`, `22653.94` and
|
||||
> a negative `-1679`; the concession claves the receipts are keyed by
|
||||
> (`12-T -012`, `14-D -014`) appear nowhere in the database. Matching on that
|
||||
> column could never hit — and because every row already has a value, the
|
||||
> `[field]: null` guards on learning and on the blank-service fill would
|
||||
> never fire either, so every page would return to review every bimester
|
||||
> forever. The clave moves to `meterNumber`, joining gas and Tijuana predial,
|
||||
> and the first confirm teaches the match. This is the same trap as
|
||||
> `policies.total` and `PROPERTY_TAX.accountNumber`: a legacy column whose
|
||||
> name promises an identifier and whose contents are something else.
|
||||
> 3. **The payable figure is not the printed subtotal.** The municipality rounds
|
||||
> to whole pesos and prints the difference as its own `Ajuste Ley Hacienda
|
||||
> Mpal` line — `-$0.05` against a 591.05 subtotal, `$0.21` against 2,872.79.
|
||||
> The "Total a pagar" box that carries the rounded figure sits on a grey fill
|
||||
> and OCR'd on 1 of 8 pages; the SubTotal row read on 8 of 8. So the amount
|
||||
> is the rounded subtotal, cross-checked against the printed box wherever it
|
||||
> survives (it agreed).
|
||||
> 4. **The office's own highlighter is an OCR failure mode.** Both pages that
|
||||
> lost their clave lost it to a marker stroke drawn across the `Clave:` line
|
||||
> — not to scan quality, which was otherwise fine. The clave is printed twice
|
||||
> (receipt and stub), which rescued a third page whose heading was struck
|
||||
> but whose stub was not; where both copies are struck, the page reports no
|
||||
> clave and goes to review rather than guessing.
|
||||
>
|
||||
> **Not attempted:** deriving the payment deadline from the bimester. It is the
|
||||
> 17th of the month after the bimester closes on a current bill, but four of
|
||||
> these eight are late — they carry a $1,000 `Multa` — and print a
|
||||
> recalculated deadline a month out. A derived date would be wrong on exactly
|
||||
> the pages a human most wants to look at, so an unreadable deadline stays
|
||||
> null.
|
||||
|
||||
### Motivation (from the meeting)
|
||||
|
||||
Each utility company (CFE, water, phone, gas...) sends 300+ individual
|
||||
@@ -330,8 +427,8 @@ Per the meeting notes' own field list:
|
||||
| Agua — Número de cuenta | `WATER` | `accountNumber` | `AGUA` | ✅ populated today |
|
||||
| Zona Fed — Número de Zona Federal | `FEDERAL_ZONE` | `accountNumber` | `ZFED` | ✅ populated today |
|
||||
| Tel — Número de teléfono | `TELEPHONE` *(new)* | `accountNumber` | `Property.phone1/2/3` (currently on `Property`, not `PropertyService`) | ⚠️ schema gap — see below |
|
||||
| Impuesto — Clave Catastral | `PROPERTY_TAX` | `accountNumber` | migrated from `PREDIAL`, **not** `CLAVE` | ⚠️ needs verification — see below |
|
||||
| Gas — Número de medidor | `GAS` | `meterNumber` | not populated — folded into free-text `notes` today | ⚠️ data gap — see below |
|
||||
| Impuesto — Clave Catastral | `PROPERTY_TAX` | `Property.cadastralKey`, plus `meterNumber` for Tijuana's municipal account | `CLAVE`; `PREDIAL` is left on `accountNumber` and never matched against | ✅ built — see the 2026-08-01 extension note |
|
||||
| Gas — Número de medidor | `GAS` | `meterNumber` | not populated — folded into free-text `notes` today | ✅ 160/334 recovered from `notes` |
|
||||
|
||||
Confidence rule of thumb once a field is confirmed populated, tune after
|
||||
seeing real statements:
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "jorgecuadros-platform",
|
||||
"version": "1.0.5",
|
||||
"version": "1.0.6",
|
||||
"private": true,
|
||||
"workspaces": [
|
||||
"apps/*",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jorgecuadros/database",
|
||||
"version": "1.0.5",
|
||||
"version": "1.0.6",
|
||||
"private": true,
|
||||
"main": "generated/client/index.js",
|
||||
"types": "generated/client/index.d.ts",
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
-- CreateTable
|
||||
CREATE TABLE `policy_ocr_batches` (
|
||||
`id` VARCHAR(191) NOT NULL,
|
||||
`provider` VARCHAR(191) NOT NULL DEFAULT 'GMX',
|
||||
`status` ENUM('UPLOADED', 'PROCESSING', 'READY_FOR_REVIEW', 'COMPLETED', 'FAILED') NOT NULL DEFAULT 'UPLOADED',
|
||||
`uploadedById` VARCHAR(191) NOT NULL,
|
||||
`label` VARCHAR(191) NULL,
|
||||
`fileCount` INTEGER NOT NULL DEFAULT 0,
|
||||
`error` TEXT NULL,
|
||||
`createdAt` DATETIME(3) NOT NULL DEFAULT CURRENT_TIMESTAMP(3),
|
||||
`completedAt` DATETIME(3) NULL,
|
||||
|
||||
INDEX `policy_ocr_batches_status_createdAt_idx`(`status`, `createdAt`),
|
||||
PRIMARY KEY (`id`)
|
||||
) DEFAULT CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci;
|
||||
|
||||
-- CreateTable
|
||||
CREATE TABLE `policy_ocr_documents` (
|
||||
`id` VARCHAR(191) NOT NULL,
|
||||
`batchId` VARCHAR(191) NOT NULL,
|
||||
`pageNumber` INTEGER NOT NULL,
|
||||
`storageKey` VARCHAR(191) NOT NULL,
|
||||
`status` ENUM('PENDING_OCR', 'OCR_FAILED', 'NEEDS_REVIEW', 'MATCHED', 'CONFIRMED', 'POSTED', 'REJECTED') NOT NULL DEFAULT 'PENDING_OCR',
|
||||
`ocrRawText` TEXT NULL,
|
||||
`ocrConfidence` DECIMAL(4, 3) NULL,
|
||||
`provider` VARCHAR(191) NULL,
|
||||
`extractedPolicyNumber` VARCHAR(191) NULL,
|
||||
`extractedInsuredName` VARCHAR(191) NULL,
|
||||
`extractedAdditionalInsured` VARCHAR(191) NULL,
|
||||
`extractedAgentName` VARCHAR(191) NULL,
|
||||
`extractedLegalAddress` TEXT NULL,
|
||||
`extractedZip` VARCHAR(191) NULL,
|
||||
`extractedPolicyFrom` DATETIME(3) NULL,
|
||||
`extractedPolicyTo` DATETIME(3) NULL,
|
||||
`extractedPolicyDate` DATETIME(3) NULL,
|
||||
`extractedCurrency` VARCHAR(191) NULL,
|
||||
`extractedNetPremium` DECIMAL(12, 2) NULL,
|
||||
`extractedPolicyFee` DECIMAL(12, 2) NULL,
|
||||
`extractedBrokerFee` DECIMAL(12, 2) NULL,
|
||||
`extractedTotal` DECIMAL(12, 2) NULL,
|
||||
`extractedCoveragesJson` JSON NULL,
|
||||
`extractedPremiumPayment` VARCHAR(191) NULL,
|
||||
`matchedPolicyId` VARCHAR(191) NULL,
|
||||
`matchedCustomerId` VARCHAR(191) NULL,
|
||||
`matchCandidates` JSON NULL,
|
||||
`matchNote` VARCHAR(191) NULL,
|
||||
`reviewedById` VARCHAR(191) NULL,
|
||||
`reviewedAt` DATETIME(3) NULL,
|
||||
`createdPolicyId` VARCHAR(191) NULL,
|
||||
`postedTransactionId` VARCHAR(191) NULL,
|
||||
`createdAt` DATETIME(3) NOT NULL DEFAULT CURRENT_TIMESTAMP(3),
|
||||
|
||||
UNIQUE INDEX `policy_ocr_documents_createdPolicyId_key`(`createdPolicyId`),
|
||||
UNIQUE INDEX `policy_ocr_documents_postedTransactionId_key`(`postedTransactionId`),
|
||||
INDEX `policy_ocr_documents_status_idx`(`status`),
|
||||
INDEX `policy_ocr_documents_matchedCustomerId_idx`(`matchedCustomerId`),
|
||||
UNIQUE INDEX `policy_ocr_documents_batchId_pageNumber_key`(`batchId`, `pageNumber`),
|
||||
PRIMARY KEY (`id`)
|
||||
) DEFAULT CHARACTER SET utf8mb4 COLLATE utf8mb4_unicode_ci;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_batches` ADD CONSTRAINT `policy_ocr_batches_uploadedById_fkey` FOREIGN KEY (`uploadedById`) REFERENCES `users`(`id`) ON DELETE RESTRICT ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_batchId_fkey` FOREIGN KEY (`batchId`) REFERENCES `policy_ocr_batches`(`id`) ON DELETE CASCADE ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_matchedPolicyId_fkey` FOREIGN KEY (`matchedPolicyId`) REFERENCES `policies`(`id`) ON DELETE SET NULL ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_matchedCustomerId_fkey` FOREIGN KEY (`matchedCustomerId`) REFERENCES `customers`(`id`) ON DELETE SET NULL ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_reviewedById_fkey` FOREIGN KEY (`reviewedById`) REFERENCES `users`(`id`) ON DELETE SET NULL ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_createdPolicyId_fkey` FOREIGN KEY (`createdPolicyId`) REFERENCES `policies`(`id`) ON DELETE SET NULL ON UPDATE CASCADE;
|
||||
|
||||
-- AddForeignKey
|
||||
ALTER TABLE `policy_ocr_documents` ADD CONSTRAINT `policy_ocr_documents_postedTransactionId_fkey` FOREIGN KEY (`postedTransactionId`) REFERENCES `transactions`(`id`) ON DELETE SET NULL ON UPDATE CASCADE;
|
||||
@@ -122,6 +122,7 @@ model Customer {
|
||||
transactions Transaction[]
|
||||
|
||||
statementDocuments StatementDocument[]
|
||||
policyOcrDocuments PolicyOcrDocument[] @relation("PolicyOcrDocumentCustomer")
|
||||
|
||||
@@map("customers")
|
||||
}
|
||||
@@ -210,6 +211,9 @@ model Policy {
|
||||
properties Property[]
|
||||
renewalNotices RenewalNotice[]
|
||||
|
||||
ocrMatchedDocuments PolicyOcrDocument[] @relation("PolicyOcrDocumentPolicy")
|
||||
ocrCreatedDocuments PolicyOcrDocument[] @relation("PolicyOcrDocumentCreatedPolicy")
|
||||
|
||||
@@unique([legacySourceDb, legacySourceTable, legacyId])
|
||||
@@index([policyNumber])
|
||||
@@map("policies")
|
||||
@@ -363,6 +367,115 @@ model PolicyDocument {
|
||||
@@map("policy_documents")
|
||||
}
|
||||
|
||||
/// Insurance OCR intake (mirrors statement_batches / statement_documents for
|
||||
/// the utility side). One upload session of policy PDFs from a provider
|
||||
/// portal (GMX, etc.) — the parser proposes policyNumber → existing Policy
|
||||
/// (or "new, pick customer"), staff confirms, and the system attaches the
|
||||
/// source PDF and optionally writes a premium Transaction.
|
||||
model PolicyOcrBatch {
|
||||
id String @id @default(uuid())
|
||||
/// Which insurance provider portal the batch came from. "GMX" today;
|
||||
/// future providers (AXA, GNP, …) extend the parser, not this table.
|
||||
provider String @default("GMX")
|
||||
status PolicyOcrBatchStatus @default(UPLOADED)
|
||||
uploadedById String
|
||||
uploadedBy User @relation("PolicyOcrBatchUploader", fields: [uploadedById], references: [id])
|
||||
label String?
|
||||
fileCount Int @default(0)
|
||||
/// Set when the pipeline fails as a whole (bad PDF, OCR binaries missing).
|
||||
error String? @db.Text
|
||||
createdAt DateTime @default(now())
|
||||
completedAt DateTime?
|
||||
|
||||
documents PolicyOcrDocument[]
|
||||
|
||||
@@index([status, createdAt])
|
||||
@@map("policy_ocr_batches")
|
||||
}
|
||||
|
||||
enum PolicyOcrBatchStatus {
|
||||
UPLOADED
|
||||
PROCESSING
|
||||
READY_FOR_REVIEW
|
||||
COMPLETED
|
||||
FAILED
|
||||
}
|
||||
|
||||
/// One parsed policy page — one Policy → one Customer (after staff confirms).
|
||||
model PolicyOcrDocument {
|
||||
id String @id @default(uuid())
|
||||
batchId String
|
||||
batch PolicyOcrBatch @relation(fields: [batchId], references: [id], onDelete: Cascade)
|
||||
pageNumber Int
|
||||
/// The rendered page image in object storage. Source PDF kept too on the
|
||||
/// batch (statement pattern) so re-running a corrected parser is possible.
|
||||
storageKey String
|
||||
status PolicyOcrDocumentStatus @default(PENDING_OCR)
|
||||
|
||||
ocrRawText String? @db.Text
|
||||
ocrConfidence Decimal? @db.Decimal(4, 3)
|
||||
/// Which parser claimed the page ("GMX" today).
|
||||
provider String?
|
||||
|
||||
// Extracted header fields, all staff-editable in review.
|
||||
extractedPolicyNumber String?
|
||||
extractedInsuredName String?
|
||||
extractedAdditionalInsured String?
|
||||
extractedAgentName String?
|
||||
extractedLegalAddress String? @db.Text
|
||||
extractedZip String?
|
||||
extractedPolicyFrom DateTime?
|
||||
extractedPolicyTo DateTime?
|
||||
extractedPolicyDate DateTime?
|
||||
extractedCurrency String?
|
||||
extractedNetPremium Decimal? @db.Decimal(12, 2)
|
||||
extractedPolicyFee Decimal? @db.Decimal(12, 2)
|
||||
extractedBrokerFee Decimal? @db.Decimal(12, 2)
|
||||
extractedTotal Decimal? @db.Decimal(12, 2)
|
||||
/// Per-coverage rows from the GMX "Material damages" / "Additional risk"
|
||||
/// tables — preserved verbatim so a missing premium receipt still leaves
|
||||
/// the coverages auditable.
|
||||
extractedCoveragesJson Json?
|
||||
extractedPremiumPayment String?
|
||||
|
||||
// Match by `Policy.policyNumber` → existing Policy / Customer.
|
||||
matchedPolicyId String?
|
||||
matchedPolicy Policy? @relation("PolicyOcrDocumentPolicy", fields: [matchedPolicyId], references: [id])
|
||||
matchedCustomerId String?
|
||||
matchedCustomer Customer? @relation("PolicyOcrDocumentCustomer", fields: [matchedCustomerId], references: [id])
|
||||
/// All policies carrying the same number, with their customer. One is
|
||||
/// normal; >1 means the policy number is shared across customers and a
|
||||
/// human must pick.
|
||||
matchCandidates Json?
|
||||
matchNote String?
|
||||
|
||||
reviewedById String?
|
||||
reviewedBy User? @relation("PolicyOcrDocumentReviewer", fields: [reviewedById], references: [id])
|
||||
reviewedAt DateTime?
|
||||
|
||||
createdPolicyId String? @unique
|
||||
createdPolicy Policy? @relation("PolicyOcrDocumentCreatedPolicy", fields: [createdPolicyId], references: [id])
|
||||
postedTransactionId String? @unique
|
||||
postedTransaction Transaction? @relation("PolicyOcrDocumentTransaction", fields: [postedTransactionId], references: [id])
|
||||
|
||||
createdAt DateTime @default(now())
|
||||
|
||||
@@unique([batchId, pageNumber])
|
||||
@@index([status])
|
||||
@@index([matchedCustomerId])
|
||||
@@map("policy_ocr_documents")
|
||||
}
|
||||
|
||||
enum PolicyOcrDocumentStatus {
|
||||
PENDING_OCR
|
||||
OCR_FAILED
|
||||
NEEDS_REVIEW
|
||||
MATCHED
|
||||
CONFIRMED
|
||||
POSTED
|
||||
REJECTED
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Utilities domain
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -399,9 +512,8 @@ model Property {
|
||||
documents ServiceDocument[]
|
||||
trustAccount TrustAccount?
|
||||
|
||||
@@index([cadastralKey])
|
||||
|
||||
@@unique([legacySourceTable, legacyId])
|
||||
@@index([cadastralKey])
|
||||
@@map("properties")
|
||||
}
|
||||
|
||||
@@ -624,14 +736,15 @@ model Transaction {
|
||||
|
||||
/// Set only on OCR-posted rows — the statement page this came from.
|
||||
statementDocument StatementDocument?
|
||||
policyOcrDocument PolicyOcrDocument? @relation("PolicyOcrDocumentTransaction")
|
||||
|
||||
@@unique([legacySourceDb, legacySourceTable, legacyId])
|
||||
@@index([customerId, transactionDate])
|
||||
// By-check reconciliation (billing.byCheck / the cheque-count report) looks
|
||||
// rows up by check number alone — the legacy EDITA CHEQUE COUNT lookup.
|
||||
@@index([checkNumber])
|
||||
// Drives the duplicate-post guard in BillingService.createBatch.
|
||||
@@index([captureRef])
|
||||
@@unique([legacySourceDb, legacySourceTable, legacyId])
|
||||
@@map("transactions")
|
||||
}
|
||||
|
||||
@@ -745,6 +858,9 @@ model User {
|
||||
statementBatches StatementBatch[] @relation("StatementBatchUploader")
|
||||
statementsReviewed StatementDocument[] @relation("StatementDocumentReviewer")
|
||||
|
||||
policyOcrBatches PolicyOcrBatch[] @relation("PolicyOcrBatchUploader")
|
||||
policyOcrReviewed PolicyOcrDocument[] @relation("PolicyOcrDocumentReviewer")
|
||||
|
||||
@@map("users")
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user