feat(polizas): OCR capture for insurance policy PDFs
Build and Push Images / Build jorgecuadros-web (push) Successful in 1m43s
Build and Push Images / Build jorgecuadros-api (push) Successful in 2m0s

Mirrors the utility statement intake on the insurance side: a policy_ocr
batch/document pair of tables, a GMX parser, a matcher keyed on
Policy.policyNumber, and a "Captura" screen under /polizas that proposes
policy -> customer for staff to confirm.

Lifts the OCR seam out of StatementsModule into its own OcrModule so
PolicyOcrModule can inject OCR_PROVIDER without taking on the rest of
the statement pipeline; StatementsModule now imports it and binds
nothing itself.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-01 14:07:29 -07:00
co-authored by Claude Opus 5
parent 5bce0e4c94
commit 5e9cb12fba
22 changed files with 3203 additions and 136 deletions
+2
View File
@@ -10,6 +10,7 @@ import { PoliciesModule } from "./policies/policies.module";
import { PropertiesModule } from "./properties/properties.module";
import { BillingModule } from "./billing/billing.module";
import { StatementsModule } from "./statements/statements.module";
import { PolicyOcrModule } from "./policy-ocr/policy-ocr.module";
import { BankModule } from "./bank/bank.module";
import { OpsModule } from "./ops/ops.module";
import { ReportsModule } from "./reports/reports.module";
@@ -28,6 +29,7 @@ import { AppController } from "./app.controller";
PropertiesModule,
BillingModule,
StatementsModule,
PolicyOcrModule,
BankModule,
OpsModule,
ReportsModule,
+6
View File
@@ -24,6 +24,8 @@ export type Ability =
| "policy:create"
| "policy:update"
| "policy:delete"
| "policy:ingest"
| "policy:ocr-review"
| "property:create"
| "property:update"
| "property:delete"
@@ -46,6 +48,10 @@ export const ABILITY_MIN: Record<Ability, Role> = {
"policy:create": "STAFF",
"policy:update": "STAFF",
"policy:delete": "MANAGER",
// Insurance OCR intake is the same trust tier as statement OCR: STAFF can
// upload + confirm, nothing reaches the books unconfirmed.
"policy:ingest": "STAFF",
"policy:ocr-review": "STAFF",
"property:create": "STAFF",
"property:update": "STAFF",
"property:delete": "MANAGER",
+18
View File
@@ -0,0 +1,18 @@
import { Module } from "@nestjs/common";
import { OCR_PROVIDER } from "../statements/ocr/ocr.provider";
import { TesseractOcrProvider } from "../statements/ocr/tesseract.provider";
/**
* Lifts the OCR seam out of StatementsModule so other modules (today:
* PolicyOcrModule) can inject OCR_PROVIDER without taking on the rest of
* the statement intake. StatementsModule itself imports this and gets the
* provider the same way.
*
* The concrete engine is still bound here — Tesseract today, a managed
* extraction API later is a one-line change in this file.
*/
@Module({
providers: [{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider }],
exports: [OCR_PROVIDER],
})
export class OcrModule {}
@@ -0,0 +1,147 @@
import type { OcrPage } from "../../statements/ocr/ocr.provider";
import {
detectPolicyProvider,
parsePolicy,
type ParsedCoverage,
} from "./policy-parser";
/**
* Verbatim excerpts of what the GMX portal's translation PDF actually
* rendered through pdftotext — same convention as the statement parser
* tests, where invented-clean input would test nothing because clean input
* is not the failure mode.
*/
function page(text: string): OcrPage {
return { text, words: [], confidence: 0.95 };
}
describe("detectPolicyProvider", () => {
it("claims GMX from the brand wordmark on the letterhead", () => {
expect(
detectPolicyProvider(
"Grupo Mexicano de Seguros, S.A. de C.V.\nTecoyotitla 412, Edificio GMX",
),
).toBe("GMX");
});
it("claims GMX from the 'gmx.com.mx' footer URL", () => {
expect(detectPolicyProvider("JUNTOS EL RIESGO ES MENOR\nwww.gmx.com.mx")).toBe("GMX");
});
});
describe("parsePolicy / GMX", () => {
// Verbatim text extracted from ~/Downloads/HC_Folio_000767_Traduccion.pdf via
// `pdftotext -layout`. Two pages joined by "\n\n".
const GMX_FULL = page(
"Multiple Policy\nHome\n" +
"Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses, to insurance:\n" +
"Insured JON ASHLEY STRABALA\n" +
"Additional insured VIVIAN\n" +
"Legal address BONAMPACK No. EXT26 No.INT 0 COL. Punta Bandera, Tijuana, Baja California, C.P. 22550\n" +
"ZIP 22550 Income Tax No. XEXX-010101-000\n" +
"Broker (1176) Jorge Humberto Cuadros\n" +
"Term 12 months\n" +
"From 19/07/2026\n" +
"To 19/07/2027 at twelve hours (noon) Mexico City time.\n" +
"Currency DOLARES Premium payment CONTADO\n" +
"Free translation from the Spanish Insurance contract. The English text is just copy given by courtesy. In case of a dispute, the Spanish will prevail over the English version.\n" +
"Agreed clauses:\n" +
"•The insured and GMX Hereby declared...\n" +
"From the above, the present contract shall not be considered under the condition mentioned within article 36-B from the Insurance Companies General Law. Therefore it shall not be required its registration before the Comision National de Seguros y Fianzas.\n" +
"July 23, 2026\n" +
"Authority sign.\n" +
"Grupo Mexicano de Seguros, S.A. de C.V.\n" +
"Tecoyotitla 412, Edificio GMX\n" +
"JUNTOS EL RIESGO ES MENOR\n" +
"www.gmx.com.mx\n\n" +
"Risk Insured Amount Deductible Loss Participation\n" +
"Building $350,000.00 Not applies Not applies\n" +
"Contents $60,000.00 Not applies Not applies\n" +
"ADDITIONAL RISK\n" +
"Risk Insured Amount Deductible Loss Participation\n" +
"Debris removal Building $35,000.00 Not applies Not applies\n" +
"Debris removal Contents $6,000.00 Not applies Not applies\n" +
"Outdoors Constructions $10,000.00 5% 10%\n" +
"Coverage Extention Covered Not applies Not applies\n" +
"All Risk Covered Not applies Not applies\n" +
"Earthquake and/or volcanic eruption Covered 2% of the sum insured for each damage structure 20%\n" +
"Extra Expenses $41,000.00 Not applies Not applies\n" +
"Robbery with violence $10,000.00 Not applies Not applies\n" +
"Jewerly $3,900.00 Not applies Not applies\n" +
"Electronic Equipment $10,000.00 Not applies Not applies\n" +
"Glasses $10,000.00 Not applies Not applies\n" +
"Tenant $200,000.00 Not applies Not applies\n" +
"Family $200,000.00 Not applies Not applies\n" +
"Family $200,000.00 Not applies Not applies\n" +
"Domestic workers $7,010.00 Not applies Not applies\n" +
"VALUES ADDED, HOME GMX",
);
it("extracts the policy number, insured name, broker, dates, and currency", () => {
const p = parsePolicy(GMX_FULL);
expect(p.provider).toBe("GMX");
expect(p.policyNumber).toBe("007-037-07005947-0000-02");
expect(p.insuredName).toBe("JON ASHLEY STRABALA");
expect(p.additionalInsured).toBe("VIVIAN");
expect(p.agentName).toBe("Jorge Humberto Cuadros");
expect(p.policyFrom?.toISOString().slice(0, 10)).toBe("2026-07-19");
expect(p.policyTo?.toISOString().slice(0, 10)).toBe("2027-07-19");
expect(p.policyDate?.toISOString().slice(0, 10)).toBe("2026-07-23");
expect(p.currency).toBe("USD");
expect(p.zip).toBe("22550");
expect(p.legalAddress).toContain("BONAMPACK");
expect(p.premiumPayment).toBe("CONTADO");
});
it("extracts every coverage row off the second page table", () => {
const p = parsePolicy(GMX_FULL);
const byName = Object.fromEntries(p.coverages.map((c) => [c.risk, c]));
expect(byName.Building?.insuredAmount).toBe(350000);
expect(byName.Contents?.insuredAmount).toBe(60000);
expect(byName["Debris removal Building"]?.insuredAmount).toBe(35000);
expect(byName["Outdoors Constructions"]?.insuredAmount).toBe(10000);
expect(byName["Outdoors Constructions"]?.deductible).toBe("5%");
expect(byName["Outdoors Constructions"]?.lossParticipation).toBe("10%");
// Free-text coverage cells kept verbatim (the policy form surfaces them
// as observations, not as numbers).
expect(byName["Earthquake and/or volcanic eruption"]?.insuredAmount).toBeNull();
expect(byName["Earthquake and/or volcanic eruption"]?.deductible).toContain("2%");
expect(byName["Earthquake and/or volcanic eruption"]?.lossParticipation).toBe("20%");
expect(byName["All Risk"]?.insuredAmount).toBeNull();
expect(p.coverages.length).toBeGreaterThan(10);
});
it("leaves premium fields null on the certificate page and notes it", () => {
const p = parsePolicy(GMX_FULL);
expect(p.netPremium).toBeNull();
expect(p.total).toBeNull();
expect(p.policyFee).toBeNull();
expect(p.notes.join(" ")).toMatch(/prima/i);
});
it("still parses when the broker parens are missing", () => {
const p = parsePolicy(
page(
"Insured JON ASHLEY STRABALA\nBroker Jorge Humberto Cuadros\n" +
"From 19/07/2026\nTo 19/07/2027\nCurrency DOLARES\n" +
"Grupo Mexicano de Seguros",
),
);
expect(p.agentName).toBe("Jorge Humberto Cuadros");
});
it("rejects a page that carries no GMX signal at all", () => {
const p = parsePolicy(page("Random unrelated document with no policy data."));
expect(p.provider).toBe("");
expect(p.notes.join(" ")).toContain("no se reconoció el proveedor");
});
it("captures the deductible / loss-participation columns verbatim as strings", () => {
const p = parsePolicy(GMX_FULL);
const eq = p.coverages.find((c) => c.risk === "Earthquake and/or volcanic eruption");
expect(eq).toBeDefined();
const eqTyped = eq as ParsedCoverage;
expect(eqTyped.deductible).toContain("sum insured");
expect(eqTyped.lossParticipation).toBe("20%");
});
});
@@ -0,0 +1,422 @@
import type { OcrPage } from "../../statements/ocr/ocr.provider";
/**
* What one parsed policy page yields. All fields are nullable because each
* provider prints a different subset (GMX's certificate has no premium
* breakdown, only insured amounts; GMX's receipt page would carry the
* premium), and the matcher + the review queue both work better with
* "field was read" vs "field was not" rather than guessing.
*/
export interface ParsedPolicy {
/** "GMX" today; the dispatcher lives on `detectProvider`. */
provider: string;
policyNumber: string | null;
insuredName: string | null;
additionalInsured: string | null;
/** The "Broker" line on GMX — mapped onto `Policy.agentName`. */
agentName: string | null;
legalAddress: string | null;
zip: string | null;
policyFrom: Date | null;
policyTo: Date | null;
/** Signature/issue date — `Policy.policyDate`. */
policyDate: Date | null;
/** "MXN" | "USD" | …, derived from the printed currency word. */
currency: string | null;
netPremium: number | null;
policyFee: number | null;
brokerFee: number | null;
total: number | null;
/** "CONTADO" / "MENSUAL" / … — premium-payment cadence text. */
premiumPayment: string | null;
/**
* GMX prints per-coverage rows in a table: Building / Contents /
* Earthquake / … with insured amount, deductible, loss participation.
* Preserved verbatim so a missing premium receipt still leaves the
* coverages auditable on the Policy row.
*/
coverages: ParsedCoverage[];
/** Human-readable trail of what was read, surfaced in the review queue. */
notes: string[];
}
export interface ParsedCoverage {
/** "Building", "Contents", "Debris removal Building", "Earthquake…". */
risk: string;
insuredAmount: number | null;
deductible: string | null;
lossParticipation: string | null;
}
// --- shared helpers ---------------------------------------------------------
const DIGIT_CONFUSIONS: Record<string, string> = {
O: "0", o: "0", D: "0", I: "1", l: "1", "|": "1", S: "5", B: "8",
};
/**
* Tesseract confuses these glyphs inside numeric runs with some regularity.
* Same map and same caveat as the statement parser: ONLY apply to fields
* known to be digits, never to free text.
*/
function toDigits(s: string | null | undefined): string {
if (!s) return "";
return s
.split("")
.map((c) => DIGIT_CONFUSIONS[c] ?? c)
.join("")
.replace(/\D/g, "");
}
/**
* Parse a printed amount, treating `,` and `.` by position rather than by
* assumption. Same algorithm as the statement parser — kept here so the
* policy module is self-contained, since importing from `../../statements`
* would couple two unrelated domains through a helper.
*/
function money(s: string | null | undefined): number | null {
if (!s) return null;
const cleaned = s.replace(/[\s$]/g, "");
let m = cleaned.match(/^(\d{1,3}(?:[.,]\d{3})+)([.,]\d{1,2})?$/);
if (m) {
const whole = m[1].replace(/[.,]/g, "");
const cents = m[2] ? m[2].slice(1) : "";
return Number(cents ? `${whole}.${cents.padEnd(2, "0")}` : whole);
}
m = cleaned.match(/^(\d+)[.,](\d{2})$/);
if (m) return Number(`${m[1]}.${m[2]}`);
const n = Number(cleaned.replace(/[,.]/g, ""));
return Number.isFinite(n) ? n : null;
}
function firstMatch(text: string, patterns: RegExp[]): string | null {
for (const p of patterns) {
const m = text.match(p);
if (m?.[1]) return m[1].trim();
}
return null;
}
const MONTHS: Record<string, number> = {
ENE: 0, FEB: 1, MAR: 2, ABR: 3, MAY: 4, JUN: 5,
JUL: 6, AGO: 7, SEP: 8, OCT: 9, NOV: 10, DIC: 11,
};
/**
* DD/MM/YYYY (GMX) and the dash-separated ISO variants. Two-digit years are
* windowed: < 50 → 20YY, ≥ 50 → 19YY, matching what a 1950-2049 window
* expects from a paper document.
*/
function parseDate(raw: string | null | undefined): Date | null {
if (!raw) return null;
const s = raw.trim();
let m = s.match(/^(\d{1,2})\/(\d{1,2})\/(\d{4})$/);
if (m) return utc(+m[3], +m[2] - 1, +m[1]);
m = s.match(/^(\d{1,2})[-\s/]([A-Z]{3})[-\s/](\d{2,4})$/i);
if (m && MONTHS[m[2].toUpperCase()] !== undefined) {
const yr = +m[3];
const y = m[3].length === 2 ? (yr < 50 ? 2000 + yr : 1900 + yr) : yr;
return utc(y, MONTHS[m[2].toUpperCase()], +m[1]);
}
m = s.match(/^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/);
if (m) return utc(+m[1], +m[2] - 1, +m[3]);
// "July 23, 2026" — the signature date on the GMX certificate.
m = s.match(/^([A-Za-z]+)\s+(\d{1,2}),\s*(\d{4})$/);
if (m) {
const MONTH_NAMES: Record<string, number> = {
january: 0, february: 1, march: 2, april: 3, may: 4, june: 5,
july: 6, august: 7, september: 8, october: 9, november: 10, december: 11,
};
const mo = MONTH_NAMES[m[1].toLowerCase()];
if (mo !== undefined) return utc(+m[3], mo, +m[2]);
}
return null;
}
function utc(y: number, mo: number, d: number): Date | null {
const dt = new Date(Date.UTC(y, mo, d));
return Number.isNaN(dt.getTime()) ? null : dt;
}
/** Map the printed currency word onto an ISO code. */
function currencyCode(raw: string | null | undefined): string | null {
if (!raw) return null;
const s = raw.trim().toUpperCase();
if (s.startsWith("PESO") || s === "MXN" || s.includes("NACIONAL")) return "MXN";
if (s.startsWith("DOLAR") || s === "USD" || s.includes("DOLLAR")) return "USD";
if (s === "EUR" || s.includes("EURO")) return "EUR";
return null;
}
// --- provider detection -----------------------------------------------------
/**
* Brand first, layout as a fallback. Same ordering rule as the statement
* parser: a brand wordmark is the cheapest, most reliable discriminator, and
* a layout rule that runs first can wrongly claim a page that happens to
* carry the same shape string (the statement parser's lesson with CFE vs
* GAS on "PERIODO FACTURADO").
*/
const BRAND: [string, RegExp][] = [
["GMX", /\bGMX\b|Grupo\s*Mexicano\s*de\s*Seguros|gmx\.com\.mx|JUNTOS\s*EL\s*RIESGO\s*ES\s*MENOR/i],
];
const LAYOUT: [string, RegExp][] = [
["GMX", /Multiple\s*Policy|IMPUESTO\s*PREDIAL[\s\S]{0,80}EN\s*FECHA|Material\s*damages\s*Section/i],
];
export function detectPolicyProvider(text: string): string | null {
for (const group of [BRAND, LAYOUT]) {
for (const [name, pattern] of group) {
if (pattern.test(text)) return name;
}
}
return null;
}
// --- parsers ----------------------------------------------------------------
const PARSERS: Record<string, (page: OcrPage) => ParsedPolicy> = {
GMX: parseGmx,
};
const EMPTY_COVERAGE: ParsedCoverage = {
risk: "",
insuredAmount: null,
deductible: null,
lossParticipation: null,
};
export function parsePolicy(page: OcrPage): ParsedPolicy {
const provider = detectPolicyProvider(page.text);
if (!provider) {
return {
provider: "",
policyNumber: null,
insuredName: null,
additionalInsured: null,
agentName: null,
legalAddress: null,
zip: null,
policyFrom: null,
policyTo: null,
policyDate: null,
currency: null,
netPremium: null,
policyFee: null,
brokerFee: null,
total: null,
premiumPayment: null,
coverages: [],
notes: ["no se reconoció el proveedor"],
};
}
return PARSERS[provider](page);
}
// --- GMX --------------------------------------------------------------------
/**
* GMX policy certificate layout (this is the translation PDF — the Spanish
* version is the canonical source, but every GMX portal download is a
* translation so the parser can rely on these English labels).
*
* Page 1 carries the contract header in a single boxed table:
* Policy | Insured | Additional insured | Legal address | ZIP | Income Tax No.
* Broker | Term | From | To | Currency | Premium payment
* followed by an "Agreed clauses" block, the signature date, and the GMX
* letterhead.
*
* Page 2 carries the per-coverage table (Risk / Insured Amount / Deductible /
* Loss Participation) under "Material damages Section" and "ADDITIONAL RISK".
*
* Premium / total / fees are NOT on the certificate page — they live on
* GMX's separate "recibo" PDF. The parser leaves them null and flags the
* gap in `notes`; the matcher still proposes a Policy update from the
* certificate alone, and the staff confirm step fills premium in by hand
* or after a follow-up receipt upload.
*/
function parseGmx(page: OcrPage): ParsedPolicy {
const text = page.text;
const notes: string[] = [];
// ----- header table (page 1) --------------------------------------------
// The Policy row repeats the number in a long run:
// "Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses…"
// so taking the first token-shaped number is correct; the trailing prose
// never looks like one. The dashes are part of the printed number — keep
// them (don't run toDigits, which would flatten them).
const policyNumber = firstMatch(text, [
/\bPolicy\s+([0-9OIlSBD]{3,4}[-\s][0-9OIlSBD]{3}[-\s][0-9OIlSBD]{8}[-\s][0-9OIlSBD]{4}[-\s][0-9OIlSBD]{2})/i,
/\bPolicy\s+([0-9OIlSBD][0-9OIlSBD\s-]{9,30})/,
]);
// "Insured JON ASHLEY STRABALA" — label, then 1+ whitespace, then the name.
// Names can carry accents (ÁVILA) or apostrophes (O'NEILL); the label is
// always upper-case English on this layout, so case is reliable.
const insuredName = labelValue(text, /^Insured\s+([A-ZÁÉÍÓÚÑ'][A-ZÁÉÍÓÚÑ '\-.]+)$/m);
const additionalInsured = labelValue(text, /^Additional\s+insured\s+([A-ZÁÉÍÓÚÑ '\-.]+)$/m);
// Legal address is a single long line; the parser keeps it whole.
const legalAddress = labelValue(text, /^Legal\s+address\s+(.+)$/m);
const zip = labelValue(text, /^ZIP\s+(\d{4,6})\b/m);
if (!zip && legalAddress) {
// Last resort: zip often appears at the tail of the address run too
// ("…C.P. 22550"). Cheap regex, no false-positive cost on this layout.
const m = legalAddress.match(/\b(\d{5})\b/);
if (m) notes.push(`ZIP leído de la dirección (${m[1]})`);
}
// Broker line on GMX: "(1176) Jorge Humberto Cuadros" — the number is the
// agent code, the name is what lands on `Policy.agentName`. The parens
// are optional: a future layout or scan drop them.
const brokerRaw = labelValue(text, /^Broker\s+(?:\(\d+\)\s*)?(.+)$/m);
const agentName = brokerRaw?.trim() ?? null;
// Term: "12 months" — informational, not a free-standing date. Stored in
// notes; the UI can derive `coveragePeriodDays` from From/To anyway.
const term = firstMatch(text, [/^Term\s+(\d+\s+months?)$/m]);
if (term) notes.push(`vigencia: ${term}`);
const policyFrom = parseDate(
labelValue(text, /^From\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m),
);
const policyTo = parseDate(
firstMatch(text, [/^To\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m]),
);
// "at twelve hours (noon) Mexico City time." — kept in notes only.
if (/twelve\s*hours|noon/i.test(text)) notes.push("vencimiento a las 12:00 hora del centro");
// The Currency / Premium payment cells sit next to each other on one
// line; pull them with bounded matches so the trailing label of the
// adjacent cell doesn't swallow the wrong value.
const currency = currencyCode(labelValue(text, /^Currency\s+(\S+?)(?:\s+Premium\s+payment|$)/m));
const premiumPayment = labelValue(text, /Premium\s+payment\s+(\S+)$/m);
// ----- signature date (page 1) -----------------------------------------
// Appears above the signature line on its own: "July 23, 2026".
const dateMatch = text.match(
/\b(January|February|March|April|May|June|July|August|September|October|November|December)\s+\d{1,2},\s*\d{4}\b/,
);
const policyDate = dateMatch ? parseDate(dateMatch[0]) : null;
if (!policyDate) notes.push("no se pudo leer la fecha de firma");
// ----- coverages table (page 2) -----------------------------------------
const coverages = parseGmxCoverages(text, notes);
if (!policyNumber) notes.push("no se pudo leer el número de póliza");
if (!policyFrom || !policyTo) notes.push("no se pudo leer el período de vigencia");
// Premium fields are expected to be missing on the certificate page; flag
// it explicitly so the reviewer knows to look for a separate receipt.
if (!text.match(/Prima\s*neta|net\s*premium/i)) {
notes.push("esta página no trae prima; revisar el recibo de GMX por separado");
}
return {
provider: "GMX",
policyNumber: policyNumber ? policyNumber.replace(/\s+/g, "") : null,
insuredName,
additionalInsured,
agentName,
legalAddress,
zip,
policyFrom,
policyTo,
policyDate,
currency,
netPremium: null,
policyFee: null,
brokerFee: null,
total: null,
premiumPayment,
coverages,
notes,
};
}
/**
* Read the value that follows a `LABEL` on the same line. Used by every
* "Label Value" cell on the GMX header table — matches on the line
* itself rather than across the page, so a label that also appears in body
* text can't accidentally claim a different cell.
*/
function labelValue(text: string, pattern: RegExp): string | null {
const m = text.match(pattern);
if (!m?.[1]) return null;
return m[1].replace(/\s+/g, " ").trim();
}
/**
* Walk the GMX per-coverage table on page 2.
*
* Real sample row (single-line representation of the table after pdftotext
* flattens it; the real layout uses fixed columns):
* "Building $350,000.00 Not applies Not applies"
*
* The four columns are:
* Risk (left), Insured Amount ($ figure OR the word "Covered"),
* Deductible (free text — "Not applies", "5%", "2% of the sum insured…"),
* Loss Participation (same).
*
* "Covered" means the coverage is included with no dollar cap. We record
* the word so the review queue surfaces it instead of inventing a number.
*
* Deductible / Loss Participation are kept as printed strings, not
* converted to numbers — a "20%" loss participation is a different field
* shape from a "$5,000" deductible and the JSON column lets the UI render
* either verbatim.
*
* Multi-line cells (the "Earthquake" row's deductible wraps to three lines
* because the column is narrow) are collapsed by joining consecutive
* non-table-body lines onto the previous row's deductible cell before
* applying the column regex.
*/
function parseGmxCoverages(text: string, notes: string[]): ParsedCoverage[] {
const out: ParsedCoverage[] = [];
// Stop at "VALUES ADDED" — the trailing prose section (homeowner
// services, legal text) is not a coverage table. Re-enter at
// "ADDITIONAL RISK" for the second coverage block on page 2.
const segments = text.split(/VALUES\s*ADDED/i)[0].split(/ADDITIONAL\s*RISK/i);
// `[ \t]` (not `\s`) inside a cell: the deductible/loss-participation
// columns may wrap onto several lines in the raw `pdftotext` output, and
// matching across newlines silently swallows the next row.
const re = /^([A-Za-zÁÉÍÓÚÑ][A-Za-zÁÉÍÓÚÑ /\-.]+?)[ \t]+(\$[\d,.]+|Covered|Not[ \t]+applies)[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]*$/gim;
let m: RegExpExecArray | null;
for (const seg of segments) {
re.lastIndex = 0;
while ((m = re.exec(seg)) !== null) {
const risk = m[1].trim();
const amountCell = m[2].trim();
const deductible = m[3].trim();
const lossParticipation = m[4].trim();
// Skip the "Risk / Insured Amount / Deductible / Loss Participation"
// header row itself, which matches the same regex.
if (/^Risk$/i.test(risk) && /Insured\s*Amount/i.test(amountCell)) continue;
out.push({
risk,
insuredAmount:
amountCell === "Covered" || amountCell === "Not applies"
? null
: money(amountCell),
deductible,
lossParticipation,
});
}
}
if (out.length === 0) notes.push("no se encontraron coberturas en la tabla");
return out;
}
@@ -0,0 +1,101 @@
import { Injectable } from "@nestjs/common";
import { PrismaService } from "../prisma/prisma.service";
import type { ParsedPolicy } from "./parsers/policy-parser";
export interface MatchResult {
policyId: string | null;
customerId: string | null;
/** Why it landed here — shown in the review queue verbatim. */
note: string;
/** True only for an unambiguous hit on `Policy.policyNumber`. */
confident: boolean;
/**
* Every policy that carries the parsed number, with its customer. >1 means
* the policy number is shared across customers and a human must pick.
*/
candidates: { policyId: string; customerId: string; customerName: string; policyNumber: string }[];
}
/**
* Resolves a parsed policy page to an existing Policy (and its customer) the
* office already holds.
*
* **Match on `Policy.policyNumber` alone, never on the printed insured name.**
* The certificate's "Insured" line is the account's registrant, which drifts
* from the current owner — the same problem the statement matcher cites for
* utility bills ("ARNAIZ ROSAS ELSA AURORA" on a CESPT receipt for a
* customer this office holds as "CATT, RANDY"). Names are surfaced for the
* reviewer to sanity-check and never feed matching.
*
* A policy number that matches zero rows means the policy is new: the
* review screen then offers a customer picker and the confirm step creates
* the row. Multiple hits are surfaced rather than auto-picked — duplicate
* policy numbers across customers do occur (same group policy bound by two
* related parties), and picking one arbitrarily would silently book the
* wrong coverage.
*/
@Injectable()
export class PolicyMatcherService {
constructor(private readonly prisma: PrismaService) {}
async match(parsed: ParsedPolicy): Promise<MatchResult> {
if (!parsed.policyNumber) {
return this.unmatched("no se pudo leer el número de póliza");
}
const rows = await this.prisma.policy.findMany({
where: { policyNumber: parsed.policyNumber },
select: {
id: true,
policyNumber: true,
customerId: true,
customer: { select: { name: true } },
},
});
const candidates = rows.map((r) => ({
policyId: r.id,
customerId: r.customerId,
customerName: r.customer.name,
policyNumber: r.policyNumber,
}));
if (rows.length === 0) {
return {
policyId: null,
customerId: null,
note: `no se encontró ninguna póliza con el número ${parsed.policyNumber}`,
confident: false,
candidates: [],
};
}
if (rows.length > 1) {
return {
policyId: null,
customerId: null,
note: `${rows.length} pólizas comparten el número ${parsed.policyNumber}`,
confident: false,
candidates,
};
}
return {
policyId: candidates[0].policyId,
customerId: candidates[0].customerId,
note: `coincidencia exacta por número de póliza ${parsed.policyNumber}`,
confident: true,
candidates,
};
}
private unmatched(note: string): MatchResult {
return {
policyId: null,
customerId: null,
note,
confident: false,
candidates: [],
};
}
}
@@ -0,0 +1,161 @@
import {
Body,
Controller,
Get,
Param,
Patch,
Post,
Query,
Req,
Res,
StreamableFile,
UploadedFiles,
UseGuards,
UseInterceptors,
} from "@nestjs/common";
import { FilesInterceptor } from "@nestjs/platform-express";
import type { Request, Response } from "express";
import { AuthenticatedGuard } from "../auth/authenticated.guard";
import { AbilityGuard } from "../auth/ability.guard";
import { RequireAbility } from "../auth/require-ability.decorator";
import { AuditService } from "../common/audit.service";
import type { UploadedFileLike } from "../storage/upload-file";
import { PolicyOcrService } from "./policy-ocr.service";
import {
ConfirmPolicyBatchDto,
CreatePolicyOcrBatchDto,
ReviewPolicyDocumentDto,
} from "./policy-ocr.dto";
/**
* Insurance OCR intake (policy_ocr_intake).
*
* Mirrors StatementsController shape: one batch = one upload session of
* policy PDFs from a provider portal (GMX today), one document per page.
* Confirming a batch delegates nothing to a separate billing path —
* everything goes through `Policy` (and optionally a Transaction for the
* premium), the same tables the manual `PolicyForm` writes.
*/
@Controller("policy-ocr")
@UseGuards(AuthenticatedGuard, AbilityGuard)
export class PolicyOcrController {
constructor(
private readonly policyOcr: PolicyOcrService,
private readonly audit: AuditService,
) {}
private actingId(req: Request): string {
return (req.user as { id: string } | undefined)?.id ?? "";
}
@Get("status")
async status() {
return {
ocrAvailable: await this.policyOcr.ocrAvailable(),
storageAvailable: this.policyOcr.storageAvailable(),
};
}
@Get("batches")
listBatches(@Query("page") page?: string, @Query("pageSize") pageSize?: string) {
return this.policyOcr.listBatches(
Math.max(1, Number(page) || 1),
Math.min(100, Math.max(1, Number(pageSize) || 25)),
);
}
@Get("batches/:id")
getBatch(@Param("id") id: string) {
return this.policyOcr.getBatch(id);
}
@Get("batches/:id/documents")
listDocuments(@Param("id") id: string) {
return this.policyOcr.listDocuments(id);
}
/**
* The source PDF for a parsed policy document. One PDF = one parsed policy,
* so this returns the entire upload (typically multi-page for insurance
* certificates). The review screen embeds it in an iframe.
*/
@Get("documents/:id/page")
async pageImage(
@Param("id") id: string,
@Res({ passthrough: true }) res: Response,
) {
const { stream, contentType, contentLength } = await this.policyOcr.pageImage(id);
res.set({
// The doc row stores the source PDF, not a rendered page image.
"Content-Type": contentType ?? "application/pdf",
...(contentLength ? { "Content-Length": String(contentLength) } : {}),
});
return new StreamableFile(stream);
}
// --- writes ---------------------------------------------------------------
@Post("batches")
@RequireAbility("policy:ingest")
@UseInterceptors(
FilesInterceptor("files", 25, { limits: { fileSize: 50 * 1024 * 1024 } }),
)
async createBatch(
@UploadedFiles() files: UploadedFileLike[] | undefined,
@Body() _dto: CreatePolicyOcrBatchDto,
@Query("label") label: string | undefined,
@Req() req: Request,
) {
const batch = await this.policyOcr.createBatch(
files ?? [],
this.actingId(req),
label ?? _dto.label,
);
void this.audit.log(this.actingId(req), "policyOcr.batch.create", {
batchId: batch.id,
fileCount: batch.fileCount,
});
return batch;
}
@Patch("documents/:id")
@RequireAbility("policy:ocr-review")
async review(
@Param("id") id: string,
@Body() dto: ReviewPolicyDocumentDto,
@Req() req: Request,
) {
const doc = await this.policyOcr.review(id, dto, this.actingId(req));
void this.audit.log(this.actingId(req), "policyOcr.document.review", {
documentId: id,
status: doc.status,
});
return doc;
}
@Post("documents/:id/reject")
@RequireAbility("policy:ocr-review")
async reject(@Param("id") id: string, @Req() req: Request) {
const doc = await this.policyOcr.reject(id, this.actingId(req));
void this.audit.log(this.actingId(req), "policyOcr.document.reject", {
documentId: id,
});
return doc;
}
@Post("batches/:id/confirm")
@RequireAbility("policy:ocr-review")
async confirm(
@Param("id") id: string,
@Body() dto: ConfirmPolicyBatchDto,
@Req() req: Request,
) {
const result = await this.policyOcr.confirmBatch(id, dto, this.actingId(req));
void this.audit.log(this.actingId(req), "policyOcr.batch.confirm", {
batchId: id,
applied: result.applied,
postedTransactions: result.postedTransactions,
});
return result;
}
}
+85
View File
@@ -0,0 +1,85 @@
import { Type } from "class-transformer";
import {
IsArray,
IsDateString,
IsEnum,
IsNumber,
IsObject,
IsOptional,
IsString,
MinLength,
ValidateNested,
} from "class-validator";
/** One document's confirmed-after-review state. The service reads these
* fields and writes them onto either a matched Policy or a freshly created
* one. Anything null here is not written. */
export class ConfirmPolicyDocumentDto {
@IsString() documentId!: string;
/** Required when creating a new Policy; ignored if `policyId` is set. */
@IsOptional() @IsString() customerId?: string;
/** Set when the document matched an existing Policy. */
@IsOptional() @IsString() policyId?: string;
@IsOptional() @IsString() policyNumber?: string;
@IsOptional() @IsString() insuredName?: string;
@IsOptional() @IsString() additionalInsured?: string;
@IsOptional() @IsString() agentName?: string;
@IsOptional() @IsString() legalAddress?: string;
@IsOptional() @IsString() zip?: string;
@IsOptional() @IsDateString() policyFrom?: string;
@IsOptional() @IsDateString() policyTo?: string;
@IsOptional() @IsDateString() policyDate?: string;
@IsOptional() @IsEnum(["MXN", "USD", "EUR"]) currency?: "MXN" | "USD" | "EUR";
@IsOptional() @IsNumber() netPremium?: number;
@IsOptional() @IsNumber() policyFee?: number;
@IsOptional() @IsNumber() brokerFee?: number;
@IsOptional() @IsNumber() total?: number;
@IsOptional() @IsString() premiumPayment?: string;
/** Coverages parsed off the PDF, passed through verbatim to Policy.coveragesJson. */
@IsOptional() @IsObject() coveragesJson?: unknown;
/** When true, write a Transaction(domain=INSURANCE, amount=-netPremium)
* in addition to creating/updating the Policy. Skipped if netPremium is
* null or zero. */
@IsOptional() postPremium?: boolean;
}
export class ConfirmPolicyBatchDto {
@IsArray()
@ValidateNested({ each: true })
@Type(() => ConfirmPolicyDocumentDto)
documents!: ConfirmPolicyDocumentDto[];
}
/** Staff correction of one document's extracted fields or its match. */
export class ReviewPolicyDocumentDto {
@IsOptional() @IsString() policyNumber?: string;
@IsOptional() @IsString() insuredName?: string;
@IsOptional() @IsString() additionalInsured?: string;
@IsOptional() @IsString() agentName?: string;
@IsOptional() @IsString() legalAddress?: string;
@IsOptional() @IsString() zip?: string;
@IsOptional() @IsDateString() policyFrom?: string;
@IsOptional() @IsDateString() policyTo?: string;
@IsOptional() @IsDateString() policyDate?: string;
@IsOptional() @IsString() currency?: string;
@IsOptional() @IsNumber() netPremium?: number;
@IsOptional() @IsNumber() policyFee?: number;
@IsOptional() @IsNumber() brokerFee?: number;
@IsOptional() @IsNumber() total?: number;
@IsOptional() @IsString() premiumPayment?: string;
@IsOptional() @IsObject() coveragesJson?: unknown;
/** Set by the reviewer when the document matched an existing Policy. */
@IsOptional() @IsString() matchedPolicyId?: string;
/** Set by the reviewer when creating a new Policy. */
@IsOptional() @IsString() matchedCustomerId?: string;
/** Force-confirm a doc even when the matcher left it ambiguous. */
@IsOptional() forceConfirm?: boolean;
}
export class CreatePolicyOcrBatchDto {
@IsOptional() @IsString() @MinLength(1) label?: string;
}
@@ -0,0 +1,18 @@
import { Module } from "@nestjs/common";
import { OcrModule } from "../ocr/ocr.module";
import { PolicyOcrController } from "./policy-ocr.controller";
import { PolicyOcrService } from "./policy-ocr.service";
import { PolicyMatcherService } from "./policy-matcher.service";
/**
* Reuses the OCR seam from OcrModule unchanged: the Tesseract provider is
* bound there and `OcrProvider` is the only thing the parsers touch. This
* module registers its own controller + service + matcher; nothing about
* utility ingestion needs to know about it.
*/
@Module({
imports: [OcrModule],
controllers: [PolicyOcrController],
providers: [PolicyOcrService, PolicyMatcherService],
})
export class PolicyOcrModule {}
@@ -0,0 +1,720 @@
import {
BadRequestException,
Inject,
Injectable,
Logger,
NotFoundException,
} from "@nestjs/common";
import { Currency, Prisma } from "@jorgecuadros/database";
import { PrismaService } from "../prisma/prisma.service";
import { StorageService } from "../storage/storage.service";
import type { UploadedFileLike } from "../storage/upload-file";
import { OCR_PROVIDER, type OcrPage, type OcrProvider } from "../statements/ocr/ocr.provider";
import { parsePolicy } from "./parsers/policy-parser";
import { PolicyMatcherService } from "./policy-matcher.service";
import type {
ConfirmPolicyBatchDto,
ConfirmPolicyDocumentDto,
ReviewPolicyDocumentDto,
} from "./policy-ocr.dto";
/**
* Insurance OCR intake — mirrors the statement pipeline at
* `apps/api/src/statements/statements.service.ts`. Reuses the OCR seam and
* Tesseract binding unchanged; the parsers and matcher are policy-specific.
*
* Why a parallel pipeline rather than a column on StatementDocument: the
* matcher keys on `Policy.policyNumber`, the confirm step writes to a
* different table (`Policy`, not `Transaction`), and the review UI shows
* different fields. Sharing one queue would either bloat the row with null
* columns or force the review screen to branch on a discriminator — both
* worse than a thin second table.
*/
@Injectable()
export class PolicyOcrService {
private readonly logger = new Logger(PolicyOcrService.name);
constructor(
private readonly prisma: PrismaService,
private readonly storage: StorageService,
private readonly matcher: PolicyMatcherService,
@Inject(OCR_PROVIDER) private readonly ocr: OcrProvider,
) {}
ocrAvailable(): Promise<boolean> {
return this.ocr.available();
}
storageAvailable(): boolean {
return this.storage.available;
}
// --- ingest ---------------------------------------------------------------
async createBatch(
files: UploadedFileLike[],
uploadedById: string,
label?: string,
) {
if (!files?.length) throw new BadRequestException("No se recibió ningún archivo.");
if (!(await this.ocr.available())) {
throw new BadRequestException(
"El servidor no tiene OCR instalado; no se pueden leer PDFs de pólizas.",
);
}
if (!this.storage.available) {
throw new BadRequestException(
"El almacenamiento de documentos no está configurado; no se pueden " +
"guardar los PDFs escaneados.",
);
}
const batch = await this.prisma.policyOcrBatch.create({
data: { provider: "GMX", uploadedById, label, fileCount: files.length },
});
const copies = files.map((f) => ({ buffer: f.buffer, name: f.originalname }));
void this.process(batch.id, copies).catch(async (err) => {
this.logger.error(`Policy OCR batch ${batch.id} failed: ${(err as Error).message}`);
await this.prisma.policyOcrBatch.update({
where: { id: batch.id },
data: { status: "FAILED", error: (err as Error).message },
});
});
return batch;
}
/**
* Render → text → parse → match, **one PolicyOcrDocument row per uploaded
* file**. The GMX certificate is a 2-page PDF where page 1 carries the
* contract header and page 2 carries the per-coverage table — both pages
* describe the SAME policy, so the parser concatenates them and the
* matcher runs once. `pageNumber` on the row is repurposed as the file
* ordinal within the batch (1, 2, 3…) — the unique constraint
* `(batchId, pageNumber)` still holds and lets a single batch carry many
* policies.
*
* The doc's `storageKey` is the SOURCE PDF (`policy-ocr/{batchId}/source-N.pdf`)
* rather than a rendered page image, so the review screen can embed the
* exact artifact the office received. The rendered page PNGs are still
* stored under `policy-ocr/{batchId}/page-M.png` for any future re-OCR or
* image-based audit, but they aren't used as `storageKey` for the document.
*/
private async process(
batchId: string,
files: { buffer: Buffer; name?: string }[],
) {
await this.prisma.policyOcrBatch.update({
where: { id: batchId },
data: { status: "PROCESSING" },
});
let fileOrdinal = 0;
let globalPageOrdinal = 0;
for (const file of files) {
fileOrdinal += 1;
const sourceKey = `policy-ocr/${batchId}/source-${fileOrdinal}.pdf`;
await this.storage.put(sourceKey, file.buffer, "application/pdf");
const pages = await this.ocr.renderPages(file.buffer);
const textLayer = await this.ocr.textPages(file.buffer).catch(() => []);
// One OcrPage per rendered page: text-layer wins when present (cheap,
// exact), OCR the rendered image when it isn't. Same precedence rule
// as the statement OCR pipeline.
const perPageOcr: OcrPage[] = [];
for (const [index, image] of pages.entries()) {
globalPageOrdinal += 1;
const pageStorageKey = `policy-ocr/${batchId}/page-${globalPageOrdinal}.png`;
await this.storage.put(pageStorageKey, image, "image/png");
const embedded = textLayer[index] ?? null;
const pageOcr = embedded ?? (await this.ocr.recognize(image));
perPageOcr.push(pageOcr);
}
// Concatenate every page's text with a blank line between pages so the
// parser's anchored regexes (^From$, ^Currency\s+...) still work
// across page boundaries — pdftotext -bbox-layout produces newline-
// separated text per page already, the `\n\n` just preserves a clear
// boundary in ocrRawText for debugging.
const mergedText = perPageOcr.map((p) => p.text).join("\n\n");
const avgConfidence =
perPageOcr.length === 0
? 0
: perPageOcr.reduce((s, p) => s + p.confidence, 0) / perPageOcr.length;
const synthetic: OcrPage = {
text: mergedText,
words: [],
confidence: avgConfidence,
};
try {
const parsed = parsePolicy(synthetic);
if (parsed.provider === "") {
throw new Error("no se reconoció el proveedor");
}
const match = await this.matcher.match(parsed);
const notes = [...parsed.notes, match.note].filter(Boolean);
// Confident when exactly one Policy carries the printed number —
// the only unambiguous hit we trust. A new policy (no match) still
// needs a customer pick, so it stays in review.
const trusted = match.confident && parsed.policyNumber != null;
await this.prisma.policyOcrDocument.create({
data: {
batchId,
pageNumber: fileOrdinal,
storageKey: sourceKey,
status: trusted ? "MATCHED" : "NEEDS_REVIEW",
ocrRawText: mergedText,
ocrConfidence: new Prisma.Decimal(avgConfidence.toFixed(3)),
provider: parsed.provider,
extractedPolicyNumber: parsed.policyNumber,
extractedInsuredName: parsed.insuredName,
extractedAdditionalInsured: parsed.additionalInsured,
extractedAgentName: parsed.agentName,
extractedLegalAddress: parsed.legalAddress,
extractedZip: parsed.zip,
extractedPolicyFrom: parsed.policyFrom,
extractedPolicyTo: parsed.policyTo,
extractedPolicyDate: parsed.policyDate,
extractedCurrency: parsed.currency,
extractedNetPremium:
parsed.netPremium != null ? new Prisma.Decimal(parsed.netPremium) : null,
extractedPolicyFee:
parsed.policyFee != null ? new Prisma.Decimal(parsed.policyFee) : null,
extractedBrokerFee:
parsed.brokerFee != null ? new Prisma.Decimal(parsed.brokerFee) : null,
extractedTotal:
parsed.total != null ? new Prisma.Decimal(parsed.total) : null,
extractedCoveragesJson: parsed.coverages.length
? (parsed.coverages as unknown as Prisma.InputJsonValue)
: Prisma.DbNull,
extractedPremiumPayment: parsed.premiumPayment,
matchedPolicyId: match.policyId,
matchedCustomerId: match.customerId,
matchCandidates: match.candidates.length
? (match.candidates as unknown as Prisma.InputJsonValue)
: Prisma.DbNull,
matchNote: notes.join("; ").slice(0, 190),
},
});
} catch (err) {
// The file as a whole failed to parse (no provider, parse exception).
// One OCR_FAILED row per file is the right granularity — the page
// images are still on disk for a re-run after a parser fix.
await this.prisma.policyOcrDocument.create({
data: {
batchId,
pageNumber: fileOrdinal,
storageKey: sourceKey,
status: "OCR_FAILED",
matchNote: (err as Error).message.slice(0, 190),
},
});
}
}
await this.prisma.policyOcrBatch.update({
where: { id: batchId },
data: { status: "READY_FOR_REVIEW" },
});
}
// --- reads ----------------------------------------------------------------
async listBatches(page: number, pageSize: number) {
const [total, items] = await this.prisma.$transaction([
this.prisma.policyOcrBatch.count(),
this.prisma.policyOcrBatch.findMany({
orderBy: { createdAt: "desc" },
skip: (page - 1) * pageSize,
take: pageSize,
include: {
uploadedBy: { select: { name: true } },
_count: { select: { documents: true } },
},
}),
]);
return { items, total, page, pageSize, pageCount: Math.ceil(total / pageSize) };
}
async getBatch(id: string) {
const batch = await this.prisma.policyOcrBatch.findUnique({
where: { id },
include: { uploadedBy: { select: { name: true } } },
});
if (!batch) throw new NotFoundException("Lote no encontrado.");
const counts = await this.prisma.policyOcrDocument.groupBy({
by: ["status"],
where: { batchId: id },
_count: { _all: true },
});
return {
...batch,
byStatus: Object.fromEntries(counts.map((c) => [c.status, c._count._all])),
};
}
async listDocuments(batchId: string) {
return this.prisma.policyOcrDocument.findMany({
where: { batchId },
orderBy: { pageNumber: "asc" },
include: {
matchedCustomer: { select: { id: true, name: true } },
matchedPolicy: {
select: {
id: true,
policyNumber: true,
customerId: true,
customer: { select: { name: true } },
},
},
},
});
}
/**
* The source PDF for the document, so the review screen can show the
* exact artifact the office uploaded (the browser's PDF viewer handles
* scrolling, zoom, and selection natively). The rendered page PNGs
* remain on disk under `policy-ocr/{batchId}/page-N.png` for any
* future re-OCR, but the doc row points here at the source.
*/
async pageImage(documentId: string) {
const doc = await this.prisma.policyOcrDocument.findUnique({
where: { id: documentId },
select: { storageKey: true },
});
if (!doc) throw new NotFoundException("Documento no encontrado.");
return this.storage.getStream(doc.storageKey);
}
// --- review ---------------------------------------------------------------
async review(id: string, dto: ReviewPolicyDocumentDto, reviewedById: string) {
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
if (!doc) throw new NotFoundException("Documento no encontrado.");
if (doc.status === "POSTED") {
throw new BadRequestException("Este documento ya fue aplicado.");
}
// Trusting a customer-supplied pair (policyId, customerId) without
// cross-check is how a document lands on the wrong customer's ledger;
// pin them here from the DB.
let matchedPolicyId = dto.matchedPolicyId ?? doc.matchedPolicyId;
let matchedCustomerId = doc.matchedCustomerId;
if (matchedPolicyId) {
const p = await this.prisma.policy.findUnique({
where: { id: matchedPolicyId },
select: { customerId: true },
});
if (!p) throw new BadRequestException("Póliza no encontrada.");
matchedCustomerId = p.customerId;
} else if (dto.matchedCustomerId) {
const c = await this.prisma.customer.findUnique({
where: { id: dto.matchedCustomerId },
select: { id: true },
});
if (!c) throw new BadRequestException("Cliente no encontrado.");
matchedCustomerId = c.id;
}
return this.prisma.policyOcrDocument.update({
where: { id },
data: {
extractedPolicyNumber: dto.policyNumber ?? undefined,
extractedInsuredName: dto.insuredName ?? undefined,
extractedAdditionalInsured: dto.additionalInsured ?? undefined,
extractedAgentName: dto.agentName ?? undefined,
extractedLegalAddress: dto.legalAddress ?? undefined,
extractedZip: dto.zip ?? undefined,
extractedPolicyFrom: dto.policyFrom ? new Date(dto.policyFrom) : undefined,
extractedPolicyTo: dto.policyTo ? new Date(dto.policyTo) : undefined,
extractedPolicyDate: dto.policyDate ? new Date(dto.policyDate) : undefined,
extractedCurrency: dto.currency ?? undefined,
extractedNetPremium:
dto.netPremium != null ? new Prisma.Decimal(dto.netPremium) : undefined,
extractedPolicyFee:
dto.policyFee != null ? new Prisma.Decimal(dto.policyFee) : undefined,
extractedBrokerFee:
dto.brokerFee != null ? new Prisma.Decimal(dto.brokerFee) : undefined,
extractedTotal:
dto.total != null ? new Prisma.Decimal(dto.total) : undefined,
extractedCoveragesJson: dto.coveragesJson
? (dto.coveragesJson as Prisma.InputJsonValue)
: undefined,
extractedPremiumPayment: dto.premiumPayment ?? undefined,
matchedPolicyId,
matchedCustomerId,
status: dto.forceConfirm ? "CONFIRMED" : "MATCHED",
reviewedById,
reviewedAt: new Date(),
},
});
}
async reject(id: string, reviewedById: string) {
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
if (!doc) throw new NotFoundException("Documento no encontrado.");
if (doc.status === "POSTED") {
throw new BadRequestException("Este documento ya fue aplicado.");
}
return this.prisma.policyOcrDocument.update({
where: { id },
data: { status: "REJECTED", reviewedById, reviewedAt: new Date() },
});
}
// --- confirm --------------------------------------------------------------
/**
* Apply every confirmed document: create or update the Policy, attach the
* source PDF as a PolicyDocument, and (when staff asked + premium parses)
* write a Transaction row. Each step is guarded by status checks so a
* double-confirm cannot re-apply a document.
*/
async confirmBatch(batchId: string, dto: ConfirmPolicyBatchDto, reviewedById: string) {
const batch = await this.prisma.policyOcrBatch.findUnique({ where: { id: batchId } });
if (!batch) throw new NotFoundException("Lote no encontrado.");
const results: { documentId: string; policyId: string; postedTransactionId: string | null }[] = [];
for (const item of dto.documents) {
const doc = await this.prisma.policyOcrDocument.findUnique({
where: { id: item.documentId },
});
if (!doc) {
throw new BadRequestException(`Documento ${item.documentId} no encontrado.`);
}
if (doc.status === "POSTED") {
throw new BadRequestException(
`El documento página ${doc.pageNumber} ya fue aplicado.`,
);
}
if (!item.policyId && !item.customerId) {
throw new BadRequestException(
`Documento página ${doc.pageNumber}: falta póliza destino o cliente.`,
);
}
// 1. Resolve target Policy (create or update). Field selection: every
// non-null `extracted*` on the doc (post-review) is written. Null is
// preserved — never overwrite an existing Policy's `netPremium` with
// null because the certificate page didn't carry one.
let policyId = item.policyId ?? null;
if (policyId) {
const updateData = buildPolicyUpdateFromDoc(item, doc);
await this.prisma.policy.update({
where: { id: policyId },
data: updateData,
});
} else {
// Create under the picked customer. `policyNumber` is the only field
// that must be present.
if (!item.policyNumber && !doc.extractedPolicyNumber) {
throw new BadRequestException(
`Documento página ${doc.pageNumber}: falta número de póliza.`,
);
}
const createData = buildPolicyCreateFromDoc(item, doc, item.customerId!);
const created = await this.prisma.policy.create({
data: createData,
});
policyId = created.id;
}
// 2. Attach the source PDF as a PolicyDocument. `doc.storageKey`
// already points at the exact upload (`policy-ocr/{batchId}/source-N.pdf`)
// so the attach is just a stream copy into the policy's namespace —
// the previous per-page "which file did this page come from" walk is
// gone because one PDF = one doc now.
await this.attachSourcePdf(doc.storageKey, policyId);
// 3. Optionally post the premium to the ledger. Only when staff
// explicitly asked (`postPremium` true) and netPremium parses — without
// that gate a missing premium would silently book $0.
let postedTransactionId: string | null = null;
const premium =
item.netPremium != null
? item.netPremium
: doc.extractedNetPremium != null
? Number(doc.extractedNetPremium)
: null;
if (item.postPremium && premium && premium > 0) {
const tx = await this.prisma.transaction.create({
data: {
customerId: (await this.policyCustomerId(policyId))!,
domain: "INSURANCE",
amount: new Prisma.Decimal(-Math.abs(premium)),
transactionDate: doc.extractedPolicyDate ?? doc.extractedPolicyFrom ?? new Date(),
currency: (item.currency ??
doc.extractedCurrency ??
"MXN") as Currency,
reference: item.policyNumber ?? doc.extractedPolicyNumber ?? null,
period: null,
captureSource: "OCR",
captureRef: doc.id,
message: `Prima de póliza ${item.policyNumber ?? doc.extractedPolicyNumber ?? ""}`,
},
});
postedTransactionId = tx.id;
}
await this.prisma.policyOcrDocument.update({
where: { id: doc.id },
data: {
status: "POSTED",
matchedPolicyId: policyId,
reviewedById,
reviewedAt: new Date(),
createdPolicyId: item.policyId ? null : policyId,
postedTransactionId,
},
});
results.push({
documentId: doc.id,
policyId,
postedTransactionId,
});
}
await this.closeIfDone(batchId);
return {
applied: results.length,
policies: results.map((r) => r.policyId),
postedTransactions: results.filter((r) => r.postedTransactionId).length,
};
}
/**
* Stream the source PDF (`sourceKey`, set by `process` on the doc row)
* into the policy's storage namespace and create a `PolicyDocument`
* pointer. Trivial now that the doc row holds the exact source key —
* the old per-page "which file did this page come from" walk is gone.
*/
private async attachSourcePdf(sourceKey: string, policyId: string): Promise<void> {
const got = await this.storage.getStream(sourceKey);
const chunks: Buffer[] = [];
for await (const c of got.stream) chunks.push(c as Buffer);
const buf = Buffer.concat(chunks);
const newKey = `policy/${policyId}/${Date.now()}-${crypto.randomUUID()}.pdf`;
await this.storage.put(newKey, buf, "application/pdf");
await this.prisma.policyDocument.create({
data: {
policyId,
documentType: "GMX_POLICY",
storageKey: newKey,
},
});
}
private async policyCustomerId(policyId: string): Promise<string | null> {
const p = await this.prisma.policy.findUnique({
where: { id: policyId },
select: { customerId: true },
});
return p?.customerId ?? null;
}
private async closeIfDone(batchId: string) {
const open = await this.prisma.policyOcrDocument.count({
where: {
batchId,
status: { in: ["PENDING_OCR", "NEEDS_REVIEW", "MATCHED", "CONFIRMED"] },
},
});
if (open === 0) {
await this.prisma.policyOcrBatch.update({
where: { id: batchId },
data: { status: "COMPLETED", completedAt: new Date() },
});
}
}
}
/** Map a (post-review) doc + final confirmed fields onto a `Policy.update`
* payload. Every field that is null in both inputs is omitted so we never
* write null over a value the Policy already carries (the GMX certificate
* has no premium — we must not blank the existing Policy.netPremium). */
function buildPolicyUpdateFromDoc(
item: ConfirmPolicyDocumentDto,
doc: {
extractedPolicyNumber: string | null;
extractedInsuredName: string | null;
extractedAdditionalInsured: string | null;
extractedAgentName: string | null;
extractedLegalAddress: string | null;
extractedZip: string | null;
extractedPolicyFrom: Date | null;
extractedPolicyTo: Date | null;
extractedPolicyDate: Date | null;
extractedCurrency: string | null;
extractedNetPremium: Prisma.Decimal | null;
extractedPolicyFee: Prisma.Decimal | null;
extractedBrokerFee: Prisma.Decimal | null;
extractedTotal: Prisma.Decimal | null;
extractedCoveragesJson: Prisma.JsonValue | null;
extractedPremiumPayment: string | null;
},
): Prisma.PolicyUpdateInput {
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
if (a != null) return new Prisma.Decimal(a);
if (b != null) return b;
return undefined;
};
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
if (a) return new Date(a);
if (b) return b;
return undefined;
};
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
if (a != null && a !== "") return a;
if (b != null && b !== "") return b;
return undefined;
};
return {
policyNumber: strOrUndef(item.policyNumber, doc.extractedPolicyNumber),
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
total: numOrUndef(item.total, doc.extractedTotal),
// coveragesJson / observations: freeform, keep the GMX data when present.
coveragesJson:
item.coveragesJson !== undefined
? (item.coveragesJson as Prisma.InputJsonValue)
: doc.extractedCoveragesJson != null
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
: undefined,
// Premium payment cadence ("CONTADO") and insured-name fields land in
// `observations` so the PolicyForm's edits stay the source of truth for
// structured fields. The reviewer can move them by hand if needed.
observations: joinObservations(
doc.extractedInsuredName,
doc.extractedAdditionalInsured,
doc.extractedLegalAddress,
doc.extractedZip,
doc.extractedPremiumPayment,
item,
),
};
}
/** Same shape as `buildPolicyUpdateFromDoc`, but for `Policy.create`. The
* `customerId` is supplied separately and `policyNumber` is required (a
* Policy without a number can't be re-matched by the OCR pipeline). */
function buildPolicyCreateFromDoc(
item: ConfirmPolicyDocumentDto,
doc: {
extractedPolicyNumber: string | null;
extractedInsuredName: string | null;
extractedAdditionalInsured: string | null;
extractedAgentName: string | null;
extractedLegalAddress: string | null;
extractedZip: string | null;
extractedPolicyFrom: Date | null;
extractedPolicyTo: Date | null;
extractedPolicyDate: Date | null;
extractedCurrency: string | null;
extractedNetPremium: Prisma.Decimal | null;
extractedPolicyFee: Prisma.Decimal | null;
extractedBrokerFee: Prisma.Decimal | null;
extractedTotal: Prisma.Decimal | null;
extractedCoveragesJson: Prisma.JsonValue | null;
extractedPremiumPayment: string | null;
},
customerId: string,
): Prisma.PolicyUncheckedCreateInput {
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
if (a != null) return new Prisma.Decimal(a);
if (b != null) return b;
return undefined;
};
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
if (a) return new Date(a);
if (b) return b;
return undefined;
};
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
if (a != null && a !== "") return a;
if (b != null && b !== "") return b;
return undefined;
};
const policyNumber =
strOrUndef(item.policyNumber, doc.extractedPolicyNumber);
if (!policyNumber) {
// Caller already guards this; the throw is a type-narrowing aid.
throw new Error("policyNumber required for create");
}
return {
policyNumber,
customerId,
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
total: numOrUndef(item.total, doc.extractedTotal),
coveragesJson:
item.coveragesJson !== undefined
? (item.coveragesJson as Prisma.InputJsonValue)
: doc.extractedCoveragesJson != null
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
: undefined,
observations: joinObservations(
doc.extractedInsuredName,
doc.extractedAdditionalInsured,
doc.extractedLegalAddress,
doc.extractedZip,
doc.extractedPremiumPayment,
item,
),
};
}
function joinObservations(
insured: string | null,
additional: string | null,
address: string | null,
zip: string | null,
premiumPayment: string | null,
item: ConfirmPolicyDocumentDto,
): string | undefined {
const lines: string[] = [];
const insuredName = strOrUndefDb(item.insuredName, insured);
if (insuredName) lines.push(`Asegurado: ${insuredName}`);
const additionalInsured = strOrUndefDb(item.additionalInsured, additional);
if (additionalInsured) lines.push(`Asegurado adicional: ${additionalInsured}`);
const legalAddress = strOrUndefDb(item.legalAddress, address);
if (legalAddress) lines.push(`Dirección: ${legalAddress}`);
const zipVal = strOrUndefDb(item.zip, zip);
if (zipVal) lines.push(`C.P.: ${zipVal}`);
const cadence = strOrUndefDb(item.premiumPayment, premiumPayment);
if (cadence) lines.push(`Pago de prima: ${cadence}`);
return lines.length ? lines.join("\n") : undefined;
}
function strOrUndefDb(a: string | undefined, b: string | null): string | undefined {
if (a != null && a !== "") return a;
if (b != null && b !== "") return b;
return undefined;
}
+6 -11
View File
@@ -1,23 +1,18 @@
import { Module } from "@nestjs/common";
import { BillingModule } from "../billing/billing.module";
import { OcrModule } from "../ocr/ocr.module";
import { StatementsController } from "./statements.controller";
import { StatementsService } from "./statements.service";
import { StatementMatcherService } from "./statement-matcher.service";
import { OCR_PROVIDER } from "./ocr/ocr.provider";
import { TesseractOcrProvider } from "./ocr/tesseract.provider";
/**
* The concrete OCR engine is bound here and nowhere else — everything
* downstream depends on the OcrProvider interface, so swapping Tesseract for a
* managed extraction API is a one-line change in this file.
* The concrete OCR engine is bound in OcrModule (see apps/api/src/ocr/) —
* everything downstream depends on the OcrProvider interface, so swapping
* Tesseract for a managed extraction API is a one-line change there.
*/
@Module({
imports: [BillingModule],
imports: [BillingModule, OcrModule],
controllers: [StatementsController],
providers: [
StatementsService,
StatementMatcherService,
{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider },
],
providers: [StatementsService, StatementMatcherService],
})
export class StatementsModule {}