feat(polizas): OCR capture for insurance policy PDFs
Mirrors the utility statement intake on the insurance side: a policy_ocr batch/document pair of tables, a GMX parser, a matcher keyed on Policy.policyNumber, and a "Captura" screen under /polizas that proposes policy -> customer for staff to confirm. Lifts the OCR seam out of StatementsModule into its own OcrModule so PolicyOcrModule can inject OCR_PROVIDER without taking on the rest of the statement pipeline; StatementsModule now imports it and binds nothing itself. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -10,6 +10,7 @@ import { PoliciesModule } from "./policies/policies.module";
|
||||
import { PropertiesModule } from "./properties/properties.module";
|
||||
import { BillingModule } from "./billing/billing.module";
|
||||
import { StatementsModule } from "./statements/statements.module";
|
||||
import { PolicyOcrModule } from "./policy-ocr/policy-ocr.module";
|
||||
import { BankModule } from "./bank/bank.module";
|
||||
import { OpsModule } from "./ops/ops.module";
|
||||
import { ReportsModule } from "./reports/reports.module";
|
||||
@@ -28,6 +29,7 @@ import { AppController } from "./app.controller";
|
||||
PropertiesModule,
|
||||
BillingModule,
|
||||
StatementsModule,
|
||||
PolicyOcrModule,
|
||||
BankModule,
|
||||
OpsModule,
|
||||
ReportsModule,
|
||||
|
||||
@@ -24,6 +24,8 @@ export type Ability =
|
||||
| "policy:create"
|
||||
| "policy:update"
|
||||
| "policy:delete"
|
||||
| "policy:ingest"
|
||||
| "policy:ocr-review"
|
||||
| "property:create"
|
||||
| "property:update"
|
||||
| "property:delete"
|
||||
@@ -46,6 +48,10 @@ export const ABILITY_MIN: Record<Ability, Role> = {
|
||||
"policy:create": "STAFF",
|
||||
"policy:update": "STAFF",
|
||||
"policy:delete": "MANAGER",
|
||||
// Insurance OCR intake is the same trust tier as statement OCR: STAFF can
|
||||
// upload + confirm, nothing reaches the books unconfirmed.
|
||||
"policy:ingest": "STAFF",
|
||||
"policy:ocr-review": "STAFF",
|
||||
"property:create": "STAFF",
|
||||
"property:update": "STAFF",
|
||||
"property:delete": "MANAGER",
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { OCR_PROVIDER } from "../statements/ocr/ocr.provider";
|
||||
import { TesseractOcrProvider } from "../statements/ocr/tesseract.provider";
|
||||
|
||||
/**
|
||||
* Lifts the OCR seam out of StatementsModule so other modules (today:
|
||||
* PolicyOcrModule) can inject OCR_PROVIDER without taking on the rest of
|
||||
* the statement intake. StatementsModule itself imports this and gets the
|
||||
* provider the same way.
|
||||
*
|
||||
* The concrete engine is still bound here — Tesseract today, a managed
|
||||
* extraction API later is a one-line change in this file.
|
||||
*/
|
||||
@Module({
|
||||
providers: [{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider }],
|
||||
exports: [OCR_PROVIDER],
|
||||
})
|
||||
export class OcrModule {}
|
||||
@@ -0,0 +1,147 @@
|
||||
import type { OcrPage } from "../../statements/ocr/ocr.provider";
|
||||
import {
|
||||
detectPolicyProvider,
|
||||
parsePolicy,
|
||||
type ParsedCoverage,
|
||||
} from "./policy-parser";
|
||||
|
||||
/**
|
||||
* Verbatim excerpts of what the GMX portal's translation PDF actually
|
||||
* rendered through pdftotext — same convention as the statement parser
|
||||
* tests, where invented-clean input would test nothing because clean input
|
||||
* is not the failure mode.
|
||||
*/
|
||||
function page(text: string): OcrPage {
|
||||
return { text, words: [], confidence: 0.95 };
|
||||
}
|
||||
|
||||
describe("detectPolicyProvider", () => {
|
||||
it("claims GMX from the brand wordmark on the letterhead", () => {
|
||||
expect(
|
||||
detectPolicyProvider(
|
||||
"Grupo Mexicano de Seguros, S.A. de C.V.\nTecoyotitla 412, Edificio GMX",
|
||||
),
|
||||
).toBe("GMX");
|
||||
});
|
||||
|
||||
it("claims GMX from the 'gmx.com.mx' footer URL", () => {
|
||||
expect(detectPolicyProvider("JUNTOS EL RIESGO ES MENOR\nwww.gmx.com.mx")).toBe("GMX");
|
||||
});
|
||||
});
|
||||
|
||||
describe("parsePolicy / GMX", () => {
|
||||
// Verbatim text extracted from ~/Downloads/HC_Folio_000767_Traduccion.pdf via
|
||||
// `pdftotext -layout`. Two pages joined by "\n\n".
|
||||
const GMX_FULL = page(
|
||||
"Multiple Policy\nHome\n" +
|
||||
"Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses, to insurance:\n" +
|
||||
"Insured JON ASHLEY STRABALA\n" +
|
||||
"Additional insured VIVIAN\n" +
|
||||
"Legal address BONAMPACK No. EXT26 No.INT 0 COL. Punta Bandera, Tijuana, Baja California, C.P. 22550\n" +
|
||||
"ZIP 22550 Income Tax No. XEXX-010101-000\n" +
|
||||
"Broker (1176) Jorge Humberto Cuadros\n" +
|
||||
"Term 12 months\n" +
|
||||
"From 19/07/2026\n" +
|
||||
"To 19/07/2027 at twelve hours (noon) Mexico City time.\n" +
|
||||
"Currency DOLARES Premium payment CONTADO\n" +
|
||||
"Free translation from the Spanish Insurance contract. The English text is just copy given by courtesy. In case of a dispute, the Spanish will prevail over the English version.\n" +
|
||||
"Agreed clauses:\n" +
|
||||
"•The insured and GMX Hereby declared...\n" +
|
||||
"From the above, the present contract shall not be considered under the condition mentioned within article 36-B from the Insurance Companies General Law. Therefore it shall not be required its registration before the Comision National de Seguros y Fianzas.\n" +
|
||||
"July 23, 2026\n" +
|
||||
"Authority sign.\n" +
|
||||
"Grupo Mexicano de Seguros, S.A. de C.V.\n" +
|
||||
"Tecoyotitla 412, Edificio GMX\n" +
|
||||
"JUNTOS EL RIESGO ES MENOR\n" +
|
||||
"www.gmx.com.mx\n\n" +
|
||||
"Risk Insured Amount Deductible Loss Participation\n" +
|
||||
"Building $350,000.00 Not applies Not applies\n" +
|
||||
"Contents $60,000.00 Not applies Not applies\n" +
|
||||
"ADDITIONAL RISK\n" +
|
||||
"Risk Insured Amount Deductible Loss Participation\n" +
|
||||
"Debris removal Building $35,000.00 Not applies Not applies\n" +
|
||||
"Debris removal Contents $6,000.00 Not applies Not applies\n" +
|
||||
"Outdoors Constructions $10,000.00 5% 10%\n" +
|
||||
"Coverage Extention Covered Not applies Not applies\n" +
|
||||
"All Risk Covered Not applies Not applies\n" +
|
||||
"Earthquake and/or volcanic eruption Covered 2% of the sum insured for each damage structure 20%\n" +
|
||||
"Extra Expenses $41,000.00 Not applies Not applies\n" +
|
||||
"Robbery with violence $10,000.00 Not applies Not applies\n" +
|
||||
"Jewerly $3,900.00 Not applies Not applies\n" +
|
||||
"Electronic Equipment $10,000.00 Not applies Not applies\n" +
|
||||
"Glasses $10,000.00 Not applies Not applies\n" +
|
||||
"Tenant $200,000.00 Not applies Not applies\n" +
|
||||
"Family $200,000.00 Not applies Not applies\n" +
|
||||
"Family $200,000.00 Not applies Not applies\n" +
|
||||
"Domestic workers $7,010.00 Not applies Not applies\n" +
|
||||
"VALUES ADDED, HOME GMX",
|
||||
);
|
||||
|
||||
it("extracts the policy number, insured name, broker, dates, and currency", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
expect(p.provider).toBe("GMX");
|
||||
expect(p.policyNumber).toBe("007-037-07005947-0000-02");
|
||||
expect(p.insuredName).toBe("JON ASHLEY STRABALA");
|
||||
expect(p.additionalInsured).toBe("VIVIAN");
|
||||
expect(p.agentName).toBe("Jorge Humberto Cuadros");
|
||||
expect(p.policyFrom?.toISOString().slice(0, 10)).toBe("2026-07-19");
|
||||
expect(p.policyTo?.toISOString().slice(0, 10)).toBe("2027-07-19");
|
||||
expect(p.policyDate?.toISOString().slice(0, 10)).toBe("2026-07-23");
|
||||
expect(p.currency).toBe("USD");
|
||||
expect(p.zip).toBe("22550");
|
||||
expect(p.legalAddress).toContain("BONAMPACK");
|
||||
expect(p.premiumPayment).toBe("CONTADO");
|
||||
});
|
||||
|
||||
it("extracts every coverage row off the second page table", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
const byName = Object.fromEntries(p.coverages.map((c) => [c.risk, c]));
|
||||
expect(byName.Building?.insuredAmount).toBe(350000);
|
||||
expect(byName.Contents?.insuredAmount).toBe(60000);
|
||||
expect(byName["Debris removal Building"]?.insuredAmount).toBe(35000);
|
||||
expect(byName["Outdoors Constructions"]?.insuredAmount).toBe(10000);
|
||||
expect(byName["Outdoors Constructions"]?.deductible).toBe("5%");
|
||||
expect(byName["Outdoors Constructions"]?.lossParticipation).toBe("10%");
|
||||
// Free-text coverage cells kept verbatim (the policy form surfaces them
|
||||
// as observations, not as numbers).
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.insuredAmount).toBeNull();
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.deductible).toContain("2%");
|
||||
expect(byName["Earthquake and/or volcanic eruption"]?.lossParticipation).toBe("20%");
|
||||
expect(byName["All Risk"]?.insuredAmount).toBeNull();
|
||||
expect(p.coverages.length).toBeGreaterThan(10);
|
||||
});
|
||||
|
||||
it("leaves premium fields null on the certificate page and notes it", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
expect(p.netPremium).toBeNull();
|
||||
expect(p.total).toBeNull();
|
||||
expect(p.policyFee).toBeNull();
|
||||
expect(p.notes.join(" ")).toMatch(/prima/i);
|
||||
});
|
||||
|
||||
it("still parses when the broker parens are missing", () => {
|
||||
const p = parsePolicy(
|
||||
page(
|
||||
"Insured JON ASHLEY STRABALA\nBroker Jorge Humberto Cuadros\n" +
|
||||
"From 19/07/2026\nTo 19/07/2027\nCurrency DOLARES\n" +
|
||||
"Grupo Mexicano de Seguros",
|
||||
),
|
||||
);
|
||||
expect(p.agentName).toBe("Jorge Humberto Cuadros");
|
||||
});
|
||||
|
||||
it("rejects a page that carries no GMX signal at all", () => {
|
||||
const p = parsePolicy(page("Random unrelated document with no policy data."));
|
||||
expect(p.provider).toBe("");
|
||||
expect(p.notes.join(" ")).toContain("no se reconoció el proveedor");
|
||||
});
|
||||
|
||||
it("captures the deductible / loss-participation columns verbatim as strings", () => {
|
||||
const p = parsePolicy(GMX_FULL);
|
||||
const eq = p.coverages.find((c) => c.risk === "Earthquake and/or volcanic eruption");
|
||||
expect(eq).toBeDefined();
|
||||
const eqTyped = eq as ParsedCoverage;
|
||||
expect(eqTyped.deductible).toContain("sum insured");
|
||||
expect(eqTyped.lossParticipation).toBe("20%");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,422 @@
|
||||
import type { OcrPage } from "../../statements/ocr/ocr.provider";
|
||||
|
||||
/**
|
||||
* What one parsed policy page yields. All fields are nullable because each
|
||||
* provider prints a different subset (GMX's certificate has no premium
|
||||
* breakdown, only insured amounts; GMX's receipt page would carry the
|
||||
* premium), and the matcher + the review queue both work better with
|
||||
* "field was read" vs "field was not" rather than guessing.
|
||||
*/
|
||||
export interface ParsedPolicy {
|
||||
/** "GMX" today; the dispatcher lives on `detectProvider`. */
|
||||
provider: string;
|
||||
policyNumber: string | null;
|
||||
insuredName: string | null;
|
||||
additionalInsured: string | null;
|
||||
/** The "Broker" line on GMX — mapped onto `Policy.agentName`. */
|
||||
agentName: string | null;
|
||||
legalAddress: string | null;
|
||||
zip: string | null;
|
||||
policyFrom: Date | null;
|
||||
policyTo: Date | null;
|
||||
/** Signature/issue date — `Policy.policyDate`. */
|
||||
policyDate: Date | null;
|
||||
/** "MXN" | "USD" | …, derived from the printed currency word. */
|
||||
currency: string | null;
|
||||
netPremium: number | null;
|
||||
policyFee: number | null;
|
||||
brokerFee: number | null;
|
||||
total: number | null;
|
||||
/** "CONTADO" / "MENSUAL" / … — premium-payment cadence text. */
|
||||
premiumPayment: string | null;
|
||||
/**
|
||||
* GMX prints per-coverage rows in a table: Building / Contents /
|
||||
* Earthquake / … with insured amount, deductible, loss participation.
|
||||
* Preserved verbatim so a missing premium receipt still leaves the
|
||||
* coverages auditable on the Policy row.
|
||||
*/
|
||||
coverages: ParsedCoverage[];
|
||||
/** Human-readable trail of what was read, surfaced in the review queue. */
|
||||
notes: string[];
|
||||
}
|
||||
|
||||
export interface ParsedCoverage {
|
||||
/** "Building", "Contents", "Debris removal Building", "Earthquake…". */
|
||||
risk: string;
|
||||
insuredAmount: number | null;
|
||||
deductible: string | null;
|
||||
lossParticipation: string | null;
|
||||
}
|
||||
|
||||
// --- shared helpers ---------------------------------------------------------
|
||||
|
||||
const DIGIT_CONFUSIONS: Record<string, string> = {
|
||||
O: "0", o: "0", D: "0", I: "1", l: "1", "|": "1", S: "5", B: "8",
|
||||
};
|
||||
|
||||
/**
|
||||
* Tesseract confuses these glyphs inside numeric runs with some regularity.
|
||||
* Same map and same caveat as the statement parser: ONLY apply to fields
|
||||
* known to be digits, never to free text.
|
||||
*/
|
||||
function toDigits(s: string | null | undefined): string {
|
||||
if (!s) return "";
|
||||
return s
|
||||
.split("")
|
||||
.map((c) => DIGIT_CONFUSIONS[c] ?? c)
|
||||
.join("")
|
||||
.replace(/\D/g, "");
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a printed amount, treating `,` and `.` by position rather than by
|
||||
* assumption. Same algorithm as the statement parser — kept here so the
|
||||
* policy module is self-contained, since importing from `../../statements`
|
||||
* would couple two unrelated domains through a helper.
|
||||
*/
|
||||
function money(s: string | null | undefined): number | null {
|
||||
if (!s) return null;
|
||||
const cleaned = s.replace(/[\s$]/g, "");
|
||||
|
||||
let m = cleaned.match(/^(\d{1,3}(?:[.,]\d{3})+)([.,]\d{1,2})?$/);
|
||||
if (m) {
|
||||
const whole = m[1].replace(/[.,]/g, "");
|
||||
const cents = m[2] ? m[2].slice(1) : "";
|
||||
return Number(cents ? `${whole}.${cents.padEnd(2, "0")}` : whole);
|
||||
}
|
||||
|
||||
m = cleaned.match(/^(\d+)[.,](\d{2})$/);
|
||||
if (m) return Number(`${m[1]}.${m[2]}`);
|
||||
|
||||
const n = Number(cleaned.replace(/[,.]/g, ""));
|
||||
return Number.isFinite(n) ? n : null;
|
||||
}
|
||||
|
||||
function firstMatch(text: string, patterns: RegExp[]): string | null {
|
||||
for (const p of patterns) {
|
||||
const m = text.match(p);
|
||||
if (m?.[1]) return m[1].trim();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const MONTHS: Record<string, number> = {
|
||||
ENE: 0, FEB: 1, MAR: 2, ABR: 3, MAY: 4, JUN: 5,
|
||||
JUL: 6, AGO: 7, SEP: 8, OCT: 9, NOV: 10, DIC: 11,
|
||||
};
|
||||
|
||||
/**
|
||||
* DD/MM/YYYY (GMX) and the dash-separated ISO variants. Two-digit years are
|
||||
* windowed: < 50 → 20YY, ≥ 50 → 19YY, matching what a 1950-2049 window
|
||||
* expects from a paper document.
|
||||
*/
|
||||
function parseDate(raw: string | null | undefined): Date | null {
|
||||
if (!raw) return null;
|
||||
const s = raw.trim();
|
||||
|
||||
let m = s.match(/^(\d{1,2})\/(\d{1,2})\/(\d{4})$/);
|
||||
if (m) return utc(+m[3], +m[2] - 1, +m[1]);
|
||||
|
||||
m = s.match(/^(\d{1,2})[-\s/]([A-Z]{3})[-\s/](\d{2,4})$/i);
|
||||
if (m && MONTHS[m[2].toUpperCase()] !== undefined) {
|
||||
const yr = +m[3];
|
||||
const y = m[3].length === 2 ? (yr < 50 ? 2000 + yr : 1900 + yr) : yr;
|
||||
return utc(y, MONTHS[m[2].toUpperCase()], +m[1]);
|
||||
}
|
||||
|
||||
m = s.match(/^(\d{4})[-/](\d{1,2})[-/](\d{1,2})$/);
|
||||
if (m) return utc(+m[1], +m[2] - 1, +m[3]);
|
||||
|
||||
// "July 23, 2026" — the signature date on the GMX certificate.
|
||||
m = s.match(/^([A-Za-z]+)\s+(\d{1,2}),\s*(\d{4})$/);
|
||||
if (m) {
|
||||
const MONTH_NAMES: Record<string, number> = {
|
||||
january: 0, february: 1, march: 2, april: 3, may: 4, june: 5,
|
||||
july: 6, august: 7, september: 8, october: 9, november: 10, december: 11,
|
||||
};
|
||||
const mo = MONTH_NAMES[m[1].toLowerCase()];
|
||||
if (mo !== undefined) return utc(+m[3], mo, +m[2]);
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function utc(y: number, mo: number, d: number): Date | null {
|
||||
const dt = new Date(Date.UTC(y, mo, d));
|
||||
return Number.isNaN(dt.getTime()) ? null : dt;
|
||||
}
|
||||
|
||||
/** Map the printed currency word onto an ISO code. */
|
||||
function currencyCode(raw: string | null | undefined): string | null {
|
||||
if (!raw) return null;
|
||||
const s = raw.trim().toUpperCase();
|
||||
if (s.startsWith("PESO") || s === "MXN" || s.includes("NACIONAL")) return "MXN";
|
||||
if (s.startsWith("DOLAR") || s === "USD" || s.includes("DOLLAR")) return "USD";
|
||||
if (s === "EUR" || s.includes("EURO")) return "EUR";
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- provider detection -----------------------------------------------------
|
||||
|
||||
/**
|
||||
* Brand first, layout as a fallback. Same ordering rule as the statement
|
||||
* parser: a brand wordmark is the cheapest, most reliable discriminator, and
|
||||
* a layout rule that runs first can wrongly claim a page that happens to
|
||||
* carry the same shape string (the statement parser's lesson with CFE vs
|
||||
* GAS on "PERIODO FACTURADO").
|
||||
*/
|
||||
const BRAND: [string, RegExp][] = [
|
||||
["GMX", /\bGMX\b|Grupo\s*Mexicano\s*de\s*Seguros|gmx\.com\.mx|JUNTOS\s*EL\s*RIESGO\s*ES\s*MENOR/i],
|
||||
];
|
||||
|
||||
const LAYOUT: [string, RegExp][] = [
|
||||
["GMX", /Multiple\s*Policy|IMPUESTO\s*PREDIAL[\s\S]{0,80}EN\s*FECHA|Material\s*damages\s*Section/i],
|
||||
];
|
||||
|
||||
export function detectPolicyProvider(text: string): string | null {
|
||||
for (const group of [BRAND, LAYOUT]) {
|
||||
for (const [name, pattern] of group) {
|
||||
if (pattern.test(text)) return name;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- parsers ----------------------------------------------------------------
|
||||
|
||||
const PARSERS: Record<string, (page: OcrPage) => ParsedPolicy> = {
|
||||
GMX: parseGmx,
|
||||
};
|
||||
|
||||
const EMPTY_COVERAGE: ParsedCoverage = {
|
||||
risk: "",
|
||||
insuredAmount: null,
|
||||
deductible: null,
|
||||
lossParticipation: null,
|
||||
};
|
||||
|
||||
export function parsePolicy(page: OcrPage): ParsedPolicy {
|
||||
const provider = detectPolicyProvider(page.text);
|
||||
if (!provider) {
|
||||
return {
|
||||
provider: "",
|
||||
policyNumber: null,
|
||||
insuredName: null,
|
||||
additionalInsured: null,
|
||||
agentName: null,
|
||||
legalAddress: null,
|
||||
zip: null,
|
||||
policyFrom: null,
|
||||
policyTo: null,
|
||||
policyDate: null,
|
||||
currency: null,
|
||||
netPremium: null,
|
||||
policyFee: null,
|
||||
brokerFee: null,
|
||||
total: null,
|
||||
premiumPayment: null,
|
||||
coverages: [],
|
||||
notes: ["no se reconoció el proveedor"],
|
||||
};
|
||||
}
|
||||
return PARSERS[provider](page);
|
||||
}
|
||||
|
||||
// --- GMX --------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* GMX policy certificate layout (this is the translation PDF — the Spanish
|
||||
* version is the canonical source, but every GMX portal download is a
|
||||
* translation so the parser can rely on these English labels).
|
||||
*
|
||||
* Page 1 carries the contract header in a single boxed table:
|
||||
* Policy | Insured | Additional insured | Legal address | ZIP | Income Tax No.
|
||||
* Broker | Term | From | To | Currency | Premium payment
|
||||
* followed by an "Agreed clauses" block, the signature date, and the GMX
|
||||
* letterhead.
|
||||
*
|
||||
* Page 2 carries the per-coverage table (Risk / Insured Amount / Deductible /
|
||||
* Loss Participation) under "Material damages Section" and "ADDITIONAL RISK".
|
||||
*
|
||||
* Premium / total / fees are NOT on the certificate page — they live on
|
||||
* GMX's separate "recibo" PDF. The parser leaves them null and flags the
|
||||
* gap in `notes`; the matcher still proposes a Policy update from the
|
||||
* certificate alone, and the staff confirm step fills premium in by hand
|
||||
* or after a follow-up receipt upload.
|
||||
*/
|
||||
function parseGmx(page: OcrPage): ParsedPolicy {
|
||||
const text = page.text;
|
||||
const notes: string[] = [];
|
||||
|
||||
// ----- header table (page 1) --------------------------------------------
|
||||
// The Policy row repeats the number in a long run:
|
||||
// "Policy 007-037-07005947-0000-02 in accordance with the enclosed clauses…"
|
||||
// so taking the first token-shaped number is correct; the trailing prose
|
||||
// never looks like one. The dashes are part of the printed number — keep
|
||||
// them (don't run toDigits, which would flatten them).
|
||||
const policyNumber = firstMatch(text, [
|
||||
/\bPolicy\s+([0-9OIlSBD]{3,4}[-\s][0-9OIlSBD]{3}[-\s][0-9OIlSBD]{8}[-\s][0-9OIlSBD]{4}[-\s][0-9OIlSBD]{2})/i,
|
||||
/\bPolicy\s+([0-9OIlSBD][0-9OIlSBD\s-]{9,30})/,
|
||||
]);
|
||||
|
||||
// "Insured JON ASHLEY STRABALA" — label, then 1+ whitespace, then the name.
|
||||
// Names can carry accents (ÁVILA) or apostrophes (O'NEILL); the label is
|
||||
// always upper-case English on this layout, so case is reliable.
|
||||
const insuredName = labelValue(text, /^Insured\s+([A-ZÁÉÍÓÚÑ'][A-ZÁÉÍÓÚÑ '\-.]+)$/m);
|
||||
const additionalInsured = labelValue(text, /^Additional\s+insured\s+([A-ZÁÉÍÓÚÑ '\-.]+)$/m);
|
||||
|
||||
// Legal address is a single long line; the parser keeps it whole.
|
||||
const legalAddress = labelValue(text, /^Legal\s+address\s+(.+)$/m);
|
||||
const zip = labelValue(text, /^ZIP\s+(\d{4,6})\b/m);
|
||||
if (!zip && legalAddress) {
|
||||
// Last resort: zip often appears at the tail of the address run too
|
||||
// ("…C.P. 22550"). Cheap regex, no false-positive cost on this layout.
|
||||
const m = legalAddress.match(/\b(\d{5})\b/);
|
||||
if (m) notes.push(`ZIP leído de la dirección (${m[1]})`);
|
||||
}
|
||||
|
||||
// Broker line on GMX: "(1176) Jorge Humberto Cuadros" — the number is the
|
||||
// agent code, the name is what lands on `Policy.agentName`. The parens
|
||||
// are optional: a future layout or scan drop them.
|
||||
const brokerRaw = labelValue(text, /^Broker\s+(?:\(\d+\)\s*)?(.+)$/m);
|
||||
const agentName = brokerRaw?.trim() ?? null;
|
||||
|
||||
// Term: "12 months" — informational, not a free-standing date. Stored in
|
||||
// notes; the UI can derive `coveragePeriodDays` from From/To anyway.
|
||||
const term = firstMatch(text, [/^Term\s+(\d+\s+months?)$/m]);
|
||||
if (term) notes.push(`vigencia: ${term}`);
|
||||
|
||||
const policyFrom = parseDate(
|
||||
labelValue(text, /^From\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m),
|
||||
);
|
||||
const policyTo = parseDate(
|
||||
firstMatch(text, [/^To\s+(\d{1,2}\/\d{1,2}\/\d{4})\b/m]),
|
||||
);
|
||||
|
||||
// "at twelve hours (noon) Mexico City time." — kept in notes only.
|
||||
if (/twelve\s*hours|noon/i.test(text)) notes.push("vencimiento a las 12:00 hora del centro");
|
||||
|
||||
// The Currency / Premium payment cells sit next to each other on one
|
||||
// line; pull them with bounded matches so the trailing label of the
|
||||
// adjacent cell doesn't swallow the wrong value.
|
||||
const currency = currencyCode(labelValue(text, /^Currency\s+(\S+?)(?:\s+Premium\s+payment|$)/m));
|
||||
const premiumPayment = labelValue(text, /Premium\s+payment\s+(\S+)$/m);
|
||||
|
||||
// ----- signature date (page 1) -----------------------------------------
|
||||
// Appears above the signature line on its own: "July 23, 2026".
|
||||
const dateMatch = text.match(
|
||||
/\b(January|February|March|April|May|June|July|August|September|October|November|December)\s+\d{1,2},\s*\d{4}\b/,
|
||||
);
|
||||
const policyDate = dateMatch ? parseDate(dateMatch[0]) : null;
|
||||
if (!policyDate) notes.push("no se pudo leer la fecha de firma");
|
||||
|
||||
// ----- coverages table (page 2) -----------------------------------------
|
||||
const coverages = parseGmxCoverages(text, notes);
|
||||
|
||||
if (!policyNumber) notes.push("no se pudo leer el número de póliza");
|
||||
if (!policyFrom || !policyTo) notes.push("no se pudo leer el período de vigencia");
|
||||
// Premium fields are expected to be missing on the certificate page; flag
|
||||
// it explicitly so the reviewer knows to look for a separate receipt.
|
||||
if (!text.match(/Prima\s*neta|net\s*premium/i)) {
|
||||
notes.push("esta página no trae prima; revisar el recibo de GMX por separado");
|
||||
}
|
||||
|
||||
return {
|
||||
provider: "GMX",
|
||||
policyNumber: policyNumber ? policyNumber.replace(/\s+/g, "") : null,
|
||||
insuredName,
|
||||
additionalInsured,
|
||||
agentName,
|
||||
legalAddress,
|
||||
zip,
|
||||
policyFrom,
|
||||
policyTo,
|
||||
policyDate,
|
||||
currency,
|
||||
netPremium: null,
|
||||
policyFee: null,
|
||||
brokerFee: null,
|
||||
total: null,
|
||||
premiumPayment,
|
||||
coverages,
|
||||
notes,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the value that follows a `LABEL` on the same line. Used by every
|
||||
* "Label Value" cell on the GMX header table — matches on the line
|
||||
* itself rather than across the page, so a label that also appears in body
|
||||
* text can't accidentally claim a different cell.
|
||||
*/
|
||||
function labelValue(text: string, pattern: RegExp): string | null {
|
||||
const m = text.match(pattern);
|
||||
if (!m?.[1]) return null;
|
||||
return m[1].replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Walk the GMX per-coverage table on page 2.
|
||||
*
|
||||
* Real sample row (single-line representation of the table after pdftotext
|
||||
* flattens it; the real layout uses fixed columns):
|
||||
* "Building $350,000.00 Not applies Not applies"
|
||||
*
|
||||
* The four columns are:
|
||||
* Risk (left), Insured Amount ($ figure OR the word "Covered"),
|
||||
* Deductible (free text — "Not applies", "5%", "2% of the sum insured…"),
|
||||
* Loss Participation (same).
|
||||
*
|
||||
* "Covered" means the coverage is included with no dollar cap. We record
|
||||
* the word so the review queue surfaces it instead of inventing a number.
|
||||
*
|
||||
* Deductible / Loss Participation are kept as printed strings, not
|
||||
* converted to numbers — a "20%" loss participation is a different field
|
||||
* shape from a "$5,000" deductible and the JSON column lets the UI render
|
||||
* either verbatim.
|
||||
*
|
||||
* Multi-line cells (the "Earthquake" row's deductible wraps to three lines
|
||||
* because the column is narrow) are collapsed by joining consecutive
|
||||
* non-table-body lines onto the previous row's deductible cell before
|
||||
* applying the column regex.
|
||||
*/
|
||||
function parseGmxCoverages(text: string, notes: string[]): ParsedCoverage[] {
|
||||
const out: ParsedCoverage[] = [];
|
||||
|
||||
// Stop at "VALUES ADDED" — the trailing prose section (homeowner
|
||||
// services, legal text) is not a coverage table. Re-enter at
|
||||
// "ADDITIONAL RISK" for the second coverage block on page 2.
|
||||
const segments = text.split(/VALUES\s*ADDED/i)[0].split(/ADDITIONAL\s*RISK/i);
|
||||
|
||||
// `[ \t]` (not `\s`) inside a cell: the deductible/loss-participation
|
||||
// columns may wrap onto several lines in the raw `pdftotext` output, and
|
||||
// matching across newlines silently swallows the next row.
|
||||
const re = /^([A-Za-zÁÉÍÓÚÑ][A-Za-zÁÉÍÓÚÑ /\-.]+?)[ \t]+(\$[\d,.]+|Covered|Not[ \t]+applies)[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]+(\S+(?:[ \t]\S+){0,8})[ \t]*$/gim;
|
||||
let m: RegExpExecArray | null;
|
||||
for (const seg of segments) {
|
||||
re.lastIndex = 0;
|
||||
while ((m = re.exec(seg)) !== null) {
|
||||
const risk = m[1].trim();
|
||||
const amountCell = m[2].trim();
|
||||
const deductible = m[3].trim();
|
||||
const lossParticipation = m[4].trim();
|
||||
|
||||
// Skip the "Risk / Insured Amount / Deductible / Loss Participation"
|
||||
// header row itself, which matches the same regex.
|
||||
if (/^Risk$/i.test(risk) && /Insured\s*Amount/i.test(amountCell)) continue;
|
||||
|
||||
out.push({
|
||||
risk,
|
||||
insuredAmount:
|
||||
amountCell === "Covered" || amountCell === "Not applies"
|
||||
? null
|
||||
: money(amountCell),
|
||||
deductible,
|
||||
lossParticipation,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (out.length === 0) notes.push("no se encontraron coberturas en la tabla");
|
||||
return out;
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import { PrismaService } from "../prisma/prisma.service";
|
||||
import type { ParsedPolicy } from "./parsers/policy-parser";
|
||||
|
||||
export interface MatchResult {
|
||||
policyId: string | null;
|
||||
customerId: string | null;
|
||||
/** Why it landed here — shown in the review queue verbatim. */
|
||||
note: string;
|
||||
/** True only for an unambiguous hit on `Policy.policyNumber`. */
|
||||
confident: boolean;
|
||||
/**
|
||||
* Every policy that carries the parsed number, with its customer. >1 means
|
||||
* the policy number is shared across customers and a human must pick.
|
||||
*/
|
||||
candidates: { policyId: string; customerId: string; customerName: string; policyNumber: string }[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves a parsed policy page to an existing Policy (and its customer) the
|
||||
* office already holds.
|
||||
*
|
||||
* **Match on `Policy.policyNumber` alone, never on the printed insured name.**
|
||||
* The certificate's "Insured" line is the account's registrant, which drifts
|
||||
* from the current owner — the same problem the statement matcher cites for
|
||||
* utility bills ("ARNAIZ ROSAS ELSA AURORA" on a CESPT receipt for a
|
||||
* customer this office holds as "CATT, RANDY"). Names are surfaced for the
|
||||
* reviewer to sanity-check and never feed matching.
|
||||
*
|
||||
* A policy number that matches zero rows means the policy is new: the
|
||||
* review screen then offers a customer picker and the confirm step creates
|
||||
* the row. Multiple hits are surfaced rather than auto-picked — duplicate
|
||||
* policy numbers across customers do occur (same group policy bound by two
|
||||
* related parties), and picking one arbitrarily would silently book the
|
||||
* wrong coverage.
|
||||
*/
|
||||
@Injectable()
|
||||
export class PolicyMatcherService {
|
||||
constructor(private readonly prisma: PrismaService) {}
|
||||
|
||||
async match(parsed: ParsedPolicy): Promise<MatchResult> {
|
||||
if (!parsed.policyNumber) {
|
||||
return this.unmatched("no se pudo leer el número de póliza");
|
||||
}
|
||||
|
||||
const rows = await this.prisma.policy.findMany({
|
||||
where: { policyNumber: parsed.policyNumber },
|
||||
select: {
|
||||
id: true,
|
||||
policyNumber: true,
|
||||
customerId: true,
|
||||
customer: { select: { name: true } },
|
||||
},
|
||||
});
|
||||
|
||||
const candidates = rows.map((r) => ({
|
||||
policyId: r.id,
|
||||
customerId: r.customerId,
|
||||
customerName: r.customer.name,
|
||||
policyNumber: r.policyNumber,
|
||||
}));
|
||||
|
||||
if (rows.length === 0) {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note: `no se encontró ninguna póliza con el número ${parsed.policyNumber}`,
|
||||
confident: false,
|
||||
candidates: [],
|
||||
};
|
||||
}
|
||||
|
||||
if (rows.length > 1) {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note: `${rows.length} pólizas comparten el número ${parsed.policyNumber}`,
|
||||
confident: false,
|
||||
candidates,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
policyId: candidates[0].policyId,
|
||||
customerId: candidates[0].customerId,
|
||||
note: `coincidencia exacta por número de póliza ${parsed.policyNumber}`,
|
||||
confident: true,
|
||||
candidates,
|
||||
};
|
||||
}
|
||||
|
||||
private unmatched(note: string): MatchResult {
|
||||
return {
|
||||
policyId: null,
|
||||
customerId: null,
|
||||
note,
|
||||
confident: false,
|
||||
candidates: [],
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
import {
|
||||
Body,
|
||||
Controller,
|
||||
Get,
|
||||
Param,
|
||||
Patch,
|
||||
Post,
|
||||
Query,
|
||||
Req,
|
||||
Res,
|
||||
StreamableFile,
|
||||
UploadedFiles,
|
||||
UseGuards,
|
||||
UseInterceptors,
|
||||
} from "@nestjs/common";
|
||||
import { FilesInterceptor } from "@nestjs/platform-express";
|
||||
import type { Request, Response } from "express";
|
||||
import { AuthenticatedGuard } from "../auth/authenticated.guard";
|
||||
import { AbilityGuard } from "../auth/ability.guard";
|
||||
import { RequireAbility } from "../auth/require-ability.decorator";
|
||||
import { AuditService } from "../common/audit.service";
|
||||
import type { UploadedFileLike } from "../storage/upload-file";
|
||||
import { PolicyOcrService } from "./policy-ocr.service";
|
||||
import {
|
||||
ConfirmPolicyBatchDto,
|
||||
CreatePolicyOcrBatchDto,
|
||||
ReviewPolicyDocumentDto,
|
||||
} from "./policy-ocr.dto";
|
||||
|
||||
/**
|
||||
* Insurance OCR intake (policy_ocr_intake).
|
||||
*
|
||||
* Mirrors StatementsController shape: one batch = one upload session of
|
||||
* policy PDFs from a provider portal (GMX today), one document per page.
|
||||
* Confirming a batch delegates nothing to a separate billing path —
|
||||
* everything goes through `Policy` (and optionally a Transaction for the
|
||||
* premium), the same tables the manual `PolicyForm` writes.
|
||||
*/
|
||||
@Controller("policy-ocr")
|
||||
@UseGuards(AuthenticatedGuard, AbilityGuard)
|
||||
export class PolicyOcrController {
|
||||
constructor(
|
||||
private readonly policyOcr: PolicyOcrService,
|
||||
private readonly audit: AuditService,
|
||||
) {}
|
||||
|
||||
private actingId(req: Request): string {
|
||||
return (req.user as { id: string } | undefined)?.id ?? "";
|
||||
}
|
||||
|
||||
@Get("status")
|
||||
async status() {
|
||||
return {
|
||||
ocrAvailable: await this.policyOcr.ocrAvailable(),
|
||||
storageAvailable: this.policyOcr.storageAvailable(),
|
||||
};
|
||||
}
|
||||
|
||||
@Get("batches")
|
||||
listBatches(@Query("page") page?: string, @Query("pageSize") pageSize?: string) {
|
||||
return this.policyOcr.listBatches(
|
||||
Math.max(1, Number(page) || 1),
|
||||
Math.min(100, Math.max(1, Number(pageSize) || 25)),
|
||||
);
|
||||
}
|
||||
|
||||
@Get("batches/:id")
|
||||
getBatch(@Param("id") id: string) {
|
||||
return this.policyOcr.getBatch(id);
|
||||
}
|
||||
|
||||
@Get("batches/:id/documents")
|
||||
listDocuments(@Param("id") id: string) {
|
||||
return this.policyOcr.listDocuments(id);
|
||||
}
|
||||
|
||||
/**
|
||||
* The source PDF for a parsed policy document. One PDF = one parsed policy,
|
||||
* so this returns the entire upload (typically multi-page for insurance
|
||||
* certificates). The review screen embeds it in an iframe.
|
||||
*/
|
||||
@Get("documents/:id/page")
|
||||
async pageImage(
|
||||
@Param("id") id: string,
|
||||
@Res({ passthrough: true }) res: Response,
|
||||
) {
|
||||
const { stream, contentType, contentLength } = await this.policyOcr.pageImage(id);
|
||||
res.set({
|
||||
// The doc row stores the source PDF, not a rendered page image.
|
||||
"Content-Type": contentType ?? "application/pdf",
|
||||
...(contentLength ? { "Content-Length": String(contentLength) } : {}),
|
||||
});
|
||||
return new StreamableFile(stream);
|
||||
}
|
||||
|
||||
// --- writes ---------------------------------------------------------------
|
||||
|
||||
@Post("batches")
|
||||
@RequireAbility("policy:ingest")
|
||||
@UseInterceptors(
|
||||
FilesInterceptor("files", 25, { limits: { fileSize: 50 * 1024 * 1024 } }),
|
||||
)
|
||||
async createBatch(
|
||||
@UploadedFiles() files: UploadedFileLike[] | undefined,
|
||||
@Body() _dto: CreatePolicyOcrBatchDto,
|
||||
@Query("label") label: string | undefined,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const batch = await this.policyOcr.createBatch(
|
||||
files ?? [],
|
||||
this.actingId(req),
|
||||
label ?? _dto.label,
|
||||
);
|
||||
void this.audit.log(this.actingId(req), "policyOcr.batch.create", {
|
||||
batchId: batch.id,
|
||||
fileCount: batch.fileCount,
|
||||
});
|
||||
return batch;
|
||||
}
|
||||
|
||||
@Patch("documents/:id")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async review(
|
||||
@Param("id") id: string,
|
||||
@Body() dto: ReviewPolicyDocumentDto,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const doc = await this.policyOcr.review(id, dto, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.document.review", {
|
||||
documentId: id,
|
||||
status: doc.status,
|
||||
});
|
||||
return doc;
|
||||
}
|
||||
|
||||
@Post("documents/:id/reject")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async reject(@Param("id") id: string, @Req() req: Request) {
|
||||
const doc = await this.policyOcr.reject(id, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.document.reject", {
|
||||
documentId: id,
|
||||
});
|
||||
return doc;
|
||||
}
|
||||
|
||||
@Post("batches/:id/confirm")
|
||||
@RequireAbility("policy:ocr-review")
|
||||
async confirm(
|
||||
@Param("id") id: string,
|
||||
@Body() dto: ConfirmPolicyBatchDto,
|
||||
@Req() req: Request,
|
||||
) {
|
||||
const result = await this.policyOcr.confirmBatch(id, dto, this.actingId(req));
|
||||
void this.audit.log(this.actingId(req), "policyOcr.batch.confirm", {
|
||||
batchId: id,
|
||||
applied: result.applied,
|
||||
postedTransactions: result.postedTransactions,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
import { Type } from "class-transformer";
|
||||
import {
|
||||
IsArray,
|
||||
IsDateString,
|
||||
IsEnum,
|
||||
IsNumber,
|
||||
IsObject,
|
||||
IsOptional,
|
||||
IsString,
|
||||
MinLength,
|
||||
ValidateNested,
|
||||
} from "class-validator";
|
||||
|
||||
/** One document's confirmed-after-review state. The service reads these
|
||||
* fields and writes them onto either a matched Policy or a freshly created
|
||||
* one. Anything null here is not written. */
|
||||
export class ConfirmPolicyDocumentDto {
|
||||
@IsString() documentId!: string;
|
||||
|
||||
/** Required when creating a new Policy; ignored if `policyId` is set. */
|
||||
@IsOptional() @IsString() customerId?: string;
|
||||
/** Set when the document matched an existing Policy. */
|
||||
@IsOptional() @IsString() policyId?: string;
|
||||
|
||||
@IsOptional() @IsString() policyNumber?: string;
|
||||
@IsOptional() @IsString() insuredName?: string;
|
||||
@IsOptional() @IsString() additionalInsured?: string;
|
||||
@IsOptional() @IsString() agentName?: string;
|
||||
@IsOptional() @IsString() legalAddress?: string;
|
||||
@IsOptional() @IsString() zip?: string;
|
||||
@IsOptional() @IsDateString() policyFrom?: string;
|
||||
@IsOptional() @IsDateString() policyTo?: string;
|
||||
@IsOptional() @IsDateString() policyDate?: string;
|
||||
@IsOptional() @IsEnum(["MXN", "USD", "EUR"]) currency?: "MXN" | "USD" | "EUR";
|
||||
@IsOptional() @IsNumber() netPremium?: number;
|
||||
@IsOptional() @IsNumber() policyFee?: number;
|
||||
@IsOptional() @IsNumber() brokerFee?: number;
|
||||
@IsOptional() @IsNumber() total?: number;
|
||||
@IsOptional() @IsString() premiumPayment?: string;
|
||||
/** Coverages parsed off the PDF, passed through verbatim to Policy.coveragesJson. */
|
||||
@IsOptional() @IsObject() coveragesJson?: unknown;
|
||||
|
||||
/** When true, write a Transaction(domain=INSURANCE, amount=-netPremium)
|
||||
* in addition to creating/updating the Policy. Skipped if netPremium is
|
||||
* null or zero. */
|
||||
@IsOptional() postPremium?: boolean;
|
||||
}
|
||||
|
||||
export class ConfirmPolicyBatchDto {
|
||||
@IsArray()
|
||||
@ValidateNested({ each: true })
|
||||
@Type(() => ConfirmPolicyDocumentDto)
|
||||
documents!: ConfirmPolicyDocumentDto[];
|
||||
}
|
||||
|
||||
/** Staff correction of one document's extracted fields or its match. */
|
||||
export class ReviewPolicyDocumentDto {
|
||||
@IsOptional() @IsString() policyNumber?: string;
|
||||
@IsOptional() @IsString() insuredName?: string;
|
||||
@IsOptional() @IsString() additionalInsured?: string;
|
||||
@IsOptional() @IsString() agentName?: string;
|
||||
@IsOptional() @IsString() legalAddress?: string;
|
||||
@IsOptional() @IsString() zip?: string;
|
||||
@IsOptional() @IsDateString() policyFrom?: string;
|
||||
@IsOptional() @IsDateString() policyTo?: string;
|
||||
@IsOptional() @IsDateString() policyDate?: string;
|
||||
@IsOptional() @IsString() currency?: string;
|
||||
@IsOptional() @IsNumber() netPremium?: number;
|
||||
@IsOptional() @IsNumber() policyFee?: number;
|
||||
@IsOptional() @IsNumber() brokerFee?: number;
|
||||
@IsOptional() @IsNumber() total?: number;
|
||||
@IsOptional() @IsString() premiumPayment?: string;
|
||||
@IsOptional() @IsObject() coveragesJson?: unknown;
|
||||
|
||||
/** Set by the reviewer when the document matched an existing Policy. */
|
||||
@IsOptional() @IsString() matchedPolicyId?: string;
|
||||
/** Set by the reviewer when creating a new Policy. */
|
||||
@IsOptional() @IsString() matchedCustomerId?: string;
|
||||
/** Force-confirm a doc even when the matcher left it ambiguous. */
|
||||
@IsOptional() forceConfirm?: boolean;
|
||||
}
|
||||
|
||||
export class CreatePolicyOcrBatchDto {
|
||||
@IsOptional() @IsString() @MinLength(1) label?: string;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { OcrModule } from "../ocr/ocr.module";
|
||||
import { PolicyOcrController } from "./policy-ocr.controller";
|
||||
import { PolicyOcrService } from "./policy-ocr.service";
|
||||
import { PolicyMatcherService } from "./policy-matcher.service";
|
||||
|
||||
/**
|
||||
* Reuses the OCR seam from OcrModule unchanged: the Tesseract provider is
|
||||
* bound there and `OcrProvider` is the only thing the parsers touch. This
|
||||
* module registers its own controller + service + matcher; nothing about
|
||||
* utility ingestion needs to know about it.
|
||||
*/
|
||||
@Module({
|
||||
imports: [OcrModule],
|
||||
controllers: [PolicyOcrController],
|
||||
providers: [PolicyOcrService, PolicyMatcherService],
|
||||
})
|
||||
export class PolicyOcrModule {}
|
||||
@@ -0,0 +1,720 @@
|
||||
import {
|
||||
BadRequestException,
|
||||
Inject,
|
||||
Injectable,
|
||||
Logger,
|
||||
NotFoundException,
|
||||
} from "@nestjs/common";
|
||||
import { Currency, Prisma } from "@jorgecuadros/database";
|
||||
import { PrismaService } from "../prisma/prisma.service";
|
||||
import { StorageService } from "../storage/storage.service";
|
||||
import type { UploadedFileLike } from "../storage/upload-file";
|
||||
import { OCR_PROVIDER, type OcrPage, type OcrProvider } from "../statements/ocr/ocr.provider";
|
||||
import { parsePolicy } from "./parsers/policy-parser";
|
||||
import { PolicyMatcherService } from "./policy-matcher.service";
|
||||
import type {
|
||||
ConfirmPolicyBatchDto,
|
||||
ConfirmPolicyDocumentDto,
|
||||
ReviewPolicyDocumentDto,
|
||||
} from "./policy-ocr.dto";
|
||||
|
||||
/**
|
||||
* Insurance OCR intake — mirrors the statement pipeline at
|
||||
* `apps/api/src/statements/statements.service.ts`. Reuses the OCR seam and
|
||||
* Tesseract binding unchanged; the parsers and matcher are policy-specific.
|
||||
*
|
||||
* Why a parallel pipeline rather than a column on StatementDocument: the
|
||||
* matcher keys on `Policy.policyNumber`, the confirm step writes to a
|
||||
* different table (`Policy`, not `Transaction`), and the review UI shows
|
||||
* different fields. Sharing one queue would either bloat the row with null
|
||||
* columns or force the review screen to branch on a discriminator — both
|
||||
* worse than a thin second table.
|
||||
*/
|
||||
@Injectable()
|
||||
export class PolicyOcrService {
|
||||
private readonly logger = new Logger(PolicyOcrService.name);
|
||||
|
||||
constructor(
|
||||
private readonly prisma: PrismaService,
|
||||
private readonly storage: StorageService,
|
||||
private readonly matcher: PolicyMatcherService,
|
||||
@Inject(OCR_PROVIDER) private readonly ocr: OcrProvider,
|
||||
) {}
|
||||
|
||||
ocrAvailable(): Promise<boolean> {
|
||||
return this.ocr.available();
|
||||
}
|
||||
|
||||
storageAvailable(): boolean {
|
||||
return this.storage.available;
|
||||
}
|
||||
|
||||
// --- ingest ---------------------------------------------------------------
|
||||
|
||||
async createBatch(
|
||||
files: UploadedFileLike[],
|
||||
uploadedById: string,
|
||||
label?: string,
|
||||
) {
|
||||
if (!files?.length) throw new BadRequestException("No se recibió ningún archivo.");
|
||||
if (!(await this.ocr.available())) {
|
||||
throw new BadRequestException(
|
||||
"El servidor no tiene OCR instalado; no se pueden leer PDFs de pólizas.",
|
||||
);
|
||||
}
|
||||
if (!this.storage.available) {
|
||||
throw new BadRequestException(
|
||||
"El almacenamiento de documentos no está configurado; no se pueden " +
|
||||
"guardar los PDFs escaneados.",
|
||||
);
|
||||
}
|
||||
|
||||
const batch = await this.prisma.policyOcrBatch.create({
|
||||
data: { provider: "GMX", uploadedById, label, fileCount: files.length },
|
||||
});
|
||||
|
||||
const copies = files.map((f) => ({ buffer: f.buffer, name: f.originalname }));
|
||||
void this.process(batch.id, copies).catch(async (err) => {
|
||||
this.logger.error(`Policy OCR batch ${batch.id} failed: ${(err as Error).message}`);
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batch.id },
|
||||
data: { status: "FAILED", error: (err as Error).message },
|
||||
});
|
||||
});
|
||||
|
||||
return batch;
|
||||
}
|
||||
|
||||
/**
|
||||
* Render → text → parse → match, **one PolicyOcrDocument row per uploaded
|
||||
* file**. The GMX certificate is a 2-page PDF where page 1 carries the
|
||||
* contract header and page 2 carries the per-coverage table — both pages
|
||||
* describe the SAME policy, so the parser concatenates them and the
|
||||
* matcher runs once. `pageNumber` on the row is repurposed as the file
|
||||
* ordinal within the batch (1, 2, 3…) — the unique constraint
|
||||
* `(batchId, pageNumber)` still holds and lets a single batch carry many
|
||||
* policies.
|
||||
*
|
||||
* The doc's `storageKey` is the SOURCE PDF (`policy-ocr/{batchId}/source-N.pdf`)
|
||||
* rather than a rendered page image, so the review screen can embed the
|
||||
* exact artifact the office received. The rendered page PNGs are still
|
||||
* stored under `policy-ocr/{batchId}/page-M.png` for any future re-OCR or
|
||||
* image-based audit, but they aren't used as `storageKey` for the document.
|
||||
*/
|
||||
private async process(
|
||||
batchId: string,
|
||||
files: { buffer: Buffer; name?: string }[],
|
||||
) {
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "PROCESSING" },
|
||||
});
|
||||
|
||||
let fileOrdinal = 0;
|
||||
let globalPageOrdinal = 0;
|
||||
for (const file of files) {
|
||||
fileOrdinal += 1;
|
||||
const sourceKey = `policy-ocr/${batchId}/source-${fileOrdinal}.pdf`;
|
||||
await this.storage.put(sourceKey, file.buffer, "application/pdf");
|
||||
|
||||
const pages = await this.ocr.renderPages(file.buffer);
|
||||
const textLayer = await this.ocr.textPages(file.buffer).catch(() => []);
|
||||
|
||||
// One OcrPage per rendered page: text-layer wins when present (cheap,
|
||||
// exact), OCR the rendered image when it isn't. Same precedence rule
|
||||
// as the statement OCR pipeline.
|
||||
const perPageOcr: OcrPage[] = [];
|
||||
for (const [index, image] of pages.entries()) {
|
||||
globalPageOrdinal += 1;
|
||||
const pageStorageKey = `policy-ocr/${batchId}/page-${globalPageOrdinal}.png`;
|
||||
await this.storage.put(pageStorageKey, image, "image/png");
|
||||
|
||||
const embedded = textLayer[index] ?? null;
|
||||
const pageOcr = embedded ?? (await this.ocr.recognize(image));
|
||||
perPageOcr.push(pageOcr);
|
||||
}
|
||||
|
||||
// Concatenate every page's text with a blank line between pages so the
|
||||
// parser's anchored regexes (^From$, ^Currency\s+...) still work
|
||||
// across page boundaries — pdftotext -bbox-layout produces newline-
|
||||
// separated text per page already, the `\n\n` just preserves a clear
|
||||
// boundary in ocrRawText for debugging.
|
||||
const mergedText = perPageOcr.map((p) => p.text).join("\n\n");
|
||||
const avgConfidence =
|
||||
perPageOcr.length === 0
|
||||
? 0
|
||||
: perPageOcr.reduce((s, p) => s + p.confidence, 0) / perPageOcr.length;
|
||||
const synthetic: OcrPage = {
|
||||
text: mergedText,
|
||||
words: [],
|
||||
confidence: avgConfidence,
|
||||
};
|
||||
|
||||
try {
|
||||
const parsed = parsePolicy(synthetic);
|
||||
if (parsed.provider === "") {
|
||||
throw new Error("no se reconoció el proveedor");
|
||||
}
|
||||
const match = await this.matcher.match(parsed);
|
||||
const notes = [...parsed.notes, match.note].filter(Boolean);
|
||||
// Confident when exactly one Policy carries the printed number —
|
||||
// the only unambiguous hit we trust. A new policy (no match) still
|
||||
// needs a customer pick, so it stays in review.
|
||||
const trusted = match.confident && parsed.policyNumber != null;
|
||||
|
||||
await this.prisma.policyOcrDocument.create({
|
||||
data: {
|
||||
batchId,
|
||||
pageNumber: fileOrdinal,
|
||||
storageKey: sourceKey,
|
||||
status: trusted ? "MATCHED" : "NEEDS_REVIEW",
|
||||
ocrRawText: mergedText,
|
||||
ocrConfidence: new Prisma.Decimal(avgConfidence.toFixed(3)),
|
||||
provider: parsed.provider,
|
||||
extractedPolicyNumber: parsed.policyNumber,
|
||||
extractedInsuredName: parsed.insuredName,
|
||||
extractedAdditionalInsured: parsed.additionalInsured,
|
||||
extractedAgentName: parsed.agentName,
|
||||
extractedLegalAddress: parsed.legalAddress,
|
||||
extractedZip: parsed.zip,
|
||||
extractedPolicyFrom: parsed.policyFrom,
|
||||
extractedPolicyTo: parsed.policyTo,
|
||||
extractedPolicyDate: parsed.policyDate,
|
||||
extractedCurrency: parsed.currency,
|
||||
extractedNetPremium:
|
||||
parsed.netPremium != null ? new Prisma.Decimal(parsed.netPremium) : null,
|
||||
extractedPolicyFee:
|
||||
parsed.policyFee != null ? new Prisma.Decimal(parsed.policyFee) : null,
|
||||
extractedBrokerFee:
|
||||
parsed.brokerFee != null ? new Prisma.Decimal(parsed.brokerFee) : null,
|
||||
extractedTotal:
|
||||
parsed.total != null ? new Prisma.Decimal(parsed.total) : null,
|
||||
extractedCoveragesJson: parsed.coverages.length
|
||||
? (parsed.coverages as unknown as Prisma.InputJsonValue)
|
||||
: Prisma.DbNull,
|
||||
extractedPremiumPayment: parsed.premiumPayment,
|
||||
matchedPolicyId: match.policyId,
|
||||
matchedCustomerId: match.customerId,
|
||||
matchCandidates: match.candidates.length
|
||||
? (match.candidates as unknown as Prisma.InputJsonValue)
|
||||
: Prisma.DbNull,
|
||||
matchNote: notes.join("; ").slice(0, 190),
|
||||
},
|
||||
});
|
||||
} catch (err) {
|
||||
// The file as a whole failed to parse (no provider, parse exception).
|
||||
// One OCR_FAILED row per file is the right granularity — the page
|
||||
// images are still on disk for a re-run after a parser fix.
|
||||
await this.prisma.policyOcrDocument.create({
|
||||
data: {
|
||||
batchId,
|
||||
pageNumber: fileOrdinal,
|
||||
storageKey: sourceKey,
|
||||
status: "OCR_FAILED",
|
||||
matchNote: (err as Error).message.slice(0, 190),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "READY_FOR_REVIEW" },
|
||||
});
|
||||
}
|
||||
|
||||
// --- reads ----------------------------------------------------------------
|
||||
|
||||
async listBatches(page: number, pageSize: number) {
|
||||
const [total, items] = await this.prisma.$transaction([
|
||||
this.prisma.policyOcrBatch.count(),
|
||||
this.prisma.policyOcrBatch.findMany({
|
||||
orderBy: { createdAt: "desc" },
|
||||
skip: (page - 1) * pageSize,
|
||||
take: pageSize,
|
||||
include: {
|
||||
uploadedBy: { select: { name: true } },
|
||||
_count: { select: { documents: true } },
|
||||
},
|
||||
}),
|
||||
]);
|
||||
return { items, total, page, pageSize, pageCount: Math.ceil(total / pageSize) };
|
||||
}
|
||||
|
||||
async getBatch(id: string) {
|
||||
const batch = await this.prisma.policyOcrBatch.findUnique({
|
||||
where: { id },
|
||||
include: { uploadedBy: { select: { name: true } } },
|
||||
});
|
||||
if (!batch) throw new NotFoundException("Lote no encontrado.");
|
||||
|
||||
const counts = await this.prisma.policyOcrDocument.groupBy({
|
||||
by: ["status"],
|
||||
where: { batchId: id },
|
||||
_count: { _all: true },
|
||||
});
|
||||
return {
|
||||
...batch,
|
||||
byStatus: Object.fromEntries(counts.map((c) => [c.status, c._count._all])),
|
||||
};
|
||||
}
|
||||
|
||||
async listDocuments(batchId: string) {
|
||||
return this.prisma.policyOcrDocument.findMany({
|
||||
where: { batchId },
|
||||
orderBy: { pageNumber: "asc" },
|
||||
include: {
|
||||
matchedCustomer: { select: { id: true, name: true } },
|
||||
matchedPolicy: {
|
||||
select: {
|
||||
id: true,
|
||||
policyNumber: true,
|
||||
customerId: true,
|
||||
customer: { select: { name: true } },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* The source PDF for the document, so the review screen can show the
|
||||
* exact artifact the office uploaded (the browser's PDF viewer handles
|
||||
* scrolling, zoom, and selection natively). The rendered page PNGs
|
||||
* remain on disk under `policy-ocr/{batchId}/page-N.png` for any
|
||||
* future re-OCR, but the doc row points here at the source.
|
||||
*/
|
||||
async pageImage(documentId: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({
|
||||
where: { id: documentId },
|
||||
select: { storageKey: true },
|
||||
});
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
return this.storage.getStream(doc.storageKey);
|
||||
}
|
||||
|
||||
// --- review ---------------------------------------------------------------
|
||||
|
||||
async review(id: string, dto: ReviewPolicyDocumentDto, reviewedById: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException("Este documento ya fue aplicado.");
|
||||
}
|
||||
|
||||
// Trusting a customer-supplied pair (policyId, customerId) without
|
||||
// cross-check is how a document lands on the wrong customer's ledger;
|
||||
// pin them here from the DB.
|
||||
let matchedPolicyId = dto.matchedPolicyId ?? doc.matchedPolicyId;
|
||||
let matchedCustomerId = doc.matchedCustomerId;
|
||||
|
||||
if (matchedPolicyId) {
|
||||
const p = await this.prisma.policy.findUnique({
|
||||
where: { id: matchedPolicyId },
|
||||
select: { customerId: true },
|
||||
});
|
||||
if (!p) throw new BadRequestException("Póliza no encontrada.");
|
||||
matchedCustomerId = p.customerId;
|
||||
} else if (dto.matchedCustomerId) {
|
||||
const c = await this.prisma.customer.findUnique({
|
||||
where: { id: dto.matchedCustomerId },
|
||||
select: { id: true },
|
||||
});
|
||||
if (!c) throw new BadRequestException("Cliente no encontrado.");
|
||||
matchedCustomerId = c.id;
|
||||
}
|
||||
|
||||
return this.prisma.policyOcrDocument.update({
|
||||
where: { id },
|
||||
data: {
|
||||
extractedPolicyNumber: dto.policyNumber ?? undefined,
|
||||
extractedInsuredName: dto.insuredName ?? undefined,
|
||||
extractedAdditionalInsured: dto.additionalInsured ?? undefined,
|
||||
extractedAgentName: dto.agentName ?? undefined,
|
||||
extractedLegalAddress: dto.legalAddress ?? undefined,
|
||||
extractedZip: dto.zip ?? undefined,
|
||||
extractedPolicyFrom: dto.policyFrom ? new Date(dto.policyFrom) : undefined,
|
||||
extractedPolicyTo: dto.policyTo ? new Date(dto.policyTo) : undefined,
|
||||
extractedPolicyDate: dto.policyDate ? new Date(dto.policyDate) : undefined,
|
||||
extractedCurrency: dto.currency ?? undefined,
|
||||
extractedNetPremium:
|
||||
dto.netPremium != null ? new Prisma.Decimal(dto.netPremium) : undefined,
|
||||
extractedPolicyFee:
|
||||
dto.policyFee != null ? new Prisma.Decimal(dto.policyFee) : undefined,
|
||||
extractedBrokerFee:
|
||||
dto.brokerFee != null ? new Prisma.Decimal(dto.brokerFee) : undefined,
|
||||
extractedTotal:
|
||||
dto.total != null ? new Prisma.Decimal(dto.total) : undefined,
|
||||
extractedCoveragesJson: dto.coveragesJson
|
||||
? (dto.coveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
extractedPremiumPayment: dto.premiumPayment ?? undefined,
|
||||
matchedPolicyId,
|
||||
matchedCustomerId,
|
||||
status: dto.forceConfirm ? "CONFIRMED" : "MATCHED",
|
||||
reviewedById,
|
||||
reviewedAt: new Date(),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
async reject(id: string, reviewedById: string) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({ where: { id } });
|
||||
if (!doc) throw new NotFoundException("Documento no encontrado.");
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException("Este documento ya fue aplicado.");
|
||||
}
|
||||
return this.prisma.policyOcrDocument.update({
|
||||
where: { id },
|
||||
data: { status: "REJECTED", reviewedById, reviewedAt: new Date() },
|
||||
});
|
||||
}
|
||||
|
||||
// --- confirm --------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Apply every confirmed document: create or update the Policy, attach the
|
||||
* source PDF as a PolicyDocument, and (when staff asked + premium parses)
|
||||
* write a Transaction row. Each step is guarded by status checks so a
|
||||
* double-confirm cannot re-apply a document.
|
||||
*/
|
||||
async confirmBatch(batchId: string, dto: ConfirmPolicyBatchDto, reviewedById: string) {
|
||||
const batch = await this.prisma.policyOcrBatch.findUnique({ where: { id: batchId } });
|
||||
if (!batch) throw new NotFoundException("Lote no encontrado.");
|
||||
|
||||
const results: { documentId: string; policyId: string; postedTransactionId: string | null }[] = [];
|
||||
|
||||
for (const item of dto.documents) {
|
||||
const doc = await this.prisma.policyOcrDocument.findUnique({
|
||||
where: { id: item.documentId },
|
||||
});
|
||||
if (!doc) {
|
||||
throw new BadRequestException(`Documento ${item.documentId} no encontrado.`);
|
||||
}
|
||||
if (doc.status === "POSTED") {
|
||||
throw new BadRequestException(
|
||||
`El documento página ${doc.pageNumber} ya fue aplicado.`,
|
||||
);
|
||||
}
|
||||
if (!item.policyId && !item.customerId) {
|
||||
throw new BadRequestException(
|
||||
`Documento página ${doc.pageNumber}: falta póliza destino o cliente.`,
|
||||
);
|
||||
}
|
||||
|
||||
// 1. Resolve target Policy (create or update). Field selection: every
|
||||
// non-null `extracted*` on the doc (post-review) is written. Null is
|
||||
// preserved — never overwrite an existing Policy's `netPremium` with
|
||||
// null because the certificate page didn't carry one.
|
||||
let policyId = item.policyId ?? null;
|
||||
|
||||
if (policyId) {
|
||||
const updateData = buildPolicyUpdateFromDoc(item, doc);
|
||||
await this.prisma.policy.update({
|
||||
where: { id: policyId },
|
||||
data: updateData,
|
||||
});
|
||||
} else {
|
||||
// Create under the picked customer. `policyNumber` is the only field
|
||||
// that must be present.
|
||||
if (!item.policyNumber && !doc.extractedPolicyNumber) {
|
||||
throw new BadRequestException(
|
||||
`Documento página ${doc.pageNumber}: falta número de póliza.`,
|
||||
);
|
||||
}
|
||||
const createData = buildPolicyCreateFromDoc(item, doc, item.customerId!);
|
||||
const created = await this.prisma.policy.create({
|
||||
data: createData,
|
||||
});
|
||||
policyId = created.id;
|
||||
}
|
||||
|
||||
// 2. Attach the source PDF as a PolicyDocument. `doc.storageKey`
|
||||
// already points at the exact upload (`policy-ocr/{batchId}/source-N.pdf`)
|
||||
// so the attach is just a stream copy into the policy's namespace —
|
||||
// the previous per-page "which file did this page come from" walk is
|
||||
// gone because one PDF = one doc now.
|
||||
await this.attachSourcePdf(doc.storageKey, policyId);
|
||||
|
||||
// 3. Optionally post the premium to the ledger. Only when staff
|
||||
// explicitly asked (`postPremium` true) and netPremium parses — without
|
||||
// that gate a missing premium would silently book $0.
|
||||
let postedTransactionId: string | null = null;
|
||||
const premium =
|
||||
item.netPremium != null
|
||||
? item.netPremium
|
||||
: doc.extractedNetPremium != null
|
||||
? Number(doc.extractedNetPremium)
|
||||
: null;
|
||||
if (item.postPremium && premium && premium > 0) {
|
||||
const tx = await this.prisma.transaction.create({
|
||||
data: {
|
||||
customerId: (await this.policyCustomerId(policyId))!,
|
||||
domain: "INSURANCE",
|
||||
amount: new Prisma.Decimal(-Math.abs(premium)),
|
||||
transactionDate: doc.extractedPolicyDate ?? doc.extractedPolicyFrom ?? new Date(),
|
||||
currency: (item.currency ??
|
||||
doc.extractedCurrency ??
|
||||
"MXN") as Currency,
|
||||
reference: item.policyNumber ?? doc.extractedPolicyNumber ?? null,
|
||||
period: null,
|
||||
captureSource: "OCR",
|
||||
captureRef: doc.id,
|
||||
message: `Prima de póliza ${item.policyNumber ?? doc.extractedPolicyNumber ?? ""}`,
|
||||
},
|
||||
});
|
||||
postedTransactionId = tx.id;
|
||||
}
|
||||
|
||||
await this.prisma.policyOcrDocument.update({
|
||||
where: { id: doc.id },
|
||||
data: {
|
||||
status: "POSTED",
|
||||
matchedPolicyId: policyId,
|
||||
reviewedById,
|
||||
reviewedAt: new Date(),
|
||||
createdPolicyId: item.policyId ? null : policyId,
|
||||
postedTransactionId,
|
||||
},
|
||||
});
|
||||
|
||||
results.push({
|
||||
documentId: doc.id,
|
||||
policyId,
|
||||
postedTransactionId,
|
||||
});
|
||||
}
|
||||
|
||||
await this.closeIfDone(batchId);
|
||||
|
||||
return {
|
||||
applied: results.length,
|
||||
policies: results.map((r) => r.policyId),
|
||||
postedTransactions: results.filter((r) => r.postedTransactionId).length,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Stream the source PDF (`sourceKey`, set by `process` on the doc row)
|
||||
* into the policy's storage namespace and create a `PolicyDocument`
|
||||
* pointer. Trivial now that the doc row holds the exact source key —
|
||||
* the old per-page "which file did this page come from" walk is gone.
|
||||
*/
|
||||
private async attachSourcePdf(sourceKey: string, policyId: string): Promise<void> {
|
||||
const got = await this.storage.getStream(sourceKey);
|
||||
const chunks: Buffer[] = [];
|
||||
for await (const c of got.stream) chunks.push(c as Buffer);
|
||||
const buf = Buffer.concat(chunks);
|
||||
|
||||
const newKey = `policy/${policyId}/${Date.now()}-${crypto.randomUUID()}.pdf`;
|
||||
await this.storage.put(newKey, buf, "application/pdf");
|
||||
await this.prisma.policyDocument.create({
|
||||
data: {
|
||||
policyId,
|
||||
documentType: "GMX_POLICY",
|
||||
storageKey: newKey,
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
private async policyCustomerId(policyId: string): Promise<string | null> {
|
||||
const p = await this.prisma.policy.findUnique({
|
||||
where: { id: policyId },
|
||||
select: { customerId: true },
|
||||
});
|
||||
return p?.customerId ?? null;
|
||||
}
|
||||
|
||||
private async closeIfDone(batchId: string) {
|
||||
const open = await this.prisma.policyOcrDocument.count({
|
||||
where: {
|
||||
batchId,
|
||||
status: { in: ["PENDING_OCR", "NEEDS_REVIEW", "MATCHED", "CONFIRMED"] },
|
||||
},
|
||||
});
|
||||
if (open === 0) {
|
||||
await this.prisma.policyOcrBatch.update({
|
||||
where: { id: batchId },
|
||||
data: { status: "COMPLETED", completedAt: new Date() },
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Map a (post-review) doc + final confirmed fields onto a `Policy.update`
|
||||
* payload. Every field that is null in both inputs is omitted so we never
|
||||
* write null over a value the Policy already carries (the GMX certificate
|
||||
* has no premium — we must not blank the existing Policy.netPremium). */
|
||||
function buildPolicyUpdateFromDoc(
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
doc: {
|
||||
extractedPolicyNumber: string | null;
|
||||
extractedInsuredName: string | null;
|
||||
extractedAdditionalInsured: string | null;
|
||||
extractedAgentName: string | null;
|
||||
extractedLegalAddress: string | null;
|
||||
extractedZip: string | null;
|
||||
extractedPolicyFrom: Date | null;
|
||||
extractedPolicyTo: Date | null;
|
||||
extractedPolicyDate: Date | null;
|
||||
extractedCurrency: string | null;
|
||||
extractedNetPremium: Prisma.Decimal | null;
|
||||
extractedPolicyFee: Prisma.Decimal | null;
|
||||
extractedBrokerFee: Prisma.Decimal | null;
|
||||
extractedTotal: Prisma.Decimal | null;
|
||||
extractedCoveragesJson: Prisma.JsonValue | null;
|
||||
extractedPremiumPayment: string | null;
|
||||
},
|
||||
): Prisma.PolicyUpdateInput {
|
||||
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
|
||||
if (a != null) return new Prisma.Decimal(a);
|
||||
if (b != null) return b;
|
||||
return undefined;
|
||||
};
|
||||
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
|
||||
if (a) return new Date(a);
|
||||
if (b) return b;
|
||||
return undefined;
|
||||
};
|
||||
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
};
|
||||
|
||||
return {
|
||||
policyNumber: strOrUndef(item.policyNumber, doc.extractedPolicyNumber),
|
||||
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
|
||||
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
|
||||
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
|
||||
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
|
||||
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
|
||||
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
|
||||
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
|
||||
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
|
||||
total: numOrUndef(item.total, doc.extractedTotal),
|
||||
// coveragesJson / observations: freeform, keep the GMX data when present.
|
||||
coveragesJson:
|
||||
item.coveragesJson !== undefined
|
||||
? (item.coveragesJson as Prisma.InputJsonValue)
|
||||
: doc.extractedCoveragesJson != null
|
||||
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
// Premium payment cadence ("CONTADO") and insured-name fields land in
|
||||
// `observations` so the PolicyForm's edits stay the source of truth for
|
||||
// structured fields. The reviewer can move them by hand if needed.
|
||||
observations: joinObservations(
|
||||
doc.extractedInsuredName,
|
||||
doc.extractedAdditionalInsured,
|
||||
doc.extractedLegalAddress,
|
||||
doc.extractedZip,
|
||||
doc.extractedPremiumPayment,
|
||||
item,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
/** Same shape as `buildPolicyUpdateFromDoc`, but for `Policy.create`. The
|
||||
* `customerId` is supplied separately and `policyNumber` is required (a
|
||||
* Policy without a number can't be re-matched by the OCR pipeline). */
|
||||
function buildPolicyCreateFromDoc(
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
doc: {
|
||||
extractedPolicyNumber: string | null;
|
||||
extractedInsuredName: string | null;
|
||||
extractedAdditionalInsured: string | null;
|
||||
extractedAgentName: string | null;
|
||||
extractedLegalAddress: string | null;
|
||||
extractedZip: string | null;
|
||||
extractedPolicyFrom: Date | null;
|
||||
extractedPolicyTo: Date | null;
|
||||
extractedPolicyDate: Date | null;
|
||||
extractedCurrency: string | null;
|
||||
extractedNetPremium: Prisma.Decimal | null;
|
||||
extractedPolicyFee: Prisma.Decimal | null;
|
||||
extractedBrokerFee: Prisma.Decimal | null;
|
||||
extractedTotal: Prisma.Decimal | null;
|
||||
extractedCoveragesJson: Prisma.JsonValue | null;
|
||||
extractedPremiumPayment: string | null;
|
||||
},
|
||||
customerId: string,
|
||||
): Prisma.PolicyUncheckedCreateInput {
|
||||
const numOrUndef = (a: number | undefined, b: Prisma.Decimal | null): Prisma.Decimal | undefined => {
|
||||
if (a != null) return new Prisma.Decimal(a);
|
||||
if (b != null) return b;
|
||||
return undefined;
|
||||
};
|
||||
const dateOrUndef = (a: string | undefined, b: Date | null): Date | undefined => {
|
||||
if (a) return new Date(a);
|
||||
if (b) return b;
|
||||
return undefined;
|
||||
};
|
||||
const strOrUndef = (a: string | undefined, b: string | null): string | undefined => {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
};
|
||||
|
||||
const policyNumber =
|
||||
strOrUndef(item.policyNumber, doc.extractedPolicyNumber);
|
||||
if (!policyNumber) {
|
||||
// Caller already guards this; the throw is a type-narrowing aid.
|
||||
throw new Error("policyNumber required for create");
|
||||
}
|
||||
|
||||
return {
|
||||
policyNumber,
|
||||
customerId,
|
||||
agentName: strOrUndef(item.agentName, doc.extractedAgentName),
|
||||
policyFrom: dateOrUndef(item.policyFrom, doc.extractedPolicyFrom),
|
||||
policyTo: dateOrUndef(item.policyTo, doc.extractedPolicyTo),
|
||||
policyDate: dateOrUndef(item.policyDate, doc.extractedPolicyDate),
|
||||
currency: strOrUndef(item.currency, doc.extractedCurrency) as Currency | undefined,
|
||||
netPremium: numOrUndef(item.netPremium, doc.extractedNetPremium),
|
||||
policyFee: numOrUndef(item.policyFee, doc.extractedPolicyFee),
|
||||
brokerFee: numOrUndef(item.brokerFee, doc.extractedBrokerFee),
|
||||
total: numOrUndef(item.total, doc.extractedTotal),
|
||||
coveragesJson:
|
||||
item.coveragesJson !== undefined
|
||||
? (item.coveragesJson as Prisma.InputJsonValue)
|
||||
: doc.extractedCoveragesJson != null
|
||||
? (doc.extractedCoveragesJson as Prisma.InputJsonValue)
|
||||
: undefined,
|
||||
observations: joinObservations(
|
||||
doc.extractedInsuredName,
|
||||
doc.extractedAdditionalInsured,
|
||||
doc.extractedLegalAddress,
|
||||
doc.extractedZip,
|
||||
doc.extractedPremiumPayment,
|
||||
item,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
function joinObservations(
|
||||
insured: string | null,
|
||||
additional: string | null,
|
||||
address: string | null,
|
||||
zip: string | null,
|
||||
premiumPayment: string | null,
|
||||
item: ConfirmPolicyDocumentDto,
|
||||
): string | undefined {
|
||||
const lines: string[] = [];
|
||||
const insuredName = strOrUndefDb(item.insuredName, insured);
|
||||
if (insuredName) lines.push(`Asegurado: ${insuredName}`);
|
||||
const additionalInsured = strOrUndefDb(item.additionalInsured, additional);
|
||||
if (additionalInsured) lines.push(`Asegurado adicional: ${additionalInsured}`);
|
||||
const legalAddress = strOrUndefDb(item.legalAddress, address);
|
||||
if (legalAddress) lines.push(`Dirección: ${legalAddress}`);
|
||||
const zipVal = strOrUndefDb(item.zip, zip);
|
||||
if (zipVal) lines.push(`C.P.: ${zipVal}`);
|
||||
const cadence = strOrUndefDb(item.premiumPayment, premiumPayment);
|
||||
if (cadence) lines.push(`Pago de prima: ${cadence}`);
|
||||
return lines.length ? lines.join("\n") : undefined;
|
||||
}
|
||||
|
||||
function strOrUndefDb(a: string | undefined, b: string | null): string | undefined {
|
||||
if (a != null && a !== "") return a;
|
||||
if (b != null && b !== "") return b;
|
||||
return undefined;
|
||||
}
|
||||
@@ -1,23 +1,18 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { BillingModule } from "../billing/billing.module";
|
||||
import { OcrModule } from "../ocr/ocr.module";
|
||||
import { StatementsController } from "./statements.controller";
|
||||
import { StatementsService } from "./statements.service";
|
||||
import { StatementMatcherService } from "./statement-matcher.service";
|
||||
import { OCR_PROVIDER } from "./ocr/ocr.provider";
|
||||
import { TesseractOcrProvider } from "./ocr/tesseract.provider";
|
||||
|
||||
/**
|
||||
* The concrete OCR engine is bound here and nowhere else — everything
|
||||
* downstream depends on the OcrProvider interface, so swapping Tesseract for a
|
||||
* managed extraction API is a one-line change in this file.
|
||||
* The concrete OCR engine is bound in OcrModule (see apps/api/src/ocr/) —
|
||||
* everything downstream depends on the OcrProvider interface, so swapping
|
||||
* Tesseract for a managed extraction API is a one-line change there.
|
||||
*/
|
||||
@Module({
|
||||
imports: [BillingModule],
|
||||
imports: [BillingModule, OcrModule],
|
||||
controllers: [StatementsController],
|
||||
providers: [
|
||||
StatementsService,
|
||||
StatementMatcherService,
|
||||
{ provide: OCR_PROVIDER, useClass: TesseractOcrProvider },
|
||||
],
|
||||
providers: [StatementsService, StatementMatcherService],
|
||||
})
|
||||
export class StatementsModule {}
|
||||
|
||||
Reference in New Issue
Block a user