feat(statements): OCR intake for scanned utility bills
Staff key 300+ utility statements per company per month by hand. This adds the ingest -> split -> OCR -> match -> review pipeline that proposes customer and amount per page instead (RECEIPT_CAPTURE_SPEC §2), posting through the existing BillingService.createBatch seam with source=OCR and a per-document captureRef so machine and hand capture share one write path and audit trail. Everything was designed against 10 real scanned statements (46 pages of CFE, CESPT and Telnor bills) rather than from the sample-free spec. The scans have no text layer at all — they are camera images — so OCR is mandatory, and they arrive bundled one customer per page. Measured on those pages the parser identifies the provider 46/46 and reads an account reference 43/46; against the dev database that is 39/46 (85%) exact auto-match, 40/46 identified, with the rest genuine review cases. That closes the OCR-provider question in favour of self-hosted Tesseract: it clears the bar for a queue where a human confirms every row, and OcrProvider keeps a managed API a one-line swap. The samples corrected three things the spec had wrong or unknown: - Clave catastral is NOT predial. DATMEX.clave (934 rows) is what CESPT and predial bills print; DATMEX.predial, which PROPERTY_TAX.accountNumber holds, has 663 distinct values across 1135 rows and appears on no statement. The clave now lives on Property.cadastralKey as the matcher's secondary key; predial is left untouched. This had been blocking predial matching. - Gas was recoverable: 160 of 334 DATMEX.gas values are real account numbers (the rest are ESTACIONARIO/CILINDRO descriptors), now in GAS.meterNumber. - Phone is one billed line per property (534/18/1 across phone1/2/3), so the new TELEPHONE ServiceKind backfills from phone1 only, not three rows. Matching is scoped to one column per service kind and never reads the customer name — a CESPT receipt prints ARNAIZ ROSAS ELSA AURORA for an account this office holds under CATT, RANDY, because the printed name is the registrant, not the current owner. Where a provider prints a payment barcode it beats the printed label (one CFE label OCR'd a digit too many while its barcode was correct) and the two cross-check, with disagreement forcing review. Confirming a document whose service had no reference writes it back, so gas and any other cold start is a one-time cost rather than a permanent queue. Verified end to end against the live dev API and MinIO: real scans uploaded over HTTP, matched, confirmed against a check, and the resulting rows checked in MySQL (negative amounts, captureSource=OCR, concept derived from the batch kind, captureRef linking back to each page). Re-confirming a posted batch is refused. Test data was removed afterwards. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,53 @@
|
||||
/**
|
||||
* The OCR seam. Everything above this interface works in terms of page text and
|
||||
* word boxes, so the concrete engine is swappable without touching the parsers,
|
||||
* the matcher, or the schema.
|
||||
*
|
||||
* The shipped implementation is self-hosted Tesseract (see tesseract.provider).
|
||||
* That choice is evidence-based rather than assumed: run against 46 pages of
|
||||
* real scanned CFE, CESPT and Telnor statements, it identified the provider on
|
||||
* 46/46 and extracted a usable account reference on 43/46, which is well past
|
||||
* the bar for a queue whose whole point is that a human confirms every row. A
|
||||
* managed document-extraction API (Textract, Document Intelligence, Document
|
||||
* AI) fits behind this same interface if per-page accuracy ever proves
|
||||
* insufficient, with no schema change — but at 300+ pages/month/company it
|
||||
* would carry a real recurring cost for accuracy that is not currently the
|
||||
* bottleneck.
|
||||
*/
|
||||
|
||||
/** One OCR'd word, with where it sits on the page. */
|
||||
export interface OcrWord {
|
||||
text: string;
|
||||
/** Pixel box in the rendered page image. */
|
||||
left: number;
|
||||
top: number;
|
||||
width: number;
|
||||
height: number;
|
||||
/** Engine confidence for this word, 0..1. */
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export interface OcrPage {
|
||||
/** Full page text, reading order, newline-separated. */
|
||||
text: string;
|
||||
/**
|
||||
* Word boxes. Needed because two of the three real layouts are *tables* —
|
||||
* the CESPT "RECIBO" prints `No. DE CUENTA` as a column header with the
|
||||
* value in the row beneath it, which line-oriented text cannot associate.
|
||||
* Parsers fall back to geometry for exactly those fields.
|
||||
*/
|
||||
words: OcrWord[];
|
||||
/** Mean word confidence across the page, 0..1. */
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export interface OcrProvider {
|
||||
/** True when the engine is actually usable in this deployment. */
|
||||
available(): Promise<boolean>;
|
||||
/** Split a PDF into one rendered page image per page. */
|
||||
renderPages(pdf: Buffer): Promise<Buffer[]>;
|
||||
/** OCR a single rendered page image. */
|
||||
recognize(pageImage: Buffer): Promise<OcrPage>;
|
||||
}
|
||||
|
||||
export const OCR_PROVIDER = Symbol("OCR_PROVIDER");
|
||||
@@ -0,0 +1,195 @@
|
||||
import { Injectable, Logger, ServiceUnavailableException } from "@nestjs/common";
|
||||
import { ConfigService } from "@nestjs/config";
|
||||
import { execFile } from "node:child_process";
|
||||
import { mkdtemp, readFile, readdir, rm, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { promisify } from "node:util";
|
||||
import type { OcrPage, OcrProvider, OcrWord } from "./ocr.provider";
|
||||
|
||||
const run = promisify(execFile);
|
||||
|
||||
/**
|
||||
* Self-hosted OCR: `pdftoppm` (poppler) to rasterise, `tesseract` to read.
|
||||
*
|
||||
* Both are external binaries rather than a native npm addon, which keeps the
|
||||
* pnpm workspace free of a compiled dependency and makes the alpine runtime
|
||||
* image a two-package change (see docker/api.Dockerfile). Like StorageService,
|
||||
* a missing binary degrades rather than crashes the API: the module reports
|
||||
* itself unavailable and statement ingest returns 503, while every other
|
||||
* feature keeps working.
|
||||
*
|
||||
* The settings below are not arbitrary — they were measured against the real
|
||||
* scanned samples:
|
||||
* - 300 DPI grayscale. The source scans are phone photos of paper at ~5MB a
|
||||
* page; below 300 the small print (RMU, clave catastral) stops resolving,
|
||||
* above it costs time for no additional fields.
|
||||
* - `--psm 6` ("assume a single uniform block of text"). The default page
|
||||
* segmentation splits these dense forms into columns and interleaves them,
|
||||
* which destroys the label-then-value adjacency every parser depends on.
|
||||
* - Spanish traineddata, with a graceful fall back to English if the language
|
||||
* pack is absent — an accented label reads worse but the digits, which are
|
||||
* what actually gets matched, are unaffected.
|
||||
*/
|
||||
@Injectable()
|
||||
export class TesseractOcrProvider implements OcrProvider {
|
||||
private readonly logger = new Logger(TesseractOcrProvider.name);
|
||||
private readonly dpi: number;
|
||||
private readonly lang: string;
|
||||
private probe: Promise<boolean> | null = null;
|
||||
|
||||
constructor(config: ConfigService) {
|
||||
this.dpi = Number(config.get("OCR_DPI") ?? 300);
|
||||
this.lang = config.get<string>("OCR_LANG") ?? "spa";
|
||||
}
|
||||
|
||||
/** Cached — the binaries do not appear or vanish while the process runs. */
|
||||
available(): Promise<boolean> {
|
||||
if (!this.probe) {
|
||||
this.probe = (async () => {
|
||||
try {
|
||||
await Promise.all([
|
||||
run("tesseract", ["--version"]),
|
||||
run("pdftoppm", ["-v"]),
|
||||
]);
|
||||
return true;
|
||||
} catch {
|
||||
this.logger.warn(
|
||||
"OCR unavailable: `tesseract` and/or `pdftoppm` not found on PATH. " +
|
||||
"Statement ingest is disabled; every other feature is unaffected.",
|
||||
);
|
||||
return false;
|
||||
}
|
||||
})();
|
||||
}
|
||||
return this.probe;
|
||||
}
|
||||
|
||||
private async require(): Promise<void> {
|
||||
if (!(await this.available())) {
|
||||
throw new ServiceUnavailableException(
|
||||
"El servicio de OCR no está disponible en este servidor.",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
private async scratch<T>(fn: (dir: string) => Promise<T>): Promise<T> {
|
||||
const dir = await mkdtemp(join(tmpdir(), "stmt-ocr-"));
|
||||
try {
|
||||
return await fn(dir);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
async renderPages(pdf: Buffer): Promise<Buffer[]> {
|
||||
await this.require();
|
||||
return this.scratch(async (dir) => {
|
||||
const src = join(dir, "in.pdf");
|
||||
await writeFile(src, pdf);
|
||||
// -gray: these are grayscale scans already; colour triples the bytes
|
||||
// handed to tesseract for no gain in character recognition.
|
||||
await run("pdftoppm", [
|
||||
"-r",
|
||||
String(this.dpi),
|
||||
"-gray",
|
||||
"-png",
|
||||
src,
|
||||
join(dir, "page"),
|
||||
]);
|
||||
const files = (await readdir(dir))
|
||||
.filter((f) => f.startsWith("page") && f.endsWith(".png"))
|
||||
// pdftoppm zero-pads its page numbers, so lexical order is page order.
|
||||
.sort();
|
||||
return Promise.all(files.map((f) => readFile(join(dir, f))));
|
||||
});
|
||||
}
|
||||
|
||||
async recognize(pageImage: Buffer): Promise<OcrPage> {
|
||||
await this.require();
|
||||
return this.scratch(async (dir) => {
|
||||
const img = join(dir, "page.png");
|
||||
await writeFile(img, pageImage);
|
||||
|
||||
// One tesseract invocation produces both outputs; TSV carries the word
|
||||
// boxes and per-word confidence, and its text can be reassembled into
|
||||
// reading order, so there is no need to run the engine twice.
|
||||
const out = join(dir, "out");
|
||||
try {
|
||||
await run("tesseract", [img, out, "-l", this.lang, "--psm", "6", "tsv"]);
|
||||
} catch (err) {
|
||||
if (this.lang !== "eng") {
|
||||
this.logger.warn(
|
||||
`Tesseract failed with lang "${this.lang}", retrying with "eng": ${
|
||||
(err as Error).message
|
||||
}`,
|
||||
);
|
||||
await run("tesseract", [img, out, "-l", "eng", "--psm", "6", "tsv"]);
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
const tsv = await readFile(`${out}.tsv`, "utf8");
|
||||
return parseTsv(tsv);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn tesseract's TSV into words plus reassembled text.
|
||||
*
|
||||
* Columns are: level, page_num, block_num, par_num, line_num, word_num, left,
|
||||
* top, width, height, conf, text. Rows with level < 5 are structural (page,
|
||||
* block, paragraph, line) and carry no text; only level 5 is a word. A conf of
|
||||
* -1 marks a structural row, so those are dropped rather than averaged in —
|
||||
* including them would drag every page's confidence toward zero.
|
||||
*/
|
||||
export function parseTsv(tsv: string): OcrPage {
|
||||
const lines = tsv.split("\n");
|
||||
const header = lines[0]?.split("\t") ?? [];
|
||||
const col = (name: string) => header.indexOf(name);
|
||||
const iLeft = col("left");
|
||||
const iTop = col("top");
|
||||
const iWidth = col("width");
|
||||
const iHeight = col("height");
|
||||
const iConf = col("conf");
|
||||
const iText = col("text");
|
||||
const iLine = col("line_num");
|
||||
const iBlock = col("block_num");
|
||||
|
||||
const words: OcrWord[] = [];
|
||||
// Keyed by block+line so the reassembled text preserves the engine's own
|
||||
// reading order instead of sorting words by raw y, which interleaves columns.
|
||||
const byLine = new Map<string, string[]>();
|
||||
|
||||
for (let i = 1; i < lines.length; i++) {
|
||||
const f = lines[i].split("\t");
|
||||
if (f.length <= iText) continue;
|
||||
const text = f[iText]?.trim();
|
||||
if (!text) continue;
|
||||
const confidence = Number(f[iConf]);
|
||||
if (!Number.isFinite(confidence) || confidence < 0) continue;
|
||||
|
||||
words.push({
|
||||
text,
|
||||
left: Number(f[iLeft]) || 0,
|
||||
top: Number(f[iTop]) || 0,
|
||||
width: Number(f[iWidth]) || 0,
|
||||
height: Number(f[iHeight]) || 0,
|
||||
confidence: confidence / 100,
|
||||
});
|
||||
|
||||
const key = `${f[iBlock]}:${f[iLine]}`;
|
||||
const bucket = byLine.get(key);
|
||||
if (bucket) bucket.push(text);
|
||||
else byLine.set(key, [text]);
|
||||
}
|
||||
|
||||
const text = [...byLine.values()].map((w) => w.join(" ")).join("\n");
|
||||
const confidence = words.length
|
||||
? words.reduce((sum, w) => sum + w.confidence, 0) / words.length
|
||||
: 0;
|
||||
|
||||
return { text, words, confidence };
|
||||
}
|
||||
Reference in New Issue
Block a user