phase-3: ingestion, from a QR in the camera to a card in the bandeja
The whole pipeline: storage behind one driver interface (local disk and S3), a portable job queue with a poller, QR and CDC parsing, OCR through the Anthropic API, dedupe, manual entry, and the bandeja that turns all of it into one decision per card. Scanning tries the trustworthy door first: a QR is parsed and prefilled from its CDC; a photo without one is queued for OCR when a key is configured and otherwise opens the manual form against the stored file. A QR that will not parse records an ingest error and falls through rather than losing the photo. Job claiming is the only dialect divergence, as SPEC allows: FOR UPDATE SKIP LOCKED on Postgres, a conditional UPDATE against SQLite's single writer. Retry backoff follows SPEC exactly and a job abandoned by a killed process returns to the queue once its lock goes stale, which is the phase 3 acceptance case. OCR uses structured outputs rather than parsing prose, so the model cannot return anything but the RULES.md schema, and every field is nullable because unreadable is a real answer. Web: scan with live QR decoding (BarcodeDetector, ZXing fallback, wasm served from our own origin), manual entry with the IVA split worked out from the total, the bandeja with swipe, buttons and keyboard all doing the same thing, and the documents list and detail with an editable classification. The seed now carries Maria's 34 purchases and 8 sales and Carlos's 6, all classified through the real rules, plus the two open ingest errors. Two defects found and fixed with tests: seeded documents could be dated in the future, which would corrupt any projection computed from them, and the category buttons announced their keyboard shortcut as part of their name. 255 vitest tests, 37 Playwright tests, rules coverage still 100%, typecheck and lint clean. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
0d7651b17c
commit
b074456b70
@@ -0,0 +1,126 @@
|
||||
import Anthropic from '@anthropic-ai/sdk';
|
||||
import { zodOutputFormat } from '@anthropic-ai/sdk/helpers/zod';
|
||||
import { z } from 'zod';
|
||||
import type { Env } from '../../lib/env';
|
||||
|
||||
/**
|
||||
* RULES.md section 9. The model returns this and nothing else: structured outputs
|
||||
* constrain the response to the schema, so there is no prose to strip and no JSON to
|
||||
* repair. Every field is nullable, because "unreadable" is a real answer and guessing a
|
||||
* number onto a tax document is the one thing this must never do.
|
||||
*/
|
||||
export const OcrExtraction = z.object({
|
||||
emitter_ruc: z.string().nullable(),
|
||||
emitter_dv: z.string().nullable(),
|
||||
emitter_name: z.string().nullable(),
|
||||
receiver_doc: z.string().nullable(),
|
||||
doc_number: z.string().nullable(),
|
||||
issue_date: z.string().nullable(),
|
||||
total: z.number().int().nullable(),
|
||||
amount_iva10: z.number().int().nullable(),
|
||||
amount_iva5: z.number().int().nullable(),
|
||||
amount_exenta: z.number().int().nullable(),
|
||||
iva10: z.number().int().nullable(),
|
||||
iva5: z.number().int().nullable(),
|
||||
confidence: z.record(z.string(), z.number()),
|
||||
});
|
||||
export type OcrExtraction = z.infer<typeof OcrExtraction>;
|
||||
|
||||
export interface OcrProvider {
|
||||
extract(args: { data: Uint8Array; mime: string }): Promise<OcrExtraction>;
|
||||
}
|
||||
|
||||
const SYSTEM_PROMPT = `Sos un extractor de datos de comprobantes fiscales paraguayos (facturas, autofacturas, notas de credito y debito).
|
||||
|
||||
Leé la imagen y devolvé unicamente los campos del esquema.
|
||||
|
||||
Reglas:
|
||||
- Las facturas paraguayas imprimen las columnas de IVA como "10%", "5%" y "Exentas".
|
||||
amount_iva10, amount_iva5 y amount_exenta son las bases gravadas de cada columna;
|
||||
iva10 e iva5 son los impuestos liquidados de cada una.
|
||||
- Los importes se imprimen con punto como separador de miles y no llevan decimales.
|
||||
Devolvelos como enteros sin separadores: "1.234.567" es 1234567.
|
||||
- issue_date en formato YYYY-MM-DD.
|
||||
- emitter_ruc sin el digito verificador; emitter_dv es ese digito.
|
||||
- Si un campo no se lee con claridad, devolvé null. Nunca adivines un numero.
|
||||
- confidence lleva una entrada por campo que sí leiste, de 0 a 1.`;
|
||||
|
||||
/** Guaranies have no cents, so a component sum may legitimately differ by rounding. */
|
||||
const TOTAL_TOLERANCE_GS = 1;
|
||||
/** RULES.md section 9: money fields drop to this when the components do not add up. */
|
||||
const MISMATCH_CONFIDENCE = 0.5;
|
||||
|
||||
/**
|
||||
* Returns null when no key is configured. Every caller treats that as "OCR is off" and
|
||||
* falls back to manual entry rather than failing the scan (SPEC.md section 8).
|
||||
*/
|
||||
export function createOcrProvider(env: Env): OcrProvider | null {
|
||||
if (!env.ANTHROPIC_API_KEY) return null;
|
||||
|
||||
const client = new Anthropic({ apiKey: env.ANTHROPIC_API_KEY });
|
||||
|
||||
return {
|
||||
async extract({ data, mime }) {
|
||||
const message = await client.messages.parse({
|
||||
model: env.OCR_MODEL,
|
||||
max_tokens: 4096,
|
||||
system: SYSTEM_PROMPT,
|
||||
output_config: { format: zodOutputFormat(OcrExtraction) },
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
content: [
|
||||
{
|
||||
type: 'image',
|
||||
source: {
|
||||
type: 'base64',
|
||||
media_type: imageMediaType(mime),
|
||||
data: Buffer.from(data).toString('base64'),
|
||||
},
|
||||
},
|
||||
{ type: 'text', text: 'Extraé los datos de este comprobante.' },
|
||||
],
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
const parsed = message.parsed_output;
|
||||
if (!parsed) throw new Error('OCR returned no parseable output');
|
||||
return lowerConfidenceOnMismatch(OcrExtraction.parse(parsed));
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* RULES.md section 9: when the total and the components are both present and disagree,
|
||||
* the money fields are not trusted, whatever the model said about them.
|
||||
*/
|
||||
export function lowerConfidenceOnMismatch(extraction: OcrExtraction): OcrExtraction {
|
||||
const { total, amount_iva10, amount_iva5, amount_exenta, iva10, iva5 } = extraction;
|
||||
const components = [amount_iva10, amount_iva5, amount_exenta, iva10, iva5];
|
||||
if (total === null || components.some((value) => value === null)) return extraction;
|
||||
|
||||
const derived =
|
||||
(amount_iva10 ?? 0) + (amount_iva5 ?? 0) + (amount_exenta ?? 0) + (iva10 ?? 0) + (iva5 ?? 0);
|
||||
if (Math.abs(total - derived) <= TOTAL_TOLERANCE_GS) return extraction;
|
||||
|
||||
const confidence = { ...extraction.confidence };
|
||||
for (const field of ['total', 'amount_iva10', 'amount_iva5', 'amount_exenta', 'iva10', 'iva5']) {
|
||||
if (field in confidence) confidence[field] = Math.min(confidence[field] ?? 1, MISMATCH_CONFIDENCE);
|
||||
else confidence[field] = MISMATCH_CONFIDENCE;
|
||||
}
|
||||
return { ...extraction, confidence };
|
||||
}
|
||||
|
||||
type ImageMediaType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/webp';
|
||||
|
||||
function imageMediaType(mime: string): ImageMediaType {
|
||||
switch (mime) {
|
||||
case 'image/png':
|
||||
case 'image/gif':
|
||||
case 'image/webp':
|
||||
return mime;
|
||||
default:
|
||||
return 'image/jpeg';
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user