docs+perf: document Google Vision key setup, consolidate OCR calls per screenshot

- README: step-by-step Google Cloud Vision API key creation and setup
  (enable API, create service-account key, place as google_credentials.json)
- ocrLocate: split into getOcrWords()/matchWordsForText() so a screenshot is
  OCR'd once and matched locally against multiple text targets, instead of a
  separate billed Vision API call per target (patient-row and Exam steps each
  cut from 2-3 calls down to 1)
This commit is contained in:
2026-07-30 23:27:33 -04:00
parent 5cdd3452f2
commit 9efa8f6e0a
3 changed files with 100 additions and 26 deletions

View File

@@ -1,7 +1,7 @@
import axios from "axios";
import FormData from "form-data";
interface OcrWord {
export interface OcrWord {
text: string;
left: number;
top: number;
@@ -11,23 +11,19 @@ interface OcrWord {
cy: number;
}
// Finds every occurrence of exact text in a screenshot using real OCR (Google Vision via
// PaymentOCRService, already running for payment-document extraction) instead of asking an AI
// to visually estimate coordinates. Google Vision's word bounding boxes give an exact pixel
// center, so unlike locateAllOnScreenshot's AI-estimated top/bottom edges, there's no systematic
// bias to correct for.
//
// `text` may be multiple words (e.g. "Single click") — Google Vision reports one bounding box
// per individual word, not per phrase, so a multi-word target is matched as a run of consecutive
// words in the OCR result (which preserves reading order) and their boxes are merged into one.
export async function locateAllViaOcr(imageBase64: string, text: string): Promise<{ x: number; y: number }[]> {
// Runs OCR (Google Vision via PaymentOCRService, already running for payment-document
// extraction) on a screenshot ONCE and returns every detected word with its exact pixel bounding
// box. Callers that need to look up several different pieces of text in the same screenshot
// (e.g. a header, a last name, and a first name all in one patient-list crop) should call this
// once and match against the result multiple times with matchWordsForText, rather than paying
// for a separate billed Vision API call per text target.
export async function getOcrWords(imageBase64: string): Promise<OcrWord[]> {
const form = new FormData();
form.append("file", Buffer.from(imageBase64, "base64"), {
filename: "screenshot.png",
contentType: "image/png",
});
let words: OcrWord[];
try {
const resp = await axios.post<{ words: OcrWord[] }>("http://localhost:5003/extract/words", form, {
headers: form.getHeaders(),
@@ -35,13 +31,22 @@ export async function locateAllViaOcr(imageBase64: string, text: string): Promis
maxContentLength: Infinity,
timeout: 30000,
});
words = resp.data?.words ?? [];
return resp.data?.words ?? [];
} catch (err: any) {
const status = err?.response?.status;
const detail = err?.response?.data?.detail || err?.message || "Unknown error";
throw new Error(`OCR request failed${status ? ` (${status})` : ""}: ${detail}`);
}
}
// Finds every occurrence of exact text within an already-OCR'd word list. Google Vision's word
// bounding boxes give an exact pixel center, so unlike locateAllOnScreenshot's AI-estimated
// top/bottom edges, there's no systematic bias to correct for.
//
// `text` may be multiple words (e.g. "Single click") — Google Vision reports one bounding box
// per individual word, not per phrase, so a multi-word target is matched as a run of consecutive
// words in the OCR result (which preserves reading order) and their boxes are merged into one.
export function matchWordsForText(words: OcrWord[], text: string): { x: number; y: number }[] {
const targetTokens = text.trim().toLowerCase().split(/\s+/).filter(Boolean);
if (targetTokens.length === 0) return [];
@@ -59,3 +64,10 @@ export async function locateAllViaOcr(imageBase64: string, text: string): Promis
}
return matches;
}
// Convenience wrapper for callers that only need a single text lookup per screenshot (one OCR
// call, one match pass) — e.g. the Save button step, which only ever searches for one thing.
export async function locateAllViaOcr(imageBase64: string, text: string): Promise<{ x: number; y: number }[]> {
const words = await getOcrWords(imageBase64);
return matchWordsForText(words, text);
}

View File

@@ -9,7 +9,7 @@ import {
locateOnScreenshot,
WindowBounds,
} from "./visionLocate";
import { locateAllViaOcr } from "./ocrLocate";
import { getOcrWords, locateAllViaOcr, matchWordsForText } from "./ocrLocate";
import { backupTypeAgentScreenshot, logTypeAgentStep } from "../utils/screenshotBackup";
export type StepStatus = "running" | "done" | "error";
@@ -195,7 +195,12 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
const { region } = await captureWindowScreenshot(ctx);
backupTypeAgentScreenshot(ctx.runId, "patient_row_locate", region);
const headerMatches = await locateAllViaOcr(region, "PatNum");
// One OCR call for the whole region, matched against locally for each of the three text
// targets below — Vision bills per call, and all three targets live in this same
// screenshot, so there's no reason to pay for three separate OCR passes over it.
const ocrWords = await getOcrWords(region);
const headerMatches = matchWordsForText(ocrWords, "PatNum");
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "PatNum header", matches: toScreenPoints(ctx, headerMatches) });
if (headerMatches.length === 0) throw new Error('Could not find the "PatNum" column header');
const header = headerMatches.reduce((a, b) => (a.y < b.y ? a : b));
@@ -207,8 +212,8 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
// Claude to estimate each word's top/bottom edge) also had a small but consistent upward
// bias since it was an LLM's visual estimate, not a measurement. OCR bounding-box centers
// are exact, so the same-row pairing below needs no bias correction.
const lastNameMatches = await locateAllViaOcr(region, ctx.patientLastName);
const firstNameMatches = await locateAllViaOcr(region, ctx.patientFirstName);
const lastNameMatches = matchWordsForText(ocrWords, ctx.patientLastName);
const firstNameMatches = matchWordsForText(ocrWords, ctx.patientFirstName);
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "last name", matches: toScreenPoints(ctx, lastNameMatches) });
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "first name", matches: toScreenPoints(ctx, firstNameMatches) });
if (lastNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientLastName}" in the patient list`);
@@ -274,8 +279,11 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
// 2, hard-failing the whole step). "Single click" is a hint label that only ever appears
// once, directly above the correct clickable list — so the "Exam" occurrence nearest it in
// x is structurally the right one, without needing to know where the panel boundaries are.
const examMatches = await locateAllViaOcr(region, "Exam");
const singleClickMatches = await locateAllViaOcr(region, "Single click");
// One OCR call for the whole region, matched locally against both text targets — same
// reasoning as the patient-row step above.
const ocrWords = await getOcrWords(region);
const examMatches = matchWordsForText(ocrWords, "Exam");
const singleClickMatches = matchWordsForText(ocrWords, "Single click");
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Exam", matches: toScreenPoints(ctx, examMatches) });
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Single click", matches: toScreenPoints(ctx, singleClickMatches) });

View File

@@ -1,13 +1,67 @@
# Medical Billing OCR API (FastAPI)
# Payment OCR Service (FastAPI)
FastAPI wrapper around a Google Cloud Vision OCR pipeline. Used by the Backend for
payment/EOB document extraction (`/extract/json`, `/extract/csv`, `/extract/pdf/json`)
and by the Windows Type Agent for exact-pixel text location (`/extract/words`).
## 1) Prereqs
- Google Cloud Vision service-account JSON.
- `GOOGLE_APPLICATION_CREDENTIALS` env var pointing to that JSON.
- Tesseract installed (for fallback OCR), and on PATH.
## 2) Install & run (local)
- Python 3.
- A Google Cloud Vision service-account key (see step 2 below).
## 2) Create a Google Cloud Vision API key
1. Go to [console.cloud.google.com](https://console.cloud.google.com) and select (or
create) the project this service should use.
2. **APIs & Services → Library** → search for **"Cloud Vision API"** → click **Enable**
(skip if already enabled).
3. **IAM & Admin → Service Accounts** → either pick an existing service account for
this app, or **Create Service Account** (any name, e.g. `ocr-service`; no special
roles are required beyond default — Vision API access comes from the API being
enabled on the project, not a role grant).
4. Open that service account → **Keys** tab → **Add Key → Create new key → JSON**.
This immediately downloads a `.json` file to your browser's Downloads folder —
**this is the only time the private key content is shown**, so keep the file safe
(a password manager or secure backup, not just Downloads).
## 3) Install the key
1. Move (don't just copy, to avoid leaving stray copies around) the downloaded JSON
file into this folder (`apps/PaymentOCRService/`).
2. Rename it to exactly `google_credentials.json` — this is the filename `.env`
already expects:
```bash
mv ~/Downloads/<your-downloaded-file>.json apps/PaymentOCRService/google_credentials.json
```
3. This filename is gitignored on purpose — **never commit it**. If a key is ever
accidentally exposed (committed, pasted, screenshotted), go back to the Keys tab
in step 2 and delete it, then generate a new one.
## 4) Install & run (local)
```bash
python -m venv .venv && source .venv/bin/activate
pip install -r requirements.txt
export GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/service-account.json
uvicorn app.main:app --reload --port 8080
uvicorn main:app --host 0.0.0.0 --port 5003
```
Or just `python main.py` — it reads `HOST`/`PORT` from `.env` (defaults to
`0.0.0.0:5003`) and `GOOGLE_APPLICATION_CREDENTIALS=google_credentials.json` from the
same file.
Verify it's working:
```bash
curl localhost:5003/health
# should report "GOOGLE_APPLICATION_CREDENTIALS set: True"
```
## 5) Endpoints
- `POST /extract/json`, `/extract/csv`, `/extract/csvtext` — payment/EOB document
images → structured rows (deskew + line-grouping + domain extraction pipeline).
- `POST /extract/pdf/json` — same, for remittance-advice PDFs.
- `POST /extract/words` — raw OCR only: a flat list of every detected word with its
exact pixel bounding box (`left`, `top`, `w`, `h`, `cx`, `cy`). Used by the Windows
Type Agent to click exact text locations instead of relying on an AI-vision
estimate.
- `GET /health`, `GET /status` — liveness/credential check, active/queued job counts.