docs+perf: document Google Vision key setup, consolidate OCR calls per screenshot

- README: step-by-step Google Cloud Vision API key creation and setup
  (enable API, create service-account key, place as google_credentials.json)
- ocrLocate: split into getOcrWords()/matchWordsForText() so a screenshot is
  OCR'd once and matched locally against multiple text targets, instead of a
  separate billed Vision API call per target (patient-row and Exam steps each
  cut from 2-3 calls down to 1)
This commit is contained in:
2026-07-30 23:27:33 -04:00
parent 5cdd3452f2
commit 9efa8f6e0a
3 changed files with 100 additions and 26 deletions
+25 -13
View File
@@ -1,7 +1,7 @@
import axios from "axios";
import FormData from "form-data";
interface OcrWord {
export interface OcrWord {
text: string;
left: number;
top: number;
@@ -11,23 +11,19 @@ interface OcrWord {
cy: number;
}
// Finds every occurrence of exact text in a screenshot using real OCR (Google Vision via
// PaymentOCRService, already running for payment-document extraction) instead of asking an AI
// to visually estimate coordinates. Google Vision's word bounding boxes give an exact pixel
// center, so unlike locateAllOnScreenshot's AI-estimated top/bottom edges, there's no systematic
// bias to correct for.
//
// `text` may be multiple words (e.g. "Single click") — Google Vision reports one bounding box
// per individual word, not per phrase, so a multi-word target is matched as a run of consecutive
// words in the OCR result (which preserves reading order) and their boxes are merged into one.
export async function locateAllViaOcr(imageBase64: string, text: string): Promise<{ x: number; y: number }[]> {
// Runs OCR (Google Vision via PaymentOCRService, already running for payment-document
// extraction) on a screenshot ONCE and returns every detected word with its exact pixel bounding
// box. Callers that need to look up several different pieces of text in the same screenshot
// (e.g. a header, a last name, and a first name all in one patient-list crop) should call this
// once and match against the result multiple times with matchWordsForText, rather than paying
// for a separate billed Vision API call per text target.
export async function getOcrWords(imageBase64: string): Promise<OcrWord[]> {
const form = new FormData();
form.append("file", Buffer.from(imageBase64, "base64"), {
filename: "screenshot.png",
contentType: "image/png",
});
let words: OcrWord[];
try {
const resp = await axios.post<{ words: OcrWord[] }>("http://localhost:5003/extract/words", form, {
headers: form.getHeaders(),
@@ -35,13 +31,22 @@ export async function locateAllViaOcr(imageBase64: string, text: string): Promis
maxContentLength: Infinity,
timeout: 30000,
});
words = resp.data?.words ?? [];
return resp.data?.words ?? [];
} catch (err: any) {
const status = err?.response?.status;
const detail = err?.response?.data?.detail || err?.message || "Unknown error";
throw new Error(`OCR request failed${status ? ` (${status})` : ""}: ${detail}`);
}
}
// Finds every occurrence of exact text within an already-OCR'd word list. Google Vision's word
// bounding boxes give an exact pixel center, so unlike locateAllOnScreenshot's AI-estimated
// top/bottom edges, there's no systematic bias to correct for.
//
// `text` may be multiple words (e.g. "Single click") — Google Vision reports one bounding box
// per individual word, not per phrase, so a multi-word target is matched as a run of consecutive
// words in the OCR result (which preserves reading order) and their boxes are merged into one.
export function matchWordsForText(words: OcrWord[], text: string): { x: number; y: number }[] {
const targetTokens = text.trim().toLowerCase().split(/\s+/).filter(Boolean);
if (targetTokens.length === 0) return [];
@@ -59,3 +64,10 @@ export async function locateAllViaOcr(imageBase64: string, text: string): Promis
}
return matches;
}
// Convenience wrapper for callers that only need a single text lookup per screenshot (one OCR
// call, one match pass) — e.g. the Save button step, which only ever searches for one thing.
export async function locateAllViaOcr(imageBase64: string, text: string): Promise<{ x: number; y: number }[]> {
const words = await getOcrWords(imageBase64);
return matchWordsForText(words, text);
}
+14 -6
View File
@@ -9,7 +9,7 @@ import {
locateOnScreenshot,
WindowBounds,
} from "./visionLocate";
import { locateAllViaOcr } from "./ocrLocate";
import { getOcrWords, locateAllViaOcr, matchWordsForText } from "./ocrLocate";
import { backupTypeAgentScreenshot, logTypeAgentStep } from "../utils/screenshotBackup";
export type StepStatus = "running" | "done" | "error";
@@ -195,7 +195,12 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
const { region } = await captureWindowScreenshot(ctx);
backupTypeAgentScreenshot(ctx.runId, "patient_row_locate", region);
const headerMatches = await locateAllViaOcr(region, "PatNum");
// One OCR call for the whole region, matched against locally for each of the three text
// targets below — Vision bills per call, and all three targets live in this same
// screenshot, so there's no reason to pay for three separate OCR passes over it.
const ocrWords = await getOcrWords(region);
const headerMatches = matchWordsForText(ocrWords, "PatNum");
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "PatNum header", matches: toScreenPoints(ctx, headerMatches) });
if (headerMatches.length === 0) throw new Error('Could not find the "PatNum" column header');
const header = headerMatches.reduce((a, b) => (a.y < b.y ? a : b));
@@ -207,8 +212,8 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
// Claude to estimate each word's top/bottom edge) also had a small but consistent upward
// bias since it was an LLM's visual estimate, not a measurement. OCR bounding-box centers
// are exact, so the same-row pairing below needs no bias correction.
const lastNameMatches = await locateAllViaOcr(region, ctx.patientLastName);
const firstNameMatches = await locateAllViaOcr(region, ctx.patientFirstName);
const lastNameMatches = matchWordsForText(ocrWords, ctx.patientLastName);
const firstNameMatches = matchWordsForText(ocrWords, ctx.patientFirstName);
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "last name", matches: toScreenPoints(ctx, lastNameMatches) });
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "first name", matches: toScreenPoints(ctx, firstNameMatches) });
if (lastNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientLastName}" in the patient list`);
@@ -274,8 +279,11 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
// 2, hard-failing the whole step). "Single click" is a hint label that only ever appears
// once, directly above the correct clickable list — so the "Exam" occurrence nearest it in
// x is structurally the right one, without needing to know where the panel boundaries are.
const examMatches = await locateAllViaOcr(region, "Exam");
const singleClickMatches = await locateAllViaOcr(region, "Single click");
// One OCR call for the whole region, matched locally against both text targets — same
// reasoning as the patient-row step above.
const ocrWords = await getOcrWords(region);
const examMatches = matchWordsForText(ocrWords, "Exam");
const singleClickMatches = matchWordsForText(ocrWords, "Single click");
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Exam", matches: toScreenPoints(ctx, examMatches) });
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Single click", matches: toScreenPoints(ctx, singleClickMatches) });