fix: locate patient row by real text position, drop unreliable AI row-verify

Row selection previously computed the target row's y arithmetically
(header.y + rowHeight * rowIndex), which drifted further off-row the
deeper the target patient was in the filtered list, and an AI
"onTarget" verification step added to catch drift instead confirmed a
click point that was visibly in blank space below the grid.

locateAllOnScreenshot now reports each text match's top/bottom edges
instead of a self-estimated center, and the patient row step locates
the target's first and last name directly, pairing whichever
occurrences land on the same row instead of trusting arithmetic or a
qualitative AI judgment call.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-29 21:53:12 -04:00
parent ad2e15ec62
commit 65dc795eb2
2 changed files with 74 additions and 31 deletions

View File

@@ -134,8 +134,10 @@ async function detectAndTrackNewWindow(ctx: RunContext, label: string): Promise<
// outside it (the main window's title bar, side panels) can ever be matched by mistake.
// 3. Click the Last Name field (found by vision, now unambiguous since the window is small
// and cropped) and type the first three letters — no first name is typed.
// 4. Find the topmost actual patient row in the now-filtered results grid and double-click
// it (opens Edit Appointment).
// 4. Locate the target patient's first and last name text directly in the now-filtered results
// grid (the 3-letter filter can return more than one patient) and pair up whichever
// occurrences share a row to find the target row's real position, then double-click it
// (opens Edit Appointment).
// 5. Diff again to find the Edit Appointment window's bounds, then click Exam and Save
// within it.
const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
@@ -189,7 +191,7 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
delayAfterMs: 200,
},
{
label: "Find and double-click the first patient row",
label: "Find and double-click the matching patient row",
execute: async (ctx) => {
const { region } = await captureWindowScreenshot(ctx);
backupTypeAgentScreenshot(ctx.runId, "patient_row_locate", region);
@@ -199,25 +201,49 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
if (headerMatches.length === 0) throw new Error('Could not find the "PatNum" column header');
const header = headerMatches.reduce((a, b) => (a.y < b.y ? a : b));
// A separate "find the first data row, not the header" locate call kept landing too close
// to (or on) the header — the two rows are visually similar and only ~15px apart. Since
// "PatNum" is an unambiguous anchor and the row height is measurable, the first row's
// position is computed directly (headerY + one row height) instead of asking the AI to
// visually tell two adjacent, similarly-styled rows apart a second time — nothing left to
// confuse once it's arithmetic on two already-known values.
// Locate the target patient's first and last name directly, rather than computing the row's
// y arithmetically (headerY + rowHeight * rowIndex) — a previous version did that and the
// click drifted further off-row the more rows down the target was, since any small error in
// the estimated row height got multiplied by the row index. Two real, independently-detected
// text positions and a same-row pairing between them is grounded in what's actually on
// screen instead of compounding an estimate.
const lastNameMatches = await locateAllOnScreenshot(ctx.userId, region, ctx.patientLastName);
const firstNameMatches = await locateAllOnScreenshot(ctx.userId, region, ctx.patientFirstName);
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "last name", matches: toScreenPoints(ctx, lastNameMatches) });
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "first name", matches: toScreenPoints(ctx, firstNameMatches) });
if (lastNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientLastName}" in the patient list`);
if (firstNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientFirstName}" in the patient list`);
// Row height is only used here as a "same row" tolerance for the pairing below, not
// multiplied by anything — so any imprecision in it no longer compounds with row distance.
const rowHeightRatio = await detectRowHeightRatio(ctx.userId, region);
const rowHeightPx = Math.round(rowHeightRatio * (ctx.windowBounds?.height ?? 0));
if (rowHeightPx <= 0) throw new Error(`AI reported an invalid row height ratio: ${rowHeightRatio}`);
const finalY = header.y + rowHeightPx;
const screen = toScreenPoint(ctx, { x: header.x, y: finalY });
let best: { last: { x: number; y: number }; first: { x: number; y: number }; dist: number } | null = null;
for (const last of lastNameMatches) {
for (const first of firstNameMatches) {
const dist = Math.abs(last.y - first.y);
if (dist <= rowHeightPx / 2 && (!best || dist < best.dist)) {
best = { last, first, dist };
}
}
}
if (!best) {
throw new Error(
`Could not find a row where "${ctx.patientFirstName}" and "${ctx.patientLastName}" are on the same line`
);
}
const screen = toScreenPoint(ctx, { x: header.x, y: best.last.y });
logTypeAgentStep(ctx.runId, {
event: "locate",
label: "first patient row",
label: "matching patient row",
pixelX: screen.x,
finalY: screen.y,
lastNameMatch: toScreenPoint(ctx, best.last),
firstNameMatch: toScreenPoint(ctx, best.first),
rowHeightPx,
headerY: toScreenPoint(ctx, { x: 0, y: header.y }).y,
});
ctx.beforeNextWindow = (await captureScreenshot(ctx.ip)).image;

View File

@@ -4,7 +4,7 @@ import { resolveAiProvider, getLlm } from "../ai/llm-factory";
// Side length (px) of the debug crop saved around a click point — big enough to show
// surrounding context (e.g. neighboring grid rows) when reviewing what a run actually clicked.
const DEBUG_CROP_SIZE = 220;
const DEBUG_CROP_SIZE = 440;
interface ParsedLocation {
found: boolean;
@@ -148,11 +148,17 @@ export async function cropAroundPoint(imageBase64: string, x: number, y: number)
return cropped.toString("base64");
}
// Parses {"matches": [{"xRatio":.., "yRatio":..}, ...]}, tolerating near-miss JSON the same
// way parseLocationResponse does — falls back to regex-extracting every xRatio/yRatio pair in
// the response if strict parsing fails, rather than treating a malformed-but-salvageable
interface RawMatch {
xRatio: number;
yTopRatio: number;
yBottomRatio: number;
}
// Parses {"matches": [{"xRatio":.., "yTopRatio":.., "yBottomRatio":..}, ...]}, tolerating
// near-miss JSON the same way parseLocationResponse does — falls back to regex-extracting every
// triple in the response if strict parsing fails, rather than treating a malformed-but-salvageable
// response as zero matches.
function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[] | null {
function parseMatchesResponse(raw: string): RawMatch[] | null {
const block = raw.match(/\{[\s\S]*\}/)?.[0];
if (!block) return null;
@@ -163,13 +169,14 @@ function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[]
// fall through to lenient repair below
}
const pairs: { xRatio: number; yRatio: number }[] = [];
const pairRegex = /"xRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yRatio"\s*:\s*(-?\d+(?:\.\d+)?)/g;
const triples: RawMatch[] = [];
const tripleRegex =
/"xRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yTopRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yBottomRatio"\s*:\s*(-?\d+(?:\.\d+)?)/g;
let match: RegExpExecArray | null;
while ((match = pairRegex.exec(block))) {
pairs.push({ xRatio: Number(match[1]), yRatio: Number(match[2]) });
while ((match = tripleRegex.exec(block))) {
triples.push({ xRatio: Number(match[1]), yTopRatio: Number(match[2]), yBottomRatio: Number(match[3]) });
}
return pairs.length > 0 ? pairs : null;
return triples.length > 0 ? triples : null;
}
// Like locateOnScreenshot, but finds every occurrence of an exact piece of text rather than
@@ -178,6 +185,11 @@ function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[]
// in the search box and once per matching patient row, so the caller pairs it against another
// text's occurrences (a first name) that share the same row instead of relying on a single
// fuzzy vision judgment call.
//
// Reports each match's TOP and BOTTOM pixel edge rather than asking the model to self-estimate a
// "center" — a center is an abstract judgment call, whereas the top/bottom of a glyph is a
// concrete, visible thing to point at. The vertical midpoint used for x/y below is then computed
// here in code from those two real edges, not trusted as a direct AI estimate.
export async function locateAllOnScreenshot(
userId: number,
imageBase64: string,
@@ -202,9 +214,12 @@ export async function locateAllOnScreenshot(
`This is a screenshot of a Windows desktop application. Find EVERY occurrence of the text ` +
`"${text}" visible anywhere in this screenshot (case-insensitive) — there may be zero, one, ` +
"or several. Respond with strict JSON only, no prose, no markdown fences: " +
'{"matches": [{"xRatio": <0 to 1>, "yRatio": <0 to 1>}, ...]} — one entry per occurrence, ' +
"at the center of that occurrence's text, using the same fraction-of-image-width/height " +
"convention as before. Use an empty array if there are no occurrences.",
'{"matches": [{"xRatio": <0 to 1>, "yTopRatio": <0 to 1>, "yBottomRatio": <0 to 1>}, ...]} — ' +
"one entry per occurrence. xRatio is how far across the image the occurrence's horizontal " +
"center is, as a fraction of the TOTAL image width. yTopRatio is where the TOP edge of that " +
"occurrence's text (the top of its tallest letters) is, and yBottomRatio is where its BOTTOM " +
"edge (the bottom of its lowest letters, including descenders) is — both as a fraction of " +
"the TOTAL image height. Use an empty array if there are no occurrences.",
},
{
type: "image_url",
@@ -226,7 +241,7 @@ export async function locateAllOnScreenshot(
}
return matches.map((m) => ({
x: Math.round(m.xRatio * width),
y: Math.round(m.yRatio * height),
y: Math.round(((m.yTopRatio + m.yBottomRatio) / 2) * height),
}));
}
@@ -253,10 +268,12 @@ export async function detectRowHeightRatio(userId: number, imageBase64: string):
{
type: "text",
text:
"This is a screenshot containing a results-grid list of rows (e.g. a patient list), each " +
"row the same height. Measure the height of a single row. Respond with strict JSON only, " +
'no prose, no markdown fences: {"rowHeightRatio": <0 to 1>} — the height of one row as a ' +
"fraction of the TOTAL image height. Use two decimal places of precision.",
"This is a screenshot containing a results-grid list of rows (e.g. a patient list) with a " +
"header row followed by data rows, all data rows the same height. IGNORE the header row — " +
"it is taller than a data row and would skew the measurement. Look at ALL the visible data " +
"rows (not just one) and measure their AVERAGE height. Respond with strict JSON only, no " +
'prose, no markdown fences: {"rowHeightRatio": <0 to 1>} — the average data row height as ' +
"a fraction of the TOTAL image height. Use three decimal places of precision.",
},
{
type: "image_url",