diff --git a/apps/Backend/src/services/typeAgentRunner.ts b/apps/Backend/src/services/typeAgentRunner.ts index 3c2e1a1a..8b6755f8 100644 --- a/apps/Backend/src/services/typeAgentRunner.ts +++ b/apps/Backend/src/services/typeAgentRunner.ts @@ -134,8 +134,10 @@ async function detectAndTrackNewWindow(ctx: RunContext, label: string): Promise< // outside it (the main window's title bar, side panels) can ever be matched by mistake. // 3. Click the Last Name field (found by vision, now unambiguous since the window is small // and cropped) and type the first three letters — no first name is typed. -// 4. Find the topmost actual patient row in the now-filtered results grid and double-click -// it (opens Edit Appointment). +// 4. Locate the target patient's first and last name text directly in the now-filtered results +// grid (the 3-letter filter can return more than one patient) and pair up whichever +// occurrences share a row to find the target row's real position, then double-click it +// (opens Edit Appointment). // 5. Diff again to find the Edit Appointment window's bounds, then click Exam and Save // within it. const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [ @@ -189,7 +191,7 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [ delayAfterMs: 200, }, { - label: "Find and double-click the first patient row", + label: "Find and double-click the matching patient row", execute: async (ctx) => { const { region } = await captureWindowScreenshot(ctx); backupTypeAgentScreenshot(ctx.runId, "patient_row_locate", region); @@ -199,25 +201,49 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [ if (headerMatches.length === 0) throw new Error('Could not find the "PatNum" column header'); const header = headerMatches.reduce((a, b) => (a.y < b.y ? a : b)); - // A separate "find the first data row, not the header" locate call kept landing too close - // to (or on) the header — the two rows are visually similar and only ~15px apart. Since - // "PatNum" is an unambiguous anchor and the row height is measurable, the first row's - // position is computed directly (headerY + one row height) instead of asking the AI to - // visually tell two adjacent, similarly-styled rows apart a second time — nothing left to - // confuse once it's arithmetic on two already-known values. + // Locate the target patient's first and last name directly, rather than computing the row's + // y arithmetically (headerY + rowHeight * rowIndex) — a previous version did that and the + // click drifted further off-row the more rows down the target was, since any small error in + // the estimated row height got multiplied by the row index. Two real, independently-detected + // text positions and a same-row pairing between them is grounded in what's actually on + // screen instead of compounding an estimate. + const lastNameMatches = await locateAllOnScreenshot(ctx.userId, region, ctx.patientLastName); + const firstNameMatches = await locateAllOnScreenshot(ctx.userId, region, ctx.patientFirstName); + logTypeAgentStep(ctx.runId, { event: "locate_all", label: "last name", matches: toScreenPoints(ctx, lastNameMatches) }); + logTypeAgentStep(ctx.runId, { event: "locate_all", label: "first name", matches: toScreenPoints(ctx, firstNameMatches) }); + if (lastNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientLastName}" in the patient list`); + if (firstNameMatches.length === 0) throw new Error(`Could not find "${ctx.patientFirstName}" in the patient list`); + + // Row height is only used here as a "same row" tolerance for the pairing below, not + // multiplied by anything — so any imprecision in it no longer compounds with row distance. const rowHeightRatio = await detectRowHeightRatio(ctx.userId, region); const rowHeightPx = Math.round(rowHeightRatio * (ctx.windowBounds?.height ?? 0)); if (rowHeightPx <= 0) throw new Error(`AI reported an invalid row height ratio: ${rowHeightRatio}`); - const finalY = header.y + rowHeightPx; - const screen = toScreenPoint(ctx, { x: header.x, y: finalY }); + let best: { last: { x: number; y: number }; first: { x: number; y: number }; dist: number } | null = null; + for (const last of lastNameMatches) { + for (const first of firstNameMatches) { + const dist = Math.abs(last.y - first.y); + if (dist <= rowHeightPx / 2 && (!best || dist < best.dist)) { + best = { last, first, dist }; + } + } + } + if (!best) { + throw new Error( + `Could not find a row where "${ctx.patientFirstName}" and "${ctx.patientLastName}" are on the same line` + ); + } + + const screen = toScreenPoint(ctx, { x: header.x, y: best.last.y }); logTypeAgentStep(ctx.runId, { event: "locate", - label: "first patient row", + label: "matching patient row", pixelX: screen.x, finalY: screen.y, + lastNameMatch: toScreenPoint(ctx, best.last), + firstNameMatch: toScreenPoint(ctx, best.first), rowHeightPx, - headerY: toScreenPoint(ctx, { x: 0, y: header.y }).y, }); ctx.beforeNextWindow = (await captureScreenshot(ctx.ip)).image; diff --git a/apps/Backend/src/services/visionLocate.ts b/apps/Backend/src/services/visionLocate.ts index 7cd2154c..6e7027d6 100644 --- a/apps/Backend/src/services/visionLocate.ts +++ b/apps/Backend/src/services/visionLocate.ts @@ -4,7 +4,7 @@ import { resolveAiProvider, getLlm } from "../ai/llm-factory"; // Side length (px) of the debug crop saved around a click point — big enough to show // surrounding context (e.g. neighboring grid rows) when reviewing what a run actually clicked. -const DEBUG_CROP_SIZE = 220; +const DEBUG_CROP_SIZE = 440; interface ParsedLocation { found: boolean; @@ -148,11 +148,17 @@ export async function cropAroundPoint(imageBase64: string, x: number, y: number) return cropped.toString("base64"); } -// Parses {"matches": [{"xRatio":.., "yRatio":..}, ...]}, tolerating near-miss JSON the same -// way parseLocationResponse does — falls back to regex-extracting every xRatio/yRatio pair in -// the response if strict parsing fails, rather than treating a malformed-but-salvageable +interface RawMatch { + xRatio: number; + yTopRatio: number; + yBottomRatio: number; +} + +// Parses {"matches": [{"xRatio":.., "yTopRatio":.., "yBottomRatio":..}, ...]}, tolerating +// near-miss JSON the same way parseLocationResponse does — falls back to regex-extracting every +// triple in the response if strict parsing fails, rather than treating a malformed-but-salvageable // response as zero matches. -function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[] | null { +function parseMatchesResponse(raw: string): RawMatch[] | null { const block = raw.match(/\{[\s\S]*\}/)?.[0]; if (!block) return null; @@ -163,13 +169,14 @@ function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[] // fall through to lenient repair below } - const pairs: { xRatio: number; yRatio: number }[] = []; - const pairRegex = /"xRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yRatio"\s*:\s*(-?\d+(?:\.\d+)?)/g; + const triples: RawMatch[] = []; + const tripleRegex = + /"xRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yTopRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yBottomRatio"\s*:\s*(-?\d+(?:\.\d+)?)/g; let match: RegExpExecArray | null; - while ((match = pairRegex.exec(block))) { - pairs.push({ xRatio: Number(match[1]), yRatio: Number(match[2]) }); + while ((match = tripleRegex.exec(block))) { + triples.push({ xRatio: Number(match[1]), yTopRatio: Number(match[2]), yBottomRatio: Number(match[3]) }); } - return pairs.length > 0 ? pairs : null; + return triples.length > 0 ? triples : null; } // Like locateOnScreenshot, but finds every occurrence of an exact piece of text rather than @@ -178,6 +185,11 @@ function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[] // in the search box and once per matching patient row, so the caller pairs it against another // text's occurrences (a first name) that share the same row instead of relying on a single // fuzzy vision judgment call. +// +// Reports each match's TOP and BOTTOM pixel edge rather than asking the model to self-estimate a +// "center" — a center is an abstract judgment call, whereas the top/bottom of a glyph is a +// concrete, visible thing to point at. The vertical midpoint used for x/y below is then computed +// here in code from those two real edges, not trusted as a direct AI estimate. export async function locateAllOnScreenshot( userId: number, imageBase64: string, @@ -202,9 +214,12 @@ export async function locateAllOnScreenshot( `This is a screenshot of a Windows desktop application. Find EVERY occurrence of the text ` + `"${text}" visible anywhere in this screenshot (case-insensitive) — there may be zero, one, ` + "or several. Respond with strict JSON only, no prose, no markdown fences: " + - '{"matches": [{"xRatio": <0 to 1>, "yRatio": <0 to 1>}, ...]} — one entry per occurrence, ' + - "at the center of that occurrence's text, using the same fraction-of-image-width/height " + - "convention as before. Use an empty array if there are no occurrences.", + '{"matches": [{"xRatio": <0 to 1>, "yTopRatio": <0 to 1>, "yBottomRatio": <0 to 1>}, ...]} — ' + + "one entry per occurrence. xRatio is how far across the image the occurrence's horizontal " + + "center is, as a fraction of the TOTAL image width. yTopRatio is where the TOP edge of that " + + "occurrence's text (the top of its tallest letters) is, and yBottomRatio is where its BOTTOM " + + "edge (the bottom of its lowest letters, including descenders) is — both as a fraction of " + + "the TOTAL image height. Use an empty array if there are no occurrences.", }, { type: "image_url", @@ -226,7 +241,7 @@ export async function locateAllOnScreenshot( } return matches.map((m) => ({ x: Math.round(m.xRatio * width), - y: Math.round(m.yRatio * height), + y: Math.round(((m.yTopRatio + m.yBottomRatio) / 2) * height), })); } @@ -253,10 +268,12 @@ export async function detectRowHeightRatio(userId: number, imageBase64: string): { type: "text", text: - "This is a screenshot containing a results-grid list of rows (e.g. a patient list), each " + - "row the same height. Measure the height of a single row. Respond with strict JSON only, " + - 'no prose, no markdown fences: {"rowHeightRatio": <0 to 1>} — the height of one row as a ' + - "fraction of the TOTAL image height. Use two decimal places of precision.", + "This is a screenshot containing a results-grid list of rows (e.g. a patient list) with a " + + "header row followed by data rows, all data rows the same height. IGNORE the header row — " + + "it is taller than a data row and would skew the measurement. Look at ALL the visible data " + + "rows (not just one) and measure their AVERAGE height. Respond with strict JSON only, no " + + 'prose, no markdown fences: {"rowHeightRatio": <0 to 1>} — the average data row height as ' + + "a fraction of the TOTAL image height. Use three decimal places of precision.", }, { type: "image_url",