feat: window-diff-based Type Agent locating for Open Dental appointment flow
Replaces fragile whole-screen text matching with a screenshot-diff step that finds each dialog's actual pixel bounds, then restricts every subsequent AI locate/click to that cropped region — eliminating false matches from text elsewhere on screen (title bars, side panels). Adds column-boundary and row-height detection so the patient row and Exam procedure click positions are computed geometrically instead of relying on repeated fuzzy AI guesses for visually similar neighbors. Also adds per-run screenshot/debug-crop backups and a structured run.log for diagnosing failed runs, plus a cmd:move primitive on the Windows agent for pre-click confirmation crops. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -29,6 +29,11 @@ socket.on("cmd:screenshot", (_payload, ack) => {
|
||||
ack({ image: FAKE_SCREENSHOT });
|
||||
});
|
||||
|
||||
socket.on("cmd:move", ({ x, y }, ack) => {
|
||||
console.log(`🖱️ move to (${x}, ${y}) (no click)`);
|
||||
ack({ ok: true });
|
||||
});
|
||||
|
||||
socket.on("cmd:click", ({ x, y }, ack) => {
|
||||
console.log(`🖱️ click at (${x}, ${y})`);
|
||||
ack({ ok: true });
|
||||
|
||||
@@ -2,7 +2,7 @@ import crypto from "crypto";
|
||||
import { Router, Request, Response } from "express";
|
||||
import { getAgentStatus, io } from "../socket";
|
||||
import { captureScreenshot, click, disconnectAgent, doubleClick, doubleClickCurrent, pressKey, typeText } from "../services/windowsAgentBridge";
|
||||
import { runExistingPatientOpenDental } from "../services/typeAgentRunner";
|
||||
import { runExistingPatientOpenDental, runExistingPatientOpenDentalRatioTest } from "../services/typeAgentRunner";
|
||||
import { storage } from "../storage";
|
||||
|
||||
const router = Router();
|
||||
@@ -80,7 +80,7 @@ router.post("/run", async (req: Request, res: Response) => {
|
||||
return;
|
||||
}
|
||||
|
||||
const { software, action, ip, patientId } = req.body ?? {};
|
||||
const { software, action, ip, patientId, ratioTest } = req.body ?? {};
|
||||
if (software !== "open-dental" || action !== "existing-patient") {
|
||||
res.status(400).json({ error: "This software/action combination isn't automated yet." });
|
||||
return;
|
||||
@@ -99,7 +99,8 @@ router.post("/run", async (req: Request, res: Response) => {
|
||||
const runId = crypto.randomUUID();
|
||||
res.json({ runId });
|
||||
|
||||
runExistingPatientOpenDental(userId, ip, patient.lastName, patient.firstName, (index, total, label, status, error) => {
|
||||
const runFn = ratioTest ? runExistingPatientOpenDentalRatioTest : runExistingPatientOpenDental;
|
||||
runFn(userId, runId, ip, patient.lastName, patient.firstName, (index, total, label, status, error) => {
|
||||
io?.emit(`type-agent:run:${runId}`, { index, total, label, status, error });
|
||||
}).catch((err) => {
|
||||
console.error("[type-agent/run] failed:", err);
|
||||
|
||||
@@ -1,5 +1,17 @@
|
||||
import { captureScreenshot, click, doubleClick, doubleClickCurrent, typeText } from "./windowsAgentBridge";
|
||||
import { locateOnScreenshot } from "./visionLocate";
|
||||
import sharp from "sharp";
|
||||
import { captureScreenshot, click, doubleClick, doubleClickCurrent, moveMouse, typeText } from "./windowsAgentBridge";
|
||||
import {
|
||||
confirmTextTyped,
|
||||
cropAroundPoint,
|
||||
cropToRegion,
|
||||
detectColumnBoundaries,
|
||||
detectRowHeightRatio,
|
||||
diffBoundingBox,
|
||||
locateAllOnScreenshot,
|
||||
locateOnScreenshot,
|
||||
WindowBounds,
|
||||
} from "./visionLocate";
|
||||
import { backupTypeAgentScreenshot, logTypeAgentStep } from "../utils/screenshotBackup";
|
||||
|
||||
export type StepStatus = "running" | "done" | "error";
|
||||
export type ProgressCallback = (
|
||||
@@ -13,8 +25,15 @@ export type ProgressCallback = (
|
||||
interface RunContext {
|
||||
ip: string;
|
||||
userId: number;
|
||||
runId: string;
|
||||
patientLastName: string;
|
||||
patientFirstName: string;
|
||||
// Set once a dialog's bounds are detected (via diffBoundingBox) — every locate/click after
|
||||
// that point is restricted to this region until it's replaced by the next dialog's bounds.
|
||||
windowBounds: WindowBounds | null;
|
||||
// The screenshot taken right before the action that's expected to open the *next* dialog —
|
||||
// stashed here so the following step can diff against it without re-capturing.
|
||||
beforeNextWindow: string | null;
|
||||
}
|
||||
|
||||
interface RunStep {
|
||||
@@ -24,23 +43,334 @@ interface RunStep {
|
||||
delayAfterMs: number;
|
||||
}
|
||||
|
||||
async function visionClick(ctx: RunContext, goal: string, opts?: { doubleClick?: boolean }) {
|
||||
// Captures the current screen and, if a window has been detected, crops it down to just that
|
||||
// window — so every locate call after that point only ever sees the relevant dialog, never the
|
||||
// rest of the screen (the main window's title bar, side panels, etc.).
|
||||
async function captureWindowScreenshot(ctx: RunContext): Promise<{ full: string; region: string }> {
|
||||
const { image } = await captureScreenshot(ctx.ip);
|
||||
const loc = await locateOnScreenshot(ctx.userId, image, goal);
|
||||
if (!loc) throw new Error(`Could not find on screen: ${goal}`);
|
||||
console.log(`[type-agent] ${opts?.doubleClick ? "double-click" : "click"} at (${loc.x}, ${loc.y}) — goal: ${goal.slice(0, 80)}...`);
|
||||
const region = ctx.windowBounds ? await cropToRegion(image, ctx.windowBounds) : image;
|
||||
return { full: image, region };
|
||||
}
|
||||
|
||||
// Translates a point from window-cropped-image coordinates back to real screen coordinates.
|
||||
function toScreenPoint(ctx: RunContext, point: { x: number; y: number }): { x: number; y: number } {
|
||||
if (!ctx.windowBounds) return point;
|
||||
return { x: point.x + ctx.windowBounds.left, y: point.y + ctx.windowBounds.top };
|
||||
}
|
||||
|
||||
// Same, for a whole list — used only when logging matches, so run.log always reports real
|
||||
// screen coordinates (matching the full-screen debug screenshots) even though the underlying
|
||||
// comparisons that pick a target run in cropped-image space.
|
||||
function toScreenPoints(ctx: RunContext, points: { x: number; y: number }[]): { x: number; y: number }[] {
|
||||
return points.map((p) => toScreenPoint(ctx, p));
|
||||
}
|
||||
|
||||
// Keeps a point inside the currently-tracked window — every movement/click is restricted to
|
||||
// this region once one is set, so a bad locate can't send the cursor wandering off into the
|
||||
// rest of the app.
|
||||
function clampToWindow(ctx: RunContext, x: number, y: number): { x: number; y: number } {
|
||||
if (!ctx.windowBounds) return { x, y };
|
||||
const { left, top, width, height } = ctx.windowBounds;
|
||||
return {
|
||||
x: Math.min(Math.max(x, left), left + width - 1),
|
||||
y: Math.min(Math.max(y, top), top + height - 1),
|
||||
};
|
||||
}
|
||||
|
||||
// Moves to (x, y) — clamped to the tracked window — saves a small marked debug crop of what's
|
||||
// there (for after-the-fact review, not fed back to any AI call), then clicks.
|
||||
async function debugClickAt(
|
||||
ctx: RunContext,
|
||||
rawX: number,
|
||||
rawY: number,
|
||||
label: string,
|
||||
opts?: { doubleClick?: boolean }
|
||||
) {
|
||||
const { x, y } = clampToWindow(ctx, rawX, rawY);
|
||||
await moveMouse(x, y, ctx.ip);
|
||||
const { image } = await captureScreenshot(ctx.ip);
|
||||
const crop = await cropAroundPoint(image, x, y);
|
||||
backupTypeAgentScreenshot(ctx.runId, `${label}_debug`, crop);
|
||||
|
||||
const action = opts?.doubleClick ? "double_click" : "click";
|
||||
console.log(`[type-agent] ${opts?.doubleClick ? "double-click" : "click"} at (${x}, ${y}) — ${label}`);
|
||||
logTypeAgentStep(ctx.runId, { event: action, label, pixelX: x, pixelY: y });
|
||||
if (opts?.doubleClick) {
|
||||
await doubleClick(loc.x, loc.y, ctx.ip);
|
||||
await doubleClick(x, y, ctx.ip);
|
||||
} else {
|
||||
await click(loc.x, loc.y, ctx.ip);
|
||||
await click(x, y, ctx.ip);
|
||||
}
|
||||
}
|
||||
|
||||
// Open Dental, "make an appointment for an existing patient" — the exact sequence confirmed
|
||||
// against a real Open Dental instance: double-click the pre-positioned schedule slot (opens
|
||||
// Select Patient) → type last name → double-click the matching patient row (opens Edit
|
||||
// Appointment with the appointment already created) → click the Exam procedure → click Save.
|
||||
// Detects a newly-opened dialog by diffing the screenshot stashed in ctx.beforeNextWindow
|
||||
// against the current screen, and adopts its bounds as the window every subsequent locate/click
|
||||
// is restricted to. The four corners (A/B/C/D) are just this rectangle's corners, logged for
|
||||
// visibility.
|
||||
async function detectAndTrackNewWindow(ctx: RunContext, label: string): Promise<void> {
|
||||
if (!ctx.beforeNextWindow) {
|
||||
throw new Error(`No "before" screenshot captured for detecting the ${label} window`);
|
||||
}
|
||||
const { image: after } = await captureScreenshot(ctx.ip);
|
||||
const bounds = await diffBoundingBox(ctx.beforeNextWindow, after);
|
||||
if (!bounds) {
|
||||
throw new Error(`Could not detect the ${label} window opening (no significant screen change)`);
|
||||
}
|
||||
ctx.windowBounds = bounds;
|
||||
ctx.beforeNextWindow = null;
|
||||
logTypeAgentStep(ctx.runId, {
|
||||
event: "window_detected",
|
||||
label,
|
||||
A: { x: bounds.left, y: bounds.top },
|
||||
B: { x: bounds.left + bounds.width, y: bounds.top },
|
||||
C: { x: bounds.left + bounds.width, y: bounds.top + bounds.height },
|
||||
D: { x: bounds.left, y: bounds.top + bounds.height },
|
||||
});
|
||||
}
|
||||
|
||||
// Open Dental, "make an appointment for an existing patient":
|
||||
// 1. Screenshot, then double-click the pre-positioned schedule slot.
|
||||
// 2. Diff against that screenshot to find the newly-opened Select Patient dialog's bounds —
|
||||
// every following locate/click in this dialog is restricted to that region, so nothing
|
||||
// outside it (the main window's title bar, side panels) can ever be matched by mistake.
|
||||
// 3. Click the Last Name field (found by vision, now unambiguous since the window is small
|
||||
// and cropped) and type the first three letters — no first name is typed.
|
||||
// 4. Find the topmost actual patient row in the now-filtered results grid and double-click
|
||||
// it (opens Edit Appointment).
|
||||
// 5. Diff again to find the Edit Appointment window's bounds, then click Exam and Save
|
||||
// within it.
|
||||
const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
|
||||
{
|
||||
label: "Screenshot, then double-click the schedule at the cursor position",
|
||||
execute: async (ctx) => {
|
||||
const { image } = await captureScreenshot(ctx.ip);
|
||||
ctx.beforeNextWindow = image;
|
||||
await doubleClickCurrent(ctx.ip);
|
||||
},
|
||||
delayAfterMs: 1000, // Select Patient window opening
|
||||
},
|
||||
{
|
||||
label: "Detect the Select Patient window",
|
||||
execute: async (ctx) => {
|
||||
await detectAndTrackNewWindow(ctx, "Select Patient");
|
||||
},
|
||||
delayAfterMs: 100,
|
||||
},
|
||||
{
|
||||
label: "Find and click the Last Name field, type the first three letters",
|
||||
execute: async (ctx) => {
|
||||
const lastThree = ctx.patientLastName.slice(0, 3);
|
||||
const { region } = await captureWindowScreenshot(ctx);
|
||||
const loc = await locateOnScreenshot(
|
||||
ctx.userId,
|
||||
region,
|
||||
'the text input field immediately to the right of the "Last Name" label'
|
||||
);
|
||||
if (!loc) throw new Error('Could not find the "Last Name" field in the Select Patient window');
|
||||
const screen = toScreenPoint(ctx, loc);
|
||||
logTypeAgentStep(ctx.runId, { event: "locate", label: "Last Name field", pixelX: screen.x, pixelY: screen.y });
|
||||
|
||||
await debugClickAt(ctx, screen.x, screen.y, "last_name_field");
|
||||
await typeText(lastThree, ctx.ip);
|
||||
},
|
||||
delayAfterMs: 600, // patient list filtering
|
||||
},
|
||||
{
|
||||
label: "Verify the last name fragment was typed",
|
||||
execute: async (ctx) => {
|
||||
const { full } = await captureWindowScreenshot(ctx);
|
||||
backupTypeAgentScreenshot(ctx.runId, "typed_check", full);
|
||||
const lastThree = ctx.patientLastName.slice(0, 3);
|
||||
const typed = await confirmTextTyped(ctx.userId, full, lastThree);
|
||||
logTypeAgentStep(ctx.runId, { event: "typed_check", typed });
|
||||
if (!typed) {
|
||||
throw new Error(`"${lastThree}" doesn't appear to have been typed anywhere on screen`);
|
||||
}
|
||||
},
|
||||
delayAfterMs: 200,
|
||||
},
|
||||
{
|
||||
label: "Find and double-click the first patient row",
|
||||
execute: async (ctx) => {
|
||||
const { region } = await captureWindowScreenshot(ctx);
|
||||
backupTypeAgentScreenshot(ctx.runId, "patient_row_locate", region);
|
||||
|
||||
const headerMatches = await locateAllOnScreenshot(ctx.userId, region, "PatNum");
|
||||
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "PatNum header", matches: toScreenPoints(ctx, headerMatches) });
|
||||
if (headerMatches.length === 0) throw new Error('Could not find the "PatNum" column header');
|
||||
const header = headerMatches.reduce((a, b) => (a.y < b.y ? a : b));
|
||||
|
||||
// A separate "find the first data row, not the header" locate call kept landing too close
|
||||
// to (or on) the header — the two rows are visually similar and only ~15px apart. Since
|
||||
// "PatNum" is an unambiguous anchor and the row height is measurable, the first row's
|
||||
// position is computed directly (headerY + one row height) instead of asking the AI to
|
||||
// visually tell two adjacent, similarly-styled rows apart a second time — nothing left to
|
||||
// confuse once it's arithmetic on two already-known values.
|
||||
const rowHeightRatio = await detectRowHeightRatio(ctx.userId, region);
|
||||
const rowHeightPx = Math.round(rowHeightRatio * (ctx.windowBounds?.height ?? 0));
|
||||
if (rowHeightPx <= 0) throw new Error(`AI reported an invalid row height ratio: ${rowHeightRatio}`);
|
||||
const finalY = header.y + rowHeightPx;
|
||||
|
||||
const screen = toScreenPoint(ctx, { x: header.x, y: finalY });
|
||||
logTypeAgentStep(ctx.runId, {
|
||||
event: "locate",
|
||||
label: "first patient row",
|
||||
pixelX: screen.x,
|
||||
finalY: screen.y,
|
||||
rowHeightPx,
|
||||
headerY: toScreenPoint(ctx, { x: 0, y: header.y }).y,
|
||||
});
|
||||
|
||||
ctx.beforeNextWindow = (await captureScreenshot(ctx.ip)).image;
|
||||
await debugClickAt(ctx, screen.x, screen.y, "patient_row", { doubleClick: true });
|
||||
},
|
||||
delayAfterMs: 1200, // Edit Appointment window opening
|
||||
},
|
||||
{
|
||||
label: "Detect the Edit Appointment window",
|
||||
execute: async (ctx) => {
|
||||
await detectAndTrackNewWindow(ctx, "Edit Appointment");
|
||||
},
|
||||
delayAfterMs: 100,
|
||||
},
|
||||
{
|
||||
label: "Find and click the Exam procedure in the upper middle column",
|
||||
execute: async (ctx) => {
|
||||
const { region } = await captureWindowScreenshot(ctx);
|
||||
backupTypeAgentScreenshot(ctx.runId, "exam_locate", region);
|
||||
|
||||
// The Edit Appointment window is laid out in three big panels: Patient Info (left), the
|
||||
// clickable single-click procedures list (middle), and the already-added procedures grid
|
||||
// plus comm log (right). "Exam" appears both as a clickable item in the middle list AND
|
||||
// as an already-added row in the right grid — telling them apart by y-position alone isn't
|
||||
// reliable since both can land at a similar height. Finding the actual column dividers and
|
||||
// keeping only the "Exam" match inside the middle column resolves this structurally.
|
||||
const dividers = await detectColumnBoundaries(ctx.userId, region);
|
||||
logTypeAgentStep(ctx.runId, { event: "column_boundaries", dividers: dividers.map((x) => toScreenPoint(ctx, { x, y: 0 }).x) });
|
||||
if (dividers.length !== 2) {
|
||||
throw new Error(`Expected 2 column dividers (3 columns), found ${dividers.length}`);
|
||||
}
|
||||
const [middleLeft, middleRight] = dividers as [number, number];
|
||||
|
||||
const examMatches = await locateAllOnScreenshot(ctx.userId, region, "Exam");
|
||||
const singleClickMatches = await locateAllOnScreenshot(ctx.userId, region, "Single click");
|
||||
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Exam", matches: toScreenPoints(ctx, examMatches) });
|
||||
logTypeAgentStep(ctx.runId, { event: "locate_all", label: "Single click", matches: toScreenPoints(ctx, singleClickMatches) });
|
||||
|
||||
if (examMatches.length === 0) throw new Error('Could not find "Exam" in the Edit Appointment window');
|
||||
if (singleClickMatches.length === 0) throw new Error('Could not find "Single click" in the Edit Appointment window');
|
||||
|
||||
const inMiddleColumn = examMatches.filter((m) => m.x > middleLeft && m.x < middleRight);
|
||||
if (inMiddleColumn.length === 0) {
|
||||
throw new Error('Found "Exam", but none were in the middle column of the window');
|
||||
}
|
||||
// "Upper" — the topmost match in that column, in case more than one qualifies.
|
||||
const examLoc = inMiddleColumn.reduce((a, b) => (a.y < b.y ? a : b));
|
||||
|
||||
// "Single click" is the hint text right above this list — same row-height/grid-average
|
||||
// technique used for the patient list, anchored on that text instead of a column header,
|
||||
// so the click lands toward the middle of the Exam row rather than its top edge.
|
||||
const singleClickY = Math.min(...singleClickMatches.map((m) => m.y));
|
||||
const rowHeightRatio = await detectRowHeightRatio(ctx.userId, region);
|
||||
const rowHeightPx = Math.round(rowHeightRatio * (ctx.windowBounds?.height ?? 0));
|
||||
if (rowHeightPx <= 0) throw new Error(`AI reported an invalid row height ratio: ${rowHeightRatio}`);
|
||||
const rowIndex = Math.max(1, Math.round((examLoc.y - singleClickY) / rowHeightPx));
|
||||
const gridY = singleClickY + rowHeightPx * rowIndex;
|
||||
const finalY = Math.round((examLoc.y + gridY) / 2);
|
||||
|
||||
const screen = toScreenPoint(ctx, { x: examLoc.x, y: finalY });
|
||||
logTypeAgentStep(ctx.runId, {
|
||||
event: "exam_match",
|
||||
pixelX: screen.x,
|
||||
directY: toScreenPoint(ctx, { x: 0, y: examLoc.y }).y,
|
||||
gridY: toScreenPoint(ctx, { x: 0, y: gridY }).y,
|
||||
finalY: screen.y,
|
||||
rowHeightPx,
|
||||
});
|
||||
|
||||
await debugClickAt(ctx, screen.x, screen.y, "exam");
|
||||
},
|
||||
delayAfterMs: 500,
|
||||
},
|
||||
{
|
||||
label: "Find and click Save",
|
||||
execute: async (ctx) => {
|
||||
const { region } = await captureWindowScreenshot(ctx);
|
||||
backupTypeAgentScreenshot(ctx.runId, "save_locate", region);
|
||||
const loc = await locateOnScreenshot(
|
||||
ctx.userId,
|
||||
region,
|
||||
"the Save button in the Edit Appointment window — it's in the LOWER part of the " +
|
||||
"RIGHTMOST of the window's three big columns (the one with the already-added " +
|
||||
"procedures grid and comm log above it), below the other action buttons in that column " +
|
||||
'(e.g. "Delete", "To Task List", "Audit Trail")'
|
||||
);
|
||||
if (!loc) throw new Error("Could not find the Save button in the Edit Appointment window");
|
||||
const screen = toScreenPoint(ctx, loc);
|
||||
logTypeAgentStep(ctx.runId, { event: "locate", label: "Save", pixelX: screen.x, pixelY: screen.y });
|
||||
|
||||
await debugClickAt(ctx, screen.x, screen.y, "save");
|
||||
},
|
||||
delayAfterMs: 300,
|
||||
},
|
||||
];
|
||||
|
||||
async function runSteps(ctx: RunContext, steps: RunStep[], onProgress: ProgressCallback): Promise<void> {
|
||||
logTypeAgentStep(ctx.runId, {
|
||||
event: "run_start",
|
||||
patientLastName: ctx.patientLastName,
|
||||
patientFirstName: ctx.patientFirstName,
|
||||
ip: ctx.ip,
|
||||
totalSteps: steps.length,
|
||||
});
|
||||
for (let i = 0; i < steps.length; i++) {
|
||||
const step = steps[i]!;
|
||||
onProgress(i, steps.length, step.label, "running");
|
||||
logTypeAgentStep(ctx.runId, { event: "step_start", index: i, label: step.label });
|
||||
try {
|
||||
await step.execute(ctx);
|
||||
onProgress(i, steps.length, step.label, "done");
|
||||
logTypeAgentStep(ctx.runId, { event: "step_done", index: i, label: step.label });
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
onProgress(i, steps.length, step.label, "error", message);
|
||||
logTypeAgentStep(ctx.runId, { event: "step_error", index: i, label: step.label, error: message });
|
||||
throw err;
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, step.delayAfterMs));
|
||||
}
|
||||
logTypeAgentStep(ctx.runId, { event: "run_done" });
|
||||
}
|
||||
|
||||
export async function runExistingPatientOpenDental(
|
||||
userId: number,
|
||||
runId: string,
|
||||
ip: string,
|
||||
patientLastName: string,
|
||||
patientFirstName: string,
|
||||
onProgress: ProgressCallback
|
||||
): Promise<void> {
|
||||
const ctx: RunContext = {
|
||||
ip,
|
||||
userId,
|
||||
runId,
|
||||
patientLastName,
|
||||
patientFirstName,
|
||||
windowBounds: null,
|
||||
beforeNextWindow: null,
|
||||
};
|
||||
await runSteps(ctx, OPEN_DENTAL_EXISTING_PATIENT, onProgress);
|
||||
}
|
||||
|
||||
// TEMPORARY TEST PATH — types the full last name directly (relying on the Select Patient
|
||||
// window's default focus, no field click) and then moves to a hardcoded ratio (0.31, 0.21) —
|
||||
// the position manually read off a real "Allowed, Allen" row screenshot — instead of any
|
||||
// vision lookup. Exists purely to test the ratio→pixel conversion and move/double-click
|
||||
// mechanics in isolation from AI locate mistakes. Kept separate from the real flow above,
|
||||
// which now types into both name fields individually and uses window-relative locates instead.
|
||||
const TEST_ROW_X_RATIO = 0.31;
|
||||
const TEST_ROW_Y_RATIO = 0.21;
|
||||
|
||||
const OPEN_DENTAL_EXISTING_PATIENT_RATIO_TEST: RunStep[] = [
|
||||
{
|
||||
label: "Double-click the schedule at the cursor position",
|
||||
execute: async (ctx) => {
|
||||
@@ -49,90 +379,56 @@ const OPEN_DENTAL_EXISTING_PATIENT: RunStep[] = [
|
||||
delayAfterMs: 1000, // Select Patient window opening
|
||||
},
|
||||
{
|
||||
label: "Find and click the Last Name field",
|
||||
execute: async (ctx) => {
|
||||
await visionClick(
|
||||
ctx,
|
||||
"the Last Name text input field in the Select Patient window — it's in the \"Search by:\" " +
|
||||
"panel on the RIGHT half of the window (if you split the window into a left half and a " +
|
||||
"right half), not in the results grid on the left"
|
||||
);
|
||||
},
|
||||
delayAfterMs: 300,
|
||||
},
|
||||
{
|
||||
label: "Type the patient's last name",
|
||||
label: "Type the patient's last name (no field click first — relies on default focus)",
|
||||
execute: async (ctx) => {
|
||||
await typeText(ctx.patientLastName, ctx.ip);
|
||||
},
|
||||
delayAfterMs: 600, // patient list filtering
|
||||
},
|
||||
{
|
||||
label: "Find and double-click the matching patient row",
|
||||
label: "TEST: verify the last name was typed",
|
||||
execute: async (ctx) => {
|
||||
const lastPrefix = ctx.patientLastName.slice(0, 3);
|
||||
const firstPrefix = ctx.patientFirstName.slice(0, 3);
|
||||
await visionClick(
|
||||
ctx,
|
||||
`Search the LEFT half of the Select Patient window for the row of patient data whose ` +
|
||||
`LastName column starts with "${lastPrefix}" AND whose First Name column starts with ` +
|
||||
`"${firstPrefix}" — matching is case-insensitive and only needs to match the first 3 ` +
|
||||
`letters of each name, not the full name exactly. Both the last name prefix and first name ` +
|
||||
`prefix must match — there may be other rows with a matching last name (family members, ` +
|
||||
`e.g. a sibling or spouse) but a different first name; do NOT click those, they are the ` +
|
||||
`wrong patient. That row is likely shaded/highlighted in LIGHT BLUE as the currently-selected ` +
|
||||
`row. Do NOT click the window's own title bar — a DARK navy-blue bar at the very top of the ` +
|
||||
`window that just reads "Select Patient" in white text; that is window chrome, not a patient ` +
|
||||
`row, and is well above the results grid. Also do not click the column header row (labels ` +
|
||||
`like "PatNum", "LastName", "First Name"). Click on the actual patient text in the matching data row.`,
|
||||
{ doubleClick: true }
|
||||
);
|
||||
const { image } = await captureScreenshot(ctx.ip);
|
||||
backupTypeAgentScreenshot(ctx.runId, `ratio_test_typed_check_${ctx.patientLastName}`, image);
|
||||
const typed = await confirmTextTyped(ctx.userId, image, ctx.patientLastName);
|
||||
if (!typed) {
|
||||
throw new Error(`"${ctx.patientLastName}" doesn't appear to have been typed anywhere on screen`);
|
||||
}
|
||||
},
|
||||
delayAfterMs: 1200, // Edit Appointment window opening
|
||||
delayAfterMs: 200,
|
||||
},
|
||||
{
|
||||
label: "Find and click the Exam procedure",
|
||||
label: `TEST: move to fixed ratio (${TEST_ROW_X_RATIO}, ${TEST_ROW_Y_RATIO}) and double-click`,
|
||||
execute: async (ctx) => {
|
||||
await visionClick(
|
||||
ctx,
|
||||
'the "Exam" item in the procedures list of the Edit Appointment window — it\'s in the ' +
|
||||
"UPPER-MIDDLE area of the window (if you split the window into an upper half and a lower " +
|
||||
"half, and again into left/middle/right thirds, it's in the upper half, middle third), " +
|
||||
'in a plain text list that also contains items like "Ex,Pro,Flo", "Prophy-Adult", "Pano"'
|
||||
);
|
||||
const { image } = await captureScreenshot(ctx.ip);
|
||||
const { width = 0, height = 0 } = await sharp(Buffer.from(image, "base64")).metadata();
|
||||
const x = Math.round(TEST_ROW_X_RATIO * width);
|
||||
const y = Math.round(TEST_ROW_Y_RATIO * height);
|
||||
|
||||
console.log(`[type-agent] TEST double-click at ratio (${TEST_ROW_X_RATIO}, ${TEST_ROW_Y_RATIO}) -> pixel (${x}, ${y})`);
|
||||
await moveMouse(x, y, ctx.ip);
|
||||
await doubleClick(x, y, ctx.ip);
|
||||
},
|
||||
delayAfterMs: 500,
|
||||
},
|
||||
{
|
||||
label: "Find and click Save",
|
||||
execute: async (ctx) => {
|
||||
await visionClick(ctx, "the Save button in the lower right corner of the Edit Appointment window");
|
||||
},
|
||||
delayAfterMs: 300,
|
||||
delayAfterMs: 1200,
|
||||
},
|
||||
];
|
||||
|
||||
export async function runExistingPatientOpenDental(
|
||||
export async function runExistingPatientOpenDentalRatioTest(
|
||||
userId: number,
|
||||
runId: string,
|
||||
ip: string,
|
||||
patientLastName: string,
|
||||
patientFirstName: string,
|
||||
onProgress: ProgressCallback
|
||||
): Promise<void> {
|
||||
const ctx: RunContext = { ip, userId, patientLastName, patientFirstName };
|
||||
const steps = OPEN_DENTAL_EXISTING_PATIENT;
|
||||
|
||||
for (let i = 0; i < steps.length; i++) {
|
||||
const step = steps[i]!;
|
||||
onProgress(i, steps.length, step.label, "running");
|
||||
try {
|
||||
await step.execute(ctx);
|
||||
onProgress(i, steps.length, step.label, "done");
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
onProgress(i, steps.length, step.label, "error", message);
|
||||
throw err;
|
||||
}
|
||||
await new Promise((resolve) => setTimeout(resolve, step.delayAfterMs));
|
||||
}
|
||||
const ctx: RunContext = {
|
||||
ip,
|
||||
userId,
|
||||
runId,
|
||||
patientLastName,
|
||||
patientFirstName,
|
||||
windowBounds: null,
|
||||
beforeNextWindow: null,
|
||||
};
|
||||
await runSteps(ctx, OPEN_DENTAL_EXISTING_PATIENT_RATIO_TEST, onProgress);
|
||||
}
|
||||
|
||||
@@ -1,17 +1,22 @@
|
||||
import sharp from "sharp";
|
||||
import { storage } from "../storage";
|
||||
import { resolveAiProvider, getLlm } from "../ai/llm-factory";
|
||||
|
||||
// Side length (px) of the debug crop saved around a click point — big enough to show
|
||||
// surrounding context (e.g. neighboring grid rows) when reviewing what a run actually clicked.
|
||||
const DEBUG_CROP_SIZE = 220;
|
||||
|
||||
interface ParsedLocation {
|
||||
found: boolean;
|
||||
x?: number;
|
||||
y?: number;
|
||||
xRatio?: number;
|
||||
yRatio?: number;
|
||||
}
|
||||
|
||||
// Parses the model's response into {found, x, y}, tolerating near-miss JSON rather than
|
||||
// only accepting strictly valid JSON. Seen in practice: {"found": true, "x": 1023, 187} —
|
||||
// a dropped "y" key with the value still present positionally. Since this response should
|
||||
// only ever contain two numbers (the coordinates), falling back to "first two numbers after
|
||||
// found" is safe and recovers without burning another AI call.
|
||||
// Parses the model's response into {found, xRatio, yRatio}, tolerating near-miss JSON rather
|
||||
// than only accepting strictly valid JSON. Seen in practice with the old pixel-based format:
|
||||
// {"found": true, "x": 1023, 187} — a dropped key with the value still present positionally.
|
||||
// Since this response should only ever contain two numbers (the coordinates), falling back to
|
||||
// "first two numbers after found" is safe and recovers without burning another AI call.
|
||||
function parseLocationResponse(raw: string): ParsedLocation | null {
|
||||
const match = raw.match(/\{[\s\S]*\}/);
|
||||
if (!match) return null;
|
||||
@@ -32,17 +37,24 @@ function parseLocationResponse(raw: string): ParsedLocation | null {
|
||||
const numbers = afterFound.match(/-?\d+(\.\d+)?/g);
|
||||
if (!numbers || numbers.length < 2) return null;
|
||||
|
||||
return { found: true, x: Number(numbers[0]), y: Number(numbers[1]) };
|
||||
return { found: true, xRatio: Number(numbers[0]), yRatio: Number(numbers[1]) };
|
||||
}
|
||||
|
||||
// Asks the configured vision AI to find a described UI element in a screenshot and return
|
||||
// its pixel coordinates — the core primitive the Type Agent step runner clicks/types through.
|
||||
// its position — the core primitive the Type Agent step runner clicks/types through.
|
||||
// Reuses the same Claude-vision pattern as /api/ai/detect-eligibility-info.
|
||||
//
|
||||
// The AI reports position as a 0-1 ratio of image width/height rather than absolute pixels.
|
||||
// Vision models are more reliable at "this is 4/10 of the way across, 3/10 down" than at
|
||||
// naming an exact pixel — asking for pixels directly produced coordinates that were close but
|
||||
// occasionally off by enough to land on a neighboring row or the title bar above it. A ratio
|
||||
// is resolution-independent by construction, so it's converted to real screen pixels here
|
||||
// using the actual screenshot dimensions rather than trusting the model to know the image size.
|
||||
export async function locateOnScreenshot(
|
||||
userId: number,
|
||||
imageBase64: string,
|
||||
goal: string
|
||||
): Promise<{ x: number; y: number } | null> {
|
||||
): Promise<{ x: number; y: number; xRatio: number; yRatio: number } | null> {
|
||||
const aiSettings = await storage.getAiSettings(userId);
|
||||
const activeAi = resolveAiProvider(aiSettings ?? {});
|
||||
if (!activeAi) {
|
||||
@@ -53,6 +65,7 @@ export async function locateOnScreenshot(
|
||||
}
|
||||
|
||||
const llm = getLlm(activeAi.provider, activeAi.key, activeAi.model);
|
||||
const { width = 0, height = 0 } = await sharp(Buffer.from(imageBase64, "base64")).metadata();
|
||||
|
||||
const content = [
|
||||
{
|
||||
@@ -60,9 +73,12 @@ export async function locateOnScreenshot(
|
||||
text:
|
||||
`This is a screenshot of a Windows desktop application. Find this UI element: "${goal}". ` +
|
||||
"Respond with strict JSON only, no prose, no markdown fences: " +
|
||||
'{"found": true, "x": <pixel x>, "y": <pixel y>} if you can locate it (coordinates must be ' +
|
||||
'pixel positions within this exact image, at the center of the element), or {"found": false} ' +
|
||||
"if it isn't visible in this screenshot.",
|
||||
'{"found": true, "xRatio": <0 to 1>, "yRatio": <0 to 1>} if you can locate it, or ' +
|
||||
'{"found": false} if it isn\'t visible in this screenshot. xRatio is how far across the ' +
|
||||
"image the element's center is, as a fraction of the TOTAL image width (0 = left edge, " +
|
||||
"0.5 = horizontal center, 1 = right edge). yRatio is how far down the element's center is, " +
|
||||
"as a fraction of the TOTAL image height (0 = top edge, 1 = bottom edge). Use two decimal " +
|
||||
"places of precision — do not report pixel coordinates.",
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
@@ -86,11 +102,353 @@ export async function locateOnScreenshot(
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!parsed.found || typeof parsed.x !== "number" || typeof parsed.y !== "number") {
|
||||
if (!parsed.found || typeof parsed.xRatio !== "number" || typeof parsed.yRatio !== "number") {
|
||||
return null;
|
||||
}
|
||||
return { x: Math.round(parsed.x), y: Math.round(parsed.y) };
|
||||
return {
|
||||
x: Math.round(parsed.xRatio * width),
|
||||
y: Math.round(parsed.yRatio * height),
|
||||
xRatio: parsed.xRatio,
|
||||
yRatio: parsed.yRatio,
|
||||
};
|
||||
}
|
||||
|
||||
throw lastError ?? new Error("AI returned an unparseable response");
|
||||
}
|
||||
|
||||
// Crops a small square around (x, y) and draws a red crosshair marker at that exact point —
|
||||
// a debug artifact saved before each click so a person can see afterward exactly where the
|
||||
// run was about to click. A real screen capture won't reliably include the OS mouse cursor
|
||||
// (BitBlt-based grabs typically omit it), so this synthetic marker stands in for it.
|
||||
export async function cropAroundPoint(imageBase64: string, x: number, y: number): Promise<string> {
|
||||
const source = sharp(Buffer.from(imageBase64, "base64"));
|
||||
const { width = 0, height = 0 } = await source.metadata();
|
||||
|
||||
const cropWidth = Math.min(DEBUG_CROP_SIZE, width);
|
||||
const cropHeight = Math.min(DEBUG_CROP_SIZE, height);
|
||||
const left = Math.round(Math.max(0, Math.min(x - cropWidth / 2, width - cropWidth)));
|
||||
const top = Math.round(Math.max(0, Math.min(y - cropHeight / 2, height - cropHeight)));
|
||||
const markerX = x - left;
|
||||
const markerY = y - top;
|
||||
|
||||
const crosshair = Buffer.from(
|
||||
`<svg width="${cropWidth}" height="${cropHeight}">` +
|
||||
`<line x1="${markerX - 12}" y1="${markerY}" x2="${markerX + 12}" y2="${markerY}" stroke="red" stroke-width="2"/>` +
|
||||
`<line x1="${markerX}" y1="${markerY - 12}" x2="${markerX}" y2="${markerY + 12}" stroke="red" stroke-width="2"/>` +
|
||||
`<circle cx="${markerX}" cy="${markerY}" r="7" fill="none" stroke="red" stroke-width="2"/>` +
|
||||
`</svg>`
|
||||
);
|
||||
|
||||
const cropped = await source
|
||||
.extract({ left, top, width: cropWidth, height: cropHeight })
|
||||
.composite([{ input: crosshair }])
|
||||
.png()
|
||||
.toBuffer();
|
||||
|
||||
return cropped.toString("base64");
|
||||
}
|
||||
|
||||
// Parses {"matches": [{"xRatio":.., "yRatio":..}, ...]}, tolerating near-miss JSON the same
|
||||
// way parseLocationResponse does — falls back to regex-extracting every xRatio/yRatio pair in
|
||||
// the response if strict parsing fails, rather than treating a malformed-but-salvageable
|
||||
// response as zero matches.
|
||||
function parseMatchesResponse(raw: string): { xRatio: number; yRatio: number }[] | null {
|
||||
const block = raw.match(/\{[\s\S]*\}/)?.[0];
|
||||
if (!block) return null;
|
||||
|
||||
try {
|
||||
const parsed = JSON.parse(block);
|
||||
if (Array.isArray(parsed.matches)) return parsed.matches;
|
||||
} catch {
|
||||
// fall through to lenient repair below
|
||||
}
|
||||
|
||||
const pairs: { xRatio: number; yRatio: number }[] = [];
|
||||
const pairRegex = /"xRatio"\s*:\s*(-?\d+(?:\.\d+)?)\s*,\s*"yRatio"\s*:\s*(-?\d+(?:\.\d+)?)/g;
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = pairRegex.exec(block))) {
|
||||
pairs.push({ xRatio: Number(match[1]), yRatio: Number(match[2]) });
|
||||
}
|
||||
return pairs.length > 0 ? pairs : null;
|
||||
}
|
||||
|
||||
// Like locateOnScreenshot, but finds every occurrence of an exact piece of text rather than
|
||||
// one best guess for a described element. Used to disambiguate by position instead of by
|
||||
// asking the AI to judge which occurrence is "the right one" — e.g. a last name shows up once
|
||||
// in the search box and once per matching patient row, so the caller pairs it against another
|
||||
// text's occurrences (a first name) that share the same row instead of relying on a single
|
||||
// fuzzy vision judgment call.
|
||||
export async function locateAllOnScreenshot(
|
||||
userId: number,
|
||||
imageBase64: string,
|
||||
text: string
|
||||
): Promise<{ x: number; y: number }[]> {
|
||||
const aiSettings = await storage.getAiSettings(userId);
|
||||
const activeAi = resolveAiProvider(aiSettings ?? {});
|
||||
if (!activeAi) {
|
||||
throw new Error("AI is not configured. Add an API key in AI Settings.");
|
||||
}
|
||||
if (activeAi.provider !== "claude") {
|
||||
throw new Error("Vision-guided steps require Claude to be the active AI provider.");
|
||||
}
|
||||
|
||||
const llm = getLlm(activeAi.provider, activeAi.key, activeAi.model);
|
||||
const { width = 0, height = 0 } = await sharp(Buffer.from(imageBase64, "base64")).metadata();
|
||||
|
||||
const content = [
|
||||
{
|
||||
type: "text",
|
||||
text:
|
||||
`This is a screenshot of a Windows desktop application. Find EVERY occurrence of the text ` +
|
||||
`"${text}" visible anywhere in this screenshot (case-insensitive) — there may be zero, one, ` +
|
||||
"or several. Respond with strict JSON only, no prose, no markdown fences: " +
|
||||
'{"matches": [{"xRatio": <0 to 1>, "yRatio": <0 to 1>}, ...]} — one entry per occurrence, ' +
|
||||
"at the center of that occurrence's text, using the same fraction-of-image-width/height " +
|
||||
"convention as before. Use an empty array if there are no occurrences.",
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: `data:image/png;base64,${imageBase64}` },
|
||||
},
|
||||
];
|
||||
|
||||
const MAX_ATTEMPTS = 2;
|
||||
let lastError: Error | null = null;
|
||||
|
||||
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
||||
const response = await llm.invoke([{ role: "user", content }] as any);
|
||||
const raw = String(response.content).trim();
|
||||
|
||||
const matches = parseMatchesResponse(raw);
|
||||
if (!matches) {
|
||||
lastError = new Error(`AI returned an unparseable response: ${raw.slice(0, 200)}`);
|
||||
continue;
|
||||
}
|
||||
return matches.map((m) => ({
|
||||
x: Math.round(m.xRatio * width),
|
||||
y: Math.round(m.yRatio * height),
|
||||
}));
|
||||
}
|
||||
|
||||
throw lastError ?? new Error("AI returned an unparseable response");
|
||||
}
|
||||
|
||||
// Asks the AI to measure the height of one row in a results-grid list, as a ratio of the
|
||||
// image's total height — used to build a grid of valid row-center y-positions so a click can be
|
||||
// snapped toward the middle of whichever row it's closest to, rather than trusting a single
|
||||
// direct locate call's y-coordinate on its own (which tends to land near a row's top edge).
|
||||
export async function detectRowHeightRatio(userId: number, imageBase64: string): Promise<number> {
|
||||
const aiSettings = await storage.getAiSettings(userId);
|
||||
const activeAi = resolveAiProvider(aiSettings ?? {});
|
||||
if (!activeAi) {
|
||||
throw new Error("AI is not configured. Add an API key in AI Settings.");
|
||||
}
|
||||
if (activeAi.provider !== "claude") {
|
||||
throw new Error("Vision-guided steps require Claude to be the active AI provider.");
|
||||
}
|
||||
|
||||
const llm = getLlm(activeAi.provider, activeAi.key, activeAi.model);
|
||||
|
||||
const content = [
|
||||
{
|
||||
type: "text",
|
||||
text:
|
||||
"This is a screenshot containing a results-grid list of rows (e.g. a patient list), each " +
|
||||
"row the same height. Measure the height of a single row. Respond with strict JSON only, " +
|
||||
'no prose, no markdown fences: {"rowHeightRatio": <0 to 1>} — the height of one row as a ' +
|
||||
"fraction of the TOTAL image height. Use two decimal places of precision.",
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: `data:image/png;base64,${imageBase64}` },
|
||||
},
|
||||
];
|
||||
|
||||
const response = await llm.invoke([{ role: "user", content }] as any);
|
||||
const raw = String(response.content).trim();
|
||||
const match = raw.match(/"rowHeightRatio"\s*:\s*(-?\d+(?:\.\d+)?)/);
|
||||
if (!match) throw new Error(`AI returned an unparseable response: ${raw.slice(0, 200)}`);
|
||||
return Number(match[1]);
|
||||
}
|
||||
|
||||
// Asks the AI to find the x-positions of the MAJOR vertical dividers that split a window into
|
||||
// its big panels/columns (e.g. a "Patient Info" panel, a middle fields+list panel, a right-side
|
||||
// grid panel) — as opposed to minor lines inside a single table's own columns. Used to tell
|
||||
// apart same-text matches that live in different panels (e.g. "Exam" appearing both in a
|
||||
// clickable procedures list and, separately, in an already-added-procedures grid) by which
|
||||
// column they actually fall in, rather than guessing from y-position alone.
|
||||
export async function detectColumnBoundaries(userId: number, imageBase64: string): Promise<number[]> {
|
||||
const aiSettings = await storage.getAiSettings(userId);
|
||||
const activeAi = resolveAiProvider(aiSettings ?? {});
|
||||
if (!activeAi) {
|
||||
throw new Error("AI is not configured. Add an API key in AI Settings.");
|
||||
}
|
||||
if (activeAi.provider !== "claude") {
|
||||
throw new Error("Vision-guided steps require Claude to be the active AI provider.");
|
||||
}
|
||||
|
||||
const llm = getLlm(activeAi.provider, activeAi.key, activeAi.model);
|
||||
const { width = 0 } = await sharp(Buffer.from(imageBase64, "base64")).metadata();
|
||||
|
||||
const content = [
|
||||
{
|
||||
type: "text",
|
||||
text:
|
||||
"This screenshot shows a Windows dialog laid out in a small number of large vertical " +
|
||||
"sections (columns) — e.g. a distinct panel on the left, a middle section with fields " +
|
||||
'and/or a list, and another panel or grid on the right. Find the x-position of each MAJOR ' +
|
||||
"vertical divider between these big sections (a visible line, border, or clear gap — NOT " +
|
||||
"the minor column lines inside a single table). Respond with strict JSON only, no prose, " +
|
||||
'no markdown fences: {"dividerXRatios": [<0 to 1>, ...]} — one entry per major divider, ' +
|
||||
"sorted left to right, as a fraction of the TOTAL image width. Use two decimal places of " +
|
||||
"precision.",
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: `data:image/png;base64,${imageBase64}` },
|
||||
},
|
||||
];
|
||||
|
||||
const response = await llm.invoke([{ role: "user", content }] as any);
|
||||
const raw = String(response.content).trim();
|
||||
const block = raw.match(/\{[\s\S]*\}/)?.[0];
|
||||
if (!block) throw new Error(`AI returned an unparseable response: ${raw.slice(0, 200)}`);
|
||||
|
||||
let ratios: number[];
|
||||
try {
|
||||
const parsed = JSON.parse(block);
|
||||
if (!Array.isArray(parsed.dividerXRatios)) throw new Error("not an array");
|
||||
ratios = parsed.dividerXRatios;
|
||||
} catch {
|
||||
const arrayMatch = block.match(/"dividerXRatios"\s*:\s*\[([^\]]*)\]/);
|
||||
if (!arrayMatch) throw new Error(`AI returned an unparseable response: ${raw.slice(0, 200)}`);
|
||||
ratios = arrayMatch[1]!
|
||||
.split(",")
|
||||
.map((s) => Number(s.trim()))
|
||||
.filter((n) => !Number.isNaN(n));
|
||||
}
|
||||
|
||||
return ratios.map((r) => Math.round(r * width)).sort((a, b) => a - b);
|
||||
}
|
||||
|
||||
export interface WindowBounds {
|
||||
left: number;
|
||||
top: number;
|
||||
width: number;
|
||||
height: number;
|
||||
}
|
||||
|
||||
// Compares two full screenshots and returns the bounding box of whatever changed
|
||||
// significantly between them — used to find a newly-opened dialog's exact position/size with
|
||||
// pixel math instead of an AI call, so later steps can crop every screenshot down to just that
|
||||
// window and never even show the AI anything outside it (the earlier bug where a name prefix
|
||||
// happened to also match unrelated text elsewhere on screen, e.g. "All Received" in a side
|
||||
// panel, is structurally impossible once matching only ever sees the cropped window).
|
||||
//
|
||||
// Rather than a raw min/max over any differing pixel, this counts changed pixels per row and
|
||||
// per column and only counts a row/column as part of the new window once a meaningful fraction
|
||||
// of it changed — a real window opening changes a large contiguous block, whereas incidental
|
||||
// noise elsewhere (a blinking clock, a flickering tray icon) only touches a handful of pixels
|
||||
// and would otherwise blow the bounding box out to include it.
|
||||
export async function diffBoundingBox(
|
||||
beforeBase64: string,
|
||||
afterBase64: string
|
||||
): Promise<WindowBounds | null> {
|
||||
const [before, after] = await Promise.all([
|
||||
sharp(Buffer.from(beforeBase64, "base64")).ensureAlpha().raw().toBuffer({ resolveWithObject: true }),
|
||||
sharp(Buffer.from(afterBase64, "base64")).ensureAlpha().raw().toBuffer({ resolveWithObject: true }),
|
||||
]);
|
||||
const { data: b, info } = before;
|
||||
const { data: a, info: infoAfter } = after;
|
||||
if (info.width !== infoAfter.width || info.height !== infoAfter.height) return null;
|
||||
|
||||
const { width, height, channels } = info;
|
||||
const PIXEL_DIFF_THRESHOLD = 40; // sum of |dR| + |dG| + |dB| to count a pixel as "changed"
|
||||
const rowCounts = new Uint32Array(height);
|
||||
const colCounts = new Uint32Array(width);
|
||||
|
||||
for (let y = 0; y < height; y++) {
|
||||
const rowBase = y * width * channels;
|
||||
for (let x = 0; x < width; x++) {
|
||||
const idx = rowBase + x * channels;
|
||||
const diff = Math.abs(a[idx]! - b[idx]!) + Math.abs(a[idx + 1]! - b[idx + 1]!) + Math.abs(a[idx + 2]! - b[idx + 2]!);
|
||||
if (diff > PIXEL_DIFF_THRESHOLD) {
|
||||
rowCounts[y] = (rowCounts[y] ?? 0) + 1;
|
||||
colCounts[x] = (colCounts[x] ?? 0) + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const rowThreshold = width * 0.05;
|
||||
const colThreshold = height * 0.05;
|
||||
|
||||
let top = -1;
|
||||
let bottom = -1;
|
||||
for (let y = 0; y < height; y++) {
|
||||
if (rowCounts[y]! > rowThreshold) {
|
||||
if (top === -1) top = y;
|
||||
bottom = y;
|
||||
}
|
||||
}
|
||||
let left = -1;
|
||||
let right = -1;
|
||||
for (let x = 0; x < width; x++) {
|
||||
if (colCounts[x]! > colThreshold) {
|
||||
if (left === -1) left = x;
|
||||
right = x;
|
||||
}
|
||||
}
|
||||
|
||||
if (top === -1 || left === -1) return null;
|
||||
return { left, top, width: right - left + 1, height: bottom - top + 1 };
|
||||
}
|
||||
|
||||
// Plain crop, no marker — used to restrict a screenshot to a previously-detected window's
|
||||
// bounds before handing it to any locate call.
|
||||
export async function cropToRegion(imageBase64: string, region: WindowBounds): Promise<string> {
|
||||
const cropped = await sharp(Buffer.from(imageBase64, "base64"))
|
||||
.extract({ left: region.left, top: region.top, width: region.width, height: region.height })
|
||||
.png()
|
||||
.toBuffer();
|
||||
return cropped.toString("base64");
|
||||
}
|
||||
|
||||
// Sanity check for steps that type without first vision-clicking into a field (relying on
|
||||
// whatever already has focus) — confirms the text actually landed somewhere visible on
|
||||
// screen before moving on, catching the case where focus wasn't where it was assumed to be.
|
||||
export async function confirmTextTyped(
|
||||
userId: number,
|
||||
fullScreenshotBase64: string,
|
||||
expectedText: string
|
||||
): Promise<boolean> {
|
||||
const aiSettings = await storage.getAiSettings(userId);
|
||||
const activeAi = resolveAiProvider(aiSettings ?? {});
|
||||
if (!activeAi) {
|
||||
throw new Error("AI is not configured. Add an API key in AI Settings.");
|
||||
}
|
||||
if (activeAi.provider !== "claude") {
|
||||
throw new Error("Vision-guided steps require Claude to be the active AI provider.");
|
||||
}
|
||||
|
||||
const llm = getLlm(activeAi.provider, activeAi.key, activeAi.model);
|
||||
|
||||
const content = [
|
||||
{
|
||||
type: "text",
|
||||
text:
|
||||
`This is a screenshot of a Windows desktop application. Is the text "${expectedText}" ` +
|
||||
"visible typed into some text input field on screen? Respond with strict JSON only, no " +
|
||||
'prose, no markdown fences: {"confirmed": true} if that text is visible in a field, or ' +
|
||||
'{"confirmed": false} if it is not visible anywhere.',
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: `data:image/png;base64,${fullScreenshotBase64}` },
|
||||
},
|
||||
];
|
||||
|
||||
const response = await llm.invoke([{ role: "user", content }] as any);
|
||||
const raw = String(response.content).trim();
|
||||
const match = raw.match(/"confirmed"\s*:\s*(true|false)/i);
|
||||
return match?.[1]?.toLowerCase() === "true";
|
||||
}
|
||||
|
||||
@@ -28,6 +28,12 @@ export function captureScreenshot(ip?: string) {
|
||||
return sendCommand<{ image: string }>("cmd:screenshot", {}, ip);
|
||||
}
|
||||
|
||||
// Moves the cursor without clicking — lets a step take a confirmation screenshot before
|
||||
// committing to a click.
|
||||
export function moveMouse(x: number, y: number, ip?: string) {
|
||||
return sendCommand<{ ok: boolean }>("cmd:move", { x, y }, ip);
|
||||
}
|
||||
|
||||
export function click(x: number, y: number, ip?: string) {
|
||||
return sendCommand<{ ok: boolean }>("cmd:click", { x, y }, ip);
|
||||
}
|
||||
|
||||
@@ -19,3 +19,38 @@ export function backupScreenshots(files: { originalname: string; buffer: Buffer
|
||||
console.error("[screenshotBackup] failed to write backup:", err);
|
||||
}
|
||||
}
|
||||
|
||||
function getTypeAgentRunDir(runId: string): string {
|
||||
const dateDir = new Date().toISOString().slice(0, 10); // YYYY-MM-DD
|
||||
return path.join(BACKUP_ROOT, "type-agent", dateDir, runId);
|
||||
}
|
||||
|
||||
// Same convention as backupScreenshots, but grouped under a "type-agent/<date>/<runId>"
|
||||
// subfolder so every screenshot from one Type Agent run stays together for debugging
|
||||
// (e.g. "locate" and "confirm" images per step, in click order via the timestamp prefix).
|
||||
export function backupTypeAgentScreenshot(runId: string, label: string, imageBase64: string): void {
|
||||
try {
|
||||
const dir = getTypeAgentRunDir(runId);
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const safeLabel = label.replace(/[/\\?%*:|"<>]/g, "-").slice(0, 80);
|
||||
const fileName = `${Date.now()}_${safeLabel}.png`;
|
||||
fs.writeFileSync(path.join(dir, fileName), Buffer.from(imageBase64, "base64"));
|
||||
} catch (err) {
|
||||
console.error("[screenshotBackup] failed to write type-agent backup:", err);
|
||||
}
|
||||
}
|
||||
|
||||
// One human-readable line per event (step start/attempt/locate/confirm/click/error), appended
|
||||
// to run.log in the same per-run folder as that run's screenshots — so opening one folder
|
||||
// shows both what was clicked and what it looked like at the time, including the exact
|
||||
// AI-reported ratio for every locate attempt.
|
||||
export function logTypeAgentStep(runId: string, entry: Record<string, unknown>): void {
|
||||
try {
|
||||
const dir = getTypeAgentRunDir(runId);
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
const line = `[${new Date().toISOString()}] ${JSON.stringify(entry)}\n`;
|
||||
fs.appendFileSync(path.join(dir, "run.log"), line);
|
||||
} catch (err) {
|
||||
console.error("[screenshotBackup] failed to write type-agent log:", err);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ import { Keyboard, CalendarPlus, UserPlus, CreditCard, Circle, CheckCircle2, Mon
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Badge } from "@/components/ui/badge";
|
||||
import { Card, CardContent, CardHeader, CardTitle, CardDescription } from "@/components/ui/card";
|
||||
import { Checkbox } from "@/components/ui/checkbox";
|
||||
import {
|
||||
Select,
|
||||
SelectContent,
|
||||
@@ -86,6 +87,10 @@ export default function AiTypeAgentPage() {
|
||||
const [runAction, setRunAction] = useState<ActionId | null>(null);
|
||||
const [agents, setAgents] = useState<ConnectedAgent[]>([]);
|
||||
const [selectedAgentIp, setSelectedAgentIp] = useState<string | null>(null);
|
||||
// Temporary debug toggle: skips vision lookup for the patient-row click and moves to a
|
||||
// hardcoded ratio instead, to test move/double-click mechanics in isolation. Remove once
|
||||
// the vision-based row click is confirmed reliable.
|
||||
const [ratioTest, setRatioTest] = useState(false);
|
||||
|
||||
useEffect(() => {
|
||||
apiRequest("GET", "/api/type-agent/status")
|
||||
@@ -152,6 +157,7 @@ export default function AiTypeAgentPage() {
|
||||
action: runAction,
|
||||
ip: selectedAgentIp,
|
||||
patientId: selectedPatient.id,
|
||||
ratioTest,
|
||||
})
|
||||
.then((res) => res.json())
|
||||
.then((data) => {
|
||||
@@ -175,7 +181,7 @@ export default function AiTypeAgentPage() {
|
||||
if (liveRunId) socket.off(`type-agent:run:${liveRunId}`, handleProgress);
|
||||
};
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [isAutomated, runAction, selectedAgentIp, selectedPatient?.id]);
|
||||
}, [isAutomated, runAction, selectedAgentIp, selectedPatient?.id, ratioTest]);
|
||||
|
||||
const getPatientName = (patient: Patient) =>
|
||||
patient.firstName && patient.lastName
|
||||
@@ -264,6 +270,13 @@ export default function AiTypeAgentPage() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Temporary debug toggle — remove once the vision-based patient-row click is
|
||||
confirmed reliable */}
|
||||
<label className="flex items-center gap-2 text-xs text-muted-foreground">
|
||||
<Checkbox checked={ratioTest} onCheckedChange={(v) => setRatioTest(v === true)} />
|
||||
Ratio test: skip AI lookup for the patient row, move to fixed ratio (0.31, 0.21) instead
|
||||
</label>
|
||||
|
||||
{/* Connected PCs */}
|
||||
<div className="space-y-1.5">
|
||||
<p className="text-xs font-medium text-muted-foreground">Windows PC to control</p>
|
||||
|
||||
@@ -107,6 +107,13 @@ class DentalAgent:
|
||||
encoded = base64.b64encode(buf.getvalue()).decode("ascii")
|
||||
return {"image": encoded}
|
||||
|
||||
@sio.on("cmd:move", namespace="/agent")
|
||||
def on_move(data):
|
||||
# Moves the cursor without clicking — used to double-check a location (via a
|
||||
# follow-up cropped screenshot) before committing to a click.
|
||||
pyautogui.moveTo(data["x"], data["y"])
|
||||
return {"ok": True}
|
||||
|
||||
@sio.on("cmd:click", namespace="/agent")
|
||||
def on_click(data):
|
||||
pyautogui.click(data["x"], data["y"])
|
||||
|
||||
Reference in New Issue
Block a user