fix: tolerate single-letter typos in CDT code text matching

Adds Levenshtein-based fuzzy word matching to lookupCdtCodes so
misspellings like "comprehensiv exam" resolve the same as
"comprehensive exam". Both the AI chat bot and the AI claim column
share this matcher, so the fix applies to both.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-20 00:01:41 -04:00
co-authored by Claude Sonnet 5
parent 374f96399e
commit ad9d311a99
+52 -3
View File
@@ -386,6 +386,48 @@ function parseRctCode(input: string): CdtMatch | null {
return { code, description: row?.Description ?? code, input, toothNumber: String(toothNum) }; return { code, description: row?.Description ?? code, input, toothNumber: String(toothNum) };
} }
/**
* Levenshtein edit distance between two strings.
*/
function editDistance(a: string, b: string): number {
const dp: number[][] = Array.from({ length: a.length + 1 }, () => new Array(b.length + 1).fill(0));
for (let i = 0; i <= a.length; i++) dp[i]![0] = i;
for (let j = 0; j <= b.length; j++) dp[0]![j] = j;
for (let i = 1; i <= a.length; i++) {
for (let j = 1; j <= b.length; j++) {
dp[i]![j] = a[i - 1] === b[j - 1]
? dp[i - 1]![j - 1]!
: 1 + Math.min(dp[i - 1]![j]!, dp[i]![j - 1]!, dp[i - 1]![j - 1]!);
}
}
return dp[a.length]![b.length]!;
}
/**
* True if two words are the same or a plausible typo of each other
* (one edit, only for words long enough that a 1-char slip is unambiguous).
*/
function wordsMatch(a: string, b: string): boolean {
if (a === b) return true;
if (a.length < 5 || b.length < 5) return false;
return editDistance(a, b) <= 1;
}
/**
* Fuzzy lookup into ALIAS_MAP: tolerates a single typo per word
* (e.g. "comprehensiv exam" → "comprehensive exam").
*/
function fuzzyAliasLookup(cleaned: string): string | undefined {
if (ALIAS_MAP[cleaned]) return ALIAS_MAP[cleaned];
const queryWords = cleaned.split(/\s+/).filter(Boolean);
for (const key of Object.keys(ALIAS_MAP)) {
const keyWords = key.split(/\s+/).filter(Boolean);
if (keyWords.length !== queryWords.length) continue;
if (keyWords.every((w, i) => wordsMatch(w, queryWords[i]!))) return ALIAS_MAP[key];
}
return undefined;
}
/** /**
* Score how well a set of query tokens matches a code's description tokens. * Score how well a set of query tokens matches a code's description tokens.
* Each matched token contributes 1 point; shorter descriptions get a bonus * Each matched token contributes 1 point; shorter descriptions get a bonus
@@ -394,7 +436,14 @@ function parseRctCode(input: string): CdtMatch | null {
function score(queryTokens: string[], entry: { tokens: Set<string>; description: string }): number { function score(queryTokens: string[], entry: { tokens: Set<string>; description: string }): number {
let hits = 0; let hits = 0;
for (const t of queryTokens) { for (const t of queryTokens) {
if (entry.tokens.has(t)) hits++; if (entry.tokens.has(t)) {
hits++;
} else if (t.length >= 5) {
// Tolerate a single typo (e.g. "comprehensiv" → "comprehensive")
for (const dt of entry.tokens) {
if (wordsMatch(t, dt)) { hits++; break; }
}
}
} }
if (hits === 0) return 0; if (hits === 0) return 0;
// Tie-break: prefer shorter descriptions (more specific match) // Tie-break: prefer shorter descriptions (more specific match)
@@ -425,8 +474,8 @@ function matchOne(input: string): CdtMatch | null {
return { code, description: row?.Description ?? code, input }; return { code, description: row?.Description ?? code, input };
} }
// Apply alias before tokenizing // Apply alias before tokenizing (fuzzy-tolerant of single-letter typos)
const normalized = ALIAS_MAP[cleaned] ?? cleaned; const normalized = fuzzyAliasLookup(cleaned) ?? cleaned;
const queryTokens = normalized const queryTokens = normalized
.replace(/[^a-z0-9\s]/g, " ") .replace(/[^a-z0-9\s]/g, " ")
.split(/\s+/) .split(/\s+/)