fix: tolerate single-letter typos in CDT code text matching
Adds Levenshtein-based fuzzy word matching to lookupCdtCodes so misspellings like "comprehensiv exam" resolve the same as "comprehensive exam". Both the AI chat bot and the AI claim column share this matcher, so the fix applies to both. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -386,6 +386,48 @@ function parseRctCode(input: string): CdtMatch | null {
|
|||||||
return { code, description: row?.Description ?? code, input, toothNumber: String(toothNum) };
|
return { code, description: row?.Description ?? code, input, toothNumber: String(toothNum) };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Levenshtein edit distance between two strings.
|
||||||
|
*/
|
||||||
|
function editDistance(a: string, b: string): number {
|
||||||
|
const dp: number[][] = Array.from({ length: a.length + 1 }, () => new Array(b.length + 1).fill(0));
|
||||||
|
for (let i = 0; i <= a.length; i++) dp[i]![0] = i;
|
||||||
|
for (let j = 0; j <= b.length; j++) dp[0]![j] = j;
|
||||||
|
for (let i = 1; i <= a.length; i++) {
|
||||||
|
for (let j = 1; j <= b.length; j++) {
|
||||||
|
dp[i]![j] = a[i - 1] === b[j - 1]
|
||||||
|
? dp[i - 1]![j - 1]!
|
||||||
|
: 1 + Math.min(dp[i - 1]![j]!, dp[i]![j - 1]!, dp[i - 1]![j - 1]!);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dp[a.length]![b.length]!;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* True if two words are the same or a plausible typo of each other
|
||||||
|
* (one edit, only for words long enough that a 1-char slip is unambiguous).
|
||||||
|
*/
|
||||||
|
function wordsMatch(a: string, b: string): boolean {
|
||||||
|
if (a === b) return true;
|
||||||
|
if (a.length < 5 || b.length < 5) return false;
|
||||||
|
return editDistance(a, b) <= 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fuzzy lookup into ALIAS_MAP: tolerates a single typo per word
|
||||||
|
* (e.g. "comprehensiv exam" → "comprehensive exam").
|
||||||
|
*/
|
||||||
|
function fuzzyAliasLookup(cleaned: string): string | undefined {
|
||||||
|
if (ALIAS_MAP[cleaned]) return ALIAS_MAP[cleaned];
|
||||||
|
const queryWords = cleaned.split(/\s+/).filter(Boolean);
|
||||||
|
for (const key of Object.keys(ALIAS_MAP)) {
|
||||||
|
const keyWords = key.split(/\s+/).filter(Boolean);
|
||||||
|
if (keyWords.length !== queryWords.length) continue;
|
||||||
|
if (keyWords.every((w, i) => wordsMatch(w, queryWords[i]!))) return ALIAS_MAP[key];
|
||||||
|
}
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Score how well a set of query tokens matches a code's description tokens.
|
* Score how well a set of query tokens matches a code's description tokens.
|
||||||
* Each matched token contributes 1 point; shorter descriptions get a bonus
|
* Each matched token contributes 1 point; shorter descriptions get a bonus
|
||||||
@@ -394,7 +436,14 @@ function parseRctCode(input: string): CdtMatch | null {
|
|||||||
function score(queryTokens: string[], entry: { tokens: Set<string>; description: string }): number {
|
function score(queryTokens: string[], entry: { tokens: Set<string>; description: string }): number {
|
||||||
let hits = 0;
|
let hits = 0;
|
||||||
for (const t of queryTokens) {
|
for (const t of queryTokens) {
|
||||||
if (entry.tokens.has(t)) hits++;
|
if (entry.tokens.has(t)) {
|
||||||
|
hits++;
|
||||||
|
} else if (t.length >= 5) {
|
||||||
|
// Tolerate a single typo (e.g. "comprehensiv" → "comprehensive")
|
||||||
|
for (const dt of entry.tokens) {
|
||||||
|
if (wordsMatch(t, dt)) { hits++; break; }
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (hits === 0) return 0;
|
if (hits === 0) return 0;
|
||||||
// Tie-break: prefer shorter descriptions (more specific match)
|
// Tie-break: prefer shorter descriptions (more specific match)
|
||||||
@@ -425,8 +474,8 @@ function matchOne(input: string): CdtMatch | null {
|
|||||||
return { code, description: row?.Description ?? code, input };
|
return { code, description: row?.Description ?? code, input };
|
||||||
}
|
}
|
||||||
|
|
||||||
// Apply alias before tokenizing
|
// Apply alias before tokenizing (fuzzy-tolerant of single-letter typos)
|
||||||
const normalized = ALIAS_MAP[cleaned] ?? cleaned;
|
const normalized = fuzzyAliasLookup(cleaned) ?? cleaned;
|
||||||
const queryTokens = normalized
|
const queryTokens = normalized
|
||||||
.replace(/[^a-z0-9\s]/g, " ")
|
.replace(/[^a-z0-9\s]/g, " ")
|
||||||
.split(/\s+/)
|
.split(/\s+/)
|
||||||
|
|||||||
Reference in New Issue
Block a user