diff --git a/apps/Backend/src/ai/cdt-lookup.ts b/apps/Backend/src/ai/cdt-lookup.ts index d641ab85..a76ecc4b 100644 --- a/apps/Backend/src/ai/cdt-lookup.ts +++ b/apps/Backend/src/ai/cdt-lookup.ts @@ -386,6 +386,48 @@ function parseRctCode(input: string): CdtMatch | null { return { code, description: row?.Description ?? code, input, toothNumber: String(toothNum) }; } +/** + * Levenshtein edit distance between two strings. + */ +function editDistance(a: string, b: string): number { + const dp: number[][] = Array.from({ length: a.length + 1 }, () => new Array(b.length + 1).fill(0)); + for (let i = 0; i <= a.length; i++) dp[i]![0] = i; + for (let j = 0; j <= b.length; j++) dp[0]![j] = j; + for (let i = 1; i <= a.length; i++) { + for (let j = 1; j <= b.length; j++) { + dp[i]![j] = a[i - 1] === b[j - 1] + ? dp[i - 1]![j - 1]! + : 1 + Math.min(dp[i - 1]![j]!, dp[i]![j - 1]!, dp[i - 1]![j - 1]!); + } + } + return dp[a.length]![b.length]!; +} + +/** + * True if two words are the same or a plausible typo of each other + * (one edit, only for words long enough that a 1-char slip is unambiguous). + */ +function wordsMatch(a: string, b: string): boolean { + if (a === b) return true; + if (a.length < 5 || b.length < 5) return false; + return editDistance(a, b) <= 1; +} + +/** + * Fuzzy lookup into ALIAS_MAP: tolerates a single typo per word + * (e.g. "comprehensiv exam" → "comprehensive exam"). + */ +function fuzzyAliasLookup(cleaned: string): string | undefined { + if (ALIAS_MAP[cleaned]) return ALIAS_MAP[cleaned]; + const queryWords = cleaned.split(/\s+/).filter(Boolean); + for (const key of Object.keys(ALIAS_MAP)) { + const keyWords = key.split(/\s+/).filter(Boolean); + if (keyWords.length !== queryWords.length) continue; + if (keyWords.every((w, i) => wordsMatch(w, queryWords[i]!))) return ALIAS_MAP[key]; + } + return undefined; +} + /** * Score how well a set of query tokens matches a code's description tokens. * Each matched token contributes 1 point; shorter descriptions get a bonus @@ -394,7 +436,14 @@ function parseRctCode(input: string): CdtMatch | null { function score(queryTokens: string[], entry: { tokens: Set; description: string }): number { let hits = 0; for (const t of queryTokens) { - if (entry.tokens.has(t)) hits++; + if (entry.tokens.has(t)) { + hits++; + } else if (t.length >= 5) { + // Tolerate a single typo (e.g. "comprehensiv" → "comprehensive") + for (const dt of entry.tokens) { + if (wordsMatch(t, dt)) { hits++; break; } + } + } } if (hits === 0) return 0; // Tie-break: prefer shorter descriptions (more specific match) @@ -425,8 +474,8 @@ function matchOne(input: string): CdtMatch | null { return { code, description: row?.Description ?? code, input }; } - // Apply alias before tokenizing - const normalized = ALIAS_MAP[cleaned] ?? cleaned; + // Apply alias before tokenizing (fuzzy-tolerant of single-letter typos) + const normalized = fuzzyAliasLookup(cleaned) ?? cleaned; const queryTokens = normalized .replace(/[^a-z0-9\s]/g, " ") .split(/\s+/)