Skip to content

Commit 89029f0

Browse files
committed
Add safe typo tolerance for names and IO fields.
1 parent d4d0a60 commit 89029f0

2 files changed

Lines changed: 122 additions & 9 deletions

File tree

‎src/services/componentSearchIndex.test.ts‎

Lines changed: 52 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -322,6 +322,58 @@ describe("lexicalSearch", () => {
322322
expect(lexicalSearch(index, "classif")[0]?.digest).toBe("prefix");
323323
});
324324

325+
it("applies typo tolerance to component names and input/output fields", () => {
326+
const index = buildSearchIndex([
327+
makeSourced({
328+
digest: "name-typo",
329+
spec: {
330+
name: "filter_rows",
331+
inputs: [],
332+
outputs: [],
333+
implementation: { container: { image: "x" } },
334+
},
335+
}),
336+
makeSourced({
337+
digest: "io-typo",
338+
spec: {
339+
name: "prepare_data",
340+
inputs: [{ name: "dataset" }],
341+
outputs: [{ name: "clean_table" }],
342+
implementation: { container: { image: "x" } },
343+
},
344+
}),
345+
]);
346+
347+
expect(lexicalSearch(index, "filtr")[0]?.digest).toBe("name-typo");
348+
expect(lexicalSearch(index, "datset")[0]?.digest).toBe("io-typo");
349+
});
350+
351+
it("does not apply typo tolerance to descriptions or implementation text", () => {
352+
const index = buildSearchIndex([
353+
makeSourced({
354+
digest: "description-only",
355+
spec: {
356+
name: "generic_component",
357+
description: "Runs an xgboost classifier.",
358+
inputs: [],
359+
outputs: [],
360+
implementation: { container: { image: "x" } },
361+
},
362+
}),
363+
makeSourced({
364+
digest: "implementation-only",
365+
spec: {
366+
name: "generic_runner",
367+
inputs: [],
368+
outputs: [],
369+
implementation: { container: { image: "python:3.11-xgboost" } },
370+
},
371+
}),
372+
]);
373+
374+
expect(lexicalSearch(index, "xgbost")).toHaveLength(0);
375+
});
376+
325377
it("boosts rare tokens over common tokens", () => {
326378
// Both candidates contain BOTH query tokens (so the all-tokens bonus applies
327379
// equally) and differ only in which token sits in the high-weight name

‎src/services/componentSearchIndex.ts‎

Lines changed: 70 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -410,6 +410,7 @@ const FIELD_PHRASE_BONUS: Record<MatchField, number> = {
410410
};
411411

412412
const PREFIX_MATCH_BONUS_MULTIPLIER = 0.5;
413+
const FUZZY_MATCH_BONUS_MULTIPLIER = 0.75;
413414
const ALL_QUERY_TOKENS_BONUS = 6;
414415

415416
const SEARCH_FIELDS: MatchField[] = [
@@ -419,6 +420,7 @@ const SEARCH_FIELDS: MatchField[] = [
419420
"implementation",
420421
"metadata",
421422
];
423+
const FUZZY_SEARCH_FIELDS: MatchField[] = ["name", "io"];
422424

423425
interface SearchOptions {
424426
/** Max results to return. Default 20. */
@@ -434,6 +436,55 @@ function searchableTokens(text: string): string[] {
434436
return text.split(/[^a-z0-9]+/).filter(isNonEmptyString);
435437
}
436438

439+
function maxTypoDistance(token: string): number {
440+
if (token.length < 4) return 0;
441+
if (token.length < 7) return 1;
442+
return 2;
443+
}
444+
445+
function isEditDistanceAtMost(
446+
left: string,
447+
right: string,
448+
maxDistance: number,
449+
): boolean {
450+
if (maxDistance === 0) return left === right;
451+
if (Math.abs(left.length - right.length) > maxDistance) return false;
452+
453+
let previous = Array.from({ length: right.length + 1 }, (_, index) => index);
454+
for (let leftIndex = 1; leftIndex <= left.length; leftIndex++) {
455+
const current = [leftIndex];
456+
let rowMinimum = current[0];
457+
458+
for (let rightIndex = 1; rightIndex <= right.length; rightIndex++) {
459+
const substitutionCost =
460+
left[leftIndex - 1] === right[rightIndex - 1] ? 0 : 1;
461+
const value = Math.min(
462+
previous[rightIndex] + 1,
463+
current[rightIndex - 1] + 1,
464+
previous[rightIndex - 1] + substitutionCost,
465+
);
466+
current[rightIndex] = value;
467+
rowMinimum = Math.min(rowMinimum, value);
468+
}
469+
470+
if (rowMinimum > maxDistance) return false;
471+
previous = current;
472+
}
473+
474+
return previous[right.length] <= maxDistance;
475+
}
476+
477+
function hasFuzzyTokenMatch(fieldText: string, token: string): boolean {
478+
const maxDistance = maxTypoDistance(token);
479+
if (maxDistance === 0) return false;
480+
return searchableTokens(fieldText).some(
481+
(fieldToken) =>
482+
!fieldToken.includes(token) &&
483+
!token.includes(fieldToken) &&
484+
isEditDistanceAtMost(token, fieldToken, maxDistance),
485+
);
486+
}
487+
437488
function entryMatchesToken(entry: IndexEntry, token: string): boolean {
438489
return SEARCH_FIELDS.some((field) => entry.searchable[field].includes(token));
439490
}
@@ -493,17 +544,27 @@ function scoreEntry(
493544
const tokenWeight = tokenWeights.get(token) ?? 1;
494545
for (const field of SEARCH_FIELDS) {
495546
const fieldText = entry.searchable[field];
496-
if (!fieldText.includes(token)) continue;
497-
498547
const fieldWeight = FIELD_WEIGHTS[field];
499-
score += fieldWeight * tokenWeight;
500-
matched.add(field);
501548

502-
const hasPrefixMatch = fieldTokensFor(field).some((fieldToken) =>
503-
fieldToken.startsWith(token),
504-
);
505-
if (hasPrefixMatch) {
506-
score += fieldWeight * PREFIX_MATCH_BONUS_MULTIPLIER * tokenWeight;
549+
if (fieldText.includes(token)) {
550+
score += fieldWeight * tokenWeight;
551+
matched.add(field);
552+
553+
const hasPrefixMatch = fieldTokensFor(field).some((fieldToken) =>
554+
fieldToken.startsWith(token),
555+
);
556+
if (hasPrefixMatch) {
557+
score += fieldWeight * PREFIX_MATCH_BONUS_MULTIPLIER * tokenWeight;
558+
}
559+
continue;
560+
}
561+
562+
if (
563+
FUZZY_SEARCH_FIELDS.includes(field) &&
564+
hasFuzzyTokenMatch(fieldText, token)
565+
) {
566+
score += fieldWeight * FUZZY_MATCH_BONUS_MULTIPLIER * tokenWeight;
567+
matched.add(field);
507568
}
508569
}
509570
}

0 commit comments

Comments
 (0)