feat: expand ingredient-alias seed list to ~137 entries, ignore accents in name matching (v0.86.0)

Ingredient-alias matching (pantry, can-cook, auto-deduct, shopping-list generation) now covers common bilingual EN/FR ingredients across all 8 grocery categories, up from ~10 staples.

Also: matching now ignores accents as well as case (NFD-normalize + strip combining marks) in both the alias resolver (lib/ingredient-match.ts) and the older name-fallback matcher (pantry-shopping-match.ts) — "café"/"Café"/"cafe" and "Épinard"/"epinard" all recognize as the same ingredient without needing every accent variant manually listed as an alias.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Arnaud
2026-07-24 19:07:57 +02:00
parent 4ae0bd580e
commit b243cc3b8b
9 changed files with 206 additions and 39 deletions
+18 -16
View File
@@ -1,14 +1,17 @@
import { db, ingredients, sql } from "@epicure/db";
import { db, ingredients } from "@epicure/db";
export type IngredientAliasIndex = Map<string, string>;
// Case- and accent-insensitive so "café"/"cafe" and "Épinard"/"epinard"
// compare equal — NFD splits an accented letter into base letter +
// combining mark, then the combining marks (U+0300-U+036F) are stripped.
function normalize(name: string): string {
return name.trim().toLowerCase();
return name.trim().toLowerCase().normalize("NFD").replace(/[̀-ͯ]/g, "");
}
/**
* Loads every canonical ingredient's name + aliases into a flat
* lowercased-string -> canonical-ingredient-id map, once per request. Used
* normalized-string -> canonical-ingredient-id map, once per request. Used
* to recognize that "sel", "sel fin", and "table salt" are all the same
* ingredient, without requiring every recipe/pantry row to carry a stored
* ingredientId (they don't — this resolves purely from the free-text name
@@ -27,25 +30,24 @@ export async function loadIngredientAliasIndex(): Promise<IngredientAliasIndex>
}
/** Canonical ingredient id if `rawName` matches a known name/alias exactly
* (case/whitespace-insensitive); otherwise the normalized rawName itself,
* so unmatched items still compare equal to other unmatched items with the
* exact same text (today's behavior, unchanged for anything not seeded). */
* (case/accent/whitespace-insensitive); otherwise the normalized rawName
* itself, so unmatched items still compare equal to other unmatched items
* with the same text (today's behavior, unchanged for anything not seeded). */
export function resolveIngredientKey(rawName: string, index: IngredientAliasIndex): string {
const normalized = normalize(rawName);
return index.get(normalized) ?? normalized;
}
/** Single-name lookup (pantry add/edit) — a direct query rather than
* loading the whole table, since this runs once per add/rename rather than
* in a loop. Returns null when there's no canonical match, meaning the item
* stays a plain freeform pantry entry. */
/** Single-name lookup (pantry add/edit). Loads the same small table as
* loadIngredientAliasIndex and compares in JS rather than in SQL — accent
* stripping via NFD has no simple SQL equivalent without the `unaccent`
* extension, which isn't guaranteed to be installed. The ingredients table
* is small (tens to low hundreds of rows), so this is cheap. Returns null
* when there's no canonical match, meaning the item stays a plain freeform
* pantry entry. */
export async function findIngredientIdByName(rawName: string): Promise<string | null> {
const normalized = normalize(rawName);
if (!normalized) return null;
const [match] = await db
.select({ id: ingredients.id })
.from(ingredients)
.where(sql`lower(${ingredients.name}) = ${normalized} or exists (select 1 from unnest(${ingredients.aliases}) a where lower(a) = ${normalized})`)
.limit(1);
return match?.id ?? null;
const index = await loadIngredientAliasIndex();
return index.get(normalized) ?? null;
}