batchCooking/apps/api/src/lib/recipe-matching/ingredient-matcher.ts
Nicolas 5cc70d4244
Some checks failed
CI / test (push) Failing after 1s
CI / lint (push) Successful in 5m10s
CI / build (push) Successful in 6m29s
CI / intent-service-test (push) Successful in 13m15s
CI / e2e (push) Successful in 10m8s
feat(recipes): permet d'ajouter des ingredients hors-catalogue
Quand le catalogue seede ne couvre pas un ingredient, l'utilisateur pouvait
etre bloque (creation manuelle) ou perdre silencieusement la ligne (import).
Une ligne de recette accepte desormais `placeholderName` (texte libre) au
lieu de `ingredientId` : l'API cree une ligne `Ingredient` `isPlaceholder`
(cle `placeholder:<uuid>`, `displayName`, `createdById`) dans la transaction
de la recette, et emet `ingredient.placeholder_created`. Ces lignes sont
exclues de `GET /reference/ingredients` et de `ingredient-matcher`.

Front : bouton "Ajouter << ... >>" dans l'etat vide de `IngredientPicker`
(formulaire + import), badge "a completer" sur la ligne, helper
`ingredientLabel` applique partout ou un libelle d'ingredient est rendu.

Admin : `/admin/catalog/*` (+ page `apps/admin-web`) liste les placeholders
regroupes par nom normalise, "marquer traite" (`reviewedAt`) et purge des
orphelins. La promotion en vraie entree catalogue reste manuelle.

Migration `ingredient_placeholder` ecrite a la main (Postgres indisponible).
Suites Mocha DB-backed ecrites, non executees en session ; test pur
`normalizePlaceholderName` + Cypress admin-web/web verts.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-28 19:06:15 +02:00

471 lines
22 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import {
INGREDIENT_LABEL_SYNONYMS_EN,
INGREDIENT_LABEL_SYNONYMS_FR,
INGREDIENT_LABELS_EN,
INGREDIENT_LABELS_FR,
UNIT_LABELS_EN,
UNIT_LABELS_FR,
} from "@batch-cooking/shared";
import { prisma } from "../../db/prisma.js";
import { normalizeText } from "./tech-step-matcher.js";
/**
* Resolves the free-text `name`/`unit`/`quantity` a `RecipeSourceAdapter`
* lifts from a recipe source (`ParsedRecipeIngredient`) against our own
* `Ingredient`/`Unit` reference catalogs — the ingredient-side counterpart
* to `tech-step-matcher.ts`'s technique detection, built for the same
* reason: a source's raw text has no idea our catalogs even exist.
*
* Unlike tech steps (regex mappings hand-authored per technique),
* ingredient/unit labels are plain hand-written text, one table per
* supported `locale` (`INGREDIENT_LABELS_EN`/`INGREDIENT_LABELS_FR`/
* `UNIT_LABELS_EN`/`UNIT_LABELS_FR`, `packages/shared`) — matching them
* against arbitrary free text (extra adjectives, plurals, "large diced
* yellow onion" for a catalog entry that's just "Onion") needs its own,
* lighter algorithm: word-tokenize both sides, naively stem for plurals,
* then look for the catalog phrase's tokens as a contiguous run inside the
* ingredient text's tokens. The longest (most specific) matching catalog
* phrase wins, same "specificity resolves overlaps" idea as
* `matchTechSteps`, just without weights (there's no need to rank two
* *unrelated* ingredients — only a phrase against its own substrings, e.g.
* "chicken breast" beating bare "chicken").
*
* `matchIngredientName`/`matchUnit`/`extractQuantity` are pure, so they're
* unit-testable without a database (see `test/ingredient-matcher.test.ts`);
* `loadIngredientCatalog`/`loadUnitCatalog` are the only DB-touching pieces,
* meant to be fetched once per request and reused across every ingredient
* line, the same "don't requery per item" convention
* `tech-step-matcher.ts`'s `TechStepClassifierService` follows for its own
* one-time training pass.
*/
/** One `Ingredient` row trimmed to what {@link matchIngredientName} needs, alongside its matching label in whichever locale it was loaded for. */
export interface IngredientMatchEntry {
ingredientId: number;
/** Matching label — English (`INGREDIENT_LABELS_EN`) or French (`INGREDIENT_LABELS_FR`) depending on which locale {@link loadIngredientCatalog} was called with, e.g. `"Chicken breast"`/`"Blanc de poulet"` — matched against free text, never displayed. */
label: string;
}
/** One `Unit` row trimmed to what {@link matchUnit} needs, alongside its accepted spellings in whichever locale it was loaded for. */
export interface UnitMatchEntry {
unitId: number;
/** Accepted spellings — English (`UNIT_LABELS_EN`) or French (`UNIT_LABELS_FR`), e.g. `["tbsp", "tbs", "tablespoon", "tablespoons"]` or `["cuillère à soupe", "cas", ...]`. Unlike English, a French entry can be genuinely multi-word — see `matchUnit`'s own doc comment. */
synonyms: string[];
}
/**
* Naive English stemmer — strips a plural suffix so "tomato"/"tomatoes",
* "onion"/"onions", "cherry"/"cherries" compare equal after stemming.
* Deliberately not a real linguistic stemmer: it only has to be
* *consistent* (the same word always stems the same way) since both sides
* of every comparison go through it, not linguistically correct on its
* own — see the module doc comment.
*/
function stemWordEn(word: string): string {
if (word.endsWith("ies") && word.length > 4) return `${word.slice(0, -3)}y`;
if (word.endsWith("es") && word.length > 3) return word.slice(0, -2);
if (word.endsWith("s") && !word.endsWith("ss") && word.length > 3) return word.slice(0, -1);
return word;
}
/**
* Naive French stemmer — French regular plurals are overwhelmingly just
* "+s" on the singular (`"carotte"`/`"carottes"`, `"pomme"`/`"pommes"`),
* unlike English's several suffix patterns, so this only strips a single
* trailing "s". Deliberately *not* {@link stemWordEn}'s `"es"` rule reused
* here: applying it to French would silently corrupt any word whose
* singular itself ends in "e" plus a consonant before the final "s" — e.g.
* `"carottes"` would wrongly stem to `"carott"` (dropping the "e" that's
* actually part of the singular `"carotte"`) instead of `"carotte"`,
* exactly the class of near-miss that made ingredient matching
* French-locale silently broken before this stemmer existed (almost every
* regular French plural ends in "es" this way — it's not an edge case).
* Irregular plurals (`"cheval"`/`"chevaux"`, `"chou"`/`"choux"`) aren't
* handled — same "consistent, not linguistically perfect" trade-off as
* {@link stemWordEn}.
*/
function stemWordFr(word: string): string {
if (word.endsWith("s") && !word.endsWith("ss") && word.length > 3) return word.slice(0, -1);
return word;
}
/** Dispatches to {@link stemWordEn}/{@link stemWordFr} by `locale` — any locale other than `"fr"` uses the English rules (the long-standing default, unchanged for every existing caller that doesn't pass a locale at all). */
function stemWord(word: string, locale: string): string {
return locale === "fr" ? stemWordFr(word) : stemWordEn(word);
}
/** Lowercases, strips diacritics (via `normalizeText`) and splits `text` into stemmed word tokens — empty/non-letter runs are dropped, not kept as empty tokens. `locale` picks the stemming rules (see {@link stemWord}); defaults to `"en"`, the original behavior every pre-existing caller still gets without passing one. */
function tokenize(text: string, locale = "en"): string[] {
return normalizeText(text)
.split(/[^a-z]+/)
.filter((word) => word.length > 0)
.map((word) => stemWord(word, locale));
}
/** Whether `needle` appears as a contiguous run inside `haystack`, at any starting position. */
function containsSubsequence(haystack: string[], needle: string[]): boolean {
if (needle.length === 0 || needle.length > haystack.length) return false;
for (let start = 0; start <= haystack.length - needle.length; start++) {
if (needle.every((word, i) => haystack[start + i] === word)) return true;
}
return false;
}
/** One stemmed word from {@link tokenizeWithOffsets}, alongside its `[start, end)` span in the *original* (un-normalized) text it came from. */
interface OffsetToken {
word: string;
start: number;
end: number;
}
/** Matches a run of letters (any script, diacritics included) — the same "word" unit {@link tokenize} splits `normalizeText`'d text on (`/[^a-z]+/`), applied here directly to the *original* text instead so each token keeps its real character offsets. Digits/punctuation are never part of a run, same separator role they play for `tokenize` (a leading quantity is `extractQuantity`'s job, not this module's word-tokenizer's). */
const LETTER_RUN_PATTERN = /\p{L}+/gu;
/**
* {@link tokenize}'s positional twin: same stemmed/normalized words, but
* each one keeps the `[start, end)` span it occupies in `text` — needed by
* {@link findIngredientMentions} to report *where* a mention is, not just
* that the catalog has a matching label somewhere. Splitting the original
* text into letter-runs first (rather than normalizing the whole string up
* front, the way `tokenize` does, then losing track of offsets) works
* safely here because `normalizeText` only ever rewrites a character's own
* form (case/diacritics) — see `_DiacriticsNormalizer`'s doc comment on the
* Python side, ported from the same guarantee — never merges or splits
* words, so normalizing one already-isolated run in place can't shift its
* boundaries relative to the un-normalized text.
*/
function tokenizeWithOffsets(text: string, locale = "en"): OffsetToken[] {
const tokens: OffsetToken[] = [];
for (const match of text.matchAll(LETTER_RUN_PATTERN)) {
const raw = match[0];
const start = match.index ?? 0;
const word = stemWord(normalizeText(raw), locale);
if (word.length === 0) continue;
tokens.push({ word, start, end: start + raw.length });
}
return tokens;
}
/**
* Matches a quantity (integer/decimal/fraction/mixed number, same shapes as
* {@link extractQuantity}) immediately followed by an optional unit
* word/phrase (up to three words, e.g. "cuillères à soupe") and an optional
* connector ("de"/"d'"/"of"/"a"/"an"), anchored at the *end* of whatever
* string it's tested against (`$`) rather than the start. Anchoring at the
* end — not the start — is what lets {@link findQuantityBeforeIngredient}
* test the *whole* text preceding a mention without first having to guess
* where an unrelated preamble ("ajouter", "puis", an earlier sentence) ends
* and the quantity phrase begins: whatever doesn't fit the pattern
* immediately before the ingredient simply isn't part of the match, no
* separate boundary-finding step needed.
*/
const QUANTITY_BEFORE_INGREDIENT_PATTERN =
/(\d+\s+\d+\/\d+|\d+\/\d+|\d+(?:[.,]\d+)?)\s*((?:\p{L}+\s+){0,2}\p{L}*)\s*(?:de\s|d[']|of\s|a\s|an\s)?$/u;
/**
* Best-effort quantity+unit lookup for an ingredient mention {@link findIngredientMentions}
* just found at `mentionStart` in `text` — looks *only* at what immediately
* precedes the mention (see {@link QUANTITY_BEFORE_INGREDIENT_PATTERN}), the
* dominant French/English recipe phrasing ("200g de beurre", "2 cuillères à
* soupe d'huile", "3 œufs"). Both `null` when nothing recognizable precedes
* it (no leading digit at all) — same "no match, not an error" posture as
* {@link extractQuantity}. Doesn't detect a quantity that *follows* its
* ingredient ("du beurre, 50g") — an accepted gap, same trade-off
* {@link extractQuantity} already documents for the leading-only case it
* was built for.
*/
function findQuantityBeforeIngredient(
text: string,
mentionStart: number,
unitCatalog: UnitMatchEntry[],
locale: string,
): { quantity: number | null; unitId: number | null } {
const match = QUANTITY_BEFORE_INGREDIENT_PATTERN.exec(text.slice(0, mentionStart));
if (!match) return { quantity: null, unitId: null };
const { quantity } = extractQuantity(match[1] ?? "");
const unitId = matchUnit(match[2] ?? "", unitCatalog, locale);
return { quantity, unitId };
}
/** One ingredient mention {@link findIngredientMentions} found in a free-text clause, alongside its `[start, end)` span (same convention as `TechStepMatch`, `tech-step-matcher.ts`) and any quantity+unit resolved immediately before it (see {@link findQuantityBeforeIngredient}) — both `null` when the clause names the ingredient with no quantity ("ajouter le sel"). */
export interface IngredientMention {
ingredientId: number;
start: number;
end: number;
quantity: number | null;
unitId: number | null;
}
/**
* Scans `text` (typically one technique's clause, see `tech-step-matcher.ts`'s
* `splitIntoClauses`) for every mention of a catalog ingredient, left to
* right, non-overlapping — the free-text-*scanning* counterpart to
* {@link matchIngredientName} (which resolves one *already-isolated*
* ingredient-line string to a single winner, not several mentions spread
* across a longer text). Same "longest catalog label wins" rule as
* {@link matchIngredientName}, applied at every token position in turn: once
* a mention is found, scanning resumes right after it rather than
* considering a shorter label starting inside an already-matched longer one.
*
* `locale` must match whatever `ingredientCatalog`/`unitCatalog` were loaded
* in (see {@link loadIngredientCatalog}/{@link loadUnitCatalog}) — defaults
* to `"en"`, same as every other function in this module.
*/
export function findIngredientMentions(
text: string,
ingredientCatalog: IngredientMatchEntry[],
unitCatalog: UnitMatchEntry[],
locale = "en",
): IngredientMention[] {
const tokens = tokenizeWithOffsets(text, locale);
if (tokens.length === 0) return [];
const candidates = ingredientCatalog
.map((entry) => ({
ingredientId: entry.ingredientId,
labelTokens: tokenize(entry.label, locale),
}))
.filter((entry) => entry.labelTokens.length > 0);
const mentions: IngredientMention[] = [];
let i = 0;
while (i < tokens.length) {
let best: { ingredientId: number; tokenCount: number } | null = null;
for (const candidate of candidates) {
const { labelTokens } = candidate;
if (i + labelTokens.length > tokens.length) continue;
const matches = labelTokens.every((word, offset) => tokens[i + offset]?.word === word);
if (!matches) continue;
if (
best === null ||
labelTokens.length > best.tokenCount ||
(labelTokens.length === best.tokenCount && candidate.ingredientId < best.ingredientId)
) {
best = { ingredientId: candidate.ingredientId, tokenCount: labelTokens.length };
}
}
if (best === null) {
i += 1;
continue;
}
const startToken = tokens[i];
const endToken = tokens[i + best.tokenCount - 1];
if (startToken === undefined || endToken === undefined) {
// Unreachable — `best` was only ever set above after confirming
// `i + labelTokens.length <= tokens.length`, so both tokens exist.
// Satisfies `noUncheckedIndexedAccess`.
i += 1;
continue;
}
const { quantity, unitId } = findQuantityBeforeIngredient(
text,
startToken.start,
unitCatalog,
locale,
);
mentions.push({
ingredientId: best.ingredientId,
start: startToken.start,
end: endToken.end,
quantity,
unitId,
});
i += best.tokenCount;
}
return mentions;
}
/**
* Resolves free-text `name` (a `ParsedRecipeIngredient.name`, e.g. `"large
* diced yellow onions"`) to the best-matching `Ingredient` in `catalog`, or
* `null` if nothing matches. Among every catalog entry whose label's words
* all appear as a contiguous run in `name`, **in the same order**, the one
* with the most words wins (most specific — "chicken breast" over bare
* "chicken"); ties break on the lowest `ingredientId`, for a deterministic
* result independent of catalog order. `locale` must match whatever
* `catalog`'s labels were loaded in (see {@link loadIngredientCatalog}) —
* defaults to `"en"`.
*/
export function matchIngredientName(
name: string,
catalog: IngredientMatchEntry[],
locale = "en",
): number | null {
const nameTokens = tokenize(name, locale);
if (nameTokens.length === 0) return null;
let best: { ingredientId: number; tokenCount: number } | null = null;
for (const entry of catalog) {
const labelTokens = tokenize(entry.label, locale);
if (!containsSubsequence(nameTokens, labelTokens)) continue;
if (
best === null ||
labelTokens.length > best.tokenCount ||
(labelTokens.length === best.tokenCount && entry.ingredientId < best.ingredientId)
) {
best = {
ingredientId: entry.ingredientId,
tokenCount: labelTokens.length,
};
}
}
return best?.ingredientId ?? null;
}
/**
* Resolves free-text `unitText` (e.g. `"tbsp"`, `"Cups"`, `"cuillères à
* soupe de farine"`) to the best-matching `Unit` in `catalog`, or `null` if
* nothing matches. Same ordered-contiguous-run search as
* {@link matchIngredientName}, longest match wins — **not** the
* single-first-word equality check this function used before French
* support existed: every English unit synonym happens to be one word, so
* comparing only `unitText`'s first token against each *whole* synonym
* string used to be enough, but a French unit can be genuinely multi-word
* (`"cuillère à soupe"`, see `UNIT_LABELS_FR`) — a whole multi-word phrase
* (spaces and all) can never equal a single extracted token, so that
* approach would have silently matched nothing for any French unit
* requiring more than one word. `locale` must match whatever `catalog`'s
* synonyms were loaded in (see {@link loadUnitCatalog}) — defaults to
* `"en"`.
*/
export function matchUnit(
unitText: string,
catalog: UnitMatchEntry[],
locale = "en",
): number | null {
const textTokens = tokenize(unitText, locale);
if (textTokens.length === 0) return null;
let best: { unitId: number; tokenCount: number } | null = null;
for (const entry of catalog) {
for (const synonym of entry.synonyms) {
const synonymTokens = tokenize(synonym, locale);
if (!containsSubsequence(textTokens, synonymTokens)) continue;
if (
best === null ||
synonymTokens.length > best.tokenCount ||
(synonymTokens.length === best.tokenCount && entry.unitId < best.unitId)
) {
best = { unitId: entry.unitId, tokenCount: synonymTokens.length };
}
}
}
return best?.unitId ?? null;
}
/** What {@link extractQuantity} pulls out of a leading numeric expression, alongside what's left of the string after it. */
export interface ExtractedQuantity {
quantity: number | null;
/** `rawText` with the leading quantity (and any separating whitespace) removed — `rawText` unchanged if none was found. */
remainder: string;
}
// Leading "1 1/2", "1/2", "1.5", "1,5" or "2" (optionally followed by a
// hyphenated range like "2-3", in which case only the first number counts —
// good enough for a best-effort quantity, not meant to model ranges. The
// `[.,]` decimal separator already covers French recipe text ("1,5") as-is,
// same pattern used for English ("1.5") — no locale-specific handling
// needed here, unlike tokenize/stemWord above.
const LEADING_QUANTITY_PATTERN = /^(\d+)\s+(\d+)\/(\d+)|^(\d+)\/(\d+)|^(\d+(?:[.,]\d+)?)/;
/**
* Pulls a leading quantity off `rawText` (e.g. `"1 1/2 cups flour"` ->
* `{ quantity: 1.5, remainder: "cups flour" }`), supporting a plain
* integer/decimal, a simple fraction (`"1/2"`), or a mixed number
* (`"1 1/2"`). `quantity: null` (remainder === rawText, trimmed) when
* `rawText` doesn't start with a recognizable number — e.g. `"salt to
* taste"`, which has none.
*/
export function extractQuantity(rawText: string): ExtractedQuantity {
const trimmed = rawText.trim();
const match = LEADING_QUANTITY_PATTERN.exec(trimmed);
if (!match) return { quantity: null, remainder: trimmed };
let quantity: number;
if (match[1] !== undefined) {
// Mixed number: "1 1/2".
quantity = Number(match[1]) + Number(match[2]) / Number(match[3]);
} else if (match[4] !== undefined) {
// Simple fraction: "1/2".
quantity = Number(match[4]) / Number(match[5]);
} else {
// Plain integer/decimal: "2" or "1.5"/"1,5" — group 6 is guaranteed
// defined here (the only remaining alternative in the pattern), `?? ""`
// is just satisfying the indexed-access type, not a real fallback.
quantity = Number((match[6] ?? "").replace(",", "."));
}
return { quantity, remainder: trimmed.slice(match[0].length).trim() };
}
/**
* Per-locale matching label tables {@link loadIngredientCatalog}/
* {@link loadUnitCatalog} pick from — the only two locales with any
* matching data authored yet (see `packages/shared/src/data/`). A locale
* with no entry here (anything but `"en"`/`"fr"`) falls back to empty
* tables in both loaders below — the same "no matching-language data,
* degrade to doing nothing rather than guess" behavior `tech-step-matcher.ts`
* already has for a locale with no trained mappings, not a thrown error.
*/
const INGREDIENT_LABELS_BY_LOCALE: Record<string, Record<string, string>> = {
en: INGREDIENT_LABELS_EN,
fr: INGREDIENT_LABELS_FR,
};
const INGREDIENT_LABEL_SYNONYMS_BY_LOCALE: Record<string, Record<string, string[]>> = {
en: INGREDIENT_LABEL_SYNONYMS_EN,
fr: INGREDIENT_LABEL_SYNONYMS_FR,
};
const UNIT_LABELS_BY_LOCALE: Record<string, Record<string, string[]>> = {
en: UNIT_LABELS_EN,
fr: UNIT_LABELS_FR,
};
/** Loads the full `Ingredient` catalog as {@link IngredientMatchEntry}s for `locale` (default `"en"`) — one entry per key with an authored label in that locale, plus one extra entry per alternate wording (`INGREDIENT_LABEL_SYNONYMS_EN`/`_FR`, e.g. "Vanilla pod" alongside "Vanilla bean" — see issue #54) sharing the same `ingredientId`; `matchIngredientName` doesn't need to know synonyms exist, it just sees more candidate labels for the same ingredient. An ingredient with no label in `locale` yet is silently skipped, never a matching target — same for every ingredient when `locale` itself has no label table at all (see {@link INGREDIENT_LABELS_BY_LOCALE}). Meant to be fetched once per request and reused across every ingredient line, not re-queried per line. */
export async function loadIngredientCatalog(locale = "en"): Promise<IngredientMatchEntry[]> {
try {
const labels = INGREDIENT_LABELS_BY_LOCALE[locale] ?? {};
const synonyms = INGREDIENT_LABEL_SYNONYMS_BY_LOCALE[locale] ?? {};
const ingredients = await prisma.ingredient.findMany({
// Placeholder rows (user free-text, `placeholder:<uuid>` key) have no
// authored label so they'd be skipped by the `label === undefined`
// check below anyway — filtered here too so an import never even
// considers resolving one raw line to another line's placeholder.
where: { isPlaceholder: false },
select: { id: true, key: true },
});
const catalog: IngredientMatchEntry[] = [];
for (const ingredient of ingredients) {
const label = labels[ingredient.key];
if (label === undefined) continue;
catalog.push({ ingredientId: ingredient.id, label });
for (const synonym of synonyms[ingredient.key] ?? []) {
catalog.push({ ingredientId: ingredient.id, label: synonym });
}
}
return catalog;
} catch (err) {
// Rethrown as-is — the caller (`sources.service.ts`/`recipe-translation.ts`)
// already handles/logs failures centrally; this function just isn't
// allowed a bare `await` per the repo's async/try-catch convention.
throw err;
}
}
/** Loads the full `Unit` catalog as {@link UnitMatchEntry}s for `locale` (default `"en"`) — one entry per key with authored synonyms in that locale (see {@link UNIT_LABELS_BY_LOCALE}); a unit with none yet is silently skipped. Meant to be fetched once per request, same reasoning as {@link loadIngredientCatalog}. */
export async function loadUnitCatalog(locale = "en"): Promise<UnitMatchEntry[]> {
try {
const labels = UNIT_LABELS_BY_LOCALE[locale] ?? {};
const units = await prisma.unit.findMany({
select: { id: true, key: true },
});
const catalog: UnitMatchEntry[] = [];
for (const unit of units) {
const synonyms = labels[unit.key];
if (synonyms !== undefined) catalog.push({ unitId: unit.id, synonyms });
}
return catalog;
} catch (err) {
throw err; // see loadIngredientCatalog()'s catch comment above
}
}