Le matching par regex ne généralisait jamais au-delà de son propre vocabulaire — une étape décrivant la fonte du beurre comme "jusqu'à ce que le beurre ait disparu dans la poêle" ne contient aucun verbe sur lequel une regex pourrait s'ancrer, alors que le sens est sans ambiguïté. Nouveau pipeline en 3 étapes (TechStepClassifierService, node-nlp 4.27.0 — la 5.x est encore alpha, non retenue) : 1. NER (entités enum) trouve les mentions candidates + leur position exacte, à partir de listes de synonymes (tech-step-training-data.ts) plutôt que de regex écrites à la main. ner.threshold: 1 (exact, après normalisation) — le défaut à 0.8 faisait matcher "faire" (verbe auxiliaire omniprésent) contre "frire" par pure proximité de chaîne. 2. La description est découpée en clauses autour de ces candidats (splitIntoClauses, pure/testable sans modèle). 3. Le NlpManager classe chaque clause individuellement, entraîné sur des phrases qui n'emploient jamais le verbe de la technique — c'est ce qui apporte la compréhension du sens. En dessous de CONFIDENCE_THRESHOLD (0.65, ajusté empiriquement), retombe sur la technique impliquée par l'ancre NER plutôt que d'abandonner un match clairement ancré sur un mot-clé. TechStepMapping (table de regex par technique/locale) supprimée — migration 20260821130000_drop_tech_step_mapping — plus aucune table n'est interrogée à l'exécution, les données de matching vivent en code. TECH_STEPS (reference-seed-data.ts) simplifié en simple liste de uid, les mappings ayant disparu. Deux pièges trouvés en construisant ce pipeline, corrigés à la source : - db/prisma.ts construisait PrismaClient sans importer config/env.ts — un run de test isolé pouvait faire gagner la course au .env interne de Prisma (dev) contre .env.test. Fixé en important config/env.js en tout premier, pour effet de bord. - NlpManager a autoSave/autoLoad: true par défaut — persiste le modèle entraîné dans model.nlp et le recharge au lieu de ré-entraîner au prochain démarrage. Les deux désactivés explicitement (sinon un modèle obsolète masquerait silencieusement toute mise à jour du corpus/seuil) ; model.nlp ajouté au .gitignore en garde-fou. apps/api/src/db/prisma.ts, recipe.service.ts, sources.service.ts et recipe-translation.ts adaptés à la matching async (le classifieur entraîné remplace le couple loadTechStepMappingRules+matchTechStepSpans synchrone) ; server.ts appelle techStepClassifier.warmUp() avant d'accepter du trafic (le tout premier appel réel à NlpManager.process() charge les ressources par langue de node-nlp, plusieurs secondes). Vérifié : tsc --noEmit, biome check (0 erreur), build complet des 6 packages, 308 tests API (dont un test-support/reset-db.ts corrigé — référençait encore tech_step_mapping dans son TRUNCATE). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
478 lines
17 KiB
TypeScript
478 lines
17 KiB
TypeScript
import { expect } from "chai";
|
|
import { prisma } from "../../src/db/prisma.js";
|
|
import type {
|
|
IngredientMatchEntry,
|
|
UnitMatchEntry,
|
|
} from "../../src/lib/recipe-matching/ingredient-matcher.js";
|
|
import {
|
|
mergeDuplicateIngredients,
|
|
type TranslatedRecipeIngredient,
|
|
translateRecipe,
|
|
translateRecipeIngredients,
|
|
translateRecipeSteps,
|
|
type UnitConversionEntry,
|
|
} from "../../src/lib/recipe-matching/recipe-translation.js";
|
|
import type {
|
|
ParsedRecipe,
|
|
ParsedRecipeIngredient,
|
|
} from "../../src/lib/recipe-sources/recipe-source-adapter.js";
|
|
import { resetDatabase } from "../../test-support/reset-db.js";
|
|
|
|
/** A minimal fixture `ParsedRecipe` — only `steps` (built from `descriptions`) matters for most tests here, the rest is filler to prove it survives translation untouched. */
|
|
function buildParsedRecipe(descriptions: string[]): ParsedRecipe {
|
|
return {
|
|
name: "Test Recipe",
|
|
description: "A recipe for testing",
|
|
picture: "https://example.test/recipe.jpg",
|
|
portions: 4,
|
|
sourceUrl: "https://example.test/recipes/1",
|
|
ingredients: [{ rawText: "1 egg", quantity: null, unit: null, name: "egg" }],
|
|
steps: descriptions.map((description, i) => ({
|
|
description,
|
|
picture: i === 0 ? "https://example.test/step1.jpg" : null,
|
|
})),
|
|
};
|
|
}
|
|
|
|
describe("recipe-translation", () => {
|
|
// `translateRecipeSteps` now goes through `techStepClassifier` (a
|
|
// trained model, not a pure regex test against a caller-supplied
|
|
// mapping list — see `tech-step-matcher.ts`), so these tests exercise
|
|
// the real training corpus (`tech-step-training-data.ts`) against a real
|
|
// `TechStep` catalog rather than synthetic fixtures — same posture
|
|
// `tech-step-matcher.test.ts`'s own `techStepClassifier` describe block
|
|
// takes, for the same reason.
|
|
describe("translateRecipeSteps", () => {
|
|
let simmerId: number;
|
|
let preheatId: number;
|
|
let meltId: number;
|
|
|
|
beforeEach(async () => {
|
|
await resetDatabase();
|
|
simmerId = (await prisma.techStep.findFirstOrThrow({ where: { key: "simmer" } })).id;
|
|
preheatId = (await prisma.techStep.findFirstOrThrow({ where: { key: "preheat" } })).id;
|
|
meltId = (await prisma.techStep.findFirstOrThrow({ where: { key: "melt" } })).id;
|
|
});
|
|
|
|
after(async () => {
|
|
await prisma.$disconnect();
|
|
});
|
|
|
|
it("declares each step's technique sequence, preserving order", async () => {
|
|
const recipe = buildParsedRecipe([
|
|
"Préchauffer la poêle, puis faire fondre le beurre",
|
|
"Servir immédiatement",
|
|
"Faire mijoter à feu doux",
|
|
]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([
|
|
[preheatId, meltId],
|
|
[],
|
|
[simmerId],
|
|
]);
|
|
});
|
|
|
|
it("leaves description/picture untouched on each step", async () => {
|
|
const recipe = buildParsedRecipe(["Faire mijoter à feu doux", "Servir immédiatement"]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.steps[0]).to.deep.equal({
|
|
description: "Faire mijoter à feu doux",
|
|
picture: "https://example.test/step1.jpg",
|
|
techStepIds: [simmerId],
|
|
});
|
|
expect(translated.steps[1]).to.deep.equal({
|
|
description: "Servir immédiatement",
|
|
picture: null,
|
|
techStepIds: [],
|
|
});
|
|
});
|
|
|
|
it("passes every other field through unchanged", async () => {
|
|
const recipe = buildParsedRecipe(["Servir immédiatement"]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.name).to.equal(recipe.name);
|
|
expect(translated.description).to.equal(recipe.description);
|
|
expect(translated.picture).to.equal(recipe.picture);
|
|
expect(translated.portions).to.equal(recipe.portions);
|
|
expect(translated.sourceUrl).to.equal(recipe.sourceUrl);
|
|
});
|
|
|
|
it("stubs every ingredient's ingredientId/unitId to null, leaving the rest of it untouched — actual ingredient matching is translateRecipeIngredients' job", async () => {
|
|
const recipe = buildParsedRecipe(["Servir immédiatement"]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.ingredients).to.deep.equal([
|
|
{ ...recipe.ingredients[0], ingredientId: null, unitId: null },
|
|
]);
|
|
});
|
|
|
|
it("gives every step an empty sequence when nothing in it means a known technique", async () => {
|
|
const recipe = buildParsedRecipe([
|
|
"Servir immédiatement",
|
|
"Ranger les couverts dans le tiroir",
|
|
]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([[], []]);
|
|
});
|
|
|
|
it("handles a recipe with no steps without error", async () => {
|
|
const recipe = buildParsedRecipe([]);
|
|
|
|
const translated = await translateRecipeSteps(recipe, "fr");
|
|
|
|
expect(translated.steps).to.deep.equal([]);
|
|
});
|
|
});
|
|
|
|
describe("translateRecipeIngredients", () => {
|
|
const tomato: IngredientMatchEntry = { ingredientId: 1, label: "Tomato" };
|
|
const chicken: IngredientMatchEntry = { ingredientId: 2, label: "Chicken" };
|
|
const chickenBreast: IngredientMatchEntry = { ingredientId: 3, label: "Chicken breast" };
|
|
const onion: IngredientMatchEntry = { ingredientId: 4, label: "Onion" };
|
|
|
|
const gram: UnitMatchEntry = { unitId: 10, synonyms: ["g", "gram", "grams"] };
|
|
const cup: UnitMatchEntry = { unitId: 11, synonyms: ["cup", "cups"] };
|
|
|
|
function buildIngredient(
|
|
overrides: Partial<ParsedRecipeIngredient> & { rawText: string; name: string },
|
|
): ParsedRecipeIngredient {
|
|
return { quantity: null, unit: null, ...overrides };
|
|
}
|
|
|
|
it("resolves ingredientId from free-text name, tolerating extra descriptive words and plurals", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[
|
|
buildIngredient({
|
|
rawText: "2 large diced yellow onions",
|
|
name: "large diced yellow onions",
|
|
}),
|
|
],
|
|
[tomato, chicken, chickenBreast, onion],
|
|
[],
|
|
);
|
|
|
|
expect(translated[0].ingredientId).to.equal(onion.ingredientId);
|
|
});
|
|
|
|
it("prefers the more specific multi-word label over a shorter one it contains", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[
|
|
buildIngredient({
|
|
rawText: "2 boneless chicken breasts",
|
|
name: "boneless chicken breasts",
|
|
}),
|
|
],
|
|
[tomato, chicken, chickenBreast, onion],
|
|
[],
|
|
);
|
|
|
|
expect(translated[0].ingredientId).to.equal(chickenBreast.ingredientId);
|
|
});
|
|
|
|
it("returns a null ingredientId when nothing in the catalog matches", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[buildIngredient({ rawText: "1 mango", name: "mango" })],
|
|
[tomato, chicken, chickenBreast, onion],
|
|
[],
|
|
);
|
|
|
|
expect(translated[0].ingredientId).to.equal(null);
|
|
});
|
|
|
|
it("extracts a mixed-number quantity and unit from rawText when the source left them null", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[buildIngredient({ rawText: "1 1/2 cups chicken breast", name: "chicken breast" })],
|
|
[chickenBreast],
|
|
[cup],
|
|
);
|
|
|
|
expect(translated[0].quantity).to.equal(1.5);
|
|
expect(translated[0].unitId).to.equal(cup.unitId);
|
|
});
|
|
|
|
it("trusts the source's own quantity/unit over re-deriving them from rawText", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[
|
|
buildIngredient({
|
|
rawText: "some raw text that happens to mention cups",
|
|
name: "tomato",
|
|
quantity: 3,
|
|
unit: "g",
|
|
}),
|
|
],
|
|
[tomato],
|
|
[gram, cup],
|
|
);
|
|
|
|
expect(translated[0].quantity).to.equal(3);
|
|
expect(translated[0].unitId).to.equal(gram.unitId);
|
|
});
|
|
|
|
it("falls back to the generic 'piece' unit when a quantity was found but no unit word was (issue #53)", () => {
|
|
const piece: UnitMatchEntry = { unitId: 12, synonyms: ["piece", "pieces", "pc", "pcs"] };
|
|
|
|
const translated = translateRecipeIngredients(
|
|
[buildIngredient({ rawText: "4 Egg Yolks", name: "Egg Yolks" })],
|
|
[],
|
|
[gram, cup, piece],
|
|
);
|
|
|
|
expect(translated[0].quantity).to.equal(4);
|
|
expect(translated[0].unitId).to.equal(piece.unitId);
|
|
});
|
|
|
|
it("does not fall back to 'piece' when there's no quantity at all to count", () => {
|
|
const piece: UnitMatchEntry = { unitId: 12, synonyms: ["piece", "pieces", "pc", "pcs"] };
|
|
|
|
const translated = translateRecipeIngredients(
|
|
[buildIngredient({ rawText: "salt to taste", name: "salt" })],
|
|
[],
|
|
[piece],
|
|
);
|
|
|
|
expect(translated[0].quantity).to.equal(null);
|
|
expect(translated[0].unitId).to.equal(null);
|
|
});
|
|
|
|
it("leaves quantity/unitId null when rawText has neither a leading number nor a recognizable unit", () => {
|
|
const translated = translateRecipeIngredients(
|
|
[buildIngredient({ rawText: "salt to taste", name: "salt" })],
|
|
[],
|
|
[],
|
|
);
|
|
|
|
expect(translated[0].quantity).to.equal(null);
|
|
expect(translated[0].unitId).to.equal(null);
|
|
});
|
|
|
|
it("leaves rawText/name/description untouched", () => {
|
|
const ingredient = buildIngredient({ rawText: "1 cup onions", name: "onions" });
|
|
|
|
const translated = translateRecipeIngredients([ingredient], [onion], [cup]);
|
|
|
|
expect(translated[0].rawText).to.equal(ingredient.rawText);
|
|
expect(translated[0].name).to.equal(ingredient.name);
|
|
});
|
|
});
|
|
|
|
describe("mergeDuplicateIngredients", () => {
|
|
const gram: UnitConversionEntry = { id: 1, type: "MASS", toBaseFactor: 1 };
|
|
const kilogram: UnitConversionEntry = { id: 2, type: "MASS", toBaseFactor: 1000 };
|
|
const milliliter: UnitConversionEntry = { id: 3, type: "VOLUME", toBaseFactor: 1 };
|
|
const piece: UnitConversionEntry = { id: 4, type: "COUNT", toBaseFactor: 1 };
|
|
const slice: UnitConversionEntry = { id: 5, type: "COUNT", toBaseFactor: 1 };
|
|
|
|
function buildLine(
|
|
overrides: Partial<TranslatedRecipeIngredient> & { rawText: string },
|
|
): TranslatedRecipeIngredient {
|
|
return {
|
|
name: overrides.rawText,
|
|
quantity: null,
|
|
unit: null,
|
|
ingredientId: null,
|
|
unitId: null,
|
|
...overrides,
|
|
};
|
|
}
|
|
|
|
it("sums the quantity of two lines resolving to the same ingredient, same unit (issue #53 follow-up)", () => {
|
|
const merged = mergeDuplicateIngredients(
|
|
[
|
|
buildLine({ rawText: "100g Sugar", ingredientId: 1, quantity: 100, unitId: gram.id }),
|
|
buildLine({ rawText: "45g Sugar", ingredientId: 1, quantity: 45, unitId: gram.id }),
|
|
],
|
|
[gram, kilogram, milliliter, piece, slice],
|
|
);
|
|
|
|
expect(merged).to.have.length(1);
|
|
expect(merged[0].quantity).to.equal(145);
|
|
expect(merged[0].unitId).to.equal(gram.id);
|
|
expect(merged[0].rawText).to.equal("100g Sugar + 45g Sugar");
|
|
});
|
|
|
|
it("converts through toBaseFactor when the duplicate uses a different unit of the same type", () => {
|
|
const merged = mergeDuplicateIngredients(
|
|
[
|
|
buildLine({ rawText: "500g Flour", ingredientId: 1, quantity: 500, unitId: gram.id }),
|
|
buildLine({
|
|
rawText: "0.5kg Flour",
|
|
ingredientId: 1,
|
|
quantity: 0.5,
|
|
unitId: kilogram.id,
|
|
}),
|
|
],
|
|
[gram, kilogram, milliliter, piece, slice],
|
|
);
|
|
|
|
expect(merged).to.have.length(1);
|
|
expect(merged[0].quantity).to.equal(1000);
|
|
expect(merged[0].unitId).to.equal(gram.id);
|
|
});
|
|
|
|
it("keeps duplicate COUNT-unit lines separate rather than guessing — a slice isn't a piece", () => {
|
|
const merged = mergeDuplicateIngredients(
|
|
[
|
|
buildLine({ rawText: "2 piece Bread", ingredientId: 1, quantity: 2, unitId: piece.id }),
|
|
buildLine({ rawText: "3 slice Bread", ingredientId: 1, quantity: 3, unitId: slice.id }),
|
|
],
|
|
[gram, kilogram, milliliter, piece, slice],
|
|
);
|
|
|
|
expect(merged).to.have.length(2);
|
|
});
|
|
|
|
it("keeps duplicate lines with incompatible unit types separate (MASS vs VOLUME)", () => {
|
|
const merged = mergeDuplicateIngredients(
|
|
[
|
|
buildLine({ rawText: "200g Milk", ingredientId: 1, quantity: 200, unitId: gram.id }),
|
|
buildLine({
|
|
rawText: "200ml Milk",
|
|
ingredientId: 1,
|
|
quantity: 200,
|
|
unitId: milliliter.id,
|
|
}),
|
|
],
|
|
[gram, kilogram, milliliter, piece, slice],
|
|
);
|
|
|
|
expect(merged).to.have.length(2);
|
|
});
|
|
|
|
it("never merges two unresolved lines (ingredientId: null) together", () => {
|
|
const merged = mergeDuplicateIngredients(
|
|
[
|
|
buildLine({ rawText: "1 vanilla pod", quantity: 1 }),
|
|
buildLine({ rawText: "1 vanilla pod", quantity: 1 }),
|
|
],
|
|
[gram, kilogram, milliliter, piece, slice],
|
|
);
|
|
|
|
expect(merged).to.have.length(2);
|
|
});
|
|
|
|
it("leaves non-duplicate lines untouched and preserves order", () => {
|
|
const sugar = buildLine({ rawText: "Sugar", ingredientId: 1, quantity: 1, unitId: gram.id });
|
|
const salt = buildLine({ rawText: "Salt", ingredientId: 2, quantity: 1, unitId: gram.id });
|
|
|
|
const merged = mergeDuplicateIngredients([sugar, salt], [gram, kilogram, milliliter]);
|
|
|
|
expect(merged).to.deep.equal([sugar, salt]);
|
|
});
|
|
});
|
|
|
|
describe("translateRecipe", () => {
|
|
beforeEach(async () => {
|
|
await resetDatabase();
|
|
});
|
|
|
|
after(async () => {
|
|
await prisma.$disconnect();
|
|
});
|
|
|
|
it("resolves real TechStep ids from the seeded French catalog", async () => {
|
|
const simmer = await prisma.techStep.findFirstOrThrow({ where: { key: "simmer" } });
|
|
const chop = await prisma.techStep.findFirstOrThrow({ where: { key: "chop" } });
|
|
const recipe = buildParsedRecipe(["Hacher les oignons", "Faire mijoter à feu doux"]);
|
|
|
|
const translated = await translateRecipe(recipe, "fr");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([
|
|
[chop.id],
|
|
[simmer.id],
|
|
]);
|
|
});
|
|
|
|
it("finds nothing for English text against the French catalog — locales are separate rule sets, never mixed", async () => {
|
|
// A step lifted verbatim from a real TheMealDB recipe.
|
|
const recipe = buildParsedRecipe([
|
|
"Bring a large saucepan of salted water to the boil",
|
|
"Chop the onions finely",
|
|
]);
|
|
|
|
const translated = await translateRecipe(recipe, "fr");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([[], []]);
|
|
});
|
|
|
|
it("resolves real TechStep ids from the seeded English catalog", async () => {
|
|
const boil = await prisma.techStep.findFirstOrThrow({ where: { key: "boil" } });
|
|
const chop = await prisma.techStep.findFirstOrThrow({ where: { key: "chop" } });
|
|
// Same two steps as the French/English mismatch test above, this
|
|
// time matched against the matching-language catalog.
|
|
const recipe = buildParsedRecipe([
|
|
"Bring a large saucepan of salted water to the boil",
|
|
"Chop the onions finely",
|
|
]);
|
|
|
|
const translated = await translateRecipe(recipe, "en");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([
|
|
[boil.id],
|
|
[chop.id],
|
|
]);
|
|
});
|
|
|
|
it("finds nothing for a locale with no mappings at all", async () => {
|
|
const recipe = buildParsedRecipe(["Faire mijoter à feu doux"]);
|
|
|
|
const translated = await translateRecipe(recipe, "de");
|
|
|
|
expect(translated.steps.map((step) => step.techStepIds)).to.deep.equal([[]]);
|
|
});
|
|
|
|
it("resolves real Ingredient/Unit ids from the seeded English catalog for an 'en' translation", async () => {
|
|
const onion = await prisma.ingredient.findFirstOrThrow({ where: { key: "onion" } });
|
|
const cup = await prisma.unit.findFirstOrThrow({ where: { key: "cup" } });
|
|
const recipe: ParsedRecipe = {
|
|
...buildParsedRecipe(["Chop the onions finely"]),
|
|
ingredients: [
|
|
{ rawText: "1 cup onions, chopped", quantity: null, unit: null, name: "onions" },
|
|
],
|
|
};
|
|
|
|
const translated = await translateRecipe(recipe, "en");
|
|
|
|
expect(translated.ingredients).to.deep.equal([
|
|
{
|
|
rawText: "1 cup onions, chopped",
|
|
quantity: 1,
|
|
unit: null,
|
|
name: "onions",
|
|
ingredientId: onion.id,
|
|
unitId: cup.id,
|
|
},
|
|
]);
|
|
});
|
|
|
|
it("leaves ingredients untouched (no quantity extraction either) for a non-English locale — no matching data exists yet, and the DB isn't even queried for it", async () => {
|
|
const recipe: ParsedRecipe = {
|
|
...buildParsedRecipe(["Faire mijoter à feu doux"]),
|
|
ingredients: [
|
|
{ rawText: "1 cup onions, chopped", quantity: null, unit: null, name: "onions" },
|
|
],
|
|
};
|
|
|
|
const translated = await translateRecipe(recipe, "fr");
|
|
|
|
expect(translated.ingredients).to.deep.equal([
|
|
{
|
|
rawText: "1 cup onions, chopped",
|
|
quantity: null,
|
|
unit: null,
|
|
name: "onions",
|
|
ingredientId: null,
|
|
unitId: null,
|
|
},
|
|
]);
|
|
});
|
|
});
|
|
});
|