From 8f94a1acda9a994bc84f436273f242891a88df0c Mon Sep 17 00:00:00 2001 From: Jules Neny Date: Mon, 28 Sep 2026 16:48:24 +0200 Subject: [PATCH] =?UTF-8?q?fix(bifrost):=20tiers=20sur=20gpt-oss=20(Groq?= =?UTF-8?q?=20a=20retir=C3=A9=20les=20llama)=20+=20marge=20de=20raisonneme?= =?UTF-8?q?nt?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RAPIDE groq/llama-3.1-8b-instant → groq/openai/gpt-oss-20b ; APPROFONDI groq/llama-3.3-70b-versatile → groq/openai/gpt-oss-120b ; replis inchangés. Les chatbots vivaient sur le repli Gemini depuis le retrait. gpt-oss raisonne : ses tokens de raisonnement sont décomptés de max_tokens. REASONING_MARGIN_TOKENS (800) ajouté au budget des trois routes, et bifrostContent() signale au journal une réponse tronquée (finish_reason length). Le champ message.reasoning est ignoré. Défaut WORKER_MODEL aligné sur la prod. En-têtes « Mistral Small » corrigés, PIPE-IA-DOC §11.3. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01AzGo8bxhHp6M9RxuAFbyLU --- PIPE-IA-DOC.md | 39 ++++++++++++++++++++++++++- server/api/chatbot-reseaux.post.ts | 8 +++--- server/api/chatbot-taff.post.ts | 8 +++--- server/api/chatbot.post.ts | 10 +++---- server/utils/bifrost.ts | 43 +++++++++++++++++++++++++++--- worker/enrich.js | 2 +- 6 files changed, 91 insertions(+), 19 deletions(-) diff --git a/PIPE-IA-DOC.md b/PIPE-IA-DOC.md index 761ce33..7ed966b 100644 --- a/PIPE-IA-DOC.md +++ b/PIPE-IA-DOC.md @@ -365,7 +365,7 @@ fait ici pour ne pas tester un ALTER contre la prod sans l'avoir vérifié. | Sujet | NAV V2 (§1-10 ci-dessus) | AEP B5 | |---|---|---| -| LLM | Mistral Nemo, appel direct | **Bifrost** (`${BIFROST_URL}/v1/chat/completions`, header `x-bf-vk`), modèle `WORKER_MODEL` (défaut `groq/llama-3.1-8b-instant`) | +| LLM | Mistral Nemo, appel direct | **Bifrost** (`${BIFROST_URL}/v1/chat/completions`, header `x-bf-vk`), modèle `WORKER_MODEL` (défaut `groq/openai/gpt-oss-20b` depuis le 28/09, voir §11.3) | | Scrape | crawl4ai (Python, `AsyncHTTPCrawlerStrategy`) | **fetch natif Node 22** — timeout 8 s, corps plafonné 500 Ko, extraction titre + meta description + `og:*` + texte visible tronqué à 4000 caractères, `User-Agent: AEP/2.0 contact@trans-former.fr` | | Notification | Resend (email Jules) | **ntfy** (`POST https://ntfy.sh/$NTFY_TOPIC`) — le message ne contient JAMAIS l'email ni le texte libre du contributeur, seulement id NocoDB / nom suggéré / type / confiance | | Seuil « 5 fiches pending » | Email si ≥ 5 en attente | **Retiré** (décision MOE : volume faible, une notif par fiche traitée suffit — à réactiver si le volume monte) | @@ -381,3 +381,40 @@ Le mode `--dry-run` (`node enrich.js --dry-run`) lit `worker/fixtures/dry-run-rows.json` au lieu de NocoDB, n'écrit rien, et par défaut mock aussi le scrape et l'appel Bifrost (`DRY_RUN_LIVE=1` pour forcer de vrais appels réseau sans jamais toucher NocoDB). + +### 11.3 Modèles Groq retirés — bascule sur gpt-oss (2026-09-28) + +Groq ne sert plus `llama-3.1-8b-instant` ni `llama-3.3-70b-versatile` +(404 `model_not_found` via Bifrost, constaté au test M3 du 28/09). + +| Consommateur | Avant | Après | Où | +|---|---|---|---| +| Worker (`worker/enrich.js`) | `groq/llama-3.1-8b-instant` | `groq/openai/gpt-oss-20b` | prod : `WORKER_MODEL` dans `/opt/aep/.env` depuis le 28/09 ; défaut du code aligné | +| Chatbots, tier RAPIDE | `groq/llama-3.1-8b-instant` | `groq/openai/gpt-oss-20b` | `server/utils/bifrost.ts`, replis inchangés | +| Chatbots, tier APPROFONDI | `groq/llama-3.3-70b-versatile` | `groq/openai/gpt-oss-120b` | `server/utils/bifrost.ts`, replis inchangés | +| RAG Pensées (LightRAG) | `gemini-oai/models/gemini-2.5-flash-lite` | inchangé | `/opt/lightrag/.env`, hors de ce routage | + +Entre le retrait et ce correctif, chaque appel chatbot partait sur le +premier repli Gemini (`stats_usage` depuis le 23/09 environ) : service +rendu, modèle primaire jamais servi. + +**gpt-oss est un modèle à raisonnement.** Deux conséquences : +- la réponse porte `choices[0].message.reasoning` à côté de `content`. Les + trois routes (`chatbot`, `chatbot-reseaux`, `chatbot-taff`) et le worker ne + lisent que `content` (JSON), `usage` et `extra_fields` : le champ est + ignoré, rien n'en dépend. `chatbot-pensees` lit la réponse de LightRAG + (`response`, `references`), pas celle de Bifrost ; +- les tokens de raisonnement sont décomptés de `max_tokens`. Si le + fournisseur rend un JSON tronqué en HTTP 200, Bifrost ne bascule pas et + l'usager lit « Je n'ai pas pu analyser ta demande » (s'il le rejette en + 400, le repli Gemini prend le relais : dégradé, pas cassé). Les routes chatbot ajoutent donc + `REASONING_MARGIN_TOKENS` (800) à leur budget de contenu (600 ou 700), et + `bifrostContent()` écrit un avertissement au journal quand + `finish_reason` vaut `length`. Le worker garde `max_tokens: 800` (sortie + courte, M3 passé en 1,6 s) ; à relever si `tokens_out` approche 800 dans + `stats_usage`. + +**Non testé depuis Windows** (Bifrost écoute sur le loopback du VPS). Test +réel au checkpoint : un appel par route, relever `extra_fields` (modèle +réellement servi), `usage.completion_tokens`, `finish_reason`, et le +modèle inscrit dans `stats_usage` pour `/api/chatbot`. diff --git a/server/api/chatbot-reseaux.post.ts b/server/api/chatbot-reseaux.post.ts index c905657..8268d80 100644 --- a/server/api/chatbot-reseaux.post.ts +++ b/server/api/chatbot-reseaux.post.ts @@ -1,12 +1,12 @@ /** * POST /api/chatbot-reseaux * Chatbot Réseaux AEP — Carte 2 "Réseaux de bifurcation" - * Keyword search sur reseaux-bifurcation.json + Mistral Small. + * Keyword search sur reseaux-bifurcation.json + LLM via Bifrost (server/utils/bifrost.ts). */ // @ts-ignore — JSON import résolu par Rollup import reseauxData from '../../public/data/reseaux-bifurcation.json' import { checkRateLimitJson } from '~/server/utils/rateLimitJson' -import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost' +import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost' interface Structure { id: string @@ -99,7 +99,7 @@ export default defineEventHandler(async (event) => { model: tier.model, fallbacks: tier.fallbacks, temperature: 0.3, - max_tokens: 700, + max_tokens: 700 + REASONING_MARGIN_TOKENS, response_format: { type: 'json_object' }, messages: [ { role: 'system', content: systemPrompt }, @@ -108,7 +108,7 @@ export default defineEventHandler(async (event) => { }), } ) - mistralRaw = res.choices?.[0]?.message?.content ?? '{}' + mistralRaw = bifrostContent(res, 'chatbot-reseaux') } catch { throw createError({ statusCode: 502, message: 'Erreur IA — réessaie dans quelques instants.' }) } diff --git a/server/api/chatbot-taff.post.ts b/server/api/chatbot-taff.post.ts index 3a76585..7c7a3a8 100644 --- a/server/api/chatbot-taff.post.ts +++ b/server/api/chatbot-taff.post.ts @@ -1,13 +1,13 @@ /** * POST /api/chatbot-taff * Chatbot d'aiguillage — Carte 3 "Trouver du taf" - * Lit plateformes-taff.json, appelle Mistral Small, retourne recommandations. + * Lit plateformes-taff.json, appelle un LLM via Bifrost (server/utils/bifrost.ts), retourne recommandations. */ // @ts-ignore — JSON import résolu par Vite/Rollup import taffData from '../../public/data/plateformes-taff.json' import { checkRateLimitJson } from '~/server/utils/rateLimitJson' -import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost' +import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost' interface PlateformeMinimal { id: string @@ -110,7 +110,7 @@ export default defineEventHandler(async (event) => { model: tier.model, fallbacks: tier.fallbacks, temperature: 0.3, - max_tokens: 700, + max_tokens: 700 + REASONING_MARGIN_TOKENS, response_format: { type: 'json_object' }, messages: [ { role: 'system', content: systemPrompt }, @@ -119,7 +119,7 @@ export default defineEventHandler(async (event) => { }), } ) - mistralRaw = res.choices?.[0]?.message?.content ?? '{}' + mistralRaw = bifrostContent(res, 'chatbot-taff') } catch { throw createError({ statusCode: 502, statusMessage: 'Erreur IA — réessaie dans quelques instants.' }) } diff --git a/server/api/chatbot.post.ts b/server/api/chatbot.post.ts index f5ca706..7d8057b 100644 --- a/server/api/chatbot.post.ts +++ b/server/api/chatbot.post.ts @@ -1,14 +1,14 @@ /** * POST /api/chatbot * - * Chatbot recherche sémantique — Mistral Small + * Chatbot recherche sémantique — LLM via Bifrost (tiers et replis : server/utils/bifrost.ts) * Spec : F §7 (endpoint), F §8 (rate limit), E-spec §6 (détails chatbot) * * Flow : * 1. Rate limit : 10 req/IP/jour (JSON fichier, SHA-256) * 2. Circuit breaker : budget 20€/mois * 3. Fetch top-N fiches (keyword match sur nom+description+fonctions) - * 4. Appel Mistral Small avec contexte JSON compact + * 4. Appel Bifrost (tier RAPIDE par défaut) avec contexte JSON compact * 5. Parse JSON → { reponse_texte, fiches_recommandees } * 6. Log stats_usage * @@ -19,7 +19,7 @@ import { checkRateLimitJson } from '~/server/utils/rateLimitJson' import { checkBudget, calcCoutMistralSmall } from '~/server/utils/circuitBreaker' -import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost' +import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost' // ── Types ────────────────────────────────────────────────────────────────────── @@ -277,7 +277,7 @@ export default defineEventHandler(async (event) => { model: tier.model, fallbacks: tier.fallbacks, temperature: 0.3, - max_tokens: 600, + max_tokens: 600 + REASONING_MARGIN_TOKENS, response_format: { type: 'json_object' }, messages: [ { role: 'system', content: systemPrompt }, @@ -286,7 +286,7 @@ export default defineEventHandler(async (event) => { }), }) - mistralRaw = bifrostRes.choices?.[0]?.message?.content ?? '{}' + mistralRaw = bifrostContent(bifrostRes, 'chatbot') tokensIn = bifrostRes.usage?.prompt_tokens ?? 0 tokensOut = bifrostRes.usage?.completion_tokens ?? 0 realModel = diff --git a/server/utils/bifrost.ts b/server/utils/bifrost.ts index f6de8c0..1075299 100644 --- a/server/utils/bifrost.ts +++ b/server/utils/bifrost.ts @@ -3,10 +3,24 @@ * Endpoint OpenAI-compatible : POST {bifrostUrl}/v1/chat/completions * Auth : header x-bf-vk * - * 2 tiers validés (Mission M3, build Bifrost) : + * 2 tiers (Mission M3 du build Bifrost ; modèles primaires changés le 28/09, session AEP front 4) : * RAPIDE — défaut, pas de toggle UI mode rapide/approfondi sur le site actuellement * APPROFONDI — activable via body.mode === 'approfondi' (prêt pour un futur toggle front) * + * ⚠ 28/09 : Groq a retiré groq/llama-3.1-8b-instant (RAPIDE) et groq/llama-3.3-70b-versatile + * (APPROFONDI) — 404 model_not_found via Bifrost. Depuis le 23/09 environ, chaque appel + * chatbot tombait sur le premier repli Gemini (stats_usage). Primaires passés sur + * groq/openai/gpt-oss-20b et groq/openai/gpt-oss-120b ; replis inchangés. Le worker + * /opt/aep-worker tourne sur gpt-oss-20b en prod depuis le 28/09 (WORKER_MODEL, json_object) : + * Bifrost accepte ce nom de modèle à slash côté Groq. + * ⚠ gpt-oss est un modèle à raisonnement : + * - la réponse porte un champ message.reasoning à côté de message.content. Les routes + * chatbot ne lisent que content (JSON), usage et extra_fields : le champ est ignoré ; + * - les tokens de raisonnement sont décomptés de max_tokens. Sans marge, le JSON de la + * réponse peut être tronqué ; rendu en HTTP 200, Bifrost ne bascule pas, JSON.parse + * échoue → « Je n'ai pas pu analyser ta demande ». Déjà vu le 15/07 : gpt-oss-120b (via Cerebras), 3 appels + * corrects sur 4, « tronque parfois à max_tokens 700 ». D'où REASONING_MARGIN_TOKENS, ajouté par + * chaque route à son budget de contenu. * ⚠ openrouter-oai exclu (bug Bifrost confirmé — 404 HTML sur modèles avec slash) * ⚠ gemini-oai exige le préfixe "models/" (sinon 403 silencieux) * ⚠ cerebras/gemma-4-31b RETIRÉ du tier RAPIDE (M4, 15/07) : en JSON mode avec un contexte @@ -19,7 +33,7 @@ */ export const BIFROST_TIER_RAPIDE = { - model: 'groq/llama-3.1-8b-instant', + model: 'groq/openai/gpt-oss-20b', fallbacks: [ 'gemini-oai/models/gemini-2.5-flash-lite', 'cohere/command-r-08-2024', @@ -27,7 +41,7 @@ export const BIFROST_TIER_RAPIDE = { } export const BIFROST_TIER_APPROFONDI = { - model: 'groq/llama-3.3-70b-versatile', + model: 'groq/openai/gpt-oss-120b', fallbacks: [ 'gemini-oai/models/gemini-2.5-flash', 'mistral/mistral-large-latest', @@ -35,13 +49,34 @@ export const BIFROST_TIER_APPROFONDI = { ], } +/** + * Tokens ajoutés au budget de contenu de chaque route (max_tokens = contenu + marge) pour + * absorber le raisonnement de gpt-oss. Plafond seulement : les replis sans raisonnement + * n'écrivent pas plus long, les prompts bornent la réponse (200-250 mots). Rester modeste : + * la limite de tokens par minute d'une clé Groq peut compter max_tokens avec le prompt. + */ +export const REASONING_MARGIN_TOKENS = 800 + /** Sélectionne le tier selon le param optionnel body.mode. */ export function pickBifrostTier(mode?: string) { return mode === 'approfondi' ? BIFROST_TIER_APPROFONDI : BIFROST_TIER_RAPIDE } export interface BifrostChatResponse { - choices: { message: { content: string } }[] + /** reasoning : présent avec gpt-oss, jamais lu (la réponse utile est dans content). */ + choices: { message: { content: string; reasoning?: string }; finish_reason?: string }[] usage?: { prompt_tokens: number; completion_tokens: number } extra_fields?: { provider?: string; resolved_model_used?: string } } + +/** + * Contenu JSON de la réponse. Une troncature (finish_reason 'length') ne lève aucune erreur + * et finit en message générique côté usager : on la signale au journal pour la voir. + */ +export function bifrostContent(res: BifrostChatResponse, route: string): string { + const choice = res.choices?.[0] + if (choice?.finish_reason === 'length') { + console.warn(`[${route}] réponse tronquée (finish_reason=length) : relever REASONING_MARGIN_TOKENS`) + } + return choice?.message?.content ?? '{}' +} diff --git a/worker/enrich.js b/worker/enrich.js index 050d543..729959e 100644 --- a/worker/enrich.js +++ b/worker/enrich.js @@ -33,7 +33,7 @@ const NOCODB_TABLE_STATS = process.env.NOCODB_TABLE_STATS; const BIFROST_URL = process.env.BIFROST_URL || 'http://127.0.0.1:8080'; const BIFROST_VK = process.env.BIFROST_VK; -const WORKER_MODEL = process.env.WORKER_MODEL || 'groq/llama-3.1-8b-instant'; +const WORKER_MODEL = process.env.WORKER_MODEL || 'groq/openai/gpt-oss-20b'; // llama-3.1-8b-instant retiré par Groq (28/09) const NTFY_TOPIC = process.env.NTFY_TOPIC;