fix(bifrost): tiers sur gpt-oss (Groq a retiré les llama) + marge de raisonnement

RAPIDE groq/llama-3.1-8b-instant → groq/openai/gpt-oss-20b ; APPROFONDI
groq/llama-3.3-70b-versatile → groq/openai/gpt-oss-120b ; replis inchangés.
Les chatbots vivaient sur le repli Gemini depuis le retrait.

gpt-oss raisonne : ses tokens de raisonnement sont décomptés de max_tokens.
REASONING_MARGIN_TOKENS (800) ajouté au budget des trois routes, et
bifrostContent() signale au journal une réponse tronquée (finish_reason
length). Le champ message.reasoning est ignoré. Défaut WORKER_MODEL aligné
sur la prod. En-têtes « Mistral Small » corrigés, PIPE-IA-DOC §11.3.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01AzGo8bxhHp6M9RxuAFbyLU
This commit is contained in:
Jules Neny
2026-09-28 16:48:24 +02:00
co-authored by Claude Opus 5.5
parent 288a9c9126
commit 8f94a1acda
6 changed files with 91 additions and 19 deletions
+4 -4
View File
@@ -1,12 +1,12 @@
/**
* POST /api/chatbot-reseaux
* Chatbot Réseaux AEP — Carte 2 "Réseaux de bifurcation"
* Keyword search sur reseaux-bifurcation.json + Mistral Small.
* Keyword search sur reseaux-bifurcation.json + LLM via Bifrost (server/utils/bifrost.ts).
*/
// @ts-ignore — JSON import résolu par Rollup
import reseauxData from '../../public/data/reseaux-bifurcation.json'
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
interface Structure {
id: string
@@ -99,7 +99,7 @@ export default defineEventHandler(async (event) => {
model: tier.model,
fallbacks: tier.fallbacks,
temperature: 0.3,
max_tokens: 700,
max_tokens: 700 + REASONING_MARGIN_TOKENS,
response_format: { type: 'json_object' },
messages: [
{ role: 'system', content: systemPrompt },
@@ -108,7 +108,7 @@ export default defineEventHandler(async (event) => {
}),
}
)
mistralRaw = res.choices?.[0]?.message?.content ?? '{}'
mistralRaw = bifrostContent(res, 'chatbot-reseaux')
} catch {
throw createError({ statusCode: 502, message: 'Erreur IA — réessaie dans quelques instants.' })
}
+4 -4
View File
@@ -1,13 +1,13 @@
/**
* POST /api/chatbot-taff
* Chatbot d'aiguillage — Carte 3 "Trouver du taf"
* Lit plateformes-taff.json, appelle Mistral Small, retourne recommandations.
* Lit plateformes-taff.json, appelle un LLM via Bifrost (server/utils/bifrost.ts), retourne recommandations.
*/
// @ts-ignore — JSON import résolu par Vite/Rollup
import taffData from '../../public/data/plateformes-taff.json'
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
interface PlateformeMinimal {
id: string
@@ -110,7 +110,7 @@ export default defineEventHandler(async (event) => {
model: tier.model,
fallbacks: tier.fallbacks,
temperature: 0.3,
max_tokens: 700,
max_tokens: 700 + REASONING_MARGIN_TOKENS,
response_format: { type: 'json_object' },
messages: [
{ role: 'system', content: systemPrompt },
@@ -119,7 +119,7 @@ export default defineEventHandler(async (event) => {
}),
}
)
mistralRaw = res.choices?.[0]?.message?.content ?? '{}'
mistralRaw = bifrostContent(res, 'chatbot-taff')
} catch {
throw createError({ statusCode: 502, statusMessage: 'Erreur IA — réessaie dans quelques instants.' })
}
+5 -5
View File
@@ -1,14 +1,14 @@
/**
* POST /api/chatbot
*
* Chatbot recherche sémantique — Mistral Small
* Chatbot recherche sémantique — LLM via Bifrost (tiers et replis : server/utils/bifrost.ts)
* Spec : F §7 (endpoint), F §8 (rate limit), E-spec §6 (détails chatbot)
*
* Flow :
* 1. Rate limit : 10 req/IP/jour (JSON fichier, SHA-256)
* 2. Circuit breaker : budget 20€/mois
* 3. Fetch top-N fiches (keyword match sur nom+description+fonctions)
* 4. Appel Mistral Small avec contexte JSON compact
* 4. Appel Bifrost (tier RAPIDE par défaut) avec contexte JSON compact
* 5. Parse JSON → { reponse_texte, fiches_recommandees }
* 6. Log stats_usage
*
@@ -19,7 +19,7 @@
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
import { checkBudget, calcCoutMistralSmall } from '~/server/utils/circuitBreaker'
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
// ── Types ──────────────────────────────────────────────────────────────────────
@@ -277,7 +277,7 @@ export default defineEventHandler(async (event) => {
model: tier.model,
fallbacks: tier.fallbacks,
temperature: 0.3,
max_tokens: 600,
max_tokens: 600 + REASONING_MARGIN_TOKENS,
response_format: { type: 'json_object' },
messages: [
{ role: 'system', content: systemPrompt },
@@ -286,7 +286,7 @@ export default defineEventHandler(async (event) => {
}),
})
mistralRaw = bifrostRes.choices?.[0]?.message?.content ?? '{}'
mistralRaw = bifrostContent(bifrostRes, 'chatbot')
tokensIn = bifrostRes.usage?.prompt_tokens ?? 0
tokensOut = bifrostRes.usage?.completion_tokens ?? 0
realModel =
+39 -4
View File
@@ -3,10 +3,24 @@
* Endpoint OpenAI-compatible : POST {bifrostUrl}/v1/chat/completions
* Auth : header x-bf-vk
*
* 2 tiers validés (Mission M3, build Bifrost) :
* 2 tiers (Mission M3 du build Bifrost ; modèles primaires changés le 28/09, session AEP front 4) :
* RAPIDE — défaut, pas de toggle UI mode rapide/approfondi sur le site actuellement
* APPROFONDI — activable via body.mode === 'approfondi' (prêt pour un futur toggle front)
*
* ⚠ 28/09 : Groq a retiré groq/llama-3.1-8b-instant (RAPIDE) et groq/llama-3.3-70b-versatile
* (APPROFONDI) — 404 model_not_found via Bifrost. Depuis le 23/09 environ, chaque appel
* chatbot tombait sur le premier repli Gemini (stats_usage). Primaires passés sur
* groq/openai/gpt-oss-20b et groq/openai/gpt-oss-120b ; replis inchangés. Le worker
* /opt/aep-worker tourne sur gpt-oss-20b en prod depuis le 28/09 (WORKER_MODEL, json_object) :
* Bifrost accepte ce nom de modèle à slash côté Groq.
* ⚠ gpt-oss est un modèle à raisonnement :
* - la réponse porte un champ message.reasoning à côté de message.content. Les routes
* chatbot ne lisent que content (JSON), usage et extra_fields : le champ est ignoré ;
* - les tokens de raisonnement sont décomptés de max_tokens. Sans marge, le JSON de la
* réponse peut être tronqué ; rendu en HTTP 200, Bifrost ne bascule pas, JSON.parse
* échoue → « Je n'ai pas pu analyser ta demande ». Déjà vu le 15/07 : gpt-oss-120b (via Cerebras), 3 appels
* corrects sur 4, « tronque parfois à max_tokens 700 ». D'où REASONING_MARGIN_TOKENS, ajouté par
* chaque route à son budget de contenu.
* ⚠ openrouter-oai exclu (bug Bifrost confirmé — 404 HTML sur modèles avec slash)
* ⚠ gemini-oai exige le préfixe "models/" (sinon 403 silencieux)
* ⚠ cerebras/gemma-4-31b RETIRÉ du tier RAPIDE (M4, 15/07) : en JSON mode avec un contexte
@@ -19,7 +33,7 @@
*/
export const BIFROST_TIER_RAPIDE = {
model: 'groq/llama-3.1-8b-instant',
model: 'groq/openai/gpt-oss-20b',
fallbacks: [
'gemini-oai/models/gemini-2.5-flash-lite',
'cohere/command-r-08-2024',
@@ -27,7 +41,7 @@ export const BIFROST_TIER_RAPIDE = {
}
export const BIFROST_TIER_APPROFONDI = {
model: 'groq/llama-3.3-70b-versatile',
model: 'groq/openai/gpt-oss-120b',
fallbacks: [
'gemini-oai/models/gemini-2.5-flash',
'mistral/mistral-large-latest',
@@ -35,13 +49,34 @@ export const BIFROST_TIER_APPROFONDI = {
],
}
/**
* Tokens ajoutés au budget de contenu de chaque route (max_tokens = contenu + marge) pour
* absorber le raisonnement de gpt-oss. Plafond seulement : les replis sans raisonnement
* n'écrivent pas plus long, les prompts bornent la réponse (200-250 mots). Rester modeste :
* la limite de tokens par minute d'une clé Groq peut compter max_tokens avec le prompt.
*/
export const REASONING_MARGIN_TOKENS = 800
/** Sélectionne le tier selon le param optionnel body.mode. */
export function pickBifrostTier(mode?: string) {
return mode === 'approfondi' ? BIFROST_TIER_APPROFONDI : BIFROST_TIER_RAPIDE
}
export interface BifrostChatResponse {
choices: { message: { content: string } }[]
/** reasoning : présent avec gpt-oss, jamais lu (la réponse utile est dans content). */
choices: { message: { content: string; reasoning?: string }; finish_reason?: string }[]
usage?: { prompt_tokens: number; completion_tokens: number }
extra_fields?: { provider?: string; resolved_model_used?: string }
}
/**
* Contenu JSON de la réponse. Une troncature (finish_reason 'length') ne lève aucune erreur
* et finit en message générique côté usager : on la signale au journal pour la voir.
*/
export function bifrostContent(res: BifrostChatResponse, route: string): string {
const choice = res.choices?.[0]
if (choice?.finish_reason === 'length') {
console.warn(`[${route}] réponse tronquée (finish_reason=length) : relever REASONING_MARGIN_TOKENS`)
}
return choice?.message?.content ?? '{}'
}