fix(bifrost): tiers sur gpt-oss (Groq a retiré les llama) + marge de raisonnement
RAPIDE groq/llama-3.1-8b-instant → groq/openai/gpt-oss-20b ; APPROFONDI groq/llama-3.3-70b-versatile → groq/openai/gpt-oss-120b ; replis inchangés. Les chatbots vivaient sur le repli Gemini depuis le retrait. gpt-oss raisonne : ses tokens de raisonnement sont décomptés de max_tokens. REASONING_MARGIN_TOKENS (800) ajouté au budget des trois routes, et bifrostContent() signale au journal une réponse tronquée (finish_reason length). Le champ message.reasoning est ignoré. Défaut WORKER_MODEL aligné sur la prod. En-têtes « Mistral Small » corrigés, PIPE-IA-DOC §11.3. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01AzGo8bxhHp6M9RxuAFbyLU
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
288a9c9126
commit
8f94a1acda
@@ -1,12 +1,12 @@
|
||||
/**
|
||||
* POST /api/chatbot-reseaux
|
||||
* Chatbot Réseaux AEP — Carte 2 "Réseaux de bifurcation"
|
||||
* Keyword search sur reseaux-bifurcation.json + Mistral Small.
|
||||
* Keyword search sur reseaux-bifurcation.json + LLM via Bifrost (server/utils/bifrost.ts).
|
||||
*/
|
||||
// @ts-ignore — JSON import résolu par Rollup
|
||||
import reseauxData from '../../public/data/reseaux-bifurcation.json'
|
||||
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
|
||||
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
|
||||
interface Structure {
|
||||
id: string
|
||||
@@ -99,7 +99,7 @@ export default defineEventHandler(async (event) => {
|
||||
model: tier.model,
|
||||
fallbacks: tier.fallbacks,
|
||||
temperature: 0.3,
|
||||
max_tokens: 700,
|
||||
max_tokens: 700 + REASONING_MARGIN_TOKENS,
|
||||
response_format: { type: 'json_object' },
|
||||
messages: [
|
||||
{ role: 'system', content: systemPrompt },
|
||||
@@ -108,7 +108,7 @@ export default defineEventHandler(async (event) => {
|
||||
}),
|
||||
}
|
||||
)
|
||||
mistralRaw = res.choices?.[0]?.message?.content ?? '{}'
|
||||
mistralRaw = bifrostContent(res, 'chatbot-reseaux')
|
||||
} catch {
|
||||
throw createError({ statusCode: 502, message: 'Erreur IA — réessaie dans quelques instants.' })
|
||||
}
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
/**
|
||||
* POST /api/chatbot-taff
|
||||
* Chatbot d'aiguillage — Carte 3 "Trouver du taf"
|
||||
* Lit plateformes-taff.json, appelle Mistral Small, retourne recommandations.
|
||||
* Lit plateformes-taff.json, appelle un LLM via Bifrost (server/utils/bifrost.ts), retourne recommandations.
|
||||
*/
|
||||
|
||||
// @ts-ignore — JSON import résolu par Vite/Rollup
|
||||
import taffData from '../../public/data/plateformes-taff.json'
|
||||
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
|
||||
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
|
||||
interface PlateformeMinimal {
|
||||
id: string
|
||||
@@ -110,7 +110,7 @@ export default defineEventHandler(async (event) => {
|
||||
model: tier.model,
|
||||
fallbacks: tier.fallbacks,
|
||||
temperature: 0.3,
|
||||
max_tokens: 700,
|
||||
max_tokens: 700 + REASONING_MARGIN_TOKENS,
|
||||
response_format: { type: 'json_object' },
|
||||
messages: [
|
||||
{ role: 'system', content: systemPrompt },
|
||||
@@ -119,7 +119,7 @@ export default defineEventHandler(async (event) => {
|
||||
}),
|
||||
}
|
||||
)
|
||||
mistralRaw = res.choices?.[0]?.message?.content ?? '{}'
|
||||
mistralRaw = bifrostContent(res, 'chatbot-taff')
|
||||
} catch {
|
||||
throw createError({ statusCode: 502, statusMessage: 'Erreur IA — réessaie dans quelques instants.' })
|
||||
}
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
/**
|
||||
* POST /api/chatbot
|
||||
*
|
||||
* Chatbot recherche sémantique — Mistral Small
|
||||
* Chatbot recherche sémantique — LLM via Bifrost (tiers et replis : server/utils/bifrost.ts)
|
||||
* Spec : F §7 (endpoint), F §8 (rate limit), E-spec §6 (détails chatbot)
|
||||
*
|
||||
* Flow :
|
||||
* 1. Rate limit : 10 req/IP/jour (JSON fichier, SHA-256)
|
||||
* 2. Circuit breaker : budget 20€/mois
|
||||
* 3. Fetch top-N fiches (keyword match sur nom+description+fonctions)
|
||||
* 4. Appel Mistral Small avec contexte JSON compact
|
||||
* 4. Appel Bifrost (tier RAPIDE par défaut) avec contexte JSON compact
|
||||
* 5. Parse JSON → { reponse_texte, fiches_recommandees }
|
||||
* 6. Log stats_usage
|
||||
*
|
||||
@@ -19,7 +19,7 @@
|
||||
|
||||
import { checkRateLimitJson } from '~/server/utils/rateLimitJson'
|
||||
import { checkBudget, calcCoutMistralSmall } from '~/server/utils/circuitBreaker'
|
||||
import { pickBifrostTier, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
import { pickBifrostTier, bifrostContent, REASONING_MARGIN_TOKENS, type BifrostChatResponse } from '~/server/utils/bifrost'
|
||||
|
||||
// ── Types ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -277,7 +277,7 @@ export default defineEventHandler(async (event) => {
|
||||
model: tier.model,
|
||||
fallbacks: tier.fallbacks,
|
||||
temperature: 0.3,
|
||||
max_tokens: 600,
|
||||
max_tokens: 600 + REASONING_MARGIN_TOKENS,
|
||||
response_format: { type: 'json_object' },
|
||||
messages: [
|
||||
{ role: 'system', content: systemPrompt },
|
||||
@@ -286,7 +286,7 @@ export default defineEventHandler(async (event) => {
|
||||
}),
|
||||
})
|
||||
|
||||
mistralRaw = bifrostRes.choices?.[0]?.message?.content ?? '{}'
|
||||
mistralRaw = bifrostContent(bifrostRes, 'chatbot')
|
||||
tokensIn = bifrostRes.usage?.prompt_tokens ?? 0
|
||||
tokensOut = bifrostRes.usage?.completion_tokens ?? 0
|
||||
realModel =
|
||||
|
||||
Reference in New Issue
Block a user