#!/usr/bin/env node /** * Contrôle des citations d'un chatbot LightRAG : chaque segment entre guillemets * de la réponse doit figurer mot pour mot dans le contexte réellement récupéré. * * Usage : * node scripts/verif-citations.mjs --rag http://HOTE:9621 --query "Cite ..." \ * [--route http://localhost:3334/api/chatbot-pensees] [--mode hybrid] [--auteur slug] \ * [--response-file reponse.txt] [--ctx-query-file autre-query.txt] [--min-mots 6] [--json] * * - Sans --response-file : la réponse est demandée à --route (POST {query, mode, auteur_slug}). * - Contexte : POST {rag}/query {query, mode, only_need_context:true}. --ctx-query-file ajoute * un second contexte (ex. la requête telle qu'envoyée avant amendement, préambule inclus) ; * un segment est « trouvé » s'il figure dans l'un ou l'autre. * - Segment >= min-mots (6) absent du contexte = inventé. Plus court : signalé, non compté. * Lecture seule côté LightRAG. Réutilisable pour l'instance 9623 (--rag http://HOTE:9623). */ import { readFileSync } from 'node:fs' const args = Object.fromEntries(process.argv.slice(2).reduce((a, x, i, arr) => { if (x.startsWith('--')) a.push([x.slice(2), arr[i + 1] && !arr[i + 1].startsWith('--') ? arr[i + 1] : true]) return a }, [])) if (!args.rag || !args.query) { console.error('--rag et --query requis'); process.exit(2) } const mode = args.mode || 'hybrid' const minMots = Number(args['min-mots'] || 6) // Normalisation : sans accents ni ponctuation/apostrophes/antislashs (le contexte LightRAG est du JSON échappé). const norm = s => s.normalize('NFD').replace(/[\u0300-\u036f]/g, '').toLowerCase() .replace(/[^\p{L}\p{N}]+/gu, ' ') .replace(/\s+/g, ' ').trim() async function post(url, body, timeout = 120000) { const r = await fetch(url, { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body), signal: AbortSignal.timeout(timeout) }) if (!r.ok) throw new Error(`${url} -> HTTP ${r.status}`) return r.json() } async function contexte(q) { const j = await post(`${args.rag}/query`, { query: q, mode, only_need_context: true }) return typeof j.response === 'string' ? j.response : JSON.stringify(j) } let reponse if (args['response-file']) reponse = readFileSync(args['response-file'], 'utf8') else { if (!args.route) { console.error('--route ou --response-file requis'); process.exit(2) } const b = { query: args.query, mode } if (args.auteur) b.auteur_slug = args.auteur reponse = (await post(args.route, b)).response } const ctxs = [await contexte(args.query)] if (args['ctx-query-file']) ctxs.push(await contexte(readFileSync(args['ctx-query-file'], 'utf8'))) const ctxNorm = norm(ctxs.join('\n')) // « » d'abord (les "..." imbriqués ne doivent pas casser l'appariement), puis "..." et “...” sur le reste. const segs = [] let reste = reponse.replace(/«s*([^»]+?)s*»/g, (_, s) => { segs.push(s); return ' ' }) for (const re of [/"([^"]+?)"/g, /“([^”]+?)”/g]) { for (const m of reste.matchAll(re)) segs.push(m[1]) } const lignes = segs.map(s => { const n = norm(s); const mots = n ? n.split(' ').length : 0 return { segment: s, mots, trouve: n.length > 0 && ctxNorm.includes(n) } }) const longs = lignes.filter(l => l.mots >= minMots) const res = { query: args.query, mode, segments: lignes.length, longs: longs.length, longs_trouves: longs.filter(l => l.trouve).length, inventes: longs.filter(l => !l.trouve).length, courts_non_comptes: lignes.length - longs.length, contexte_chars: ctxs.map(c => c.length), detail: lignes, reponse, } if (args.json) console.log(JSON.stringify(res, null, 2)) else { console.log(`Q: ${res.query}\nsegments=${res.segments} longs(>=${minMots} mots)=${res.longs} trouvés=${res.longs_trouves} INVENTÉS=${res.inventes} courts=${res.courts_non_comptes}`) for (const l of lignes) console.log(` [${l.mots >= minMots ? (l.trouve ? 'OK ' : 'INV') : 'crt'}] (${l.mots}) ${l.segment.slice(0, 160)}`) }