RAPIDE groq/llama-3.1-8b-instant → groq/openai/gpt-oss-20b ; APPROFONDI groq/llama-3.3-70b-versatile → groq/openai/gpt-oss-120b ; replis inchangés. Les chatbots vivaient sur le repli Gemini depuis le retrait. gpt-oss raisonne : ses tokens de raisonnement sont décomptés de max_tokens. REASONING_MARGIN_TOKENS (800) ajouté au budget des trois routes, et bifrostContent() signale au journal une réponse tronquée (finish_reason length). Le champ message.reasoning est ignoré. Défaut WORKER_MODEL aligné sur la prod. En-têtes « Mistral Small » corrigés, PIPE-IA-DOC §11.3. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01AzGo8bxhHp6M9RxuAFbyLU
598 lines
25 KiB
JavaScript
598 lines
25 KiB
JavaScript
#!/usr/bin/env node
|
||
/**
|
||
* AEP — Worker enrichissement IA (B5-M2)
|
||
* Lancé via systemd timer toutes les 15 minutes (deploy/aep-worker/aep-worker.timer)
|
||
* Pipeline : fetch pending → scrape léger (fetch natif) → Bifrost (LLM) → update NocoDB → ntfy
|
||
*
|
||
* Écarts à la version NAV V2 d'origine (PIPE-IA-DOC.md, fusionné pas réécrit) :
|
||
* - LLM : Bifrost (passerelle OpenAI-compatible en prod) au lieu de Mistral direct.
|
||
* - Scrape : fetch natif Node 22 (timeout 8s, 500 Ko max, texte tronqué 4000 car.)
|
||
* au lieu de crawl4ai/Python (disque VPS tendu, décision MOE 27/09).
|
||
* - Notification : ntfy (https://ntfy.sh/$NTFY_TOPIC) au lieu de Resend (abandonné 15/07).
|
||
* Le message ne contient JAMAIS l'email ni le texte libre du contributeur.
|
||
* - Le formulaire assoupli (B5-M1) écrit un `nom` placeholder ("[à qualifier] host") et
|
||
* éventuellement "Type : non précisé" en tête de description_user. Le worker propose un
|
||
* vrai nom et un type — voir §Sortie JSON ci-dessous et PIPE-IA-DOC.md.
|
||
* - Seuil « email à 5 fiches pending » PAS réactivé (décision MOE : volume faible, une
|
||
* notif par fiche traitée suffit — cf. B5-proposer-pipe.md §Points d'attention).
|
||
*/
|
||
|
||
import { spawnSync, execSync } from 'child_process';
|
||
import { existsSync, writeFileSync, unlinkSync, readFileSync, mkdirSync } from 'fs';
|
||
import { dirname, join } from 'path';
|
||
import { fileURLToPath } from 'url';
|
||
|
||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||
|
||
// ─── CONFIG DEPUIS .env ───────────────────────────────────────────────────────
|
||
const NOCODB_URL = process.env.NOCODB_URL || 'http://localhost:8070';
|
||
const NOCODB_TOKEN = process.env.NOCODB_TOKEN;
|
||
const NOCODB_BASE = process.env.NOCODB_BASE;
|
||
const NOCODB_TABLE_ORGAS = process.env.NOCODB_TABLE_ORGAS;
|
||
const NOCODB_TABLE_STATS = process.env.NOCODB_TABLE_STATS;
|
||
|
||
const BIFROST_URL = process.env.BIFROST_URL || 'http://127.0.0.1:8080';
|
||
const BIFROST_VK = process.env.BIFROST_VK;
|
||
const WORKER_MODEL = process.env.WORKER_MODEL || 'groq/openai/gpt-oss-20b'; // llama-3.1-8b-instant retiré par Groq (28/09)
|
||
|
||
const NTFY_TOPIC = process.env.NTFY_TOPIC;
|
||
|
||
const BUDGET_MAX_EUR = parseFloat(process.env.BUDGET_MAX_EUR || '20');
|
||
const WORKER_LIMIT = parseInt(process.env.WORKER_LIMIT || '5');
|
||
const LOCK_FILE = process.env.WORKER_LOCK_FILE || '/tmp/aep-worker.lock';
|
||
|
||
// Prix par 1M tokens (USD) — par défaut 0 (Groq gratuit dans le tier RAPIDE Bifrost).
|
||
// Réglable si le worker bascule un jour sur un modèle payant.
|
||
const WORKER_PRICE_IN = parseFloat(process.env.WORKER_PRICE_IN_USD_PER_M || '0') / 1_000_000;
|
||
const WORKER_PRICE_OUT = parseFloat(process.env.WORKER_PRICE_OUT_USD_PER_M || '0') / 1_000_000;
|
||
const USD_TO_EUR = 0.93;
|
||
|
||
// Scrape léger
|
||
const SCRAPE_TIMEOUT_MS = 8_000;
|
||
const SCRAPE_MAX_BYTES = 500_000;
|
||
const SCRAPE_TEXT_MAX_CHARS = 4_000;
|
||
const SCRAPE_USER_AGENT = 'AEP/2.0 contact@trans-former.fr';
|
||
|
||
// ─── CLI ──────────────────────────────────────────────────────────────────────
|
||
const DRY_RUN = process.argv.includes('--dry-run');
|
||
// DRY_RUN_LIVE=1 : en dry-run, tente quand même de vrais appels réseau (scrape + Bifrost)
|
||
// pour un test manuel avec accès réseau — ne touche JAMAIS NocoDB en dry-run, dans tous les cas.
|
||
const DRY_RUN_LIVE = process.env.DRY_RUN_LIVE === '1';
|
||
const DRY_RUN_FIXTURE = process.env.DRY_RUN_FIXTURE || join(__dirname, 'fixtures', 'dry-run-rows.json');
|
||
|
||
// ─── TAXONOMIE VALIDE (apostrophe typographique U+2019 comme NocoDB) ─────────
|
||
const VALID_FONCTIONS = [
|
||
'Juridique', 'Technique', 'Économique', 'Administratif', 'Chantier',
|
||
'Comptabilité', 'Développement', 'Formation', 'Gestion d’agence', 'Santé mentale'
|
||
];
|
||
|
||
const VALID_SUBMISSION_TYPES = ['ecosysteme', 'reseau', 'job', 'outil'];
|
||
|
||
// ─── MAPPING NORMALISATION TAGS ───────────────────────────────────────────────
|
||
const TAG_MAP = [
|
||
[['juridique', 'droit', 'litige', 'contrat', 'déontologie', 'décennale', 'médiation', 'pi ', 'propriété intellectuelle', 'ccag', 'marchés publics droit'], 'Juridique'],
|
||
[['technique', 're2020', 'thermique', 'structure', 'bim', 'dtu', 'acoustique', 'matériaux', 'simulation', 'opr', 'réserves', 'acv', 'pcd', 'eurocodes', 'pmc'], 'Technique'],
|
||
[['économique', 'prix', 'tarif', 'honoraire', 'devis', 'roi', 'financement', 'subvention', 'cee', 'maprimerenov', 'anah', 'business plan', 'pricing'], 'Économique'],
|
||
[['administratif', 'permis', 'plu', 'plui', 'erp', 'autorisation travaux', 'abf', 'patrimoine', 'marchés publics procédure', 'cctp', 'dpgf', 'concours', 'urbanisme'], 'Administratif'],
|
||
[['chantier', 'coordination chantier', 'det', 'suivi travaux', 'sps', 'sécurité chantier', 'planning chantier', 'entreprise', 'sous-traitance', 'réception travaux'], 'Chantier'],
|
||
[['comptabilité', 'fiscal', 'tva', 'bnc', 'bic', 'expert-comptable', 'bilan', 'trésorerie', 'transmission agence', 'création agence', 'micro'], 'Comptabilité'],
|
||
[['développement', 'prospection', 'commercial', 'client', 'réseau', 'candidature', 'consultation', 'acquisition', 'marketing', 'notoriété', 'ao '], 'Développement'],
|
||
[['formation', 'école', 'mooc', 'organisme', 'formation continue', 'cpf', 'dpc', 'cfaa'], 'Formation'],
|
||
[['gestion d’agence', 'gestion d\'agence', 'rh', 'recrutement', 'emploi', 'salaire', 'ccn', 'convention collective', 'idcc', 'temps de travail', 'management'], 'Gestion d’agence'],
|
||
[['santé mentale', 'burn-out', 'épuisement', 'souffrance', 'bien-être', 'harcèlement', 'stress', 'psychologique', 'équilibre'], 'Santé mentale'],
|
||
];
|
||
|
||
function normalizeTag(raw) {
|
||
const t = String(raw).toLowerCase().trim();
|
||
const exact = VALID_FONCTIONS.find(v => v.toLowerCase() === t);
|
||
if (exact) return exact;
|
||
for (const [patterns, normalized] of TAG_MAP) {
|
||
if (patterns.some(p => t.includes(p))) return normalized;
|
||
}
|
||
return null;
|
||
}
|
||
|
||
// ─── UTILITAIRES LOG ─────────────────────────────────────────────────────────
|
||
function log(...args) {
|
||
const ts = new Date().toISOString();
|
||
console.log(`[${ts}]`, ...args);
|
||
}
|
||
|
||
// ─── LOCK ANTI-OVERLAP ────────────────────────────────────────────────────────
|
||
function acquireLock() {
|
||
if (existsSync(LOCK_FILE)) {
|
||
const content = execSync(`cat ${LOCK_FILE}`).toString().trim();
|
||
const pid = parseInt(content);
|
||
try {
|
||
execSync(`kill -0 ${pid} 2>/dev/null`);
|
||
return false; // Process encore vivant
|
||
} catch {
|
||
log('Lock orphelin détecté, suppression');
|
||
unlinkSync(LOCK_FILE);
|
||
}
|
||
}
|
||
writeFileSync(LOCK_FILE, process.pid.toString());
|
||
return true;
|
||
}
|
||
|
||
function releaseLock() {
|
||
try { unlinkSync(LOCK_FILE); } catch {}
|
||
}
|
||
|
||
// ─── NOCODB API ──────────────────────────────────────────────────────────────
|
||
async function nocodbGet(path) {
|
||
const res = await fetch(`${NOCODB_URL}/api/v1/db/data/noco/${NOCODB_BASE}/${path}`, {
|
||
headers: { 'xc-token': NOCODB_TOKEN }
|
||
});
|
||
if (!res.ok) throw new Error(`NocoDB GET ${path} → ${res.status}: ${await res.text()}`);
|
||
return res.json();
|
||
}
|
||
|
||
async function nocodbPatch(tableId, rowId, data) {
|
||
const res = await fetch(`${NOCODB_URL}/api/v1/db/data/noco/${NOCODB_BASE}/${tableId}/${rowId}`, {
|
||
method: 'PATCH',
|
||
headers: { 'xc-token': NOCODB_TOKEN, 'Content-Type': 'application/json' },
|
||
body: JSON.stringify(data)
|
||
});
|
||
if (!res.ok) throw new Error(`NocoDB PATCH ${tableId}/${rowId} → ${res.status}: ${await res.text()}`);
|
||
return res.json();
|
||
}
|
||
|
||
async function nocodbPost(tableId, data) {
|
||
const res = await fetch(`${NOCODB_URL}/api/v1/db/data/noco/${NOCODB_BASE}/${tableId}`, {
|
||
method: 'POST',
|
||
headers: { 'xc-token': NOCODB_TOKEN, 'Content-Type': 'application/json' },
|
||
body: JSON.stringify(data)
|
||
});
|
||
if (!res.ok) throw new Error(`NocoDB POST ${tableId} → ${res.status}: ${await res.text()}`);
|
||
return res.json();
|
||
}
|
||
|
||
/** Écrit une mise à jour de fiche — no-op loggé en dry-run (jamais de PATCH réel). */
|
||
async function patchRow(rowId, data) {
|
||
if (DRY_RUN) {
|
||
log(`[dry-run] PATCH fiche ${rowId} :`, JSON.stringify(data));
|
||
return;
|
||
}
|
||
return nocodbPatch(NOCODB_TABLE_ORGAS, rowId, data);
|
||
}
|
||
|
||
// ─── BUDGET CIRCUIT BREAKER ──────────────────────────────────────────────────
|
||
async function getBudgetMoisCourant() {
|
||
if (DRY_RUN) return 0;
|
||
const now = new Date();
|
||
const year = now.getFullYear();
|
||
const month = now.getMonth(); // 0-indexed
|
||
// NocoDB ne supporte pas bien le filtre datetime — on récupère tout et filtre en JS
|
||
try {
|
||
const data = await nocodbGet(`${NOCODB_TABLE_STATS}?limit=1000&sort=-timestamp`);
|
||
const total = (data.list || []).reduce((sum, row) => {
|
||
const ts = new Date(row.timestamp || row.CreatedAt || 0);
|
||
if (ts.getFullYear() === year && ts.getMonth() === month) {
|
||
return sum + (parseFloat(row.cout_eur) || 0);
|
||
}
|
||
return sum;
|
||
}, 0);
|
||
return total;
|
||
} catch (e) {
|
||
log('Erreur lecture budget:', e.message);
|
||
return 0;
|
||
}
|
||
}
|
||
|
||
async function logUsage(usage, model, endpoint, orgaId) {
|
||
const tokensIn = usage?.prompt_tokens || 0;
|
||
const tokensOut = usage?.completion_tokens || 0;
|
||
const coutEur = ((tokensIn * WORKER_PRICE_IN) + (tokensOut * WORKER_PRICE_OUT)) * USD_TO_EUR;
|
||
|
||
if (DRY_RUN) {
|
||
log(`[dry-run] Usage (non loggé) : ${tokensIn}in + ${tokensOut}out = €${coutEur.toFixed(6)} (${model})`);
|
||
return coutEur;
|
||
}
|
||
|
||
await nocodbPost(NOCODB_TABLE_STATS, {
|
||
model,
|
||
endpoint,
|
||
tokens_in: tokensIn,
|
||
tokens_out: tokensOut,
|
||
cout_eur: parseFloat(coutEur.toFixed(6)),
|
||
timestamp: new Date().toISOString(),
|
||
orga_id: orgaId || null
|
||
});
|
||
|
||
log(`Usage log: ${tokensIn}in + ${tokensOut}out = €${coutEur.toFixed(6)} (${model})`);
|
||
return coutEur;
|
||
}
|
||
|
||
// ─── FETCH FICHES PENDING ────────────────────────────────────────────────────
|
||
async function fetchPendingRows() {
|
||
const data = await nocodbGet(
|
||
`${NOCODB_TABLE_ORGAS}?where=(moderation_status,eq,pending)~and(ai_processed,eq,false)&limit=${WORKER_LIMIT}&sort=submitted_at`
|
||
);
|
||
return data.list || [];
|
||
}
|
||
|
||
function loadFixtureRows() {
|
||
if (!existsSync(DRY_RUN_FIXTURE)) {
|
||
throw new Error(`Fixture dry-run introuvable : ${DRY_RUN_FIXTURE}`);
|
||
}
|
||
const raw = JSON.parse(readFileSync(DRY_RUN_FIXTURE, 'utf-8'));
|
||
return Array.isArray(raw) ? raw : [raw];
|
||
}
|
||
|
||
// ─── LIENS — extraits de url + description_user (format B5-M1 "Liens :\n...") ─
|
||
function extractLinks(row) {
|
||
const text = `${row.url || ''}\n${row.description_user || row.description || ''}`;
|
||
const matches = text.match(/https?:\/\/[^\s)"'<>]+/g) || [];
|
||
const cleaned = matches.map(u => u.replace(/[.,;:!?]+$/, ''));
|
||
return [...new Set(cleaned)];
|
||
}
|
||
|
||
// ─── SCRAPE LÉGER (fetch natif, pas de crawl4ai) ─────────────────────────────
|
||
async function fetchCapped(url) {
|
||
const controller = new AbortController();
|
||
const timer = setTimeout(() => controller.abort(), SCRAPE_TIMEOUT_MS);
|
||
try {
|
||
const res = await fetch(url, {
|
||
signal: controller.signal,
|
||
redirect: 'follow',
|
||
headers: { 'User-Agent': SCRAPE_USER_AGENT },
|
||
});
|
||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||
|
||
const reader = res.body?.getReader();
|
||
if (!reader) return await res.text();
|
||
|
||
const chunks = [];
|
||
let total = 0;
|
||
while (true) {
|
||
const { done, value } = await reader.read();
|
||
if (done) break;
|
||
chunks.push(value);
|
||
total += value.length;
|
||
if (total >= SCRAPE_MAX_BYTES) {
|
||
await reader.cancel().catch(() => {});
|
||
break;
|
||
}
|
||
}
|
||
const buf = Buffer.concat(chunks.map(c => Buffer.from(c)));
|
||
return buf.subarray(0, SCRAPE_MAX_BYTES).toString('utf-8');
|
||
} finally {
|
||
clearTimeout(timer);
|
||
}
|
||
}
|
||
|
||
function extractMeta(html, key) {
|
||
const re1 = new RegExp(`<meta[^>]*(?:name|property)=["']${key}["'][^>]*content=["']([^"']*)["']`, 'i');
|
||
const re2 = new RegExp(`<meta[^>]*content=["']([^"']*)["'][^>]*(?:name|property)=["']${key}["']`, 'i');
|
||
return (html.match(re1) || html.match(re2))?.[1]?.trim() || null;
|
||
}
|
||
|
||
function extractOgTags(html) {
|
||
const og = {};
|
||
const re1 = /<meta[^>]*property=["']og:([a-zA-Z:_-]+)["'][^>]*content=["']([^"']*)["']/gi;
|
||
const re2 = /<meta[^>]*content=["']([^"']*)["'][^>]*property=["']og:([a-zA-Z:_-]+)["']/gi;
|
||
let m;
|
||
while ((m = re1.exec(html))) og[m[1]] = m[2];
|
||
while ((m = re2.exec(html))) if (!(m[2] in og)) og[m[2]] = m[1];
|
||
return og;
|
||
}
|
||
|
||
function decodeEntities(s) {
|
||
return s
|
||
.replace(/ /g, ' ')
|
||
.replace(/&/g, '&')
|
||
.replace(/</g, '<')
|
||
.replace(/>/g, '>')
|
||
.replace(/"/g, '"')
|
||
.replace(/�?39;/g, "'");
|
||
}
|
||
|
||
function extractVisibleText(html) {
|
||
let s = html
|
||
.replace(/<!--[\s\S]*?-->/g, ' ')
|
||
.replace(/<script[\s\S]*?<\/script>/gi, ' ')
|
||
.replace(/<style[\s\S]*?<\/style>/gi, ' ')
|
||
.replace(/<nav[\s\S]*?<\/nav>/gi, ' ')
|
||
.replace(/<footer[\s\S]*?<\/footer>/gi, ' ')
|
||
.replace(/<head[\s\S]*?<\/head>/gi, ' ');
|
||
s = s.replace(/<[^>]+>/g, ' ');
|
||
s = decodeEntities(s);
|
||
s = s.replace(/\s+/g, ' ').trim();
|
||
return s.slice(0, SCRAPE_TEXT_MAX_CHARS);
|
||
}
|
||
|
||
function extractTitle(html) {
|
||
return html.match(/<title[^>]*>([^<]*)<\/title>/i)?.[1]?.trim() || null;
|
||
}
|
||
|
||
async function scrapeLight(url) {
|
||
log(`Scraping léger: ${url}`);
|
||
const html = await fetchCapped(url);
|
||
return {
|
||
title: extractTitle(html),
|
||
metaDescription: extractMeta(html, 'description'),
|
||
og: extractOgTags(html),
|
||
text: extractVisibleText(html),
|
||
};
|
||
}
|
||
|
||
const MOCK_SCRAPE_RESULT = {
|
||
title: '[mock dry-run] Titre de la page',
|
||
metaDescription: '[mock dry-run] Meta description factice, aucun réseau contacté.',
|
||
og: { site_name: '[mock dry-run]' },
|
||
text: '[mock dry-run] Extrait de texte factice utilisé quand DRY_RUN_LIVE n’est pas activé.',
|
||
};
|
||
|
||
// ─── APPEL LLM VIA BIFROST ────────────────────────────────────────────────────
|
||
const SYSTEM_PROMPT = `Tu es un assistant qui aide à qualifier des ressources soumises pour une cartographie collaborative de l'écosystème professionnel de l'architecture en France (AEP).
|
||
|
||
RÈGLES ABSOLUES :
|
||
1. Tu ne dois JAMAIS inventer d'informations non présentes dans les sources fournies.
|
||
2. Si une information est absente ou incertaine, retourne \`null\` pour ce champ.
|
||
3. Tu dois retourner UNIQUEMENT un objet JSON valide, sans texte avant ou après.
|
||
4. "description" : neutre, factuelle, en français, max 300 caractères, sans jugement de valeur.
|
||
5. "nom" : un nom de fiche court et identifiable, sans le préfixe "[à qualifier]".
|
||
6. "tags" : 1 à 5 valeurs, uniquement parmi la liste autorisée.
|
||
|
||
TAGS AUTORISÉS : "Juridique" | "Technique" | "Économique" | "Administratif" | "Chantier" | "Comptabilité" | "Développement" | "Formation" | "Gestion d’agence" | "Santé mentale"
|
||
TYPE_SUGGERE AUTORISÉ (une seule valeur ou null) : "ecosysteme" | "reseau" | "job" | "outil"
|
||
|
||
FORMAT DE SORTIE JSON :
|
||
{
|
||
"nom": "string | null",
|
||
"description": "string (max 300 chars, français, neutre, factuel) | null",
|
||
"type_suggere": "ecosysteme" | "reseau" | "job" | "outil" | null,
|
||
"ville": "string | null",
|
||
"tags": ["string", "..."],
|
||
"confiance": "haute" | "moyenne" | "faible"
|
||
}
|
||
|
||
Le champ "confiance" reflète ta certitude globale :
|
||
- "haute" : contenu scrapé riche, informations claires.
|
||
- "moyenne" : contenu partiel, ou texte du contributeur seul mais clair.
|
||
- "faible" : aucun contenu scrapé et texte du contributeur vague, inférences importantes.`;
|
||
|
||
function buildUserPrompt(row, scrapeData, autresLiens) {
|
||
return `RESSOURCE À QUALIFIER :
|
||
|
||
Nom actuel (placeholder à remplacer) : ${row.nom}
|
||
Lien principal : ${row.url || 'non fourni'}
|
||
${autresLiens.length ? `Autres liens mentionnés par le contributeur : ${autresLiens.join(', ')}` : ''}
|
||
Texte du contributeur (pourquoi c'est pertinent pour lui) : ${row.description_user || row.description || 'non fourni'}
|
||
|
||
CONTENU EXTRAIT DU SITE (scraping léger) :
|
||
Titre : ${scrapeData?.title || 'non disponible'}
|
||
Meta description : ${scrapeData?.metaDescription || 'non disponible'}
|
||
Open Graph : ${scrapeData && Object.keys(scrapeData.og || {}).length ? JSON.stringify(scrapeData.og) : 'non disponible'}
|
||
Texte visible (extrait) : ${scrapeData?.text || 'Site non accessible ou lien non fourni.'}
|
||
|
||
---
|
||
|
||
Qualifie cette ressource selon les règles du system prompt. Retourne uniquement le JSON.`;
|
||
}
|
||
|
||
async function callBifrostWithRetry(row, scrapeData, autresLiens, maxRetries = 2) {
|
||
const userPrompt = buildUserPrompt(row, scrapeData, autresLiens);
|
||
|
||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||
if (attempt > 0) {
|
||
log(`Retry ${attempt}/${maxRetries} pour fiche ${row.Id}`);
|
||
await new Promise(r => setTimeout(r, 2000 * attempt));
|
||
}
|
||
|
||
try {
|
||
const res = await fetch(`${BIFROST_URL}/v1/chat/completions`, {
|
||
method: 'POST',
|
||
headers: {
|
||
'x-bf-vk': BIFROST_VK,
|
||
'Content-Type': 'application/json'
|
||
},
|
||
body: JSON.stringify({
|
||
model: WORKER_MODEL,
|
||
temperature: 0.2,
|
||
max_tokens: 800,
|
||
response_format: { type: 'json_object' },
|
||
messages: [
|
||
{ role: 'system', content: SYSTEM_PROMPT },
|
||
{ role: 'user', content: userPrompt }
|
||
]
|
||
}),
|
||
signal: AbortSignal.timeout(60_000)
|
||
});
|
||
|
||
if (!res.ok) {
|
||
const err = await res.text();
|
||
throw new Error(`Bifrost API ${res.status}: ${err}`);
|
||
}
|
||
|
||
const data = await res.json();
|
||
const content = data.choices?.[0]?.message?.content;
|
||
if (!content) throw new Error('Réponse Bifrost vide');
|
||
|
||
const parsed = JSON.parse(content);
|
||
parsed._usage = data.usage;
|
||
parsed._raw = content;
|
||
return parsed;
|
||
|
||
} catch (e) {
|
||
log(`Erreur Bifrost (tentative ${attempt + 1}): ${e.message}`);
|
||
if (attempt === maxRetries) return null;
|
||
}
|
||
}
|
||
return null;
|
||
}
|
||
|
||
const MOCK_BIFROST_RESULT = {
|
||
nom: '[mock dry-run] Nom suggéré',
|
||
description: '[mock dry-run] Description factice générée sans appel réseau.',
|
||
type_suggere: 'ecosysteme',
|
||
ville: null,
|
||
tags: ['Développement'],
|
||
confiance: 'faible',
|
||
_usage: { prompt_tokens: 0, completion_tokens: 0 },
|
||
_raw: '{"mock":"dry-run"}',
|
||
};
|
||
|
||
// ─── NOTIFICATION NTFY (jamais l'email ni le texte du contributeur) ──────────
|
||
async function notifyNtfy(title, message) {
|
||
if (DRY_RUN) {
|
||
log(`[dry-run] ntfy (non envoyé) — ${title} :`, message.replace(/\n/g, ' | '));
|
||
return;
|
||
}
|
||
if (!NTFY_TOPIC) {
|
||
log('NTFY_TOPIC absent, notification skippée');
|
||
return;
|
||
}
|
||
try {
|
||
const res = await fetch(`https://ntfy.sh/${NTFY_TOPIC}`, {
|
||
method: 'POST',
|
||
headers: { Title: title, Priority: '3', Tags: 'inbox_tray' },
|
||
body: message,
|
||
});
|
||
if (!res.ok) log('ntfy erreur:', await res.text());
|
||
else log('ntfy envoyé:', title);
|
||
} catch (e) {
|
||
log('ntfy exception:', e.message);
|
||
}
|
||
}
|
||
|
||
// ─── MAIN ─────────────────────────────────────────────────────────────────────
|
||
async function run() {
|
||
if (!DRY_RUN && !acquireLock()) {
|
||
log('Worker déjà en cours, skip');
|
||
process.exit(0);
|
||
}
|
||
|
||
const startTime = Date.now();
|
||
log(`=== Worker AEP enrichissement démarré${DRY_RUN ? ' (--dry-run)' : ''} ===`);
|
||
|
||
try {
|
||
if (!DRY_RUN) {
|
||
if (!NOCODB_TOKEN || !NOCODB_BASE || !NOCODB_TABLE_ORGAS || !BIFROST_VK) {
|
||
throw new Error('Variables .env manquantes (NOCODB_TOKEN, NOCODB_BASE, NOCODB_TABLE_ORGAS, BIFROST_VK)');
|
||
}
|
||
}
|
||
|
||
const budgetMois = await getBudgetMoisCourant();
|
||
log(`Budget mois courant: €${budgetMois.toFixed(4)} / €${BUDGET_MAX_EUR}`);
|
||
|
||
if (budgetMois >= BUDGET_MAX_EUR) {
|
||
log('Budget épuisé pour ce mois. Worker en pause.');
|
||
await notifyNtfy('[AEP] Budget IA épuisé', `Le budget de ${BUDGET_MAX_EUR}€ a été atteint. Worker en pause jusqu'au 1er du mois prochain.`);
|
||
return;
|
||
}
|
||
|
||
const rows = DRY_RUN ? loadFixtureRows() : await fetchPendingRows();
|
||
log(`${rows.length} fiche(s) à traiter${DRY_RUN ? ' (fixture)' : ''}`);
|
||
|
||
if (rows.length === 0) {
|
||
log('Rien à traiter.');
|
||
return;
|
||
}
|
||
|
||
let processedCount = 0;
|
||
|
||
for (const row of rows) {
|
||
const rowStart = Date.now();
|
||
log(`--- Traitement fiche ${row.Id}: ${row.nom} ---`);
|
||
|
||
const budgetCheck = await getBudgetMoisCourant();
|
||
if (budgetCheck >= BUDGET_MAX_EUR) {
|
||
log('Budget atteint mid-pipeline, arrêt.');
|
||
break;
|
||
}
|
||
|
||
const liens = extractLinks(row);
|
||
const primaryUrl = row.url && row.url.trim() ? row.url.trim() : null;
|
||
const autresLiens = liens.filter(l => l !== primaryUrl);
|
||
|
||
// Scraping (uniquement le lien principal — un seul champ scrape_content en base)
|
||
let scrapeData = null;
|
||
const shouldScrape = primaryUrl && (row.scrape_status === 'pending' || !row.scrape_status);
|
||
const useNetwork = !DRY_RUN || DRY_RUN_LIVE;
|
||
|
||
if (shouldScrape) {
|
||
try {
|
||
scrapeData = useNetwork ? await scrapeLight(primaryUrl) : MOCK_SCRAPE_RESULT;
|
||
await patchRow(row.Id, {
|
||
scrape_status: 'scraped',
|
||
scrape_content: JSON.stringify(scrapeData),
|
||
});
|
||
} catch (e) {
|
||
log(`Scrape échoué: ${e.message}`);
|
||
await patchRow(row.Id, { scrape_status: 'failed' });
|
||
}
|
||
} else if (!primaryUrl) {
|
||
await patchRow(row.Id, { scrape_status: 'no_link' });
|
||
}
|
||
|
||
// Appel LLM (Bifrost)
|
||
const enriched = useNetwork
|
||
? await callBifrostWithRetry(row, scrapeData, autresLiens)
|
||
: MOCK_BIFROST_RESULT;
|
||
|
||
if (!enriched) {
|
||
log(`Échec Bifrost sur fiche ${row.Id}, flag ai_error`);
|
||
await patchRow(row.Id, { moderation_status: 'ai_error', ai_processed: true });
|
||
continue;
|
||
}
|
||
|
||
// Normalisation tags
|
||
const rawTags = enriched.tags || enriched.tags_fonction || [];
|
||
const normalizedTags = [...new Set(rawTags.map(normalizeTag).filter(Boolean))];
|
||
|
||
const updateData = {
|
||
description_enrichie: enriched.description || null,
|
||
tags_fonction: normalizedTags.join(','),
|
||
moderation_status: 'ai_processed',
|
||
ai_processed: true,
|
||
ai_raw_output: JSON.stringify({ output: enriched, confiance: enriched.confiance }),
|
||
};
|
||
|
||
// Nom : ne remplacer que le placeholder posé par le formulaire assoupli (B5-M1).
|
||
if (enriched.nom && typeof row.nom === 'string' && row.nom.startsWith('[à qualifier]')) {
|
||
updateData.nom = enriched.nom;
|
||
}
|
||
if (enriched.ville && !row.localisation_ville) {
|
||
updateData.localisation_ville = enriched.ville;
|
||
}
|
||
// submission_type : ne réajuster que si l'utilisateur n'avait pas choisi de chip
|
||
// (marqueur posé par utils/submitLibre.ts::toOrgaPayload — "Type : non précisé").
|
||
const typeNonPrecise = typeof row.description_user === 'string'
|
||
&& row.description_user.startsWith('Type : non précisé');
|
||
if (typeNonPrecise && VALID_SUBMISSION_TYPES.includes(enriched.type_suggere)) {
|
||
updateData.submission_type = enriched.type_suggere;
|
||
}
|
||
|
||
await patchRow(row.Id, updateData);
|
||
await logUsage(enriched._usage, WORKER_MODEL, 'enrichissement', row.Id);
|
||
|
||
await notifyNtfy(
|
||
'[AEP] Fiche enrichie',
|
||
[
|
||
`Id NocoDB : ${row.Id}`,
|
||
`Nom suggéré : ${updateData.nom || row.nom}`,
|
||
`Type : ${updateData.submission_type || row.submission_type || 'nc'}`,
|
||
`Confiance : ${enriched.confiance || 'nc'}`,
|
||
].join('\n'),
|
||
);
|
||
|
||
const elapsed = ((Date.now() - rowStart) / 1000).toFixed(1);
|
||
log(`Fiche ${row.Id} traitée en ${elapsed}s — confiance: ${enriched.confiance || 'nc'}`);
|
||
processedCount++;
|
||
}
|
||
|
||
log(`=== Run terminé: ${processedCount}/${rows.length} fiches traitées en ${((Date.now() - startTime) / 1000).toFixed(1)}s ===`);
|
||
|
||
} catch (e) {
|
||
log('ERREUR WORKER:', e.message);
|
||
console.error(e.stack);
|
||
process.exitCode = 1;
|
||
} finally {
|
||
if (!DRY_RUN) releaseLock();
|
||
}
|
||
}
|
||
|
||
run();
|