Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018BAUeCFpDkRD6tU5wGsc1C
128 lines
4.9 KiB
JavaScript
128 lines
4.9 KiB
JavaScript
#!/usr/bin/env node
|
|
/**
|
|
* Construit le corpus de connaissances de l'assistant a partir du wiki du site.
|
|
*
|
|
* Le wiki n'est pas ecrit en dur dans des pages : son contenu vit dans les
|
|
* fichiers de traduction du frontend, sous `dashboard.wikiPages`. C'est donc la
|
|
* source de verite, et la meme que celle que lit l'utilisateur — une reponse de
|
|
* l'assistant et la page wiki citee ne peuvent pas diverger.
|
|
*
|
|
* Le corpus est ecrit dans le backend et versionne : l'image backend ne doit
|
|
* pas dependre des fichiers du frontend a l'execution.
|
|
*
|
|
* Usage : npm run knowledge:build
|
|
*/
|
|
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
|
|
const ROOT = path.resolve(__dirname, '../../../..');
|
|
const MESSAGES = path.join(ROOT, 'apps/frontend/messages');
|
|
const OUT = path.resolve(__dirname, '../../src/infrastructure/ai/knowledge/wiki-corpus.json');
|
|
|
|
const LOCALES = ['fr', 'en'];
|
|
|
|
/** Les cles de mise en page ne portent aucune connaissance. */
|
|
const LAYOUT_KEYS = /^(col[A-Z]|.*Title$|.*Label$|backToWiki)/;
|
|
|
|
/** `documentsTransport` -> `documents-transport`, l'URL de la page wiki. */
|
|
const toSlug = key => key.replace(/([a-z0-9])([A-Z])/g, '$1-$2').toLowerCase();
|
|
|
|
const humanize = key =>
|
|
key
|
|
.replace(/([a-z0-9])([A-Z])/g, '$1 $2')
|
|
.replace(/^./, c => c.toUpperCase())
|
|
.trim();
|
|
|
|
/**
|
|
* Nomme un champ d'objet dans la langue du wiki.
|
|
*
|
|
* Les cles de traduction sont en anglais (`code`, `name`, `description`) mais
|
|
* chaque sujet publie deja ses en-tetes de colonnes (`colCode`, `colName`...) :
|
|
* les reutiliser evite d'ecrire « Name: » au milieu d'un fragment francais.
|
|
*/
|
|
const labelFor = (topic, key) => topic[`col${key[0].toUpperCase()}${key.slice(1)}`] ?? humanize(key);
|
|
|
|
/** Aplatit une valeur de traduction en lignes lisibles par un modele. */
|
|
function toLines(value, topic) {
|
|
if (typeof value === 'string') return [value];
|
|
if (typeof value === 'number' || typeof value === 'boolean') return [String(value)];
|
|
if (Array.isArray(value)) return value.flatMap(item => toLines(item, topic));
|
|
|
|
if (value && typeof value === 'object') {
|
|
// Un objet de table se lit mieux sur une ligne qu'eclate en champs :
|
|
// « Code: 40 00 — Nom: Mise en Libre Pratique — Description: ... ».
|
|
const entries = Object.entries(value).filter(([, v]) => v !== null && v !== undefined);
|
|
const scalars = entries.filter(([, v]) => typeof v === 'string' || typeof v === 'number');
|
|
const rest = entries.filter(([, v]) => typeof v === 'object');
|
|
|
|
const head = scalars.map(([k, v]) => `${labelFor(topic, k)}: ${v}`).join(' — ');
|
|
return [
|
|
head,
|
|
...rest.flatMap(([k, v]) => toLines(v, topic).map(line => `${labelFor(topic, k)}: ${line}`)),
|
|
].filter(Boolean);
|
|
}
|
|
|
|
return [];
|
|
}
|
|
|
|
/**
|
|
* Un fragment par section du sujet. Une section = un champ de premier niveau,
|
|
* intitule par son `*Title` voisin quand il existe. Decouper plus finement
|
|
* casserait les tableaux (un Incoterm isole de sa colonne « risque ») ;
|
|
* decouper moins finement noierait la reponse sous 4 000 caracteres.
|
|
*/
|
|
function chunksForTopic(locale, topicKey, topic) {
|
|
const title = topic.title ?? humanize(topicKey);
|
|
const href = `/dashboard/wiki/${toSlug(topicKey)}`;
|
|
const chunks = [];
|
|
|
|
const header = [topic.title, topic.description].filter(Boolean).join('\n');
|
|
if (header) {
|
|
chunks.push({ section: title, text: header });
|
|
}
|
|
|
|
for (const [key, value] of Object.entries(topic)) {
|
|
if (key === 'title' || key === 'description') continue;
|
|
if (LAYOUT_KEYS.test(key)) continue;
|
|
|
|
const lines = toLines(value, topic).filter(Boolean);
|
|
if (!lines.length) continue;
|
|
|
|
const section = topic[`${key}Title`] ?? humanize(key);
|
|
chunks.push({ section, text: `${section}\n${lines.map(line => `- ${line}`).join('\n')}` });
|
|
}
|
|
|
|
return chunks.map((chunk, index) => ({
|
|
id: `${locale}:${topicKey}:${index}`,
|
|
locale,
|
|
topic: topicKey,
|
|
title,
|
|
section: chunk.section,
|
|
href,
|
|
text: chunk.text,
|
|
}));
|
|
}
|
|
|
|
const documents = [];
|
|
|
|
for (const locale of LOCALES) {
|
|
const file = path.join(MESSAGES, `${locale}.json`);
|
|
const wiki = JSON.parse(fs.readFileSync(file, 'utf8')).dashboard?.wikiPages;
|
|
if (!wiki) throw new Error(`dashboard.wikiPages introuvable dans ${file}`);
|
|
|
|
for (const [topicKey, topic] of Object.entries(wiki)) {
|
|
// Les libelles partages (`responsibleLabel`...) sont des chaines, pas des sujets.
|
|
if (!topic || typeof topic !== 'object' || Array.isArray(topic)) continue;
|
|
documents.push(...chunksForTopic(locale, topicKey, topic));
|
|
}
|
|
}
|
|
|
|
fs.mkdirSync(path.dirname(OUT), { recursive: true });
|
|
fs.writeFileSync(OUT, JSON.stringify({ documents }, null, 2) + '\n');
|
|
|
|
const byLocale = LOCALES.map(l => `${l}: ${documents.filter(d => d.locale === l).length}`).join(', ');
|
|
const chars = documents.reduce((sum, d) => sum + d.text.length, 0);
|
|
console.log(`${documents.length} fragments (${byLocale}) — ${chars} caracteres`);
|
|
console.log(`écrit dans ${path.relative(ROOT, OUT)}`);
|