271 lines
9.6 KiB
TypeScript
271 lines
9.6 KiB
TypeScript
import { ConfigService } from '@nestjs/config';
|
|
import { CachePort } from '@domain/ports/out/cache.port';
|
|
import { TradeEmbeddingPort } from '@domain/ports/out/trade-assistant.port';
|
|
import {
|
|
WikiContribution,
|
|
WikiContributionStatus,
|
|
} from '@domain/entities/wiki-contribution.entity';
|
|
import { WikiRetriever, normalizeQuestion, pack, unpack } from './wiki-retriever';
|
|
|
|
/**
|
|
* Embedder deterministe : un sac de mots sur un vocabulaire metier reduit. Le
|
|
* classement obtenu est donc reellement lexical, ce qui permet d'affirmer
|
|
* qu'une question sur la douane remonte la page douane.
|
|
*/
|
|
const VOCABULARY = [
|
|
'douane',
|
|
'douanieres',
|
|
'douaniers',
|
|
'incoterm',
|
|
'incoterms',
|
|
'conteneur',
|
|
'conteneurs',
|
|
'assurance',
|
|
'vgm',
|
|
'imdg',
|
|
];
|
|
|
|
/** Dimensions de reserve, pour les textes sans mot du vocabulaire metier. */
|
|
const BUCKETS = 64;
|
|
|
|
function fakeVector(text: string): number[] {
|
|
const words = normalizeQuestion(text).split(' ');
|
|
const vector = VOCABULARY.map(term => words.filter(word => word === term).length);
|
|
vector.push(...new Array<number>(BUCKETS).fill(0));
|
|
|
|
const norm = Math.sqrt(vector.reduce((sum, v) => sum + v * v, 0));
|
|
if (norm > 0) return vector.map(v => v / norm);
|
|
|
|
// Sans terme commun, deux textes doivent etre quasi orthogonaux. Un vecteur
|
|
// uniforme les rendait au contraire identiques : tout ressemblait a tout, et
|
|
// aucun seuil de pertinence n'etait observable.
|
|
//
|
|
// Le retriever compose ses documents en « titre — section\ntexte » : ce
|
|
// separateur les distingue d'une question. Les deux familles occupent des
|
|
// moities de dimensions disjointes, pour qu'aucune collision fortuite ne
|
|
// rapproche une question d'un document qui n'a rien a voir avec elle.
|
|
const half = BUCKETS / 2;
|
|
const isDocument = text.includes(' — ');
|
|
const hash = [...normalizeQuestion(text)].reduce(
|
|
(acc, char) => (acc * 31 + char.charCodeAt(0)) % half,
|
|
7
|
|
);
|
|
|
|
vector[VOCABULARY.length + (isDocument ? hash : half + hash)] = 1;
|
|
return vector;
|
|
}
|
|
|
|
function memoryCache(): CachePort & { store: Map<string, unknown> } {
|
|
const store = new Map<string, unknown>();
|
|
return {
|
|
store,
|
|
async get<T>(key: string): Promise<T | null> {
|
|
return (store.get(key) as T) ?? null;
|
|
},
|
|
async set<T>(key: string, value: T): Promise<void> {
|
|
store.set(key, value);
|
|
},
|
|
async delete(key: string) {
|
|
store.delete(key);
|
|
},
|
|
async deleteMany(keys: string[]) {
|
|
keys.forEach(key => store.delete(key));
|
|
},
|
|
async exists(key: string) {
|
|
return store.has(key);
|
|
},
|
|
async ttl() {
|
|
return -1;
|
|
},
|
|
async clear() {
|
|
store.clear();
|
|
},
|
|
async getStats() {
|
|
return { hits: 0, misses: 0, hitRate: 0, keyCount: store.size };
|
|
},
|
|
};
|
|
}
|
|
|
|
const config = new ConfigService({});
|
|
|
|
function embedder(): jest.Mocked<TradeEmbeddingPort> {
|
|
return {
|
|
isAvailable: jest.fn().mockReturnValue(true),
|
|
embed: jest.fn(async (texts: string[]) => texts.map(fakeVector)),
|
|
};
|
|
}
|
|
|
|
describe('WikiRetriever', () => {
|
|
it('ranks the wiki page that matches the question', async () => {
|
|
const retriever = new WikiRetriever(embedder(), memoryCache(), config);
|
|
|
|
const [best] = await retriever.search('Quels sont les régimes douaniers ?', 'fr');
|
|
|
|
expect(best.href).toBe('/dashboard/wiki/douanes');
|
|
expect(best.text).toContain('Mise en Libre Pratique');
|
|
expect(best.score).toBeGreaterThan(0);
|
|
});
|
|
|
|
it('vectorises the corpus once per process, however many searches', async () => {
|
|
const embeddings = embedder();
|
|
const retriever = new WikiRetriever(embeddings, memoryCache(), config);
|
|
|
|
await retriever.search('douane', 'fr');
|
|
await retriever.search('conteneur', 'fr');
|
|
await retriever.search('incoterms', 'fr');
|
|
|
|
// Un appel pour le corpus, puis un par question inedite.
|
|
const corpusCalls = embeddings.embed.mock.calls.filter(([texts]) => texts.length > 1);
|
|
expect(corpusCalls).toHaveLength(1);
|
|
});
|
|
|
|
it('reuses the cached index after a restart, without re-embedding', async () => {
|
|
const cache = memoryCache();
|
|
await new WikiRetriever(embedder(), cache, config).search('douane', 'fr');
|
|
|
|
const afterRestart = embedder();
|
|
await new WikiRetriever(afterRestart, cache, config).search('incoterms', 'fr');
|
|
|
|
// Seule la question inedite est vectorisee : le corpus vient du cache.
|
|
expect(afterRestart.embed).toHaveBeenCalledTimes(1);
|
|
expect(afterRestart.embed.mock.calls[0][0]).toEqual(['incoterms']);
|
|
});
|
|
|
|
it('does not re-embed a question already asked, whatever the wording noise', async () => {
|
|
const cache = memoryCache();
|
|
await new WikiRetriever(embedder(), cache, config).search('Quels documents ?', 'fr');
|
|
|
|
const second = embedder();
|
|
await new WikiRetriever(second, cache, config).search(' quels documents ', 'fr');
|
|
|
|
expect(second.embed).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('falls back to lexical search when no provider key is configured', async () => {
|
|
const embeddings = embedder();
|
|
embeddings.isAvailable.mockReturnValue(false);
|
|
|
|
const [best] = await new WikiRetriever(embeddings, memoryCache(), config).search(
|
|
'régimes douaniers dédouanées',
|
|
'fr'
|
|
);
|
|
|
|
expect(embeddings.embed).not.toHaveBeenCalled();
|
|
expect(best.href).toBe('/dashboard/wiki/douanes');
|
|
});
|
|
|
|
it('answers in the requested language and falls back to French', async () => {
|
|
const retriever = new WikiRetriever(embedder(), memoryCache(), config);
|
|
|
|
const [english] = await retriever.search('incoterms', 'en');
|
|
const [unknown] = await retriever.search('incoterms', 'de');
|
|
|
|
expect(english.id.startsWith('en:')).toBe(true);
|
|
expect(unknown.id.startsWith('fr:')).toBe(true);
|
|
});
|
|
|
|
it('returns nothing for a question the wiki does not cover', async () => {
|
|
// Sous le seuil, l'assistant citait des pages sans rapport sous une reponse
|
|
// produite par les outils : mieux vaut ne rien citer que citer a cote.
|
|
const retriever = new WikiRetriever(embedder(), memoryCache(), config);
|
|
|
|
// Aucun mot du vocabulaire metier : la similarite reste sous 0,45.
|
|
expect(await retriever.search('combien de reservations ai-je', 'fr')).toEqual([]);
|
|
});
|
|
|
|
it('keeps answering when the cache is unavailable', async () => {
|
|
const broken = memoryCache();
|
|
broken.get = jest.fn().mockRejectedValue(new Error('redis down'));
|
|
broken.set = jest.fn().mockRejectedValue(new Error('redis down'));
|
|
|
|
const results = await new WikiRetriever(embedder(), broken, config).search('douane', 'fr');
|
|
|
|
expect(results.length).toBeGreaterThan(0);
|
|
});
|
|
|
|
/* ------------------------------------------------------------------------ */
|
|
/* Complements ecrits par l'assistant */
|
|
/* ------------------------------------------------------------------------ */
|
|
|
|
describe('contributions', () => {
|
|
const page = WikiContribution.fromPersistence({
|
|
id: 'w1',
|
|
version: 1,
|
|
locale: 'fr',
|
|
topic: 'vgm',
|
|
title: 'VGM et pesée',
|
|
section: 'Méthodes',
|
|
// Les mots du vocabulaire de test portent tout le score.
|
|
body: 'vgm vgm vgm conteneur conteneurs',
|
|
status: WikiContributionStatus.PUBLISHED,
|
|
authorUserId: 'user',
|
|
authorOrganizationId: 'org',
|
|
createdAt: new Date(),
|
|
updatedAt: new Date(),
|
|
});
|
|
|
|
const repository = (pages: WikiContribution[]) => ({
|
|
findPublished: jest.fn().mockResolvedValue(pages),
|
|
findForReview: jest.fn().mockResolvedValue([]),
|
|
findById: jest.fn().mockResolvedValue(null),
|
|
findByTitle: jest.fn().mockResolvedValue(null),
|
|
save: jest.fn(),
|
|
revision: jest.fn().mockResolvedValue(`${pages.length}:r1`),
|
|
});
|
|
|
|
it('cites a contributed page alongside the published wiki', async () => {
|
|
const contributions = repository([page]);
|
|
const retriever = new WikiRetriever(embedder(), memoryCache(), config, contributions);
|
|
|
|
// La limite est ouverte : ce qui se verifie ici est que le complement
|
|
// concourt avec le wiki publie, pas qu'il le devance.
|
|
const results = await retriever.search('vgm vgm vgm', 'fr', 10);
|
|
|
|
expect(results.map(r => r.href)).toContain(page.href);
|
|
});
|
|
|
|
it('reuses the index while the revision holds, and rebuilds when it moves', async () => {
|
|
const contributions = repository([page]);
|
|
const retriever = new WikiRetriever(embedder(), memoryCache(), config, contributions);
|
|
|
|
await retriever.search('vgm', 'fr');
|
|
await retriever.search('vgm', 'fr');
|
|
expect(contributions.findPublished).toHaveBeenCalledTimes(1);
|
|
|
|
contributions.revision.mockResolvedValue('2:r2');
|
|
await retriever.search('vgm', 'fr');
|
|
expect(contributions.findPublished).toHaveBeenCalledTimes(2);
|
|
});
|
|
|
|
it('answers from the published wiki when the contributions are unreachable', async () => {
|
|
const contributions = repository([]);
|
|
contributions.revision.mockRejectedValue(new Error('db down'));
|
|
|
|
const results = await new WikiRetriever(
|
|
embedder(),
|
|
memoryCache(),
|
|
config,
|
|
contributions
|
|
).search('douane', 'fr');
|
|
|
|
expect(results.length).toBeGreaterThan(0);
|
|
});
|
|
});
|
|
});
|
|
|
|
describe('vector packing', () => {
|
|
it('survives a round trip through the cache', () => {
|
|
const vector = Float32Array.from([0.5, -0.25, 0.125]);
|
|
expect([...unpack(pack(vector))]).toEqual([0.5, -0.25, 0.125]);
|
|
});
|
|
});
|
|
|
|
describe('normalizeQuestion', () => {
|
|
it('collapses case, accents and punctuation so one wording is one vector', () => {
|
|
expect(normalizeQuestion(' Quels DOCUMENTS, pour la douane ? ')).toBe(
|
|
'quels documents pour la douane'
|
|
);
|
|
expect(normalizeQuestion('dédouanées')).toBe('dedouanees');
|
|
});
|
|
});
|