BREAKING CHANGE: regex fallback is now opt-in (patterns no longer defaults
to presets.swiss). With no fallback an LLM failure throws AnonymizationError
(fail-closed) and the pre-filter is bypassed; at least one of llm/patterns is
required. AnonymizationResult gains a required `legend`; anonymizeChunks seed
is now { mapping, legend? } and returns legend.
- prompt: model may coin new UPPERCASE abbreviations and returns a 'legende'
explaining every abbreviation used (French); backfilled by DEFAULT_LEGEND
- PatternDef.meaning surfaces in the legend; swiss/generic presets get meanings
- AnonymizationError (exported) wraps the cause on fail-closed
- README: drop the chatbot provenance line; add 'How it works' + nLPD sections
- 34 tests / 99% coverage; bump to 0.2.0
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
182 lines
8.4 KiB
TypeScript
182 lines
8.4 KiB
TypeScript
import { describe, it, expect, vi } from 'vitest';
|
|
import { Anonymizer, AnonymizationError, presets, type LlmProvider } from '../src/index.js';
|
|
|
|
/** An LLM provider that is configured but always fails → forces the regex fallback. */
|
|
const failingLlm = (overrides: Partial<LlmProvider> = {}): LlmProvider => ({
|
|
isConfigured: () => true,
|
|
anonymize: vi.fn().mockRejectedValue(new Error('LLM_NOT_CONFIGURED')),
|
|
anonymizeBatch: vi.fn().mockRejectedValue(new Error('LLM_NOT_CONFIGURED')),
|
|
...overrides,
|
|
});
|
|
|
|
describe('Anonymizer (Swiss preset)', () => {
|
|
const svc = new Anonymizer({ llm: failingLlm(), patterns: presets.swiss });
|
|
|
|
it('skips anonymization (no LLM call) when there is no PII', async () => {
|
|
const llm = failingLlm();
|
|
const s = new Anonymizer({ llm, patterns: presets.swiss });
|
|
const r = await s.anonymize('Explique la différence entre INNER JOIN et LEFT JOIN');
|
|
expect(r.mapping).toEqual({});
|
|
expect(r.anon).toContain('INNER JOIN');
|
|
expect(llm.anonymize).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('falls back to regex for structured PII (email/phone/AVS)', async () => {
|
|
const r = await svc.anonymize('Contact: jean@exemple.ch, +41 79 123 45 67, AVS 756.1234.5678.90');
|
|
expect(r.anon).not.toContain('jean@exemple.ch');
|
|
expect(r.anon).not.toContain('756.1234.5678.90');
|
|
expect(Object.values(r.mapping)).toContain('jean@exemple.ch');
|
|
expect(svc.deanonymize(r.anon, r.mapping)).toContain('jean@exemple.ch');
|
|
});
|
|
|
|
it('deanonymizes a placeholder split across stream tokens (critical)', () => {
|
|
const mapping = { '[PER_1.NOM:M]': 'Alain JACCARD' };
|
|
const d = svc.makeStreamDeanonymizer(mapping);
|
|
let out = '';
|
|
out += d.push('Bonjour [PER_');
|
|
out += d.push('1.NOM');
|
|
out += d.push(':M], ravi');
|
|
out += d.flush();
|
|
expect(out).toBe('Bonjour Alain JACCARD, ravi');
|
|
expect(out).not.toContain('[PER_');
|
|
});
|
|
|
|
it('streams plain text through unchanged', () => {
|
|
const d = svc.makeStreamDeanonymizer({});
|
|
expect(d.push('Un INNER JOIN ') + d.push('retourne...') + d.flush()).toBe('Un INNER JOIN retourne...');
|
|
});
|
|
|
|
describe('anonymizeChunks', () => {
|
|
it('PII-free chunks → no LLM call, returned as-is', async () => {
|
|
const llm = failingLlm();
|
|
const s = new Anonymizer({ llm, patterns: presets.swiss });
|
|
const r = await s.anonymizeChunks(['Un INNER JOIN combine deux tables.'], { mapping: {} });
|
|
expect(llm.anonymizeBatch).not.toHaveBeenCalled();
|
|
expect(r.anon[0]).toContain('INNER JOIN');
|
|
expect(r.mapping).toEqual({});
|
|
});
|
|
|
|
it('reuses the question placeholder for the same person (deterministic, no LLM)', async () => {
|
|
const llm = failingLlm();
|
|
const s = new Anonymizer({ llm, patterns: presets.swiss });
|
|
const seed = { '[PER_1.NOM:M]': 'Alain JACCARD' };
|
|
const r = await s.anonymizeChunks(['Le dossier de Alain JACCARD est complet.'], { mapping: seed });
|
|
expect(llm.anonymizeBatch).not.toHaveBeenCalled();
|
|
expect(r.anon[0]).toContain('[PER_1.NOM:M]');
|
|
expect(r.anon[0]).not.toContain('Alain JACCARD');
|
|
expect(r.mapping).toEqual(seed);
|
|
});
|
|
|
|
it('renumbers a NEW person that collides with the question placeholder', async () => {
|
|
const anonymizeBatch = vi.fn().mockResolvedValue({
|
|
segments: ['[PER_1.NOM:M] a signé.'],
|
|
mapping: { '[PER_1.NOM:M]': 'Bob Martin' },
|
|
legend: { PER: 'Personne', NOM: 'Nom de famille', M: 'Masculin' },
|
|
});
|
|
const s = new Anonymizer({ llm: failingLlm({ anonymizeBatch }), patterns: presets.swiss });
|
|
const seed = { '[PER_1.NOM:M]': 'Alain JACCARD' };
|
|
const r = await s.anonymizeChunks(['Bob Martin a signé.'], { mapping: seed });
|
|
expect(anonymizeBatch).toHaveBeenCalledTimes(1);
|
|
expect(r.anon[0]).toContain('[PER_2.NOM:M]');
|
|
expect(r.mapping['[PER_1.NOM:M]']).toBe('Alain JACCARD');
|
|
expect(r.mapping['[PER_2.NOM:M]']).toBe('Bob Martin');
|
|
// legend covers the abbreviations used in the anonymized chunk
|
|
expect(r.legend).toMatchObject({ PER: 'Personne', NOM: 'Nom de famille', M: 'Masculin' });
|
|
});
|
|
|
|
it('falls back to regex (structured ids) when the LLM is unavailable', async () => {
|
|
const s = new Anonymizer({ llm: failingLlm(), patterns: presets.swiss });
|
|
const r = await s.anonymizeChunks(['Contact : jean@exemple.ch'], { mapping: {} });
|
|
expect(r.anon[0]).not.toContain('jean@exemple.ch');
|
|
expect(Object.values(r.mapping)).toContain('jean@exemple.ch');
|
|
});
|
|
});
|
|
|
|
describe('regex-only mode (no LLM provider)', () => {
|
|
const s = new Anonymizer({ patterns: presets.swiss });
|
|
|
|
it('still anonymizes structured PII without any provider', async () => {
|
|
const r = await s.anonymize('Écris à jean@exemple.ch');
|
|
expect(r.anon).not.toContain('jean@exemple.ch');
|
|
expect(Object.values(r.mapping)).toContain('jean@exemple.ch');
|
|
});
|
|
|
|
it('leaves a bare proper name untouched (no LLM to catch it)', async () => {
|
|
const r = await s.anonymize('Alain Jaccard a réussi');
|
|
// The name-hint flags it, but with no LLM the regex fallback finds no structured id.
|
|
expect(r.anon).toContain('Alain Jaccard');
|
|
expect(r.mapping).toEqual({});
|
|
});
|
|
});
|
|
|
|
describe('configurable presets', () => {
|
|
it('generic preset anonymizes an IPv4 address', async () => {
|
|
const s = new Anonymizer({ patterns: presets.generic });
|
|
const r = await s.anonymize('Serveur 192.168.1.42 indisponible');
|
|
expect(r.anon).not.toContain('192.168.1.42');
|
|
expect(Object.values(r.mapping)).toContain('192.168.1.42');
|
|
});
|
|
|
|
it('accepts a fully custom pattern set', async () => {
|
|
const s = new Anonymizer({ patterns: [{ tag: 'TICKET', re: /\bJIRA-\d+\b/g }] });
|
|
const r = await s.anonymize('Voir JIRA-123');
|
|
expect(r.anon).toContain('[TICKET_1]');
|
|
expect(r.mapping['[TICKET_1]']).toBe('JIRA-123');
|
|
});
|
|
});
|
|
|
|
describe('legend', () => {
|
|
it('the regex fallback produces a French legend from tag meanings', async () => {
|
|
const s = new Anonymizer({ patterns: presets.swiss });
|
|
const r = await s.anonymize('Écris à jean@exemple.ch');
|
|
expect(r.legend).toEqual({ EMAIL: 'Adresse e-mail' });
|
|
});
|
|
|
|
it('backfills a missing legend entry from the built-in default (LLM omitted it)', async () => {
|
|
const anonymize = vi.fn().mockResolvedValue({
|
|
anon: 'Dossier de [PER_1.NOM:M]',
|
|
mapping: { '[PER_1.NOM:M]': 'Alain Jaccard' },
|
|
legend: {}, // model returned no legend
|
|
});
|
|
const s = new Anonymizer({ llm: failingLlm({ anonymize }), patterns: presets.swiss });
|
|
const r = await s.anonymize('Dossier de Alain Jaccard');
|
|
expect(r.legend).toEqual({ PER: 'Personne', NOM: 'Nom de famille', M: 'Masculin' });
|
|
});
|
|
|
|
it('a custom tag with a meaning surfaces it in the legend', async () => {
|
|
const s = new Anonymizer({
|
|
patterns: [{ tag: 'TICKET', re: /\bJIRA-\d+\b/g, meaning: 'Ticket de suivi' }],
|
|
});
|
|
const r = await s.anonymize('Voir JIRA-123');
|
|
expect(r.legend).toEqual({ TICKET: 'Ticket de suivi' });
|
|
});
|
|
});
|
|
|
|
describe('fail-closed (optional fallback)', () => {
|
|
it('throws if neither an LLM nor patterns are provided', () => {
|
|
expect(() => new Anonymizer({})).toThrowError(AnonymizationError);
|
|
});
|
|
|
|
it('LLM-only: a provider failure throws AnonymizationError with the cause', async () => {
|
|
const cause = new Error('LLM_HTTP_500');
|
|
const s = new Anonymizer({ llm: failingLlm({ anonymize: vi.fn().mockRejectedValue(cause) }) });
|
|
await expect(s.anonymize('Contact: jean@exemple.ch')).rejects.toBeInstanceOf(AnonymizationError);
|
|
await expect(s.anonymize('Contact: jean@exemple.ch')).rejects.toMatchObject({ cause });
|
|
});
|
|
|
|
it('LLM-only: bypasses the pre-filter so PII-free text still hits the provider', async () => {
|
|
const anonymize = vi.fn().mockResolvedValue({ anon: 'INNER JOIN', mapping: {}, legend: {} });
|
|
const s = new Anonymizer({ llm: failingLlm({ anonymize }) });
|
|
await s.anonymize('Explique INNER JOIN'); // no structured PII, no name hint
|
|
expect(anonymize).toHaveBeenCalledTimes(1);
|
|
});
|
|
|
|
it('LLM-only: anonymizeChunks rejects with AnonymizationError on provider failure', async () => {
|
|
const s = new Anonymizer({ llm: failingLlm() }); // anonymizeBatch rejects
|
|
await expect(s.anonymizeChunks(['Bob Martin a signé.'], { mapping: {} })).rejects.toBeInstanceOf(
|
|
AnonymizationError,
|
|
);
|
|
});
|
|
});
|
|
});
|