- contentHash-Feld im VaultNode (SHA-256, erste 16 Zeichen vom Body ohne Frontmatter) - findDuplicateGroups(): gruppiert Notizen nach Hash, nur Gruppen mit >= 2 Nodes - 99_System/Index.md: neue Sektion 'Moegliche Duplikate' (oberste Stelle, vor den Typ-Gruppen) - 6 Unit-Tests: leere Liste, einzelne Notiz, 2/3 gleicher Hash, mehrere Gruppen, Sortierung Hinweis: Embedding-basierte inhaltliche Aehnlichkeit (z.B. cosine > 0.9) braucht QMD und ist Phase 2.5-erweitert. Aktuell nur exakte Hash-Matches.
68 lines
1.9 KiB
TypeScript
68 lines
1.9 KiB
TypeScript
// Tests fuer 2.5 Duplikat-Erkennung
|
|
// Phase 2.5
|
|
|
|
import { describe, it, expect } from 'vitest';
|
|
import { findDuplicateGroups, type VaultNode } from '../src/cli/index-cmd.js';
|
|
|
|
function node(id: string, hash: string): VaultNode {
|
|
return {
|
|
id,
|
|
title: id,
|
|
type: 'inbox',
|
|
cluster: null,
|
|
status: null,
|
|
created: null,
|
|
aktualisiert: null,
|
|
tags: [],
|
|
size: 0,
|
|
veraltet: false,
|
|
contentHash: hash,
|
|
};
|
|
}
|
|
|
|
describe('findDuplicateGroups', () => {
|
|
it('leere Liste liefert keine Gruppen', () => {
|
|
expect(findDuplicateGroups([])).toEqual([]);
|
|
});
|
|
|
|
it('einzelne Notizen liefern keine Gruppen', () => {
|
|
expect(findDuplicateGroups([node('a', 'h1'), node('b', 'h2')])).toEqual([]);
|
|
});
|
|
|
|
it('zwei Notizen mit gleichem Hash ergeben eine Gruppe', () => {
|
|
const groups = findDuplicateGroups([node('a', 'h1'), node('b', 'h1')]);
|
|
expect(groups.length).toBe(1);
|
|
expect(groups[0].hash).toBe('h1');
|
|
expect(groups[0].nodes.map((n) => n.id)).toEqual(['a', 'b']);
|
|
});
|
|
|
|
it('drei Notizen mit gleichem Hash in einer Gruppe', () => {
|
|
const groups = findDuplicateGroups([node('a', 'h1'), node('b', 'h1'), node('c', 'h1')]);
|
|
expect(groups.length).toBe(1);
|
|
expect(groups[0].nodes.length).toBe(3);
|
|
});
|
|
|
|
it('mehrere Hash-Gruppen werden separat geliefert', () => {
|
|
const groups = findDuplicateGroups([
|
|
node('a', 'h1'),
|
|
node('b', 'h1'),
|
|
node('c', 'h2'),
|
|
node('d', 'h2'),
|
|
]);
|
|
expect(groups.length).toBe(2);
|
|
});
|
|
|
|
it('sortiert nach Hash dann nach ID', () => {
|
|
const groups = findDuplicateGroups([
|
|
node('z.md', 'h1'),
|
|
node('a.md', 'h1'),
|
|
node('b.md', 'h2'),
|
|
node('y.md', 'h2'),
|
|
]);
|
|
expect(groups.length).toBe(2);
|
|
// h1 < h2, dann alphabetisch nach ID innerhalb der Gruppe
|
|
expect(groups[0].nodes.map((n) => n.id)).toEqual(['a.md', 'z.md']);
|
|
expect(groups[1].nodes.map((n) => n.id)).toEqual(['b.md', 'y.md']);
|
|
});
|
|
});
|