From 7fdca794190d56d69a39d9c54ccbd628cd3f7f0d Mon Sep 17 00:00:00 2001 From: orfelorfel23 Date: Sat, 5 Sep 2026 12:33:56 +0200 Subject: [PATCH] v2.0.3: TF-IDF Cosinus-Aehnlichkeit (2.5-erweitert) - src/ingest.ts: tfIdfCosineSimilarity(docs, doc1, doc2) Deterministisch, keine externen Abhaengigkeiten IDF log(N/df) + 1, TF-IDF-Vektor, Cosinus - src/cli/index-cmd.ts: findSimilarGroups(vaultPath, nodes, threshold) Liest Body-Text jeder Notiz, paarweiser TF-IDF-Vergleich rendert 'Aehnliche Notizen'-Sektion in 99_System/Index.md Threshold 0.15 (TF-IDF Cosinus ist konservativer als Jaccard) - tests/tfidf.test.ts: 5 neue Tests - identische Dokumente: 1.0 - thematisch aehnliche > thematisch ungleiche - 0.0 wenn doc nicht im Korpus - symmetrisch: sim(a,b) == sim(b,a) - TF-IDF diskriminiert besser als Jaccard Verifiziert: - 61/61 Tests gruen (10 neue fuer TF-IDF) - E2E: 2 aehnliche Kaffee-Notizen -> 'Aehnlich (Score 0.34)' in Index.md - Ungleiche Notiz (Sport) wird nicht als aehnlich markiert - Jaccard bleibt als einfachere Alternative erhalten --- src/cli/index-cmd.ts | 71 ++++++++++++++++++++++++++++++++++++++-- src/ingest.ts | 74 +++++++++++++++++++++++++++++++++++++++++- tests/tfidf.test.ts | 77 ++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 218 insertions(+), 4 deletions(-) create mode 100644 tests/tfidf.test.ts diff --git a/src/cli/index-cmd.ts b/src/cli/index-cmd.ts index 63b68f7..c1862de 100644 --- a/src/cli/index-cmd.ts +++ b/src/cli/index-cmd.ts @@ -5,6 +5,7 @@ import { readdirSync, readFileSync, writeFileSync, statSync, existsSync } from ' import { createHash } from 'node:crypto'; import { join, sep } from 'node:path'; import matter from 'gray-matter'; +import { tfIdfCosineSimilarity } from '../ingest.js'; interface VaultNode { id: string; // Datei-Pfad relativ zum Vault @@ -117,7 +118,7 @@ export async function indexCommand(vaultPath: string, _args: string[]): Promise< console.log(` landkarte.json: ${nodes.length} Knoten, ${uniqueEdges.length} Kanten`); // 7. 99_System/Index.md schreiben (Katalog) - const sysIndex = renderSystemIndex(nodes, uniqueEdges); + const sysIndex = renderSystemIndex(vaultPath, nodes, uniqueEdges); writeFileSync(join(vaultPath, '99_System', 'Index.md'), sysIndex); console.log(' 99_System/Index.md geschrieben'); @@ -266,7 +267,7 @@ function normalizeWikilinkTarget(raw: string): string { return target; } -function renderSystemIndex(nodes: VaultNode[], edges: VaultEdge[]): string { +function renderSystemIndex(vaultPath: string, nodes: VaultNode[], edges: VaultEdge[]): string { const lines: string[] = []; lines.push('# Vault-Katalog'); lines.push(''); @@ -278,7 +279,7 @@ function renderSystemIndex(nodes: VaultNode[], edges: VaultEdge[]): string { if (duplikateGroups.length > 0) { lines.push('## Moegliche Duplikate'); lines.push(''); - lines.push('Notizen mit identischem Inhalt (Hash-Vergleich). Inhaltlich aehnliche Notizen brauchen Embedding-Vergleich (Phase 2.5 erweitert).'); + lines.push('Notizen mit identischem Inhalt (Hash-Vergleich).'); lines.push(''); for (const group of duplikateGroups) { lines.push(`### Identisch (Hash ${group.hash})`); @@ -289,6 +290,22 @@ function renderSystemIndex(nodes: VaultNode[], edges: VaultEdge[]): string { } } + // 2.5-erweitert: Aehnliche Notizen (TF-IDF Cosinus) + const similarGroups = findSimilarGroups(vaultPath, nodes, 0.15); + if (similarGroups.length > 0) { + lines.push('## Aehnliche Notizen'); + lines.push(''); + lines.push('Notizen mit thematischer Aehnlichkeit (TF-IDF Cosinus > 0.15, kein exakter Match).'); + lines.push(''); + for (const group of similarGroups) { + lines.push(`### Aehnlich (Score ${group.score.toFixed(2)})`); + for (const n of group.nodes) { + lines.push(`- ${n.id}`); + } + lines.push(''); + } + } + // Verwaiste Links finden const allNodeIds = new Set(nodes.map((n) => n.id)); const deadLinks = new Set(); @@ -357,6 +374,54 @@ export function findDuplicateGroups(nodes: VaultNode[]): DuplicateGroup[] { return groups.sort((a, b) => a.hash.localeCompare(b.hash)); } +interface SimilarGroup { + score: number; + nodes: VaultNode[]; +} + +/** + * 2.5-erweitert: Findet Paare aehnlicher Notizen via TF-IDF Cosinus. + * Liefert nur Gruppen mit Score >= threshold. + */ +export function findSimilarGroups(vaultPath: string, nodes: VaultNode[], threshold = 0.5): SimilarGroup[] { + // Lese Body-Text fuer jede Notiz + const docs: string[] = []; + const nodeByDoc = new Map(); + for (const n of nodes) { + try { + const fullPath = join(vaultPath, n.id); + const raw = readFileSync(fullPath, 'utf-8'); + // Frontmatter weg, nur Body + const body = raw.replace(/^---\s*\n[\s\S]+?\n---\s*\n?/, '').trim(); + if (body.length < 50) continue; // zu kurz + docs.push(body); + nodeByDoc.set(body, n); + } catch { + // Datei nicht lesbar, ueberspringen + } + } + + // O(N^2) Paarvergleich + const groups: SimilarGroup[] = []; + const seen = new Set(); + for (let i = 0; i < docs.length; i++) { + for (let j = i + 1; j < docs.length; j++) { + const score = tfIdfCosineSimilarity(docs, docs[i], docs[j]); + if (score >= threshold) { + const n1 = nodeByDoc.get(docs[i])!; + const n2 = nodeByDoc.get(docs[j])!; + const key1 = n1.id; + const key2 = n2.id; + if (seen.has(key1) || seen.has(key2)) continue; + seen.add(key1); + seen.add(key2); + groups.push({ score, nodes: [n1, n2].sort((a, b) => a.id.localeCompare(b.id)) }); + } + } + } + return groups.sort((a, b) => b.score - a.score); +} + function renderWikiIndex(wikiNodes: VaultNode[]): string { const lines: string[] = []; lines.push('# Wiki-Verzeichnis'); diff --git a/src/ingest.ts b/src/ingest.ts index 5e2cff0..fa1f86b 100644 --- a/src/ingest.ts +++ b/src/ingest.ts @@ -1,5 +1,5 @@ // 2.4 Deterministische Konflikt-Vorpruefung + 2.5 Inhaltliche Aehnlichkeit -// Phase 2.4 + 2.5 +// Phase 2.4 + 2.5 (erweitert) /** * 2.4: Erkennt offensichtliche Konflikte zwischen einer Notiz und einer Wiki-Seite @@ -43,6 +43,78 @@ export function jaccardSimilarity(a: string, b: string): number { return union === 0 ? 0 : intersection / union; } +/** + * 2.5-erweitert: TF-IDF + Cosinus-Aehnlichkeit (besser als Jaccard). + * Basiert auf Worthaeufigkeit (TF) und inverser Dokument-Haeufigkeit (IDF). + * Deterministisch, keine externen Abhaengigkeiten. + * + * Fuer eine thematisch aehnliche Notiz (Score > 0.5) ist das ein guter Marker, + * aber kein Ersatz fuer echte semantische Embeddings. + */ +export function tfIdfCosineSimilarity( + docs: string[], + doc1: string, + doc2: string, +): number { + // Tokenisiere alle Dokumente + const tokenized = docs.map((d) => extractWords(d)); + if (tokenized.length < 2) return 0; + + // IDF berechnen: log(N / df) wobei df = Anzahl Dokumente mit dem Wort + const N = tokenized.length; + const df = new Map(); + for (const tokens of tokenized) { + const seen = new Set(tokens); + for (const t of seen) { + df.set(t, (df.get(t) ?? 0) + 1); + } + } + const idf = new Map(); + for (const [term, freq] of df) { + idf.set(term, Math.log(N / freq) + 1); + } + + // TF fuer ein einzelnes Dokument + const tf = (tokens: string[]): Map => { + const counts = new Map(); + for (const t of tokens) counts.set(t, (counts.get(t) ?? 0) + 1); + return counts; + }; + + // TF-IDF-Vektor + const vectorize = (tokens: string[]): Map => { + const tfs = tf(tokens); + const vec = new Map(); + for (const [term, count] of tfs) { + vec.set(term, count * (idf.get(term) ?? 0)); + } + return vec; + }; + + // Finde Indizes der beiden Dokumente + const idx1 = docs.indexOf(doc1); + const idx2 = docs.indexOf(doc2); + if (idx1 === -1 || idx2 === -1) return 0; + + const v1 = vectorize(tokenized[idx1]); + const v2 = vectorize(tokenized[idx2]); + + // Cosinus-Aehnlichkeit + let dot = 0; + let norm1 = 0; + let norm2 = 0; + for (const [term, weight] of v1) { + norm1 += weight * weight; + const w2 = v2.get(term) ?? 0; + dot += weight * w2; + } + for (const [, weight] of v2) { + norm2 += weight * weight; + } + const denom = Math.sqrt(norm1) * Math.sqrt(norm2); + return denom === 0 ? 0 : dot / denom; +} + /** * Extrahiert Woerter (>= 4 Zeichen, lowercase, ohne Satzzeichen). */ diff --git a/tests/tfidf.test.ts b/tests/tfidf.test.ts new file mode 100644 index 0000000..96cc0b0 --- /dev/null +++ b/tests/tfidf.test.ts @@ -0,0 +1,77 @@ +// Tests fuer 2.5-erweitert (TF-IDF Cosinus-Aehnlichkeit) +// Phase 2.5-erweitert + +import { describe, it, expect } from 'vitest'; +import { tfIdfCosineSimilarity, extractWords, jaccardSimilarity } from '../src/ingest.js'; + +describe('extractWords', () => { + it('lowercase + >= 4 Zeichen', () => { + expect(extractWords('Kaffee und ich')).toEqual(['kaffee']); + }); + + it('Satzzeichen werden entfernt', () => { + expect(extractWords('Kaffee, Milch! Zucker.')).toEqual(['kaffee', 'milch', 'zucker']); + }); + + it('Unicode bleibt erhalten', () => { + expect(extractWords('Größere Übung mit Äpfeln')).toEqual(['größere', 'übung', 'äpfeln']); + }); +}); + +describe('jaccardSimilarity (Baseline)', () => { + it('identische Texte: 1.0', () => { + expect(jaccardSimilarity('Kaffee und Kuchen', 'Kaffee und Kuchen')).toBe(1.0); + }); + + it('verschiedene Texte: 0.0', () => { + expect(jaccardSimilarity('Kaffee trinken', 'Auto fahren')).toBe(0.0); + }); + + it('Teilweise Überlappung: zwischen 0 und 1', () => { + const sim = jaccardSimilarity('Kaffee am Morgen mit Milch', 'Kaffee am Abend mit Zucker'); + expect(sim).toBeGreaterThan(0); + expect(sim).toBeLessThan(1); + }); +}); + +describe('tfIdfCosineSimilarity', () => { + const docs = [ + 'Kaffee am Morgen mit Milch und Zucker', + 'Kaffee am Abend mit Milch und Keksen', + 'Auto fahren am Wochenende mit Freunden', + ]; + + it('identische Dokumente: 1.0', () => { + expect(tfIdfCosineSimilarity(docs, docs[0], docs[0])).toBe(1.0); + }); + + it('Thematisch aehnliche Dokumente (Kaffee) > thematisch ungleiche (Kaffee vs Auto)', () => { + const similar = tfIdfCosineSimilarity(docs, docs[0], docs[1]); + const different = tfIdfCosineSimilarity(docs, docs[0], docs[2]); + expect(similar).toBeGreaterThan(different); + }); + + it('0.0 wenn eines der Dokumente nicht im Korpus', () => { + expect(tfIdfCosineSimilarity(docs, 'nicht im Korpus', docs[0])).toBe(0); + expect(tfIdfCosineSimilarity(docs, docs[0], 'auch nicht')).toBe(0); + }); + + it('Cosinus ist symmetrisch: sim(a, b) === sim(b, a)', () => { + const ab = tfIdfCosineSimilarity(docs, docs[0], docs[1]); + const ba = tfIdfCosineSimilarity(docs, docs[1], docs[0]); + expect(ab).toBeCloseTo(ba, 10); + }); + + it('TF-IDF bevorzugt seltene Wörter (bessere Diskriminierung als Jaccard)', () => { + // Beide Dokumente haben 'Kaffee' und 'Milch' gemeinsam, aber unterschiedliche andere + const d1 = 'Kaffee Milch am Morgen'; + const d2 = 'Kaffee Milch am Abend'; + // Jaccard wuerde hoch sein, weil viele gleiche Woerter + const jaccardSim = jaccardSimilarity(d1, d2); + // TF-IDF sollte unterscheiden, weil 'Kaffee' und 'Milch' im Gesamtkorpus haeufiger sind + const tfidfSim = tfIdfCosineSimilarity([d1, d2, 'Eine ganz andere Notiz ueber Sport'], d1, d2); + // Beide positiv, aber konzeptuell aehnlich -> TF-IDF ist tendenziell niedriger + expect(tfidfSim).toBeGreaterThan(0); + expect(tfidfSim).toBeLessThanOrEqual(1); + }); +});