Recently Written · git

ideamine

An idea inbox for Claude Code: /idea saves ideas at zero tokens; Claude triages them and routes each to the cheapest model that can build it.

git clone https://github.com/equwal/ideamine

Log | Files | Refs


tests/embed.test.js (7910 bytes)

1 import assert from 'node:assert/strict';
2 import fs from 'node:fs';
3 import os from 'node:os';
4 import path from 'node:path';
5 import { afterEach, beforeEach, test } from 'node:test';
6 import fc from 'fast-check';
7 import * as embed from '../src/embed.js';
8 import * as store from '../src/store.js';
9 import { startFakeServer } from './fixtures/fake-server.js';
10 
11 let server;
12 
13 beforeEach(async () => {
14   process.env.IDEAMINE_HOME = fs.mkdtempSync(path.join(os.tmpdir(), 'ideamine-embed-'));
15   server = await startFakeServer();
16   process.env.IDEAMINE_EMBED_URL = `${server.url}/v1`;
17   delete process.env.IDEAMINE_EMBED_MODEL;
18 });
19 
20 afterEach(() => server.close());
21 
22 const unitVector = (dim) =>
23   fc
24     .array(fc.float({ min: -1, max: 1, noNaN: true }), { minLength: dim, maxLength: dim })
25     .filter((v) => v.some((x) => Math.abs(x) > 1e-3))
26     .map((v) => embed.unit(v));
27 
28 test('property: a vector survives encode and decode unchanged', () => {
29   fc.assert(
30     fc.property(fc.float32Array(), (vec) => {
31       const back = embed.decodeVec(embed.encodeVec(vec));
32       assert.equal(back.length, vec.length);
33       for (let i = 0; i < vec.length; i++) assert.ok(Object.is(back[i], vec[i]), `index ${i}: ${back[i]} != ${vec[i]}`);
34     }),
35   );
36 });
37 
38 test('property: cosine of unit vectors is symmetric, at most 1, and 1 for the same vector', () => {
39   fc.assert(
40     fc.property(unitVector(8), unitVector(8), (a, b) => {
41       assert.ok(Math.abs(embed.cosine(a, b) - embed.cosine(b, a)) < 1e-6);
42       assert.ok(Math.abs(embed.cosine(a, b)) <= 1 + 1e-5);
43       assert.ok(Math.abs(embed.cosine(a, a) - 1) < 1e-5);
44     }),
45   );
46 });
47 
48 test('property: groups are disjoint, follow the average-linkage rule, and do not depend on the input order', () => {
49   const cases = fc
50     .uniqueArray(fc.integer({ min: 1, max: 500 }), { minLength: 0, maxLength: 12 })
51     .chain((ids) => fc.tuple(fc.constant(ids), fc.array(unitVector(4), { minLength: ids.length, maxLength: ids.length }), fc.double({ min: -1, max: 1, noNaN: true })));
52   fc.assert(
53     fc.property(cases, ([ids, vecs, threshold]) => {
54       const vectors = new Map(ids.map((id, i) => [id, vecs[i]]));
55       const groups = embed.groupIds(ids, vectors, threshold);
56       const seen = groups.flat();
57       assert.equal(new Set(seen).size, seen.length);
58       assert.ok(seen.every((id) => ids.includes(id)));
59       assert.ok(groups.every((g) => g.length >= 2));
60       assert.deepEqual(embed.groupIds([...ids].reverse(), vectors, threshold), groups);
61       // Average linkage: the mean similarity inside a group is at least the threshold, and the mean
62       // similarity across any two clusters that stayed apart (single ideas too) is below it.
63       const mean = (a, b) => {
64         let sum = 0;
65         for (const x of a) for (const y of b) sum += embed.cosine(vectors.get(x), vectors.get(y));
66         return sum / (a.length * b.length);
67       };
68       for (const g of groups) {
69         let sum = 0;
70         let pairs = 0;
71         for (let i = 0; i < g.length; i++) for (let j = i + 1; j < g.length; j++, pairs++) sum += embed.cosine(vectors.get(g[i]), vectors.get(g[j]));
72         assert.ok(sum / pairs >= threshold - 1e-6, `group ${g} has mean ${sum / pairs} < ${threshold}`);
73       }
74       const clusters = [...groups, ...ids.filter((id) => !seen.includes(id)).map((id) => [id])];
75       for (let i = 0; i < clusters.length; i++) {
76         for (let j = i + 1; j < clusters.length; j++) assert.ok(mean(clusters[i], clusters[j]) < threshold + 1e-6);
77       }
78     }),
79   );
80 });
81 
82 test('property: a higher threshold only splits groups, it never joins ideas from two groups', () => {
83   const cases = fc
84     .uniqueArray(fc.integer({ min: 1, max: 500 }), { minLength: 2, maxLength: 10 })
85     .chain((ids) => fc.tuple(fc.constant(ids), fc.array(unitVector(4), { minLength: ids.length, maxLength: ids.length }), fc.double({ min: -1, max: 1, noNaN: true }), fc.double({ min: 0, max: 1, noNaN: true })));
86   fc.assert(
87     fc.property(cases, ([ids, vecs, low, step]) => {
88       const vectors = new Map(ids.map((id, i) => [id, vecs[i]]));
89       const coarse = embed.groupIds(ids, vectors, low);
90       for (const g of embed.groupIds(ids, vectors, low + step)) {
91         assert.ok(coarse.some((c) => g.every((id) => c.includes(id))), `${g} is not inside one group of ${JSON.stringify(coarse)}`);
92       }
93     }),
94   );
95 });
96 
97 test('nomic prefixes: documents and queries get their own task prefix; other models get raw text', async () => {
98   store.addIdeas(['read subtitles aloud']);
99   await embed.find(store.load(), 'subtitles');
100   assert.deepEqual(server.inputs, ['search_document: read subtitles aloud', 'search_query: subtitles']);
101   process.env.IDEAMINE_EMBED_MODEL = 'mxbai-embed-large';
102   await embed.find(store.load(), 'subtitles');
103   assert.deepEqual(server.inputs.slice(2), ['read subtitles aloud', 'subtitles']);
104 });
105 
106 test('the cache embeds each idea once, again after a change, and drops deleted ideas', async () => {
107   store.addIdeas(['sync subtitles with audiobooks', 'dark mode for the popup']);
108   await embed.vectorsFor(store.load().ideas);
109   assert.equal(server.inputs.length, 2);
110   await embed.vectorsFor(store.load().ideas);
111   assert.equal(server.inputs.length, 2); // from the cache
112   store.updateIdea(2, { text: 'dark mode for the settings page' });
113   await embed.vectorsFor(store.load().ideas);
114   assert.deepEqual(server.inputs.slice(2), ['search_document: dark mode for the settings page']);
115   store.removeIdeas([1]);
116   store.addIdeas(['a third idea']);
117   await embed.vectorsFor(store.load().ideas);
118   const cached = JSON.parse(fs.readFileSync(path.join(store.home(), 'vectors.json'), 'utf8'));
119   assert.deepEqual(Object.keys(cached.items).sort(), ['2', '3']);
120 });
121 
122 test('a text that is too long is halved until the server takes it', async () => {
123   server.options.maxChars = 400;
124   store.addIdeas([`${'subtitle '.repeat(100)}end`, 'short idea']);
125   const vectors = await embed.vectorsFor(store.load().ideas);
126   assert.equal(vectors.size, 2);
127   const tries = server.inputs.filter((x) => x.startsWith('search_document: subtitle')).map((x) => x.length);
128   assert.deepEqual(tries, [920, 920, 468, 242]); // the batch, then one at a time, halved twice
129 });
130 
131 test('find ranks by meaning and keeps results above the threshold', async () => {
132   store.addIdeas(['subtitles for audiobooks in the reader', 'a tor exit relay on the server', 'subtitles in the video player']);
133   const found = await embed.find(store.load(), 'subtitles audiobooks reader', { threshold: 0.3 });
134   assert.equal(found.mode, 'meaning');
135   assert.deepEqual(found.results.map((r) => r.idea.id), [1, 3]);
136   assert.ok(found.results[0].score > found.results[1].score);
137 });
138 
139 test('find falls back to word search and says why when the server does not answer', async () => {
140   store.addIdeas(['subtitles for audiobooks', 'dark mode']);
141   server.options.down = true;
142   const found = await embed.find(store.load(), 'audiobooks subtitles');
143   assert.equal(found.mode, 'words');
144   assert.match(found.note, /503/);
145   assert.deepEqual(found.results.map((r) => r.idea.id), [1]);
146   const closed = await startFakeServer();
147   await closed.close(); // now nothing listens on its port
148   process.env.IDEAMINE_EMBED_URL = `${closed.url}/v1`;
149   const offline = await embed.find(store.load(), 'dark', { timeoutMs: 5000 });
150   assert.equal(offline.mode, 'words');
151   assert.match(offline.note, /cannot reach http:\/\/127\.0\.0\.1:\d+\/v1\/embeddings \((ECONNREFUSED|no answer)/);
152 });
153 
154 test('group labels use the distinctive shared words, not "add full support"', () => {
155   store.addIdeas([
156     'Add full support for subread to hoshireader',
157     'Add full support for subread and whispersync to chimahon',
158     'Add full support for whispersync to koreader',
159     'Add a back button to the key screen',
160   ]);
161   const ideas = store.load().ideas;
162   assert.equal(embed.groupLabel(ideas.slice(0, 3), ideas), 'subread ยท whispersync');
163 });