Files
SRCmail/e2e/electron-ai-local-index.spec.ts
T
Bernd Rodler 7b2047681e fix(mail-index): local AI retrieval returned 0 hits for real questions — AND-every-token FTS matching killed on stop words
Found by a new real Electron e2e test built specifically to prove the
`local` AI class genuinely works end-to-end in the packaged desktop shell:
a real local Ollama model answering a real question, grounded in the real
encrypted SQLite/FTS5 mail index — not a browser tab, not a mock.

First run surfaced a genuine bug: toFtsMatchQuery() AND-joins every token,
which is right for a deliberate search-box query but wrong for the natural-
language questions the AI retrieval surface (/api/offline/search — see its
own module header, "THE RETRIEVAL SURFACE") actually receives. "When is
check-in for the Villa sul Lago booking, and what time?" shares almost none
of its own function words with the email that answers it, so ANDing every
token — including "when"/"is"/"for"/"the"/"and"/"what" — returned 0 hits
against an index that correctly returns the right email for "Villa sul Lago
check-in".

Fix: new toFtsMatchQueryAny() (lib/mail-index/store.ts) — drops a small,
well-known English stop-word list, OR-joins what's left, and lets the
existing bm25 ranking pick the winner among partial matches. Deliberately a
NEW function, not a change to toFtsMatchQuery itself: that one's own tests
rely on "AND"/"OR"/"NOT" surviving verbatim as literal search terms
(FTS5-keyword-injection safety) — a different guarantee than this one's job
of turning a question into a good search. search() gains a `mode: 'and' |
'any'` option (default 'and', so every existing caller is unaffected); the
offline-search route passes 'any', since its one real caller is exactly
this AI-question shape.

Also added, to make the e2e test possible at all: electron/main.ts's
VNCMAIL_TEST_FIXED_PORT — a narrow, off-by-default escape hatch so
DEV_MOCK_JMAP's JMAP_SERVER_URL can point at this same standalone server's
own /api/dev-jmap. Needed because the encrypted index's key channel
(fd-3/safeStorage) only gets wired up in startStandaloneServer()'s own
random-port launch path, never when ELECTRON_LOAD_URL bypasses it for a
plain `next dev` target — so this was the only way to exercise the real
index without a full Stalwart+SMTP Docker fixture.

Verified live in the real packaged Electron shell, not just unit tests:
real dev-mode login, real multi-round /api/offline/sync + /api/offline/reindex
(39 mail/35 calendar/23 contacts indexed), real local-discovery banner
(11 real Ollama models on this machine), real "Connect", a real question
through the real Settings UI, a real direct renderer->Ollama /api/chat call
(confirmed via network log, never proxied through this app's backend), and
the model's own answer citing the exact right fact: "Saturday 28 March at
15:00" — a fact that exists nowhere except in the one indexed email.

4 new unit tests for toFtsMatchQueryAny. Full gate: tsc clean, eslint
clean, 2502/2502 tests passing, build clean, e2e/electron-ai-local-index.spec.ts
passing against the real standalone server + real Electron + real Ollama.
2026-08-06 18:09:14 +02:00

200 lines
11 KiB
TypeScript

import { test, expect, _electron as electron } from '@playwright/test';
import type { ElectronApplication, Page } from '@playwright/test';
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* Proves the two hardest-to-fake claims about the AI Assistant's `local`
* class in the REAL packaged desktop shell, not a browser tab:
*
* 1. The LLM genuinely runs locally — a direct browser-side fetch to
* this machine's own Ollama (127.0.0.1:11434), never proxied through
* this app's backend.
* 2. It is genuinely grounded in the ENCRYPTED LOCAL SQLITE/FTS5 MAIL
* INDEX (lib/mail-index/**), not the separate real-JMAP-embeddings
* server leg (lib/ai/retrieval/mail-embeddings.ts) — AI_SERVER_BASE_URL
* is deliberately left UNSET here so only the local FTS leg can
* supply retrieval context. If this test passes, the local index
* leg is the only possible source of the grounded answer.
*
* Needs a real launch through electron/main.ts's startStandaloneServer(),
* not ELECTRON_LOAD_URL — that's the only code path that wires up the
* fd-3 key channel / safeStorage the encrypted index depends on (see
* integration/tests/12-electron-mail-index.spec.ts's header for the full
* reasoning). That function picks a random free port every launch, which
* would make it impossible to also point DEV_MOCK_JMAP's JMAP_SERVER_URL
* at this same server's own /api/dev-jmap route — hence
* VNCMAIL_TEST_FIXED_PORT, a narrow, off-by-default escape hatch added to
* electron/main.ts specifically to make this test possible without a real
* Stalwart fixture.
*
* Requires a real Ollama already running on this machine with at least one
* completion-capable model installed — skips (not fails) otherwise, since
* "no local LLM on this machine" is an environment fact, not a bug.
*/
const projectRoot = path.resolve(__dirname, '..');
const FIXED_PORT = 39217;
const ORIGIN = `http://127.0.0.1:${FIXED_PORT}`;
async function ollamaIsUp(): Promise<boolean> {
try {
const res = await fetch('http://127.0.0.1:11434/api/tags');
if (!res.ok) return false;
const body = (await res.json()) as { models?: Array<{ capabilities?: string[] }> };
return (body.models ?? []).some((m) => !m.capabilities || m.capabilities.includes('completion'));
} catch {
return false;
}
}
test.describe('Electron desktop shell - local LLM answers from the real encrypted mail index', () => {
let electronApp: ElectronApplication;
let appWindow: Page;
let userDataDir: string;
test.beforeAll(async () => {
if (!(await ollamaIsUp())) {
test.skip(true, 'No local Ollama with a completion-capable model reachable on this machine — environment fact, not a failure.');
}
userDataDir = fs.mkdtempSync(path.join(os.tmpdir(), 'vncmail-ai-index-test-'));
electronApp = await electron.launch({
args: [projectRoot, `--user-data-dir=${userDataDir}`],
env: {
...process.env,
VNCMAIL_TEST_FIXED_PORT: String(FIXED_PORT),
DEV_MOCK_JMAP: 'true',
JMAP_SERVER_URL: `${ORIGIN}/api/dev-jmap`,
SESSION_SECRET: 'electron-ai-local-index-verify-32-chars-min',
// Deliberately UNSET: isolates grounding to the local FTS leg (see
// module header) — the server embeddings leg 404s cleanly instead
// of silently also being able to answer the question.
AI_SERVER_BASE_URL: '',
NODE_ENV: 'production',
},
});
appWindow = await electronApp.firstWindow();
await appWindow.waitForLoadState('domcontentloaded');
});
test.afterAll(async () => {
await electronApp?.close();
if (userDataDir) fs.rmSync(userDataDir, { recursive: true, force: true });
});
test('logs in, builds the real encrypted index, and a local Ollama model answers a mail question grounded in it', async () => {
// ── 1. Real dev-mode login (sets the real session cookie the offline
// index and every other server-side-identity feature need). ──
const devLoginContainer = appWindow.locator('div', { hasText: 'Dev mode - logging in as dev@localhost' }).last();
await devLoginContainer.getByRole('button').click();
await appWindow.waitForURL((url) => !url.pathname.includes('login'), { timeout: 20000 });
// ── 2. Build the real encrypted local index: delta-sync the mock
// account's mail into the replica store, then write it into SQLite/FTS5.
// Chains /api/offline/sync while unfinishedWork is true, capped so a
// real bug can't hang the test forever. ──
const syncOutcome = await appWindow.evaluate(async () => {
let unfinished = true;
let calls = 0;
const statuses: number[] = [];
while (unfinished && calls < 10) {
const res = await fetch('/api/offline/sync', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: '{}' });
statuses.push(res.status);
if (!res.ok) break;
const body = await res.json();
unfinished = body.unfinishedWork === true;
calls++;
}
const reindexRes = await fetch('/api/offline/reindex', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ catchUp: true }) });
return { syncStatuses: statuses, syncCalls: calls, reindexStatus: reindexRes.status, reindexBody: await reindexRes.json().catch(() => null) };
});
console.log('[ai-local-index] sync+reindex outcome:', JSON.stringify(syncOutcome));
expect(syncOutcome.syncStatuses.every((s) => s === 200)).toBe(true);
expect(syncOutcome.reindexStatus).toBe(200);
// ── 3. Prove the local index itself is real and queryable BEFORE
// touching the LLM at all — isolates "is the SQLite/FTS5 index working"
// from "did the model use it correctly". ──
const directSearch = await appWindow.evaluate(async () => {
const res = await fetch(`/api/offline/search?q=${encodeURIComponent('Villa sul Lago check-in')}&limit=6`);
return { status: res.status, body: await res.json().catch(() => null) };
});
console.log('[ai-local-index] direct /api/offline/search result:', JSON.stringify(directSearch.body));
expect(directSearch.status, 'the encrypted local index must be reachable (200), not 404 (feature disabled) or 503 (no key channel)').toBe(200);
expect(directSearch.body?.ok).toBe(true);
const hitTitles = (directSearch.body?.hits ?? []).map((h: { title?: string }) => h.title ?? '');
expect(hitTitles.some((t: string) => /villa sul lago/i.test(t)), `expected a "Villa sul Lago" hit in the real index, got: ${JSON.stringify(hitTitles)}`).toBe(true);
// ── 4. Navigate to the real AI Assistant settings UI and use the
// local-discovery "Connect" banner — the exact flow a real user takes,
// proving discovery -> connect -> ask works as one integrated feature.
// Deliberately an in-app SPA navigation (click the real sidebar link),
// NOT appWindow.goto() — a full page reload drops whatever client-only
// session state the dev-mode login established (confirmed: goto('/settings')
// bounces straight back to /login even though the JMAP session cookie
// from step 2/3 is still valid), so the click is load-bearing, not
// cosmetic. ──
await appWindow.locator('a[href="/settings"], a[href*="/settings"]').first().click();
const searchBox = appWindow.locator('input[type="search"]').first();
await searchBox.fill('AI Assistant');
await appWindow.getByRole('button', { name: 'AI Assistant' }).click();
const connectButton = appWindow.getByRole('button', { name: /Connect/i });
await expect(connectButton, 'the local-discovery banner should appear since a real Ollama is running on this machine').toBeVisible({ timeout: 10000 });
const bannerText = await appWindow.locator('text=Local AI found on this machine').locator('..').innerText();
console.log('[ai-local-index] discovery banner text:', bannerText);
await connectButton.click();
// ── 5. Ask a question only answerable by combining the local LLM
// with the local index's actual content. ──
const questionBox = appWindow.getByPlaceholder(/What did legal say/i);
await questionBox.fill('When is check-in for the Villa sul Lago booking, and what time?');
const askButton = appWindow.getByRole('button', { name: /^Ask$/ });
await expect(askButton, 'Ask must be enabled immediately after Connect pre-fills provider+model').toBeEnabled({ timeout: 5000 });
const chatRequests: string[] = [];
const offlineSearchCalls: Array<{ url: string; status: number; body: unknown }> = [];
appWindow.on('request', (req) => {
if (req.url().includes('11434')) chatRequests.push(`${req.method()} ${req.url()}`);
});
appWindow.on('response', async (res) => {
if (res.url().includes('/api/offline/search')) {
offlineSearchCalls.push({ url: res.url(), status: res.status(), body: await res.json().catch(() => null) });
}
});
// Log the exact prompt Ollama actually received, straight from the
// request body — the ground truth for "did retrieval even fire".
const ollamaChatPayloads: unknown[] = [];
await appWindow.route('**/api/chat', async (route) => {
try {
ollamaChatPayloads.push(JSON.parse(route.request().postData() ?? 'null'));
} catch { /* ignore parse failure, still let the request through */ }
await route.continue();
});
await askButton.click();
const answerLocator = appWindow.locator('p.whitespace-pre-wrap').first();
await expect(answerLocator, 'the local Ollama model should produce an answer within a generous timeout').toBeVisible({ timeout: 60000 });
const answerText = await answerLocator.innerText();
console.log('[ai-local-index] final answer:', answerText);
console.log('[ai-local-index] direct-to-Ollama requests observed:', chatRequests);
console.log('[ai-local-index] /api/offline/search calls during Ask:', JSON.stringify(offlineSearchCalls));
console.log('[ai-local-index] exact payload(s) sent to Ollama /api/chat:', JSON.stringify(ollamaChatPayloads));
// The real proof: the model's own words contain the fact that only
// exists in the indexed email (28 March, 15:00), and the request log
// shows the renderer talked to Ollama's loopback address directly.
expect(answerText).toMatch(/28\s*march|march\s*28/i);
expect(answerText).toMatch(/15:00|3\s*pm|3:00\s*pm/i);
expect(chatRequests.some((r) => r.includes('/api/chat')), `expected a direct renderer -> Ollama /api/chat request, saw: ${JSON.stringify(chatRequests)}`).toBe(true);
await appWindow.screenshot({ path: path.join(projectRoot, 'electron-ai-local-index-result.png'), fullPage: true });
});
});