import { test, expect, _electron as electron } from '@playwright/test'; import type { ElectronApplication, Page } from '@playwright/test'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; /** * Proves the two hardest-to-fake claims about the AI Assistant's `local` * class in the REAL packaged desktop shell, not a browser tab: * * 1. The LLM genuinely runs locally — a direct browser-side fetch to * this machine's own Ollama (127.0.0.1:11434), never proxied through * this app's backend. * 2. It is genuinely grounded in the ENCRYPTED LOCAL SQLITE/FTS5 MAIL * INDEX (lib/mail-index/**), not the separate real-JMAP-embeddings * server leg (lib/ai/retrieval/mail-embeddings.ts) — AI_SERVER_BASE_URL * is deliberately left UNSET here so only the local FTS leg can * supply retrieval context. If this test passes, the local index * leg is the only possible source of the grounded answer. * * Needs a real launch through electron/main.ts's startStandaloneServer(), * not ELECTRON_LOAD_URL — that's the only code path that wires up the * fd-3 key channel / safeStorage the encrypted index depends on (see * integration/tests/12-electron-mail-index.spec.ts's header for the full * reasoning). That function picks a random free port every launch, which * would make it impossible to also point DEV_MOCK_JMAP's JMAP_SERVER_URL * at this same server's own /api/dev-jmap route — hence * VNCMAIL_TEST_FIXED_PORT, a narrow, off-by-default escape hatch added to * electron/main.ts specifically to make this test possible without a real * Stalwart fixture. * * Requires a real Ollama already running on this machine with at least one * completion-capable model installed — skips (not fails) otherwise, since * "no local LLM on this machine" is an environment fact, not a bug. */ const projectRoot = path.resolve(__dirname, '..'); const FIXED_PORT = 39217; const ORIGIN = `http://127.0.0.1:${FIXED_PORT}`; async function ollamaIsUp(): Promise { try { const res = await fetch('http://127.0.0.1:11434/api/tags'); if (!res.ok) return false; const body = (await res.json()) as { models?: Array<{ capabilities?: string[] }> }; return (body.models ?? []).some((m) => !m.capabilities || m.capabilities.includes('completion')); } catch { return false; } } test.describe('Electron desktop shell - local LLM answers from the real encrypted mail index', () => { let electronApp: ElectronApplication; let appWindow: Page; let userDataDir: string; test.beforeAll(async () => { if (!(await ollamaIsUp())) { test.skip(true, 'No local Ollama with a completion-capable model reachable on this machine — environment fact, not a failure.'); } userDataDir = fs.mkdtempSync(path.join(os.tmpdir(), 'vncmail-ai-index-test-')); electronApp = await electron.launch({ args: [projectRoot, `--user-data-dir=${userDataDir}`], env: { ...process.env, VNCMAIL_TEST_FIXED_PORT: String(FIXED_PORT), DEV_MOCK_JMAP: 'true', JMAP_SERVER_URL: `${ORIGIN}/api/dev-jmap`, SESSION_SECRET: 'electron-ai-local-index-verify-32-chars-min', // Deliberately UNSET: isolates grounding to the local FTS leg (see // module header) — the server embeddings leg 404s cleanly instead // of silently also being able to answer the question. AI_SERVER_BASE_URL: '', NODE_ENV: 'production', }, }); appWindow = await electronApp.firstWindow(); await appWindow.waitForLoadState('domcontentloaded'); }); test.afterAll(async () => { await electronApp?.close(); if (userDataDir) fs.rmSync(userDataDir, { recursive: true, force: true }); }); test('logs in, builds the real encrypted index, and a local Ollama model answers a mail question grounded in it', async () => { // ── 1. Real dev-mode login (sets the real session cookie the offline // index and every other server-side-identity feature need). ── const devLoginContainer = appWindow.locator('div', { hasText: 'Dev mode - logging in as dev@localhost' }).last(); await devLoginContainer.getByRole('button').click(); await appWindow.waitForURL((url) => !url.pathname.includes('login'), { timeout: 20000 }); // ── 2. Build the real encrypted local index: delta-sync the mock // account's mail into the replica store, then write it into SQLite/FTS5. // Chains /api/offline/sync while unfinishedWork is true, capped so a // real bug can't hang the test forever. ── const syncOutcome = await appWindow.evaluate(async () => { let unfinished = true; let calls = 0; const statuses: number[] = []; while (unfinished && calls < 10) { const res = await fetch('/api/offline/sync', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: '{}' }); statuses.push(res.status); if (!res.ok) break; const body = await res.json(); unfinished = body.unfinishedWork === true; calls++; } const reindexRes = await fetch('/api/offline/reindex', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ catchUp: true }) }); return { syncStatuses: statuses, syncCalls: calls, reindexStatus: reindexRes.status, reindexBody: await reindexRes.json().catch(() => null) }; }); console.log('[ai-local-index] sync+reindex outcome:', JSON.stringify(syncOutcome)); expect(syncOutcome.syncStatuses.every((s) => s === 200)).toBe(true); expect(syncOutcome.reindexStatus).toBe(200); // ── 3. Prove the local index itself is real and queryable BEFORE // touching the LLM at all — isolates "is the SQLite/FTS5 index working" // from "did the model use it correctly". ── const directSearch = await appWindow.evaluate(async () => { const res = await fetch(`/api/offline/search?q=${encodeURIComponent('Villa sul Lago check-in')}&limit=6`); return { status: res.status, body: await res.json().catch(() => null) }; }); console.log('[ai-local-index] direct /api/offline/search result:', JSON.stringify(directSearch.body)); expect(directSearch.status, 'the encrypted local index must be reachable (200), not 404 (feature disabled) or 503 (no key channel)').toBe(200); expect(directSearch.body?.ok).toBe(true); const hitTitles = (directSearch.body?.hits ?? []).map((h: { title?: string }) => h.title ?? ''); expect(hitTitles.some((t: string) => /villa sul lago/i.test(t)), `expected a "Villa sul Lago" hit in the real index, got: ${JSON.stringify(hitTitles)}`).toBe(true); // ── 4. Navigate to the real AI Assistant settings UI and use the // local-discovery "Connect" banner — the exact flow a real user takes, // proving discovery -> connect -> ask works as one integrated feature. // Deliberately an in-app SPA navigation (click the real sidebar link), // NOT appWindow.goto() — a full page reload drops whatever client-only // session state the dev-mode login established (confirmed: goto('/settings') // bounces straight back to /login even though the JMAP session cookie // from step 2/3 is still valid), so the click is load-bearing, not // cosmetic. ── await appWindow.locator('a[href="/settings"], a[href*="/settings"]').first().click(); const searchBox = appWindow.locator('input[type="search"]').first(); await searchBox.fill('AI Assistant'); await appWindow.getByRole('button', { name: 'AI Assistant' }).click(); const connectButton = appWindow.getByRole('button', { name: /Connect/i }); await expect(connectButton, 'the local-discovery banner should appear since a real Ollama is running on this machine').toBeVisible({ timeout: 10000 }); const bannerText = await appWindow.locator('text=Local AI found on this machine').locator('..').innerText(); console.log('[ai-local-index] discovery banner text:', bannerText); await connectButton.click(); // ── 5. Ask a question only answerable by combining the local LLM // with the local index's actual content. ── const questionBox = appWindow.getByPlaceholder(/What did legal say/i); await questionBox.fill('When is check-in for the Villa sul Lago booking, and what time?'); const askButton = appWindow.getByRole('button', { name: /^Ask$/ }); await expect(askButton, 'Ask must be enabled immediately after Connect pre-fills provider+model').toBeEnabled({ timeout: 5000 }); const chatRequests: string[] = []; const offlineSearchCalls: Array<{ url: string; status: number; body: unknown }> = []; appWindow.on('request', (req) => { if (req.url().includes('11434')) chatRequests.push(`${req.method()} ${req.url()}`); }); appWindow.on('response', async (res) => { if (res.url().includes('/api/offline/search')) { offlineSearchCalls.push({ url: res.url(), status: res.status(), body: await res.json().catch(() => null) }); } }); // Log the exact prompt Ollama actually received, straight from the // request body — the ground truth for "did retrieval even fire". const ollamaChatPayloads: unknown[] = []; await appWindow.route('**/api/chat', async (route) => { try { ollamaChatPayloads.push(JSON.parse(route.request().postData() ?? 'null')); } catch { /* ignore parse failure, still let the request through */ } await route.continue(); }); await askButton.click(); const answerLocator = appWindow.locator('p.whitespace-pre-wrap').first(); await expect(answerLocator, 'the local Ollama model should produce an answer within a generous timeout').toBeVisible({ timeout: 60000 }); const answerText = await answerLocator.innerText(); console.log('[ai-local-index] final answer:', answerText); console.log('[ai-local-index] direct-to-Ollama requests observed:', chatRequests); console.log('[ai-local-index] /api/offline/search calls during Ask:', JSON.stringify(offlineSearchCalls)); console.log('[ai-local-index] exact payload(s) sent to Ollama /api/chat:', JSON.stringify(ollamaChatPayloads)); // The real proof: the model's own words contain the fact that only // exists in the indexed email (28 March, 15:00), and the request log // shows the renderer talked to Ollama's loopback address directly. expect(answerText).toMatch(/28\s*march|march\s*28/i); expect(answerText).toMatch(/15:00|3\s*pm|3:00\s*pm/i); expect(chatRequests.some((r) => r.includes('/api/chat')), `expected a direct renderer -> Ollama /api/chat request, saw: ${JSON.stringify(chatRequests)}`).toBe(true); await appWindow.screenshot({ path: path.join(projectRoot, 'electron-ai-local-index-result.png'), fullPage: true }); }); });