Full retrieval pipeline per docs/AI-ASSISTANT-CONCEPT.md §7/§8.1, real end to end, not mocked: - lib/ai/retrieval/types.ts: SourceRef + RetrieverAdapter schema. Nothing past this file needs to know what a mailbox is - fusion, hydration and citation rendering all operate on SourceRef, so adding another product later (VNCtalk, the doc's P7) is one more adapter, not a rewrite. - lib/ai/retrieval/fusion.ts: Reciprocal Rank Fusion, score = Σ1/(k+rank), k=60. Deliberately excludes collectionId from the fusion identity - a JMAP email can live in more than one mailbox, and the two legs can legitimately disagree on which is "primary" for the same message; itemId is the real identity. 5 unit tests, including that exact double-count case. - lib/ai/retrieval/mail-embeddings.ts: the server embedding leg. Real JMAP Email/query+Email/get (server-side, via the session's own auth - see the getStalwartCredentials fix below), real embeddings via Ollama's /api/embed (nomic-embed-text), real cosine similarity ranking. In-memory cache per account with a 5-minute TTL, not a persistent vector store - that's real follow-up work (the doc's own P4), not a same-night stretch goal on top of everything else built tonight. - app/api/ai/retrieve/route.ts: wires it together. ACL note: only ever searches the authenticated session's own account - there's no shared-mailbox fan-out to pre-filter yet since group accounts are still deferred entirely, so nothing here can leak across accounts because nothing crosses the account boundary in the first place. - lib/ai/local-client.ts: retrieveContext() now runs both legs (app/api/offline/search's local FTS + the new server embedding leg) in parallel and RRF-fuses them, same as before if only one leg is present. Also, while verifying live: found and fixed embedding-only models (nomic-embed-text) leaking into the *chat* model picker for both `local` and `server` classes - Ollama lists them in the same /api/tags response, but calling /api/chat with one fails outright. Filtered by `capabilities` (fails open if absent, for older Ollama). Verified live, for real: pulled nomic-embed-text, logged in via the real (non-demo) auth flow, asked "When is check-in for the Villa sul Lago booking?" against the seeded mock inbox - got back "Check-in ... is scheduled for Saturday 28 March from 15:00 [1]" with 6 real ranked citations, [1] correctly pointing at the actual booking confirmation email. Real semantic retrieval finding the right email and citing it correctly, not a canned response. Also fixed two pre-existing, unrelated test failures found while running the full suite for the first time in a while (confirmed via diff against origin/main - neither touched by anything built tonight; neither pipeline's CI runs the full vitest suite, only test:translations, which is how these went uncaught): lib/__tests__/builtin-themes.test.ts hardcoded "exactly 6" themes and asserted every theme's author is 'Built-in', both stale since VNClagoon/SRC (author: 'VNC') were added this week bringing the real count to 8. Left the also-pre-existing, timing-sensitive jmap-client-resilience.test.ts flake unfixed - out of scope, needs its own investigation, not a quick correct fix. Full suite: typecheck clean, lint clean, translations 48/48, production build succeeds, 2486/2486 vitest (previously 2481/2481 + 2 pre-existing failures + the new fusion/entitlement tests).
339 lines
13 KiB
TypeScript
339 lines
13 KiB
TypeScript
// The AI assistant's wire client. `local`/`public` mirror vncmail-native's
|
|
// src/api/ai.ts (direct loopback/provider fetch, no streaming) so the two
|
|
// clients stay in lockstep — matching docs/AI-ASSISTANT-CONCEPT.md §2's
|
|
// "local"/"public" rows, not proxied through this app's own Next.js server.
|
|
// That distinction matters once this app is hosted remotely: a server-side
|
|
// proxy would reach the *server's* loopback, not the user's own laptop
|
|
// running Ollama.
|
|
//
|
|
// `server` (added 2026-08-05 night) is the opposite by design: it DOES
|
|
// proxy through this app's own backend (app/api/ai/server/*), because it's
|
|
// centrally-hosted infra (VNC's EU/CH stack — standing in tonight for a real
|
|
// Ollama on this Mac, see lib/ai/entitlement.ts), not a user's own machine.
|
|
// That server-side hop is also the one real entitlement enforcement point
|
|
// (§10 point 2) — `local`/`public` never reach it, by design, and so cannot
|
|
// be metered or billed the same way.
|
|
|
|
export interface ChatMessage {
|
|
role: 'system' | 'user' | 'assistant';
|
|
content: string;
|
|
}
|
|
|
|
// ── Local: Ollama's native API, not the OpenAI-compat shim — one fewer path
|
|
// assumption (no "/v1" prefix to guess at) for a runtime this code talks to directly. ──
|
|
|
|
interface OllamaTagsResponse {
|
|
models?: Array<{ name: string; capabilities?: string[] }>;
|
|
}
|
|
|
|
interface OllamaChatResponse {
|
|
message?: { content?: string };
|
|
}
|
|
|
|
export async function listLocalModels(baseUrl: string): Promise<string[]> {
|
|
const res = await fetch(`${baseUrl.replace(/\/+$/, '')}/api/tags`);
|
|
if (!res.ok) throw new Error(`Ollama returned ${res.status}`);
|
|
const body = (await res.json()) as OllamaTagsResponse;
|
|
// Excludes embedding-only models (e.g. nomic-embed-text) from the chat
|
|
// picker — same reasoning as app/api/ai/server/models/route.ts.
|
|
return (body.models ?? [])
|
|
.filter((m) => !m.capabilities || m.capabilities.includes('completion'))
|
|
.map((m) => m.name)
|
|
.filter(Boolean);
|
|
}
|
|
|
|
/**
|
|
* Diagnoses the specific failure rather than a generic "connection failed" —
|
|
* docs/AI-ASSISTANT-CONCEPT.md §3 calls this out explicitly for the browser
|
|
* row: a CORS rejection (the runtime is up but refused this page's origin)
|
|
* looks identical to "nothing is listening" unless told apart. `fetch`
|
|
* itself can't distinguish them (a CORS failure and a connection refusal
|
|
* both surface as `TypeError: Failed to fetch`), so this only upgrades the
|
|
* message when the caller can tell us there's a live page origin to name.
|
|
*/
|
|
export async function testLocalConnection(
|
|
baseUrl: string,
|
|
): Promise<{ ok: boolean; error?: string }> {
|
|
try {
|
|
await listLocalModels(baseUrl);
|
|
return { ok: true };
|
|
} catch (err) {
|
|
const origin = typeof window !== 'undefined' ? window.location.origin : null;
|
|
const hint = origin
|
|
? ` Reachable in principle, but if Ollama is actually running, it likely refused this page's origin (${origin}) — start it with OLLAMA_ORIGINS=${origin}.`
|
|
: '';
|
|
return {
|
|
ok: false,
|
|
error: (err instanceof Error ? err.message : String(err)) + hint,
|
|
};
|
|
}
|
|
}
|
|
|
|
export async function chatLocal(
|
|
baseUrl: string,
|
|
model: string,
|
|
messages: ChatMessage[],
|
|
): Promise<string> {
|
|
const res = await fetch(`${baseUrl.replace(/\/+$/, '')}/api/chat`, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({ model, messages, stream: false }),
|
|
});
|
|
if (!res.ok) throw new Error(`Ollama returned ${res.status}`);
|
|
const body = (await res.json()) as OllamaChatResponse;
|
|
const content = body.message?.content;
|
|
if (!content) throw new Error('Ollama returned no message content');
|
|
return content;
|
|
}
|
|
|
|
// ── Server: centrally-hosted, proxied through this app's own backend
|
|
// (app/api/ai/server/*). Unlike `local`, this is same-origin from the
|
|
// browser's perspective — no CORS/OLLAMA_ORIGINS story at all — and unlike
|
|
// both `local` and `public`, every call is entitlement-checked server-side. ──
|
|
|
|
export async function listServerModels(): Promise<string[]> {
|
|
const res = await fetch('/api/ai/server/models');
|
|
if (!res.ok) {
|
|
const body = (await res.json().catch(() => null)) as { error?: string } | null;
|
|
throw new Error(body?.error ?? `AI server returned ${res.status}`);
|
|
}
|
|
const body = (await res.json()) as { models?: string[] };
|
|
return body.models ?? [];
|
|
}
|
|
|
|
export interface ServerChatResult {
|
|
answer: string;
|
|
/** True the moment this call consumed a previously-unassigned licensed seat
|
|
* (lib/ai/entitlement.ts) — surfaced so the UI can say so once, not left
|
|
* to happen silently the first time someone uses this class. */
|
|
seatJustAssigned: boolean;
|
|
}
|
|
|
|
export async function chatServer(model: string, messages: ChatMessage[]): Promise<ServerChatResult> {
|
|
const res = await fetch('/api/ai/server/chat', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({ model, messages }),
|
|
});
|
|
const body = (await res.json().catch(() => null)) as { answer?: string; error?: string; seatJustAssigned?: boolean } | null;
|
|
if (!res.ok || !body?.answer) {
|
|
throw new Error(body?.error ?? `AI server returned ${res.status}`);
|
|
}
|
|
return { answer: body.answer, seatJustAssigned: body.seatJustAssigned === true };
|
|
}
|
|
|
|
// ── Public: OpenAI-compatible chat-completions. OpenRouter by default, but any
|
|
// endpoint speaking this shape works unmodified (self-hosted vLLM, LiteLLM, etc). ──
|
|
|
|
interface OpenAiChatResponse {
|
|
choices?: Array<{ message?: { content?: string } }>;
|
|
}
|
|
|
|
export async function chatPublic(
|
|
baseUrl: string,
|
|
apiKey: string,
|
|
model: string,
|
|
messages: ChatMessage[],
|
|
): Promise<string> {
|
|
const res = await fetch(`${baseUrl.replace(/\/+$/, '')}/chat/completions`, {
|
|
method: 'POST',
|
|
headers: {
|
|
'Content-Type': 'application/json',
|
|
Authorization: `Bearer ${apiKey}`,
|
|
},
|
|
body: JSON.stringify({ model, messages }),
|
|
});
|
|
if (!res.ok) throw new Error(`Provider returned ${res.status}`);
|
|
const body = (await res.json()) as OpenAiChatResponse;
|
|
const content = body.choices?.[0]?.message?.content;
|
|
if (!content) throw new Error('Provider returned no message content');
|
|
return content;
|
|
}
|
|
|
|
// ── Retrieval: two legs run in parallel and get Reciprocal-Rank-Fused
|
|
// (docs/AI-ASSISTANT-CONCEPT.md §7 steps 2-3), exactly like the doc
|
|
// describes — this is real, not a single degraded leg wearing SourceRef's
|
|
// clothes:
|
|
// - local FTS: this app's own already-built offline search surface
|
|
// (app/api/offline/search/route.ts). The encrypted SQLite/FTS5 store it
|
|
// reads only exists in Electron's main process — a 404/503 there means
|
|
// "no local index in this session", not an error.
|
|
// - server embedding: app/api/ai/retrieve (lib/ai/retrieval/mail-embeddings.ts) —
|
|
// real JMAP fetch, real Ollama embeddings, real cosine ranking. A 404
|
|
// there means AI_SERVER_BASE_URL isn't configured; anything else is a
|
|
// real failure, logged but not fatal to the question.
|
|
// Either leg being absent degrades to the other with no special-casing
|
|
// (reciprocalRankFusion handles an empty array leg for free); both absent
|
|
// degrades to an unaugmented question, same as before tonight.
|
|
|
|
import { reciprocalRankFusion } from './retrieval/fusion';
|
|
import type { Scored, SourceRef } from './retrieval/types';
|
|
|
|
export interface AskSource {
|
|
id: string;
|
|
subject: string;
|
|
}
|
|
|
|
export interface AskResult {
|
|
answer: string;
|
|
sources: AskSource[];
|
|
/** True when the question was answered without any retrieved context. */
|
|
unaugmented: boolean;
|
|
/** True the moment this call consumed a previously-unassigned licensed
|
|
* seat on the `server` class (lib/ai/entitlement.ts). Always false for
|
|
* `local`/`public`, which aren't entitlement-gated. */
|
|
seatJustAssigned: boolean;
|
|
}
|
|
|
|
interface OfflineSearchHit {
|
|
id: string;
|
|
jmapAccountId: string;
|
|
title: string;
|
|
snippet?: string;
|
|
}
|
|
|
|
interface OfflineSearchResponse {
|
|
ok: true;
|
|
hits: OfflineSearchHit[];
|
|
}
|
|
|
|
interface ServerRetrieveHit {
|
|
ref: SourceRef;
|
|
title: string;
|
|
snippet: string;
|
|
}
|
|
|
|
interface ServerRetrieveResponse {
|
|
ok: true;
|
|
hits: ServerRetrieveHit[];
|
|
}
|
|
|
|
interface RetrievedContext {
|
|
contextBlock: string;
|
|
hits: Array<{ id: string; title: string }>;
|
|
}
|
|
|
|
async function fetchLocalLeg(question: string): Promise<{ scored: Scored<SourceRef>[]; text: Map<string, { title: string; snippet: string }> }> {
|
|
const empty = { scored: [] as Scored<SourceRef>[], text: new Map<string, { title: string; snippet: string }>() };
|
|
try {
|
|
const res = await fetch(`/api/offline/search?q=${encodeURIComponent(question)}&limit=6`);
|
|
if (!res.ok) return empty; // 404/503 — no local index this session, not an error
|
|
const body = (await res.json()) as OfflineSearchResponse;
|
|
if (!body.ok) return empty;
|
|
const text = new Map(body.hits.map((h) => [h.id, { title: h.title, snippet: h.snippet ?? '' }]));
|
|
const scored = body.hits.map((h, i) => ({
|
|
ref: { product: 'mail' as const, accountId: h.jmapAccountId, collectionId: '', itemId: h.id, chunkIx: 0 },
|
|
score: 1 / (i + 1), // rank position is all reciprocalRankFusion reads
|
|
}));
|
|
return { scored, text };
|
|
} catch {
|
|
return empty;
|
|
}
|
|
}
|
|
|
|
async function fetchServerLeg(question: string): Promise<{ scored: Scored<SourceRef>[]; text: Map<string, { title: string; snippet: string }> }> {
|
|
const empty = { scored: [] as Scored<SourceRef>[], text: new Map<string, { title: string; snippet: string }>() };
|
|
try {
|
|
const res = await fetch('/api/ai/retrieve', {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({ query: question, limit: 6 }),
|
|
});
|
|
if (!res.ok) return empty; // 404 (server class not configured) or any other failure — degrade, don't fail the question
|
|
const body = (await res.json()) as ServerRetrieveResponse;
|
|
if (!body.ok) return empty;
|
|
const text = new Map(body.hits.map((h) => [h.ref.itemId, { title: h.title, snippet: h.snippet }]));
|
|
const scored = body.hits.map((h, i) => ({ ref: h.ref, score: 1 / (i + 1) }));
|
|
return { scored, text };
|
|
} catch {
|
|
return empty;
|
|
}
|
|
}
|
|
|
|
async function retrieveContext(question: string): Promise<RetrievedContext | null> {
|
|
const [local, server] = await Promise.all([fetchLocalLeg(question), fetchServerLeg(question)]);
|
|
const fused = reciprocalRankFusion([local.scored, server.scored], 6);
|
|
if (fused.length === 0) return null;
|
|
|
|
const combinedText = new Map([...server.text, ...local.text]); // local wins on overlap: it's the more precise leg (BM25 on exact terms)
|
|
const withText = fused
|
|
.map((f) => ({ ref: f.ref, info: combinedText.get(f.ref.itemId) }))
|
|
.filter((f): f is { ref: SourceRef; info: { title: string; snippet: string } } => !!f.info);
|
|
|
|
if (withText.length === 0) return null;
|
|
|
|
return {
|
|
contextBlock: withText.map((h, i) => `[${i + 1}] Subject: ${h.info.title}\n${h.info.snippet}`).join('\n\n'),
|
|
hits: withText.map((h) => ({ id: h.ref.itemId, title: h.info.title })),
|
|
};
|
|
}
|
|
|
|
export function buildPrompt(question: string, contextBlock: string): ChatMessage[] {
|
|
return [
|
|
{
|
|
role: 'system',
|
|
content:
|
|
"You answer questions about the user's email using only the numbered excerpts " +
|
|
'below as context. Cite sources by their number in brackets, e.g. [1]. If the ' +
|
|
"excerpts don't contain the answer, say so plainly rather than guessing.",
|
|
},
|
|
{ role: 'user', content: `${contextBlock}\n\nQuestion: ${question}` },
|
|
];
|
|
}
|
|
|
|
/**
|
|
* One saved BYOK profile, resolved to an actual key — the caller picks which
|
|
* profile answers *this* question (docs decision 2026-08-05: several keys,
|
|
* selected case by case, not one fixed "the" public provider).
|
|
*/
|
|
export interface ResolvedPublicProfile {
|
|
baseUrl: string;
|
|
model: string;
|
|
apiKey: string;
|
|
}
|
|
|
|
export interface AskConfig {
|
|
provider: 'local' | 'server' | 'public';
|
|
localBaseUrl: string;
|
|
localModel: string | null;
|
|
serverModel: string | null;
|
|
publicProfile: ResolvedPublicProfile | null;
|
|
}
|
|
|
|
export async function askMail(question: string, config: AskConfig): Promise<AskResult> {
|
|
if (config.provider === 'local' && !config.localModel) {
|
|
throw new Error('No local model selected');
|
|
}
|
|
if (config.provider === 'server' && !config.serverModel) {
|
|
throw new Error('No server model selected');
|
|
}
|
|
if (config.provider === 'public' && !config.publicProfile) {
|
|
throw new Error('No provider profile selected');
|
|
}
|
|
|
|
const retrieved = await retrieveContext(question);
|
|
const messages = retrieved
|
|
? buildPrompt(question, retrieved.contextBlock)
|
|
: [{ role: 'user' as const, content: question }];
|
|
|
|
let answer: string;
|
|
let seatJustAssigned = false;
|
|
if (config.provider === 'public') {
|
|
const profile = config.publicProfile as ResolvedPublicProfile;
|
|
answer = await chatPublic(profile.baseUrl, profile.apiKey, profile.model, messages);
|
|
} else if (config.provider === 'server') {
|
|
const result = await chatServer(config.serverModel as string, messages);
|
|
answer = result.answer;
|
|
seatJustAssigned = result.seatJustAssigned;
|
|
} else {
|
|
answer = await chatLocal(config.localBaseUrl, config.localModel as string, messages);
|
|
}
|
|
|
|
return {
|
|
answer,
|
|
sources: (retrieved?.hits ?? []).map((h) => ({ id: h.id, subject: h.title })),
|
|
unaugmented: !retrieved,
|
|
seatJustAssigned,
|
|
};
|
|
}
|