This commit is contained in:
316
packages/server/src/local/routes/agent-run.ts
Normal file
316
packages/server/src/local/routes/agent-run.ts
Normal file
@@ -0,0 +1,316 @@
|
||||
// CC Sesija A A3.1 follow-up — /api/agent/run sidecar route.
|
||||
//
|
||||
// Backs the Tauri run_agent_query command for the structured-action retrieval
|
||||
// loop (runRetrievalAgentLoop), which is shape-aware. Distinct from /api/chat
|
||||
// (runAgentLoop, conversational, multi-turn message history) because shapes
|
||||
// are designed for one-shot Q→A retrieval flows — not for conversational
|
||||
// chat. Two coexisting paths matches actual two-product reality:
|
||||
// /api/chat → runAgentLoop (conversation, no shapes)
|
||||
// /api/agent/run → runRetrievalAgentLoop (research / one-shot, shape-driven)
|
||||
//
|
||||
// Streaming via SSE matches /api/chat pattern (reply.hijack + writeHead +
|
||||
// raw.write event blocks). Per-step progress events come from the agent
|
||||
// loop's onProgress callback (Phase 3.4 — AgentRunProgressEvent).
|
||||
|
||||
import type { FastifyPluginAsync } from 'fastify';
|
||||
import { HybridSearch } from '@waggle/core';
|
||||
import {
|
||||
runRetrievalAgentLoop,
|
||||
listShapes,
|
||||
registerShape,
|
||||
claudeGen1V1Shape,
|
||||
qwenThinkingGen1V1Shape,
|
||||
type LlmCallFn,
|
||||
type LlmCallInput,
|
||||
type LlmCallResult,
|
||||
type RetrievalSearchFn,
|
||||
resolveModelForClass,
|
||||
LIGHTWEIGHT_MODEL,
|
||||
} from '@waggle/agent';
|
||||
|
||||
// CC Sesija A A3.2 (2026-04-30): register Faza 1 GEPA-evolved variants into
|
||||
// the prompt-shape REGISTRY at module-load. Phase 5 LOCKED scope is just two
|
||||
// shapes (claude-gen1-v1 + qwen-thinking-gen1-v1) — gen1-v2 variants stay
|
||||
// out per decisions/2026-04-29-phase-5-scope-LOCKED.md (Faza 2 OVERFIT
|
||||
// exposed in Checkpoint C). Registration is idempotent: registerShape()
|
||||
// overwrites by name, so duplicate route loads (HMR, tests) are safe.
|
||||
registerShape('claude-gen1-v1', claudeGen1V1Shape);
|
||||
registerShape('qwen-thinking-gen1-v1', qwenThinkingGen1V1Shape);
|
||||
|
||||
// Module-load defaults are LAST-RESORT only. The live values come from server
|
||||
// state at REQUEST time (see resolveLlmEndpoint): when LiteLLM is unavailable,
|
||||
// service.ts falls back to the built-in Anthropic proxy and writes the real
|
||||
// URL/key into server.localConfig.litellmUrl + server.agentState.litellmApiKey
|
||||
// at runtime. Reading the snapshot here would route /api/agent/run at the dead
|
||||
// LiteLLM default — broken for the common no-LiteLLM (Anthropic-only) case.
|
||||
const DEFAULT_LITELLM_URL = process.env.WAGGLE_LITELLM_URL ?? 'http://localhost:4000';
|
||||
const LITELLM_KEY =
|
||||
process.env.LITELLM_API_KEY ?? process.env.LITELLM_MASTER_KEY ?? 'sk-waggle-dev';
|
||||
const DEFAULT_MODEL = 'claude-sonnet-4-6';
|
||||
const DEFAULT_PERSONA = 'general-purpose';
|
||||
const DEFAULT_MAX_STEPS = 5;
|
||||
|
||||
/**
|
||||
* Resolve the LiteLLM endpoint URL + API key from LIVE server state at request
|
||||
* time. service.ts mutates these at runtime when it falls back from LiteLLM to
|
||||
* the built-in Anthropic proxy (server.localConfig.litellmUrl = self-proxy URL,
|
||||
* server.agentState.litellmApiKey = wsSessionToken), so a module-load snapshot
|
||||
* would miss the fallback. Falls back to the module defaults only when state is
|
||||
* absent (e.g. very early boot) so callers never get an empty endpoint.
|
||||
*/
|
||||
export function resolveLlmEndpoint(server: {
|
||||
localConfig?: { litellmUrl?: string };
|
||||
agentState?: { litellmApiKey?: string };
|
||||
}): { url: string; apiKey: string } {
|
||||
return {
|
||||
url: server.localConfig?.litellmUrl ?? DEFAULT_LITELLM_URL,
|
||||
apiKey: server.agentState?.litellmApiKey ?? LITELLM_KEY,
|
||||
};
|
||||
}
|
||||
|
||||
interface AgentRunBody {
|
||||
question: string;
|
||||
shape?: string;
|
||||
model?: string;
|
||||
persona?: string;
|
||||
workspace?: string;
|
||||
workspaceId?: string;
|
||||
maxSteps?: number;
|
||||
maxRetrievalsPerStep?: number;
|
||||
}
|
||||
|
||||
export const agentRunRoutes: FastifyPluginAsync = async (server) => {
|
||||
/**
|
||||
* Build the LiteLLM-backed llmCall. Mirrors the benchmark/faza-1 caller
|
||||
* pattern (benchmarks/gepa/scripts/faza-1/run-gen-1.ts) — single retry on
|
||||
* transient failures is intentionally omitted here (sidecar callers can
|
||||
* retry at the request level if needed; agent loop runs in tokio task on
|
||||
* Tauri side and will surface error events).
|
||||
*/
|
||||
function makeLlmCall(): LlmCallFn {
|
||||
return async (input: LlmCallInput): Promise<LlmCallResult> => {
|
||||
const started = Date.now();
|
||||
// Deterministic capability-aware routing: a declared-lightweight internal
|
||||
// call (compaction etc.) drops to Haiku-on-proxy (cheap, universal — works
|
||||
// for a FREE user with no local model). privacyRequired keeps the call
|
||||
// on-device and never downgrades to the cloud budget model. The agent's
|
||||
// selected model is used as the on-device candidate when it is local.
|
||||
const currentModel = server.agentState?.currentModel;
|
||||
const localModel = currentModel?.startsWith('ollama/') ? currentModel : undefined;
|
||||
// Fail closed: a privacy-required call with no on-device model must not
|
||||
// touch the cloud — return an error rather than leaking the conversation.
|
||||
if (input.privacyRequired && !localModel) {
|
||||
return {
|
||||
content: '', inTokens: 0, outTokens: 0, costUsd: 0,
|
||||
latencyMs: Date.now() - started,
|
||||
error: 'privacyRequired: no on-device model is configured',
|
||||
};
|
||||
}
|
||||
// Don't reroute a local (Ollama) session's lightweight calls to cloud
|
||||
// Haiku — keep them on-device. The lightweight→cheap-cloud override
|
||||
// applies only when the session is already cloud-backed.
|
||||
const model = resolveModelForClass(input.model, {
|
||||
class: localModel ? undefined : input.class,
|
||||
privacyRequired: input.privacyRequired,
|
||||
lightweightModel: LIGHTWEIGHT_MODEL,
|
||||
localModel,
|
||||
});
|
||||
const isQwen = model.includes('qwen');
|
||||
const payload: Record<string, unknown> = {
|
||||
model,
|
||||
messages: input.messages,
|
||||
max_tokens: input.maxTokens ?? (isQwen ? 16384 : 4096),
|
||||
};
|
||||
if (model.startsWith('claude-opus')) {
|
||||
payload.temperature = 1.0;
|
||||
} else if (model === 'gpt-5.4' || model === 'minimax-m27-via-openrouter') {
|
||||
// omit temperature (model rejects it)
|
||||
} else {
|
||||
payload.temperature = input.temperature ?? 0.3;
|
||||
}
|
||||
if (isQwen && input.thinking !== undefined) {
|
||||
payload.extra_body = { enable_thinking: input.thinking };
|
||||
}
|
||||
|
||||
try {
|
||||
// Read URL/key from LIVE server state per request — picks up the
|
||||
// runtime Anthropic-proxy fallback installed by service.ts.
|
||||
const { url: litellmUrl, apiKey: litellmKey } = resolveLlmEndpoint(server);
|
||||
const resp = await fetch(`${litellmUrl}/chat/completions`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
Authorization: `Bearer ${litellmKey}`,
|
||||
},
|
||||
body: JSON.stringify(payload),
|
||||
});
|
||||
const data = (await resp.json()) as {
|
||||
error?: { message?: string };
|
||||
choices?: Array<{ message?: { content?: string } }>;
|
||||
usage?: { prompt_tokens?: number; completion_tokens?: number; total_cost?: number };
|
||||
};
|
||||
if (data.error) {
|
||||
return {
|
||||
content: '',
|
||||
inTokens: 0,
|
||||
outTokens: 0,
|
||||
costUsd: 0,
|
||||
latencyMs: Date.now() - started,
|
||||
error: data.error.message ?? 'LiteLLM error',
|
||||
};
|
||||
}
|
||||
const content = data.choices?.[0]?.message?.content ?? '';
|
||||
return {
|
||||
content,
|
||||
inTokens: data.usage?.prompt_tokens ?? 0,
|
||||
outTokens: data.usage?.completion_tokens ?? 0,
|
||||
costUsd: data.usage?.total_cost ?? 0,
|
||||
latencyMs: Date.now() - started,
|
||||
};
|
||||
} catch (err) {
|
||||
return {
|
||||
content: '',
|
||||
inTokens: 0,
|
||||
outTokens: 0,
|
||||
costUsd: 0,
|
||||
latencyMs: Date.now() - started,
|
||||
error: err instanceof Error ? err.message : 'LiteLLM fetch failed',
|
||||
};
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/** Build the HybridSearch-backed search fn for the requested workspace. */
|
||||
function makeSearch(workspaceId: string | undefined): RetrievalSearchFn | null {
|
||||
const personalMindDb = server.multiMind?.personal;
|
||||
if (!personalMindDb) return null;
|
||||
const embedder = server.embeddingProvider;
|
||||
if (!embedder) return null;
|
||||
|
||||
const targetMindDb =
|
||||
workspaceId && workspaceId !== 'personal'
|
||||
? server.agentState?.getWorkspaceMindDb?.(workspaceId) ?? personalMindDb
|
||||
: personalMindDb;
|
||||
|
||||
const hybrid = new HybridSearch(targetMindDb, embedder);
|
||||
|
||||
return async ({ query, limit }) => {
|
||||
const hits = await hybrid.search(query, { limit: limit ?? 8 });
|
||||
return {
|
||||
formattedResults:
|
||||
hits.length === 0
|
||||
? ''
|
||||
: hits
|
||||
.map(
|
||||
(s, i) =>
|
||||
`[result ${i + 1}, score ${s.finalScore.toFixed(3)}]\n${s.frame.content}`,
|
||||
)
|
||||
.join('\n\n---\n\n'),
|
||||
resultCount: hits.length,
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
// POST /api/agent/run — shape-aware structured-retrieval agent run with SSE
|
||||
// streaming. Body: { question, shape?, model?, persona?, workspace?,
|
||||
// maxSteps?, maxRetrievalsPerStep? }.
|
||||
server.post<{ Body: AgentRunBody }>('/api/agent/run', async (request, reply) => {
|
||||
const body = request.body ?? ({} as AgentRunBody);
|
||||
const { question, shape, model, persona, maxSteps, maxRetrievalsPerStep } = body;
|
||||
const workspaceId = body.workspace ?? body.workspaceId;
|
||||
|
||||
// Validation BEFORE hijack — once hijacked, reply.status() is a no-op.
|
||||
if (!question || typeof question !== 'string') {
|
||||
return reply.status(400).send({ error: 'question is required' });
|
||||
}
|
||||
|
||||
const search = makeSearch(workspaceId);
|
||||
if (!search) {
|
||||
return reply.status(503).send({
|
||||
error: 'multi-mind or embedding provider not initialized',
|
||||
});
|
||||
}
|
||||
|
||||
// Validate shape via REGISTRY membership. Unknown shapes log + fall back
|
||||
// to the model-alias-derived default (selectShape's normal behavior),
|
||||
// so a user with a stale UI cache + a removed shape still gets a working
|
||||
// run instead of a 400.
|
||||
let shapeOverride: string | undefined;
|
||||
if (shape) {
|
||||
const available = listShapes();
|
||||
if (available.includes(shape)) {
|
||||
shapeOverride = shape;
|
||||
} else {
|
||||
request.log.warn(
|
||||
{ shape, available },
|
||||
'[agent-run] unknown shape requested, falling back to model-default',
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Hijack response for SSE.
|
||||
await reply.hijack();
|
||||
const raw = reply.raw;
|
||||
raw.writeHead(200, {
|
||||
'Content-Type': 'text/event-stream',
|
||||
'Cache-Control': 'no-cache',
|
||||
Connection: 'keep-alive',
|
||||
'X-Accel-Buffering': 'no',
|
||||
});
|
||||
|
||||
const sendEvent = (event: string, data: unknown): void => {
|
||||
try {
|
||||
raw.write(`event: ${event}\n`);
|
||||
raw.write(`data: ${JSON.stringify(data)}\n\n`);
|
||||
} catch {
|
||||
// Connection closed by client — agent loop will continue but events
|
||||
// are dropped. This is fine; final state still lands in the trace.
|
||||
}
|
||||
};
|
||||
|
||||
sendEvent('started', {
|
||||
shape: shapeOverride ?? '(model-default)',
|
||||
shapeRequested: shape ?? null,
|
||||
shapeRecognized: shapeOverride !== undefined || shape === undefined,
|
||||
model: model ?? DEFAULT_MODEL,
|
||||
});
|
||||
|
||||
try {
|
||||
const result = await runRetrievalAgentLoop({
|
||||
modelAlias: model ?? DEFAULT_MODEL,
|
||||
persona: persona ?? DEFAULT_PERSONA,
|
||||
question,
|
||||
llmCall: makeLlmCall(),
|
||||
search,
|
||||
promptShapeOverride: shapeOverride,
|
||||
maxSteps: maxSteps ?? DEFAULT_MAX_STEPS,
|
||||
maxRetrievalsPerStep: maxRetrievalsPerStep ?? 8,
|
||||
onProgress: (event) => sendEvent('progress', event),
|
||||
});
|
||||
|
||||
sendEvent('finalized', {
|
||||
rawResponse: result.rawResponse,
|
||||
normalizedResponse: result.normalizedResponse,
|
||||
promptShapeName: result.promptShapeName,
|
||||
stepsTaken: result.stepsTaken,
|
||||
retrievalCalls: result.retrievalCalls,
|
||||
loopExhausted: result.loopExhausted,
|
||||
totalTokensIn: result.totalTokensIn,
|
||||
totalTokensOut: result.totalTokensOut,
|
||||
totalCostUsd: result.totalCostUsd,
|
||||
totalLatencyMs: result.totalLatencyMs,
|
||||
});
|
||||
} catch (err) {
|
||||
sendEvent('error', {
|
||||
error: err instanceof Error ? err.message : 'agent run failed',
|
||||
});
|
||||
} finally {
|
||||
sendEvent('done', { ok: true });
|
||||
try {
|
||||
raw.end();
|
||||
} catch {
|
||||
// already closed
|
||||
}
|
||||
}
|
||||
});
|
||||
};
|
||||
Reference in New Issue
Block a user