feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
import { describe, it, expect, vi } from 'vitest';
|
|
|
|
|
import { ChatService, MAX_ITERATIONS, TOOL_NAME_SEPARATOR, type ChatToolDispatcher } from '../src/services/chat.service.js';
|
|
|
|
|
import type { AgentService } from '../src/services/agent.service.js';
|
|
|
|
|
import type { LlmService } from '../src/services/llm.service.js';
|
|
|
|
|
import type { LlmAdapterRegistry } from '../src/services/llm/dispatcher.js';
|
|
|
|
|
import type { LlmAdapter, NonStreamingResult, InferContext } from '../src/services/llm/types.js';
|
|
|
|
|
import type { IChatRepository } from '../src/repositories/chat.repository.js';
|
|
|
|
|
import type { IPromptRepository } from '../src/repositories/prompt.repository.js';
|
2026-04-26 19:27:59 +01:00
|
|
|
import type { IPersonalityRepository } from '../src/repositories/personality.repository.js';
|
|
|
|
|
import type { ChatMessage, ChatThread, Prompt, Personality, PersonalityPrompt } from '@prisma/client';
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
|
|
|
|
|
const NOW = new Date();
|
|
|
|
|
|
|
|
|
|
function mockChatRepo(): IChatRepository & { _msgs: ChatMessage[]; _threads: ChatThread[] } {
|
|
|
|
|
const msgs: ChatMessage[] = [];
|
|
|
|
|
const threads: ChatThread[] = [];
|
|
|
|
|
let idCounter = 1;
|
|
|
|
|
|
|
|
|
|
return {
|
|
|
|
|
_msgs: msgs,
|
|
|
|
|
_threads: threads,
|
|
|
|
|
createThread: vi.fn(async ({ agentId, ownerId, title }) => {
|
|
|
|
|
const t: ChatThread = {
|
|
|
|
|
id: `thread-${String(idCounter++)}`,
|
|
|
|
|
agentId,
|
|
|
|
|
ownerId,
|
|
|
|
|
title: title ?? '',
|
|
|
|
|
lastTurnAt: NOW,
|
|
|
|
|
createdAt: NOW,
|
|
|
|
|
updatedAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
threads.push(t);
|
|
|
|
|
return t;
|
|
|
|
|
}),
|
|
|
|
|
findThread: vi.fn(async (id: string) => threads.find((t) => t.id === id) ?? null),
|
|
|
|
|
listThreadsByAgent: vi.fn(async (agentId: string) => threads.filter((t) => t.agentId === agentId)),
|
|
|
|
|
listMessages: vi.fn(async (threadId: string) =>
|
|
|
|
|
msgs.filter((m) => m.threadId === threadId).sort((a, b) => a.turnIndex - b.turnIndex)),
|
|
|
|
|
appendMessage: vi.fn(async (input) => {
|
|
|
|
|
const turnIndex = input.turnIndex ?? msgs.filter((m) => m.threadId === input.threadId).length;
|
|
|
|
|
const m: ChatMessage = {
|
|
|
|
|
id: `msg-${String(idCounter++)}`,
|
|
|
|
|
threadId: input.threadId,
|
|
|
|
|
turnIndex,
|
|
|
|
|
role: input.role,
|
|
|
|
|
content: input.content,
|
|
|
|
|
toolCalls: (input.toolCalls ?? null) as ChatMessage['toolCalls'],
|
|
|
|
|
toolCallId: input.toolCallId ?? null,
|
|
|
|
|
status: input.status ?? 'complete',
|
|
|
|
|
createdAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
msgs.push(m);
|
|
|
|
|
return m;
|
|
|
|
|
}),
|
|
|
|
|
updateStatus: vi.fn(async (id: string, status) => {
|
|
|
|
|
const m = msgs.find((x) => x.id === id);
|
|
|
|
|
if (!m) throw new Error('not found');
|
|
|
|
|
m.status = status;
|
|
|
|
|
return m;
|
|
|
|
|
}),
|
|
|
|
|
markPendingAsError: vi.fn(async (threadId: string) => {
|
|
|
|
|
let n = 0;
|
|
|
|
|
for (const m of msgs) {
|
|
|
|
|
if (m.threadId === threadId && m.status === 'pending') {
|
|
|
|
|
m.status = 'error';
|
|
|
|
|
n += 1;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return n;
|
|
|
|
|
}),
|
|
|
|
|
touchThread: vi.fn(async () => undefined),
|
|
|
|
|
nextTurnIndex: vi.fn(async (threadId: string) =>
|
|
|
|
|
msgs.filter((m) => m.threadId === threadId).length),
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function mockPromptRepo(rows: Prompt[] = []): IPromptRepository {
|
|
|
|
|
return {
|
|
|
|
|
findAll: vi.fn(async () => rows),
|
2026-04-26 19:27:59 +01:00
|
|
|
findGlobal: vi.fn(async () => rows.filter((p) => p.projectId === null && p.agentId === null)),
|
|
|
|
|
findByAgent: vi.fn(async (agentId: string) => rows.filter((p) => p.agentId === agentId)),
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
findById: vi.fn(async (id: string) => rows.find((p) => p.id === id) ?? null),
|
|
|
|
|
findByNameAndProject: vi.fn(async () => null),
|
2026-04-26 19:27:59 +01:00
|
|
|
findByNameAndAgent: vi.fn(async () => null),
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
create: vi.fn(),
|
|
|
|
|
update: vi.fn(),
|
|
|
|
|
delete: vi.fn(),
|
|
|
|
|
} as unknown as IPromptRepository;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function mockTools(impl: Partial<ChatToolDispatcher> = {}): ChatToolDispatcher {
|
|
|
|
|
return {
|
|
|
|
|
listTools: impl.listTools ?? vi.fn(async () => []),
|
|
|
|
|
callTool: impl.callTool ?? vi.fn(async () => ({ ok: true })),
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
2026-04-26 19:27:59 +01:00
|
|
|
function mockAgents(opts: { defaultPersonality?: { id: string; name: string } | null } = {}): AgentService {
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
return {
|
|
|
|
|
getByName: vi.fn(async (name: string) => ({
|
|
|
|
|
id: `agent-${name}`,
|
|
|
|
|
name,
|
|
|
|
|
description: 'desc',
|
|
|
|
|
systemPrompt: 'You are a helpful agent.',
|
|
|
|
|
llm: { id: 'llm-1', name: 'qwen3-thinking' },
|
|
|
|
|
project: name === 'no-project'
|
|
|
|
|
? null
|
|
|
|
|
: { id: 'proj-1', name: 'mcpctl-dev' },
|
2026-04-26 19:27:59 +01:00
|
|
|
defaultPersonality: opts.defaultPersonality ?? null,
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
proxyModelName: null,
|
|
|
|
|
defaultParams: { temperature: 0.5 },
|
|
|
|
|
extras: {},
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
version: 1,
|
|
|
|
|
createdAt: NOW,
|
|
|
|
|
updatedAt: NOW,
|
|
|
|
|
})),
|
|
|
|
|
} as unknown as AgentService;
|
|
|
|
|
}
|
|
|
|
|
|
feat(db+mcpd): Agent lifecycle + chat.service kind=virtual branch (v3 Stage 1)
Two pieces of v3 plumbing — schema + the latent v1 chat.service bug.
Schema (db):
- Agent gains kind/providerSessionId/lastHeartbeatAt/status/inactiveSince
mirroring Llm's v1 lifecycle. Reuses LlmKind / LlmStatus enums; no
new types. Existing rows backfill kind=public/status=active so v1
CRUD is unaffected.
- @@index([kind, status]) for the GC sweep, @@index([providerSessionId])
for disconnect-cascade lookups.
- 4 new prisma-level tests cover defaults, persisting virtual fields,
the (kind, status) GC index, and providerSessionId lookups.
Total agent-schema tests: 20/20.
chat.service (mcpd) — fixes the v1 latent bug:
- LlmView's kind is now plumbed through prepareContext as ctx.llmKind.
- Two new private helpers, runOneInference / streamInference, branch
on ctx.llmKind: 'public' goes through the existing adapter
registry, 'virtual' relays through VirtualLlmService.enqueueInferTask
(mirrors the route-handler branch from v1 Stage 3).
- Streaming bridges VirtualLlmService's onChunk callback API to an
async iterator via a small queue + wake pattern.
- ChatService gains an optional virtualLlms constructor parameter;
main.ts wires it in. Older test wirings without it raise a clear
"virtualLlms dispatcher not wired" error when the row is virtual,
rather than silently falling through to the public path against an
empty URL.
This unblocks any Agent (public OR future v3-virtual) pinned to a
kind=virtual Llm. Pre-this-stage, those agents 502'd against the
empty url field.
Tests: 4 new chat-service-virtual-llm.test.ts cover the relay path
non-streaming, streaming, missing-dispatcher error, and rejection
surfacing. mcpd suite: 841/841 (was 833, +8 across stages 1+v3-Stage-1).
Workspace: 2054/2054 across 153 files.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-27 16:55:02 +01:00
|
|
|
function mockLlms(opts: { kind?: 'public' | 'virtual' } = {}): LlmService {
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
return {
|
|
|
|
|
getByName: vi.fn(async (name: string) => ({
|
|
|
|
|
id: 'llm-1', name, type: 'openai', model: 'qwen3-thinking',
|
|
|
|
|
url: '', tier: 'fast', description: '',
|
|
|
|
|
apiKeyRef: null, extraConfig: {},
|
feat(db+mcpd): Agent lifecycle + chat.service kind=virtual branch (v3 Stage 1)
Two pieces of v3 plumbing — schema + the latent v1 chat.service bug.
Schema (db):
- Agent gains kind/providerSessionId/lastHeartbeatAt/status/inactiveSince
mirroring Llm's v1 lifecycle. Reuses LlmKind / LlmStatus enums; no
new types. Existing rows backfill kind=public/status=active so v1
CRUD is unaffected.
- @@index([kind, status]) for the GC sweep, @@index([providerSessionId])
for disconnect-cascade lookups.
- 4 new prisma-level tests cover defaults, persisting virtual fields,
the (kind, status) GC index, and providerSessionId lookups.
Total agent-schema tests: 20/20.
chat.service (mcpd) — fixes the v1 latent bug:
- LlmView's kind is now plumbed through prepareContext as ctx.llmKind.
- Two new private helpers, runOneInference / streamInference, branch
on ctx.llmKind: 'public' goes through the existing adapter
registry, 'virtual' relays through VirtualLlmService.enqueueInferTask
(mirrors the route-handler branch from v1 Stage 3).
- Streaming bridges VirtualLlmService's onChunk callback API to an
async iterator via a small queue + wake pattern.
- ChatService gains an optional virtualLlms constructor parameter;
main.ts wires it in. Older test wirings without it raise a clear
"virtualLlms dispatcher not wired" error when the row is virtual,
rather than silently falling through to the public path against an
empty URL.
This unblocks any Agent (public OR future v3-virtual) pinned to a
kind=virtual Llm. Pre-this-stage, those agents 502'd against the
empty url field.
Tests: 4 new chat-service-virtual-llm.test.ts cover the relay path
non-streaming, streaming, missing-dispatcher error, and rejection
surfacing. mcpd suite: 841/841 (was 833, +8 across stages 1+v3-Stage-1).
Workspace: 2054/2054 across 153 files.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-27 16:55:02 +01:00
|
|
|
kind: opts.kind ?? 'public',
|
|
|
|
|
status: 'active',
|
|
|
|
|
lastHeartbeatAt: null,
|
|
|
|
|
inactiveSince: null,
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
version: 1, createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
})),
|
|
|
|
|
resolveApiKey: vi.fn(async () => 'fake-key'),
|
|
|
|
|
} as unknown as LlmService;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Adapter that yields a scripted sequence of canned responses, one per call. */
|
|
|
|
|
function scriptedAdapter(responses: NonStreamingResult[]): LlmAdapter {
|
|
|
|
|
let i = 0;
|
|
|
|
|
return {
|
|
|
|
|
kind: 'scripted',
|
|
|
|
|
infer: vi.fn(async (_ctx: InferContext) => {
|
|
|
|
|
const r = responses[i] ?? responses[responses.length - 1];
|
|
|
|
|
i += 1;
|
|
|
|
|
if (r === undefined) throw new Error('no scripted response');
|
|
|
|
|
return r;
|
|
|
|
|
}),
|
|
|
|
|
stream: async function*(_ctx: InferContext) {
|
|
|
|
|
yield { data: '[DONE]', done: true };
|
|
|
|
|
},
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function adapterRegistry(adapter: LlmAdapter): LlmAdapterRegistry {
|
|
|
|
|
return { get: () => adapter } as unknown as LlmAdapterRegistry;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function chatCompletion(content: string): NonStreamingResult {
|
|
|
|
|
return {
|
|
|
|
|
status: 200,
|
|
|
|
|
body: {
|
|
|
|
|
id: 'cmpl-1',
|
|
|
|
|
object: 'chat.completion',
|
|
|
|
|
choices: [{ index: 0, message: { role: 'assistant', content }, finish_reason: 'stop' }],
|
|
|
|
|
},
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function toolCall(name: string, args: Record<string, unknown>): NonStreamingResult {
|
|
|
|
|
return {
|
|
|
|
|
status: 200,
|
|
|
|
|
body: {
|
|
|
|
|
id: 'cmpl-1',
|
|
|
|
|
object: 'chat.completion',
|
|
|
|
|
choices: [{
|
|
|
|
|
index: 0,
|
|
|
|
|
message: {
|
|
|
|
|
role: 'assistant',
|
|
|
|
|
content: '',
|
|
|
|
|
tool_calls: [{
|
|
|
|
|
id: `call-${name}`,
|
|
|
|
|
type: 'function',
|
|
|
|
|
function: { name, arguments: JSON.stringify(args) },
|
|
|
|
|
}],
|
|
|
|
|
},
|
|
|
|
|
finish_reason: 'tool_calls',
|
|
|
|
|
}],
|
|
|
|
|
},
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
describe('ChatService', () => {
|
|
|
|
|
it('plain text turn — persists user + assistant rows and returns the reply', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('hello back')]);
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
const result = await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
expect(result.assistant).toBe('hello back');
|
|
|
|
|
const stored = chatRepo._msgs.filter((m) => m.threadId === result.threadId);
|
|
|
|
|
expect(stored.map((m) => m.role)).toEqual(['user', 'assistant']);
|
|
|
|
|
expect(stored[1]?.status).toBe('complete');
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('runs a full tool-use round-trip and ends with a text reply', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [{
|
|
|
|
|
name: `grafana${TOOL_NAME_SEPARATOR}query`,
|
|
|
|
|
description: 'query grafana',
|
|
|
|
|
parameters: { type: 'object', properties: {} },
|
|
|
|
|
}]),
|
|
|
|
|
callTool: vi.fn(async () => ({ rows: [{ value: 42 }] })),
|
|
|
|
|
});
|
|
|
|
|
const adapter = scriptedAdapter([
|
|
|
|
|
toolCall(`grafana${TOOL_NAME_SEPARATOR}query`, { q: 'cpu' }),
|
|
|
|
|
chatCompletion('the answer is 42'),
|
|
|
|
|
]);
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
const result = await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'what is cpu?',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
expect(result.assistant).toBe('the answer is 42');
|
|
|
|
|
expect(tools.callTool).toHaveBeenCalledWith({
|
|
|
|
|
projectId: 'proj-1',
|
|
|
|
|
serverName: 'grafana',
|
|
|
|
|
toolName: 'query',
|
|
|
|
|
args: { q: 'cpu' },
|
|
|
|
|
});
|
|
|
|
|
const stored = chatRepo._msgs.filter((m) => m.threadId === result.threadId);
|
|
|
|
|
expect(stored.map((m) => m.role)).toEqual(['user', 'assistant', 'tool', 'assistant']);
|
|
|
|
|
// No `pending` rows leaked.
|
|
|
|
|
expect(stored.every((m) => m.status === 'complete')).toBe(true);
|
|
|
|
|
// Tool turn's toolCallId links back.
|
|
|
|
|
const toolTurn = stored.find((m) => m.role === 'tool');
|
|
|
|
|
expect(toolTurn?.toolCallId).toBe(`call-grafana${TOOL_NAME_SEPARATOR}query`);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('caps the loop at MAX_ITERATIONS when the model never settles', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [{
|
|
|
|
|
name: `g${TOOL_NAME_SEPARATOR}t`,
|
|
|
|
|
description: '',
|
|
|
|
|
parameters: { type: 'object' },
|
|
|
|
|
}]),
|
|
|
|
|
callTool: vi.fn(async () => ({})),
|
|
|
|
|
});
|
|
|
|
|
// Always return a tool_call → the loop never reaches a terminal turn.
|
|
|
|
|
const adapter = scriptedAdapter([toolCall(`g${TOOL_NAME_SEPARATOR}t`, {})]);
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await expect(svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'loop forever',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
})).rejects.toThrow(new RegExp(`exceeded ${String(MAX_ITERATIONS)}`));
|
|
|
|
|
|
|
|
|
|
// After failure, no row should remain `pending`.
|
|
|
|
|
expect(chatRepo._msgs.every((m) => m.status !== 'pending')).toBe(true);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('flips pending rows to error when the adapter throws mid-loop', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [{
|
|
|
|
|
name: `g${TOOL_NAME_SEPARATOR}t`, description: '', parameters: {},
|
|
|
|
|
}]),
|
|
|
|
|
callTool: vi.fn(async () => ({})),
|
|
|
|
|
});
|
|
|
|
|
const adapter: LlmAdapter = {
|
|
|
|
|
kind: 'fail-after-one',
|
|
|
|
|
infer: vi.fn()
|
|
|
|
|
.mockResolvedValueOnce(toolCall(`g${TOOL_NAME_SEPARATOR}t`, {}))
|
|
|
|
|
.mockRejectedValueOnce(new Error('upstream blew up')),
|
|
|
|
|
stream: async function*() { yield { data: '[DONE]', done: true }; },
|
|
|
|
|
};
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await expect(svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'go',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
})).rejects.toThrow('upstream blew up');
|
|
|
|
|
|
|
|
|
|
expect(chatRepo._msgs.some((m) => m.status === 'error')).toBe(false);
|
|
|
|
|
expect(chatRepo._msgs.every((m) => m.status !== 'pending')).toBe(true);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('merges per-call params over agent.defaultParams (override wins)', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
params: { temperature: 0.9, max_tokens: 256 },
|
|
|
|
|
});
|
|
|
|
|
const ctx = inferSpy.mock.calls[0][0] as InferContext;
|
|
|
|
|
expect(ctx.body.temperature).toBe(0.9);
|
|
|
|
|
expect(ctx.body.max_tokens).toBe(256);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('forwards `extra` keys into the body for provider-specific knobs', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
params: { extra: { metadata: { user_id: 'abc' }, repetition_penalty: 1.05 } },
|
|
|
|
|
});
|
|
|
|
|
const ctx = inferSpy.mock.calls[0][0] as InferContext;
|
|
|
|
|
expect((ctx.body as Record<string, unknown>)['repetition_penalty']).toBe(1.05);
|
|
|
|
|
expect((ctx.body as Record<string, unknown>)['metadata']).toEqual({ user_id: 'abc' });
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('builds a system block from agent.systemPrompt + project prompts (priority desc)', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const prompts: Prompt[] = [
|
|
|
|
|
{
|
|
|
|
|
id: 'p1', name: 'low', content: 'LOW prompt',
|
|
|
|
|
projectId: 'proj-1', priority: 1, summary: null, chapters: null,
|
|
|
|
|
linkTarget: null, version: 1, createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
id: 'p2', name: 'high', content: 'HIGH prompt',
|
|
|
|
|
projectId: 'proj-1', priority: 9, summary: null, chapters: null,
|
|
|
|
|
linkTarget: null, version: 1, createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
},
|
|
|
|
|
];
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(prompts), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({ agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1' });
|
|
|
|
|
const ctx = inferSpy.mock.calls[0][0] as InferContext;
|
|
|
|
|
const sys = ctx.body.messages.find((m) => m.role === 'system');
|
|
|
|
|
expect(typeof sys?.content).toBe('string');
|
|
|
|
|
const text = sys?.content as string;
|
|
|
|
|
// High-priority prompt comes before low-priority.
|
|
|
|
|
expect(text.indexOf('HIGH prompt')).toBeLessThan(text.indexOf('LOW prompt'));
|
|
|
|
|
// Agent's own system prompt leads.
|
|
|
|
|
expect(text.indexOf('You are a helpful agent.')).toBeLessThan(text.indexOf('HIGH prompt'));
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('refuses tool calls when the agent has no project attached', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([toolCall(`x${TOOL_NAME_SEPARATOR}y`, {})]);
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [{ name: `x${TOOL_NAME_SEPARATOR}y`, description: '', parameters: {} }]),
|
|
|
|
|
});
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
await expect(svc.chat({
|
|
|
|
|
agentName: 'no-project',
|
|
|
|
|
userMessage: 'go',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
})).rejects.toThrow(/Project/);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('honours tools_allowlist (filters tools before sending to adapter)', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [
|
|
|
|
|
{ name: `s1${TOOL_NAME_SEPARATOR}a`, description: '', parameters: {} },
|
|
|
|
|
{ name: `s1${TOOL_NAME_SEPARATOR}b`, description: '', parameters: {} },
|
|
|
|
|
]),
|
|
|
|
|
});
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
params: { tools_allowlist: [`s1${TOOL_NAME_SEPARATOR}a`] },
|
|
|
|
|
});
|
|
|
|
|
const ctx = inferSpy.mock.calls[0][0] as InferContext;
|
|
|
|
|
expect(ctx.body.tools).toHaveLength(1);
|
|
|
|
|
expect(ctx.body.tools?.[0]?.function.name).toBe(`s1${TOOL_NAME_SEPARATOR}a`);
|
|
|
|
|
});
|
fix(agents): close gaps from /gstack-review
P1 — thread reads now enforce ownership
========================================
chat.service.ts / routes/agent-chat.ts
GET /api/v1/threads/:id/messages was previously RBAC-mapped to
view:agents (no resourceName scope) with the route comment promising
"service-level owner check enforces fine-grained access" — but the
service didn't actually check. Any caller with view:agents could read
another user's thread by guessing/learning the threadId. CUIDs are
hard to brute-force but they leak: SSE `final` chunks, agents-plugin
`_meta.threadId`, and several response bodies surface them. Now
ChatService.listMessages(threadId, ownerId) loads the thread, returns
404 (not 403, to avoid id-enumeration via differential status codes)
if ownerId doesn't match. Regression test in chat-service.test.ts
covers Alice/Bob isolation + nonexistent-thread same-shape 404.
P2 — AgentChatRequestSchema strict mode
========================================
validation/agent.schema.ts
`.merge()` does NOT inherit `.strict()` from AgentChatParamsSchema.
Typo'd fields (e.g. `temprature`) silently fell through and the agent
silently used the default — debuggable only by reading the LLM call
payload. Re-applied `.strict()` on the merged schema.
P2 — per-agent maxIterations override + clamp
==============================================
chat.service.ts
Loop cap was a hard-coded module constant (12), wrong for both
research-style agents (need higher) and cheap-probe agents (could opt
lower). Now reads `agent.extras.maxIterations`, clamps 1..50, falls
back to 12 default. The clamp is the soft-DoS guard: a hostile agent
definition with `maxIterations:1000000` can't burn unbounded LLM calls
per request. Both chat() and chatStream() use ctx.maxIterations now.
Regression test covers low-cap override (rejects with `exceeded 2`)
and hostile-value clamp (rejects with `exceeded 50`).
P3 — SSE write to closed socket
================================
routes/agent-chat.ts
When the upstream adapter throws after some chunks were already
written AND the client disconnected, the catch block tried to flush
more chunks to a closed socket. Without an `on('error')` handler
Node emits unhandled error events; once Pino is wired to alerts
this'd page on every disconnect-mid-stream. writeSseChunk now
checks `reply.raw.destroyed || writableEnded` before write.
P3 — BACKEND_TOKEN_DEAD preserves original stack
=================================================
services/secret-backend-rotator.service.ts
When wrapping mintRoleToken/lookupSelf failures as
BACKEND_TOKEN_DEAD, the new Error() discarded the original throw —
hard to tell whether the inner failure was a network blip vs an
OpenBao API mismatch vs DNS. Now uses `new Error(msg, { cause: err })`
so the inner stack survives.
P3 — .gitignore .claude/scheduled_tasks.lock
=============================================
This persisted state file was leaking into every `git status`.
Tests
=====
mcpd 761/761 (+2 regression tests). mcplocal 715/715. cli 430/430.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 23:53:19 +01:00
|
|
|
|
feat(chat): surface reasoning_content as `thinking` chunks; fix --no-stream timeout
Reasoning models (qwen3-thinking, deepseek-reasoner, OpenAI o1 family) emit
their scratchpad as `delta.reasoning_content` (or `delta.reasoning`,
or `delta.provider_specific_fields.reasoning_content` when LiteLLM passes
through from vLLM) — separate from `delta.content`. Before this commit
mcpd's parseStreamingChunk only watched `content`, so the model's 30-90s
reasoning phase looked like dead air to the REPL: streaming connection
open, no chunks, no progress. Caught during the agents-feature shakedown
when qwen3-thinking sat silent for 90s on a docmost__list_pages call.
mcpd
====
chat.service.ts
- parseStreamingChunk extracts a `reasoningDelta` from the chunk body,
accepting all four spellings (reasoning_content / reasoning /
provider_specific_fields.{reasoning_content,reasoning}). Future
providers can add their own field names by extending the
fallback chain.
- chatStream yields `{ type: 'thinking', delta }` chunks as reasoning
arrives, alongside the existing `{ type: 'text', delta }` for content.
- Reasoning is intentionally NOT persisted to the thread. It's the
model's scratchpad, not part of the conversation. Subsequent turns
don't see it.
- Adds 'thinking' to the ChatStreamChunk.type union.
CLI
===
chat.ts
- streamOnce handles 'thinking' chunks: writes them dim+italic to
stderr (ANSI 2;3m) so the model's reasoning visually flows like a
quote block while the final answer streams to stdout. Plain text
when stderr isn't a TTY (pipe to file → no escape codes leak).
- chatRequestNonStream replaces the shared ApiClient.post() for the
--no-stream path. ApiClient defaults to a 10s timeout, way too tight
for any chat that calls a tool: LLM round + tool dispatch + LLM
summary easily exceeds 10s. The new helper uses the same 600s timeout
the streaming path has been using all along.
Tests:
chat-service.test.ts (+2):
- reasoning_content deltas surface as `thinking` chunks (not text);
reasoning is NOT persisted to the assistant turn's content.
- LiteLLM's provider_specific_fields.reasoning_content shape parses
identically to the vendor-native shape.
mcpd 777/777, cli 430/430.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-26 17:04:01 +01:00
|
|
|
// Regression: reasoning_content (qwen3-thinking, deepseek-reasoner, o1)
|
|
|
|
|
// streams as `thinking` chunks, separate from `text`.
|
|
|
|
|
// Without this, the model's 30-90s reasoning phase looks like dead air to
|
|
|
|
|
// the REPL — caught by user feedback during the agents-feature shakedown.
|
|
|
|
|
it('chatStream surfaces reasoning_content deltas as `thinking` chunks', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
// Adapter that yields a sequence of openai-format chunks: 2 reasoning
|
|
|
|
|
// deltas, then 1 content delta, then [DONE].
|
|
|
|
|
const adapter: LlmAdapter = {
|
|
|
|
|
kind: 'scripted-thinking',
|
|
|
|
|
infer: vi.fn(),
|
|
|
|
|
stream: async function*() {
|
|
|
|
|
yield { data: JSON.stringify({ choices: [{ delta: { reasoning_content: 'Let me think... ' }, finish_reason: null }] }) };
|
|
|
|
|
yield { data: JSON.stringify({ choices: [{ delta: { reasoning_content: 'OK, ready.' }, finish_reason: null }] }) };
|
|
|
|
|
yield { data: JSON.stringify({ choices: [{ delta: { content: 'DONE' }, finish_reason: 'stop' }] }) };
|
|
|
|
|
yield { data: '[DONE]', done: true };
|
|
|
|
|
},
|
|
|
|
|
};
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
const chunks: Array<{ type: string; delta?: string }> = [];
|
|
|
|
|
for await (const chunk of svc.chatStream({
|
|
|
|
|
agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1',
|
|
|
|
|
})) {
|
|
|
|
|
chunks.push({ type: chunk.type, delta: chunk.delta });
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Expect: 2 thinking + 1 text + 1 final
|
|
|
|
|
expect(chunks.filter((c) => c.type === 'thinking').map((c) => c.delta))
|
|
|
|
|
.toEqual(['Let me think... ', 'OK, ready.']);
|
|
|
|
|
expect(chunks.filter((c) => c.type === 'text').map((c) => c.delta)).toEqual(['DONE']);
|
|
|
|
|
expect(chunks.find((c) => c.type === 'final')).toBeDefined();
|
|
|
|
|
|
|
|
|
|
// Reasoning is NOT persisted to the thread — only assistant content.
|
|
|
|
|
const assistantTurn = chatRepo._msgs.find((m) => m.role === 'assistant');
|
|
|
|
|
expect(assistantTurn?.content).toBe('DONE');
|
|
|
|
|
expect(assistantTurn?.content).not.toContain('Let me think');
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// Regression: provider_specific_fields.reasoning_content shape (LiteLLM
|
|
|
|
|
// passthrough from vLLM) is also recognized.
|
|
|
|
|
it('chatStream recognizes LiteLLM provider_specific_fields.reasoning_content', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter: LlmAdapter = {
|
|
|
|
|
kind: 'scripted-litellm',
|
|
|
|
|
infer: vi.fn(),
|
|
|
|
|
stream: async function*() {
|
|
|
|
|
yield { data: JSON.stringify({ choices: [{ delta: { provider_specific_fields: { reasoning_content: 'thinking via litellm...' } }, finish_reason: null }] }) };
|
|
|
|
|
yield { data: JSON.stringify({ choices: [{ delta: { content: 'ok' }, finish_reason: 'stop' }] }) };
|
|
|
|
|
yield { data: '[DONE]', done: true };
|
|
|
|
|
},
|
|
|
|
|
};
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
const chunks: Array<{ type: string; delta?: string }> = [];
|
|
|
|
|
for await (const chunk of svc.chatStream({
|
|
|
|
|
agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1',
|
|
|
|
|
})) {
|
|
|
|
|
chunks.push({ type: chunk.type, delta: chunk.delta });
|
|
|
|
|
}
|
|
|
|
|
expect(chunks.filter((c) => c.type === 'thinking').map((c) => c.delta))
|
|
|
|
|
.toEqual(['thinking via litellm...']);
|
|
|
|
|
});
|
|
|
|
|
|
fix(agents): close gaps from /gstack-review
P1 — thread reads now enforce ownership
========================================
chat.service.ts / routes/agent-chat.ts
GET /api/v1/threads/:id/messages was previously RBAC-mapped to
view:agents (no resourceName scope) with the route comment promising
"service-level owner check enforces fine-grained access" — but the
service didn't actually check. Any caller with view:agents could read
another user's thread by guessing/learning the threadId. CUIDs are
hard to brute-force but they leak: SSE `final` chunks, agents-plugin
`_meta.threadId`, and several response bodies surface them. Now
ChatService.listMessages(threadId, ownerId) loads the thread, returns
404 (not 403, to avoid id-enumeration via differential status codes)
if ownerId doesn't match. Regression test in chat-service.test.ts
covers Alice/Bob isolation + nonexistent-thread same-shape 404.
P2 — AgentChatRequestSchema strict mode
========================================
validation/agent.schema.ts
`.merge()` does NOT inherit `.strict()` from AgentChatParamsSchema.
Typo'd fields (e.g. `temprature`) silently fell through and the agent
silently used the default — debuggable only by reading the LLM call
payload. Re-applied `.strict()` on the merged schema.
P2 — per-agent maxIterations override + clamp
==============================================
chat.service.ts
Loop cap was a hard-coded module constant (12), wrong for both
research-style agents (need higher) and cheap-probe agents (could opt
lower). Now reads `agent.extras.maxIterations`, clamps 1..50, falls
back to 12 default. The clamp is the soft-DoS guard: a hostile agent
definition with `maxIterations:1000000` can't burn unbounded LLM calls
per request. Both chat() and chatStream() use ctx.maxIterations now.
Regression test covers low-cap override (rejects with `exceeded 2`)
and hostile-value clamp (rejects with `exceeded 50`).
P3 — SSE write to closed socket
================================
routes/agent-chat.ts
When the upstream adapter throws after some chunks were already
written AND the client disconnected, the catch block tried to flush
more chunks to a closed socket. Without an `on('error')` handler
Node emits unhandled error events; once Pino is wired to alerts
this'd page on every disconnect-mid-stream. writeSseChunk now
checks `reply.raw.destroyed || writableEnded` before write.
P3 — BACKEND_TOKEN_DEAD preserves original stack
=================================================
services/secret-backend-rotator.service.ts
When wrapping mintRoleToken/lookupSelf failures as
BACKEND_TOKEN_DEAD, the new Error() discarded the original throw —
hard to tell whether the inner failure was a network blip vs an
OpenBao API mismatch vs DNS. Now uses `new Error(msg, { cause: err })`
so the inner stack survives.
P3 — .gitignore .claude/scheduled_tasks.lock
=============================================
This persisted state file was leaking into every `git status`.
Tests
=====
mcpd 761/761 (+2 regression tests). mcplocal 715/715. cli 430/430.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 23:53:19 +01:00
|
|
|
// Regression: per-agent maxIterations override + clamp.
|
|
|
|
|
// Found by /gstack-review on 2026-04-25.
|
|
|
|
|
// Without the clamp, a hostile agent definition with `extras.maxIterations:1000000`
|
|
|
|
|
// could spin the loop into a near-infinite tool-call burn.
|
|
|
|
|
it('per-agent extras.maxIterations clamps below default and refuses absurd values', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const tools = mockTools({
|
|
|
|
|
listTools: vi.fn(async () => [{
|
|
|
|
|
name: `g${TOOL_NAME_SEPARATOR}t`, description: '', parameters: {},
|
|
|
|
|
}]),
|
|
|
|
|
callTool: vi.fn(async () => ({})),
|
|
|
|
|
});
|
|
|
|
|
// Agent with maxIterations=2 — only 2 tool-call rounds allowed before bail.
|
|
|
|
|
const agentsLowCap = {
|
|
|
|
|
getByName: vi.fn(async () => ({
|
|
|
|
|
id: 'agent-low', name: 'low', description: '', systemPrompt: '',
|
|
|
|
|
llm: { id: 'llm-1', name: 'qwen3-thinking' },
|
|
|
|
|
project: { id: 'proj-1', name: 'mcpctl-dev' },
|
|
|
|
|
proxyModelName: null, defaultParams: {},
|
|
|
|
|
extras: { maxIterations: 2 },
|
|
|
|
|
ownerId: 'owner-1', version: 1, createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
})),
|
|
|
|
|
} as unknown as AgentService;
|
|
|
|
|
const adapter = scriptedAdapter([toolCall(`g${TOOL_NAME_SEPARATOR}t`, {})]);
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
agentsLowCap, mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
await expect(svc.chat({
|
|
|
|
|
agentName: 'low', userMessage: 'spin', ownerId: 'owner-1',
|
|
|
|
|
})).rejects.toThrow(/exceeded 2 iterations/);
|
|
|
|
|
|
|
|
|
|
// Hostile agent with maxIterations=1000000 — must clamp to 50, not iterate forever.
|
|
|
|
|
const agentsHostile = {
|
|
|
|
|
getByName: vi.fn(async () => ({
|
|
|
|
|
id: 'agent-bad', name: 'bad', description: '', systemPrompt: '',
|
|
|
|
|
llm: { id: 'llm-1', name: 'qwen3-thinking' },
|
|
|
|
|
project: { id: 'proj-1', name: 'mcpctl-dev' },
|
|
|
|
|
proxyModelName: null, defaultParams: {},
|
|
|
|
|
extras: { maxIterations: 1_000_000 },
|
|
|
|
|
ownerId: 'owner-1', version: 1, createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
})),
|
|
|
|
|
} as unknown as AgentService;
|
|
|
|
|
const adapter2 = scriptedAdapter([toolCall(`g${TOOL_NAME_SEPARATOR}t`, {})]);
|
|
|
|
|
const chatRepo2 = mockChatRepo();
|
|
|
|
|
const svc2 = new ChatService(
|
|
|
|
|
agentsHostile, mockLlms(), adapterRegistry(adapter2),
|
|
|
|
|
chatRepo2, mockPromptRepo(), tools,
|
|
|
|
|
);
|
|
|
|
|
await expect(svc2.chat({
|
|
|
|
|
agentName: 'bad', userMessage: 'spin', ownerId: 'owner-1',
|
|
|
|
|
})).rejects.toThrow(/exceeded 50 iterations/);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// Regression: thread message reads must enforce ownership.
|
|
|
|
|
// Found by /gstack-review on 2026-04-25.
|
|
|
|
|
// Without this, any caller with `view:agents` could read another user's thread
|
|
|
|
|
// by guessing/learning the threadId (CUIDs leak through SSE chunks + tool _meta).
|
|
|
|
|
it('listMessages refuses a thread owned by another user (404, not 403, to avoid id-enumeration)', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
// Pre-seed a thread owned by 'alice'
|
|
|
|
|
await chatRepo.createThread({ agentId: 'agent-x', ownerId: 'alice' });
|
|
|
|
|
const aliceThread = chatRepo._threads[0]!;
|
|
|
|
|
await chatRepo.appendMessage({
|
|
|
|
|
threadId: aliceThread.id,
|
|
|
|
|
role: 'user',
|
|
|
|
|
content: 'private to alice',
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(scriptedAdapter([chatCompletion('ok')])),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
// Bob requests Alice's thread by id — must 404.
|
|
|
|
|
await expect(svc.listMessages(aliceThread.id, 'bob'))
|
|
|
|
|
.rejects.toThrow(/not found/i);
|
|
|
|
|
|
|
|
|
|
// Alice gets her own messages.
|
|
|
|
|
const aliceMessages = await svc.listMessages(aliceThread.id, 'alice');
|
|
|
|
|
expect(aliceMessages.map((m) => m.content)).toEqual(['private to alice']);
|
|
|
|
|
|
|
|
|
|
// Genuinely missing thread — same 404 shape (no oracle leak).
|
|
|
|
|
await expect(svc.listMessages('cnonexistent000000000000000', 'alice'))
|
|
|
|
|
.rejects.toThrow(/not found/i);
|
|
|
|
|
});
|
2026-04-26 19:27:59 +01:00
|
|
|
|
|
|
|
|
// ── Agent-direct prompts + personality overlay (Stage 3 system block) ──
|
|
|
|
|
|
|
|
|
|
it('injects agent-direct prompts BETWEEN agent.systemPrompt and project prompts', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const prompts: Prompt[] = [
|
|
|
|
|
// Project prompt
|
|
|
|
|
{
|
|
|
|
|
id: 'p-proj', name: 'proj', content: 'PROJECT_TEXT',
|
|
|
|
|
projectId: 'proj-1', agentId: null, priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
},
|
|
|
|
|
// Agent-direct prompt
|
|
|
|
|
{
|
|
|
|
|
id: 'p-direct', name: 'direct', content: 'AGENT_DIRECT_TEXT',
|
|
|
|
|
projectId: null, agentId: 'agent-reviewer', priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
},
|
|
|
|
|
];
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(prompts), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({ agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1' });
|
|
|
|
|
const sys = (inferSpy.mock.calls[0][0] as InferContext).body.messages.find((m) => m.role === 'system');
|
|
|
|
|
const text = sys?.content as string;
|
|
|
|
|
expect(text.indexOf('You are a helpful agent.')).toBeLessThan(text.indexOf('AGENT_DIRECT_TEXT'));
|
|
|
|
|
expect(text.indexOf('AGENT_DIRECT_TEXT')).toBeLessThan(text.indexOf('PROJECT_TEXT'));
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('appends personality-bound prompts after project prompts when --personality is passed', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const projectPrompt: Prompt = {
|
|
|
|
|
id: 'p-proj', name: 'proj', content: 'PROJECT_TEXT',
|
|
|
|
|
projectId: 'proj-1', agentId: null, priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
const personalityPrompt: Prompt = {
|
|
|
|
|
id: 'p-pers', name: 'pers', content: 'PERSONALITY_TEXT',
|
|
|
|
|
projectId: null, agentId: null, priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
const personalities = mockPersonalityRepo({
|
|
|
|
|
'pers-grumpy': {
|
|
|
|
|
personality: makePersonality({ id: 'pers-grumpy', name: 'grumpy', agentId: 'agent-reviewer' }),
|
|
|
|
|
bindings: [{ promptId: personalityPrompt.id, priority: 5 }],
|
|
|
|
|
},
|
|
|
|
|
}, [projectPrompt, personalityPrompt]);
|
|
|
|
|
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo([projectPrompt, personalityPrompt]), mockTools(),
|
|
|
|
|
personalities,
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
personalityName: 'grumpy',
|
|
|
|
|
});
|
|
|
|
|
const sys = (inferSpy.mock.calls[0][0] as InferContext).body.messages.find((m) => m.role === 'system');
|
|
|
|
|
const text = sys?.content as string;
|
|
|
|
|
expect(text.indexOf('PROJECT_TEXT')).toBeLessThan(text.indexOf('PERSONALITY_TEXT'));
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('falls back to agent.defaultPersonality when --personality is omitted', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const personalityPrompt: Prompt = {
|
|
|
|
|
id: 'p-pers', name: 'pers', content: 'DEFAULT_PERSONALITY_TEXT',
|
|
|
|
|
projectId: null, agentId: null, priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
const personalities = mockPersonalityRepo({
|
|
|
|
|
'pers-default': {
|
|
|
|
|
personality: makePersonality({ id: 'pers-default', name: 'default', agentId: 'agent-reviewer' }),
|
|
|
|
|
bindings: [{ promptId: personalityPrompt.id, priority: 5 }],
|
|
|
|
|
},
|
|
|
|
|
}, [personalityPrompt]);
|
|
|
|
|
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents({ defaultPersonality: { id: 'pers-default', name: 'default' } }),
|
|
|
|
|
mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo([personalityPrompt]), mockTools(),
|
|
|
|
|
personalities,
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({ agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1' });
|
|
|
|
|
const sys = (inferSpy.mock.calls[0][0] as InferContext).body.messages.find((m) => m.role === 'system');
|
|
|
|
|
expect(sys?.content as string).toContain('DEFAULT_PERSONALITY_TEXT');
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('throws when --personality references a name the agent does not own', async () => {
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const personalities = mockPersonalityRepo({});
|
|
|
|
|
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo(), mockTools(),
|
|
|
|
|
personalities,
|
|
|
|
|
);
|
|
|
|
|
await expect(svc.chat({
|
|
|
|
|
agentName: 'reviewer',
|
|
|
|
|
userMessage: 'hi',
|
|
|
|
|
ownerId: 'owner-1',
|
|
|
|
|
personalityName: 'ghost',
|
|
|
|
|
})).rejects.toThrow(/Personality not found/);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
it('preserves today\'s system block when no personality and no agent-direct prompts exist', async () => {
|
|
|
|
|
// Regression guard: backwards-compatible by construction.
|
|
|
|
|
const chatRepo = mockChatRepo();
|
|
|
|
|
const adapter = scriptedAdapter([chatCompletion('ok')]);
|
|
|
|
|
const inferSpy = adapter.infer as ReturnType<typeof vi.fn>;
|
|
|
|
|
const projectPrompt: Prompt = {
|
|
|
|
|
id: 'p-proj', name: 'proj', content: 'ONLY_PROJECT_TEXT',
|
|
|
|
|
projectId: 'proj-1', agentId: null, priority: 5, summary: null,
|
|
|
|
|
chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
};
|
|
|
|
|
const svc = new ChatService(
|
|
|
|
|
mockAgents(), mockLlms(), adapterRegistry(adapter),
|
|
|
|
|
chatRepo, mockPromptRepo([projectPrompt]), mockTools(),
|
|
|
|
|
);
|
|
|
|
|
await svc.chat({ agentName: 'reviewer', userMessage: 'hi', ownerId: 'owner-1' });
|
|
|
|
|
const sys = (inferSpy.mock.calls[0][0] as InferContext).body.messages.find((m) => m.role === 'system');
|
|
|
|
|
const text = sys?.content as string;
|
|
|
|
|
expect(text).toContain('You are a helpful agent.');
|
|
|
|
|
expect(text).toContain('ONLY_PROJECT_TEXT');
|
|
|
|
|
});
|
feat(agents): mcpd repos + Agent/Chat services with tool-use loop (Stage 2)
Layers the persistence-side logic on top of the Stage 1 schema. AgentService
mirrors LlmService's CRUD shape with name-resolved llm/project references and
yaml round-trip support; ChatService is the orchestrator that drives one chat
turn end-to-end: build the merged system block (agent.systemPrompt + project
Prompts ordered by priority desc + per-call systemAppend), persist the user
turn, run the adapter, dispatch any tool_calls through an injected
ChatToolDispatcher, persist tool turns linked back via toolCallId, and loop
until the model returns terminal text.
Per-call params resolve LiteLLM-style: request body → agent.defaultParams →
adapter default. The escape hatch `extra` is forwarded as-is so each adapter
can cherry-pick provider-specific knobs (Anthropic metadata, vLLM
repetition_penalty, etc.) without code changes here.
Persistence is non-transactional across the loop because tool calls can take
minutes; long-held DB transactions would starve other writers. Instead each
in-flight assistant turn is written `pending` and flipped to `complete` only
after its tool results land. On failure or max-iter overrun, every `pending`
row in the thread is flipped to `error` so the trail is auditable.
Tools are namespaced on the wire as `<server>__<tool>`, unmarshalled at
dispatch time; `tools_allowlist` filters before the model sees the list.
Tests:
agent-service.test.ts (7) — CRUD with name-resolved llm/project, conflict
on duplicate, llm switch, project detach, listByProject filtering,
upsertByName branch coverage.
chat-service.test.ts (9) — plain text turn, full text→tool→text loop with
toolCallId linkage, max-iter cap leaves zero pending, adapter-throws
leaves zero pending, body→defaultParams merge, `extra` passthrough,
project-Prompt priority ordering in the system block, tool-without-
project rejection, tools_allowlist filtering.
All 16 green; full mcpd suite still 737/737.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-25 16:38:38 +01:00
|
|
|
});
|
2026-04-26 19:27:59 +01:00
|
|
|
|
|
|
|
|
// ── Helpers for personality-overlay tests ──
|
|
|
|
|
|
|
|
|
|
function makePersonality(overrides: Partial<Personality> = {}): Personality {
|
|
|
|
|
return {
|
|
|
|
|
id: `pers-${Math.random().toString(36).slice(2, 8)}`,
|
|
|
|
|
name: 'p',
|
|
|
|
|
description: '',
|
|
|
|
|
agentId: 'agent-reviewer',
|
|
|
|
|
priority: 5,
|
|
|
|
|
createdAt: NOW,
|
|
|
|
|
updatedAt: NOW,
|
|
|
|
|
...overrides,
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
interface MockPersonalityFixture {
|
|
|
|
|
personality: Personality;
|
|
|
|
|
bindings: Array<{ promptId: string; priority: number }>;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function mockPersonalityRepo(
|
|
|
|
|
fixtures: Record<string, MockPersonalityFixture>,
|
|
|
|
|
prompts: Prompt[] = [],
|
|
|
|
|
): IPersonalityRepository {
|
|
|
|
|
const byId = new Map<string, MockPersonalityFixture>(Object.entries(fixtures));
|
|
|
|
|
const promptsById = new Map<string, Prompt>(prompts.map((p) => [p.id, p]));
|
|
|
|
|
return {
|
|
|
|
|
findAll: vi.fn(async () => [...byId.values()].map((f) => f.personality)),
|
|
|
|
|
findByAgent: vi.fn(async (agentId: string) =>
|
|
|
|
|
[...byId.values()].filter((f) => f.personality.agentId === agentId).map((f) => f.personality)),
|
|
|
|
|
findById: vi.fn(async (id: string) => byId.get(id)?.personality ?? null),
|
|
|
|
|
findByNameAndAgent: vi.fn(async (name: string, agentId: string) => {
|
|
|
|
|
for (const f of byId.values()) {
|
|
|
|
|
if (f.personality.name === name && f.personality.agentId === agentId) {
|
|
|
|
|
return f.personality;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return null;
|
|
|
|
|
}),
|
|
|
|
|
create: vi.fn(),
|
|
|
|
|
update: vi.fn(),
|
|
|
|
|
delete: vi.fn(),
|
|
|
|
|
listPrompts: vi.fn(async (personalityId: string) => {
|
|
|
|
|
const fixture = byId.get(personalityId);
|
|
|
|
|
if (!fixture) return [];
|
|
|
|
|
return fixture.bindings.map<PersonalityPrompt & { prompt: Prompt }>((b) => ({
|
|
|
|
|
id: `bind-${b.promptId}`,
|
|
|
|
|
personalityId,
|
|
|
|
|
promptId: b.promptId,
|
|
|
|
|
priority: b.priority,
|
|
|
|
|
createdAt: NOW,
|
|
|
|
|
prompt: promptsById.get(b.promptId) ?? ({
|
|
|
|
|
id: b.promptId, name: 'p', content: '',
|
|
|
|
|
projectId: null, agentId: null, priority: b.priority,
|
|
|
|
|
summary: null, chapters: null, linkTarget: null, version: 1,
|
|
|
|
|
createdAt: NOW, updatedAt: NOW,
|
|
|
|
|
} as Prompt),
|
|
|
|
|
}));
|
|
|
|
|
}),
|
|
|
|
|
attachPrompt: vi.fn(),
|
|
|
|
|
detachPrompt: vi.fn(),
|
|
|
|
|
findBinding: vi.fn(async () => null),
|
|
|
|
|
};
|
|
|
|
|
}
|