Real Anthropic Claude integration per explicit product decision: a ticket's AI session diagnoses the problem via a structured-output call, applies a DB-configurable confidence-band policy (FR-005), and on "proceed" reasons and acts through a small permission/risk-gated tool system (FR-011/FR-012), optionally walking a matching runbook step by step with the application — never the model — owning the step index (FR-015/FR-016). Resolution requires real tool evidence, never customer claims alone (FR-018) — verifyProductResolution is a documented fail-closed placeholder mirroring the existing malware-scanner precedent, since no real per-product operational signal exists yet. AISupportSession.status mirrors onto Ticket.status through 003-ticketing's existing AI_ANALYZING/AI_TROUBLESHOOTING/AI_VERIFYING/AI_RESOLVED/ HUMAN_ESCALATION state machine, discovered during planning to have been built anticipating this exact feature. Two circular module dependencies (escalation<->sessions, tools<->sessions) were designed around rather than found as bugs: escalation is a pure summary formatter with no state dependencies of its own, and tools stays a clean leaf module with zero dependency on ai-support/sessions. Ticket creation enqueues the first diagnosis turn via the existing queue infrastructure (off the hot path of the inbound SaaS integration endpoint); a human actor changing ticket status ends the AI session via the event-bus scaffold that existed in this codebase but had never been wired to anything. A real Prisma limitation was found and fixed before it reached tests: compound-unique upsert rejects null for a nullable key column, so AIConfidencePolicy uses find-then-update/create instead, same fix class 004 already used for the same underlying limitation. Adds 9 unit tests (confidence-band, tool-policy-gate, runbook-step- advance) and 6 integration test files, including the two constitution- required standing E2E scenarios. AI-independent tests were run against real Postgres/Redis/MinIO (88 passed, 0 failed across the full suite, including every pre-existing 002/003/004 test). The AI-dependent tests compile and skip cleanly via describe.skipIf but were not run against a live model — no ANTHROPIC_API_KEY was available in this session; a real key must be supplied before this feature can actually run. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
162 lines
6.3 KiB
TypeScript
162 lines
6.3 KiB
TypeScript
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
|
import { buildApp } from '@/app';
|
|
import { prismaClient } from '@/infrastructure/database';
|
|
import { FastifyInstance } from 'fastify';
|
|
import {
|
|
encryptCredential,
|
|
generateCredentialSecret,
|
|
issueIntegrationToken,
|
|
} from '@/modules/catalog/products';
|
|
import { sessionsService } from '@/modules/ai-support/sessions';
|
|
|
|
/** Covers specs/005-ai-support/quickstart.md Scenario 2 — requires a real ANTHROPIC_API_KEY. */
|
|
const hasRealApiKey = /^sk-ant-/.test(process.env.ANTHROPIC_API_KEY ?? '');
|
|
|
|
describe.skipIf(!hasRealApiKey)('AI clarification loop (User Story 2)', () => {
|
|
let app: FastifyInstance;
|
|
const externalProductId = `TEST_AI_ASK_PROD_${Date.now()}`;
|
|
let secret: string;
|
|
|
|
beforeAll(async () => {
|
|
app = await buildApp();
|
|
const product = await prismaClient.product.create({
|
|
data: { externalProductId, name: 'AI Ask Test Product', status: 'active' },
|
|
});
|
|
secret = generateCredentialSecret();
|
|
await prismaClient.productIntegration.create({
|
|
data: {
|
|
productId: product.id,
|
|
credentialRef: encryptCredential(secret),
|
|
authMechanism: 'signed_token',
|
|
allowedScope: { tenantIds: ['tenant-1'] },
|
|
status: 'active',
|
|
rateLimitPerMinute: 1000,
|
|
rateLimitPerUserPerMinute: 1000,
|
|
},
|
|
});
|
|
const created = await app.inject({
|
|
method: 'POST',
|
|
url: `/admin/products/${externalProductId}/knowledge`,
|
|
payload: {
|
|
code: `KB-ASK-${Date.now()}`,
|
|
type: 'faq',
|
|
problem: 'Vague, ambiguous problem report scenarios.',
|
|
},
|
|
});
|
|
await app.inject({
|
|
method: 'PATCH',
|
|
url: `/admin/knowledge/${created.json().data.code}/publish`,
|
|
});
|
|
// Force the "ask" band deterministically: an impossibly narrow high/low gap makes almost any
|
|
// confidence land in "ask", and a generous question budget lets the loop actually run.
|
|
await app.inject({
|
|
method: 'PUT',
|
|
url: `/admin/products/${externalProductId}/ai-policy`,
|
|
payload: { highThreshold: 0.999, lowThreshold: 0.001, maxClarifyingQuestions: 1 },
|
|
});
|
|
});
|
|
|
|
afterAll(async () => {
|
|
const product = await prismaClient.product.findUnique({ where: { externalProductId } });
|
|
if (product) {
|
|
const tickets = await prismaClient.ticket.findMany({ where: { productId: product.id } });
|
|
const ticketIds = tickets.map((t) => t.id);
|
|
const sessions = await prismaClient.aISupportSession.findMany({
|
|
where: { ticketId: { in: ticketIds } },
|
|
});
|
|
const sessionIds = sessions.map((s) => s.id);
|
|
await prismaClient.aIKnowledgeReference.deleteMany({
|
|
where: { sessionId: { in: sessionIds } },
|
|
});
|
|
await prismaClient.aIDiagnosis.deleteMany({ where: { sessionId: { in: sessionIds } } });
|
|
await prismaClient.aIInteraction.deleteMany({ where: { sessionId: { in: sessionIds } } });
|
|
await prismaClient.aISupportSession.deleteMany({ where: { id: { in: sessionIds } } });
|
|
await prismaClient.ticketMessage.deleteMany({ where: { ticketId: { in: ticketIds } } });
|
|
await prismaClient.ticket.deleteMany({ where: { productId: product.id } });
|
|
await prismaClient.problem.deleteMany({ where: { productId: product.id } });
|
|
await prismaClient.knowledgeEntry.deleteMany({ where: { productId: product.id } });
|
|
await prismaClient.aIConfidencePolicy.deleteMany({ where: { productId: product.id } });
|
|
}
|
|
await prismaClient.productIntegration.deleteMany({ where: { product: { externalProductId } } });
|
|
await prismaClient.product.deleteMany({ where: { externalProductId } });
|
|
await app.close();
|
|
});
|
|
|
|
it('an "ask" outcome posts a customer-visible question, and a reply produces a new diagnosis', async () => {
|
|
const token = issueIntegrationToken(secret, {
|
|
externalProductId,
|
|
tenantId: 'tenant-1',
|
|
userId: 'user-1',
|
|
});
|
|
const created = await app.inject({
|
|
method: 'POST',
|
|
url: '/v1/support/requests',
|
|
headers: { authorization: `Bearer ${token}` },
|
|
payload: {
|
|
productId: externalProductId,
|
|
tenantId: 'tenant-1',
|
|
userId: 'user-1',
|
|
source: 'test',
|
|
problem: 'It broke.',
|
|
},
|
|
});
|
|
const ticketId = created.json().data.ticketId;
|
|
await sessionsService.runFirstTurn(ticketId);
|
|
|
|
const beforeReply = await app.inject({ method: 'GET', url: `/tickets/${ticketId}/ai-session` });
|
|
const beforeStatus: string = beforeReply.json().data.status;
|
|
const diagnosesBefore = beforeReply.json().data.diagnosis;
|
|
|
|
if (beforeStatus === 'escalated') {
|
|
// The very first diagnosis already escalated (e.g. FR-006/provider failure) — the "ask"
|
|
// path specifically wasn't exercised this run; nothing further to assert here.
|
|
return;
|
|
}
|
|
|
|
const messagesResponse = await app.inject({
|
|
method: 'GET',
|
|
url: `/tickets/${ticketId}/messages`,
|
|
});
|
|
const aiMessages = messagesResponse
|
|
.json()
|
|
.data.filter((m: { type: string }) => m.type === 'AI_MESSAGE');
|
|
expect(aiMessages.length).toBeGreaterThan(0);
|
|
|
|
const replyResponse = await app.inject({
|
|
method: 'POST',
|
|
url: `/tickets/${ticketId}/ai-session/messages`,
|
|
payload: {
|
|
message: 'The problem happens specifically when I try to load the dashboard page.',
|
|
},
|
|
});
|
|
expect(replyResponse.statusCode).toBe(200);
|
|
|
|
const afterReply = await app.inject({ method: 'GET', url: `/tickets/${ticketId}/ai-session` });
|
|
// A new diagnosis must exist and be distinguishable from the first (different createdAt) —
|
|
// proving re-diagnosis happened rather than reusing the original.
|
|
expect(afterReply.json().data.diagnosis.id).not.toBe(diagnosesBefore?.id);
|
|
}, 90000);
|
|
});
|
|
|
|
/** No LLM call involved — runs unconditionally, unlike the rest of this file. */
|
|
describe('AI session message routing guard (contracts/ai-support-contract.md guarantee 1)', () => {
|
|
let app: FastifyInstance;
|
|
|
|
beforeAll(async () => {
|
|
app = await buildApp();
|
|
});
|
|
|
|
afterAll(async () => {
|
|
await app.close();
|
|
});
|
|
|
|
it('404s replying to a ticket with no active AI session', async () => {
|
|
const response = await app.inject({
|
|
method: 'POST',
|
|
url: `/tickets/nonexistent-ticket-id/ai-session/messages`,
|
|
payload: { message: 'hello' },
|
|
});
|
|
expect(response.statusCode).toBe(404);
|
|
});
|
|
});
|