Compare commits

...
13 changed files with 709 additions and 46 deletions
+19 -3
View File
@@ -26,7 +26,7 @@ async function post(request) {
if (!result.success) {
return Response.json(
{ ...result, reconstruction: result.reconstruction || null },
{ error: "Reasoning request could not be completed." },
{ status: Number(result.statusCode) || 500 },
);
}
@@ -42,9 +42,25 @@ async function post(request) {
promptVersion: result.promptVersion,
});
} catch (e) {
if (e.code === "PROVIDER_UNAVAILABLE") {
return Response.json(
{ error: "Reasoning service is temporarily unavailable." },
{ status: 503 },
);
}
const validationFailed = e?.validationFailed || e?.code === "VALIDATION_FAILED";
if (validationFailed) {
return Response.json(
{ success: false, error: "Reasoning request could not be completed." },
{ status: Number(e.statusCode) || 500 },
);
}
const statusCode = Number(e.statusCode) || 500;
return Response.json(
{ error: e.message || "Unknown server error", responseDurationMs: 0 },
{ status: 500 },
{ error: "Reasoning request could not be completed." },
{ status: statusCode },
);
}
}
+3 -1
View File
@@ -55,7 +55,9 @@ async function post(request) {
{
success: false,
stage: error.statusCode === 400 ? "request_validation" : "provider",
error: error.message ?? "Overview synthesis failed",
error: error.statusCode === 400
? "Invalid overview request"
: "Reasoning request could not be completed.",
},
{ status: error.statusCode }
);
+2 -8
View File
@@ -48,14 +48,8 @@ async function post(request) {
return Response.json(
{
success: false,
error: diagnostics.error,
validationErrors: result.validationErrors,
diagnostics: result.diagnostics,
analysisErrors: result.analysisErrors,
validationIssues: result.validationIssues,
providerApiPath: result.providerApiPath,
providerExecution: result.providerExecution,
rawResponse: result.rawResponse ?? undefined,
error: status === 400 ? diagnostics.error : "Reasoning request could not be completed.",
...(status === 400 ? { validationErrors: result.validationErrors } : {}),
},
{ status },
);
+3 -1
View File
@@ -57,7 +57,9 @@ async function post(request) {
{
success: false,
stage: error.statusCode === 400 ? "request_validation" : "provider",
error: error.message ?? "Synthesis failed",
error: error.statusCode === 400
? "Invalid synthesis request"
: "Reasoning request could not be completed.",
},
{ status: error.statusCode }
);
@@ -50,6 +50,23 @@ async function post(request) {
);
}
const REQUEST_LENGTH_LIMITS = {
answer: 10000,
question: 2048,
centralStatement: 2048,
targetLabel: 2048,
targetDescription: 2048,
};
for (const [field, limit] of Object.entries(REQUEST_LENGTH_LIMITS)) {
if (body[field] && body[field].length > limit) {
return Response.json(
{ error: `Request field "${field}" exceeds maximum length of ${limit} characters` },
{ status: 400 },
);
}
}
const prompt = buildFocusedDeconstructPrompt({
targetLabel: body.targetLabel,
targetDescription: body.targetDescription,
@@ -154,7 +171,7 @@ async function post(request) {
);
}
return Response.json(
{ error: e.message || "Unknown server error" },
{ error: "Reasoning request could not be completed." },
{ status: 500 },
);
}
+8 -2
View File
@@ -30,6 +30,9 @@ function isTechnicalSummary(summary) {
return false;
}
// ── Focused answer contract ──────────────────────────────────
const FOCUSED_ANSWER_MAX_LENGTH = 10000;
// ── Recovery state components (Phase 2) ───────────────────────
function ProviderUnavailableCard({ onRestart }) {
@@ -238,8 +241,11 @@ function FocusedQuestionBody({
{focused?.question?.trim() && processingStep !== "active" && focused.status === "formulated" && !hasAnswer && !hasActiveFollowUp && (
<div data-testid="completed-narrative">
<label htmlFor={`rw-answer-${nodeId}`} className="mb-2 block text-sm font-medium text-gray-700">Your response</label>
<textarea id={`rw-answer-${nodeId}`} value={focusedAnswer} onChange={(e) => setFocusedAnswer(e.target.value)} rows={4} data-testid="response-textarea" className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400 disabled:cursor-not-allowed disabled:opacity-60" placeholder="What do you know about this?" />
<button onClick={(e) => { e.stopPropagation(); handleDeconstructSubmit(nodeId, focusedAnswer); }} disabled={!focusedAnswer.trim() || processingStep === "active"} style={{ cursor: !focusedAnswer.trim() || processingStep === "active" ? "not-allowed" : "pointer" }} className="mt-3 rounded-lg border border-green-600 bg-white px-4 py-2 text-sm font-medium text-green-700 hover:bg-green-50 transition disabled:opacity-50">Submit response</button>
<textarea id={`rw-answer-${nodeId}`} value={focusedAnswer} onChange={(e) => setFocusedAnswer(e.target.value)} rows={4} maxLength={FOCUSED_ANSWER_MAX_LENGTH} data-testid="response-textarea" className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400 disabled:cursor-not-allowed disabled:opacity-60" placeholder="What do you know about this?" />
<div className="flex items-center justify-between mt-2">
<span className="text-xs text-gray-400">{focusedAnswer.length}/{FOCUSED_ANSWER_MAX_LENGTH}</span>
<button onClick={(e) => { e.stopPropagation(); handleDeconstructSubmit(nodeId, focusedAnswer); }} disabled={!focusedAnswer.trim() || processingStep === "active"} style={{ cursor: !focusedAnswer.trim() || processingStep === "active" ? "not-allowed" : "pointer" }} className="rounded-lg border border-green-600 bg-white px-4 py-2 text-sm font-medium text-green-700 hover:bg-green-50 transition disabled:opacity-50">Submit response</button>
</div>
</div>
)}
+160
View File
@@ -16,6 +16,13 @@
- The existing generic Retry UX remains unchanged. A live deployed outage had already proven investigation preservation.
- `/api/cases/start` and `/api/cases/update` outage sanitization remain separately unverified; this change does not claim those routes are fixed.
## Reasoning API browser error boundary hardened
- All reasoning HTTP error responses now sanitize raw provider diagnostics: `e.message` / raw internals never leak to the client.
- Verified on: `/api/cases/start`, `/api/focused-investigation/deconstruct`, `/api/cases/overview`, `/api/cases/synthesis`, `/api/analyse`.
- Provider/internal diagnostics remain server-side only.
- No reasoning semantics changed.
## v0.62d production Docker packaging — LIVE PROVEN
- `Dockerfile` — minimal multi-stage Alpine build (Node 22), Next.js standalone output mode
@@ -118,6 +125,60 @@ Jenkins SCM branch used to load the Jenkinsfile is conceptually separate from th
- CSS Grid with responsive column placement replaces original flexbox; three DOM sections ensure correct mobile stacking without duplicating content
- desktop two-column presentation (explanation left / sign-in right) preserved unchanged at `md` breakpoint and above
## Focused-investigation input hardening
- Focused answer now has a visible 10,000-character UI limit (maxLength + character counter).
- Server enforces matching 10,000-character bound on `/api/focused-investigation/deconstruct`.
- Other prompt-bearing fields retain explicit defensive bounds (2048 characters each).
- Oversized/malformed requests are rejected before reasoning with controlled HTTP 400.
- No punctuation/HTML/SQL-style content stripping introduced.
- Reasoning semantics unchanged.
## Pre-Tester Input Security Review
**Status:** closed — sufficient for controlled external-user testing. Not a general security audit, penetration test, or production-launch certification.
### SQL injection
- No raw request-driven SQL construction found.
- Persistence uses Supabase/PostgREST query-builder boundaries.
- SQL-character stripping / sanitisation is not warranted.
### XSS
- Normal user-controlled text is React-escaped.
- No user-controlled unsafe HTML sink found.
- HTML / script stripping of prose is not warranted.
### Reasoning error disclosure
- Browser-facing reasoning errors have been sanitised (commit `2cb2d55`).
### Focused user input
- Bounded to 10,000 characters server-side on `/api/focused-investigation/deconstruct`.
- UI exposes matching `maxLength` and character counter (commit `a6796c6`).
### Investigation snapshot size
- Authenticated persistence envelope currently lacks a whole-snapshot size ceiling.
- Not classified as an outstanding pre-tester blocker.
- Legitimate investigation size / turn depth is not yet known.
- No arbitrary product ceiling imposed before real-user evidence exists.
- Monitor snapshot growth later via metadata (serialized bytes / revision / contribution count) without logging investigation content.
- Malformed-envelope / runtime validation remains a separate future hardening opportunity.
### Prompt injection
- Assessed as **low risk under the current architecture** — not claimed to be impossible or "solved".
- Untrusted scenario / answer / persisted text can influence model reasoning.
- No evidence it gains application authority: no model-accessible arbitrary tools, DB targeting, auth identity control, ownership bypass, deletion, or arbitrary external requests found.
- Model/provider configuration is server-owned; model output passes through parsing/structured validation/domain boundaries before application mutation.
- No prompt-injection phrase / keyword filtering warranted; do not strip instruction-like natural-language content.
- Not a pre-tester blocker.
### Deferred non-security observations (backlog only — no code change)
- Orchestrator update flow contains an implicitly correct but indent-control-flow-unclear fall-through / else structure worth cleaning up later.
- Investigation overview validation may accept unexpected extra fields.
- Provider JSON recovery is intentionally/permissively capable of recovering malformed JSON; may merit a future robustness review.
### Tester-readiness position
Identified MEDIUM pre-tester security work is sufficiently addressed for controlled external-user testing. The application has **not** been generally security-audited, penetration-tested, or certified as production-secure or commercially launch-ready.
## CURRENT MVP DIRECTION
Initial-decomposition hardening is frozen for the current MVP stage.
@@ -348,3 +409,102 @@ Previous focused-deconstruction semantic runs before the plumbing fix remain **i
2. Initial-reconstruction schema was supplied to focused-deconstruction call instead of its own contract
These are recorded as known contamination in the v0.61 archive chapter (`docs/design-evolution/ch19/initial-decomposition-v0.61.md`). The plumbing fix is complete and verified (48/48 tests).
## Tester-Readiness Findings — Recorded for Session Continuity
### Tester-readiness position
```
controlled external testing: GO
known pre-tester security blockers: none identified
next uncertainty worth reducing: real first-time-user product value, trust and independent usability
```
Do not claim commercial launch readiness, general production scalability, security certification or proven product-market fit.
### First external-user experiment
**Primary question:** Can a first-time user, without Rob guiding the investigation, use Confidence Engine to reach a Current Understanding that they regard as materially better than the way they framed the situation at the start?
**Tester evidence should prioritise:**
- did understanding materially improve?
- did they trust the changed understanding?
- could they reach it without Rob's help?
- was the experience usable?
**Post-use discovery questions (do not seed categories or mention a mental-health interpretation):**
- Who do you think this would be useful for?
- What kinds of situations would you use this for?
- Is there a situation in your own life or work where you can imagine coming back and using this?
### Landing-page positioning
- Current landing wording is deliberately unchanged for the first cohort.
- It may naturally be interpreted broadly, potentially including mental-health/anxiety-adjacent situations.
- This is an observation to learn from, not currently a defect.
- Use unprompted tester responses to learn what category/audience/use cases users believe Confidence Engine belongs to.
- Do not reposition before this evidence exists.
### Synthetic-user testing
- AI-driven first-time-user journeys may be useful as a pre-human stress test.
- Preferred conceptual separation: **synthetic first-time user → real Confidence Engine journey → independent evaluator**.
- Synthetic testing can expose reasoning/UX/systematic failures.
- It does NOT replace human evidence about genuine trust, changed understanding, usefulness, repeat use or willingness to return.
- No synthetic-testing implementation is currently required before human testing.
### Reasoning concurrency
- Current CE application has **no** server-side per-user reasoning concurrency protection.
- No CE-level global reasoning concurrency limit.
- Multiple tabs/users can reach provider concurrently.
- Ollama/provider concurrency and queue behaviour are not controlled by CE repository code.
- Do not invent queue/latency behaviour from the 300-second request timeout.
- Controlled cohort of roughly 510 invited testers: **not currently a release blocker**.
- Observe actual behaviour before designing queue/semaphore infrastructure.
- Before wider/public access, reasoning-resource concurrency protection should be reconsidered.
### Rate limiting
- No application-level per-user reasoning rate limit currently exists.
- Generic API rate limiting is not required for the controlled tester cohort.
- Wider/public access should revisit reasoning-specific abuse/resource protection.
- Future protection should target scarce reasoning/provider capacity rather than indiscriminately throttling cheap authenticated persistence/read operations.
- `withAuthenticatedApi` establishes authenticated identity but should not automatically become a generic reasoning-throttling owner.
### Accidental duplicate submission (deferred small product/UX item)
- Trace indicates initial Analyse action is **not disabled** by `status === "loading"`.
- Rapid same-page duplicate submission may therefore be possible.
- This is distinct from server-side rate/concurrency protection.
- Retain for a future bounded correction; do not change production code now.
### Future operational evidence
If concurrency instrumentation becomes necessary, prefer metadata only:
- reasoning endpoint/action
- request start/end or duration
- number of concurrent active reasoning calls
- result/timeout classification
Do not log investigation content merely for capacity measurement.
### Investigation growth
Preserve the existing decision:
- No arbitrary whole-investigation snapshot ceiling before real-user evidence.
- Legitimate turn depth and mature investigation size are unknown.
- Later observation may use serialized snapshot bytes, revision, contribution count and finding count without recording content.
### Existing deferred hardening/cleanup (not tester blockers)
Ensure these remain visible and are not accidentally promoted to tester blockers:
- malformed persistence-envelope/runtime validation
- investigation overview unexpected-extra-field validation
- provider JSON recovery robustness
- orchestrator update-flow control/indentation clarity
- whole-snapshot size decision pending real-user evidence
- reasoning concurrency/rate protection before wider access
- initial Analyse duplicate-submit prevention
+59
View File
@@ -287,6 +287,65 @@ The following material learnings are carried forward as durable context for safe
User-selected/active investigation ownership must survive substantive ties and
question-formulation rejection. (Already documented in `docs/current-handoff.md`.)
## 11. Pre-Tester Input Security Review (closed)
**Status:** closed — sufficient for controlled external-user testing. Not a general security audit, penetration test, or production-launch certification.
### SQL injection
- No raw request-driven SQL construction found; persistence uses Supabase/PostgREST query-builder boundaries.
### XSS
- Normal user-controlled text is React-escaped; no unsafe HTML sink found.
### Reasoning error disclosure
- Browser-facing reasoning errors sanitised (commit `2cb2d55`).
### Focused input bound
- 10,000-character server-side + UI boundary on focused investigation (commit `a6796c6`).
### Investigation snapshot size envelope / investigation growth
- Authenticated persistence lacks a whole-snapshot size ceiling.
- No arbitrary whole-investigation snapshot ceiling before real-user evidence.
- Legitimate turn depth and mature investigation size are unknown.
- Later observation may use serialized snapshot bytes, revision, contribution count and finding count without recording content.
- Not a pre-tester blocker.
### Prompt injection
- **Low risk under the current architecture.** Untrusted text can influence model reasoning but no evidence it gains application authority. No model-accessible arbitrary tools, DB targeting, auth control, or privileged side effects found. Output passes structured validation before application mutation. Not a pre-tester blocker.
### Deferred non-security observations (backlog only — no code change)
- Orchestrator update flow indentation/control-flow clarity deferred.
- Investigation overview validation may accept unexpected extra fields.
- Provider JSON recovery permissiveness deferred as future robustness review.
- Malformed persistence-envelope/runtime validation deferred.
- Initial Analyse duplicate-submit prevention deferred (action not disabled by `status === "loading"`).
### Tester-readiness position
```
controlled external testing: GO
known pre-tester security blockers: none identified
next uncertainty worth reducing: real first-time-user product value, trust and independent usability
```
Do not claim commercial launch readiness, general production scalability, security certification or proven product-market fit.
### Reasoning concurrency (deferred)
- No server-side per-user reasoning concurrency protection; no CE-level global limit.
- Multiple tabs/users can reach provider concurrently — Ollama/provider behaviour not controlled by CE repository code.
- Controlled cohort of ~510 invited testers: **not a release blocker**.
- Before wider/public access, reasoning-resource concurrency protection should be reconsidered.
- If instrumentation needed later: prefer metadata only (endpoint/action, request start/end or duration, concurrent active call count, result/timeout classification). Do not log investigation content for capacity measurement.
### Rate limiting (deferred)
- No application-level per-user reasoning rate limit currently exists.
- Not required for controlled tester cohort; revisit before wider/public access.
- Future protection should target scarce reasoning/provider capacity, not indiscriminately throttle cheap authenticated persistence/read operations.
---
### RTO learning from Experiments 1417
Since the handoff document was written, further learning has emerged from Return-to-Origin work (RTO.1417):
+205
View File
@@ -0,0 +1,205 @@
import { describe, expect, it, vi } from "vitest";
// ── Mock domain seam and provider at module level ────
const mockAnalyseScenario = vi.fn();
vi.mock("@/lib/analysis", () => ({
analyseScenario: (...args) => mockAnalyseScenario(...args),
PROMPT_VERSIONS: ["v1"],
DEFAULT_PROMPT_VERSION: "v1",
}));
vi.mock("@/lib/llm/provider.js", () => ({
getProvider: () => ({}),
getProviderModelName: () => "gpt-5.6-terra",
}));
vi.mock("@/lib/supabase/api-auth.js", () => ({
withAuthenticatedApi: (handler) => handler,
}));
// ── Helpers ─────────────────────────────────────────
function makeValidScenario() {
return "The supplier changed delivery schedules without notice, causing our production line to halt.";
}
function makeRequest(body) {
return new Request("http://localhost/api/analyse", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
});
}
// ── Route tests — valid POST ────────────────────────
describe("POST /api/analyse — success contract", () => {
it("returns structured analysis on success", async () => {
mockAnalyseScenario.mockResolvedValue({
success: true,
inputClassification: "manufacturing",
reconstruction: { summary: "Validated reconstruction summary" },
evidence: [],
nextQuestion: null,
modelName: "gpt-5.6-terra",
responseDurationMs: 1200,
validationStatus: "passed",
promptVersion: "v1",
});
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(200);
const data = await res.json();
expect(typeof data.inputClassification).toBe("string");
expect(data.reconstruction.summary).toBe("Validated reconstruction summary");
});
});
// ── Route tests — request validation ────────────────
describe("POST /api/analyse — request validation", () => {
it("missing scenario → 400", async () => {
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({}));
expect(res.status).toBe(400);
const data = await res.json();
expect(data).toEqual({ error: "Request must include a 'scenario' string field" });
});
it("null scenario → 400", async () => {
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: null }));
expect(res.status).toBe(400);
});
it("non-string scenario → 400", async () => {
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: 123 }));
expect(res.status).toBe(400);
});
});
// ── Route tests — error handling ────────────────────
describe("POST /api/analyse — error boundary", () => {
it("domain seam throws PROVIDER_UNAVAILABLE → sanitized 503 with generic message", async () => {
const err = Object.assign(
new Error("Ollama /api/generate request timed out after 5 minutes"),
{ code: "PROVIDER_UNAVAILABLE", providerApiPath: "/api/generate" },
);
mockAnalyseScenario.mockImplementationOnce(async () => { throw err; });
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(503);
const data = await res.json();
expect(data.error).toBe("Reasoning service is temporarily unavailable.");
});
it("domain seam throws with diagnostics → sanitized response, preserved status", async () => {
const rawResponse = `{"reconstruction":{"observedStates":[{"id":"obs-1"${"x".repeat(2500)}}]}}`;
const err = Object.assign(
new Error("Provider unavailable"),
{
statusCode: 502,
providerApiPath: "/v1/responses",
providerExecution: { chatRequestAttempted: true },
rawResponse,
},
);
mockAnalyseScenario.mockImplementationOnce(async () => { throw err; });
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(502);
const data = await res.json();
expect(data.error).toBe("Reasoning request could not be completed.");
expect(JSON.stringify(data)).not.toMatch(/provider unavailable|generate|llama3|rawResponse/i);
});
it("domain seam throws without statusCode → 500", async () => {
mockAnalyseScenario.mockImplementationOnce(async () => { throw new Error("unknown error"); });
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(500);
const data = await res.json();
expect(data.error).toBe("Reasoning request could not be completed.");
});
it("domain seam throws validation failure → preserved status", async () => {
const err = Object.assign(new Error("Invalid input"), { statusCode: 400 });
mockAnalyseScenario.mockImplementationOnce(async () => { throw err; });
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(400);
const data = await res.json();
expect(data.error).toBe("Reasoning request could not be completed.");
});
it("domain seam returns failure result with diagnostics → sanitized response", async () => {
mockAnalyseScenario.mockResolvedValue({
success: false,
error: "Provider unavailable",
diagnostics: { modelName: "llama3" },
statusCode: 502,
analysisErrors: ["reconstruction: Required"],
validationIssues: [{ path: ["reconstruction"], code: "invalid_type", message: "Required" }],
providerApiPath: "/api/generate",
providerExecution: { generateRequestAttempted: true },
rawResponse: '{"observedStates":[{"id":"x"}]}',
});
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(502);
const data = await res.json();
expect(data).toEqual({ error: "Reasoning request could not be completed." });
expect(JSON.stringify(data)).not.toMatch(/provider unavailable|generate|llama3|rawResponse/i);
});
it("domain seam throws without exposing raw provider/internal diagnostics", async () => {
const err = Object.assign(
new Error("Internal connection reset by peer — host=10.0.0.5:8080 key=sk-abc"),
{ statusCode: 502, providerApiPath: "/internal/chat", providerExecution: { chatRequestAttempted: true } },
);
mockAnalyseScenario.mockImplementationOnce(async () => { throw err; });
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(502);
const data = await res.json();
expect(JSON.stringify(data)).not.toMatch(/10\.0\.0\.5|8080|sk-abc|providerApiPath|providerExecution|Internal connection reset/i);
});
it("domain seam returns failed analysis object → not spread into response", async () => {
mockAnalyseScenario.mockResolvedValue({
success: false,
error: "Analysis failed",
statusCode: 500,
failedAnalysisObject: { raw: true, internal: "diagnostics" },
});
const { POST } = await import("@/app/api/analyse/route.js");
const res = await POST(makeRequest({ scenario: makeValidScenario() }));
expect(res.status).toBe(500);
const data = await res.json();
expect(data).toEqual({ error: "Reasoning request could not be completed." });
expect(JSON.stringify(data)).not.toMatch(/failedAnalysisObject|internal|diagnostics/i);
});
});
+37 -1
View File
@@ -1,4 +1,4 @@
import { beforeEach, describe, expect, it, vi } from "vitest";
import { describe, expect, it, vi } from "vitest";
const mockSynthesize = vi.fn();
@@ -11,6 +11,10 @@ vi.mock("@/lib/llm/provider.js", () => ({
getProviderModelName: () => "gpt-5.6-terra",
}));
vi.mock("@/lib/supabase/api-auth.js", () => ({
withAuthenticatedApi: (handler) => handler,
}));
describe("POST /api/cases/overview provider routing", () => {
beforeEach(() => mockSynthesize.mockClear());
@@ -26,4 +30,36 @@ describe("POST /api/cases/overview provider routing", () => {
expect(response.status).toBe(200);
expect(mockSynthesize.mock.calls[0][1]).toMatchObject({ modelName: "gpt-5.6-terra" });
});
it("sanitizes provider failure details", async () => {
const error = Object.assign(
new Error("Ollama /api/generate returned 500 from private host"),
{ statusCode: 502 },
);
mockSynthesize.mockImplementationOnce(async () => {
throw error;
});
const { POST } = await import("@/app/api/cases/overview/route.js");
const response = await POST(new Request("http://localhost/api/cases/overview", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ situationGraph: { nodes: [], edges: [] } }),
}));
expect(response.status).toBe(502);
const rawData = await response.text();
const data = JSON.parse(rawData);
expect(data.success).toBe(false);
expect(data.stage).toBe("provider");
expect(data).toEqual({
success: false,
stage: "provider",
error: "Reasoning request could not be completed.",
});
// Prove raw provider diagnostics are NOT exposed at the route boundary
expect(rawData).not.toContain(error.message);
});
});
+3 -21
View File
@@ -169,7 +169,7 @@ describe("app/api/cases/start route", () => {
expect(JSON.stringify(body)).not.toMatch(/ollama|generate|timed out/i);
});
it("returns provider/internal failures as 5xx without stack traces", async () => {
it("sanitizes provider/internal failures at the browser boundary", async () => {
const rawResponse = `{"reconstruction":{"observedStates":[{"id":"obs-1"${"x".repeat(2500)}}]}}`;
mockStartCase.mockResolvedValue({
success: false,
@@ -207,26 +207,8 @@ describe("app/api/cases/start route", () => {
expect(response.status).toBe(502);
const body = await response.json();
expect(body).toHaveProperty("rawResponse");
expect(body.rawResponse).toBe(rawResponse);
expect(body.rawResponse.length).toBeGreaterThan(2000);
expect(body.analysisErrors).toEqual(["reconstruction: Required"]);
expect(body.validationIssues).toEqual([
expect.objectContaining({
path: ["reconstruction", "observedStates", 2, "description"],
code: "invalid_type",
message: "Required",
expected: "string",
received: "undefined",
}),
]);
expect(body.providerApiPath).toBe("/api/generate");
expect(body.providerExecution).toEqual({
chatCapabilityDetected: false,
chatRequestAttempted: false,
chatRequestSucceeded: false,
generateRequestAttempted: true,
});
expect(body).toEqual({ success: false, error: "Reasoning request could not be completed." });
expect(JSON.stringify(body)).not.toMatch(/provider unavailable|generate|llama3|rawResponse/i);
expect(errorSpy).toHaveBeenCalledTimes(1);
expect(errorSpy).toHaveBeenCalledWith(
"[api/cases/start] error response",
@@ -13,6 +13,10 @@ vi.mock("@/lib/llm/provider.js", () => ({
getProviderModelName: () => "gpt-5.6-terra",
}));
vi.mock("@/lib/supabase/api-auth.js", () => ({
withAuthenticatedApi: (handler) => handler,
}));
// ── Helpers ─────────────────────────────────────────────────
function makeValidGraph() {
@@ -110,10 +114,13 @@ describe("POST /api/cases/synthesis — error cases", () => {
});
it("domain seam throws with statusCode → mapped status", async () => {
mockSynthesize.mockRejectedValue(new Error("Provider failed"));
// Add statusCode property to the error object after creation
const err = Object.assign(new Error("Provider failed"), { statusCode: 502 });
mockSynthesize.mockRejectedValue(err);
const err = Object.assign(
new Error("Ollama /api/generate returned 500 from private host"),
{ statusCode: 502 },
);
mockSynthesize.mockImplementationOnce(async () => {
throw err;
});
const { POST } = await import("@/app/api/cases/synthesis/route.js");
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
@@ -122,10 +129,17 @@ describe("POST /api/cases/synthesis — error cases", () => {
const data = await res.json();
expect(data.success).toBe(false);
expect(data.stage).toBe("provider");
expect(data).toEqual({
success: false,
stage: "provider",
error: "Reasoning request could not be completed.",
});
});
it("domain seam throws without statusCode → 500", async () => {
mockSynthesize.mockRejectedValue(new Error("unknown error"));
mockSynthesize.mockImplementationOnce(async () => {
throw new Error("unknown error");
});
const { POST } = await import("@/app/api/cases/synthesis/route.js");
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
@@ -138,7 +152,9 @@ describe("POST /api/cases/synthesis — error cases", () => {
it("domain seam throws 400 → mapped to 400", async () => {
const err = Object.assign(new Error("Invalid input"), { statusCode: 400 });
mockSynthesize.mockRejectedValue(err);
mockSynthesize.mockImplementationOnce(async () => {
throw err;
});
const { POST } = await import("@/app/api/cases/synthesis/route.js");
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
+170 -2
View File
@@ -360,7 +360,7 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
expect(json.providerExecution).toBeUndefined();
});
it("preserves the 500 provider-failure contract while logging structural diagnostics", async () => {
it("sanitizes generic provider failures while logging structural diagnostics", async () => {
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({
@@ -383,7 +383,7 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
}),
}));
expect(response.status).toBe(500);
await expect(response.json()).resolves.toEqual({ error: "provider failed" });
await expect(response.json()).resolves.toEqual({ error: "Reasoning request could not be completed." });
expect(errorSpy).toHaveBeenCalledWith(
"[api/focused-investigation/deconstruct] provider failure",
expect.objectContaining({ targetNodeId: "node-id", providerApiPath: "/v1/responses" }),
@@ -393,6 +393,174 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
}
});
it("rejects oversized focused answer with 400 and does not reach reasoning seam", async () => {
const generateReconstruction = vi.fn().mockResolvedValue({
response: {},
providerApiPath: "/api/chat",
});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({ generateReconstruction }),
getProviderModelName: () => "configured-model",
}));
const largeAnswer = "x".repeat(10001);
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
targetNodeId: "node-id",
targetLabel: "label",
targetDescription: "description",
centralStatement: "central statement",
question: "question?",
answer: largeAnswer,
}),
}));
expect(response.status).toBe(400);
const json = await response.json();
expect(json.error).toMatch(/answer.*exceeds maximum length/i);
expect(generateReconstruction).not.toHaveBeenCalled();
});
it("accepts focused answer at exact server max (10000) and reaches reasoning seam", async () => {
const generateReconstruction = vi.fn().mockResolvedValue({
response: {
targetNodeId: "node-id",
observations: [],
uncertainties: [],
assumptions: [],
relationships: [],
possibleFollowUpQuestions: [],
},
providerApiPath: "/api/chat",
});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({ generateReconstruction }),
getProviderModelName: () => "configured-model",
}));
const exactMaxAnswer = "x".repeat(10000);
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
targetNodeId: "node-id",
targetLabel: "label",
targetDescription: "description",
centralStatement: "central statement",
question: "question?",
answer: exactMaxAnswer,
}),
}));
expect(response.status).toBe(200);
const json = await response.json();
expect(json.success).toBe(true);
expect(generateReconstruction).toHaveBeenCalled();
});
it("rejects malformed required field (wrong type) with 400", async () => {
const generateReconstruction = vi.fn().mockResolvedValue({
response: {}, providerApiPath: "/api/chat",
});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({ generateReconstruction }),
getProviderModelName: () => "configured-model",
}));
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
targetNodeId: ["not-a-string"],
targetLabel: "label",
targetDescription: "description",
centralStatement: "central statement",
question: "question?",
answer: "answer.",
}),
}));
expect(response.status).toBe(400);
const json = await response.json();
expect(json.error).toMatch(/targetNodeId.*string/i);
expect(generateReconstruction).not.toHaveBeenCalled();
});
it("rejects oversized targetDescription with 400", async () => {
const generateReconstruction = vi.fn().mockResolvedValue({
response: {}, providerApiPath: "/api/chat",
});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({ generateReconstruction }),
getProviderModelName: () => "configured-model",
}));
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
targetNodeId: "node-id",
targetLabel: "label",
targetDescription: "x".repeat(2049),
centralStatement: "central statement",
question: "question?",
answer: "answer.",
}),
}));
expect(response.status).toBe(400);
const json = await response.json();
expect(json.error).toMatch(/targetDescription.*exceeds maximum length/i);
expect(generateReconstruction).not.toHaveBeenCalled();
});
it("rejects malformed centralStatement (number) with 400", async () => {
const generateReconstruction = vi.fn().mockResolvedValue({
response: {}, providerApiPath: "/api/chat",
});
vi.doMock("@/lib/llm/provider", () => ({
getProvider: () => ({ generateReconstruction }),
getProviderModelName: () => "configured-model",
}));
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
targetNodeId: "node-id",
targetLabel: "label",
targetDescription: "description",
centralStatement: 12345,
question: "question?",
answer: "answer.",
}),
}));
expect(response.status).toBe(400);
const json = await response.json();
expect(json.error).toMatch(/centralStatement.*string/i);
expect(generateReconstruction).not.toHaveBeenCalled();
});
it("returns a sanitized 503 when the provider is unavailable", async () => {
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
vi.doMock("@/lib/llm/provider", () => ({