Feature/product platform foundation v0.62 #1
@@ -23,6 +23,7 @@ export async function POST(request) {
|
|||||||
validationErrors: result.validationErrors,
|
validationErrors: result.validationErrors,
|
||||||
diagnostics: result.diagnostics,
|
diagnostics: result.diagnostics,
|
||||||
analysisErrors: result.analysisErrors,
|
analysisErrors: result.analysisErrors,
|
||||||
|
validationIssues: result.validationIssues,
|
||||||
rawResponse: result.rawResponse ?? undefined,
|
rawResponse: result.rawResponse ?? undefined,
|
||||||
},
|
},
|
||||||
{ status },
|
{ status },
|
||||||
|
|||||||
@@ -45,6 +45,12 @@
|
|||||||
- The remaining semantic weakness was substitution: narrower decision prerequisites were preserved, but the full supplied intervention-fit dependency was not. v0.5 now requires preservation of the user's actual dependency level; narrower evidence prerequisites may coexist but may not replace that higher-order contingency.
|
- The remaining semantic weakness was substitution: narrower decision prerequisites were preserved, but the full supplied intervention-fit dependency was not. v0.5 now requires preservation of the user's actual dependency level; narrower evidence prerequisites may coexist but may not replace that higher-order contingency.
|
||||||
- No schema, builder, or orchestrator change was required. Deterministic tests prove this instruction exists only; live compliance remains untested. Next restart point: one live intervention-fit validation, followed by bounded repeatability if it passes.
|
- No schema, builder, or orchestrator change was required. Deterministic tests prove this instruction exists only; live compliance remains untested. Next restart point: one live intervention-fit validation, followed by bounded repeatability if it passes.
|
||||||
|
|
||||||
|
## Reconstruction validation failure observability
|
||||||
|
|
||||||
|
- The latest one-call intervention-fit validation returned HTTP 502: a real structured-output product failure, so intervention-fit semantic compliance was not evaluable.
|
||||||
|
- Previous diagnostics flattened the failure to `reconstruction: Required` and truncated raw evidence. Failure responses now retain exact validation issue paths, codes, messages, available native metadata, and complete diagnostic raw response at the existing validation boundary.
|
||||||
|
- Reasoning prompt, schema acceptance, and provider behaviour remain unchanged; no live calls occurred in this increment. Next restart point: one production-default manufacturing call—evaluate E if it succeeds, or use the new evidence to diagnose the exact structured-output defect if it fails.
|
||||||
|
|
||||||
## Current product architecture
|
## Current product architecture
|
||||||
|
|
||||||
Three distinct routes, not a single page:
|
Three distinct routes, not a single page:
|
||||||
|
|||||||
+5
-3
@@ -91,7 +91,7 @@ export async function analyseScenario(scenario, opts = {}) {
|
|||||||
try {
|
try {
|
||||||
rawResponseStr = JSON.stringify(rawResponse);
|
rawResponseStr = JSON.stringify(rawResponse);
|
||||||
} catch {
|
} catch {
|
||||||
rawResponseStr = String(rawResponse).slice(0, 2000);
|
rawResponseStr = String(rawResponse);
|
||||||
}
|
}
|
||||||
|
|
||||||
const compatibility = normaliseAnalysisResponse(rawResponse);
|
const compatibility = normaliseAnalysisResponse(rawResponse);
|
||||||
@@ -129,7 +129,7 @@ export async function analyseScenario(scenario, opts = {}) {
|
|||||||
|
|
||||||
// ── Neither schema matched — partial failure ───────
|
// ── Neither schema matched — partial failure ───────
|
||||||
return buildPartialResult(
|
return buildPartialResult(
|
||||||
rawResponseStr?.slice(0, 2000),
|
rawResponseStr,
|
||||||
resultV2.error ?? resultV1.error,
|
resultV2.error ?? resultV1.error,
|
||||||
OLLAMA_MODEL,
|
OLLAMA_MODEL,
|
||||||
duration,
|
duration,
|
||||||
@@ -218,6 +218,7 @@ function buildPartialResult(
|
|||||||
compatibility,
|
compatibility,
|
||||||
) {
|
) {
|
||||||
let errors = [];
|
let errors = [];
|
||||||
|
const validationIssues = error?.issues ?? [];
|
||||||
if (error && typeof error.flatten === "function") {
|
if (error && typeof error.flatten === "function") {
|
||||||
errors = error.flatten().fieldErrors
|
errors = error.flatten().fieldErrors
|
||||||
? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [
|
? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [
|
||||||
@@ -233,13 +234,14 @@ function buildPartialResult(
|
|||||||
validationStatus: "invalid",
|
validationStatus: "invalid",
|
||||||
modelName: model,
|
modelName: model,
|
||||||
responseDurationMs: duration,
|
responseDurationMs: duration,
|
||||||
rawResponse: rawResp?.slice(0, 2000),
|
rawResponse: rawResp,
|
||||||
promptVersion: version,
|
promptVersion: version,
|
||||||
inputClassification: null,
|
inputClassification: null,
|
||||||
reconstruction: null,
|
reconstruction: null,
|
||||||
evidence: undefined,
|
evidence: undefined,
|
||||||
nextQuestion: undefined,
|
nextQuestion: undefined,
|
||||||
errors,
|
errors,
|
||||||
|
validationIssues,
|
||||||
...buildCompatibilityDiagnostics(compatibility),
|
...buildCompatibilityDiagnostics(compatibility),
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -371,6 +371,7 @@ export async function startCase(body, dependencies = {}) {
|
|||||||
graphReferenceValidation: null,
|
graphReferenceValidation: null,
|
||||||
}),
|
}),
|
||||||
analysisErrors: analysis.errors ?? undefined,
|
analysisErrors: analysis.errors ?? undefined,
|
||||||
|
validationIssues: analysis.validationIssues ?? undefined,
|
||||||
rawResponse: analysis.rawResponse ?? undefined,
|
rawResponse: analysis.rawResponse ?? undefined,
|
||||||
statusCode: Number(analysis.statusCode) || 502,
|
statusCode: Number(analysis.statusCode) || 502,
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -105,12 +105,23 @@ describe("app/api/cases/start route", () => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
it("returns provider/internal failures as 5xx without stack traces", async () => {
|
it("returns provider/internal failures as 5xx without stack traces", async () => {
|
||||||
|
const rawResponse = `{"reconstruction":{"observedStates":[{"id":"obs-1"${"x".repeat(2500)}}]}}`;
|
||||||
mockStartCase.mockResolvedValue({
|
mockStartCase.mockResolvedValue({
|
||||||
success: false,
|
success: false,
|
||||||
error: "Provider unavailable",
|
error: "Provider unavailable",
|
||||||
diagnostics: { modelName: "llama3" },
|
diagnostics: { modelName: "llama3" },
|
||||||
statusCode: 502,
|
statusCode: 502,
|
||||||
rawResponse: '{"reconstruction":{"summary":""}}',
|
analysisErrors: ["reconstruction: Required"],
|
||||||
|
validationIssues: [
|
||||||
|
{
|
||||||
|
path: ["reconstruction", "observedStates", 2, "description"],
|
||||||
|
code: "invalid_type",
|
||||||
|
message: "Required",
|
||||||
|
expected: "string",
|
||||||
|
received: "undefined",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
rawResponse,
|
||||||
});
|
});
|
||||||
|
|
||||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||||
@@ -125,7 +136,18 @@ describe("app/api/cases/start route", () => {
|
|||||||
expect(response.status).toBe(502);
|
expect(response.status).toBe(502);
|
||||||
const body = await response.json();
|
const body = await response.json();
|
||||||
expect(body).toHaveProperty("rawResponse");
|
expect(body).toHaveProperty("rawResponse");
|
||||||
expect(body.rawResponse).toBe('{"reconstruction":{"summary":""}}');
|
expect(body.rawResponse).toBe(rawResponse);
|
||||||
|
expect(body.rawResponse.length).toBeGreaterThan(2000);
|
||||||
|
expect(body.analysisErrors).toEqual(["reconstruction: Required"]);
|
||||||
|
expect(body.validationIssues).toEqual([
|
||||||
|
expect.objectContaining({
|
||||||
|
path: ["reconstruction", "observedStates", 2, "description"],
|
||||||
|
code: "invalid_type",
|
||||||
|
message: "Required",
|
||||||
|
expected: "string",
|
||||||
|
received: "undefined",
|
||||||
|
}),
|
||||||
|
]);
|
||||||
});
|
});
|
||||||
|
|
||||||
it("returns structured 500 on malformed JSON", async () => {
|
it("returns structured 500 on malformed JSON", async () => {
|
||||||
|
|||||||
@@ -536,6 +536,36 @@ describe("lib/graph/orchestrator startCase", () => {
|
|||||||
expect(result).toHaveProperty("rawResponse");
|
expect(result).toHaveProperty("rawResponse");
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves structured reconstruction validation issues on analysis failure", async () => {
|
||||||
|
const validationIssues = [
|
||||||
|
{
|
||||||
|
path: ["reconstruction", "observedStates", 2, "description"],
|
||||||
|
code: "invalid_type",
|
||||||
|
message: "Required",
|
||||||
|
expected: "string",
|
||||||
|
received: "undefined",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
mockAnalyseScenario.mockResolvedValue({
|
||||||
|
success: false,
|
||||||
|
error: "Scenario analysis failed",
|
||||||
|
errors: ["reconstruction: Required"],
|
||||||
|
validationIssues,
|
||||||
|
rawResponse: "x".repeat(2501),
|
||||||
|
});
|
||||||
|
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||||
|
|
||||||
|
const result = await startCase({ scenario: "Scenario text" });
|
||||||
|
|
||||||
|
expect(result).toMatchObject({
|
||||||
|
success: false,
|
||||||
|
statusCode: 502,
|
||||||
|
analysisErrors: ["reconstruction: Required"],
|
||||||
|
validationIssues,
|
||||||
|
});
|
||||||
|
expect(result.rawResponse).toHaveLength(2501);
|
||||||
|
});
|
||||||
|
|
||||||
it("returns null selectedQuestion when neither analysis nor graph path yields a question", async () => {
|
it("returns null selectedQuestion when neither analysis nor graph path yields a question", async () => {
|
||||||
mockAnalyseScenario.mockResolvedValue(
|
mockAnalyseScenario.mockResolvedValue(
|
||||||
makeAnalysisResult({
|
makeAnalysisResult({
|
||||||
|
|||||||
@@ -192,6 +192,54 @@ describe("analyseScenario compatibility", () => {
|
|||||||
expect(result.nextQuestion).toBeUndefined();
|
expect(result.nextQuestion).toBeUndefined();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("preserves nested validation issues and complete raw output on reconstruction failure", async () => {
|
||||||
|
mockGenerateReconstruction.mockResolvedValue({
|
||||||
|
inputClassification: {
|
||||||
|
primaryType: "unexplained_change",
|
||||||
|
secondaryTypes: [],
|
||||||
|
reasoningModes: [],
|
||||||
|
classificationReason: "reason",
|
||||||
|
confidence: "medium",
|
||||||
|
},
|
||||||
|
reconstruction: {
|
||||||
|
summary: "summary",
|
||||||
|
actors: [],
|
||||||
|
systemsOrObjects: [],
|
||||||
|
expectedStates: [],
|
||||||
|
observedStates: [{ id: "obs-1", confidence: "high" }],
|
||||||
|
differences: [],
|
||||||
|
knownTransitions: [],
|
||||||
|
unexplainedTransitions: [],
|
||||||
|
contradictions: [],
|
||||||
|
importantUnknowns: [],
|
||||||
|
plausibleInterpretations: [],
|
||||||
|
padding: "x".repeat(2500),
|
||||||
|
},
|
||||||
|
evidence: [],
|
||||||
|
nextQuestion: {
|
||||||
|
id: "q1",
|
||||||
|
question: "What changed?",
|
||||||
|
targets: [],
|
||||||
|
reason: "reason",
|
||||||
|
expectedInformationValue: "high",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
const { analyseScenario } = await import("@/lib/analysis.js");
|
||||||
|
const result = await analyseScenario("Scenario text", {
|
||||||
|
promptVersion: "v0.3",
|
||||||
|
});
|
||||||
|
|
||||||
|
expect(result.success).toBe(false);
|
||||||
|
expect(result.rawResponse.length).toBeGreaterThan(2000);
|
||||||
|
expect(result.validationIssues).toContainEqual(expect.objectContaining({
|
||||||
|
path: ["reconstruction", "observedStates", 0, "description"],
|
||||||
|
code: "invalid_type",
|
||||||
|
message: "Required",
|
||||||
|
}));
|
||||||
|
expect(result.errors).toContain("reconstruction: Required");
|
||||||
|
});
|
||||||
|
|
||||||
it("succeeds when reported_claim is the only evidenceType mismatch", async () => {
|
it("succeeds when reported_claim is the only evidenceType mismatch", async () => {
|
||||||
mockGenerateReconstruction.mockResolvedValue({
|
mockGenerateReconstruction.mockResolvedValue({
|
||||||
inputClassification: {
|
inputClassification: {
|
||||||
|
|||||||
Reference in New Issue
Block a user