Compare commits
20
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2e4c624a8a | ||
|
|
781d6a462f | ||
|
|
48ce66dddb | ||
|
|
4affadab4b | ||
|
|
392564ed61 | ||
|
|
72ef175971 | ||
|
|
904aec7616 | ||
|
|
c3de80f203 | ||
|
|
a9bce79658 | ||
|
|
a948910ba8 | ||
|
|
cb77f955ed | ||
|
|
f3cdfce0b0 | ||
|
|
b38a6a9f2e | ||
|
|
02a6ecd0da | ||
|
|
575b8fd971 | ||
|
|
84858107b7 | ||
|
|
0ccc03c111 | ||
|
|
3c1362d8a1 | ||
|
|
79ea2f6824 | ||
|
|
d72c7c5465 |
@@ -34,3 +34,8 @@ Thumbs.db
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
|
||||
# Generated evaluation artifacts (regenerated each run)
|
||||
evaluation-results/
|
||||
provider-debug-results/
|
||||
tests-results/
|
||||
|
||||
+28
-80
@@ -1,101 +1,49 @@
|
||||
import { getConfig } from "@/lib/config";
|
||||
import { getProvider } from "@/lib/llm/provider";
|
||||
import { reconstructionSchema } from "@/lib/reconstruction/schema";
|
||||
|
||||
const MAX_SCENARIO_LENGTH = 10000;
|
||||
import {
|
||||
analyseScenario,
|
||||
PROMPT_VERSIONS,
|
||||
DEFAULT_PROMPT_VERSION,
|
||||
} from "@/lib/analysis";
|
||||
|
||||
export async function POST(request) {
|
||||
const startTime = Date.now();
|
||||
let rawResponse = null;
|
||||
|
||||
try {
|
||||
const body = await request.json();
|
||||
|
||||
|
||||
if (!body.scenario || typeof body.scenario !== "string") {
|
||||
return Response.json(
|
||||
{ error: "Request must include a 'scenario' string field" },
|
||||
{ status: 400 }
|
||||
{ status: 400 },
|
||||
);
|
||||
}
|
||||
|
||||
const trimmed = body.scenario.trim();
|
||||
|
||||
if (trimmed.length === 0) {
|
||||
// Optional prompt version override
|
||||
let promptVersion = DEFAULT_PROMPT_VERSION;
|
||||
if (body.promptVersion && PROMPT_VERSIONS.includes(body.promptVersion)) {
|
||||
promptVersion = body.promptVersion;
|
||||
}
|
||||
|
||||
const result = await analyseScenario(body.scenario, { promptVersion });
|
||||
|
||||
if (!result.success) {
|
||||
return Response.json(
|
||||
{ error: "Scenario cannot be empty" },
|
||||
{ status: 400 }
|
||||
{ ...result, reconstruction: result.reconstruction || null },
|
||||
{ status: Number(result.statusCode) || 500 },
|
||||
);
|
||||
}
|
||||
|
||||
if (trimmed.length > MAX_SCENARIO_LENGTH) {
|
||||
return Response.json(
|
||||
{ error: `Scenario must be under ${MAX_SCENARIO_LENGTH} characters` },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
const configResult = getConfig();
|
||||
if (!configResult.ok) {
|
||||
return Response.json(
|
||||
{ error: "Invalid server configuration" },
|
||||
{ status: 500 }
|
||||
);
|
||||
}
|
||||
|
||||
const { OLLAMA_BASE_URL, OLLAMA_MODEL } = configResult.config;
|
||||
const provider = getProvider();
|
||||
|
||||
// Attempt parse to capture raw for debugging
|
||||
let reconstruction;
|
||||
try {
|
||||
reconstruction = await provider.generateReconstruction(trimmed, OLLAMA_MODEL);
|
||||
} catch (e) {
|
||||
return Response.json(
|
||||
{
|
||||
error: e.message || "Unknown server error",
|
||||
responseDurationMs: Date.now() - startTime,
|
||||
modelName: OLLAMA_MODEL,
|
||||
validationStatus: "invalid",
|
||||
},
|
||||
{ status: 500 }
|
||||
);
|
||||
}
|
||||
|
||||
// Try to stringify for rawResponse display (safe even if it's already an object)
|
||||
try {
|
||||
rawResponse = JSON.stringify(reconstruction);
|
||||
} catch {
|
||||
rawResponse = String(reconstruction).slice(0, 2000);
|
||||
}
|
||||
|
||||
const duration = Date.now() - startTime;
|
||||
|
||||
// Validate with Zod schema
|
||||
const validationResult = reconstructionSchema.safeParse(reconstruction);
|
||||
|
||||
if (!validationResult.success) {
|
||||
return Response.json({
|
||||
reconstruction: null,
|
||||
modelName: OLLAMA_MODEL,
|
||||
responseDurationMs: duration,
|
||||
validationStatus: "invalid",
|
||||
rawResponse: rawResponse?.slice(0, 2000),
|
||||
errors: validationResult.error.issues.map((i) => `${i.path.join(".")}: ${i.message}`),
|
||||
});
|
||||
}
|
||||
|
||||
return Response.json({
|
||||
reconstruction: validationResult.data,
|
||||
modelName: OLLAMA_MODEL,
|
||||
responseDurationMs: duration,
|
||||
validationStatus: "valid",
|
||||
rawResponse: rawResponse?.slice(0, 2000),
|
||||
inputClassification: result.inputClassification,
|
||||
reconstruction: result.reconstruction,
|
||||
evidence: result.evidence,
|
||||
nextQuestion: result.nextQuestion,
|
||||
modelName: result.modelName,
|
||||
responseDurationMs: result.responseDurationMs,
|
||||
validationStatus: result.validationStatus,
|
||||
promptVersion: result.promptVersion,
|
||||
});
|
||||
} catch (e) {
|
||||
const duration = Date.now() - startTime;
|
||||
return Response.json(
|
||||
{ error: e.message || "Unknown server error", responseDurationMs: duration },
|
||||
{ status: 500 }
|
||||
{ error: e.message || "Unknown server error", responseDurationMs: 0 },
|
||||
{ status: 500 },
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
import { startCase } from "@/lib/graph/orchestrator.js";
|
||||
|
||||
export async function POST(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
const result = await startCase(body);
|
||||
|
||||
if (result.success) {
|
||||
return Response.json(result, { status: 200 });
|
||||
}
|
||||
|
||||
const status =
|
||||
result.statusCode === 400
|
||||
? 400
|
||||
: result.statusCode >= 500
|
||||
? result.statusCode
|
||||
: 500;
|
||||
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
error: result.error ?? "Start case failed",
|
||||
validationErrors: result.validationErrors,
|
||||
diagnostics: result.diagnostics,
|
||||
analysisErrors: result.analysisErrors,
|
||||
},
|
||||
{ status },
|
||||
);
|
||||
} catch {
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
error: "Internal server error",
|
||||
},
|
||||
{ status: 500 },
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
import { updateCase } from "@/lib/graph/orchestrator.js";
|
||||
|
||||
function mapFailureStatus(result) {
|
||||
switch (result?.stage) {
|
||||
case "request_validation":
|
||||
case "graph_validation":
|
||||
return 400;
|
||||
case "provider":
|
||||
return 502;
|
||||
case "proposal_validation":
|
||||
case "proposal_compatibility":
|
||||
case "application":
|
||||
return 422;
|
||||
case "result_validation":
|
||||
return 500;
|
||||
default:
|
||||
return 500;
|
||||
}
|
||||
}
|
||||
|
||||
function buildFailureResponse(result) {
|
||||
return {
|
||||
success: false,
|
||||
stage: result?.stage ?? "internal",
|
||||
error: result?.error ?? "Update case failed",
|
||||
validationErrors: result?.validationErrors,
|
||||
graphValidationErrors: result?.graphValidationErrors,
|
||||
proposalErrors: result?.proposalErrors,
|
||||
providerErrors: result?.providerErrors,
|
||||
errors: result?.errors,
|
||||
diagnostics: result?.diagnostics,
|
||||
};
|
||||
}
|
||||
|
||||
export async function POST(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
const result = await updateCase(body, { applyProposal: true });
|
||||
|
||||
if (result.success) {
|
||||
return Response.json(result, { status: 200 });
|
||||
}
|
||||
|
||||
return Response.json(buildFailureResponse(result), {
|
||||
status: mapFailureStatus(result),
|
||||
});
|
||||
} catch (error) {
|
||||
if (error instanceof SyntaxError) {
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid JSON request body",
|
||||
},
|
||||
{ status: 400 },
|
||||
);
|
||||
}
|
||||
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
stage: "internal",
|
||||
error: "Internal server error",
|
||||
},
|
||||
{ status: 500 },
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,3 +1,5 @@
|
||||
import React from "react";
|
||||
|
||||
const ValidationIndicator = ({ status }) => {
|
||||
const styles = {
|
||||
valid: "text-green-600",
|
||||
@@ -10,17 +12,83 @@ const ValidationIndicator = ({ status }) => {
|
||||
invalid: "❌ Validation failed",
|
||||
};
|
||||
return (
|
||||
<div className={`flex items-center gap-2 ${styles[status] || "text-gray-500"}`}>
|
||||
<div
|
||||
className={`flex items-center gap-2 ${styles[status] || "text-gray-500"}`}
|
||||
>
|
||||
<span className="font-medium">{labels[status] || status}</span>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
const validationIcons = {
|
||||
valid: "✅",
|
||||
partial: "⚠️",
|
||||
invalid: "❌",
|
||||
};
|
||||
|
||||
export default function DiagnosticsView({ result }) {
|
||||
if (!result) return null;
|
||||
|
||||
const diagnostics = result.diagnostics || result;
|
||||
|
||||
const metrics = [
|
||||
{ label: "Model", value: result.modelName || "?" },
|
||||
{ label: "Duration", value: result.responseDurationMs != null ? `${result.responseDurationMs}ms` : "?" },
|
||||
{ label: "Validation", value: <ValidationIndicator status={result.validationStatus || "invalid"} /> },
|
||||
{ label: "Model", value: diagnostics.modelName || result.modelName || "?" },
|
||||
{ label: "Provider", value: "Ollama" },
|
||||
{
|
||||
label: "Prompt version",
|
||||
value: diagnostics.promptVersion || result.promptVersion || "?",
|
||||
},
|
||||
{
|
||||
label: "Duration",
|
||||
value:
|
||||
diagnostics.responseDurationMs != null
|
||||
? `${diagnostics.responseDurationMs}ms`
|
||||
: "?",
|
||||
},
|
||||
{
|
||||
label: "Validation",
|
||||
value: (
|
||||
<ValidationIndicator
|
||||
status={diagnostics.validationStatus || result.validationStatus || "invalid"}
|
||||
/>
|
||||
),
|
||||
},
|
||||
{
|
||||
label: "Node count",
|
||||
value:
|
||||
diagnostics.nodeCount != null
|
||||
? diagnostics.nodeCount
|
||||
: diagnostics.graphNodeCount != null
|
||||
? diagnostics.graphNodeCount
|
||||
: "?",
|
||||
},
|
||||
{
|
||||
label: "Edge count",
|
||||
value:
|
||||
diagnostics.edgeCount != null
|
||||
? diagnostics.edgeCount
|
||||
: diagnostics.graphEdgeCount != null
|
||||
? diagnostics.graphEdgeCount
|
||||
: "?",
|
||||
},
|
||||
{
|
||||
label: "Graph references",
|
||||
value:
|
||||
diagnostics.graphReferenceValidation == null
|
||||
? "?"
|
||||
: diagnostics.graphReferenceValidation.valid
|
||||
? `${validationIcons.valid} valid`
|
||||
: `${validationIcons.invalid} invalid`,
|
||||
},
|
||||
];
|
||||
|
||||
const errors = [
|
||||
...(result.errors || []),
|
||||
...(result.validationErrors || []),
|
||||
...(result.graphValidationErrors || []),
|
||||
...(result.proposalErrors || []),
|
||||
...(result.providerErrors || []),
|
||||
...(result.analysisErrors || []),
|
||||
];
|
||||
|
||||
return (
|
||||
@@ -35,16 +103,32 @@ export default function DiagnosticsView({ result }) {
|
||||
))}
|
||||
</dl>
|
||||
|
||||
{/* Collapsed raw output for debugging */}
|
||||
{result.rawResponse && (
|
||||
<details className="mt-4">
|
||||
<summary className="cursor-pointer text-xs text-gray-500 underline hover:text-gray-700">
|
||||
View raw model response
|
||||
View raw model response (
|
||||
{(result.rawResponse?.length || 0).toLocaleString()} chars)
|
||||
</summary>
|
||||
<pre className="mt-2 max-h-60 overflow-auto rounded bg-gray-900 px-3 py-2 text-xs leading-relaxed text-green-400">
|
||||
{result.rawResponse}
|
||||
</pre>
|
||||
</details>
|
||||
)}
|
||||
|
||||
{/* Errors if present */}
|
||||
{errors.length > 0 && (
|
||||
<details className="mt-3">
|
||||
<summary className="cursor-pointer text-xs text-red-500 underline hover:text-red-700">
|
||||
Validation errors ({errors.length})
|
||||
</summary>
|
||||
<ul className="mt-1 space-y-0.5 text-xs text-red-600">
|
||||
{errors.map((err, i) => (
|
||||
<li key={i}>{typeof err === "string" ? err : err?.message || JSON.stringify(err)}</li>
|
||||
))}
|
||||
</ul>
|
||||
</details>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
import React from "react";
|
||||
|
||||
function ListSection({ title, items, renderItem = (item) => item }) {
|
||||
if (!items?.length) return null;
|
||||
|
||||
return (
|
||||
<section className="rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-gray-800">{title}</h3>
|
||||
<ul className="space-y-1 text-sm text-gray-700">
|
||||
{items.map((item, index) => (
|
||||
<li key={`${title}-${index}`}>{renderItem(item)}</li>
|
||||
))}
|
||||
</ul>
|
||||
</section>
|
||||
);
|
||||
}
|
||||
|
||||
export default function GraphUpdateView({ updateResult }) {
|
||||
if (!updateResult?.proposal) return null;
|
||||
|
||||
const {
|
||||
resolvedUnknownNodeIds,
|
||||
affectedNodeIds,
|
||||
previousActiveUnknownNodeId,
|
||||
newActiveUnknownNodeId,
|
||||
selectedQuestion,
|
||||
changesApplied,
|
||||
proposal,
|
||||
previousSituationGraph,
|
||||
updatedSituationGraph,
|
||||
} = updateResult;
|
||||
|
||||
const newlySurfacedUnknownNodeIds = (proposal.addedNodes || [])
|
||||
.filter((node) => node.kind === "unknown")
|
||||
.map((node) => node.id);
|
||||
|
||||
const previousNodesById = new Map(
|
||||
(previousSituationGraph?.nodes || []).map((node) => [node.id, node]),
|
||||
);
|
||||
const updatedNodesById = new Map(
|
||||
(updatedSituationGraph?.nodes || []).map((node) => [node.id, node]),
|
||||
);
|
||||
const proposalUpdatesByNodeId = new Map(
|
||||
(proposal.updatedNodes || []).map((update) => [update.nodeId, update]),
|
||||
);
|
||||
|
||||
function resolveNodePresentation(nodeId) {
|
||||
const previousNode = previousNodesById.get(nodeId) || null;
|
||||
const updatedNode = updatedNodesById.get(nodeId) || null;
|
||||
const node = updatedNode || previousNode;
|
||||
const update = proposalUpdatesByNodeId.get(nodeId) || null;
|
||||
|
||||
if (!node) {
|
||||
return (
|
||||
<div className="space-y-1">
|
||||
<div className="font-medium text-gray-900">Unknown node (ID: {nodeId})</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="space-y-1">
|
||||
<div className="font-medium text-gray-900">{node.label}</div>
|
||||
<div className="text-xs text-gray-600">
|
||||
{node.kind} · {node.confidence}
|
||||
</div>
|
||||
{(update?.previousStatus || update?.newStatus || node.status) && (
|
||||
<div className="text-xs text-gray-700">
|
||||
{update?.previousStatus ? `Previous status: ${update.previousStatus}` : null}
|
||||
{update?.previousStatus && update?.newStatus ? " → " : null}
|
||||
{update?.newStatus
|
||||
? `New status: ${update.newStatus}`
|
||||
: !update?.previousStatus
|
||||
? `Status: ${node.status}`
|
||||
: null}
|
||||
</div>
|
||||
)}
|
||||
{update?.reason && <div className="text-xs text-gray-700">{update.reason}</div>}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function resolveActiveUnknown(nodeId) {
|
||||
if (!nodeId) return null;
|
||||
|
||||
const node = updatedNodesById.get(nodeId) || previousNodesById.get(nodeId);
|
||||
if (!node) {
|
||||
return `Unknown node (ID: ${nodeId})`;
|
||||
}
|
||||
|
||||
return `${node.label} · ${node.status} · ${node.confidence}`;
|
||||
}
|
||||
|
||||
const changeItems = [
|
||||
changesApplied?.addedNodeCount
|
||||
? `${changesApplied.addedNodeCount} node(s) added`
|
||||
: null,
|
||||
changesApplied?.updatedNodeCount
|
||||
? `${changesApplied.updatedNodeCount} node(s) updated`
|
||||
: null,
|
||||
changesApplied?.addedEdgeCount
|
||||
? `${changesApplied.addedEdgeCount} edge(s) added`
|
||||
: null,
|
||||
changesApplied?.removedEdgeCount
|
||||
? `${changesApplied.removedEdgeCount} edge(s) removed`
|
||||
: null,
|
||||
changesApplied?.resolvedUnknownCount
|
||||
? `${changesApplied.resolvedUnknownCount} unknown(s) resolved`
|
||||
: null,
|
||||
].filter(Boolean);
|
||||
|
||||
return (
|
||||
<div className="space-y-4">
|
||||
<section className="rounded-lg border border-blue-200 bg-blue-50 p-4">
|
||||
<h2 className="mb-2 text-base font-semibold text-blue-900">
|
||||
Graph update applied
|
||||
</h2>
|
||||
<div className="grid gap-2 text-sm text-blue-950 sm:grid-cols-2">
|
||||
{previousActiveUnknownNodeId && (
|
||||
<div>
|
||||
<span className="font-medium">Previous active unknown:</span>{" "}
|
||||
{resolveActiveUnknown(previousActiveUnknownNodeId)}
|
||||
</div>
|
||||
)}
|
||||
{newActiveUnknownNodeId && (
|
||||
<div>
|
||||
<span className="font-medium">New active unknown:</span>{" "}
|
||||
{resolveActiveUnknown(newActiveUnknownNodeId)}
|
||||
</div>
|
||||
)}
|
||||
{selectedQuestion?.question && (
|
||||
<div>
|
||||
<span className="font-medium">Next question:</span>{" "}
|
||||
{selectedQuestion.question}
|
||||
</div>
|
||||
)}
|
||||
{!selectedQuestion?.question && !newActiveUnknownNodeId && previousActiveUnknownNodeId && (
|
||||
<div>
|
||||
<span className="font-medium">Next question status:</span> No next question selected yet.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<ListSection
|
||||
title="Resolved unknowns"
|
||||
items={resolvedUnknownNodeIds}
|
||||
renderItem={resolveNodePresentation}
|
||||
/>
|
||||
<ListSection
|
||||
title="Newly surfaced unknowns"
|
||||
items={newlySurfacedUnknownNodeIds}
|
||||
renderItem={resolveNodePresentation}
|
||||
/>
|
||||
<ListSection
|
||||
title="Affected nodes"
|
||||
items={affectedNodeIds}
|
||||
renderItem={resolveNodePresentation}
|
||||
/>
|
||||
<ListSection title="Applied changes" items={changeItems} />
|
||||
|
||||
<details className="rounded-lg border border-gray-200 bg-gray-50 p-4">
|
||||
<summary className="cursor-pointer text-sm font-medium text-gray-700 underline">
|
||||
Proposal details
|
||||
</summary>
|
||||
<pre className="mt-3 overflow-auto rounded bg-gray-900 p-3 text-xs text-green-400">
|
||||
{JSON.stringify(proposal, null, 2)}
|
||||
</pre>
|
||||
<pre className="mt-3 overflow-auto rounded bg-gray-900 p-3 text-xs text-green-400">
|
||||
{JSON.stringify(
|
||||
{
|
||||
previousActiveUnknownNodeId,
|
||||
newActiveUnknownNodeId,
|
||||
resolvedUnknownNodeIds,
|
||||
affectedNodeIds,
|
||||
},
|
||||
null,
|
||||
2,
|
||||
)}
|
||||
</pre>
|
||||
</details>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,15 +1,8 @@
|
||||
const categoryLabels = {
|
||||
observations: "Direct Observations",
|
||||
reportedClaims: "Reported Claims",
|
||||
assumptions: "Unsupported Assumptions",
|
||||
entities: "Entities",
|
||||
transitions: "Transitions",
|
||||
expectedButMissing: "Expected But Missing",
|
||||
presentButUnexpected: "Present But Unexpected",
|
||||
contradictions: "Contradictions",
|
||||
openUncertainties: "Open Uncertainties",
|
||||
};
|
||||
"use client";
|
||||
|
||||
import { useMemo } from "react";
|
||||
|
||||
// ── Confidence badge (shared) ────────────────────────
|
||||
const confidenceColor = {
|
||||
low: "text-red-600 bg-red-50 border-red-200",
|
||||
medium: "text-yellow-700 bg-yellow-50 border-yellow-200",
|
||||
@@ -17,54 +10,401 @@ const confidenceColor = {
|
||||
};
|
||||
|
||||
const ConfidenceBadge = ({ level }) => (
|
||||
<span className={`inline-block rounded-full border px-2 py-0.5 text-xs font-medium ${confidenceColor[level] || "text-gray-600 bg-gray-100"}`}>
|
||||
<span
|
||||
className={`inline-block rounded-full border px-2 py-0.5 text-xs font-medium ${confidenceColor[level] || "text-gray-600 bg-gray-100"}`}
|
||||
>
|
||||
{level}
|
||||
</span>
|
||||
);
|
||||
|
||||
function ItemList({ items, renderExtra }) {
|
||||
if (!items?.length) return <p className="text-sm italic text-gray-400">None identified</p>;
|
||||
|
||||
// ── Evidence type labels (shared) ───────────────────
|
||||
const evidenceTypeLabels = {
|
||||
direct_observation: "Direct Observation",
|
||||
reported_statement: "Reported Statement",
|
||||
interpretation: "Interpretation",
|
||||
assumption: "Assumption",
|
||||
inferred_relationship: "Inferred Relationship",
|
||||
};
|
||||
|
||||
const importanceColors = {
|
||||
incidental: "text-gray-500 bg-gray-50 border-gray-200",
|
||||
supporting: "text-blue-700 bg-blue-50 border-blue-200",
|
||||
important: "text-orange-700 bg-orange-50 border-orange-200",
|
||||
critical: "text-red-800 bg-red-50 border-red-300 font-semibold",
|
||||
};
|
||||
|
||||
const importanceLabels = {
|
||||
incidental: "Incidental",
|
||||
supporting: "Supporting",
|
||||
important: "Important",
|
||||
critical: "Critical",
|
||||
};
|
||||
|
||||
// ── Input classification display ────────────────────
|
||||
function ClassificationDisplay({ classification }) {
|
||||
if (!classification) return null;
|
||||
const p = classification.primaryType || classification.primary_type;
|
||||
const sec =
|
||||
classification.secondaryTypes || classification.secondary_types || [];
|
||||
const modes =
|
||||
classification.reasoningModes || classification.reasoning_modes || [];
|
||||
|
||||
// Normalize camelCase to snake_case for display if needed
|
||||
const primaryLabel = String(p)
|
||||
.replace(/_/g, " ")
|
||||
.replace(/\b\w/g, (c) => c.toUpperCase());
|
||||
const secLabels = sec.map((s) =>
|
||||
s.replace(/_/g, " ").replace(/\b\w/g, (c) => c.toUpperCase()),
|
||||
);
|
||||
const modeLabels = modes.map((m) =>
|
||||
m.replace(/_/g, " ").replace(/\b\w/g, (c) => c.toUpperCase()),
|
||||
);
|
||||
|
||||
return (
|
||||
<ul className="space-y-2">
|
||||
{items.map((item) => (
|
||||
<li key={item.id} className="rounded border border-gray-200 bg-white px-3 py-2 text-sm">
|
||||
<div className="flex items-center gap-2">
|
||||
<span className="font-mono text-xs text-gray-400">#{item.id}</span>
|
||||
<ConfidenceBadge level={item.confidence} />
|
||||
</div>
|
||||
<p className="mt-1">{item.description}</p>
|
||||
{renderExtra && renderExtra(item)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
<div className="rounded-lg border border-blue-200 bg-blue-50 p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-blue-700">
|
||||
Input Classification
|
||||
</h3>
|
||||
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1.5 text-sm">
|
||||
<dt className="text-blue-500">Primary type</dt>
|
||||
<dd className="font-medium">{primaryLabel}</dd>
|
||||
{secLabels.length > 0 && (
|
||||
<>
|
||||
<dt className="text-blue-500 pt-1">Secondary types</dt>
|
||||
<dd>{secLabels.join(" · ")}</dd>
|
||||
</>
|
||||
)}
|
||||
{modeLabels.length > 0 && (
|
||||
<>
|
||||
<dt className="text-blue-500 pt-1">Reasoning modes</dt>
|
||||
<dd>{modeLabels.join(" · ")}</dd>
|
||||
</>
|
||||
)}
|
||||
<dt className="text-blue-500 pt-1">Classification reason</dt>
|
||||
<dd className="italic">
|
||||
{classification.classificationReason ||
|
||||
classification.classification_reason}
|
||||
</dd>
|
||||
<dt className="text-blue-500 pt-1">Confidence</dt>
|
||||
<dd>
|
||||
<ConfidenceBadge level={classification.confidence} />
|
||||
</dd>
|
||||
</dl>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Reconstruction summary ──────────────────────────
|
||||
function SummaryDisplay({ reconstruction }) {
|
||||
if (!reconstruction?.summary) return null;
|
||||
const summary = reconstruction.summary || reconstruction.Summary;
|
||||
return (
|
||||
<div className="rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-gray-600">
|
||||
Reconstruction Summary
|
||||
</h3>
|
||||
<p className="text-sm leading-relaxed">{summary}</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Generic item list (used for multiple sections) ──
|
||||
function ItemList({ title, items, renderExtra }) {
|
||||
const count = items?.length;
|
||||
if (!count) return null; // hide empty sections entirely
|
||||
|
||||
const itemsArr = Array.isArray(items) ? items : [items];
|
||||
|
||||
return (
|
||||
<div className="mb-4 rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-gray-600">
|
||||
{title} ({count})
|
||||
</h3>
|
||||
<ul className="space-y-2">
|
||||
{itemsArr.map((item, idx) => (
|
||||
<li
|
||||
key={item.id || `${title}-${idx}`}
|
||||
className="rounded border border-gray-200 bg-white px-3 py-2 text-sm"
|
||||
>
|
||||
<div className="flex items-center gap-2">
|
||||
{item.id && (
|
||||
<span className="font-mono text-xs text-gray-400">
|
||||
#{item.id}
|
||||
</span>
|
||||
)}
|
||||
{item.confidence && <ConfidenceBadge level={item.confidence} />}
|
||||
{item.importance && (
|
||||
<span
|
||||
className={`inline-block rounded-full border px-2 py-0.5 text-xs font-medium ${importanceColors[item.importance] || "text-gray-600 bg-gray-100"}`}
|
||||
>
|
||||
{importanceLabels[item.importance]}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
<p className="mt-1">{item.description}</p>
|
||||
{renderExtra && renderExtra(item)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Plausible interpretations ───────────────────────
|
||||
function InterpretationsDisplay({ interpretations }) {
|
||||
if (!interpretations?.length) return null;
|
||||
const arr = Array.isArray(interpretations)
|
||||
? interpretations
|
||||
: [interpretations];
|
||||
|
||||
return (
|
||||
<div className="mb-4 rounded-lg border border-indigo-200 bg-indigo-50 p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-indigo-700">
|
||||
Plausible Interpretations ({arr.length})
|
||||
</h3>
|
||||
<ul className="space-y-3">
|
||||
{arr.map((interp, idx) => (
|
||||
<li
|
||||
key={interp.id || `${idx}`}
|
||||
className="rounded border border-indigo-200 bg-white px-3 py-2.5 text-sm leading-relaxed"
|
||||
>
|
||||
<div className="flex items-center gap-2 mb-1">
|
||||
<span className="font-medium text-indigo-600">
|
||||
{interp.description}
|
||||
</span>
|
||||
{interp.confidence && (
|
||||
<ConfidenceBadge level={interp.confidence} />
|
||||
)}
|
||||
</div>
|
||||
{interp.supportingEvidenceIds?.length > 0 && (
|
||||
<p className="text-xs text-gray-500">
|
||||
Supporting evidence: {interp.supportingEvidenceIds.join(", ")}
|
||||
</p>
|
||||
)}
|
||||
{interp.assumptionsRequired?.length > 0 && (
|
||||
<p className="text-xs italic text-gray-500">
|
||||
Requires assumptions: {interp.assumptionsRequired.join("; ")}
|
||||
</p>
|
||||
)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Next question (prominent) ───────────────────────
|
||||
function NextQuestionDisplay({ question }) {
|
||||
if (!question?.question) return null;
|
||||
const q = question.question || question.Question;
|
||||
const targets = question.targets || question.Targets || [];
|
||||
const reason = question.reason || question.Reason || "";
|
||||
const value =
|
||||
question.expectedInformationValue ||
|
||||
question.expected_information_value ||
|
||||
"medium";
|
||||
|
||||
const valueLabel =
|
||||
{ low: "Low", medium: "Medium", high: "High" }[value] || "Medium";
|
||||
const valueColor =
|
||||
{
|
||||
low: "bg-yellow-100 text-yellow-800",
|
||||
medium: "bg-blue-100 text-blue-800",
|
||||
high: "bg-green-100 text-green-800",
|
||||
}[value] || "";
|
||||
|
||||
return (
|
||||
<div className="rounded-lg border-2 border-green-300 bg-green-50 p-5">
|
||||
<div className="flex items-center gap-2 mb-2">
|
||||
<h3 className="text-sm font-bold text-green-800">Next Question</h3>
|
||||
<span
|
||||
className={`rounded-full px-2 py-0.5 text-xs font-medium ${valueColor}`}
|
||||
>
|
||||
{valueLabel} value
|
||||
</span>
|
||||
</div>
|
||||
<p className="mb-2 text-base font-medium text-gray-900">{q}</p>
|
||||
{targets.length > 0 && (
|
||||
<p className="text-sm text-gray-600">Targets: {targets.join(", ")}</p>
|
||||
)}
|
||||
{reason && (
|
||||
<p className="text-sm italic text-gray-500">Because: {reason}</p>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Evidence list ───────────────────────────────────
|
||||
function EvidenceDisplay({ evidence }) {
|
||||
if (!evidence?.length) return null;
|
||||
const arr = Array.isArray(evidence) ? evidence : [evidence];
|
||||
|
||||
const evidenceLabels = {
|
||||
direct_observation: "👁 Direct Observation",
|
||||
reported_statement: "🗣 Reported Statement",
|
||||
interpretation: "💡 Interpretation",
|
||||
assumption: "❓ Assumption",
|
||||
inferred_relationship: "🔗 Inferred Relationship",
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="mb-4 rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-2 text-sm font-semibold text-gray-600">
|
||||
Supporting Evidence ({arr.length})
|
||||
</h3>
|
||||
<ul className="space-y-2">
|
||||
{arr.map((item, idx) => (
|
||||
<li
|
||||
key={item.id || `${idx}`}
|
||||
className="rounded border border-gray-200 bg-white px-3 py-2 text-sm leading-relaxed"
|
||||
>
|
||||
<div className="flex items-center gap-2 mb-0.5 flex-wrap">
|
||||
{item.id && (
|
||||
<span className="font-mono text-xs text-gray-400">
|
||||
#{item.id}
|
||||
</span>
|
||||
)}
|
||||
<span
|
||||
className={`inline-block rounded px-1.5 py-0.5 text-[10px] font-medium ${importanceColors[item.importance] || "text-gray-600 bg-gray-100"}`}
|
||||
>
|
||||
{importanceLabels[item.importance]}
|
||||
</span>
|
||||
<span className="inline-block rounded px-1.5 py-0.5 text-[10px] font-medium bg-gray-100 text-gray-700">
|
||||
{evidenceLabels[item.evidenceType] || item.evidenceType}
|
||||
</span>
|
||||
{item.confidence && <ConfidenceBadge level={item.confidence} />}
|
||||
</div>
|
||||
<p className="text-sm">{item.description}</p>
|
||||
{(item.source || item.attribution) && (
|
||||
<p className="mt-0.5 text-xs text-gray-400">
|
||||
Source: {item.source || item.attribution}
|
||||
</p>
|
||||
)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Main component ──────────────────────────────────
|
||||
export default function ReconstructionView({ reconstruction, partial }) {
|
||||
// Handle both v0.2 direct object and wrapped result formats
|
||||
const data = reconstruction;
|
||||
|
||||
if (partial) {
|
||||
return (
|
||||
<div className="rounded-lg border border-yellow-300 bg-yellow-50 px-4 py-3 text-sm text-yellow-800">
|
||||
⚠ Partial result — some fields failed validation. Showing what was accepted.
|
||||
⚠ Partial result — some fields failed validation. Showing what was
|
||||
accepted.
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
const categories = Object.entries(categoryLabels).map(([key, label]) => ({
|
||||
key,
|
||||
label,
|
||||
items: reconstruction[key],
|
||||
}));
|
||||
|
||||
return (
|
||||
<div className="space-y-1">
|
||||
<h2 className="mb-3 text-lg font-semibold">Reconstruction</h2>
|
||||
{categories.map(({ key, label, items }) => (
|
||||
<div key={key} className="mb-4 rounded border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-2 text-sm font-medium text-gray-600">{label}</h3>
|
||||
<ItemList items={items} />
|
||||
</div>
|
||||
))}
|
||||
<div className="space-y-4">
|
||||
{/* Classification first */}
|
||||
{data.inputClassification && (
|
||||
<ClassificationDisplay classification={data.inputClassification} />
|
||||
)}
|
||||
|
||||
{/* Summary */}
|
||||
{data.reconstruction?.summary && (
|
||||
<SummaryDisplay reconstruction={data.reconstruction} />
|
||||
)}
|
||||
|
||||
{/* Key differences */}
|
||||
{data.reconstruction?.differences && (
|
||||
<ItemList
|
||||
title="Key Differences"
|
||||
items={data.reconstruction.differences}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Unexplained transitions */}
|
||||
{data.reconstruction?.unexplainedTransitions &&
|
||||
data.reconstruction.unexplainedTransitions.length > 0 && (
|
||||
<ItemList
|
||||
title="Unexplained Transitions"
|
||||
items={data.reconstruction.unexplainedTransitions}
|
||||
renderExtra={(i) =>
|
||||
i.entity && (
|
||||
<p className="mt-1 text-xs text-gray-500">Entity: {i.entity}</p>
|
||||
)
|
||||
}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Contradictions */}
|
||||
{data.reconstruction?.contradictions &&
|
||||
data.reconstruction.contradictions.length > 0 && (
|
||||
<ItemList
|
||||
title="Contradictions"
|
||||
items={data.reconstruction.contradictions}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Important unknowns */}
|
||||
{data.reconstruction?.importantUnknowns &&
|
||||
data.reconstruction.importantUnknowns.length > 0 && (
|
||||
<ItemList
|
||||
title="Important Unknowns"
|
||||
items={data.reconstruction.importantUnknowns}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Plausible interpretations */}
|
||||
{data.reconstruction?.plausibleInterpretations &&
|
||||
data.reconstruction.plausibleInterpretations.length > 0 && (
|
||||
<InterpretationsDisplay
|
||||
interpretations={data.reconstruction.plausibleInterpretations}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Secondary reconstruction categories (actors, systems, etc.) */}
|
||||
{data.reconstruction?.actors && data.reconstruction.actors.length > 0 && (
|
||||
<ItemList title="Actors" items={data.reconstruction.actors} />
|
||||
)}
|
||||
{data.reconstruction?.systemsOrObjects &&
|
||||
data.reconstruction.systemsOrObjects.length > 0 && (
|
||||
<ItemList
|
||||
title="Systems / Objects"
|
||||
items={data.reconstruction.systemsOrObjects}
|
||||
/>
|
||||
)}
|
||||
{data.reconstruction?.expectedStates &&
|
||||
data.reconstruction.expectedStates.length > 0 && (
|
||||
<ItemList
|
||||
title="Expected States"
|
||||
items={data.reconstruction.expectedStates}
|
||||
/>
|
||||
)}
|
||||
{data.reconstruction?.observedStates &&
|
||||
data.reconstruction.observedStates.length > 0 && (
|
||||
<ItemList
|
||||
title="Observed States"
|
||||
items={data.reconstruction.observedStates}
|
||||
/>
|
||||
)}
|
||||
{data.reconstruction?.knownTransitions &&
|
||||
data.reconstruction.knownTransitions.length > 0 && (
|
||||
<ItemList
|
||||
title="Known Transitions"
|
||||
items={data.reconstruction.knownTransitions}
|
||||
renderExtra={(i) => (
|
||||
<div className="mt-1 text-xs text-gray-500">
|
||||
{i.entity && <span>Entity: {i.entity} · </span>}
|
||||
From “{i.previousState}” → To “{i.currentState}” ("{i.explanationStatus}")
|
||||
</div>
|
||||
)}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Next question — prominent */}
|
||||
<NextQuestionDisplay question={data.nextQuestion} />
|
||||
|
||||
{/* Evidence */}
|
||||
{data.evidence && <EvidenceDisplay evidence={data.evidence} />}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
+295
-44
@@ -1,37 +1,168 @@
|
||||
"use client";
|
||||
|
||||
import React from "react";
|
||||
import { useState, useRef } from "react";
|
||||
import ReconstructionView from "@/components/reconstruction-view";
|
||||
import DiagnosticsView from "@/components/diagnostics-view";
|
||||
import GraphUpdateView from "@/components/graph-update-view";
|
||||
import SituationGraphView from "@/components/situation-graph-view";
|
||||
|
||||
const MAX_LENGTH = 10000;
|
||||
|
||||
export async function submitScenarioForStartCase(fetchImpl, scenario) {
|
||||
return fetchImpl("/api/cases/start", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ scenario }),
|
||||
});
|
||||
}
|
||||
|
||||
export async function submitAnswerForUpdateCase(
|
||||
fetchImpl,
|
||||
{ situationGraph, previousQuestion, answer },
|
||||
) {
|
||||
if (!answer?.trim()) {
|
||||
return {
|
||||
ok: false,
|
||||
skipped: true,
|
||||
data: {
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Please enter an answer before updating.",
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
const response = await fetchImpl("/api/cases/update", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ situationGraph, previousQuestion, answer }),
|
||||
});
|
||||
|
||||
return {
|
||||
ok: response.ok,
|
||||
skipped: false,
|
||||
data: await response.json(),
|
||||
};
|
||||
}
|
||||
|
||||
function normaliseStartResult(data) {
|
||||
return {
|
||||
...data,
|
||||
selectedQuestion:
|
||||
typeof data?.selectedQuestion === "string"
|
||||
? data.selectedQuestion
|
||||
: data?.selectedQuestion?.question ?? null,
|
||||
newlySurfacedNodeIds: data?.newlySurfacedNodeIds ?? [],
|
||||
};
|
||||
}
|
||||
|
||||
function normaliseUpdateSelectedQuestion(selectedQuestion) {
|
||||
if (!selectedQuestion) return null;
|
||||
if (typeof selectedQuestion === "string") return selectedQuestion;
|
||||
return selectedQuestion.question ?? null;
|
||||
}
|
||||
|
||||
export function ScenarioResultPanels({ status, result }) {
|
||||
if (!result) return null;
|
||||
|
||||
const hasGraph = Boolean(result.situationGraph);
|
||||
const hasQuestion = Boolean(result.selectedQuestion?.question);
|
||||
const hasDiagnostics = Boolean(result.diagnostics);
|
||||
|
||||
return (
|
||||
<>
|
||||
{status === "error" && (
|
||||
<div className="space-y-3">
|
||||
{result.error && (
|
||||
<div className="rounded-lg border border-red-300 bg-red-50 px-4 py-3 text-sm text-red-700 whitespace-pre-wrap">
|
||||
Error: {result.error}
|
||||
</div>
|
||||
)}
|
||||
{!hasGraph && !hasQuestion && (
|
||||
<div className="rounded-lg border border-yellow-300 bg-yellow-50 px-4 py-2 text-sm text-yellow-800">
|
||||
Validation failed — no structured graph output was produced.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{(status === "success" || hasGraph || hasQuestion) && (
|
||||
<SituationGraphView
|
||||
situationGraph={result.situationGraph}
|
||||
selectedQuestion={result.selectedQuestion}
|
||||
newlySurfacedNodeIds={result.newlySurfacedNodeIds}
|
||||
/>
|
||||
)}
|
||||
|
||||
{hasDiagnostics && <DiagnosticsView result={result} />}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
export function UpdateErrorPanel({ updateError }) {
|
||||
if (!updateError) return null;
|
||||
|
||||
const errors = [
|
||||
...(updateError.errors || []),
|
||||
...(updateError.validationErrors || []),
|
||||
...(updateError.graphValidationErrors || []),
|
||||
...(updateError.proposalErrors || []),
|
||||
...(updateError.providerErrors || []),
|
||||
];
|
||||
|
||||
return (
|
||||
<div className="space-y-3">
|
||||
<div className="rounded-lg border border-red-300 bg-red-50 px-4 py-3 text-sm text-red-700 whitespace-pre-wrap">
|
||||
Update error: {updateError.error}
|
||||
</div>
|
||||
{errors.length > 0 && (
|
||||
<details className="rounded-lg border border-red-200 bg-red-50 px-4 py-3">
|
||||
<summary className="cursor-pointer text-sm font-medium text-red-700 underline">
|
||||
Update details ({errors.length})
|
||||
</summary>
|
||||
<ul className="mt-2 space-y-1 text-sm text-red-700">
|
||||
{errors.map((item, index) => (
|
||||
<li key={index}>
|
||||
{typeof item === "string" ? item : item?.message || JSON.stringify(item)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</details>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export default function ScenarioForm() {
|
||||
const [scenario, setScenario] = useState("");
|
||||
const [status, setStatus] = useState("idle"); // idle | loading | error | success
|
||||
const [result, setResult] = useState(null);
|
||||
const [answer, setAnswer] = useState("");
|
||||
const [updateStatus, setUpdateStatus] = useState("idle"); // idle | loading | error | success
|
||||
const [updateError, setUpdateError] = useState(null);
|
||||
const [updateResult, setUpdateResult] = useState(null);
|
||||
const textareaRef = useRef(null);
|
||||
|
||||
const handleSubmit = async (e) => {
|
||||
e.preventDefault();
|
||||
setStatus("loading");
|
||||
setResult(null);
|
||||
setAnswer("");
|
||||
setUpdateStatus("idle");
|
||||
setUpdateError(null);
|
||||
setUpdateResult(null);
|
||||
|
||||
try {
|
||||
const res = await fetch("/api/analyse", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ scenario }),
|
||||
});
|
||||
const res = await submitScenarioForStartCase(fetch, scenario);
|
||||
|
||||
const data = await res.json();
|
||||
|
||||
if (res.ok && data.validationStatus === "valid") {
|
||||
if (res.ok && data.success) {
|
||||
setStatus("success");
|
||||
setResult(data);
|
||||
setResult(normaliseStartResult(data));
|
||||
} else {
|
||||
setStatus("error");
|
||||
setResult(data);
|
||||
setResult(normaliseStartResult(data));
|
||||
}
|
||||
} catch (err) {
|
||||
setStatus("error");
|
||||
@@ -39,8 +170,64 @@ export default function ScenarioForm() {
|
||||
}
|
||||
};
|
||||
|
||||
// Always show diagnostics when there's a result (even if validation failed)
|
||||
const hasDiagnostics = result && (result.reconstruction || result.modelName || result.responseDurationMs !== undefined);
|
||||
const handleUpdate = async (e) => {
|
||||
e.preventDefault();
|
||||
|
||||
const submission = await submitAnswerForUpdateCase(fetch, {
|
||||
situationGraph: result?.situationGraph,
|
||||
previousQuestion: result?.selectedQuestion,
|
||||
answer,
|
||||
});
|
||||
|
||||
if (submission.skipped) {
|
||||
setUpdateStatus("error");
|
||||
setUpdateError(submission.data);
|
||||
return;
|
||||
}
|
||||
|
||||
setUpdateStatus("loading");
|
||||
setUpdateError(null);
|
||||
|
||||
try {
|
||||
const outcome = submission.data;
|
||||
|
||||
if (submission.ok && outcome.success) {
|
||||
setUpdateStatus("success");
|
||||
setUpdateResult({
|
||||
...outcome,
|
||||
previousSituationGraph: result?.situationGraph ?? null,
|
||||
});
|
||||
setResult((current) => ({
|
||||
...current,
|
||||
situationGraph: outcome.updatedSituationGraph,
|
||||
selectedQuestion: normaliseUpdateSelectedQuestion(
|
||||
outcome.selectedQuestion,
|
||||
),
|
||||
newlySurfacedNodeIds: (outcome.proposal?.addedNodes || [])
|
||||
.filter((node) => node.kind === "unknown")
|
||||
.map((node) => node.id),
|
||||
diagnostics: outcome.diagnostics,
|
||||
}));
|
||||
setAnswer("");
|
||||
} else {
|
||||
setUpdateStatus("error");
|
||||
setUpdateError(outcome);
|
||||
}
|
||||
} catch (err) {
|
||||
setUpdateStatus("error");
|
||||
setUpdateError({ error: err.message || "Network request failed" });
|
||||
}
|
||||
};
|
||||
|
||||
const canRenderAnswerForm =
|
||||
status === "success" &&
|
||||
updateStatus === "idle" &&
|
||||
Boolean(result?.situationGraph) &&
|
||||
Boolean(result?.selectedQuestion);
|
||||
|
||||
const canRenderDisabledFollowUpForm =
|
||||
updateStatus === "success" &&
|
||||
Boolean(updateResult?.selectedQuestion?.question || result?.selectedQuestion);
|
||||
|
||||
return (
|
||||
<div className="space-y-6">
|
||||
@@ -54,7 +241,9 @@ export default function ScenarioForm() {
|
||||
className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400"
|
||||
/>
|
||||
<div className="flex items-center justify-between">
|
||||
<span className="text-xs text-gray-400">{scenario.length}/{MAX_LENGTH}</span>
|
||||
<span className="text-xs text-gray-400">
|
||||
{scenario.length}/{MAX_LENGTH}
|
||||
</span>
|
||||
<button
|
||||
type="submit"
|
||||
disabled={status === "loading" || !scenario.trim()}
|
||||
@@ -65,42 +254,104 @@ export default function ScenarioForm() {
|
||||
</div>
|
||||
</form>
|
||||
|
||||
{status === "error" && (
|
||||
<div className="space-y-3">
|
||||
{result?.error && (
|
||||
<div className="rounded-lg border border-red-300 bg-red-50 px-4 py-3 text-sm text-red-700 whitespace-pre-wrap">
|
||||
Error: {result.error}
|
||||
</div>
|
||||
)}
|
||||
{hasDiagnostics && result?.modelName && (
|
||||
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1.5 text-sm">
|
||||
<dt className="text-gray-500">Model</dt>
|
||||
<dd>{result.modelName}</dd>
|
||||
<dt className="text-gray-500">Duration</dt>
|
||||
<dd>{result.responseDurationMs != null ? `${result.responseDurationMs}ms` : "?"}</dd>
|
||||
</dl>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{status === "success" && result?.reconstruction && (
|
||||
<div className="space-y-4">
|
||||
<ReconstructionView reconstruction={result.reconstruction} />
|
||||
<DiagnosticsView result={result} />
|
||||
</div>
|
||||
)}
|
||||
|
||||
{status === "error" && result?.reconstruction && (
|
||||
<div className="space-y-3">
|
||||
<div className="rounded-lg border border-yellow-300 bg-yellow-50 px-4 py-2 text-sm text-yellow-800">
|
||||
⚠ Partial result — some fields failed validation. Showing what was accepted.
|
||||
{canRenderAnswerForm && (
|
||||
<form onSubmit={handleUpdate} className="space-y-4 rounded-lg border border-gray-200 bg-white p-4">
|
||||
<div>
|
||||
<h2 className="text-base font-semibold text-gray-900">Selected Question</h2>
|
||||
<p className="mt-1 text-sm text-gray-700">{result.selectedQuestion}</p>
|
||||
</div>
|
||||
<ReconstructionView reconstruction={result.reconstruction} partial />
|
||||
<div>
|
||||
<label htmlFor="answer-textarea" className="mb-2 block text-sm font-medium text-gray-700">
|
||||
Your answer
|
||||
</label>
|
||||
<textarea
|
||||
id="answer-textarea"
|
||||
value={answer}
|
||||
onChange={(e) => setAnswer(e.target.value)}
|
||||
rows={4}
|
||||
className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400"
|
||||
placeholder="Enter the answer to the selected question..."
|
||||
/>
|
||||
</div>
|
||||
<div className="flex items-center justify-between gap-4">
|
||||
<p className="text-xs text-gray-500">
|
||||
{updateStatus === "loading"
|
||||
? "Applying validated graph update..."
|
||||
: "One update turn only in this prototype."}
|
||||
</p>
|
||||
<button
|
||||
type="submit"
|
||||
disabled={updateStatus === "loading"}
|
||||
className="rounded-lg bg-blue-700 px-4 py-2 text-sm font-medium text-white transition hover:bg-blue-600 disabled:cursor-not-allowed disabled:opacity-40"
|
||||
>
|
||||
{updateStatus === "loading" ? "Updating..." : "Update situation"}
|
||||
</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
|
||||
<UpdateErrorPanel updateError={updateError} />
|
||||
|
||||
{updateStatus === "success" && updateResult && (
|
||||
<>
|
||||
<div className="rounded-lg border border-yellow-300 bg-yellow-50 px-4 py-3 text-sm text-yellow-800">
|
||||
{updateResult.selectedQuestion?.question
|
||||
? updateResult.selectedQuestion.question
|
||||
: "No next question selected yet."}
|
||||
</div>
|
||||
{canRenderDisabledFollowUpForm && (
|
||||
<form className="space-y-4 rounded-lg border border-gray-200 bg-white p-4 opacity-70">
|
||||
<div>
|
||||
<h2 className="text-base font-semibold text-gray-900">Selected Question</h2>
|
||||
<p className="mt-1 text-sm text-gray-700">
|
||||
{updateResult.selectedQuestion?.question || result?.selectedQuestion}
|
||||
</p>
|
||||
</div>
|
||||
<div>
|
||||
<label htmlFor="follow-up-disabled-textarea" className="mb-2 block text-sm font-medium text-gray-700">
|
||||
Your answer
|
||||
</label>
|
||||
<textarea
|
||||
id="follow-up-disabled-textarea"
|
||||
rows={4}
|
||||
disabled
|
||||
className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm opacity-70"
|
||||
placeholder="Additional submission is disabled in this one-update prototype."
|
||||
/>
|
||||
</div>
|
||||
<div className="flex items-center justify-between gap-4">
|
||||
<p className="text-xs text-gray-500">
|
||||
Additional submission is disabled in this one-update prototype.
|
||||
</p>
|
||||
<button
|
||||
type="button"
|
||||
disabled
|
||||
className="rounded-lg bg-blue-700 px-4 py-2 text-sm font-medium text-white disabled:cursor-not-allowed disabled:opacity-40"
|
||||
>
|
||||
Update situation
|
||||
</button>
|
||||
</div>
|
||||
</form>
|
||||
)}
|
||||
<GraphUpdateView updateResult={updateResult} />
|
||||
</>
|
||||
)}
|
||||
|
||||
<ScenarioResultPanels status={status} result={result} />
|
||||
|
||||
{(status === "loading" || updateStatus === "loading") && (
|
||||
<div className="py-12 text-center text-sm text-gray-400">
|
||||
Waiting for model response...
|
||||
</div>
|
||||
)}
|
||||
|
||||
{status === "loading" && (
|
||||
<div className="py-12 text-center text-sm text-gray-400">Waiting for model response...</div>
|
||||
{/* Empty state */}
|
||||
{status === "idle" && (
|
||||
<div className="rounded-lg border border-dashed border-gray-300 bg-gray-50 px-6 py-8 text-center">
|
||||
<p className="text-sm text-gray-400">
|
||||
Enter a scenario above and click Analyse to begin.
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
"use client";
|
||||
|
||||
import React from "react";
|
||||
|
||||
function NodeBadge({ children, tone = "gray" }) {
|
||||
const tones = {
|
||||
gray: "border-gray-200 bg-gray-50 text-gray-700",
|
||||
blue: "border-blue-200 bg-blue-50 text-blue-700",
|
||||
green: "border-green-200 bg-green-50 text-green-700",
|
||||
yellow: "border-yellow-200 bg-yellow-50 text-yellow-700",
|
||||
red: "border-red-200 bg-red-50 text-red-700",
|
||||
purple: "border-purple-200 bg-purple-50 text-purple-700",
|
||||
};
|
||||
|
||||
return (
|
||||
<span className={`rounded-full border px-2 py-0.5 text-xs ${tones[tone] || tones.gray}`}>
|
||||
{children}
|
||||
</span>
|
||||
);
|
||||
}
|
||||
|
||||
function NodeGroup({
|
||||
title,
|
||||
nodes,
|
||||
resolvedNodeIds = new Set(),
|
||||
newlySurfacedNodeIds = new Set(),
|
||||
activeUnknownNodeId = null,
|
||||
}) {
|
||||
if (!nodes?.length) return null;
|
||||
|
||||
return (
|
||||
<section className="rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h3 className="mb-3 text-sm font-semibold text-gray-700">
|
||||
{title} ({nodes.length})
|
||||
</h3>
|
||||
<ul className="space-y-3">
|
||||
{nodes.map((node) => (
|
||||
<li key={node.id} className="rounded border border-gray-100 bg-gray-50 p-3 text-sm">
|
||||
<div className="flex flex-wrap items-center gap-2">
|
||||
<span className="font-medium text-gray-900">{node.label}</span>
|
||||
<NodeBadge tone="blue">{node.status}</NodeBadge>
|
||||
<NodeBadge tone="green">{node.confidence}</NodeBadge>
|
||||
{resolvedNodeIds.has(node.id) && (
|
||||
<NodeBadge tone="red">resolved unknown</NodeBadge>
|
||||
)}
|
||||
{newlySurfacedNodeIds.has(node.id) && (
|
||||
<NodeBadge tone="purple">newly surfaced unknown</NodeBadge>
|
||||
)}
|
||||
{activeUnknownNodeId === node.id && (
|
||||
<NodeBadge tone="yellow">active unknown</NodeBadge>
|
||||
)}
|
||||
{node.value != null && (
|
||||
<NodeBadge tone="yellow">
|
||||
{node.value}
|
||||
{node.unit ? ` ${node.unit}` : ""}
|
||||
</NodeBadge>
|
||||
)}
|
||||
</div>
|
||||
{node.description && node.description !== node.label && (
|
||||
<p className="mt-1 text-gray-600">{node.description}</p>
|
||||
)}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</section>
|
||||
);
|
||||
}
|
||||
|
||||
export default function SituationGraphView({
|
||||
situationGraph,
|
||||
selectedQuestion,
|
||||
newlySurfacedNodeIds = [],
|
||||
}) {
|
||||
if (!situationGraph) return null;
|
||||
|
||||
const selectedQuestionText =
|
||||
typeof selectedQuestion === "string"
|
||||
? selectedQuestion
|
||||
: selectedQuestion?.question ?? null;
|
||||
|
||||
const activeUnknown = situationGraph.activeUnknownNodeId
|
||||
? situationGraph.nodes.find((node) => node.id === situationGraph.activeUnknownNodeId)
|
||||
: null;
|
||||
|
||||
const nodesByKind = situationGraph.nodes.reduce((acc, node) => {
|
||||
if (!acc[node.kind]) acc[node.kind] = [];
|
||||
acc[node.kind].push(node);
|
||||
return acc;
|
||||
}, {});
|
||||
|
||||
const resolvedNodeIdSet = new Set(situationGraph.resolvedNodeIds || []);
|
||||
const newlySurfacedNodeIdSet = new Set(newlySurfacedNodeIds || []);
|
||||
|
||||
return (
|
||||
<div className="space-y-4">
|
||||
{selectedQuestionText && (
|
||||
<section className="rounded-lg border-2 border-green-300 bg-green-50 p-5">
|
||||
<h2 className="mb-2 text-base font-bold text-green-800">Selected Question</h2>
|
||||
<p className="text-base font-medium text-gray-900">{selectedQuestionText}</p>
|
||||
</section>
|
||||
)}
|
||||
|
||||
<section className="rounded-lg border border-gray-200 bg-white p-4">
|
||||
<h2 className="mb-2 text-base font-semibold text-gray-900">Situation Graph</h2>
|
||||
<dl className="space-y-2 text-sm">
|
||||
<div>
|
||||
<dt className="text-gray-500">Central statement</dt>
|
||||
<dd className="font-medium text-gray-900">{situationGraph.centralStatement}</dd>
|
||||
</div>
|
||||
{situationGraph.currentSummary && (
|
||||
<div>
|
||||
<dt className="text-gray-500">Current summary</dt>
|
||||
<dd className="text-gray-800">{situationGraph.currentSummary}</dd>
|
||||
</div>
|
||||
)}
|
||||
{activeUnknown && (
|
||||
<div>
|
||||
<dt className="text-gray-500">Active unknown</dt>
|
||||
<dd className="text-gray-900">{activeUnknown.label}</dd>
|
||||
</div>
|
||||
)}
|
||||
<div>
|
||||
<dt className="text-gray-500">Edge count</dt>
|
||||
<dd className="text-gray-900">{situationGraph.edges.length}</dd>
|
||||
</div>
|
||||
</dl>
|
||||
</section>
|
||||
|
||||
{Object.entries(nodesByKind).map(([kind, nodes]) => (
|
||||
<NodeGroup
|
||||
key={kind}
|
||||
title={kind.replace(/_/g, " ")}
|
||||
nodes={nodes}
|
||||
resolvedNodeIds={resolvedNodeIdSet}
|
||||
newlySurfacedNodeIds={newlySurfacedNodeIdSet}
|
||||
activeUnknownNodeId={situationGraph.activeUnknownNodeId}
|
||||
/>
|
||||
))}
|
||||
|
||||
<details className="rounded-lg border border-gray-200 bg-gray-50 p-4">
|
||||
<summary className="cursor-pointer text-sm font-medium text-gray-700 underline">
|
||||
Raw graph JSON
|
||||
</summary>
|
||||
<pre className="mt-3 overflow-auto rounded bg-gray-900 p-3 text-xs text-green-400">
|
||||
{JSON.stringify(situationGraph, null, 2)}
|
||||
</pre>
|
||||
</details>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
# Orchestrator Contract — Confidence Engine v0.4
|
||||
|
||||
## 1. Exported Function Signatures & Shape (JavaScript)
|
||||
|
||||
### lib/analysis.js
|
||||
|
||||
```js
|
||||
export async function analyseScenario(scenario, opts = {})
|
||||
// @param {string} scenario
|
||||
// @param {{ promptVersion?: "v0.2" | "v0.3" }} [opts]
|
||||
// @returns {Promise<{ success: boolean, validationStatus: "valid"|"invalid",
|
||||
// modelName: string|null, responseDurationMs: number, rawResponse: string|null,
|
||||
// promptVersion: string|null, inputClassification: object|null, reconstruction: object|null,
|
||||
// evidence: object[]|undefined, nextQuestion: string|undefined, errors: string[]|undefined,
|
||||
// error: string|undefined, statusCode: number|undefined }>}
|
||||
|
||||
export const PROMPT_VERSIONS // { [key: string]: string }
|
||||
export const DEFAULT_PROMPT_VERSION // "v0.2"
|
||||
```
|
||||
|
||||
### lib/graph/schema.js
|
||||
|
||||
```js
|
||||
export const SituationKind // { observation, reported_claim, metric, state, transition, relationship, assumption, unknown, conclusion }
|
||||
export const SituationStatus // { known, unknown, provisional, supported, weakened, contradicted, resolved }
|
||||
export const ConfidenceLevel // { low, medium, high }
|
||||
export const SituationRelationship // { supports, weakens, contradicts, depends_on, causes, may_cause, measures, compares_with, updates, other }
|
||||
|
||||
export const situationNodeSchema // Zod → {@typedef SituationNode}
|
||||
export const situationEdgeSchema // Zod → {@typedef SituationEdge}
|
||||
export const situationGraphSchema // Zod → {@typedef SituationGraph}
|
||||
export const graphUpdateSchema // Zod → {@typedef GraphUpdate}
|
||||
export const startCaseRequestSchema // { scenario: string (1-10000), promptVersion?: string }
|
||||
export const updateCaseRequestSchema// { situationGraph: SituationGraph, previousQuestion: string, answer: string (1-5000), promptVersion?: string }
|
||||
|
||||
/** @param {string} label */ /** @returns {string} */ export function makeNodeId(label)
|
||||
/** @param {{ id?, label, description, kind?, status?, confidence?, value?, unit?, ... }} opts */ /** @returns {SituationNode} */ export function makeNode(opts)
|
||||
/** @param {{ id?, fromNodeId, toNodeId, relationship?, confidence?, description? }} opts */ /** @returns {SituationEdge} */ export function makeEdge(opts)
|
||||
/** @param {{ centralStatement?, nodes?, edges?, activeUnknownNodeId?, resolvedNodeIds?, currentSummary? }} opts */ /** @returns {SituationGraph} */ export function makeGraph(opts)
|
||||
```
|
||||
|
||||
### lib/graph/utils.js
|
||||
|
||||
```js
|
||||
export function validateGraphReferences(graph) // → { valid: boolean, errors: string[] }
|
||||
export function detectDuplicateNodeIds(nodes) // → { nodeId, count }[]
|
||||
export function detectDuplicateEdges(edges) // { edgeId, fromNodeId, toNodeId, relationship }[]
|
||||
export function findDependentNodes(graph, nodeId) // → string[] (transitive)
|
||||
export function findAffectedNodes(graph, nodeId) // → string[] (direct + indirect via affects/dependsOn)
|
||||
/** @param {SituationGraph} graph */ /** @param {string} nodeId */ /** @param {string} newStatus */ /** @param {*} newValue */ /** @param {string} reason */
|
||||
export function resolveUnknownNode(graph, nodeId, newStatus, newValue, reason) // → { success, error?, previousStatus?, newStatus?, previousValue?, newValue?, reason?, affectedNodes? }
|
||||
export function selectActiveUnknownCandidate(graph, resolvedNodeIds) // → { nodeId, label, score } | null
|
||||
/** @param {SituationGraph} graph */ /** @param {GraphUpdate} update */
|
||||
export function applyGraphUpdate(graph, update) // → { success: boolean, errors?, nodes?, edges?, resolvedNodeIds? }
|
||||
/** @param {SituationGraph} graph */ /** @param {GraphUpdate} update */
|
||||
export function validateGraphUpdate(graph, update) // → { valid: boolean, errors: string[] }
|
||||
```
|
||||
|
||||
### lib/graph/builder.js
|
||||
|
||||
```js
|
||||
export function buildInitialGraph(analysisData) // @param {{ reconstruction, evidence? }} → { nodes: SituationNode[], edges: SituationEdge[] }
|
||||
export function buildMinimalGraph(scenario) // @param {string} → { nodes, edges }
|
||||
export function describeGraph(graph) // @param {{ nodes, edges }} → string (summary text)
|
||||
```
|
||||
|
||||
## 2. Dependencies Between Files
|
||||
|
||||
```
|
||||
lib/analysis.js
|
||||
├── getConfig() from lib/config.js
|
||||
├── getProvider() from lib/llm/provider.js [EXTERNAL]
|
||||
├── buildPrompt() from lib/reconstruction/prompt.js
|
||||
└── reconstructionV2/V1Schema from lib/reconstruction/schema.js
|
||||
|
||||
lib/graph/utils.js ← imports situationNodeSchema, situationEdgeSchema, situationGraphSchema from schema.js
|
||||
lib/graph/builder.js ← imports situationNodeSchema, situationEdgeSchema, makeNodeId from schema.js
|
||||
docs/v0.4-handoff.md → references CaseOrchestrator.startCase()/updateCase() (not in any inspected file)
|
||||
```
|
||||
|
||||
## 3. Side Effects (LLM Calls)
|
||||
|
||||
| Function | LLM Call? | Details |
|
||||
|---|---|---|
|
||||
| `analyseScenario()` | **Yes** | `provider.generateReconstruction(prompt, model)` — POST to configured LLM. Prompt from `buildPrompt(scenario, version)`. |
|
||||
| All graph functions (`schema.js`, `utils.js`, `builder.js`) | No | Pure/deterministic only. |
|
||||
| `startCase()` / `updateCase()` (per handoff) | **Yes** | startCase: calls analyseScenario. updateCase: calls LLM via buildUpdatePrompt context + provider for GraphUpdate, then applyGraphUpdate(). |
|
||||
|
||||
## 4. Minimal Proposed Contract for API Functions
|
||||
|
||||
### startCase(body)
|
||||
- **Input:** `{ scenario: string (1-10000), promptVersion?: string }` — validated by `startCaseRequestSchema`.
|
||||
- **Flow:** validate → `analyseScenario()` → if ok, `buildInitialGraph(result)`; on failure return minimal graph via `buildMinimalGraph()`.
|
||||
- **Output (success):** `{ success: true, graphSummary: string, nodeCount: number, edgeCount: number, activeUnknownNodeId: string|undefined, nextQuestion: string }`
|
||||
- **Output (failure):** `{ success: false, error: string, graphSummary: string, nodeCount: number, edgeCount: number }`
|
||||
|
||||
### updateCase(body)
|
||||
- **Input:** `{ situationGraph: SituationGraph, previousQuestion: string (1+), answer: string (1-5000), promptVersion?: string }` — validated by `updateCaseRequestSchema`.
|
||||
- **Flow:** validate → `buildUpdatePrompt(ctx)` → LLM call for GraphUpdate proposal → `validateGraphUpdate()` → `applyGraphUpdate()` → resolve unknowns via `resolveUnknownNode()` → pick next candidate via `selectActiveUnknownCandidate()`.
|
||||
- **Output (success):** `{ success: true, graphSummary: string, nodeChanges: { added, updated, removed }, edgeChanges: { added, removed }, resolvedNodes: string[], nextQuestion: string|null }`
|
||||
- **Output (failure):** `{ success: false, error: string, graphSummary: string, nodeChanges: {}, edgeChanges: {}, resolvedNodes: [], nextQuestion: null }`
|
||||
|
||||
## 5. Missing Interfaces — TODO
|
||||
|
||||
1. **[TODO]** `CaseOrchestrator` class described in handoff but absent from all five inspected files. startCase()/updateCase() wrappers need implementation per above contract.
|
||||
2. **[TODO]** `buildUpdatePrompt(ctx)` (per handoff lives in prompt-builder.js) — not reviewed; input/output needs a separate doc once the file is available.
|
||||
3. **[TODO]** LLM provider interface (`getProvider()`, `generateReconstruction(prompt, model)`) — external dependency. Assumes rawResponse is parseable JSON matching v0.2/v0.1 schema; needs explicit contract.
|
||||
4. **[TODO]** Error handling for updateCase() on malformed LLM JSON — handoff notes "generic 500"; needs structured retry/error contract.
|
||||
5. **[TODO]** Completion heuristic `getCompletionStatus()` referenced in handoff but absent; needs contract (e.g., "complete" when no unresolved unknown nodes).
|
||||
|
||||
---
|
||||
|
||||
*End of contract.*
|
||||
@@ -0,0 +1,258 @@
|
||||
# v0.4 Handoff — Confidence Engine (confidence-engine)
|
||||
|
||||
**Date:** 2026-08-01
|
||||
**Branch:** `feature/reconstruction-v0.3`
|
||||
**Parent branch:** `main`
|
||||
|
||||
---
|
||||
|
||||
## 1. What This Project Is
|
||||
|
||||
A Next.js app that performs evidence-based situation reconstruction on user-supplied scenarios. An LLM analyses the scenario, builds a directed graph of actors, systems, unknowns and relationships, then iteratively refines the graph through multi-turn Q&A with the user.
|
||||
|
||||
---
|
||||
|
||||
## 2. Recent Commit History
|
||||
|
||||
| Commit | Message |
|
||||
|--------|---------|
|
||||
| `79ea2f6` | feat: add v0.3 normalised comparison reasoning |
|
||||
| `d72c7c5` | chore: establish clean v0.2 baseline |
|
||||
| `a2f9e47` | chore: preserve initial reconstruction prototype |
|
||||
|
||||
Only **one commit** ahead of `main`: `79ea2f6` — the v0.3 normalised comparison reasoning work.
|
||||
|
||||
---
|
||||
|
||||
## 3. Current State Summary
|
||||
|
||||
### What's done and committed to this branch
|
||||
|
||||
1. **v0.3 prompt** (`prompts/reconstruct-v0.3.md`) — a full LLM system prompt that adds:
|
||||
- Normalisation / rate reasoning guidance (distinguishing absolute counts from per-unit rates)
|
||||
- Interpretation discipline (empty array when evidence is too thin; no speculative filler)
|
||||
- "Exactly one next question" constraint (no compound questions)
|
||||
- Evidence type classification: `direct_observation`, `reported_statement`, `interpretation`, `assumption`, `inferred_relationship`
|
||||
- Importance and confidence scales
|
||||
- A strict camelCase JSON output schema with four top-level keys: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`
|
||||
|
||||
2. **v0.3 prompt versioning** (`lib/reconstruction/prompt.js`) — exports `PROMPT_VERSIONS`, `DEFAULT_PROMPT_VERSION ("v0.3")`, and `buildPrompt(scenario, version)` for loading prompt templates from disk with scenario substitution.
|
||||
|
||||
3. **Schema validation** (`lib/reconstruction/schema.js`) — Zod schemas for v0.2 output (`reconstructionV2Schema`). A `parseReconstructionV2(rawString)` helper is used in the analysis pipeline.
|
||||
|
||||
4. **v0.3 reasoning tests** (`tests/v03-reasoning.test.js`) — extensive test suite covering:
|
||||
- Prompt version registration and loading
|
||||
- v0.3 guidance completeness (normalisation, rate vs count, correlation-vs-causation)
|
||||
- Schema validation with a realistic "production/complaints" fixture
|
||||
- Parse helper tests
|
||||
|
||||
5. **Graph library** (`lib/graph/`) — the multi-turn reconstruction pipeline:
|
||||
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `schema.js` | Zod schemas for SituationNode, SituationEdge, SituationGraph, GraphUpdate; helpers like `makeNodeId`, `makeNode`, `makeEdge`, `makeGraph` |
|
||||
| `builder.js` | `buildInitialGraph(reconstruction, evidence)` — converts v0.2/v0.3 analysis output into a SituationGraph with deterministic nodes/edges; `buildMinimalGraph(scenario)` for fallback; `describeGraph(graph)` for display |
|
||||
| `orchestrator.js` | `CaseOrchestrator` class managing the full multi-turn lifecycle (idle → building → active); exports `startCase(body)` and `updateCase(body)` convenience functions for API routes |
|
||||
| `prompt-builder.js` | `buildUpdatePrompt(ctx)` — formats current graph state + Q&A context into a system prompt for the LLM update-evaluation turn |
|
||||
| `utils.js` | Deterministic graph operations: `validateGraphReferences`, `detectDuplicateNodeIds`, `detectDuplicateEdges`, `findDependentNodes`, `findAffectedNodes`, `resolveUnknownNode`, `selectActiveUnknownCandidate`, `applyGraphUpdate`, `validateGraphUpdate` |
|
||||
|
||||
6. **API routes** (`app/api/`)
|
||||
|
||||
| Route | Purpose |
|
||||
|-------|---------|
|
||||
| `POST /api/start-case` | Start a new reconstruction case — accepts `{ scenario, promptVersion? }`, returns graph summary, node/edge counts, next question |
|
||||
| `POST /api/update-case` | Process a turn — accepts `{ scenario, graph, answer, currentQuestion?, turnCount?, modelName? }`, returns updated graph summary, next question, changes summary |
|
||||
|
||||
7. **Smoke test** (`tests/smoke.test.js`) — basic integration test for the start-case API route.
|
||||
|
||||
### What's NOT yet committed (untracked files from git status)
|
||||
|
||||
| File | Description |
|
||||
|------|-------------|
|
||||
| `lib/graph/` (full directory) | The multi-turn graph library — built but NOT yet committed to any branch. These are the new untracked files: `builder.js`, `orchestrator.js`, `prompt-builder.js`, `schema.js`, `utils.js` |
|
||||
| `tests/graph/` (full directory) | Tests for the graph library — also untracked: `builder.test.js`, `orchestrator.test.js`, `prompt-builder.test.js`, `schema.test.js`, `utils.test.js` |
|
||||
| `app/api/start-case/route.js` | New API route (untracked) |
|
||||
| `app/api/update-case/route.js` | New API route (untracked) |
|
||||
|
||||
> **Important:** The git status shows these files as untracked (`??`). They exist on disk but have never been staged or committed. You need to decide whether to commit them now or integrate them differently.
|
||||
|
||||
---
|
||||
|
||||
## 4. Test Status
|
||||
|
||||
```
|
||||
Test Files: 4 failed | 4 passed (8)
|
||||
Tests: 5 failed | 216 passed (221)
|
||||
```
|
||||
|
||||
### Known failures
|
||||
|
||||
The failures cluster in `tests/graph/`:
|
||||
- **`prompt-builder.test.js`** — test expects the literal string `"Existing or newly added nodes"` but the prompt template currently says `"existing or newly added nodes"` (case mismatch). The SYSTEM_PROMPT_HEADER constant uses lowercase.
|
||||
- Other graph tests likely have similar fixture/reference issues.
|
||||
|
||||
Run `npx vitest run tests/graph/ --reporter=verbose` for full details.
|
||||
|
||||
---
|
||||
|
||||
## 5. Architecture Overview
|
||||
|
||||
```
|
||||
User scenario
|
||||
│
|
||||
▼
|
||||
┌──────────────┐ ┌─────────────────┐ ┌──────────────┐
|
||||
│ analyseScenario│──▶│ buildPrompt │──▶│ LLM (v0.3) │
|
||||
│ (lib/analysis.js) │ (reconstruction/prompt.js) │ │
|
||||
└──────────────┘ └─────────────────┘ └──────┬───────┘
|
||||
│
|
||||
▼
|
||||
┌──────────────┐
|
||||
│ Parse output │
|
||||
│ (Zod/parse │
|
||||
│ Reconstruction│
|
||||
│ V2) │
|
||||
└──────┬───────┘
|
||||
│
|
||||
┌───────────────────────────────┤
|
||||
▼ ▼
|
||||
┌──────────────┐ ┌──────────────────┐
|
||||
│buildInitialGraph│ │ buildMinimalGraph │
|
||||
│ (graph/builder)│ │ (fallback) │
|
||||
└──────┬─────────┘ └──────────────────┘
|
||||
│
|
||||
▼
|
||||
┌──────────────┐
|
||||
│SituationGraph │ ← Zod-validated graph structure
|
||||
│ {nodes, edges}│ nodes: observation/metric/unknown/...
|
||||
└──────┬───────┘ edges: supports/weakens/causes/...
|
||||
│
|
||||
(multi-turn loop via updateCase)
|
||||
│
|
||||
┌─────────▼─────────┐
|
||||
│buildUpdatePrompt │ → LLM proposes GraphUpdate
|
||||
│ │
|
||||
│applyGraphUpdate │ → deterministic, validated
|
||||
│validateGraphUpdate│ (no direct LLM mutation)
|
||||
└───────────────────┘
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 6. Key Design Decisions
|
||||
|
||||
### Normalisation / rate reasoning (v0.3 focus)
|
||||
The v0.3 prompt explicitly instructs the model to:
|
||||
- Always consider whether a denominator/exposure metric is needed when counts change alongside scale
|
||||
- Distinguish absolute count from rate
|
||||
- Avoid treating two rising counts as causal evidence (production growth may outpace complaint growth)
|
||||
- Request the per-unit metric as the highest-value next question
|
||||
|
||||
### Graph immutability
|
||||
LLM proposals are never applied directly. All mutations go through `applyGraphUpdate()` in `lib/graph/utils.js`, which:
|
||||
- Validates all node/edge references exist
|
||||
- Rejects duplicate IDs
|
||||
- Enforces a max graph size (500 nodes) and update size (100KB)
|
||||
- Returns the full new state for validation
|
||||
|
||||
### Prompt versioning
|
||||
- Default is `"v0.3"` but `PROMPT_VERSIONS` includes `"v0.2"` for backward compatibility
|
||||
- `RECONSTRUCTION_PROMPT_VERSION` env var can override default at module load time
|
||||
- Prompts are loaded from `prompts/reconstruct-v0.{version}.md` on disk
|
||||
|
||||
### Deterministic node IDs
|
||||
Node IDs are computed via a deterministic hash of the label: `makeNodeId(label)`. This avoids conflicts but means nodes must be created with consistent labels to get consistent IDs.
|
||||
|
||||
---
|
||||
|
||||
## 7. Open Questions / TODOs for Next Developer
|
||||
|
||||
1. **Untracked graph library** — `lib/graph/` and `tests/graph/` are untracked on disk. Do we commit them as part of v0.4, or keep them in a separate branch?
|
||||
|
||||
2. **Test failures** — 5 tests fail across the graph test suite. The prompt-builder case-sensitivity issue needs fixing. Review all failing tests before merging.
|
||||
|
||||
3. **Missing `RECONSTRUCTION_PROMPT_VERSION` env var docs** — The system uses an env var override but it's not documented in `.env.example`. Add it if it's intended to be configurable.
|
||||
|
||||
4. **Provider integration** — `lib/llm/provider.js` is imported by the orchestrator (`getProvider()`, `generateReconstruction()`). Verify the provider implementation matches what this code expects.
|
||||
|
||||
5. **Graph completeness heuristic** — `CaseOrchestrator.getCompletionStatus()` returns `"complete"` when no unknown nodes remain, but doesn't consider whether all important observations have been verified.
|
||||
|
||||
6. **Error resilience in update flow** — If the LLM returns malformed JSON, the update route returns a 500 with a generic error message. Consider retry logic or structured error parsing.
|
||||
|
||||
7. **`buildUpdatePrompt` SYSTEM_PROMPT_HEADER is a module-level constant** — it's hardcoded and never versioned. If v0.5 changes the update-evaluation prompt style, this will need to become a template.
|
||||
|
||||
8. **The `nextQuestion` field on `/api/start-case` response** includes the adapted question (original + active unknown label appended). The client may want the original and adapted separately.
|
||||
|
||||
---
|
||||
|
||||
## 8. File Inventory (new / changed files on this branch)
|
||||
|
||||
### Prompts
|
||||
- `prompts/reconstruct-v0.3.md` — **NEW** — v0.3 system prompt (161 lines)
|
||||
- `prompts/reconstruct-v0.2.md` — **existing** — baseline prompt
|
||||
|
||||
### Core library
|
||||
- `lib/analysis.js` — **MODIFIED** — analyseScenario function (uses v0.3 prompt by default)
|
||||
- `lib/reconstruction/prompt.js` — **MODIFIED** — prompt versioning exports
|
||||
- `lib/reconstruction/schema.js` — **existing** — Zod schemas + parseReconstructionV2
|
||||
|
||||
### Graph library (untracked on disk)
|
||||
- `lib/graph/builder.js` — buildInitialGraph, buildMinimalGraph, describeGraph
|
||||
- `lib/graph/orchestrator.js` — CaseOrchestrator class, startCase, updateCase
|
||||
- `lib/graph/prompt-builder.js` — buildUpdatePrompt + SYSTEM_PROMPT_HEADER
|
||||
- `lib/graph/schema.js` — SituationNode/Edge/Graph/Update Zod schemas
|
||||
- `lib/graph/utils.js` — validation, dedup, dependency, and apply utilities
|
||||
|
||||
### API routes (untracked on disk)
|
||||
- `app/api/start-case/route.js`
|
||||
- `app/api/update-case/route.js`
|
||||
|
||||
### Tests (untracked on disk)
|
||||
- `tests/graph/builder.test.js`
|
||||
- `tests/graph/orchestrator.test.js`
|
||||
- `tests/graph/prompt-builder.test.js`
|
||||
- `tests/graph/schema.test.js`
|
||||
- `tests/graph/utils.test.js`
|
||||
- `tests/v03-reasoning.test.js` — **committed** to current branch
|
||||
- `tests/smoke.test.js`
|
||||
|
||||
### Config changes
|
||||
- `package.json` — added dependency (verify which one)
|
||||
- `playwright.config.js` — added/modified for integration testing
|
||||
- `.env.local` — exists locally (not committed)
|
||||
|
||||
---
|
||||
|
||||
## 9. How to Run
|
||||
|
||||
```bash
|
||||
# Install dependencies
|
||||
npm install
|
||||
|
||||
# Unit tests
|
||||
npx vitest run
|
||||
|
||||
# Graph library tests (has 5 failures)
|
||||
npx vitest run tests/graph/ --reporter=verbose
|
||||
|
||||
# Start dev server
|
||||
npm run dev
|
||||
|
||||
# API endpoints
|
||||
# POST /api/start-case → { scenario: "..." }
|
||||
# POST /api/update-case → { graph: {...}, answer: "...", ... }
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 10. What to Do First (Recommended Priorities)
|
||||
|
||||
1. **Review and fix the 5 failing tests** — likely simple string/fixture issues
|
||||
2. **Decide on the untracked files** — commit them, or create a v0.4 branch from this point
|
||||
3. **Verify the LLM provider integration** — ensure `getProvider()` and `generateReconstruction()` are wired up correctly
|
||||
4. **Add env var documentation** for `RECONSTRUCTION_PROMPT_VERSION` to `.env.example`
|
||||
5. **Smoke test end-to-end** — call `/api/start-case` with a real scenario and verify the full flow
|
||||
|
||||
---
|
||||
|
||||
*End of handoff.*
|
||||
@@ -0,0 +1,25 @@
|
||||
# v0.4 Route Status
|
||||
|
||||
- `app/api/cases/start/route.js`
|
||||
- Current tracked start-case route for the v0.4 graph orchestration path.
|
||||
- Covered by `tests/app/api/cases-start-route.test.js`.
|
||||
|
||||
- `app/api/cases/update/route.js`
|
||||
- Current tracked update-case route for the v0.4 graph orchestration path.
|
||||
- Delegates to `updateCase(body, { applyProposal: true })`.
|
||||
- Covered by `tests/app/api/cases-update-route.test.js`.
|
||||
|
||||
- `app/api/start-case/route.js`
|
||||
- Earlier experiment / duplicate start route.
|
||||
- No repository UI/test references were found.
|
||||
- Deleted from the working tree during UI connection cleanup.
|
||||
|
||||
- `app/api/update-case/route.js`
|
||||
- Earlier experimental duplicate update route.
|
||||
- Removed from the working tree during route consolidation.
|
||||
|
||||
- Current UI status
|
||||
- `components/scenario-form.jsx` now calls `/api/cases/start` for the main experimental flow.
|
||||
- `/api/cases/update` is the active tracked update route.
|
||||
- `/api/analyse` remains available for legacy one-shot analysis.
|
||||
- No UI changes were required for this route milestone.
|
||||
@@ -0,0 +1,48 @@
|
||||
# v0.5 Question Priority Generalisation
|
||||
|
||||
## Hypothesis
|
||||
|
||||
The current deterministic unknown selector and graph-context question formulator should generalise across several decision types by selecting a foundational unknown before downstream implementation or pricing leaves.
|
||||
|
||||
## Scenarios
|
||||
|
||||
1. Should we hire another engineer?
|
||||
2. Should we replace the delivery vans?
|
||||
3. Should we launch in another country?
|
||||
4. Should we continue a project that is over budget?
|
||||
5. Should we introduce a paid support tier?
|
||||
|
||||
## Results
|
||||
|
||||
| Scenario | Selected unknown | Strategy | Pass/Fail |
|
||||
| ---------------------------- | --------------------------- | -------------------- | --------- |
|
||||
| Hire another engineer | `hire-success-criteria` | `decision criterion` | Pass |
|
||||
| Replace the delivery vans | `van-reliability-threshold` | `decision criterion` | Pass |
|
||||
| Launch in another country | `country-value-threshold` | `actor/customer` | Pass |
|
||||
| Continue over-budget project | `project-benefit-threshold` | `decision criterion` | Pass |
|
||||
| Introduce paid support tier | `support-value-threshold` | `actor/customer` | Pass |
|
||||
|
||||
## Repeated failure patterns
|
||||
|
||||
Two repeated structural formulation failures appeared before the final pass:
|
||||
|
||||
1. **Constraint language in surrounding graph context outranked node-local decision-threshold language** in more than one case.
|
||||
2. **Baseline language in surrounding graph context outranked node-local threshold language** in more than one case.
|
||||
|
||||
Both failures affected formulation strategy, not deterministic unknown selection.
|
||||
|
||||
## Code change made
|
||||
|
||||
A small deterministic change was made in `lib/graph/question-formulator.js`:
|
||||
|
||||
- prefer node-local `definition` language before broader criterion inference
|
||||
- prefer node-local `decision criterion` language before context-only `constraint` inference
|
||||
- only treat `baseline` or `constraint` as primary when the selected node itself carries that language, otherwise allow them as fallback strategies later
|
||||
|
||||
No architecture, UI, persistence, prompt, scoring, additional model turns, or provider calls were added.
|
||||
|
||||
## Remaining limitations
|
||||
|
||||
- In two passing cases, the selector chose a threshold-style foundational node while the formulator still used an `actor/customer` strategy because related context strongly referenced customers or recipients.
|
||||
- This experiment is fixture-driven and deterministic; it is useful for regression protection, not scientific validation.
|
||||
- The suite exercises the production path without model calls, but it does not prove behaviour over arbitrary real-world graph structures.
|
||||
@@ -0,0 +1,58 @@
|
||||
# v0.5 Release Notes
|
||||
|
||||
## Purpose of v0.5
|
||||
|
||||
v0.5 stabilises the graph-backed one-turn update flow so the engine can resolve an answered unknown, surface consequential new unknowns, prioritise the next unknown deterministically, and formulate a deterministic follow-up question without changing the UI or adding more model turns.
|
||||
|
||||
## Capabilities proven
|
||||
|
||||
v0.5 includes:
|
||||
|
||||
- resolving an existing unknown
|
||||
- surfacing consequential new unknowns
|
||||
- limiting emergent unknowns
|
||||
- deterministic information-value prioritisation
|
||||
- deterministic question formulation
|
||||
- generalisation across five decision types
|
||||
- graph-backed one-turn UI update
|
||||
|
||||
## Five-case generalisation result
|
||||
|
||||
All five deterministic fixture scenarios passed:
|
||||
|
||||
1. Should we hire another engineer?
|
||||
2. Should we replace the delivery vans?
|
||||
3. Should we launch in another country?
|
||||
4. Should we continue a project that is over budget?
|
||||
5. Should we introduce a paid support tier?
|
||||
|
||||
The selector chose a foundational unknown first in each case, avoided the downstream leaf first, required no model call, and preserved graph immutability during question formulation.
|
||||
|
||||
## Key deterministic safeguards
|
||||
|
||||
- proposal application re-selects the active unknown deterministically after validation
|
||||
- information-value scoring penalises downstream or prerequisite-blocked unknowns
|
||||
- emergent unknown validation limits additions and requires explicit answer-derived linkage
|
||||
- final question wording is reformulated from graph context without an extra model turn
|
||||
- question validation rejects compound, awkward, or pricing-led fallback phrasing
|
||||
|
||||
## Known limitation
|
||||
|
||||
A correctly selected threshold node can still be phrased using an actor/customer strategy when surrounding graph context strongly references customers or value recipients.
|
||||
|
||||
This limitation is recorded for the next experiment and is not being fixed in the v0.5 release-prep task.
|
||||
|
||||
## Deliberately excluded work
|
||||
|
||||
- no reasoning-logic expansion beyond the small deterministic formulation fixes already landed on the branch
|
||||
- no new features
|
||||
- no UI changes
|
||||
- no persistence
|
||||
- no additional model turn
|
||||
- no Ollama calls for validation
|
||||
- no evaluator-suite runs
|
||||
- no Playwright runs
|
||||
|
||||
## Next experimental question
|
||||
|
||||
Can the question formulation strategy remain aligned with the selected node's role when surrounding graph context contains competing signals?
|
||||
+242
@@ -0,0 +1,242 @@
|
||||
/**
|
||||
* Core analysis pipeline — shared by API routes and evaluation harness.
|
||||
* Calls the provider, parses output, validates against Zod schemas (v0.2 first, v0.1 fallback).
|
||||
*/
|
||||
|
||||
import { getConfig } from "../lib/config.js";
|
||||
import { getProvider } from "../lib/llm/provider.js";
|
||||
import {
|
||||
buildPrompt,
|
||||
PROMPT_VERSIONS,
|
||||
DEFAULT_PROMPT_VERSION,
|
||||
} from "../lib/reconstruction/prompt.js";
|
||||
import { normaliseAnalysisResponse } from "../lib/reconstruction/compatibility.js";
|
||||
import {
|
||||
reconstructionV2Schema,
|
||||
reconstructionSchema as reconstructionV1Schema,
|
||||
} from "../lib/reconstruction/schema.js";
|
||||
|
||||
const MAX_SCENARIO_LENGTH = 10000;
|
||||
|
||||
/**
|
||||
* Analyse a scenario string through the full pipeline.
|
||||
* @param {string} scenario - The scenario text to analyse
|
||||
* @param {object} [opts]
|
||||
* @param {"v0.1" | "v0.2"} [opts.promptVersion="v0.2"] - Prompt version to use
|
||||
* @returns {Promise<object>} Analysis result with diagnostics
|
||||
*/
|
||||
export async function analyseScenario(scenario, opts = {}) {
|
||||
const startTime = Date.now();
|
||||
|
||||
// ── Input validation ───────────────────────────────
|
||||
if (typeof scenario !== "string") {
|
||||
return buildErrorResponse("Input must be a string", startTime);
|
||||
}
|
||||
|
||||
const trimmed = scenario.trim();
|
||||
if (trimmed.length === 0) {
|
||||
return buildErrorResponse("Scenario cannot be empty", startTime);
|
||||
}
|
||||
if (trimmed.length > MAX_SCENARIO_LENGTH) {
|
||||
return buildErrorResponse(
|
||||
`Scenario must be under ${MAX_SCENARIO_LENGTH} characters`,
|
||||
startTime,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Configuration check ────────────────────────────
|
||||
const configResult = getConfig();
|
||||
if (!configResult.ok) {
|
||||
return buildErrorResponse("Invalid server configuration", startTime, "500");
|
||||
}
|
||||
|
||||
const { OLLAMA_BASE_URL: _ignored, OLLAMA_MODEL } = configResult.config;
|
||||
const promptVersion = opts.promptVersion || DEFAULT_PROMPT_VERSION;
|
||||
|
||||
// ── Build prompt ───────────────────────────────────
|
||||
let promptObj;
|
||||
try {
|
||||
promptObj = await buildPrompt(trimmed, promptVersion);
|
||||
} catch (e) {
|
||||
return buildErrorResponse(
|
||||
`Failed to build prompt: ${e.message}`,
|
||||
startTime,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Call provider ──────────────────────────────────
|
||||
const provider = getProvider();
|
||||
let rawResponse;
|
||||
try {
|
||||
rawResponse = await provider.generateReconstruction(
|
||||
promptObj.prompt,
|
||||
OLLAMA_MODEL,
|
||||
);
|
||||
} catch (e) {
|
||||
return buildErrorResponse(
|
||||
e.message || "Provider error during analysis",
|
||||
Date.now() - startTime,
|
||||
);
|
||||
}
|
||||
|
||||
const duration = Date.now() - startTime;
|
||||
|
||||
// Try to capture raw response for diagnostics
|
||||
let rawResponseStr;
|
||||
try {
|
||||
rawResponseStr = JSON.stringify(rawResponse);
|
||||
} catch {
|
||||
rawResponseStr = String(rawResponse).slice(0, 2000);
|
||||
}
|
||||
|
||||
const compatibility = normaliseAnalysisResponse(rawResponse);
|
||||
const candidateResponse = compatibility.normalised;
|
||||
|
||||
// ── Validate against v0.2 schema (preferred) ──────
|
||||
const resultV2 = tryValidateAgainstSchema(
|
||||
candidateResponse,
|
||||
reconstructionV2Schema,
|
||||
);
|
||||
if (resultV2.valid) {
|
||||
return buildSuccessResultV2(
|
||||
resultV2.data,
|
||||
OLLAMA_MODEL,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Fallback to v0.1 schema ────────────────────────
|
||||
const resultV1 = tryValidateAgainstSchema(
|
||||
candidateResponse,
|
||||
reconstructionV1Schema,
|
||||
);
|
||||
if (resultV1.valid) {
|
||||
return buildSuccessResultV1(
|
||||
resultV1.data,
|
||||
OLLAMA_MODEL,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Neither schema matched — partial failure ───────
|
||||
return buildPartialResult(
|
||||
rawResponseStr?.slice(0, 2000),
|
||||
resultV2.error ?? resultV1.error,
|
||||
OLLAMA_MODEL,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
);
|
||||
}
|
||||
|
||||
/** Attempt validation against a Zod schema */
|
||||
function tryValidateAgainstSchema(data, schema) {
|
||||
if (!schema.safeParse) {
|
||||
return {
|
||||
valid: false,
|
||||
error: new Error("Schema does not support safeParse"),
|
||||
};
|
||||
}
|
||||
const result = schema.safeParse(data);
|
||||
return result.success
|
||||
? { valid: true, data: result.data }
|
||||
: { valid: false, error: result.error };
|
||||
}
|
||||
|
||||
// ── Result builders ──────────────────────────────────
|
||||
|
||||
function buildErrorResponse(message, elapsed, statusCode = 500) {
|
||||
return {
|
||||
success: false,
|
||||
error: message,
|
||||
modelName: null,
|
||||
responseDurationMs: elapsed,
|
||||
validationStatus: "invalid",
|
||||
rawResponse: null,
|
||||
promptVersion: null,
|
||||
statusCode,
|
||||
};
|
||||
}
|
||||
|
||||
function buildCompatibilityDiagnostics(compatibility) {
|
||||
return {
|
||||
compatibilityApplied: compatibility.changesApplied.length > 0,
|
||||
compatibilityChanges: compatibility.changesApplied,
|
||||
compatibilityWarnings: compatibility.warnings,
|
||||
};
|
||||
}
|
||||
|
||||
function buildSuccessResultV2(data, model, duration, version, compatibility) {
|
||||
return {
|
||||
success: true,
|
||||
validationStatus: "valid",
|
||||
modelName: model,
|
||||
responseDurationMs: duration,
|
||||
rawResponse: JSON.stringify(data).slice(0, 3000),
|
||||
promptVersion: version,
|
||||
inputClassification: data.inputClassification,
|
||||
reconstruction: data.reconstruction,
|
||||
evidence: data.evidence,
|
||||
nextQuestion: data.nextQuestion,
|
||||
errors: undefined,
|
||||
...buildCompatibilityDiagnostics(compatibility),
|
||||
};
|
||||
}
|
||||
|
||||
function buildSuccessResultV1(data, model, duration, version, compatibility) {
|
||||
return {
|
||||
success: true,
|
||||
validationStatus: "valid",
|
||||
modelName: model,
|
||||
responseDurationMs: duration,
|
||||
rawResponse: JSON.stringify(data).slice(0, 3000),
|
||||
promptVersion: version,
|
||||
inputClassification: null,
|
||||
reconstruction: data,
|
||||
evidence: undefined,
|
||||
nextQuestion: undefined,
|
||||
errors: undefined,
|
||||
...buildCompatibilityDiagnostics(compatibility),
|
||||
};
|
||||
}
|
||||
|
||||
function buildPartialResult(
|
||||
rawResp,
|
||||
error,
|
||||
model,
|
||||
duration,
|
||||
version,
|
||||
compatibility,
|
||||
) {
|
||||
let errors = [];
|
||||
if (error && typeof error.flatten === "function") {
|
||||
errors = error.flatten().fieldErrors
|
||||
? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [
|
||||
`${k}: ${v.join(", ")}`,
|
||||
])
|
||||
: [String(error)];
|
||||
} else if (error) {
|
||||
errors = [String(error).slice(0, 500)];
|
||||
}
|
||||
|
||||
return {
|
||||
success: false,
|
||||
validationStatus: "invalid",
|
||||
modelName: model,
|
||||
responseDurationMs: duration,
|
||||
rawResponse: rawResp?.slice(0, 2000),
|
||||
promptVersion: version,
|
||||
inputClassification: null,
|
||||
reconstruction: null,
|
||||
evidence: undefined,
|
||||
nextQuestion: undefined,
|
||||
errors,
|
||||
...buildCompatibilityDiagnostics(compatibility),
|
||||
};
|
||||
}
|
||||
|
||||
export { PROMPT_VERSIONS, DEFAULT_PROMPT_VERSION };
|
||||
@@ -0,0 +1,715 @@
|
||||
import { describeGraph } from "./builder.js";
|
||||
import { formulateQuestion } from "./question-formulator.js";
|
||||
import { graphUpdateSchema, situationGraphSchema } from "./schema.js";
|
||||
import {
|
||||
applyGraphUpdate,
|
||||
detectDuplicateNodeIds,
|
||||
findAffectedNodes,
|
||||
scoreUnknownCandidate,
|
||||
selectActiveUnknownCandidate,
|
||||
validateGraphReferences,
|
||||
validateGraphUpdate,
|
||||
} from "./utils.js";
|
||||
|
||||
function cloneJsonSafe(value) {
|
||||
return JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
function zodIssuesToErrors(error) {
|
||||
return (
|
||||
error?.issues?.map((issue) => {
|
||||
const path = issue.path?.length ? `${issue.path.join(".")}: ` : "";
|
||||
return `${path}${issue.message}`;
|
||||
}) ?? ["Validation failed"]
|
||||
);
|
||||
}
|
||||
|
||||
function collectDuplicateEdgeIds(edges) {
|
||||
const counts = new Map();
|
||||
|
||||
for (const edge of edges) {
|
||||
counts.set(edge.id, (counts.get(edge.id) ?? 0) + 1);
|
||||
}
|
||||
|
||||
return [...counts.entries()]
|
||||
.filter(([, count]) => count > 1)
|
||||
.map(([edgeId, count]) => ({ edgeId, count }));
|
||||
}
|
||||
|
||||
function normaliseText(value) {
|
||||
return String(value || "")
|
||||
.toLowerCase()
|
||||
.replace(/[^a-z0-9]+/g, " ")
|
||||
.trim();
|
||||
}
|
||||
|
||||
function buildNodeById(graph, addedNodes = []) {
|
||||
return new Map(
|
||||
[...graph.nodes, ...addedNodes].map((node) => [node.id, node]),
|
||||
);
|
||||
}
|
||||
|
||||
function isCompoundQuestion(question) {
|
||||
if (typeof question !== "string") return false;
|
||||
const trimmed = question.trim();
|
||||
if (!trimmed) return false;
|
||||
|
||||
const questionMarks = (trimmed.match(/\?/g) || []).length;
|
||||
if (questionMarks > 1) return true;
|
||||
if (/\?\s*(and|or)\b/i.test(trimmed)) return true;
|
||||
if (/\b(and|or)\b[^?]{0,60}\?/i.test(trimmed) && /,/.test(trimmed))
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function validateAddedUnknowns(graph, proposal) {
|
||||
const errors = [];
|
||||
const addedUnknowns = proposal.addedNodes.filter(
|
||||
(node) => node.kind === "unknown",
|
||||
);
|
||||
|
||||
if (addedUnknowns.length > 3) {
|
||||
errors.push(
|
||||
`Proposal adds too many unknown nodes: ${addedUnknowns.length} (maximum 3)`,
|
||||
);
|
||||
}
|
||||
|
||||
const unresolvedExistingUnknowns = graph.nodes.filter(
|
||||
(node) =>
|
||||
node.kind === "unknown" &&
|
||||
!proposal.resolvedUnknownNodeIds.includes(node.id),
|
||||
);
|
||||
const seenAddedUnknownMeanings = new Map();
|
||||
const answerDerivedNodeIds = new Set([
|
||||
...proposal.updatedNodes.map((update) => update.nodeId),
|
||||
...proposal.resolvedUnknownNodeIds,
|
||||
...proposal.addedNodes
|
||||
.filter((node) => node.kind !== "unknown")
|
||||
.map((node) => node.id),
|
||||
]);
|
||||
const proposalNodeById = buildNodeById(graph, proposal.addedNodes);
|
||||
|
||||
function hasExplicitNodeReference(fromNode, toNodeId) {
|
||||
if (!fromNode || !toNodeId) return false;
|
||||
|
||||
return (
|
||||
fromNode.parentId === toNodeId ||
|
||||
fromNode.dependsOn.includes(toNodeId) ||
|
||||
fromNode.affects.includes(toNodeId) ||
|
||||
fromNode.childIds.includes(toNodeId)
|
||||
);
|
||||
}
|
||||
|
||||
function hasExplicitAnswerDerivedRelationship(unknownNode) {
|
||||
const connectedEdge = proposal.addedEdges.find(
|
||||
(edge) =>
|
||||
(edge.fromNodeId === unknownNode.id &&
|
||||
answerDerivedNodeIds.has(edge.toNodeId)) ||
|
||||
(edge.toNodeId === unknownNode.id &&
|
||||
answerDerivedNodeIds.has(edge.fromNodeId)),
|
||||
);
|
||||
|
||||
if (connectedEdge) {
|
||||
return true;
|
||||
}
|
||||
|
||||
for (const answerDerivedNodeId of answerDerivedNodeIds) {
|
||||
const answerDerivedNode = proposalNodeById.get(answerDerivedNodeId);
|
||||
|
||||
if (
|
||||
hasExplicitNodeReference(unknownNode, answerDerivedNodeId) ||
|
||||
hasExplicitNodeReference(answerDerivedNode, unknownNode.id)
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
for (const unknownNode of addedUnknowns) {
|
||||
const meaningKeys = [
|
||||
normaliseText(unknownNode.label),
|
||||
normaliseText(unknownNode.description),
|
||||
].filter(Boolean);
|
||||
|
||||
for (const meaningKey of meaningKeys) {
|
||||
if (seenAddedUnknownMeanings.has(meaningKey)) {
|
||||
errors.push(
|
||||
`Proposal adds duplicate unknown meaning: "${unknownNode.label}"`,
|
||||
);
|
||||
break;
|
||||
}
|
||||
seenAddedUnknownMeanings.set(meaningKey, unknownNode.id);
|
||||
}
|
||||
|
||||
for (const existingUnknown of unresolvedExistingUnknowns) {
|
||||
const existingMeaningKeys = [
|
||||
normaliseText(existingUnknown.label),
|
||||
normaliseText(existingUnknown.description),
|
||||
].filter(Boolean);
|
||||
if (meaningKeys.some((key) => existingMeaningKeys.includes(key))) {
|
||||
errors.push(
|
||||
`Proposal adds a node duplicating unresolved unknown: "${existingUnknown.id}"`,
|
||||
);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (
|
||||
unknownNode.description.trim() === unknownNode.label.trim() ||
|
||||
!/\b(because|matters|important|needed|relevant|so that|to determine|to decide)\b/i.test(
|
||||
unknownNode.description,
|
||||
)
|
||||
) {
|
||||
errors.push(
|
||||
`New unknown must include why it matters in its description: "${unknownNode.id}"`,
|
||||
);
|
||||
}
|
||||
|
||||
if (!hasExplicitAnswerDerivedRelationship(unknownNode)) {
|
||||
errors.push(
|
||||
`New unknown must be explicitly related to an answer-derived node: "${unknownNode.id}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return errors;
|
||||
}
|
||||
|
||||
function validateSelectedQuestion(graph, proposal) {
|
||||
const errors = [];
|
||||
const selectedQuestion = proposal.selectedQuestion;
|
||||
const nodeById = buildNodeById(graph, proposal.addedNodes);
|
||||
|
||||
if (selectedQuestion == null) {
|
||||
return { errors, selectedQuestionNodeId: null };
|
||||
}
|
||||
|
||||
const node = nodeById.get(selectedQuestion.nodeId);
|
||||
if (!node) {
|
||||
errors.push(
|
||||
`selectedQuestion references missing node: "${selectedQuestion.nodeId}"`,
|
||||
);
|
||||
return { errors, selectedQuestionNodeId: selectedQuestion.nodeId };
|
||||
}
|
||||
|
||||
if (node.kind !== "unknown") {
|
||||
errors.push(
|
||||
`selectedQuestion must reference an unknown node: "${selectedQuestion.nodeId}"`,
|
||||
);
|
||||
}
|
||||
|
||||
const resolvesNode = proposal.resolvedUnknownNodeIds.includes(
|
||||
selectedQuestion.nodeId,
|
||||
);
|
||||
const updatedStatus = proposal.updatedNodes.find(
|
||||
(update) => update.nodeId === selectedQuestion.nodeId,
|
||||
)?.newStatus;
|
||||
const effectiveStatus = updatedStatus ?? node.status;
|
||||
|
||||
if (resolvesNode || effectiveStatus === "resolved") {
|
||||
errors.push(
|
||||
`selectedQuestion must reference an unresolved node: "${selectedQuestion.nodeId}"`,
|
||||
);
|
||||
}
|
||||
|
||||
if (
|
||||
graph.activeUnknownNodeId &&
|
||||
proposal.resolvedUnknownNodeIds.includes(graph.activeUnknownNodeId) &&
|
||||
selectedQuestion.nodeId === graph.activeUnknownNodeId
|
||||
) {
|
||||
errors.push(
|
||||
`selectedQuestion cannot reselect the previous resolved unknown: "${selectedQuestion.nodeId}"`,
|
||||
);
|
||||
}
|
||||
|
||||
if (isCompoundQuestion(selectedQuestion.question)) {
|
||||
errors.push("selectedQuestion must be a single non-compound question");
|
||||
}
|
||||
|
||||
const resolvedNodeIds = [
|
||||
...(graph.resolvedNodeIds || []),
|
||||
...(proposal.resolvedUnknownNodeIds || []),
|
||||
];
|
||||
const candidateScore = scoreUnknownCandidate(
|
||||
{
|
||||
...graph,
|
||||
nodes: [...graph.nodes, ...(proposal.addedNodes || [])],
|
||||
edges: [...graph.edges, ...(proposal.addedEdges || [])],
|
||||
},
|
||||
node,
|
||||
resolvedNodeIds,
|
||||
);
|
||||
|
||||
return { errors, selectedQuestionNodeId: selectedQuestion.nodeId };
|
||||
}
|
||||
|
||||
function validateQuestionSelectionRequirement(graph, proposal) {
|
||||
const addedConsequentialUnknowns = proposal.addedNodes.filter(
|
||||
(node) => node.kind === "unknown" && node.status !== "resolved",
|
||||
);
|
||||
|
||||
if (
|
||||
proposal.selectedQuestion == null &&
|
||||
addedConsequentialUnknowns.length > 0
|
||||
) {
|
||||
return [
|
||||
"selectedQuestion is required when consequential unresolved unknowns remain after resolving the answered unknown",
|
||||
];
|
||||
}
|
||||
|
||||
return [];
|
||||
}
|
||||
|
||||
function buildResolvedUnknownUpdate(node) {
|
||||
return {
|
||||
nodeId: node.id,
|
||||
previousStatus: node.status ?? null,
|
||||
newStatus: "resolved",
|
||||
previousValue: node.value ?? null,
|
||||
newValue: node.value ?? null,
|
||||
reason:
|
||||
"Resolved because the proposal explicitly marked this unknown as resolved.",
|
||||
};
|
||||
}
|
||||
|
||||
function reconcileResolutionSemantics(graph, proposal) {
|
||||
const nextProposal = cloneJsonSafe(proposal);
|
||||
const errors = [];
|
||||
const graphNodeById = new Map(graph.nodes.map((node) => [node.id, node]));
|
||||
const updatedNodeById = new Map(
|
||||
nextProposal.updatedNodes.map((nodeUpdate) => [
|
||||
nodeUpdate.nodeId,
|
||||
nodeUpdate,
|
||||
]),
|
||||
);
|
||||
|
||||
for (const resolvedUnknownNodeId of nextProposal.resolvedUnknownNodeIds) {
|
||||
const existingNode = graphNodeById.get(resolvedUnknownNodeId);
|
||||
|
||||
if (!existingNode) {
|
||||
errors.push(
|
||||
`Resolved unknown must reference an existing node: "${resolvedUnknownNodeId}"`,
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (existingNode.kind !== "unknown") {
|
||||
errors.push(
|
||||
`Resolved unknown must reference an existing unknown node: "${resolvedUnknownNodeId}"`,
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
const existingUpdate = updatedNodeById.get(resolvedUnknownNodeId);
|
||||
if (!existingUpdate) {
|
||||
const syntheticUpdate = buildResolvedUnknownUpdate(existingNode);
|
||||
nextProposal.updatedNodes.push(syntheticUpdate);
|
||||
updatedNodeById.set(resolvedUnknownNodeId, syntheticUpdate);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (existingUpdate.newStatus !== "resolved") {
|
||||
existingUpdate.newStatus = "resolved";
|
||||
if (existingUpdate.previousStatus == null) {
|
||||
existingUpdate.previousStatus = existingNode.status ?? null;
|
||||
}
|
||||
if (existingUpdate.previousValue === undefined) {
|
||||
existingUpdate.previousValue = existingNode.value ?? null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const update of nextProposal.updatedNodes) {
|
||||
const existingNode = graphNodeById.get(update.nodeId);
|
||||
if (
|
||||
existingNode?.kind === "unknown" &&
|
||||
update.newStatus === "resolved" &&
|
||||
!nextProposal.resolvedUnknownNodeIds.includes(update.nodeId)
|
||||
) {
|
||||
errors.push(
|
||||
`Unknown node updated to resolved must also appear in resolvedUnknownNodeIds: "${update.nodeId}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
proposal: nextProposal,
|
||||
errors,
|
||||
};
|
||||
}
|
||||
|
||||
function validateSemanticDuplicateUnknowns(graph, proposal) {
|
||||
const errors = [];
|
||||
const unresolvedUnknowns = graph.nodes.filter(
|
||||
(node) =>
|
||||
node.kind === "unknown" &&
|
||||
!proposal.resolvedUnknownNodeIds.includes(node.id),
|
||||
);
|
||||
|
||||
for (const addedNode of proposal.addedNodes) {
|
||||
const addedTexts = [
|
||||
normaliseText(addedNode.label),
|
||||
normaliseText(addedNode.description),
|
||||
].filter(Boolean);
|
||||
|
||||
for (const unresolvedUnknown of unresolvedUnknowns) {
|
||||
const unresolvedTexts = [
|
||||
normaliseText(unresolvedUnknown.label),
|
||||
normaliseText(unresolvedUnknown.description),
|
||||
].filter(Boolean);
|
||||
|
||||
const duplicatesMeaning = addedTexts.some((text) =>
|
||||
unresolvedTexts.includes(text),
|
||||
);
|
||||
|
||||
if (!duplicatesMeaning) continue;
|
||||
|
||||
const linkedToUnknown = proposal.addedEdges.some(
|
||||
(edge) =>
|
||||
(edge.fromNodeId === addedNode.id &&
|
||||
edge.toNodeId === unresolvedUnknown.id) ||
|
||||
(edge.toNodeId === addedNode.id &&
|
||||
edge.fromNodeId === unresolvedUnknown.id),
|
||||
);
|
||||
|
||||
const updatedUnknown = proposal.updatedNodes.some(
|
||||
(update) => update.nodeId === unresolvedUnknown.id,
|
||||
);
|
||||
|
||||
if (!linkedToUnknown && !updatedUnknown) {
|
||||
errors.push(
|
||||
`Proposal adds a node duplicating unresolved unknown meaning without linking or resolving it: "${unresolvedUnknown.id}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return errors;
|
||||
}
|
||||
|
||||
function buildAffectedNodeIds(graph, proposal) {
|
||||
const affected = new Set(proposal.affectedNodeIds ?? []);
|
||||
|
||||
for (const update of proposal.updatedNodes ?? []) {
|
||||
affected.add(update.nodeId);
|
||||
for (const nodeId of findAffectedNodes(graph, update.nodeId)) {
|
||||
affected.add(nodeId);
|
||||
}
|
||||
}
|
||||
|
||||
for (const nodeId of proposal.resolvedUnknownNodeIds ?? []) {
|
||||
affected.add(nodeId);
|
||||
for (const affectedNodeId of findAffectedNodes(graph, nodeId)) {
|
||||
affected.add(affectedNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
return [...affected];
|
||||
}
|
||||
|
||||
function buildChangesApplied(proposal, affectedNodeIds) {
|
||||
return {
|
||||
addedNodeCount: proposal.addedNodes.length,
|
||||
addedUnknownCount: proposal.addedNodes.filter(
|
||||
(node) => node.kind === "unknown",
|
||||
).length,
|
||||
updatedNodeCount: proposal.updatedNodes.length,
|
||||
addedEdgeCount: proposal.addedEdges.length,
|
||||
removedEdgeCount: proposal.removedEdgeIds.length,
|
||||
resolvedUnknownCount: proposal.resolvedUnknownNodeIds.length,
|
||||
affectedNodeCount: affectedNodeIds.length,
|
||||
};
|
||||
}
|
||||
|
||||
export function applyValidatedProposal({ situationGraph, proposal }) {
|
||||
const graphValidation = situationGraphSchema.safeParse(situationGraph);
|
||||
const proposalValidation = graphUpdateSchema.safeParse(proposal);
|
||||
|
||||
const existingGraphReferenceValidation = graphValidation.success
|
||||
? validateGraphReferences(situationGraph)
|
||||
: null;
|
||||
|
||||
const existingDuplicateNodeIds = graphValidation.success
|
||||
? detectDuplicateNodeIds(situationGraph.nodes)
|
||||
: [];
|
||||
const existingDuplicateEdgeIds = graphValidation.success
|
||||
? collectDuplicateEdgeIds(situationGraph.edges)
|
||||
: [];
|
||||
|
||||
if (
|
||||
!graphValidation.success ||
|
||||
!existingGraphReferenceValidation?.valid ||
|
||||
existingDuplicateNodeIds.length > 0 ||
|
||||
existingDuplicateEdgeIds.length > 0
|
||||
) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "graph_validation",
|
||||
errors: [
|
||||
...(!graphValidation.success
|
||||
? zodIssuesToErrors(graphValidation.error)
|
||||
: []),
|
||||
...(!existingGraphReferenceValidation?.valid
|
||||
? existingGraphReferenceValidation.errors
|
||||
: []),
|
||||
...existingDuplicateNodeIds.map(
|
||||
({ nodeId, count }) =>
|
||||
`Graph contains duplicate node ID: "${nodeId}" (${count} occurrences)`,
|
||||
),
|
||||
...existingDuplicateEdgeIds.map(
|
||||
({ edgeId, count }) =>
|
||||
`Graph contains duplicate edge ID: "${edgeId}" (${count} occurrences)`,
|
||||
),
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
if (!proposalValidation.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "proposal_compatibility",
|
||||
errors: zodIssuesToErrors(proposalValidation.error),
|
||||
};
|
||||
}
|
||||
|
||||
const reconciledProposal = reconcileResolutionSemantics(
|
||||
situationGraph,
|
||||
proposalValidation.data,
|
||||
);
|
||||
const validatedProposal = reconciledProposal.proposal;
|
||||
const proposalCompatibilityErrors = [];
|
||||
proposalCompatibilityErrors.push(...reconciledProposal.errors);
|
||||
const proposalGraphValidation = validateGraphUpdate(
|
||||
situationGraph,
|
||||
validatedProposal,
|
||||
);
|
||||
|
||||
if (!proposalGraphValidation.valid) {
|
||||
proposalCompatibilityErrors.push(...proposalGraphValidation.errors);
|
||||
}
|
||||
|
||||
const existingEdgeIds = new Set(situationGraph.edges.map((edge) => edge.id));
|
||||
const reachableNodeIds = new Set([
|
||||
...situationGraph.nodes.map((node) => node.id),
|
||||
...validatedProposal.addedNodes.map((node) => node.id),
|
||||
]);
|
||||
const addedEdgeDuplicateIds = collectDuplicateEdgeIds(
|
||||
validatedProposal.addedEdges,
|
||||
);
|
||||
proposalCompatibilityErrors.push(
|
||||
...addedEdgeDuplicateIds.map(
|
||||
({ edgeId, count }) =>
|
||||
`Proposal contains duplicate added edge ID: "${edgeId}" (${count} occurrences)`,
|
||||
),
|
||||
);
|
||||
|
||||
for (const edge of validatedProposal.addedEdges) {
|
||||
if (existingEdgeIds.has(edge.id)) {
|
||||
proposalCompatibilityErrors.push(
|
||||
`Cannot add edge with duplicate ID: "${edge.id}"`,
|
||||
);
|
||||
}
|
||||
if (!reachableNodeIds.has(edge.fromNodeId)) {
|
||||
proposalCompatibilityErrors.push(
|
||||
`Added edge references non-existent fromNodeId: "${edge.fromNodeId}"`,
|
||||
);
|
||||
}
|
||||
if (!reachableNodeIds.has(edge.toNodeId)) {
|
||||
proposalCompatibilityErrors.push(
|
||||
`Added edge references non-existent toNodeId: "${edge.toNodeId}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const removedEdgeIds = new Set(validatedProposal.removedEdgeIds);
|
||||
for (const edgeId of removedEdgeIds) {
|
||||
if (!existingEdgeIds.has(edgeId)) {
|
||||
proposalCompatibilityErrors.push(
|
||||
`Cannot remove non-existent edge: "${edgeId}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const combinedNodeDuplicates = detectDuplicateNodeIds([
|
||||
...situationGraph.nodes,
|
||||
...validatedProposal.addedNodes,
|
||||
]);
|
||||
proposalCompatibilityErrors.push(
|
||||
...combinedNodeDuplicates.map(
|
||||
({ nodeId, count }) =>
|
||||
`Proposal would produce duplicate node ID: "${nodeId}" (${count} occurrences)`,
|
||||
),
|
||||
);
|
||||
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateSemanticDuplicateUnknowns(situationGraph, validatedProposal),
|
||||
);
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateAddedUnknowns(situationGraph, validatedProposal),
|
||||
);
|
||||
|
||||
const selectedQuestionValidation = validateSelectedQuestion(
|
||||
situationGraph,
|
||||
validatedProposal,
|
||||
);
|
||||
proposalCompatibilityErrors.push(...selectedQuestionValidation.errors);
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateQuestionSelectionRequirement(situationGraph, validatedProposal),
|
||||
);
|
||||
|
||||
if (proposalCompatibilityErrors.length > 0) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "proposal_compatibility",
|
||||
errors: proposalCompatibilityErrors,
|
||||
};
|
||||
}
|
||||
|
||||
const graphSnapshot = cloneJsonSafe(situationGraph);
|
||||
const proposalSnapshot = cloneJsonSafe(validatedProposal);
|
||||
const previousActiveUnknownNodeId = graphSnapshot.activeUnknownNodeId ?? null;
|
||||
const affectedNodeIds = buildAffectedNodeIds(graphSnapshot, proposalSnapshot);
|
||||
|
||||
const applied = applyGraphUpdate(graphSnapshot, proposalSnapshot);
|
||||
if (!applied.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "application",
|
||||
errors: applied.errors,
|
||||
};
|
||||
}
|
||||
|
||||
const updatedSituationGraph = {
|
||||
...graphSnapshot,
|
||||
nodes: applied.nodes,
|
||||
edges: applied.edges,
|
||||
resolvedNodeIds: applied.resolvedNodeIds,
|
||||
};
|
||||
|
||||
const activeUnknownWasResolved =
|
||||
previousActiveUnknownNodeId != null &&
|
||||
updatedSituationGraph.resolvedNodeIds.includes(previousActiveUnknownNodeId);
|
||||
|
||||
let newActiveUnknownNodeId = previousActiveUnknownNodeId;
|
||||
if (activeUnknownWasResolved) {
|
||||
newActiveUnknownNodeId = null;
|
||||
}
|
||||
|
||||
if (validatedProposal.selectedQuestion?.nodeId) {
|
||||
newActiveUnknownNodeId = validatedProposal.selectedQuestion.nodeId;
|
||||
}
|
||||
|
||||
const remainingUnknownExists =
|
||||
newActiveUnknownNodeId != null &&
|
||||
updatedSituationGraph.nodes.some(
|
||||
(node) =>
|
||||
node.id === newActiveUnknownNodeId &&
|
||||
node.kind === "unknown" &&
|
||||
!updatedSituationGraph.resolvedNodeIds.includes(node.id),
|
||||
);
|
||||
|
||||
if (!remainingUnknownExists) {
|
||||
newActiveUnknownNodeId =
|
||||
selectActiveUnknownCandidate(
|
||||
updatedSituationGraph,
|
||||
updatedSituationGraph.resolvedNodeIds,
|
||||
)?.nodeId ?? null;
|
||||
}
|
||||
|
||||
const deterministicSelection = selectActiveUnknownCandidate(
|
||||
updatedSituationGraph,
|
||||
updatedSituationGraph.resolvedNodeIds,
|
||||
);
|
||||
|
||||
if (deterministicSelection?.nodeId) {
|
||||
newActiveUnknownNodeId = deterministicSelection.nodeId;
|
||||
}
|
||||
|
||||
updatedSituationGraph.activeUnknownNodeId = newActiveUnknownNodeId;
|
||||
updatedSituationGraph.currentSummary = describeGraph(updatedSituationGraph);
|
||||
|
||||
const selectedNode = deterministicSelection?.nodeId
|
||||
? updatedSituationGraph.nodes.find(
|
||||
(node) => node.id === deterministicSelection.nodeId,
|
||||
)
|
||||
: null;
|
||||
const formulatedQuestion = selectedNode
|
||||
? formulateQuestion({
|
||||
node: selectedNode,
|
||||
graph: updatedSituationGraph,
|
||||
context: {
|
||||
resolvedValues: validatedProposal.updatedNodes
|
||||
.map((update) => update.newValue)
|
||||
.filter(
|
||||
(value) => typeof value === "string" && value.trim().length > 0,
|
||||
),
|
||||
},
|
||||
})
|
||||
: null;
|
||||
|
||||
const finalSelectedQuestion = deterministicSelection
|
||||
? {
|
||||
nodeId: deterministicSelection.nodeId,
|
||||
question:
|
||||
formulatedQuestion?.question || deterministicSelection.question,
|
||||
reason: formulatedQuestion?.reason || deterministicSelection.reason,
|
||||
strategy: formulatedQuestion?.strategy,
|
||||
}
|
||||
: null;
|
||||
|
||||
const resultGraphValidation = situationGraphSchema.safeParse(
|
||||
updatedSituationGraph,
|
||||
);
|
||||
const resultReferenceValidation = resultGraphValidation.success
|
||||
? validateGraphReferences(updatedSituationGraph)
|
||||
: null;
|
||||
const resultDuplicateNodeIds = resultGraphValidation.success
|
||||
? detectDuplicateNodeIds(updatedSituationGraph.nodes)
|
||||
: [];
|
||||
const resultDuplicateEdgeIds = resultGraphValidation.success
|
||||
? collectDuplicateEdgeIds(updatedSituationGraph.edges)
|
||||
: [];
|
||||
|
||||
if (
|
||||
!resultGraphValidation.success ||
|
||||
!resultReferenceValidation?.valid ||
|
||||
resultDuplicateNodeIds.length > 0 ||
|
||||
resultDuplicateEdgeIds.length > 0
|
||||
) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "result_validation",
|
||||
errors: [
|
||||
...(!resultGraphValidation.success
|
||||
? zodIssuesToErrors(resultGraphValidation.error)
|
||||
: []),
|
||||
...(!resultReferenceValidation?.valid
|
||||
? resultReferenceValidation.errors
|
||||
: []),
|
||||
...resultDuplicateNodeIds.map(
|
||||
({ nodeId, count }) =>
|
||||
`Updated graph contains duplicate node ID: "${nodeId}" (${count} occurrences)`,
|
||||
),
|
||||
...resultDuplicateEdgeIds.map(
|
||||
({ edgeId, count }) =>
|
||||
`Updated graph contains duplicate edge ID: "${edgeId}" (${count} occurrences)`,
|
||||
),
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
updatedSituationGraph,
|
||||
graphUpdate: validatedProposal,
|
||||
affectedNodeIds,
|
||||
resolvedUnknownNodeIds: validatedProposal.resolvedUnknownNodeIds,
|
||||
previousActiveUnknownNodeId,
|
||||
newActiveUnknownNodeId,
|
||||
selectedQuestion: finalSelectedQuestion,
|
||||
changesApplied: buildChangesApplied(validatedProposal, affectedNodeIds),
|
||||
graphReferenceValidation: resultReferenceValidation,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,305 @@
|
||||
/**
|
||||
* Deterministic situation graph builder — builds initial graph from scenario text.
|
||||
* Takes v0.2/v0.3 analysis output (from analyseScenario) and constructs a SituationGraph.
|
||||
*/
|
||||
|
||||
import {
|
||||
situationNodeSchema,
|
||||
situationEdgeSchema,
|
||||
makeNodeId,
|
||||
} from "./schema.js";
|
||||
|
||||
/**
|
||||
* Build an initial situation graph from a v0.3 reconstruction result.
|
||||
* @param {{ reconstruction: object, evidence: object[] | undefined }} analysisData
|
||||
* @returns {{ nodes: import("./schema.js").SituationNode[], edges: import("./schema.js").SituationEdge[] }}
|
||||
*/
|
||||
export function buildInitialGraph(analysisData) {
|
||||
const { reconstruction, evidence = [] } = analysisData;
|
||||
|
||||
if (!reconstruction || !reconstruction.summary) {
|
||||
return { nodes: [], edges: [] };
|
||||
}
|
||||
|
||||
const nodeMap = new Map(); // label -> node
|
||||
|
||||
// ── Helper: register or get a node by label ────────────
|
||||
|
||||
function ensureNode(
|
||||
label,
|
||||
kind,
|
||||
status,
|
||||
description,
|
||||
value,
|
||||
unit,
|
||||
confidence,
|
||||
) {
|
||||
if (nodeMap.has(label)) return nodeMap.get(label);
|
||||
|
||||
const id = makeNodeId(label);
|
||||
const node = situationNodeSchema.parse({
|
||||
id,
|
||||
label,
|
||||
description: description ?? label,
|
||||
kind,
|
||||
status,
|
||||
confidence,
|
||||
value: value ?? null,
|
||||
unit: unit ?? null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
});
|
||||
nodeMap.set(label, node);
|
||||
return node;
|
||||
}
|
||||
|
||||
// ── Evidence lookup ────────────────────────────────────
|
||||
|
||||
const evidenceMap = new Map();
|
||||
for (const ev of evidence) {
|
||||
if (ev.id) evidenceMap.set(ev.id, ev);
|
||||
}
|
||||
|
||||
function addEvidenceToNode(nodeId, evidenceId) {
|
||||
const node = Object.values(nodeMap).find((n) => n.id === nodeId);
|
||||
if (node && !node.evidenceIds.includes(evidenceId)) {
|
||||
node.evidenceIds.push(evidenceId);
|
||||
}
|
||||
}
|
||||
|
||||
// ── Extract observed states as nodes ────────────────────
|
||||
|
||||
const summaryNode = ensureNode(
|
||||
reconstruction.summary || "Situation Summary",
|
||||
"state",
|
||||
"provisional",
|
||||
"Summary of the situation from the scenario text",
|
||||
null,
|
||||
null,
|
||||
"medium",
|
||||
);
|
||||
|
||||
// Collect all observable quantities as metric nodes
|
||||
const metrics = new Map();
|
||||
|
||||
if (reconstruction.observedStates) {
|
||||
for (const obs of reconstruction.observedStates) {
|
||||
const node = ensureNode(
|
||||
obs.description || obs.label,
|
||||
"observation",
|
||||
"supported",
|
||||
obs.description || obs.label,
|
||||
null,
|
||||
null,
|
||||
obs.confidence || "medium",
|
||||
);
|
||||
|
||||
if (obs.id) node.evidenceIds.push(obs.id);
|
||||
}
|
||||
}
|
||||
|
||||
// Actors as states/nodes
|
||||
if (reconstruction.actors) {
|
||||
for (const actor of reconstruction.actors) {
|
||||
ensureNode(
|
||||
actor.description || actor.label,
|
||||
"observation",
|
||||
"supported",
|
||||
actor.description || actor.label,
|
||||
null,
|
||||
null,
|
||||
actor.confidence || "medium",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (reconstruction.systemsOrObjects) {
|
||||
for (const sys of reconstruction.systemsOrObjects) {
|
||||
ensureNode(
|
||||
sys.description || sys.label,
|
||||
"metric",
|
||||
"known",
|
||||
sys.description || sys.label,
|
||||
null,
|
||||
null,
|
||||
sys.confidence || "medium",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Differences as relationship nodes
|
||||
if (reconstruction.differences) {
|
||||
for (const diff of reconstruction.differences) {
|
||||
const node = ensureNode(
|
||||
diff.description || "Difference",
|
||||
"relationship",
|
||||
"supported",
|
||||
diff.description || "Difference",
|
||||
null,
|
||||
null,
|
||||
diff.confidence || "medium",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Contradictions as nodes
|
||||
if (reconstruction.contradictions) {
|
||||
for (const c of reconstruction.contradictions) {
|
||||
const node = ensureNode(
|
||||
c.description || c.label,
|
||||
"relationship",
|
||||
"supported",
|
||||
c.description || c.label,
|
||||
null,
|
||||
null,
|
||||
c.confidence || "medium",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Important unknowns as unknown nodes
|
||||
const unknownNodes = [];
|
||||
if (reconstruction.importantUnknowns) {
|
||||
for (const unk of reconstruction.importantUnknowns) {
|
||||
const node = ensureNode(
|
||||
unk.description || unk.label,
|
||||
"unknown",
|
||||
"unknown",
|
||||
unk.description || "Unknown factor in the situation",
|
||||
null,
|
||||
null,
|
||||
unk.confidence || "low",
|
||||
);
|
||||
unknownNodes.push(node);
|
||||
}
|
||||
}
|
||||
|
||||
// Plausible interpretations
|
||||
if (reconstruction.plausibleInterpretations) {
|
||||
for (const interp of reconstruction.plausibleInterpretations) {
|
||||
ensureNode(
|
||||
interp.description || interp.label,
|
||||
"assumption",
|
||||
"provisional",
|
||||
interp.description || "Plausible interpretation",
|
||||
null,
|
||||
null,
|
||||
interp.confidence || "low",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Known transitions
|
||||
if (reconstruction.knownTransitions) {
|
||||
for (const trans of reconstruction.knownTransitions) {
|
||||
ensureNode(
|
||||
`${trans.entity}: ${trans.previousState} → ${trans.currentState}`,
|
||||
"transition",
|
||||
trans.explanationStatus === "confirmed" ? "known" : "provisional",
|
||||
trans.description ||
|
||||
`Transition: ${trans.entity} from ${trans.previousState} to ${trans.currentState}`,
|
||||
null,
|
||||
null,
|
||||
trans.confidence || "medium",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ── Build edges between nodes ────────────────────────
|
||||
|
||||
const nodeArr = Array.from(nodeMap.values());
|
||||
const edges = [];
|
||||
|
||||
// Link actors → observed states as measures relationships
|
||||
let actorNodes = [];
|
||||
let metricNodes = [];
|
||||
let unknownNodeIds = [];
|
||||
|
||||
for (const n of nodeArr) {
|
||||
if (n.kind === "observation" && n.status === "supported") {
|
||||
// These are observations — link to summary
|
||||
edges.push(
|
||||
situationEdgeSchema.parse({
|
||||
id: `e-sum-${n.id}`,
|
||||
fromNodeId: n.id,
|
||||
toNodeId: summaryNode.id,
|
||||
relationship: "supports",
|
||||
confidence: n.confidence || "medium",
|
||||
description: `${n.label} supports the summary`,
|
||||
}),
|
||||
);
|
||||
}
|
||||
if (n.kind === "unknown") {
|
||||
unknownNodeIds.push(n.id);
|
||||
edges.push(
|
||||
situationEdgeSchema.parse({
|
||||
id: `e-unk-${n.id}`,
|
||||
fromNodeId: n.id,
|
||||
toNodeId: summaryNode.id,
|
||||
relationship: "depends_on",
|
||||
confidence: n.confidence || "low",
|
||||
description: `${n.label} is an unresolved factor for this situation`,
|
||||
}),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return { nodes: nodeArr, edges };
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a minimal starting graph for any scenario.
|
||||
* Used when analysis has no reconstruction data (e.g., error state).
|
||||
*/
|
||||
export function buildMinimalGraph(scenario) {
|
||||
const shortLabel = scenario.slice(0, 80);
|
||||
|
||||
return {
|
||||
nodes: [
|
||||
situationNodeSchema.parse({
|
||||
id: "n0",
|
||||
label: shortLabel,
|
||||
description: `Initial situation from: "${scenario.slice(0, 200)}"`,
|
||||
kind: "state",
|
||||
status: "provisional",
|
||||
confidence: "low",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
}),
|
||||
],
|
||||
edges: [],
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert graph nodes/edges to a human-readable summary for display.
|
||||
*/
|
||||
export function describeGraph(graph) {
|
||||
const parts = [];
|
||||
|
||||
// Count by kind
|
||||
const byKind = {};
|
||||
for (const n of graph.nodes) {
|
||||
byKind[n.kind] = (byKind[n.kind] || 0) + 1;
|
||||
}
|
||||
|
||||
parts.push(
|
||||
`Nodes: ${Object.entries(byKind)
|
||||
.map(([k, v]) => `${v} ${k}`)
|
||||
.join(", ")}`,
|
||||
);
|
||||
parts.push(`Edges: ${graph.edges.length} total`);
|
||||
parts.push(
|
||||
`Unknowns: ${graph.nodes.filter((n) => n.status === "unknown").length} unresolved`,
|
||||
);
|
||||
|
||||
return parts.join(" | ");
|
||||
}
|
||||
@@ -0,0 +1,333 @@
|
||||
/**
|
||||
* Situation Graph Case Orchestrator — manages the lifecycle of a case.
|
||||
* startCase builds initial graph from analysis; updateCase applies answers.
|
||||
*/
|
||||
|
||||
import { analyseScenario } from "../analysis.js";
|
||||
import { assertConfig } from "../config.js";
|
||||
import { getProvider } from "../llm/provider.js";
|
||||
import {
|
||||
makeGraph,
|
||||
startCaseRequestSchema,
|
||||
situationGraphSchema,
|
||||
updateCaseRequestSchema,
|
||||
} from "./schema.js";
|
||||
import { buildInitialGraph, describeGraph } from "./builder.js";
|
||||
import { applyValidatedProposal } from "./apply-proposal.js";
|
||||
import { buildGraphUpdatePrompt } from "./prompt-builder.js";
|
||||
import { parseGraphUpdateProposal } from "./update-proposal.js";
|
||||
import {
|
||||
selectActiveUnknownCandidate,
|
||||
validateGraphReferences,
|
||||
} from "./utils.js";
|
||||
|
||||
function toValidationErrors(error) {
|
||||
return (
|
||||
error?.errors?.map((issue) => ({
|
||||
path: issue.path,
|
||||
message: issue.message,
|
||||
code: issue.code,
|
||||
})) ?? [{ message: "Validation failed" }]
|
||||
);
|
||||
}
|
||||
|
||||
function buildDiagnostics({ analysis, graph, graphReferenceValidation }) {
|
||||
return {
|
||||
promptVersion: analysis?.promptVersion ?? null,
|
||||
modelName: analysis?.modelName ?? null,
|
||||
responseDurationMs: analysis?.responseDurationMs ?? null,
|
||||
validationStatus: analysis?.validationStatus ?? "invalid",
|
||||
nodeCount: graph?.nodes?.length ?? 0,
|
||||
edgeCount: graph?.edges?.length ?? 0,
|
||||
graphReferenceValidation,
|
||||
compatibilityApplied: analysis?.compatibilityApplied ?? false,
|
||||
compatibilityChanges: analysis?.compatibilityChanges ?? [],
|
||||
compatibilityWarnings: analysis?.compatibilityWarnings ?? [],
|
||||
};
|
||||
}
|
||||
|
||||
function buildUpdateDiagnostics({
|
||||
promptVersion,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied,
|
||||
graph,
|
||||
graphReferenceValidation,
|
||||
}) {
|
||||
return {
|
||||
promptVersion: promptVersion ?? "v0.4",
|
||||
modelName: modelName ?? null,
|
||||
responseDurationMs: responseDurationMs ?? null,
|
||||
validationStatus: "valid",
|
||||
nodeCount: graph?.nodes?.length ?? 0,
|
||||
edgeCount: graph?.edges?.length ?? 0,
|
||||
graphReferenceValidation: graphReferenceValidation ?? {
|
||||
valid: true,
|
||||
errors: [],
|
||||
},
|
||||
normalisationsApplied: normalisationsApplied ?? [],
|
||||
};
|
||||
}
|
||||
|
||||
export async function startCase(body) {
|
||||
const parsedRequest = startCaseRequestSchema.safeParse(body);
|
||||
|
||||
if (!parsedRequest.success) {
|
||||
return {
|
||||
success: false,
|
||||
error: "Invalid start-case request",
|
||||
validationErrors: toValidationErrors(parsedRequest.error),
|
||||
statusCode: 400,
|
||||
};
|
||||
}
|
||||
|
||||
const { scenario, promptVersion } = parsedRequest.data;
|
||||
const analysis = await analyseScenario(scenario, { promptVersion });
|
||||
|
||||
if (!analysis.success) {
|
||||
return {
|
||||
success: false,
|
||||
error: analysis.error ?? "Scenario analysis failed",
|
||||
diagnostics: buildDiagnostics({
|
||||
analysis,
|
||||
graph: null,
|
||||
graphReferenceValidation: null,
|
||||
}),
|
||||
analysisErrors: analysis.errors ?? undefined,
|
||||
rawResponse: analysis.rawResponse ?? undefined,
|
||||
statusCode: Number(analysis.statusCode) || 502,
|
||||
};
|
||||
}
|
||||
|
||||
const initialGraph = buildInitialGraph({
|
||||
reconstruction: analysis.reconstruction,
|
||||
evidence: analysis.evidence,
|
||||
});
|
||||
|
||||
const currentSummary = describeGraph(initialGraph);
|
||||
const activeUnknownNodeId =
|
||||
selectActiveUnknownCandidate(
|
||||
{
|
||||
...initialGraph,
|
||||
resolvedNodeIds: [],
|
||||
},
|
||||
[],
|
||||
)?.nodeId ?? null;
|
||||
|
||||
const situationGraph = makeGraph({
|
||||
centralStatement: scenario,
|
||||
nodes: initialGraph.nodes,
|
||||
edges: initialGraph.edges,
|
||||
activeUnknownNodeId,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary,
|
||||
});
|
||||
|
||||
situationGraphSchema.parse(situationGraph);
|
||||
|
||||
const graphReferenceValidation = validateGraphReferences(situationGraph);
|
||||
if (!graphReferenceValidation.valid) {
|
||||
return {
|
||||
success: false,
|
||||
error: "Situation graph reference validation failed",
|
||||
diagnostics: buildDiagnostics({
|
||||
analysis,
|
||||
graph: situationGraph,
|
||||
graphReferenceValidation,
|
||||
}),
|
||||
validationErrors: graphReferenceValidation.errors,
|
||||
statusCode: 500,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
situationGraph,
|
||||
selectedQuestion: analysis.nextQuestion ?? null,
|
||||
diagnostics: buildDiagnostics({
|
||||
analysis,
|
||||
graph: situationGraph,
|
||||
graphReferenceValidation,
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
export async function updateCase() {
|
||||
return updateCaseWithDependencies(...arguments);
|
||||
}
|
||||
|
||||
function sanitiseErrorMessage(error, fallbackMessage) {
|
||||
if (typeof error?.message === "string" && error.message.trim().length > 0) {
|
||||
return error.message;
|
||||
}
|
||||
|
||||
return fallbackMessage;
|
||||
}
|
||||
|
||||
async function updateCaseWithDependencies(body, dependencies = {}) {
|
||||
const parsedRequest = updateCaseRequestSchema.safeParse(body);
|
||||
|
||||
if (!parsedRequest.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid update-case request",
|
||||
validationErrors: toValidationErrors(parsedRequest.error),
|
||||
statusCode: 400,
|
||||
};
|
||||
}
|
||||
|
||||
const { situationGraph, previousQuestion, answer, promptVersion } =
|
||||
parsedRequest.data;
|
||||
|
||||
const graphSchemaValidation = situationGraphSchema.safeParse(situationGraph);
|
||||
const graphReferenceValidation = validateGraphReferences(situationGraph);
|
||||
|
||||
if (!graphSchemaValidation.success || !graphReferenceValidation.valid) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "graph_validation",
|
||||
error: "Invalid situation graph",
|
||||
graphValidationErrors: [
|
||||
...(!graphSchemaValidation.success
|
||||
? toValidationErrors(graphSchemaValidation.error)
|
||||
: []),
|
||||
...(!graphReferenceValidation.valid
|
||||
? graphReferenceValidation.errors
|
||||
: []),
|
||||
],
|
||||
statusCode: 400,
|
||||
};
|
||||
}
|
||||
|
||||
const buildPrompt =
|
||||
dependencies.buildGraphUpdatePrompt ?? buildGraphUpdatePrompt;
|
||||
const parseProposal =
|
||||
dependencies.parseGraphUpdateProposal ?? parseGraphUpdateProposal;
|
||||
const applyProposalUpdate =
|
||||
dependencies.applyValidatedProposal ?? applyValidatedProposal;
|
||||
const shouldApplyProposal = dependencies.applyProposal === true;
|
||||
|
||||
let modelName = null;
|
||||
let rawResponse;
|
||||
const startedAt = Date.now();
|
||||
|
||||
try {
|
||||
const config = dependencies.config ?? assertConfig();
|
||||
modelName = config.OLLAMA_MODEL;
|
||||
|
||||
const prompt = buildPrompt({
|
||||
situationGraph,
|
||||
previousQuestion,
|
||||
answer,
|
||||
promptVersion,
|
||||
});
|
||||
|
||||
const provider = dependencies.provider ?? getProvider();
|
||||
rawResponse = await provider.generateReconstruction(prompt, modelName);
|
||||
} catch (error) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "provider",
|
||||
error: "Graph update proposal generation failed",
|
||||
providerErrors: [
|
||||
sanitiseErrorMessage(
|
||||
error,
|
||||
"Provider failed to generate graph update proposal",
|
||||
),
|
||||
],
|
||||
diagnostics: {
|
||||
promptVersion: promptVersion ?? null,
|
||||
modelName,
|
||||
responseDurationMs: Date.now() - startedAt,
|
||||
normalisationsApplied: [],
|
||||
},
|
||||
statusCode: 502,
|
||||
};
|
||||
}
|
||||
|
||||
const parsedProposal = parseProposal(rawResponse);
|
||||
const responseDurationMs = Date.now() - startedAt;
|
||||
|
||||
if (!parsedProposal.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "proposal_validation",
|
||||
error: "Invalid graph update proposal",
|
||||
proposalErrors: parsedProposal.errors,
|
||||
diagnostics: {
|
||||
promptVersion: promptVersion ?? null,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
},
|
||||
statusCode: 502,
|
||||
};
|
||||
}
|
||||
|
||||
if (shouldApplyProposal) {
|
||||
const applicationResult = applyProposalUpdate({
|
||||
situationGraph,
|
||||
proposal: parsedProposal.proposal,
|
||||
});
|
||||
|
||||
if (!applicationResult.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: applicationResult.stage,
|
||||
errors: applicationResult.errors,
|
||||
diagnostics: {
|
||||
...buildUpdateDiagnostics({
|
||||
promptVersion,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
graph: situationGraph,
|
||||
graphReferenceValidation: graphReferenceValidation,
|
||||
}),
|
||||
},
|
||||
statusCode:
|
||||
applicationResult.stage === "application" ||
|
||||
applicationResult.stage === "result_validation"
|
||||
? 500
|
||||
: 400,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stage: "update_applied",
|
||||
updatedSituationGraph: applicationResult.updatedSituationGraph,
|
||||
proposal: applicationResult.graphUpdate,
|
||||
selectedQuestion: applicationResult.selectedQuestion,
|
||||
affectedNodeIds: applicationResult.affectedNodeIds,
|
||||
resolvedUnknownNodeIds: applicationResult.resolvedUnknownNodeIds,
|
||||
previousActiveUnknownNodeId:
|
||||
applicationResult.previousActiveUnknownNodeId,
|
||||
newActiveUnknownNodeId: applicationResult.newActiveUnknownNodeId,
|
||||
changesApplied: applicationResult.changesApplied,
|
||||
diagnostics: buildUpdateDiagnostics({
|
||||
promptVersion,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
graph: applicationResult.updatedSituationGraph,
|
||||
graphReferenceValidation: applicationResult.graphReferenceValidation,
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stage: "proposal_ready",
|
||||
proposal: parsedProposal.proposal,
|
||||
diagnostics: buildUpdateDiagnostics({
|
||||
promptVersion,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
graph: situationGraph,
|
||||
graphReferenceValidation,
|
||||
}),
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
import {
|
||||
ConfidenceLevel,
|
||||
SituationKind,
|
||||
SituationRelationship,
|
||||
SituationStatus,
|
||||
} from "./schema.js";
|
||||
|
||||
const DEFAULT_PROMPT_VERSION = "v0.4";
|
||||
|
||||
function formatEnumValues(values) {
|
||||
return Object.values(values).join(" | ");
|
||||
}
|
||||
|
||||
function formatGraph(graph) {
|
||||
return JSON.stringify(graph, null, 2);
|
||||
}
|
||||
|
||||
function formatExampleAnswerBlock() {
|
||||
return [
|
||||
"Example answer the model must be able to handle without hard-coding output:",
|
||||
'"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units."',
|
||||
"This may justify resolving a rate-related unknown or updating a metric node, but only if the current graph and answer support that proposal.",
|
||||
].join("\n");
|
||||
}
|
||||
|
||||
export function buildGraphUpdatePrompt({
|
||||
situationGraph,
|
||||
previousQuestion,
|
||||
answer,
|
||||
promptVersion = DEFAULT_PROMPT_VERSION,
|
||||
}) {
|
||||
const nodeKinds = formatEnumValues(SituationKind);
|
||||
const nodeStatuses = formatEnumValues(SituationStatus);
|
||||
const edgeRelationships = formatEnumValues(SituationRelationship);
|
||||
const confidenceLevels = formatEnumValues(ConfidenceLevel);
|
||||
|
||||
return `You are proposing a graph update for Confidence Engine ${promptVersion}.
|
||||
|
||||
Return exactly one JSON object matching the GraphUpdate contract.
|
||||
Return JSON only. Do not include markdown, explanation, or any text before or after the JSON object.
|
||||
|
||||
## Current Situation Graph
|
||||
${formatGraph(situationGraph)}
|
||||
|
||||
## Previous Selected Question
|
||||
${previousQuestion}
|
||||
|
||||
## User Answer
|
||||
${answer}
|
||||
|
||||
## Allowed Node Kinds
|
||||
${nodeKinds}
|
||||
|
||||
## Allowed Node Statuses
|
||||
${nodeStatuses}
|
||||
|
||||
## Allowed Edge Relationships
|
||||
${edgeRelationships}
|
||||
|
||||
## Allowed Confidence Values
|
||||
${confidenceLevels}
|
||||
|
||||
## Required JSON Field Names
|
||||
The JSON object must contain exactly these top-level fields:
|
||||
- addedNodes
|
||||
- updatedNodes
|
||||
- addedEdges
|
||||
- removedEdgeIds
|
||||
- resolvedUnknownNodeIds
|
||||
- affectedNodeIds
|
||||
- selectedQuestion
|
||||
|
||||
## Required Shapes
|
||||
- addedNodes: array of nodes using these exact keys:
|
||||
id, label, description, kind, status, confidence, value, unit, evidenceIds, dependsOn, affects, parentId, childIds
|
||||
- updatedNodes: array of node updates using these exact keys:
|
||||
nodeId, previousStatus, newStatus, previousValue, newValue, reason
|
||||
- addedEdges: array of edges using these exact keys:
|
||||
id, fromNodeId, toNodeId, relationship, confidence, description
|
||||
- removedEdgeIds: array of strings
|
||||
- resolvedUnknownNodeIds: array of strings
|
||||
- affectedNodeIds: array of strings
|
||||
- selectedQuestion: either null or an object using these exact keys:
|
||||
nodeId, question, reason
|
||||
|
||||
## Proposal Rules
|
||||
1. Propose changes only. Never return a replacement graph.
|
||||
2. Preserve unrelated nodes and edges by omitting them from the proposal.
|
||||
3. Reference existing node IDs when updating an existing concept.
|
||||
4. Use addedNodes only for genuinely new concepts.
|
||||
5. Resolve the answered unknown first when the answer supports it.
|
||||
6. Then inspect the answer for newly introduced consequential uncertainty.
|
||||
7. Add new unknown nodes only when the answer introduces a new decision, claim, object, measure, dependency, or unresolved term directly relevant to the case.
|
||||
8. Add at most 3 new unknown nodes.
|
||||
9. Every new unknown must be directly traceable to the user's answer and its description must state why that uncertainty matters.
|
||||
9a. In the description of every new unknown, explicitly include a short why-it-matters clause using wording such as because, so that, needed to decide, or matters because.
|
||||
10. Do not add broad generic discovery questions.
|
||||
11. Do not add duplicate unknowns.
|
||||
12. Do not expand unrelated branches.
|
||||
13. Propagate only through explicit dependencies or relationships already present in the graph, except for the minimal new edges needed to connect validated new unknowns to the relevant answer-derived decision or context node.
|
||||
13a. For every new unknown node, include at least one added edge that connects it to an existing updated/resolved node or to a newly added non-unknown node introduced from the answer.
|
||||
14. Do not invent evidence.
|
||||
15. Do not create unsupported causal edges.
|
||||
16. If consequential unresolved unknowns exist, selectedQuestion may identify one valid candidate unknown, but the engine will deterministically choose final priority after validation.
|
||||
17. selectedQuestion.nodeId must reference an unresolved unknown node that exists either already in the graph or in addedNodes.
|
||||
18. selectedQuestion.question must be one narrow non-compound question about that one unknown.
|
||||
19. Do not prioritise downstream implementation, pricing, optimisation, or speculative branches ahead of prerequisite definitions, actors, success criteria, constraints, measures, or terminology.
|
||||
20. Return selectedQuestion as null only when no consequential unresolved unknown remains.
|
||||
21. Use empty arrays when there are no changes in a category.
|
||||
22. Never return null array entries.
|
||||
23. Never use unknown enum values.
|
||||
24. Do not change existing IDs.
|
||||
25. Do not replace the whole graph, and do not restate unchanged graph content inside the proposal.
|
||||
|
||||
## Additional Guidance
|
||||
- If the answer only clarifies an existing unknown, prefer updatedNodes and resolvedUnknownNodeIds over creating duplicate nodes.
|
||||
- When an answer resolves an existing unknown, include that existing node ID in resolvedUnknownNodeIds and update that node rather than creating only a parallel observation.
|
||||
- If the answer creates a more specific decision situation, add the smallest set of new nodes and edges needed to represent that situation and only its most consequential unknowns.
|
||||
- If you add a new unknown, do not leave it floating: connect it with an added edge to the relevant decision/context node created or updated from the answer.
|
||||
- If you add a new unknown, its description must do two jobs in one sentence: what is unknown, and why resolving it matters for the case.
|
||||
- Treat selectedQuestion as a candidate only; the engine will apply deterministic information-value scoring after validation.
|
||||
- If the answer does not justify a change, return empty arrays for every category.
|
||||
|
||||
## Example Constraint Reminder
|
||||
${formatExampleAnswerBlock()}
|
||||
|
||||
## Output Contract Reminder
|
||||
Return one JSON object only, with exact field names and exact enum values.
|
||||
Never include a full graph.
|
||||
Never include any field other than the contract fields above.
|
||||
`;
|
||||
}
|
||||
|
||||
export const buildUpdatePrompt = buildGraphUpdatePrompt;
|
||||
@@ -0,0 +1,367 @@
|
||||
function normaliseText(value) {
|
||||
return String(value || "")
|
||||
.toLowerCase()
|
||||
.replace(/[^a-z0-9]+/g, " ")
|
||||
.trim();
|
||||
}
|
||||
|
||||
function sentenceCase(value) {
|
||||
const trimmed = String(value || "").trim();
|
||||
if (!trimmed) return "this uncertainty";
|
||||
return trimmed.charAt(0).toLowerCase() + trimmed.slice(1);
|
||||
}
|
||||
|
||||
function buildNodeMap(graph) {
|
||||
return new Map((graph?.nodes || []).map((node) => [node.id, node]));
|
||||
}
|
||||
|
||||
function collectRelatedNodes(node, graph) {
|
||||
if (!node || !graph) return [];
|
||||
|
||||
const nodesById = buildNodeMap(graph);
|
||||
const relatedIds = new Set([
|
||||
...(node.dependsOn || []),
|
||||
...(node.affects || []),
|
||||
...(node.childIds || []),
|
||||
]);
|
||||
|
||||
if (node.parentId) {
|
||||
relatedIds.add(node.parentId);
|
||||
}
|
||||
|
||||
for (const edge of graph.edges || []) {
|
||||
if (edge.fromNodeId === node.id) {
|
||||
relatedIds.add(edge.toNodeId);
|
||||
}
|
||||
if (edge.toNodeId === node.id) {
|
||||
relatedIds.add(edge.fromNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
return [...relatedIds].map((nodeId) => nodesById.get(nodeId)).filter(Boolean);
|
||||
}
|
||||
|
||||
function collectResolvedContextValues(graph) {
|
||||
const resolvedSet = new Set(graph?.resolvedNodeIds || []);
|
||||
|
||||
return (graph?.nodes || [])
|
||||
.filter((node) => resolvedSet.has(node.id))
|
||||
.map((node) => node.value)
|
||||
.filter((value) => typeof value === "string" && value.trim().length > 0);
|
||||
}
|
||||
|
||||
function extractMeaning(node) {
|
||||
const raw = `${node?.label || ""} ${node?.description || ""}`.trim();
|
||||
let meaning = String(
|
||||
node?.label || node?.description || "this uncertainty",
|
||||
).trim();
|
||||
|
||||
const lowered = normaliseText(raw);
|
||||
if (
|
||||
/\b(customer|user|buyer|stakeholder|recipient|audience)\b/.test(lowered)
|
||||
) {
|
||||
return "the relevant customer, user, or value recipient";
|
||||
}
|
||||
|
||||
meaning = meaning
|
||||
.replace(/^uncertainty regarding\s+/i, "")
|
||||
.replace(/^uncertainty about\s+/i, "")
|
||||
.replace(/^lack of\s+/i, "")
|
||||
.replace(/^unknown\s+/i, "")
|
||||
.replace(/^whether\s+/i, "")
|
||||
.replace(/^the\s+/, "")
|
||||
.trim();
|
||||
|
||||
if (!meaning) {
|
||||
return "this uncertainty";
|
||||
}
|
||||
|
||||
return sentenceCase(meaning);
|
||||
}
|
||||
|
||||
function extractActionPhrase(texts) {
|
||||
for (const text of texts) {
|
||||
const value = String(text || "").trim();
|
||||
if (!value) continue;
|
||||
|
||||
const matches = [
|
||||
value.match(/\b(?:whether|deciding|decision) to\s+([^.,;:]+)/i),
|
||||
value.match(/\b(?:justify|continuing|proceeding with)\s+([^.,;:]+)/i),
|
||||
value.match(
|
||||
/\b(build|launch|adopt|buy|continue|proceed|invest in|fund)\s+([^.,;:]+)/i,
|
||||
),
|
||||
].filter(Boolean);
|
||||
|
||||
const match = matches[0];
|
||||
if (!match) continue;
|
||||
|
||||
const phrase = (match[1] || `${match[1] || ""} ${match[2] || ""}`)
|
||||
.replace(/^to\s+/i, "")
|
||||
.trim();
|
||||
|
||||
if (phrase) {
|
||||
return phrase;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function toGerundPhrase(phrase) {
|
||||
const trimmed = String(phrase || "").trim();
|
||||
if (!trimmed) return "proceeding with this decision";
|
||||
|
||||
const [firstWord, ...rest] = trimmed.split(/\s+/);
|
||||
const lower = firstWord.toLowerCase();
|
||||
const irregular = {
|
||||
be: "being",
|
||||
build: "building",
|
||||
continue: "continuing",
|
||||
decide: "deciding",
|
||||
proceed: "proceeding",
|
||||
launch: "launching",
|
||||
invest: "investing",
|
||||
fund: "funding",
|
||||
buy: "buying",
|
||||
pay: "paying",
|
||||
adopt: "adopting",
|
||||
};
|
||||
|
||||
let gerund = irregular[lower];
|
||||
if (!gerund) {
|
||||
if (lower.endsWith("e") && !lower.endsWith("ee")) {
|
||||
gerund = `${lower.slice(0, -1)}ing`;
|
||||
} else {
|
||||
gerund = `${lower}ing`;
|
||||
}
|
||||
}
|
||||
|
||||
return [gerund, ...rest].join(" ");
|
||||
}
|
||||
|
||||
function detectStrategy({ node, graph, relatedNodes, combinedText, meaning }) {
|
||||
const text = normaliseText(combinedText);
|
||||
const nodeText = normaliseText(
|
||||
`${node?.label || ""} ${node?.description || ""}`,
|
||||
);
|
||||
const relatedText = normaliseText(
|
||||
relatedNodes
|
||||
.map((relatedNode) => `${relatedNode.label} ${relatedNode.description}`)
|
||||
.join(" "),
|
||||
);
|
||||
const resolvedValues = collectResolvedContextValues(graph);
|
||||
const actionPhrase = extractActionPhrase([
|
||||
...resolvedValues,
|
||||
...relatedNodes.map((relatedNode) => relatedNode.value),
|
||||
...relatedNodes.map((relatedNode) => relatedNode.label),
|
||||
...relatedNodes.map((relatedNode) => relatedNode.description),
|
||||
graph?.centralStatement,
|
||||
]);
|
||||
|
||||
const decisionContext =
|
||||
/\b(decision|whether to|build|launch|continue|proceed|invest|allocate)\b/.test(
|
||||
`${text} ${relatedText} ${resolvedValues.join(" ")}`,
|
||||
) || Boolean(actionPhrase);
|
||||
|
||||
const hasConstraintLanguage =
|
||||
/\b(constraint|limit|budget|deadline|requirement|regulation|capacity)\b/.test(
|
||||
text,
|
||||
);
|
||||
const hasPrimaryConstraintLanguage =
|
||||
/\b(constraint|limit|budget|deadline|requirement|regulation|capacity)\b/.test(
|
||||
nodeText,
|
||||
);
|
||||
|
||||
if (/\b(customer|user|buyer|stakeholder|recipient|audience)\b/.test(text)) {
|
||||
return { strategy: "actor/customer", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
const hasBaselineLanguage =
|
||||
/\b(before|previous|baseline|prior|comparable state)\b/.test(text);
|
||||
const hasPrimaryBaselineLanguage =
|
||||
/\b(before|previous|baseline|prior|comparable state)\b/.test(nodeText);
|
||||
|
||||
if (hasBaselineLanguage && hasPrimaryBaselineLanguage) {
|
||||
return { strategy: "baseline", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (/\b(when|timing|timeline|duration|sequence|milestone)\b/.test(text)) {
|
||||
return { strategy: "transition/timing", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
const hasDefinitionLanguage =
|
||||
/\b(define|definition|meaning|term|terminology)\b/.test(text);
|
||||
const hasPrimaryDefinitionLanguage =
|
||||
/\b(define|definition|meaning|term|terminology)\b/.test(nodeText);
|
||||
const hasCriteriaLanguage =
|
||||
/\b(success criteria|success threshold|threshold|decision criteria|criterion|justify|sufficient)\b/.test(
|
||||
nodeText,
|
||||
);
|
||||
const hasDecisionValueLanguage =
|
||||
decisionContext &&
|
||||
/\b(value|commercial value|commercial viability|viability|justify|sufficient|success|threshold|criterion)\b/.test(
|
||||
text,
|
||||
);
|
||||
const hasMeasurementLanguage =
|
||||
/\b(metric|measure|measurable|roi|revenue projection|benchmark)\b/.test(
|
||||
text,
|
||||
);
|
||||
|
||||
if (hasDecisionValueLanguage && hasMeasurementLanguage) {
|
||||
return { strategy: "measurement", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasPrimaryDefinitionLanguage) {
|
||||
return { strategy: "definition", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasDecisionValueLanguage || hasCriteriaLanguage) {
|
||||
return { strategy: "decision criterion", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasConstraintLanguage && hasPrimaryConstraintLanguage) {
|
||||
return { strategy: "constraint", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasDefinitionLanguage) {
|
||||
return { strategy: "definition", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasBaselineLanguage) {
|
||||
return { strategy: "baseline", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasConstraintLanguage) {
|
||||
return { strategy: "constraint", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (/\b(evidence|proof|validate|validation|signal|demand)\b/.test(text)) {
|
||||
return { strategy: "evidence", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (hasMeasurementLanguage) {
|
||||
return { strategy: "measurement", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (
|
||||
/\b(objective|goal|outcome|problem|job to be done|benefit)\b/.test(text)
|
||||
) {
|
||||
return { strategy: "objective", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
if (
|
||||
node?.kind === "reported_claim" ||
|
||||
node?.kind === "conclusion" ||
|
||||
/\b(claim|assertion|true|false)\b/.test(text)
|
||||
) {
|
||||
return { strategy: "evidence", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
return { strategy: "generic clarification", meaning, actionPhrase };
|
||||
}
|
||||
|
||||
function buildQuestion({ strategy, meaning, actionPhrase }) {
|
||||
switch (strategy) {
|
||||
case "decision criterion":
|
||||
return actionPhrase
|
||||
? `What outcome would demonstrate enough value to justify ${toGerundPhrase(actionPhrase)}?`
|
||||
: "What outcome would be sufficient to justify this decision?";
|
||||
case "definition":
|
||||
return `What does ${meaning} mean in this situation?`;
|
||||
case "evidence":
|
||||
return `What evidence would show whether ${meaning} is true?`;
|
||||
case "baseline":
|
||||
return `What was the comparable state before ${meaning}?`;
|
||||
case "actor/customer":
|
||||
return "Who experiences the problem or receives the value in this situation?";
|
||||
case "objective":
|
||||
return "What outcome is this decision or effort meant to achieve?";
|
||||
case "constraint":
|
||||
return "What constraint most limits the available options in this situation?";
|
||||
case "measurement":
|
||||
return `What measure would determine whether ${meaning} is sufficient?`;
|
||||
case "transition/timing":
|
||||
return `When does ${meaning} become relevant in the decision or change?`;
|
||||
default:
|
||||
return `What specific fact would resolve whether ${meaning} is true?`;
|
||||
}
|
||||
}
|
||||
|
||||
function isCompoundQuestion(question) {
|
||||
const trimmed = String(question || "").trim();
|
||||
const questionMarks = (trimmed.match(/\?/g) || []).length;
|
||||
|
||||
if (questionMarks !== 1) return true;
|
||||
if (/\?\s*(and|or)\b/i.test(trimmed)) return true;
|
||||
if (/\b(and|or)\b[^?]{0,80}\?/i.test(trimmed) && /,/.test(trimmed)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
function validateFormulatedQuestion(question, meaning) {
|
||||
const trimmed = String(question || "").trim();
|
||||
const lower = trimmed.toLowerCase();
|
||||
const meaningWords = normaliseText(meaning)
|
||||
.split(" ")
|
||||
.filter((word) => word.length > 3);
|
||||
const overlappingWord = meaningWords.find((word) => lower.includes(word));
|
||||
|
||||
if (!trimmed) return false;
|
||||
if ((trimmed.match(/\?/g) || []).length !== 1) return false;
|
||||
if (isCompoundQuestion(trimmed)) return false;
|
||||
if (/^what is\s+/i.test(trimmed)) return false;
|
||||
if (/^how should uncertainty regarding\b/i.test(trimmed)) return false;
|
||||
if (/^what would resolve uncertainty regarding\b/i.test(trimmed))
|
||||
return false;
|
||||
if (
|
||||
/\bprice|pricing|price point\b/i.test(trimmed) &&
|
||||
!/\bprice\b/i.test(meaning)
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
if (
|
||||
!overlappingWord &&
|
||||
!/\b(decision|evidence|constraint|customer|value|outcome)\b/i.test(trimmed)
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
export function formulateQuestion({ node, graph, context = {} }) {
|
||||
const relatedNodes = collectRelatedNodes(node, graph);
|
||||
const meaning = extractMeaning(node);
|
||||
const combinedText = [
|
||||
node?.label,
|
||||
node?.description,
|
||||
...relatedNodes.map((relatedNode) => relatedNode.label),
|
||||
...relatedNodes.map((relatedNode) => relatedNode.description),
|
||||
graph?.centralStatement,
|
||||
...(context.resolvedValues || []),
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join(" ");
|
||||
|
||||
const detected = detectStrategy({
|
||||
node,
|
||||
graph,
|
||||
relatedNodes,
|
||||
combinedText,
|
||||
meaning,
|
||||
});
|
||||
|
||||
let question = buildQuestion(detected);
|
||||
|
||||
if (!validateFormulatedQuestion(question, meaning)) {
|
||||
question = `What evidence would resolve whether ${meaning} is true?`;
|
||||
}
|
||||
|
||||
return {
|
||||
question,
|
||||
reason: `Formulated from graph context using the ${detected.strategy} strategy.`,
|
||||
strategy: detected.strategy,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,205 @@
|
||||
/**
|
||||
* Situation Graph schema — v0.4 experiment.
|
||||
* Defines types for an evolving multi-turn situation reconstruction graph.
|
||||
* Plain TypeScript interfaces implemented as Zod schemas for runtime validation.
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
|
||||
// ── Enums ────────────────────────────────────────────
|
||||
|
||||
export const SituationKind = /** @type {const} */ ({
|
||||
observation: "observation",
|
||||
reported_claim: "reported_claim",
|
||||
metric: "metric",
|
||||
state: "state",
|
||||
transition: "transition",
|
||||
relationship: "relationship",
|
||||
assumption: "assumption",
|
||||
unknown: "unknown",
|
||||
conclusion: "conclusion",
|
||||
});
|
||||
|
||||
export const SituationStatus = /** @type {const} */ ({
|
||||
known: "known",
|
||||
unknown: "unknown",
|
||||
provisional: "provisional",
|
||||
supported: "supported",
|
||||
weakened: "weakened",
|
||||
contradicted: "contradicted",
|
||||
resolved: "resolved",
|
||||
});
|
||||
|
||||
export const ConfidenceLevel = /** @type {const} */ ({
|
||||
low: "low",
|
||||
medium: "medium",
|
||||
high: "high",
|
||||
});
|
||||
|
||||
// ── SituationNode ────────────────────────────────────
|
||||
|
||||
export const situationNodeSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
label: z.string().min(1),
|
||||
description: z.string().min(1),
|
||||
kind: z.enum(Object.values(SituationKind)),
|
||||
status: z.enum(Object.values(SituationStatus)),
|
||||
confidence: z.enum(Object.values(ConfidenceLevel)),
|
||||
value: z.union([z.string(), z.number(), z.null()]).nullable().optional(),
|
||||
unit: z.string().nullable().optional(),
|
||||
evidenceIds: z.array(z.string()).default([]),
|
||||
dependsOn: z.array(z.string()).default([]),
|
||||
affects: z.array(z.string()).default([]),
|
||||
parentId: z.string().nullable().optional(),
|
||||
childIds: z.array(z.string()).default([]),
|
||||
});
|
||||
|
||||
/** @typedef {z.infer<typeof situationNodeSchema>} SituationNode */
|
||||
|
||||
// ── SituationEdge ────────────────────────────────────
|
||||
|
||||
export const SituationRelationship = /** @type {const} */ ({
|
||||
supports: "supports",
|
||||
weakens: "weakens",
|
||||
contradicts: "contradicts",
|
||||
depends_on: "depends_on",
|
||||
causes: "causes",
|
||||
may_cause: "may_cause",
|
||||
measures: "measures",
|
||||
compares_with: "compares_with",
|
||||
updates: "updates",
|
||||
other: "other",
|
||||
});
|
||||
|
||||
export const situationEdgeSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
fromNodeId: z.string().min(1),
|
||||
toNodeId: z.string().min(1),
|
||||
relationship: z.enum(Object.values(SituationRelationship)),
|
||||
confidence: z.enum(Object.values(ConfidenceLevel)),
|
||||
description: z.string().min(1),
|
||||
});
|
||||
|
||||
/** @typedef {z.infer<typeof situationEdgeSchema>} SituationEdge */
|
||||
|
||||
// ── SituationGraph ───────────────────────────────────
|
||||
|
||||
export const situationGraphSchema = z.object({
|
||||
centralStatement: z.string().min(1),
|
||||
nodes: z.array(situationNodeSchema).min(1),
|
||||
edges: z.array(situationEdgeSchema).default([]),
|
||||
activeUnknownNodeId: z.string().nullable(),
|
||||
resolvedNodeIds: z.array(z.string()).default([]),
|
||||
currentSummary: z.string().min(1),
|
||||
});
|
||||
|
||||
/** @typedef {z.infer<typeof situationGraphSchema>} SituationGraph */
|
||||
|
||||
// ── GraphUpdate (change set) ────────────────────────
|
||||
|
||||
const graphUpdateNodeChangeSchema = z.object({
|
||||
nodeId: z.string().min(1),
|
||||
previousStatus: z.enum(Object.values(SituationStatus)).nullable().optional(),
|
||||
newStatus: z.enum(Object.values(SituationStatus)).nullable().optional(),
|
||||
previousValue: z
|
||||
.union([z.string(), z.number(), z.null()])
|
||||
.nullable()
|
||||
.optional(),
|
||||
newValue: z.union([z.string(), z.number(), z.null()]).nullable().optional(),
|
||||
reason: z.string().min(1),
|
||||
});
|
||||
|
||||
export const selectedQuestionSchema = z
|
||||
.object({
|
||||
nodeId: z.string().min(1),
|
||||
question: z.string().min(1),
|
||||
reason: z.string().min(1),
|
||||
})
|
||||
.strict();
|
||||
|
||||
export const graphUpdateSchema = z.object({
|
||||
addedNodes: z.array(situationNodeSchema).default([]),
|
||||
updatedNodes: z.array(graphUpdateNodeChangeSchema).default([]),
|
||||
addedEdges: z.array(situationEdgeSchema).default([]),
|
||||
removedEdgeIds: z.array(z.string()).default([]),
|
||||
resolvedUnknownNodeIds: z.array(z.string()).default([]),
|
||||
affectedNodeIds: z.array(z.string()).default([]),
|
||||
selectedQuestion: selectedQuestionSchema.nullable().default(null),
|
||||
});
|
||||
|
||||
/** @typedef {z.infer<typeof graphUpdateSchema>} GraphUpdate */
|
||||
|
||||
// ── API request / response schemas ───────────────────
|
||||
|
||||
export const startCaseRequestSchema = z.object({
|
||||
scenario: z.string().min(1).max(10000),
|
||||
promptVersion: z.string().optional(),
|
||||
});
|
||||
|
||||
export const updateCaseRequestSchema = z.object({
|
||||
situationGraph: situationGraphSchema,
|
||||
previousQuestion: z.string().min(1),
|
||||
answer: z.string().min(1).max(5000),
|
||||
promptVersion: z.string().optional(),
|
||||
});
|
||||
|
||||
// ── Helpers ──────────────────────────────────────────
|
||||
|
||||
/** Generate a short deterministic ID from a label */
|
||||
export function makeNodeId(label) {
|
||||
return "n" + Math.abs(hashString(label)).toString(36).slice(0, 7);
|
||||
}
|
||||
|
||||
function hashString(str) {
|
||||
let h = 0;
|
||||
for (let i = 0; i < str.length; i++) {
|
||||
h = (Math.imul(31, h) + str.charCodeAt(i)) | 0;
|
||||
}
|
||||
return h;
|
||||
}
|
||||
|
||||
/** Create a minimal valid node — used in tests and fixtures */
|
||||
export function makeNode(opts) {
|
||||
const id = opts.id || makeNodeId(opts.label);
|
||||
return situationNodeSchema.parse({
|
||||
id,
|
||||
label: opts.label,
|
||||
description: opts.description ?? opts.label,
|
||||
kind: opts.kind ?? "observation",
|
||||
status: opts.status ?? "unknown",
|
||||
confidence: opts.confidence ?? "medium",
|
||||
value: opts.value ?? null,
|
||||
unit: opts.unit ?? null,
|
||||
evidenceIds: opts.evidenceIds ?? [],
|
||||
dependsOn: opts.dependsOn ?? [],
|
||||
affects: opts.affects ?? [],
|
||||
parentId: opts.parentId ?? null,
|
||||
childIds: opts.childIds ?? [],
|
||||
});
|
||||
}
|
||||
|
||||
/** Create a minimal valid edge — used in tests and fixtures */
|
||||
export function makeEdge(opts) {
|
||||
return situationEdgeSchema.parse({
|
||||
id:
|
||||
opts.id ||
|
||||
"e" + opts.fromNodeId.slice(0, 3) + "-" + opts.toNodeId.slice(0, 3),
|
||||
fromNodeId: opts.fromNodeId,
|
||||
toNodeId: opts.toNodeId,
|
||||
relationship: opts.relationship ?? "supports",
|
||||
confidence: opts.confidence ?? "medium",
|
||||
description: opts.description ?? opts.fromNodeId + " -> " + opts.toNodeId,
|
||||
});
|
||||
}
|
||||
|
||||
/** Build a minimal valid graph structure */
|
||||
export function makeGraph(opts) {
|
||||
return situationGraphSchema.parse({
|
||||
centralStatement: opts.centralStatement || "",
|
||||
nodes: opts.nodes ?? [],
|
||||
edges: opts.edges ?? [],
|
||||
activeUnknownNodeId: opts.activeUnknownNodeId ?? null,
|
||||
resolvedNodeIds: opts.resolvedNodeIds ?? [],
|
||||
currentSummary: opts.currentSummary || "",
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,157 @@
|
||||
import { graphUpdateSchema } from "./schema.js";
|
||||
|
||||
const TOP_LEVEL_ARRAY_FIELDS = [
|
||||
"addedNodes",
|
||||
"updatedNodes",
|
||||
"addedEdges",
|
||||
"removedEdgeIds",
|
||||
"resolvedUnknownNodeIds",
|
||||
"affectedNodeIds",
|
||||
];
|
||||
|
||||
const TOP_LEVEL_NULLABLE_FIELDS = ["selectedQuestion"];
|
||||
|
||||
function cloneJsonSafe(value) {
|
||||
if (value == null) return value;
|
||||
return JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
function removeNullArrayEntries(value, path = [], normalisationsApplied = []) {
|
||||
if (Array.isArray(value)) {
|
||||
const filtered = [];
|
||||
value.forEach((item, index) => {
|
||||
if (item === null) {
|
||||
normalisationsApplied.push({
|
||||
path: [...path, index],
|
||||
change: "Removed null array entry",
|
||||
});
|
||||
return;
|
||||
}
|
||||
filtered.push(
|
||||
removeNullArrayEntries(item, [...path, index], normalisationsApplied),
|
||||
);
|
||||
});
|
||||
return filtered;
|
||||
}
|
||||
|
||||
if (value && typeof value === "object") {
|
||||
return Object.fromEntries(
|
||||
Object.entries(value).map(([key, child]) => [
|
||||
key,
|
||||
removeNullArrayEntries(child, [...path, key], normalisationsApplied),
|
||||
]),
|
||||
);
|
||||
}
|
||||
|
||||
return value;
|
||||
}
|
||||
|
||||
function applyKnownEnumAliases(proposal, normalisationsApplied) {
|
||||
if (!proposal || typeof proposal !== "object") return proposal;
|
||||
|
||||
if (Array.isArray(proposal.addedNodes)) {
|
||||
proposal.addedNodes = proposal.addedNodes.map((node, index) => {
|
||||
if (node?.kind === "reported_statement") {
|
||||
normalisationsApplied.push({
|
||||
path: ["addedNodes", index, "kind"],
|
||||
change: "Converted reported_statement to reported_claim",
|
||||
});
|
||||
return { ...node, kind: "reported_claim" };
|
||||
}
|
||||
return node;
|
||||
});
|
||||
}
|
||||
|
||||
return proposal;
|
||||
}
|
||||
|
||||
function fillMissingOptionalArrays(proposal, normalisationsApplied) {
|
||||
if (!proposal || typeof proposal !== "object") return proposal;
|
||||
|
||||
for (const field of TOP_LEVEL_ARRAY_FIELDS) {
|
||||
if (!(field in proposal)) {
|
||||
proposal[field] = [];
|
||||
normalisationsApplied.push({
|
||||
path: [field],
|
||||
change: "Filled missing optional array with []",
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return proposal;
|
||||
}
|
||||
|
||||
function fillMissingNullableFields(proposal, normalisationsApplied) {
|
||||
if (!proposal || typeof proposal !== "object") return proposal;
|
||||
|
||||
for (const field of TOP_LEVEL_NULLABLE_FIELDS) {
|
||||
if (!(field in proposal)) {
|
||||
proposal[field] = null;
|
||||
normalisationsApplied.push({
|
||||
path: [field],
|
||||
change: "Filled missing optional nullable field with null",
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return proposal;
|
||||
}
|
||||
|
||||
export function parseGraphUpdateProposal(rawResponse) {
|
||||
const raw = rawResponse;
|
||||
let parsed;
|
||||
|
||||
if (typeof rawResponse === "string") {
|
||||
try {
|
||||
parsed = JSON.parse(rawResponse);
|
||||
} catch (error) {
|
||||
return {
|
||||
success: false,
|
||||
proposal: null,
|
||||
raw,
|
||||
normalisationsApplied: [],
|
||||
errors: [error.message || "Model response is not valid JSON"],
|
||||
};
|
||||
}
|
||||
} else if (rawResponse && typeof rawResponse === "object") {
|
||||
parsed = cloneJsonSafe(rawResponse);
|
||||
} else {
|
||||
return {
|
||||
success: false,
|
||||
proposal: null,
|
||||
raw,
|
||||
normalisationsApplied: [],
|
||||
errors: ["Graph update proposal must be a JSON object or JSON string"],
|
||||
};
|
||||
}
|
||||
|
||||
const normalisationsApplied = [];
|
||||
let normalised = removeNullArrayEntries(parsed, [], normalisationsApplied);
|
||||
normalised = applyKnownEnumAliases(normalised, normalisationsApplied);
|
||||
normalised = fillMissingOptionalArrays(normalised, normalisationsApplied);
|
||||
normalised = fillMissingNullableFields(normalised, normalisationsApplied);
|
||||
|
||||
const parsedProposal = graphUpdateSchema.safeParse(normalised);
|
||||
|
||||
if (!parsedProposal.success) {
|
||||
return {
|
||||
success: false,
|
||||
proposal: null,
|
||||
raw,
|
||||
normalisationsApplied,
|
||||
errors: parsedProposal.error.issues.map((issue) => ({
|
||||
path: issue.path,
|
||||
message: issue.message,
|
||||
code: issue.code,
|
||||
})),
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
proposal: parsedProposal.data,
|
||||
raw,
|
||||
normalisationsApplied,
|
||||
errors: [],
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,553 @@
|
||||
/**
|
||||
* Deterministic graph utilities for situation graph operations.
|
||||
* These functions perform safe, validated operations on the graph.
|
||||
* The LLM should never directly modify the graph — it proposes changes,
|
||||
* and these utilities apply them safely.
|
||||
*/
|
||||
|
||||
import {
|
||||
situationNodeSchema,
|
||||
situationEdgeSchema,
|
||||
situationGraphSchema,
|
||||
} from "./schema.js";
|
||||
|
||||
function normaliseText(value) {
|
||||
return String(value || "")
|
||||
.toLowerCase()
|
||||
.replace(/[^a-z0-9]+/g, " ")
|
||||
.trim();
|
||||
}
|
||||
|
||||
function collectNodeText(node) {
|
||||
return `${node?.label || ""} ${node?.description || ""}`.trim();
|
||||
}
|
||||
|
||||
function countIncomingUnknownDependencies(graph, nodeId, resolvedNodeIds) {
|
||||
const resolvedSet = new Set(resolvedNodeIds || []);
|
||||
const nodesById = new Map(graph.nodes.map((node) => [node.id, node]));
|
||||
const incoming = new Set();
|
||||
|
||||
for (const dependencyId of nodesById.get(nodeId)?.dependsOn || []) {
|
||||
const dependencyNode = nodesById.get(dependencyId);
|
||||
if (dependencyNode?.kind === "unknown" && !resolvedSet.has(dependencyId)) {
|
||||
incoming.add(dependencyId);
|
||||
}
|
||||
}
|
||||
|
||||
for (const edge of graph.edges) {
|
||||
if (edge.toNodeId !== nodeId) continue;
|
||||
const dependencyNode = nodesById.get(edge.fromNodeId);
|
||||
if (
|
||||
dependencyNode?.kind === "unknown" &&
|
||||
!resolvedSet.has(edge.fromNodeId)
|
||||
) {
|
||||
incoming.add(edge.fromNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
return incoming.size;
|
||||
}
|
||||
|
||||
function classifyUnknownPriority(text) {
|
||||
const normalised = normaliseText(text);
|
||||
|
||||
const matches = {
|
||||
objective:
|
||||
/\b(objective|goal|outcome|value|problem|job to be done|benefit|commercial value)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
actor:
|
||||
/\b(customer|user|buyer|actor|stakeholder|audience|recipient)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
criteria:
|
||||
/\b(success criteria|success threshold|threshold|decision criteria|criterion|justify|sufficient)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
measure:
|
||||
/\b(metric|measure|measurable|roi|demand|evidence|signal|proof)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
terminology: /\b(define|definition|meaning|means|term|terminology)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
constraint:
|
||||
/\b(constraint|limit|budget|deadline|requirement|regulation)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
pricing: /\b(price|pricing|price point|subscription|charge|pay for)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
implementation:
|
||||
/\b(implementation|build approach|architecture|stack|feature|technical design)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
optimisation:
|
||||
/\b(optimisation|optimi[sz]ation|improve|efficiency|performance|scale)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
speculative:
|
||||
/\b(maybe|possible|optional|future branch|nice to have|slogan|colour|color|ui)\b/.test(
|
||||
normalised,
|
||||
),
|
||||
};
|
||||
|
||||
return matches;
|
||||
}
|
||||
|
||||
export function scoreUnknownCandidate(graph, node, resolvedNodeIds = []) {
|
||||
const text = collectNodeText(node);
|
||||
const matches = classifyUnknownPriority(text);
|
||||
const downstreamCount = findDependentNodes(graph, node.id).length;
|
||||
const unresolvedParentUnknownCount = countIncomingUnknownDependencies(
|
||||
graph,
|
||||
node.id,
|
||||
resolvedNodeIds,
|
||||
);
|
||||
|
||||
let score = downstreamCount * 4;
|
||||
|
||||
if (matches.objective) score += 12;
|
||||
if (matches.actor) score += 10;
|
||||
if (matches.criteria) score += 11;
|
||||
if (matches.measure) score += 8;
|
||||
if (matches.terminology) score += 7;
|
||||
if (matches.constraint) score += 9;
|
||||
|
||||
if (matches.pricing) score -= 8;
|
||||
if (matches.implementation) score -= 10;
|
||||
if (matches.optimisation) score -= 9;
|
||||
if (matches.speculative) score -= 12;
|
||||
|
||||
if (
|
||||
matches.pricing &&
|
||||
!matches.objective &&
|
||||
!matches.criteria &&
|
||||
!matches.actor
|
||||
) {
|
||||
score -= 6;
|
||||
}
|
||||
|
||||
score -= unresolvedParentUnknownCount * 7;
|
||||
|
||||
return {
|
||||
nodeId: node.id,
|
||||
label: node.label,
|
||||
score,
|
||||
downstreamCount,
|
||||
unresolvedParentUnknownCount,
|
||||
matches,
|
||||
};
|
||||
}
|
||||
|
||||
export function buildDeterministicQuestionForUnknown(node) {
|
||||
const text = normaliseText(collectNodeText(node));
|
||||
|
||||
if (
|
||||
/\b(success criteria|success threshold|threshold|decision criteria|criterion)\b/.test(
|
||||
text,
|
||||
)
|
||||
) {
|
||||
return `What outcome would define success for ${node.label}?`;
|
||||
}
|
||||
if (
|
||||
/\b(customer|user|buyer|actor|stakeholder|audience|recipient)\b/.test(text)
|
||||
) {
|
||||
return `Who is the key actor or customer for ${node.label}?`;
|
||||
}
|
||||
if (
|
||||
/\b(define|definition|meaning|means|term|terminology|value)\b/.test(text)
|
||||
) {
|
||||
return `How should ${node.label} be defined for this decision?`;
|
||||
}
|
||||
if (
|
||||
/\b(metric|measure|measurable|roi|demand|evidence|signal|proof)\b/.test(
|
||||
text,
|
||||
)
|
||||
) {
|
||||
return `What evidence or measure would resolve ${node.label}?`;
|
||||
}
|
||||
|
||||
return `What would resolve ${node.label}?`;
|
||||
}
|
||||
|
||||
// ── Validate that all edge references point to existing nodes ──
|
||||
|
||||
export function validateGraphReferences(graph) {
|
||||
const errors = [];
|
||||
const nodeIds = new Set(graph.nodes.map((n) => n.id));
|
||||
|
||||
for (const node of graph.nodes) {
|
||||
if (node.parentId !== null && !nodeIds.has(node.parentId)) {
|
||||
errors.push(
|
||||
`Node "${node.id}" references parentId "${node.parentId}" which does not exist`,
|
||||
);
|
||||
}
|
||||
for (const cid of node.childIds) {
|
||||
if (!nodeIds.has(cid)) {
|
||||
errors.push(
|
||||
`Node "${node.id}" references childIds "${cid}" which does not exist`,
|
||||
);
|
||||
}
|
||||
}
|
||||
for (const dep of node.dependsOn) {
|
||||
if (!nodeIds.has(dep)) {
|
||||
errors.push(
|
||||
`Node "${node.id}" depends on "${dep}" which does not exist`,
|
||||
);
|
||||
}
|
||||
}
|
||||
for (const aff of node.affects) {
|
||||
if (!nodeIds.has(aff)) {
|
||||
errors.push(`Node "${node.id}" affects "${aff}" which does not exist`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const edge of graph.edges) {
|
||||
if (!nodeIds.has(edge.fromNodeId)) {
|
||||
errors.push(
|
||||
`Edge "${edge.id}" references non-existent fromNodeId "${edge.fromNodeId}"`,
|
||||
);
|
||||
}
|
||||
if (!nodeIds.has(edge.toNodeId)) {
|
||||
errors.push(
|
||||
`Edge "${edge.id}" references non-existent toNodeId "${edge.toNodeId}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return { valid: errors.length === 0, errors };
|
||||
}
|
||||
|
||||
// ── Detect duplicate node IDs ──
|
||||
|
||||
export function detectDuplicateNodeIds(nodes) {
|
||||
const countMap = new Map();
|
||||
const seen = new Set();
|
||||
|
||||
for (const node of nodes) {
|
||||
if (countMap.has(node.id)) {
|
||||
countMap.set(node.id, countMap.get(node.id) + 1);
|
||||
} else {
|
||||
countMap.set(node.id, 1);
|
||||
}
|
||||
}
|
||||
|
||||
const duplicates = [];
|
||||
for (const [id, count] of countMap.entries()) {
|
||||
if (count > 1 && !seen.has(id)) {
|
||||
duplicates.push({ nodeId: id, count });
|
||||
seen.add(id);
|
||||
}
|
||||
}
|
||||
|
||||
return duplicates;
|
||||
}
|
||||
|
||||
// ── Detect duplicate edges ──
|
||||
|
||||
export function detectDuplicateEdges(edges) {
|
||||
const seen = new Set();
|
||||
const duplicates = [];
|
||||
|
||||
for (const edge of edges) {
|
||||
const key = `${edge.fromNodeId}->${edge.toNodeId}:${edge.relationship}`;
|
||||
if (seen.has(key)) {
|
||||
duplicates.push({
|
||||
edgeId: edge.id,
|
||||
fromNodeId: edge.fromNodeId,
|
||||
toNodeId: edge.toNodeId,
|
||||
relationship: edge.relationship,
|
||||
});
|
||||
}
|
||||
seen.add(key);
|
||||
}
|
||||
|
||||
return duplicates;
|
||||
}
|
||||
|
||||
// ── Find all nodes that depend on a given node (transitive) ──
|
||||
|
||||
export function findDependentNodes(graph, nodeId) {
|
||||
const direct = graph.nodes
|
||||
.filter((n) => n.dependsOn.includes(nodeId))
|
||||
.map((n) => n.id);
|
||||
const affected = new Set(direct);
|
||||
|
||||
// Also propagate through edges where the relationship is depends_on
|
||||
for (const edge of graph.edges) {
|
||||
if (edge.toNodeId === nodeId && !affected.has(edge.fromNodeId)) {
|
||||
direct.push(edge.fromNodeId);
|
||||
affected.add(edge.fromNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
// Transitive propagation — BFS
|
||||
const queue = [...direct];
|
||||
while (queue.length > 0) {
|
||||
const current = queue.shift();
|
||||
if (!current || !affected.has(current)) continue;
|
||||
|
||||
for (const node of graph.nodes) {
|
||||
if (node.dependsOn.includes(current) && !affected.has(node.id)) {
|
||||
affected.add(node.id);
|
||||
queue.push(node.id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return [...affected];
|
||||
}
|
||||
|
||||
// ── Find all nodes that are directly or indirectly affected by a change in nodeId ──
|
||||
|
||||
export function findAffectedNodes(graph, nodeId) {
|
||||
// Direct effects: two sources
|
||||
// 1. Nodes that depend on this node (they list it in their dependsOn)
|
||||
const directFromDepends = graph.nodes
|
||||
.filter((n) => n.id !== nodeId && n.dependsOn.includes(nodeId))
|
||||
.map((n) => n.id);
|
||||
|
||||
// 2. Targets of the node's affects relationships (this node directly affects them)
|
||||
const myAffectedTargets = new Set(
|
||||
graph.nodes.find((n) => n.id === nodeId)?.affects || [],
|
||||
);
|
||||
|
||||
// Merge: also add edge targets where this node is the source
|
||||
for (const edge of graph.edges) {
|
||||
if (edge.fromNodeId === nodeId && !myAffectedTargets.has(edge.toNodeId)) {
|
||||
myAffectedTargets.add(edge.toNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
// Combine both sources
|
||||
const direct = [...new Set([...directFromDepends, ...myAffectedTargets])];
|
||||
|
||||
// Transitive propagation — BFS through dependsOn and affects of affected nodes
|
||||
const affected = new Set(direct);
|
||||
const queue = [...direct];
|
||||
while (queue.length > 0) {
|
||||
const current = queue.shift();
|
||||
if (!current || !affected.has(current)) continue;
|
||||
|
||||
for (const node of graph.nodes) {
|
||||
if (
|
||||
node.id !== nodeId &&
|
||||
!affected.has(node.id) &&
|
||||
(node.dependsOn.includes(current) || node.affects.includes(current))
|
||||
) {
|
||||
affected.add(node.id);
|
||||
queue.push(node.id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return [...affected];
|
||||
}
|
||||
|
||||
// ── Resolve an unknown node ──
|
||||
|
||||
export function resolveUnknownNode(graph, nodeId, newStatus, newValue, reason) {
|
||||
const nodeIdx = graph.nodes.findIndex((n) => n.id === nodeId);
|
||||
if (nodeIdx === -1) {
|
||||
return { success: false, error: `Node "${nodeId}" not found in graph` };
|
||||
}
|
||||
|
||||
const previousStatus = graph.nodes[nodeIdx].status;
|
||||
const previousValue = graph.nodes[nodeIdx].value;
|
||||
|
||||
return {
|
||||
success: true,
|
||||
previousStatus,
|
||||
newStatus,
|
||||
previousValue,
|
||||
newValue,
|
||||
reason,
|
||||
affectedNodes: findAffectedNodes(graph, nodeId),
|
||||
};
|
||||
}
|
||||
|
||||
// ── Select the next highest-value active unknown candidate ──
|
||||
|
||||
export function selectActiveUnknownCandidate(graph, resolvedNodeIds) {
|
||||
// Skip already resolved nodes
|
||||
const unresolved = graph.nodes.filter(
|
||||
(n) => n.kind === "unknown" && !resolvedNodeIds.includes(n.id),
|
||||
);
|
||||
|
||||
if (unresolved.length === 0) return null;
|
||||
|
||||
const scoredCandidates = unresolved.map((node) => ({
|
||||
node,
|
||||
...scoreUnknownCandidate(graph, node, resolvedNodeIds),
|
||||
}));
|
||||
|
||||
scoredCandidates.sort((a, b) => {
|
||||
if (b.score !== a.score) return b.score - a.score;
|
||||
if (b.downstreamCount !== a.downstreamCount) {
|
||||
return b.downstreamCount - a.downstreamCount;
|
||||
}
|
||||
if (a.unresolvedParentUnknownCount !== b.unresolvedParentUnknownCount) {
|
||||
return a.unresolvedParentUnknownCount - b.unresolvedParentUnknownCount;
|
||||
}
|
||||
return a.node.label.localeCompare(b.node.label);
|
||||
});
|
||||
|
||||
const best = scoredCandidates[0];
|
||||
if (!best) return null;
|
||||
|
||||
return {
|
||||
nodeId: best.node.id,
|
||||
label: best.node.label,
|
||||
score: best.score,
|
||||
question: buildDeterministicQuestionForUnknown(best.node),
|
||||
reason: `Selected for highest information value (score ${best.score}) with ${best.downstreamCount} downstream dependency node(s) and ${best.unresolvedParentUnknownCount} unresolved prerequisite unknown(s).`,
|
||||
};
|
||||
}
|
||||
|
||||
// ── Apply a graph update deterministically ──
|
||||
|
||||
export function applyGraphUpdate(graph, update) {
|
||||
const errors = [];
|
||||
const updatedNodesMap = new Map();
|
||||
|
||||
// Validate that update references existing nodes or newly added ones
|
||||
const allNodeIds = new Set(graph.nodes.map((n) => n.id));
|
||||
for (const added of update.addedNodes) {
|
||||
if (allNodeIds.has(added.id)) {
|
||||
errors.push(`Cannot add node with duplicate ID: "${added.id}"`);
|
||||
continue;
|
||||
}
|
||||
allNodeIds.add(added.id);
|
||||
}
|
||||
|
||||
// Validate updated nodes exist
|
||||
for (const upd of update.updatedNodes) {
|
||||
if (!allNodeIds.has(upd.nodeId)) {
|
||||
errors.push(`Cannot update non-existent node: "${upd.nodeId}"`);
|
||||
}
|
||||
}
|
||||
|
||||
// Validate added edges reference existing or new nodes
|
||||
for (const edge of update.addedEdges) {
|
||||
if (!allNodeIds.has(edge.fromNodeId)) {
|
||||
errors.push(
|
||||
`Added edge references non-existent fromNodeId: "${edge.fromNodeId}"`,
|
||||
);
|
||||
}
|
||||
if (!allNodeIds.has(edge.toNodeId)) {
|
||||
errors.push(
|
||||
`Added edge references non-existent toNodeId: "${edge.toNodeId}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if (errors.length > 0) return { success: false, errors };
|
||||
|
||||
// Build the new nodes list — start with a deep copy of existing
|
||||
const newNodes = graph.nodes.map((n) => ({ ...n }));
|
||||
|
||||
// Apply updated nodes
|
||||
for (const upd of update.updatedNodes) {
|
||||
const idx = newNodes.findIndex((n) => n.id === upd.nodeId);
|
||||
if (idx === -1) continue; // already validated above
|
||||
|
||||
if (upd.newStatus !== undefined && upd.newStatus !== null) {
|
||||
newNodes[idx].status = upd.newStatus;
|
||||
}
|
||||
if (upd.newValue !== undefined) {
|
||||
newNodes[idx].value = upd.newValue;
|
||||
}
|
||||
updatedNodesMap.set(upd.nodeId, newNodes[idx]);
|
||||
}
|
||||
|
||||
// Add new nodes
|
||||
for (const newNode of update.addedNodes) {
|
||||
if (!allNodeIds.has(newNode.id)) continue;
|
||||
allNodeIds.add(newNode.id);
|
||||
newNodes.push({ ...newNode });
|
||||
}
|
||||
|
||||
// Remove edges if requested
|
||||
const removedEdgeSet = new Set(update.removedEdgeIds);
|
||||
const newEdges = graph.edges.filter((e) => !removedEdgeSet.has(e.id));
|
||||
|
||||
// Add new edges
|
||||
for (const newEdge of update.addedEdges) {
|
||||
newEdges.push({ ...newEdge });
|
||||
|
||||
// Update dependsOn / affects on the nodes
|
||||
const fromNode = newNodes.find((n) => n.id === newEdge.fromNodeId);
|
||||
const toNode = newNodes.find((n) => n.id === newEdge.toNodeId);
|
||||
if (fromNode && !fromNode.childIds.includes(newEdge.toNodeId)) {
|
||||
fromNode.childIds.push(newEdge.toNodeId);
|
||||
}
|
||||
if (toNode && !toNode.dependsOn.includes(newEdge.fromNodeId)) {
|
||||
toNode.dependsOn.push(newEdge.fromNodeId);
|
||||
}
|
||||
}
|
||||
|
||||
// Add resolved node IDs
|
||||
const newResolved = [
|
||||
...new Set([...graph.resolvedNodeIds, ...update.resolvedUnknownNodeIds]),
|
||||
];
|
||||
|
||||
return {
|
||||
success: true,
|
||||
nodes: newNodes,
|
||||
edges: newEdges,
|
||||
resolvedNodeIds: newResolved,
|
||||
};
|
||||
}
|
||||
|
||||
// ── Validate a proposed graph update before application ──
|
||||
|
||||
export function validateGraphUpdate(graph, update) {
|
||||
const errors = [];
|
||||
|
||||
// Check for duplicate node IDs against existing and newly added nodes
|
||||
const extendedIds = new Set(graph.nodes.map((n) => n.id));
|
||||
for (const newNode of update.addedNodes) {
|
||||
if (extendedIds.has(newNode.id)) {
|
||||
errors.push(`Cannot add node with duplicate ID: "${newNode.id}"`);
|
||||
} else {
|
||||
extendedIds.add(newNode.id);
|
||||
}
|
||||
}
|
||||
|
||||
// Check updated nodes exist (in original graph, not newly added ones)
|
||||
const existingIds = new Set(graph.nodes.map((n) => n.id));
|
||||
for (const upd of update.updatedNodes) {
|
||||
if (!existingIds.has(upd.nodeId)) {
|
||||
errors.push(`Cannot update non-existent node: "${upd.nodeId}"`);
|
||||
}
|
||||
}
|
||||
|
||||
// Reject updates with no meaningful change
|
||||
const statusChanged = update.updatedNodes.some(
|
||||
(u) => u.previousStatus !== null && u.newStatus !== u.previousStatus,
|
||||
);
|
||||
const valueChanged = update.updatedNodes.some(
|
||||
(u) => u.previousValue !== null && u.newValue !== u.previousValue,
|
||||
);
|
||||
|
||||
const hasMeaningfulChange =
|
||||
update.addedNodes.length > 0 ||
|
||||
statusChanged ||
|
||||
valueChanged ||
|
||||
update.addedEdges.length > 0 ||
|
||||
update.removedEdgeIds.length > 0;
|
||||
|
||||
if (!hasMeaningfulChange) {
|
||||
errors.push("Update contains no meaningful change");
|
||||
}
|
||||
|
||||
// Reject oversized input
|
||||
const totalSize = JSON.stringify(update).length;
|
||||
if (totalSize > 100000) {
|
||||
errors.push(`Proposed graph update exceeds 100KB (${totalSize} bytes)`);
|
||||
}
|
||||
|
||||
return { valid: errors.length === 0, errors };
|
||||
}
|
||||
+3
-5
@@ -93,11 +93,9 @@ async function detectChatSupport(baseUrl) {
|
||||
|
||||
class OllamaLlmProvider {
|
||||
async generateReconstruction(scenario, modelName) {
|
||||
const { buildPrompt } = await import("@/lib/reconstruction/prompt");
|
||||
|
||||
let rawPrompt = buildPrompt(scenario);
|
||||
// Stronger JSON hint since we can't use format:json on older Ollama
|
||||
const prompt = rawPrompt + `\n\nReturn ONLY a valid JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.`;
|
||||
// scenario is ALREADY a fully-built prompt text (built by analyseScenario).
|
||||
// Do NOT call buildPrompt() again — that would double-wrap the prompt.
|
||||
const prompt = scenario;
|
||||
|
||||
const baseUrl = process.env.OLLAMA_BASE_URL;
|
||||
if (!baseUrl) throw new Error("OLLAMA_BASE_URL is not set");
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
function cloneJsonSafe(value) {
|
||||
if (value == null) return value;
|
||||
return JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
export function normaliseAnalysisResponse(input) {
|
||||
const normalised = cloneJsonSafe(input);
|
||||
const changesApplied = [];
|
||||
const warnings = [];
|
||||
|
||||
if (!normalised || typeof normalised !== "object") {
|
||||
return { normalised: input, changesApplied, warnings };
|
||||
}
|
||||
|
||||
if (Array.isArray(normalised.evidence)) {
|
||||
normalised.evidence = normalised.evidence.map((record, index) => {
|
||||
if (!record || typeof record !== "object") return record;
|
||||
|
||||
if (record.source === null) {
|
||||
changesApplied.push({
|
||||
path: ["evidence", index, "source"],
|
||||
change: "Converted null source to undefined",
|
||||
});
|
||||
|
||||
const { source: _removed, ...rest } = record;
|
||||
return rest;
|
||||
}
|
||||
|
||||
return record;
|
||||
});
|
||||
}
|
||||
|
||||
if (changesApplied.length > 0) {
|
||||
warnings.push(
|
||||
"Applied deterministic reconstruction compatibility normalisation",
|
||||
);
|
||||
}
|
||||
|
||||
return { normalised, changesApplied, warnings };
|
||||
}
|
||||
@@ -1,5 +1,24 @@
|
||||
export function buildPrompt(scenario) {
|
||||
return `You are a neutral analyst performing an evidence-based reconstruction of the following scenario.
|
||||
import { promises as fs } from "node:fs";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { dirname, join } from "node:path";
|
||||
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = dirname(__filename);
|
||||
const PROMPTS_DIR = join(__dirname, "../../prompts");
|
||||
|
||||
/** Available prompt versions */
|
||||
export const PROMPT_VERSIONS = ["v0.1", "v0.2", "v0.3"];
|
||||
|
||||
/** Default prompt version (override via RECONSTRUCTION_PROMPT_VERSION env var) */
|
||||
const defaultVersionFromEnv = process.env.RECONSTRUCTION_PROMPT_VERSION;
|
||||
export const DEFAULT_PROMPT_VERSION =
|
||||
defaultVersionFromEnv && PROMPT_VERSIONS.includes(defaultVersionFromEnv)
|
||||
? defaultVersionFromEnv
|
||||
: "v0.3";
|
||||
|
||||
/** Build a v0.1 (extraction-only) prompt inline for backward compatibility */
|
||||
function buildV1Prompt(scenario) {
|
||||
return `You are a neutral analyst performing an evidence-based reconstruction of the following scenario.
|
||||
|
||||
Rules:
|
||||
1. Do NOT invent facts. Only include information present in the scenario or clearly implied.
|
||||
@@ -29,3 +48,55 @@ Return valid JSON matching this structure exactly:
|
||||
|
||||
Return ONLY the JSON object. No markdown, no explanation, no preamble.`;
|
||||
}
|
||||
|
||||
/** Load a versioned prompt from disk and substitute {{SCENARIO}} */
|
||||
async function buildV2Prompt(scenario) {
|
||||
try {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.2.md"),
|
||||
"utf-8",
|
||||
);
|
||||
return content.replace("{{SCENARIO}}", scenario);
|
||||
} catch {
|
||||
// Fall back to v0.1 prompt if v0.2 file is missing
|
||||
return buildV1Prompt(scenario);
|
||||
}
|
||||
}
|
||||
|
||||
/** Load a versioned prompt from disk and substitute {{SCENARIO}} */
|
||||
async function buildV3Prompt(scenario) {
|
||||
try {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
return content.replace("{{SCENARIO}}", scenario);
|
||||
} catch {
|
||||
// Fall back to v0.2 prompt if v0.3 file is missing
|
||||
return buildV2Prompt(scenario);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an analysis prompt for the given version.
|
||||
* @param {"v0.1" | "v0.2" | "v0.3"} [version="v0.3"]
|
||||
* @returns {Promise<{prompt: string, version: string}>}
|
||||
*/
|
||||
export async function buildPrompt(scenario, version = "v0.3") {
|
||||
let prompt;
|
||||
switch (version) {
|
||||
case "v0.1":
|
||||
prompt = buildV1Prompt(scenario);
|
||||
break;
|
||||
case "v0.2":
|
||||
prompt = await buildV2Prompt(scenario);
|
||||
break;
|
||||
default: // v0.3
|
||||
prompt = await buildV3Prompt(scenario);
|
||||
break;
|
||||
}
|
||||
|
||||
const strongJsonHint =
|
||||
"\n\nReturn ONLY a valid JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.";
|
||||
return { prompt: prompt + strongJsonHint, version };
|
||||
}
|
||||
|
||||
+183
-16
@@ -1,38 +1,59 @@
|
||||
import { z } from "zod";
|
||||
|
||||
const confidenceEnum = z.enum(["low", "medium", "high"]);
|
||||
// ──────────────────────────────────────────────
|
||||
// Shared enums (v0.1 & v0.2)
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
const itemSchema = z.object({
|
||||
export const confidenceEnum = z.enum(["low", "medium", "high"]);
|
||||
const importanceEnum = z.enum([
|
||||
"incidental",
|
||||
"supporting",
|
||||
"important",
|
||||
"critical",
|
||||
]);
|
||||
const expectedInfoValueEnum = z.enum(["low", "medium", "high"]);
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// v0.1 — extraction-only schema (preserved)
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
const confidenceEnumV1 = z.enum(["low", "medium", "high"]);
|
||||
|
||||
const itemSchemaV1 = z.object({
|
||||
id: z.string().min(1),
|
||||
description: z.string().min(1),
|
||||
confidence: confidenceEnum,
|
||||
confidence: confidenceEnumV1,
|
||||
});
|
||||
|
||||
export const reconstructionSchema = z.object({
|
||||
observations: z.array(itemSchema),
|
||||
observations: z.array(itemSchemaV1),
|
||||
reportedClaims: z.array(
|
||||
itemSchema.extend({
|
||||
attributedTo: z.union([z.string().min(1), z.null()]).optional().nullable(),
|
||||
})
|
||||
itemSchemaV1.extend({
|
||||
attributedTo: z
|
||||
.union([z.string().min(1), z.null()])
|
||||
.optional()
|
||||
.nullable(),
|
||||
}),
|
||||
),
|
||||
assumptions: z.array(itemSchema),
|
||||
entities: z.array(itemSchema),
|
||||
assumptions: z.array(itemSchemaV1),
|
||||
entities: z.array(itemSchemaV1),
|
||||
transitions: z.array(
|
||||
itemSchema.extend({
|
||||
itemSchemaV1.extend({
|
||||
entity: z.string().min(1),
|
||||
previousState: z.string().min(1),
|
||||
currentState: z.string().min(1),
|
||||
explanationStatus: z.string().min(1),
|
||||
})
|
||||
}),
|
||||
),
|
||||
expectedButMissing: z.array(itemSchema),
|
||||
presentButUnexpected: z.array(itemSchema),
|
||||
contradictions: z.array(itemSchema),
|
||||
openUncertainties: z.array(itemSchema),
|
||||
expectedButMissing: z.array(itemSchemaV1),
|
||||
presentButUnexpected: z.array(itemSchemaV1),
|
||||
contradictions: z.array(itemSchemaV1),
|
||||
openUncertainties: z.array(itemSchemaV1),
|
||||
});
|
||||
|
||||
// v0.1 analyse response (used internally)
|
||||
export const analyseResponseSchema = z.object({
|
||||
reconstruction: reconstructionSchema,
|
||||
reconstruction: z.union([reconstructionSchema, z.null()]),
|
||||
modelName: z.string(),
|
||||
responseDurationMs: z.number(),
|
||||
validationStatus: z.enum(["valid", "partial", "invalid"]),
|
||||
@@ -48,6 +69,141 @@ export const healthResponseSchema = z.object({
|
||||
error: z.string().nullable(),
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// v0.2 — reasoning classification + reconstruction
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
export const inputTypes =
|
||||
/** @type {z.ZodType<typeof import("@/lib/reconstruction/schema").INPUT_TYPE_VALUE>} */ (
|
||||
z.enum([
|
||||
"observed_problem",
|
||||
"unexplained_change",
|
||||
"contradiction",
|
||||
"decision_request",
|
||||
"causal_claim",
|
||||
"reported_claim",
|
||||
"fault_report",
|
||||
"ambiguous_statement",
|
||||
"question",
|
||||
"desired_outcome",
|
||||
"insufficient_context",
|
||||
"other",
|
||||
])
|
||||
);
|
||||
|
||||
export const reasoningModes =
|
||||
/** @type {z.ZodType<typeof import("@/lib/reconstruction/schema").REASONING_MODE_VALUE>} */ (
|
||||
z.enum([
|
||||
"establish_baseline",
|
||||
"identify_difference",
|
||||
"reconstruct_transition",
|
||||
"decompose_aggregate",
|
||||
"validate_measurement",
|
||||
"validate_claim",
|
||||
"investigate_contradiction",
|
||||
"clarify_meaning",
|
||||
"decision_support",
|
||||
"fault_investigation",
|
||||
"identify_missing_information",
|
||||
"test_possible_explanations",
|
||||
"other",
|
||||
])
|
||||
);
|
||||
|
||||
const evidenceRecordSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
description: z.string().min(1),
|
||||
evidenceType: z.enum([
|
||||
"direct_observation",
|
||||
"reported_statement",
|
||||
"interpretation",
|
||||
"assumption",
|
||||
"inferred_relationship",
|
||||
]),
|
||||
source: z.string().optional(),
|
||||
attribution: z.string().nullable().optional(),
|
||||
confidence: confidenceEnum,
|
||||
importance: importanceEnum,
|
||||
});
|
||||
|
||||
const reconstructionSchemaV2 = z.object({
|
||||
summary: z.string().min(1),
|
||||
actors: z.array(itemSchemaV1),
|
||||
systemsOrObjects: z.array(itemSchemaV1),
|
||||
expectedStates: z.array(itemSchemaV1),
|
||||
observedStates: z.array(itemSchemaV1),
|
||||
differences: z.array(itemSchemaV1),
|
||||
knownTransitions: z.array(
|
||||
itemSchemaV1.extend({
|
||||
entity: z.string().min(1),
|
||||
previousState: z.string().min(1),
|
||||
currentState: z.string().min(1),
|
||||
explanationStatus: z.string().min(1),
|
||||
}),
|
||||
),
|
||||
unexplainedTransitions: z.array(
|
||||
itemSchemaV1.extend({
|
||||
entity: z.string().min(1).optional(),
|
||||
previousState: z.string().min(1).optional(),
|
||||
currentState: z.string().min(1).optional(),
|
||||
}),
|
||||
),
|
||||
contradictions: z.array(itemSchemaV1),
|
||||
importantUnknowns: z.array(itemSchemaV1),
|
||||
plausibleInterpretations: z.array(
|
||||
z.object({
|
||||
id: z.string().min(1),
|
||||
description: z.string().min(1),
|
||||
supportingEvidenceIds: z.array(z.string()),
|
||||
assumptionsRequired: z.array(z.string()).optional().default([]),
|
||||
confidence: confidenceEnum,
|
||||
}),
|
||||
),
|
||||
});
|
||||
|
||||
const inputClassificationSchema = z.object({
|
||||
primaryType: inputTypes,
|
||||
secondaryTypes: z.array(inputTypes).optional().default([]),
|
||||
reasoningModes: z.array(reasoningModes).optional().default([]),
|
||||
classificationReason: z.string().min(1),
|
||||
confidence: confidenceEnum,
|
||||
});
|
||||
|
||||
const nextQuestionSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
question: z.string().min(1),
|
||||
targets: z.array(z.string()),
|
||||
reason: z.string().min(1),
|
||||
expectedInformationValue: expectedInfoValueEnum,
|
||||
reasoningMode: reasoningModes.optional().default("other"),
|
||||
});
|
||||
|
||||
// v0.2 complete analysis response (what the model produces)
|
||||
export const reconstructionV2Schema = z.object({
|
||||
inputClassification: inputClassificationSchema,
|
||||
reconstruction: reconstructionSchemaV2,
|
||||
evidence: z.array(evidenceRecordSchema),
|
||||
nextQuestion: nextQuestionSchema,
|
||||
});
|
||||
|
||||
// Outer wrapper for API return (includes diagnostics + v0.2 data)
|
||||
export const analyseResponseV2Schema = z.object({
|
||||
inputClassification: inputClassificationSchema.optional(),
|
||||
reconstruction: reconstructionSchemaV2.optional().nullable(),
|
||||
evidence: z.array(evidenceRecordSchema).optional(),
|
||||
nextQuestion: nextQuestionSchema.optional(),
|
||||
modelName: z.string(),
|
||||
responseDurationMs: z.number(),
|
||||
validationStatus: z.enum(["valid", "partial", "invalid"]),
|
||||
rawResponse: z.string().optional(),
|
||||
errors: z.array(z.string()).optional(),
|
||||
promptVersion: z.string().optional(),
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// Parsing helpers
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
export function parseReconstruction(raw) {
|
||||
if (typeof raw === "string") {
|
||||
try {
|
||||
@@ -58,3 +214,14 @@ export function parseReconstruction(raw) {
|
||||
}
|
||||
return reconstructionSchema.parse(raw);
|
||||
}
|
||||
|
||||
export function parseReconstructionV2(raw) {
|
||||
if (typeof raw === "string") {
|
||||
try {
|
||||
raw = JSON.parse(raw);
|
||||
} catch {
|
||||
throw new SyntaxError("Model response is not valid JSON");
|
||||
}
|
||||
}
|
||||
return reconstructionV2Schema.parse(raw);
|
||||
}
|
||||
|
||||
Generated
+66
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "confidence-engine",
|
||||
"version": "0.1.0",
|
||||
"version": "0.2.0-experimental",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "confidence-engine",
|
||||
"version": "0.1.0",
|
||||
"version": "0.2.0-experimental",
|
||||
"dependencies": {
|
||||
"next": "^14.2.0",
|
||||
"react": "^18.3.0",
|
||||
@@ -14,6 +14,7 @@
|
||||
"zod": "^3.23.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@playwright/test": "^1.62.1",
|
||||
"@types/node": "^20.14.0",
|
||||
"@types/react": "^18.3.0",
|
||||
"@types/react-dom": "^18.3.0",
|
||||
@@ -888,6 +889,22 @@
|
||||
"node": ">=14"
|
||||
}
|
||||
},
|
||||
"node_modules/@playwright/test": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.62.1.tgz",
|
||||
"integrity": "sha512-DTcUc8qii+cpHvtOwggMtBRMjKZHXYWdw8syRYu2vtzuq4Wxphqq4NfCs5Zt44L6mA8rfDfj+PHnxFc/FeK6mQ==",
|
||||
"devOptional": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"playwright": "1.62.1"
|
||||
},
|
||||
"bin": {
|
||||
"playwright": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
}
|
||||
},
|
||||
"node_modules/@rollup/rollup-android-arm-eabi": {
|
||||
"version": "4.62.3",
|
||||
"resolved": "https://registry.npmjs.org/@rollup/rollup-android-arm-eabi/-/rollup-android-arm-eabi-4.62.3.tgz",
|
||||
@@ -5601,6 +5618,53 @@
|
||||
"node": ">= 6"
|
||||
}
|
||||
},
|
||||
"node_modules/playwright": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/playwright/-/playwright-1.62.1.tgz",
|
||||
"integrity": "sha512-0M+L3LAD8/nm554LOla9Ayx0j0tmFZ0FBcoQ7F1VuVHpM/XpiC8RcDzBQB8W5+hA8L22THxELzeF+2WcUzvcLg==",
|
||||
"devOptional": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"playwright-core": "1.62.1"
|
||||
},
|
||||
"bin": {
|
||||
"playwright": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"fsevents": "2.3.2"
|
||||
}
|
||||
},
|
||||
"node_modules/playwright-core": {
|
||||
"version": "1.62.1",
|
||||
"resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.62.1.tgz",
|
||||
"integrity": "sha512-wPYSwEBJY9GHraISXqyqtx0na0LpO3XEX7jNDhntbex7tzUS7kLnZsOlFruFJB4Hi/rhDMjXGqHewDZ68nYZVw==",
|
||||
"devOptional": true,
|
||||
"license": "Apache-2.0",
|
||||
"bin": {
|
||||
"playwright-core": "cli.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
}
|
||||
},
|
||||
"node_modules/playwright/node_modules/fsevents": {
|
||||
"version": "2.3.2",
|
||||
"resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz",
|
||||
"integrity": "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==",
|
||||
"dev": true,
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": "^8.16.0 || ^10.6.0 || >=11.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/possible-typed-array-names": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/possible-typed-array-names/-/possible-typed-array-names-1.1.0.tgz",
|
||||
|
||||
+2
-1
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "confidence-engine",
|
||||
"version": "0.1.0",
|
||||
"version": "0.2.0-experimental",
|
||||
"private": true,
|
||||
"description": "Experimental prototype for evidence-based situation reconstruction using local LLMs",
|
||||
"scripts": {
|
||||
@@ -19,6 +19,7 @@
|
||||
"zod": "^3.23.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@playwright/test": "^1.62.1",
|
||||
"@types/node": "^20.14.0",
|
||||
"@types/react": "^18.3.0",
|
||||
"@types/react-dom": "^18.3.0",
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
import { defineConfig } from "@playwright/test";
|
||||
export default defineConfig({
|
||||
use: { headless: true, screenshot: "only-on-failure", actionTimeout: 120000 },
|
||||
testMatch: "**/tests/smoke.test.js",
|
||||
});
|
||||
@@ -0,0 +1,122 @@
|
||||
You are a neutral analyst performing evidence-based situation reconstruction.
|
||||
|
||||
## Rules
|
||||
|
||||
1. Do NOT invent facts, context or causes. Only include information present in the scenario or clearly implied.
|
||||
2. First determine what kind of input has been supplied. Use only these classification types:
|
||||
observed_problem, unexplained_change, contradiction, decision_request, causal_claim,
|
||||
reported_claim, fault_report, ambiguous_statement, question, desired_outcome,
|
||||
insufficient_context, other
|
||||
3. Choose reasoning modes from:
|
||||
establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate,
|
||||
validate_measurement, validate_claim, investigate_contradiction, clarify_meaning,
|
||||
decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other
|
||||
4. Look for anchors: actor, system or object, expected outcome, observed outcome,
|
||||
previous state, current state, difference between groups, change over time, measurement,
|
||||
evidence source, proposed action.
|
||||
5. Identify meaningful differences (e.g., some succeed while others fail; revenue rises while cash falls).
|
||||
6. Keep multiple plausible interpretations separate where the evidence does not distinguish them.
|
||||
7. Distinguish: what was said / what it may mean / why it may have been said.
|
||||
8. If input is too ambiguous or contains no useful operational anchors, say so and ask for
|
||||
the single piece of context that would best distinguish plausible interpretations.
|
||||
|
||||
## Confidence scale
|
||||
|
||||
- low — weak evidence, speculation, or missing information
|
||||
- medium — reasonable inference from available evidence
|
||||
- high — strong evidence, direct observation, or confirmed fact
|
||||
|
||||
## Importance scale (evidence records)
|
||||
|
||||
- incidental — minor detail, unlikely to affect conclusions
|
||||
- supporting — adds context but not critical
|
||||
- important — materially affects understanding of the situation
|
||||
- critical — essential to resolving the situation; without it conclusions cannot be drawn
|
||||
|
||||
## Expected information value (next question)
|
||||
|
||||
- low — marginally useful even if answered
|
||||
- medium — meaningfully clarifies the situation
|
||||
- high — would significantly distinguish between plausible explanations or fill a gap in understanding
|
||||
|
||||
## Next question selection criteria
|
||||
|
||||
Prefer questions that:
|
||||
- clarify a major difference
|
||||
- establish a baseline
|
||||
- explain an important transition
|
||||
- test an unsupported claim
|
||||
- distinguish between plausible explanations
|
||||
- request measurable evidence
|
||||
- identify who or what is affected
|
||||
- establish timing
|
||||
|
||||
Avoid questions that:
|
||||
- have already been answered
|
||||
- assume a cause
|
||||
- jump to a solution
|
||||
- ask about motive before the observable situation is understood
|
||||
- focus on incidental wording
|
||||
- are too broad to produce useful information
|
||||
- combine many unrelated questions
|
||||
|
||||
## Output format — return this exact JSON structure
|
||||
|
||||
Return a JSON object with exactly these four top-level keys (use **camelCase**):
|
||||
|
||||
```json
|
||||
{
|
||||
"inputClassification": {
|
||||
"primaryType": "<one of: observed_problem, unexplained_change, contradiction, decision_request, causal_claim, reported_claim, fault_report, ambiguous_statement, question, desired_outcome, insufficient_context, other>",
|
||||
"secondaryTypes": ["<optional additional types from the same list>"],
|
||||
"reasoningModes": ["<one or more of: establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate, validate_measurement, validate_claim, investigate_contradiction, clarify_meaning, decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other>"],
|
||||
"classificationReason": "<brief explanation of why you chose the primary type>",
|
||||
"confidence": "<low | medium | high>"
|
||||
},
|
||||
"reconstruction": {
|
||||
"summary": "<one-sentence overview of the situation>",
|
||||
"actors": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"systemsOrObjects": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"expectedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"observedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"differences": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"knownTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}],
|
||||
"unexplainedTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "..."}],
|
||||
"contradictions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"importantUnknowns": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"plausibleInterpretations": [{"id": "...", "description": "...", "supportingEvidenceIds": ["<ids that support this interpretation>"], "assumptionsRequired": [], "confidence": "<low|medium|high>"}]
|
||||
},
|
||||
"evidence": [
|
||||
{
|
||||
"id": "<any unique string>",
|
||||
"description": "...",
|
||||
"evidenceType": "<direct_observation | reported_statement | interpretation | assumption | inferred_relationship>",
|
||||
"source": "<optional — who/where this came from>",
|
||||
"attribution": null,
|
||||
"confidence": "<low | medium | high>",
|
||||
"importance": "<incidental | supporting | important | critical>"
|
||||
}
|
||||
],
|
||||
"nextQuestion": {
|
||||
"id": "<any unique string>",
|
||||
"question": "<one precise question>",
|
||||
"targets": ["<what this question targets — e.g. 'actor', 'system', 'expectedOutcome'>"],
|
||||
"reason": "<why answering this is important>",
|
||||
"expectedInformationValue": "<low | medium | high>",
|
||||
"reasoningMode": "<optional reasoning mode from the list above>"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
CRITICAL RULES for JSON output:
|
||||
1. Use **exactly** the key names shown above (camelCase, no snake_case).
|
||||
2. The four top-level keys must be: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`.
|
||||
3. Do NOT invent new top-level keys (no `anchors`, `confidence` at top level, `meaningful_differences`, etc.).
|
||||
4. Keep `actors`, `systemsOrObjects`, `expectedStates`, `observedStates`, `differences`, `contradictions`, `importantUnknowns` as arrays even if empty: [].
|
||||
5. Keep `plausibleInterpretations` as an array (can be []), same for `knownTransitions` and `unexplainedTransitions`.
|
||||
6. Each object in arrays must have at least `id`, `description`, `confidence`.
|
||||
|
||||
Scenario:
|
||||
{{SCENARIO}}
|
||||
|
||||
Return ONLY the JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.
|
||||
@@ -0,0 +1,160 @@
|
||||
You are a neutral analyst performing evidence-based situation reconstruction.
|
||||
|
||||
## Rules
|
||||
|
||||
1. Do NOT invent facts, context or causes. Only include information present in the scenario or clearly implied.
|
||||
2. First determine what kind of input has been supplied. Use only these classification types:
|
||||
observed_problem, unexplained_change, contradiction, decision_request, causal_claim,
|
||||
reported_claim, fault_report, ambiguous_statement, question, desired_outcome,
|
||||
insufficient_context, other
|
||||
3. Choose reasoning modes from:
|
||||
establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate,
|
||||
validate_measurement, validate_claim, investigate_contradiction, clarify_meaning,
|
||||
decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other
|
||||
4. Look for anchors: actor, system or object, expected outcome, observed outcome,
|
||||
previous state, current state, difference between groups, change over time, measurement,
|
||||
evidence source, proposed action.
|
||||
5. Identify meaningful differences (e.g., some succeed while others fail; revenue rises while cash falls).
|
||||
6. Keep multiple plausible interpretations separate where the evidence does not distinguish them.
|
||||
7. Distinguish: what was said / what it may mean / why it may have been said.
|
||||
8. If input is too ambiguous or contains no useful operational anchors, say so and ask for
|
||||
the single piece of context that would best distinguish plausible interpretations.
|
||||
|
||||
## Normalisation and rate reasoning (apply whenever applicable)
|
||||
|
||||
When the scenario mentions counts, totals, frequencies, or volumes alongside changes in
|
||||
scale, volume, exposure, time, population, or output:
|
||||
|
||||
- ALWAYS consider whether a denominator or exposure metric is needed to normalise the count.
|
||||
- Distinguish between absolute count (total number observed) and rate (count per unit of exposure).
|
||||
- Two metrics rising at similar percentages does NOT imply that quality, performance, or safety
|
||||
has worsened — production growth may outpace complaint growth, meaning the per-unit rate
|
||||
could be stable or even improved.
|
||||
- Identify the possible denominator explicitly (e.g., "per unit produced", "per customer served",
|
||||
"per hour of operation").
|
||||
- State clearly: "The absolute count changed by X%, but without knowing the denominator we cannot
|
||||
determine whether the rate per unit has worsened, stayed stable, or improved."
|
||||
- Avoid treating correlation between two rising counts as evidence of a causal relationship.
|
||||
|
||||
## Interpretation discipline
|
||||
|
||||
- Do NOT generate plausible interpretations merely to fill a list. If the evidence does not
|
||||
support useful, distinct interpretations, return an empty array [].
|
||||
- Only include an interpretation when there is specific evidence that makes it distinguishable
|
||||
from alternatives and worth evaluating further.
|
||||
- Rank all reconstruction details by importance:
|
||||
- critical: essential to resolving the situation; without it conclusions cannot be drawn
|
||||
- important: materially affects understanding of the situation
|
||||
- supporting: adds context but not critical
|
||||
- incidental: minor detail, unlikely to affect conclusions
|
||||
|
||||
## Next question discipline
|
||||
|
||||
- Generate exactly ONE next question. Do NOT combine multiple questions.
|
||||
- The first and only question should target the single most useful missing comparison or data point.
|
||||
- Prefer narrow, specific questions over broad compound questions.
|
||||
- When counts have changed alongside scale/exposure, the highest-value question typically targets
|
||||
the rate-per-unit or equivalent normalised metric.
|
||||
- Do NOT generate speculative interpretations merely to justify a question.
|
||||
|
||||
## Confidence scale
|
||||
|
||||
- low — weak evidence, speculation, or missing information
|
||||
- medium — reasonable inference from available evidence
|
||||
- high — strong evidence, direct observation, or confirmed fact
|
||||
|
||||
## Importance scale (evidence records)
|
||||
|
||||
- incidental — minor detail, unlikely to affect conclusions
|
||||
- supporting — adds context but not critical
|
||||
- important — materially affects understanding of the situation
|
||||
- critical — essential to resolving the situation; without it conclusions cannot be drawn
|
||||
|
||||
## Expected information value (next question)
|
||||
|
||||
- low — marginally useful even if answered
|
||||
- medium — meaningfully clarifies the situation
|
||||
- high — would significantly distinguish between plausible explanations or fill a gap in understanding
|
||||
|
||||
## Next question selection criteria
|
||||
|
||||
Prefer questions that:
|
||||
- clarify a major difference
|
||||
- establish a baseline
|
||||
- explain an important transition
|
||||
- test an unsupported claim
|
||||
- distinguish between plausible explanations
|
||||
- request measurable evidence
|
||||
- identify who or what is affected
|
||||
- establish timing
|
||||
|
||||
Avoid questions that:
|
||||
- have already been answered
|
||||
- assume a cause
|
||||
- jump to a solution
|
||||
- ask about motive before the observable situation is understood
|
||||
- focus on incidental wording
|
||||
- are too broad to produce useful information
|
||||
- combine many unrelated questions
|
||||
|
||||
## Output format — return this exact JSON structure
|
||||
|
||||
Return a JSON object with exactly these four top-level keys (use **camelCase**):
|
||||
|
||||
```json
|
||||
{
|
||||
"inputClassification": {
|
||||
"primaryType": "<one of: observed_problem, unexplained_change, contradiction, decision_request, causal_claim, reported_claim, fault_report, ambiguous_statement, question, desired_outcome, insufficient_context, other>",
|
||||
"secondaryTypes": ["<optional additional types from the same list>"],
|
||||
"reasoningModes": ["<one or more of: establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate, validate_measurement, validate_claim, investigate_contradiction, clarify_meaning, decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other>"],
|
||||
"classificationReason": "<brief explanation of why you chose the primary type>",
|
||||
"confidence": "<low | medium | high>"
|
||||
},
|
||||
"reconstruction": {
|
||||
"summary": "<one-sentence overview of the situation>",
|
||||
"actors": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"systemsOrObjects": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"expectedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"observedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"differences": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"knownTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}],
|
||||
"unexplainedTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "..."}],
|
||||
"contradictions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"importantUnknowns": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"plausibleInterpretations": [{"id": "...", "description": "...", "supportingEvidenceIds": ["<ids that support this interpretation>"], "assumptionsRequired": [], "confidence": "<low|medium|high>"}]
|
||||
},
|
||||
"evidence": [
|
||||
{
|
||||
"id": "<any unique string>",
|
||||
"description": "...",
|
||||
"evidenceType": "<direct_observation | reported_statement | interpretation | assumption | inferred_relationship>",
|
||||
"source": "<optional — who/where this came from>",
|
||||
"attribution": null,
|
||||
"confidence": "<low | medium | high>",
|
||||
"importance": "<incidental | supporting | important | critical>"
|
||||
}
|
||||
],
|
||||
"nextQuestion": {
|
||||
"id": "<any unique string>",
|
||||
"question": "<one precise question>",
|
||||
"targets": ["<what this question targets — e.g. 'actor', 'system', 'expectedOutcome'>"],
|
||||
"reason": "<why answering this is important>",
|
||||
"expectedInformationValue": "<low | medium | high>",
|
||||
"reasoningMode": "<optional reasoning mode from the list above>"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
CRITICAL RULES for JSON output:
|
||||
1. Use **exactly** the key names shown above (camelCase, no snake_case).
|
||||
2. The four top-level keys must be: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`.
|
||||
3. Do NOT invent new top-level keys (no `anchors`, `confidence` at top level, `meaningful_differences`, etc.).
|
||||
4. Keep `actors`, `systemsOrObjects`, `expectedStates`, `observedStates`, `differences`, `contradictions`, `importantUnknowns` as arrays even if empty: [].
|
||||
5. Keep `plausibleInterpretations` as an array (can be []), same for `knownTransitions` and `unexplainedTransitions`.
|
||||
6. Each object in arrays must have at least `id`, `description`, `confidence`.
|
||||
7. **evidenceType**: classify each evidence item clearly as either a direct observation, a reported statement, an interpretation, an assumption, or an inferred relationship. Do not treat raw counts as proof of causal relationships — they may be inferred relationships only when supported by explicit reasoning about denominators or rates.
|
||||
|
||||
Scenario:
|
||||
{{SCENARIO}}
|
||||
|
||||
Return ONLY the JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.
|
||||
@@ -0,0 +1,97 @@
|
||||
import { mkdir, writeFile } from "node:fs/promises";
|
||||
|
||||
const BASE_URL =
|
||||
process.env.CONFIDENCE_ENGINE_BASE_URL || "http://127.0.0.1:3000";
|
||||
const OUTPUT_DIR = "tests-results/commercial-value-update";
|
||||
|
||||
const scenario = "I think therefore I am";
|
||||
const answer =
|
||||
"Deciding whether to build the Confidence Engine due to uncertainty about its commercial value.";
|
||||
|
||||
async function postJson(path, body) {
|
||||
const response = await fetch(`${BASE_URL}${path}`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"content-type": "application/json",
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
|
||||
const json = await response.json();
|
||||
return { status: response.status, json };
|
||||
}
|
||||
|
||||
function printLine(label, value) {
|
||||
const rendered = value === undefined ? null : value;
|
||||
console.log(`${label}: ${JSON.stringify(rendered)}`);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
await mkdir(OUTPUT_DIR, { recursive: true });
|
||||
|
||||
const startResult = await postJson("/api/cases/start", { scenario });
|
||||
await writeFile(
|
||||
`${OUTPUT_DIR}/start-response.json`,
|
||||
JSON.stringify(startResult, null, 2),
|
||||
);
|
||||
|
||||
const selectedQuestion = startResult.json?.selectedQuestion?.question || null;
|
||||
|
||||
let updateResult = {
|
||||
status: null,
|
||||
json: {
|
||||
success: false,
|
||||
stage: "request_construction",
|
||||
errors: ["Missing selected question from start response"],
|
||||
},
|
||||
};
|
||||
|
||||
if (startResult.json?.success && selectedQuestion) {
|
||||
updateResult = await postJson("/api/cases/update", {
|
||||
situationGraph: startResult.json.situationGraph,
|
||||
previousQuestion: selectedQuestion,
|
||||
answer,
|
||||
});
|
||||
}
|
||||
|
||||
await writeFile(
|
||||
`${OUTPUT_DIR}/update-response.json`,
|
||||
JSON.stringify(updateResult, null, 2),
|
||||
);
|
||||
|
||||
printLine("start success", startResult.json?.success ?? false);
|
||||
printLine("update success", updateResult.json?.success ?? false);
|
||||
printLine("update stage", updateResult.json?.stage ?? null);
|
||||
printLine(
|
||||
"proposal added nodes",
|
||||
updateResult.json?.proposal?.addedNodes?.map((node) => node.id) ?? null,
|
||||
);
|
||||
printLine(
|
||||
"proposal added edges",
|
||||
updateResult.json?.proposal?.addedEdges?.map((edge) => ({
|
||||
id: edge.id,
|
||||
fromNodeId: edge.fromNodeId,
|
||||
toNodeId: edge.toNodeId,
|
||||
relationship: edge.relationship,
|
||||
})) ?? null,
|
||||
);
|
||||
printLine(
|
||||
"proposal resolved unknown IDs",
|
||||
updateResult.json?.proposal?.resolvedUnknownNodeIds ??
|
||||
updateResult.json?.resolvedUnknownNodeIds ??
|
||||
null,
|
||||
);
|
||||
printLine(
|
||||
"errors",
|
||||
updateResult.json?.errors ??
|
||||
updateResult.json?.proposalErrors ??
|
||||
updateResult.json?.graphValidationErrors ??
|
||||
updateResult.json?.validationErrors ??
|
||||
null,
|
||||
);
|
||||
}
|
||||
|
||||
main().catch((error) => {
|
||||
console.error(error instanceof Error ? error.message : String(error));
|
||||
process.exitCode = 1;
|
||||
});
|
||||
@@ -0,0 +1,114 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const mockStartCase = vi.fn();
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", () => ({
|
||||
startCase: (...args) => mockStartCase(...args),
|
||||
}));
|
||||
|
||||
describe("app/api/cases/start route", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("delegates request body to the orchestrator", async () => {
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: true,
|
||||
situationGraph: { nodes: [{ id: "n1" }], edges: [] },
|
||||
selectedQuestion: null,
|
||||
diagnostics: {},
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const request = new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ scenario: "Scenario text" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
|
||||
await POST(request);
|
||||
|
||||
expect(mockStartCase).toHaveBeenCalledWith({ scenario: "Scenario text" });
|
||||
});
|
||||
|
||||
it("returns 200 on success", async () => {
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: true,
|
||||
situationGraph: { nodes: [{ id: "n1" }], edges: [] },
|
||||
selectedQuestion: null,
|
||||
diagnostics: {},
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ scenario: "Scenario text" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
});
|
||||
|
||||
it("returns 400 for invalid request input", async () => {
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: false,
|
||||
error: "Invalid start-case request",
|
||||
validationErrors: [{ message: "Required" }],
|
||||
statusCode: 400,
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({}),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(400);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: false,
|
||||
error: "Invalid start-case request",
|
||||
});
|
||||
});
|
||||
|
||||
it("returns provider/internal failures as 5xx without stack traces", async () => {
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: false,
|
||||
error: "Provider unavailable",
|
||||
diagnostics: { modelName: "llama3" },
|
||||
statusCode: 502,
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ scenario: "Scenario text" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(502);
|
||||
await expect(response.json()).resolves.not.toHaveProperty("stack");
|
||||
});
|
||||
|
||||
it("returns structured 500 on malformed JSON", async () => {
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const request = {
|
||||
json: vi.fn().mockRejectedValue(new Error("Unexpected token")),
|
||||
};
|
||||
|
||||
const response = await POST(request);
|
||||
|
||||
expect(response.status).toBe(500);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: false,
|
||||
error: "Internal server error",
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,304 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const mockUpdateCase = vi.fn();
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", () => ({
|
||||
updateCase: (...args) => mockUpdateCase(...args),
|
||||
}));
|
||||
|
||||
function makeSuccessResult() {
|
||||
return {
|
||||
success: true,
|
||||
stage: "update_applied",
|
||||
updatedSituationGraph: {
|
||||
centralStatement: "Scenario",
|
||||
nodes: [{ id: "n1" }],
|
||||
edges: [],
|
||||
activeUnknownNodeId: null,
|
||||
resolvedNodeIds: ["n1"],
|
||||
currentSummary: "Updated summary",
|
||||
},
|
||||
proposal: {
|
||||
addedNodes: [],
|
||||
updatedNodes: [],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: ["n1"],
|
||||
affectedNodeIds: ["n1"],
|
||||
selectedQuestion: null,
|
||||
},
|
||||
selectedQuestion: null,
|
||||
affectedNodeIds: ["n1"],
|
||||
resolvedUnknownNodeIds: ["n1"],
|
||||
previousActiveUnknownNodeId: "n0",
|
||||
newActiveUnknownNodeId: null,
|
||||
changesApplied: { updatedNodeCount: 1 },
|
||||
diagnostics: { promptVersion: "v0.4" },
|
||||
};
|
||||
}
|
||||
|
||||
describe("app/api/cases/update route", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("valid update returns HTTP 200", async () => {
|
||||
mockUpdateCase.mockResolvedValue(makeSuccessResult());
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
});
|
||||
|
||||
it("route calls updateCase with applyProposal: true", async () => {
|
||||
mockUpdateCase.mockResolvedValue(makeSuccessResult());
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const body = { situationGraph: {}, previousQuestion: "Q", answer: "A" };
|
||||
await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify(body),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(mockUpdateCase).toHaveBeenCalledWith(body, { applyProposal: true });
|
||||
});
|
||||
|
||||
it("invalid JSON returns 400", async () => {
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const request = {
|
||||
json: vi.fn().mockRejectedValue(new SyntaxError("Unexpected token")),
|
||||
};
|
||||
|
||||
const response = await POST(request);
|
||||
|
||||
expect(response.status).toBe(400);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid JSON request body",
|
||||
});
|
||||
});
|
||||
|
||||
it("request validation failure returns 400", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid update-case request",
|
||||
validationErrors: [{ message: "Required" }],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({}),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(400);
|
||||
});
|
||||
|
||||
it("graph validation failure returns 400", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "graph_validation",
|
||||
error: "Invalid situation graph",
|
||||
graphValidationErrors: ["bad graph"],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(400);
|
||||
});
|
||||
|
||||
it("provider failure returns 502", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "provider",
|
||||
error: "Graph update proposal generation failed",
|
||||
providerErrors: ["provider offline"],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(502);
|
||||
});
|
||||
|
||||
it("proposal validation failure returns 422", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "proposal_validation",
|
||||
error: "Invalid graph update proposal",
|
||||
proposalErrors: [{ message: "bad proposal" }],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(422);
|
||||
});
|
||||
|
||||
it("proposal compatibility failure returns 422", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "proposal_compatibility",
|
||||
error: "Update case failed",
|
||||
errors: ["incompatible proposal"],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(422);
|
||||
});
|
||||
|
||||
it("application failure returns 422", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "application",
|
||||
error: "Update case failed",
|
||||
errors: ["could not apply"],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(422);
|
||||
});
|
||||
|
||||
it("result validation failure returns 500", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "result_validation",
|
||||
error: "Update case failed",
|
||||
errors: ["invalid result"],
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(500);
|
||||
});
|
||||
|
||||
it("unknown failure returns 500", async () => {
|
||||
mockUpdateCase.mockRejectedValue(new Error("boom"));
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(500);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: false,
|
||||
stage: "internal",
|
||||
error: "Internal server error",
|
||||
});
|
||||
});
|
||||
|
||||
it("success response preserves updated graph fields", async () => {
|
||||
const success = makeSuccessResult();
|
||||
mockUpdateCase.mockResolvedValue(success);
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
updatedSituationGraph: success.updatedSituationGraph,
|
||||
proposal: success.proposal,
|
||||
affectedNodeIds: success.affectedNodeIds,
|
||||
resolvedUnknownNodeIds: success.resolvedUnknownNodeIds,
|
||||
previousActiveUnknownNodeId: success.previousActiveUnknownNodeId,
|
||||
newActiveUnknownNodeId: success.newActiveUnknownNodeId,
|
||||
selectedQuestion: success.selectedQuestion,
|
||||
changesApplied: success.changesApplied,
|
||||
diagnostics: success.diagnostics,
|
||||
});
|
||||
});
|
||||
|
||||
it("stack traces and raw provider output are not exposed", async () => {
|
||||
mockUpdateCase.mockResolvedValue({
|
||||
success: false,
|
||||
stage: "provider",
|
||||
error: "Graph update proposal generation failed",
|
||||
providerErrors: ["provider offline"],
|
||||
rawResponse: "secret",
|
||||
stack: "trace",
|
||||
diagnostics: {},
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ answer: "A" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
const payload = await response.json();
|
||||
|
||||
expect(payload).not.toHaveProperty("stack");
|
||||
expect(payload).not.toHaveProperty("rawResponse");
|
||||
});
|
||||
});
|
||||
+433
@@ -0,0 +1,433 @@
|
||||
import { makeEdge, makeGraph, makeNode } from "@/lib/graph/schema.js";
|
||||
|
||||
function makeScenarioGraph({
|
||||
scenario,
|
||||
decisionNode,
|
||||
answeredContextUnknown,
|
||||
foundationalUnknown,
|
||||
consequentialUnknown,
|
||||
downstreamLeaf,
|
||||
}) {
|
||||
const nodes = [
|
||||
decisionNode,
|
||||
answeredContextUnknown,
|
||||
foundationalUnknown,
|
||||
consequentialUnknown,
|
||||
downstreamLeaf,
|
||||
];
|
||||
|
||||
const edges = [
|
||||
makeEdge({
|
||||
id: `${decisionNode.id}-to-${foundationalUnknown.id}`,
|
||||
fromNodeId: decisionNode.id,
|
||||
toNodeId: foundationalUnknown.id,
|
||||
relationship: "depends_on",
|
||||
description: `${decisionNode.label} depends on ${foundationalUnknown.label}.`,
|
||||
}),
|
||||
makeEdge({
|
||||
id: `${answeredContextUnknown.id}-to-${consequentialUnknown.id}`,
|
||||
fromNodeId: answeredContextUnknown.id,
|
||||
toNodeId: consequentialUnknown.id,
|
||||
relationship: "depends_on",
|
||||
description: `${consequentialUnknown.label} was surfaced from resolved context.`,
|
||||
}),
|
||||
makeEdge({
|
||||
id: `${foundationalUnknown.id}-to-${consequentialUnknown.id}`,
|
||||
fromNodeId: foundationalUnknown.id,
|
||||
toNodeId: consequentialUnknown.id,
|
||||
relationship: "depends_on",
|
||||
description: `${consequentialUnknown.label} depends on ${foundationalUnknown.label}.`,
|
||||
}),
|
||||
makeEdge({
|
||||
id: `${consequentialUnknown.id}-to-${downstreamLeaf.id}`,
|
||||
fromNodeId: consequentialUnknown.id,
|
||||
toNodeId: downstreamLeaf.id,
|
||||
relationship: "depends_on",
|
||||
description: `${downstreamLeaf.label} depends on ${consequentialUnknown.label}.`,
|
||||
}),
|
||||
];
|
||||
|
||||
return makeGraph({
|
||||
centralStatement: scenario,
|
||||
nodes,
|
||||
edges,
|
||||
activeUnknownNodeId: answeredContextUnknown.id,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary: "Generalisation fixture graph",
|
||||
});
|
||||
}
|
||||
|
||||
export const questionPriorityGeneralisationFixtures = [
|
||||
{
|
||||
key: "hire-engineer",
|
||||
scenario: "Should we hire another engineer?",
|
||||
decisionType: "resourcing decision",
|
||||
acceptableFoundationalUnknownNodeIds: [
|
||||
"hire-success-criteria",
|
||||
"hire-bottleneck",
|
||||
],
|
||||
prohibitedFirstTopics: ["salary", "job advert", "programming language"],
|
||||
acceptableQuestionStrategies: ["decision criterion", "constraint"],
|
||||
notes:
|
||||
"The first question should establish whether more engineering capacity is justified before compensation or implementation details.",
|
||||
graph: makeScenarioGraph({
|
||||
scenario: "Should we hire another engineer?",
|
||||
decisionNode: makeNode({
|
||||
id: "hire-decision",
|
||||
label: "Hiring another engineer decision",
|
||||
description: "Decision about increasing engineering capacity.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
value: "Deciding whether to hire another engineer",
|
||||
childIds: ["hire-success-criteria"],
|
||||
}),
|
||||
answeredContextUnknown: makeNode({
|
||||
id: "hire-delays-known",
|
||||
label: "Delivery delays established",
|
||||
description:
|
||||
"Need to confirm whether recent delivery delays are real because this context determines whether a capacity decision is even relevant.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value:
|
||||
"The roadmap is slipping because the current team cannot clear the queue.",
|
||||
childIds: ["hire-bottleneck"],
|
||||
}),
|
||||
foundationalUnknown: makeNode({
|
||||
id: "hire-success-criteria",
|
||||
label: "Hiring success threshold",
|
||||
description:
|
||||
"Need the success threshold because the hiring decision depends on what improvement would justify adding headcount.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "hire-decision",
|
||||
childIds: ["hire-bottleneck"],
|
||||
}),
|
||||
consequentialUnknown: makeNode({
|
||||
id: "hire-bottleneck",
|
||||
label: "Primary delivery bottleneck",
|
||||
description:
|
||||
"Need the main bottleneck because the team must know whether another engineer would relieve the limiting constraint.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
dependsOn: ["hire-success-criteria"],
|
||||
parentId: "hire-success-criteria",
|
||||
childIds: ["hire-salary"],
|
||||
}),
|
||||
downstreamLeaf: makeNode({
|
||||
id: "hire-salary",
|
||||
label: "Engineer salary budget",
|
||||
description:
|
||||
"Need the salary range because compensation planning comes after the hiring case is established.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["hire-bottleneck"],
|
||||
parentId: "hire-bottleneck",
|
||||
}),
|
||||
}),
|
||||
},
|
||||
{
|
||||
key: "replace-vans",
|
||||
scenario: "Should we replace the delivery vans?",
|
||||
decisionType: "asset replacement decision",
|
||||
acceptableFoundationalUnknownNodeIds: [
|
||||
"van-reliability-threshold",
|
||||
"van-service-constraint",
|
||||
],
|
||||
prohibitedFirstTopics: [
|
||||
"purchase price",
|
||||
"paint colour",
|
||||
"finance provider",
|
||||
],
|
||||
acceptableQuestionStrategies: ["decision criterion", "constraint"],
|
||||
notes:
|
||||
"The first question should establish whether the fleet is failing a threshold that justifies replacement.",
|
||||
graph: makeScenarioGraph({
|
||||
scenario: "Should we replace the delivery vans?",
|
||||
decisionNode: makeNode({
|
||||
id: "van-decision",
|
||||
label: "Replace delivery vans decision",
|
||||
description: "Decision about replacing the current delivery fleet.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
value: "Deciding whether to replace the delivery vans",
|
||||
childIds: ["van-reliability-threshold"],
|
||||
}),
|
||||
answeredContextUnknown: makeNode({
|
||||
id: "van-breakdowns-known",
|
||||
label: "Breakdown trend confirmed",
|
||||
description:
|
||||
"Need to confirm whether the recent rise in breakdowns is real because that context determines whether fleet replacement is relevant.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value:
|
||||
"Breakdowns and missed deliveries have increased over the last quarter.",
|
||||
childIds: ["van-service-constraint"],
|
||||
}),
|
||||
foundationalUnknown: makeNode({
|
||||
id: "van-reliability-threshold",
|
||||
label: "Replacement justification threshold",
|
||||
description:
|
||||
"Need the threshold because the replacement decision depends on what level of reliability loss is enough to justify replacing the fleet.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "van-decision",
|
||||
childIds: ["van-service-constraint"],
|
||||
}),
|
||||
consequentialUnknown: makeNode({
|
||||
id: "van-service-constraint",
|
||||
label: "Operational service constraint",
|
||||
description:
|
||||
"Need the limiting service constraint because the team must know how vehicle unreliability is affecting deliveries before comparing purchasing options.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
dependsOn: ["van-reliability-threshold"],
|
||||
parentId: "van-reliability-threshold",
|
||||
childIds: ["van-price"],
|
||||
}),
|
||||
downstreamLeaf: makeNode({
|
||||
id: "van-price",
|
||||
label: "Exact replacement purchase price",
|
||||
description:
|
||||
"Need the exact purchase price because financing analysis comes after replacement is justified.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["van-service-constraint"],
|
||||
parentId: "van-service-constraint",
|
||||
}),
|
||||
}),
|
||||
},
|
||||
{
|
||||
key: "launch-country",
|
||||
scenario: "Should we launch in another country?",
|
||||
decisionType: "market expansion decision",
|
||||
acceptableFoundationalUnknownNodeIds: [
|
||||
"country-customer",
|
||||
"country-value-threshold",
|
||||
],
|
||||
prohibitedFirstTopics: [
|
||||
"launch date",
|
||||
"office location",
|
||||
"advertising channel",
|
||||
],
|
||||
acceptableQuestionStrategies: ["actor/customer", "decision criterion"],
|
||||
notes:
|
||||
"The first question should clarify the customer or value case for expansion before rollout logistics.",
|
||||
graph: makeScenarioGraph({
|
||||
scenario: "Should we launch in another country?",
|
||||
decisionNode: makeNode({
|
||||
id: "country-decision",
|
||||
label: "Launch in another country decision",
|
||||
description: "Decision about entering a new national market.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
value: "Deciding whether to launch in another country",
|
||||
childIds: ["country-customer"],
|
||||
}),
|
||||
answeredContextUnknown: makeNode({
|
||||
id: "country-interest-known",
|
||||
label: "Inbound interest confirmed",
|
||||
description:
|
||||
"Need to confirm whether inbound interest from another country is real because that context determines whether expansion is relevant.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
value:
|
||||
"Prospective customers from another country are asking for access.",
|
||||
childIds: ["country-value-threshold"],
|
||||
}),
|
||||
foundationalUnknown: makeNode({
|
||||
id: "country-customer",
|
||||
label: "Relevant customer in the new country",
|
||||
description:
|
||||
"Need the relevant customer because the expansion decision depends on who experiences the problem or receives the value in that market.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "country-decision",
|
||||
childIds: ["country-value-threshold"],
|
||||
}),
|
||||
consequentialUnknown: makeNode({
|
||||
id: "country-value-threshold",
|
||||
label: "Expansion value threshold",
|
||||
description:
|
||||
"Need the value threshold because the team must know what evidence of demand or value would justify entering the new country.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
dependsOn: ["country-customer"],
|
||||
parentId: "country-customer",
|
||||
childIds: ["country-launch-date"],
|
||||
}),
|
||||
downstreamLeaf: makeNode({
|
||||
id: "country-launch-date",
|
||||
label: "Country launch date",
|
||||
description:
|
||||
"Need the launch date because rollout planning follows once the expansion case is established.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["country-value-threshold"],
|
||||
parentId: "country-value-threshold",
|
||||
}),
|
||||
}),
|
||||
},
|
||||
{
|
||||
key: "over-budget-project",
|
||||
scenario: "Should we continue a project that is over budget?",
|
||||
decisionType: "continuation decision",
|
||||
acceptableFoundationalUnknownNodeIds: [
|
||||
"project-benefit-threshold",
|
||||
"project-remaining-benefit",
|
||||
],
|
||||
prohibitedFirstTopics: ["sunk cost", "project logo", "final launch date"],
|
||||
acceptableQuestionStrategies: ["decision criterion", "objective"],
|
||||
notes:
|
||||
"The first question should establish remaining value or success threshold before sunk-cost framing or launch timing.",
|
||||
graph: makeScenarioGraph({
|
||||
scenario: "Should we continue a project that is over budget?",
|
||||
decisionNode: makeNode({
|
||||
id: "project-decision",
|
||||
label: "Continue over-budget project decision",
|
||||
description:
|
||||
"Decision about continuing a project that has exceeded budget.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
value: "Deciding whether to continue the over-budget project",
|
||||
childIds: ["project-benefit-threshold"],
|
||||
}),
|
||||
answeredContextUnknown: makeNode({
|
||||
id: "project-overrun-known",
|
||||
label: "Budget overrun confirmed",
|
||||
description:
|
||||
"Need to confirm whether the project is materially over budget because that context determines whether a continuation decision is relevant.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value: "The project has exceeded its approved budget by 35 percent.",
|
||||
childIds: ["project-remaining-benefit"],
|
||||
}),
|
||||
foundationalUnknown: makeNode({
|
||||
id: "project-benefit-threshold",
|
||||
label: "Continuation success threshold",
|
||||
description:
|
||||
"Need the threshold because the continuation decision depends on what remaining benefit would still justify completing the project.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "project-decision",
|
||||
childIds: ["project-remaining-benefit"],
|
||||
}),
|
||||
consequentialUnknown: makeNode({
|
||||
id: "project-remaining-benefit",
|
||||
label: "Remaining project benefit",
|
||||
description:
|
||||
"Need the remaining benefit because the team must know what value is still achievable before deciding whether to continue.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
dependsOn: ["project-benefit-threshold"],
|
||||
parentId: "project-benefit-threshold",
|
||||
childIds: ["project-launch-date"],
|
||||
}),
|
||||
downstreamLeaf: makeNode({
|
||||
id: "project-launch-date",
|
||||
label: "Final launch date",
|
||||
description:
|
||||
"Need the final launch date because scheduling details only matter after remaining value is established.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["project-remaining-benefit"],
|
||||
parentId: "project-remaining-benefit",
|
||||
}),
|
||||
}),
|
||||
},
|
||||
{
|
||||
key: "paid-support-tier",
|
||||
scenario: "Should we introduce a paid support tier?",
|
||||
decisionType: "commercial packaging decision",
|
||||
acceptableFoundationalUnknownNodeIds: [
|
||||
"support-customer",
|
||||
"support-value-threshold",
|
||||
],
|
||||
prohibitedFirstTopics: [
|
||||
"subscription price",
|
||||
"payment provider",
|
||||
"tier name",
|
||||
],
|
||||
acceptableQuestionStrategies: ["actor/customer", "decision criterion"],
|
||||
notes:
|
||||
"The first question should establish who values paid support or what outcome would justify offering it before pricing details.",
|
||||
graph: makeScenarioGraph({
|
||||
scenario: "Should we introduce a paid support tier?",
|
||||
decisionNode: makeNode({
|
||||
id: "support-decision",
|
||||
label: "Introduce paid support tier decision",
|
||||
description: "Decision about adding a paid support offering.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
value: "Deciding whether to introduce a paid support tier",
|
||||
childIds: ["support-customer"],
|
||||
}),
|
||||
answeredContextUnknown: makeNode({
|
||||
id: "support-requests-known",
|
||||
label: "Support request pattern confirmed",
|
||||
description:
|
||||
"Need to confirm whether repeated requests for faster support responses are real because that context determines whether a paid tier is relevant.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value:
|
||||
"Some users are asking for guaranteed response times and escalation help.",
|
||||
childIds: ["support-value-threshold"],
|
||||
}),
|
||||
foundationalUnknown: makeNode({
|
||||
id: "support-customer",
|
||||
label: "Customer willing to pay for support",
|
||||
description:
|
||||
"Need the customer because the decision depends on who experiences enough support pain or receives enough value to pay for a support tier.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "support-decision",
|
||||
childIds: ["support-value-threshold"],
|
||||
}),
|
||||
consequentialUnknown: makeNode({
|
||||
id: "support-value-threshold",
|
||||
label: "Paid support value threshold",
|
||||
description:
|
||||
"Need the value threshold because the team must know what outcome would justify introducing paid support before setting packaging details.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
dependsOn: ["support-customer"],
|
||||
parentId: "support-customer",
|
||||
childIds: ["support-price"],
|
||||
}),
|
||||
downstreamLeaf: makeNode({
|
||||
id: "support-price",
|
||||
label: "Support subscription price",
|
||||
description:
|
||||
"Need the subscription price because pricing and payment setup come after the support value case is established.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["support-value-threshold"],
|
||||
parentId: "support-value-threshold",
|
||||
}),
|
||||
}),
|
||||
},
|
||||
];
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,489 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import {
|
||||
buildInitialGraph,
|
||||
buildMinimalGraph,
|
||||
describeGraph,
|
||||
} from "@/lib/graph/builder.js";
|
||||
import {
|
||||
makeNode,
|
||||
situationEdgeSchema,
|
||||
situationGraphSchema,
|
||||
situationNodeSchema,
|
||||
} from "@/lib/graph/schema.js";
|
||||
|
||||
// ── Helper: create a v0.3-style reconstruction fixture ───────────
|
||||
|
||||
function makeReconstructionFixture() {
|
||||
return {
|
||||
summary: "Company X reports revenue growth but increasing complaints",
|
||||
actors: [
|
||||
{ id: "actor-1", description: "Customer Base", confidence: "high" },
|
||||
{
|
||||
id: "actor-2",
|
||||
description: "Product Engineering Team",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
systemsOrObjects: [
|
||||
{ id: "sys-1", description: "Production Line A", confidence: "high" },
|
||||
{
|
||||
id: "sys-2",
|
||||
description: "Quality Control System",
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
expectedStates: [],
|
||||
observedStates: [
|
||||
{
|
||||
id: "obs-1",
|
||||
description: "Revenue up 15% year-over-year",
|
||||
confidence: "high",
|
||||
},
|
||||
{
|
||||
id: "obs-2",
|
||||
description: "Customer complaints up 40% year-over-year",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
differences: [
|
||||
{
|
||||
id: "diff-1",
|
||||
description: "Complaint count grew faster than revenue",
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [
|
||||
{
|
||||
id: "con-1",
|
||||
description: "Revenue growth vs complaint growth inconsistency",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
importantUnknowns: [
|
||||
{
|
||||
id: "unk-1",
|
||||
description: "Denominator for complaint rate (customers served)",
|
||||
confidence: "high",
|
||||
},
|
||||
{
|
||||
id: "unk-2",
|
||||
description: "Root cause of complaint increase",
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
plausibleInterpretations: [],
|
||||
};
|
||||
}
|
||||
|
||||
function makeEvidenceFixture() {
|
||||
return [
|
||||
{
|
||||
id: "ev-1",
|
||||
description: "Annual report data",
|
||||
evidenceType: "direct_observation",
|
||||
confidence: "high",
|
||||
importance: "critical",
|
||||
},
|
||||
{
|
||||
id: "ev-2",
|
||||
description: "Customer survey results",
|
||||
evidenceType: "reported_statement",
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
describe("buildInitialGraph", () => {
|
||||
it("builds nodes from reconstruction data", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: makeEvidenceFixture(),
|
||||
});
|
||||
|
||||
expect(result.nodes.length).toBeGreaterThan(0);
|
||||
expect(result.edges.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("creates a summary node", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const summaryNode = result.nodes.find((n) => n.kind === "state");
|
||||
expect(summaryNode).toBeDefined();
|
||||
expect(summaryNode.label).toBe(
|
||||
"Company X reports revenue growth but increasing complaints",
|
||||
);
|
||||
});
|
||||
|
||||
it("creates observation nodes from observedStates", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const observations = result.nodes.filter((n) => n.kind === "observation");
|
||||
expect(observations.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("creates unknown nodes from importantUnknowns", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const unknowns = result.nodes.filter((n) => n.kind === "unknown");
|
||||
expect(unknowns.length).toBe(2); // unk-1 and unk-2
|
||||
});
|
||||
|
||||
it("creates metric nodes from systemsOrObjects", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const metrics = result.nodes.filter((n) => n.kind === "metric");
|
||||
expect(metrics.length).toBe(2); // sys-1 and sys-2
|
||||
});
|
||||
|
||||
it("creates actor nodes as observations", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const actors = result.nodes.filter(
|
||||
(n) =>
|
||||
n.label.includes("Customer Base") ||
|
||||
n.label.includes("Product Engineering"),
|
||||
);
|
||||
expect(actors.length).toBe(2);
|
||||
});
|
||||
|
||||
it("creates difference nodes", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const differenceNode = result.nodes.find((n) =>
|
||||
n.label.includes("Complaint count grew faster than revenue"),
|
||||
);
|
||||
|
||||
expect(differenceNode).toBeDefined();
|
||||
expect(differenceNode.kind).toBe("relationship");
|
||||
});
|
||||
|
||||
it("creates contradiction nodes", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const contradictionNode = result.nodes.find((n) =>
|
||||
n.label.includes("inconsistency"),
|
||||
);
|
||||
|
||||
expect(contradictionNode).toBeDefined();
|
||||
expect(contradictionNode.kind).toBe("relationship");
|
||||
});
|
||||
|
||||
it("creates edges linking observations to summary", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const supportEdges = result.edges.filter(
|
||||
(e) => e.relationship === "supports",
|
||||
);
|
||||
expect(supportEdges.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("creates edges linking unknowns to summary as depends_on", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const depEdges = result.edges.filter(
|
||||
(e) => e.relationship === "depends_on",
|
||||
);
|
||||
expect(depEdges.length).toBe(2); // Two unknown nodes
|
||||
});
|
||||
|
||||
it("handles empty observedStates gracefully", () => {
|
||||
const reconstruction = {
|
||||
...makeReconstructionFixture(),
|
||||
observedStates: [],
|
||||
};
|
||||
const result = buildInitialGraph({ reconstruction, evidence: [] });
|
||||
|
||||
expect(result.nodes.length).toBeGreaterThan(0); // Summary + actors + systems still created
|
||||
});
|
||||
|
||||
it("handles missing reconstruction fields gracefully", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: { summary: "Minimal" },
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
expect(result.nodes.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("handles null/undefined reconstruction", () => {
|
||||
const result = buildInitialGraph({ reconstruction: null, evidence: [] });
|
||||
expect(result.nodes.length).toBe(0);
|
||||
expect(result.edges.length).toBe(0);
|
||||
});
|
||||
|
||||
it("handles missing evidence array", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
});
|
||||
|
||||
expect(result.nodes.length).toBeGreaterThan(0);
|
||||
expect(result.edges.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("generates deterministic node IDs for same labels", () => {
|
||||
const r1 = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
const r2 = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
const ids1 = r1.nodes.map((n) => n.id).sort();
|
||||
const ids2 = r2.nodes.map((n) => n.id).sort();
|
||||
expect(ids1).toEqual(ids2);
|
||||
});
|
||||
|
||||
it("produces valid schema output (no parse errors)", () => {
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: makeReconstructionFixture(),
|
||||
evidence: makeEvidenceFixture(),
|
||||
});
|
||||
|
||||
for (const node of result.nodes) {
|
||||
const parsed = situationNodeSchema.safeParse(node);
|
||||
if (!parsed.success) {
|
||||
console.error(`Invalid node: ${node.id}`, node, parsed.error.message);
|
||||
}
|
||||
expect(parsed.success).toBe(true);
|
||||
}
|
||||
|
||||
for (const edge of result.edges) {
|
||||
const parsed = situationEdgeSchema.safeParse(edge);
|
||||
if (!parsed.success) {
|
||||
console.error(`Invalid edge: ${edge.id}`, edge, parsed.error.message);
|
||||
}
|
||||
expect(parsed.success).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("creates edges for knownTransitions as transition nodes", () => {
|
||||
const reconstruction = {
|
||||
...makeReconstructionFixture(),
|
||||
knownTransitions: [
|
||||
{
|
||||
id: "trans-1",
|
||||
description: "Product shipped v2.0",
|
||||
entity: "Product",
|
||||
previousState: "v1.x",
|
||||
currentState: "v2.0",
|
||||
explanationStatus: "confirmed",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = buildInitialGraph({ reconstruction, evidence: [] });
|
||||
const transitions = result.nodes.filter((n) => n.kind === "transition");
|
||||
expect(transitions.length).toBe(1);
|
||||
});
|
||||
|
||||
it("creates nodes for unexplainedTransitions", () => {
|
||||
const reconstruction = {
|
||||
...makeReconstructionFixture(),
|
||||
unexplainedTransitions: [
|
||||
{
|
||||
id: "ut-1",
|
||||
description: "Support wait time increased",
|
||||
entity: "Support",
|
||||
previousState: "2hr",
|
||||
currentState: "8hr",
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = buildInitialGraph({ reconstruction, evidence: [] });
|
||||
expect(result.nodes.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("creates nodes for plausibleInterpretations as assumptions", () => {
|
||||
const reconstruction = {
|
||||
...makeReconstructionFixture(),
|
||||
plausibleInterpretations: [
|
||||
{
|
||||
id: "interp-1",
|
||||
description: "Quality degradation hypothesis",
|
||||
supportingEvidenceIds: ["ev-2"],
|
||||
assumptionsRequired: [],
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = buildInitialGraph({ reconstruction, evidence: [] });
|
||||
const assumptions = result.nodes.filter((n) => n.kind === "assumption");
|
||||
expect(assumptions.length).toBe(1);
|
||||
});
|
||||
|
||||
it("links evidence to observation nodes", () => {
|
||||
const reconstruction = makeReconstructionFixture();
|
||||
const evidence = [{ id: "ev-1", description: "Test evidence" }];
|
||||
|
||||
// Add a mapping from observed states to evidence IDs would require modification
|
||||
// For now, just verify the nodes have empty evidenceIds (as per current implementation)
|
||||
const result = buildInitialGraph({ reconstruction, evidence });
|
||||
for (const node of result.nodes) {
|
||||
expect(Array.isArray(node.evidenceIds)).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("handles very large reconstruction without errors", () => {
|
||||
const actors = Array.from({ length: 20 }, (_, i) => ({
|
||||
id: `actor-${i}`,
|
||||
description: `Actor ${i}`,
|
||||
confidence: "high",
|
||||
}));
|
||||
|
||||
const result = buildInitialGraph({
|
||||
reconstruction: { ...makeReconstructionFixture(), actors },
|
||||
evidence: [],
|
||||
});
|
||||
|
||||
expect(result.nodes.length).toBeGreaterThan(10);
|
||||
});
|
||||
|
||||
it("handles transition with confirmed explanation", () => {
|
||||
const reconstruction = {
|
||||
...makeReconstructionFixture(),
|
||||
knownTransitions: [
|
||||
{
|
||||
id: "t-confirmed",
|
||||
description: "Confirmed event",
|
||||
entity: "E1",
|
||||
previousState: "s1",
|
||||
currentState: "s2",
|
||||
explanationStatus: "confirmed",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = buildInitialGraph({ reconstruction, evidence: [] });
|
||||
const confirmedTransitions = result.nodes.filter(
|
||||
(n) => n.kind === "transition" && n.status === "known",
|
||||
);
|
||||
expect(confirmedTransitions.length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildMinimalGraph", () => {
|
||||
it("creates a single node with scenario text as label", () => {
|
||||
const graph = buildMinimalGraph(
|
||||
"This is a test scenario for minimal graph creation",
|
||||
);
|
||||
expect(graph.nodes.length).toBe(1);
|
||||
expect(graph.edges.length).toBe(0);
|
||||
});
|
||||
|
||||
it("truncates label to 80 chars", () => {
|
||||
const longScenario = "a".repeat(200);
|
||||
const graph = buildMinimalGraph(longScenario);
|
||||
expect(graph.nodes[0].label.length).toBeLessThanOrEqual(80);
|
||||
});
|
||||
|
||||
it("creates provisional state node", () => {
|
||||
const graph = buildMinimalGraph("Test scenario");
|
||||
expect(graph.nodes[0].kind).toBe("state");
|
||||
expect(graph.nodes[0].status).toBe("provisional");
|
||||
expect(graph.nodes[0].confidence).toBe("low");
|
||||
});
|
||||
|
||||
it("uses first 200 chars of scenario for description", () => {
|
||||
const graph = buildMinimalGraph(
|
||||
"This is a test scenario for minimal graph creation",
|
||||
);
|
||||
expect(graph.nodes[0].description).toContain("Initial situation from:");
|
||||
});
|
||||
|
||||
it("creates deterministic ID via situationNodeSchema.parse", () => {
|
||||
const graph = buildMinimalGraph("Test scenario");
|
||||
// Node has explicit id "n0" from the builder, not makeNodeId
|
||||
expect(graph.nodes[0].id).toBe("n0");
|
||||
});
|
||||
|
||||
it("creates minimal valid structure", () => {
|
||||
const graph = buildMinimalGraph("Test");
|
||||
expect(graph.nodes).toHaveLength(1);
|
||||
expect(graph.edges).toHaveLength(0);
|
||||
expect(graph.nodes[0].evidenceIds).toEqual([]);
|
||||
expect(graph.nodes[0].dependsOn).toEqual([]);
|
||||
expect(graph.nodes[0].affects).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("describeGraph", () => {
|
||||
it("returns summary string with node count by kind", () => {
|
||||
const graph = buildMinimalGraph("Test");
|
||||
const description = describeGraph(graph);
|
||||
|
||||
expect(description).toContain("Nodes:");
|
||||
expect(description).toContain("Edges:");
|
||||
expect(description).toContain("Unknowns:");
|
||||
});
|
||||
|
||||
it("shows correct edge count", () => {
|
||||
const graph = buildMinimalGraph("Test");
|
||||
const description = describeGraph(graph);
|
||||
|
||||
expect(description).toContain("Edges: 0 total");
|
||||
});
|
||||
|
||||
it("counts unresolved unknowns", () => {
|
||||
const n1 = makeNode({
|
||||
id: "n-unk",
|
||||
label: "Unknown",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
});
|
||||
const graph = situationGraphSchema.parse({
|
||||
centralStatement: "Test",
|
||||
nodes: [n1],
|
||||
edges: [],
|
||||
activeUnknownNodeId: n1.id,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary: "Test",
|
||||
});
|
||||
|
||||
const description = describeGraph(graph);
|
||||
expect(description).toContain("1"); // One unresolved unknown
|
||||
});
|
||||
|
||||
it("groups nodes by kind in output", () => {
|
||||
const graph = buildMinimalGraph("Test");
|
||||
const description = describeGraph(graph);
|
||||
|
||||
expect(description).toContain("1 state");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,792 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { validateGraphReferences } from "@/lib/graph/utils.js";
|
||||
import { makeGraph, makeNode } from "@/lib/graph/schema.js";
|
||||
|
||||
const mockAnalyseScenario = vi.fn();
|
||||
const MOCK_CONFIG = { OLLAMA_MODEL: "configured" };
|
||||
|
||||
vi.mock("@/lib/analysis.js", () => ({
|
||||
analyseScenario: (...args) => mockAnalyseScenario(...args),
|
||||
}));
|
||||
|
||||
function makeAnalysisResult(overrides = {}) {
|
||||
return {
|
||||
success: true,
|
||||
validationStatus: "valid",
|
||||
modelName: "configured-model",
|
||||
responseDurationMs: 321,
|
||||
rawResponse: undefined,
|
||||
promptVersion: "v0.3",
|
||||
reconstruction: {
|
||||
summary: "Revenue and complaints diverge",
|
||||
actors: [],
|
||||
systemsOrObjects: [],
|
||||
expectedStates: [],
|
||||
observedStates: [
|
||||
{
|
||||
id: "obs-1",
|
||||
label: "Revenue up",
|
||||
description: "Revenue up 15%",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
differences: [],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [],
|
||||
importantUnknowns: [
|
||||
{
|
||||
id: "unk-1",
|
||||
label: "Complaint rate denominator",
|
||||
description: "Need the denominator for complaint rate",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
plausibleInterpretations: [],
|
||||
},
|
||||
evidence: [],
|
||||
nextQuestion: {
|
||||
id: "q-1",
|
||||
question: "What denominator is being used for the complaint rate?",
|
||||
},
|
||||
compatibilityApplied: false,
|
||||
compatibilityChanges: [],
|
||||
compatibilityWarnings: [],
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function makeUpdateGraph() {
|
||||
const unknown = makeNode({
|
||||
id: "n-unknown",
|
||||
label: "Complaint rate denominator",
|
||||
description: "Need the denominator for the complaint rate",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
});
|
||||
const observation = makeNode({
|
||||
id: "n-observation",
|
||||
label: "Complaint count rose",
|
||||
description: "Complaint count rose faster than output",
|
||||
kind: "observation",
|
||||
status: "supported",
|
||||
confidence: "high",
|
||||
});
|
||||
|
||||
return makeGraph({
|
||||
centralStatement:
|
||||
"Complaint counts increased while production also increased.",
|
||||
nodes: [unknown, observation],
|
||||
edges: [],
|
||||
activeUnknownNodeId: unknown.id,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary: "Nodes: 1 unknown, 1 observation | Edges: 0 total",
|
||||
});
|
||||
}
|
||||
|
||||
function makeUpdateRequest(overrides = {}) {
|
||||
return {
|
||||
situationGraph: makeUpdateGraph(),
|
||||
previousQuestion: "What denominator is being used for the complaint rate?",
|
||||
answer:
|
||||
"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.",
|
||||
promptVersion: "v0.4",
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function makeProposal(overrides = {}) {
|
||||
return {
|
||||
addedNodes: [],
|
||||
updatedNodes: [
|
||||
{
|
||||
nodeId: "n-unknown",
|
||||
previousStatus: "unknown",
|
||||
newStatus: "resolved",
|
||||
previousValue: null,
|
||||
newValue: "1.9 complaints per 100 units",
|
||||
reason: "The answer directly provides the normalized rate.",
|
||||
},
|
||||
],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: ["n-unknown"],
|
||||
affectedNodeIds: [],
|
||||
selectedQuestion: null,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("lib/graph/orchestrator startCase", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("passes a valid request through to analyseScenario", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(makeAnalysisResult());
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({
|
||||
scenario: "Revenue increased while complaint counts rose faster.",
|
||||
promptVersion: "v0.3",
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(mockAnalyseScenario).toHaveBeenCalledWith(
|
||||
"Revenue increased while complaint counts rose faster.",
|
||||
{ promptVersion: "v0.3" },
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects invalid request input without throwing", async () => {
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "" });
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
error: "Invalid start-case request",
|
||||
statusCode: 400,
|
||||
});
|
||||
expect(result.validationErrors).toBeInstanceOf(Array);
|
||||
expect(mockAnalyseScenario).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("builds a valid graph on successful analysis", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(makeAnalysisResult());
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.situationGraph.centralStatement).toBe("Scenario text");
|
||||
expect(result.situationGraph.currentSummary).toContain("Nodes:");
|
||||
expect(result.diagnostics).toMatchObject({
|
||||
validationStatus: "valid",
|
||||
modelName: "configured-model",
|
||||
graphReferenceValidation: { valid: true, errors: [] },
|
||||
});
|
||||
});
|
||||
|
||||
it("applies active unknown selection to the graph", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(makeAnalysisResult());
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.situationGraph.activeUnknownNodeId).toBeTruthy();
|
||||
});
|
||||
|
||||
it("returns structured failure when graph reference validation fails", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(makeAnalysisResult());
|
||||
const utils = await import("@/lib/graph/utils.js");
|
||||
const validateSpy = vi
|
||||
.spyOn(utils, "validateGraphReferences")
|
||||
.mockReturnValue({
|
||||
valid: false,
|
||||
errors: ['Edge references non-existent toNodeId "missing"'],
|
||||
});
|
||||
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
error: "Situation graph reference validation failed",
|
||||
validationErrors: ['Edge references non-existent toNodeId "missing"'],
|
||||
statusCode: 500,
|
||||
});
|
||||
expect(result.diagnostics.graphReferenceValidation.valid).toBe(false);
|
||||
validateSpy.mockRestore();
|
||||
});
|
||||
|
||||
it("preserves analysis/provider failure details", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue({
|
||||
success: false,
|
||||
error: "Provider unavailable",
|
||||
errors: ["socket hang up"],
|
||||
rawResponse: null,
|
||||
modelName: "configured-model",
|
||||
responseDurationMs: 99,
|
||||
promptVersion: "v0.3",
|
||||
validationStatus: "invalid",
|
||||
statusCode: 502,
|
||||
});
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
error: "Provider unavailable",
|
||||
analysisErrors: ["socket hang up"],
|
||||
statusCode: 502,
|
||||
});
|
||||
});
|
||||
|
||||
it("returns null selectedQuestion when analysis has no nextQuestion", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(
|
||||
makeAnalysisResult({ nextQuestion: undefined }),
|
||||
);
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.selectedQuestion).toBeNull();
|
||||
});
|
||||
|
||||
it("includes compatibility diagnostics when provided by analysis", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(
|
||||
makeAnalysisResult({
|
||||
compatibilityApplied: true,
|
||||
compatibilityChanges: [
|
||||
{
|
||||
path: ["evidence", 0, "source"],
|
||||
change: "Converted null source to undefined",
|
||||
},
|
||||
],
|
||||
compatibilityWarnings: [
|
||||
"Applied deterministic reconstruction compatibility normalisation",
|
||||
],
|
||||
}),
|
||||
);
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result.diagnostics.compatibilityApplied).toBe(true);
|
||||
expect(result.diagnostics.compatibilityChanges).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("produces a validated update proposal for a valid request", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: true,
|
||||
stage: "proposal_ready",
|
||||
proposal: makeProposal(),
|
||||
diagnostics: {
|
||||
promptVersion: "v0.4",
|
||||
modelName: "configured",
|
||||
nodeCount: 2,
|
||||
edgeCount: 0,
|
||||
validationStatus: "valid",
|
||||
},
|
||||
});
|
||||
expect(provider.generateReconstruction).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("valid request reaches prompt builder", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const buildGraphUpdatePrompt = vi.fn().mockReturnValue("PROMPT");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
|
||||
const request = makeUpdateRequest();
|
||||
const result = await updateCase(request, {
|
||||
buildGraphUpdatePrompt,
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(buildGraphUpdatePrompt).toHaveBeenCalledWith({
|
||||
situationGraph: request.situationGraph,
|
||||
previousQuestion: request.previousQuestion,
|
||||
answer: request.answer,
|
||||
promptVersion: request.promptVersion,
|
||||
});
|
||||
expect(provider.generateReconstruction).toHaveBeenCalledWith(
|
||||
"PROMPT",
|
||||
"configured",
|
||||
);
|
||||
});
|
||||
|
||||
it("invalid request prevents provider call", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn(),
|
||||
};
|
||||
|
||||
const result = await updateCase(
|
||||
{ previousQuestion: "Q?", answer: "A" },
|
||||
{
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
},
|
||||
);
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid update-case request",
|
||||
statusCode: 400,
|
||||
});
|
||||
expect(result.validationErrors).toBeInstanceOf(Array);
|
||||
expect(provider.generateReconstruction).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("invalid graph prevents provider call", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn(),
|
||||
};
|
||||
|
||||
const graph = makeUpdateGraph();
|
||||
graph.nodes[0].dependsOn.push("missing-node");
|
||||
|
||||
const result = await updateCase(
|
||||
makeUpdateRequest({ situationGraph: graph }),
|
||||
{
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
},
|
||||
);
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
stage: "graph_validation",
|
||||
error: "Invalid situation graph",
|
||||
statusCode: 400,
|
||||
});
|
||||
expect(result.graphValidationErrors).toEqual(
|
||||
expect.arrayContaining([
|
||||
expect.stringContaining('depends on "missing-node"'),
|
||||
]),
|
||||
);
|
||||
expect(provider.generateReconstruction).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("prompt includes previous question and answer", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
const request = makeUpdateRequest();
|
||||
|
||||
await updateCase(request, {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
const prompt = provider.generateReconstruction.mock.calls[0][0];
|
||||
expect(prompt).toContain(request.previousQuestion);
|
||||
expect(prompt).toContain(request.answer);
|
||||
});
|
||||
|
||||
it("returns proposal validation failure for malformed JSON", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue("{not json"),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
stage: "proposal_validation",
|
||||
error: "Invalid graph update proposal",
|
||||
diagnostics: {
|
||||
promptVersion: "v0.4",
|
||||
modelName: "configured",
|
||||
},
|
||||
statusCode: 502,
|
||||
});
|
||||
expect(result.proposalErrors).toBeInstanceOf(Array);
|
||||
});
|
||||
|
||||
it("returns structured errors for schema-invalid proposal", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
updatedNodes: [{ nodeId: "n-unknown" }],
|
||||
}),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(false);
|
||||
expect(result.stage).toBe("proposal_validation");
|
||||
expect(result.proposalErrors).toEqual(
|
||||
expect.arrayContaining([
|
||||
expect.objectContaining({
|
||||
path: expect.any(Array),
|
||||
message: expect.any(String),
|
||||
}),
|
||||
]),
|
||||
);
|
||||
});
|
||||
|
||||
it("includes parser normalisations in diagnostics", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
updatedNodes: [],
|
||||
}),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.diagnostics.normalisationsApplied).toEqual(
|
||||
expect.arrayContaining([
|
||||
expect.objectContaining({
|
||||
change: "Filled missing optional array with []",
|
||||
}),
|
||||
]),
|
||||
);
|
||||
});
|
||||
|
||||
it("returns structured provider-stage failure", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi
|
||||
.fn()
|
||||
.mockRejectedValue(new Error("provider offline")),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: false,
|
||||
stage: "provider",
|
||||
error: "Graph update proposal generation failed",
|
||||
providerErrors: ["provider offline"],
|
||||
statusCode: 502,
|
||||
});
|
||||
});
|
||||
|
||||
it("does not mutate the input graph", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
const request = makeUpdateRequest();
|
||||
const originalGraph = JSON.parse(JSON.stringify(request.situationGraph));
|
||||
|
||||
await updateCase(request, {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(request.situationGraph).toEqual(originalGraph);
|
||||
});
|
||||
|
||||
it("does not call applyGraphUpdate", async () => {
|
||||
const utils = await import("@/lib/graph/utils.js");
|
||||
const applySpy = vi.spyOn(utils, "applyGraphUpdate");
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
|
||||
await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(applySpy).not.toHaveBeenCalled();
|
||||
applySpy.mockRestore();
|
||||
});
|
||||
|
||||
it("does not invent a next question outside the proposal", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
});
|
||||
|
||||
expect(result.selectedQuestion).toBeUndefined();
|
||||
expect(result.nextQuestion).toBeUndefined();
|
||||
expect(result.proposal.nextQuestion).toBeUndefined();
|
||||
});
|
||||
|
||||
it("returns selectedQuestion from applied update proposal", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(
|
||||
makeProposal({
|
||||
addedNodes: [
|
||||
makeNode({
|
||||
id: "n-build-decision",
|
||||
label: "Build Confidence Engine decision",
|
||||
description: "Decision introduced by the answer.",
|
||||
kind: "state",
|
||||
status: "supported",
|
||||
confidence: "medium",
|
||||
}),
|
||||
makeNode({
|
||||
id: "n-commercial-value",
|
||||
label: "Commercial value definition",
|
||||
description:
|
||||
"Need a concrete definition because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
}),
|
||||
],
|
||||
addedEdges: [
|
||||
{
|
||||
id: "e-build-commercial-value",
|
||||
fromNodeId: "n-build-decision",
|
||||
toNodeId: "n-commercial-value",
|
||||
relationship: "depends_on",
|
||||
confidence: "medium",
|
||||
description:
|
||||
"The decision depends on commercial value definition.",
|
||||
},
|
||||
],
|
||||
selectedQuestion: {
|
||||
nodeId: "n-commercial-value",
|
||||
question:
|
||||
"How should commercial value be defined for this decision?",
|
||||
reason: "Consequential unresolved uncertainty remains.",
|
||||
},
|
||||
}),
|
||||
),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
applyProposal: true,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.selectedQuestion?.nodeId).toBe("n-commercial-value");
|
||||
expect(result.newActiveUnknownNodeId).toBe("n-commercial-value");
|
||||
expect(result.selectedQuestion?.question).not.toBe(
|
||||
"How should commercial value be defined for this decision?",
|
||||
);
|
||||
expect(result.selectedQuestion?.question.toLowerCase()).not.toContain(
|
||||
"how should uncertainty regarding",
|
||||
);
|
||||
});
|
||||
|
||||
it("deterministically prioritises customer value over pricing follow-up", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(
|
||||
makeProposal({
|
||||
addedNodes: [
|
||||
makeNode({
|
||||
id: "n-value",
|
||||
label: "Customer value",
|
||||
description:
|
||||
"Need customer value because purchase decisions depend on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
}),
|
||||
makeNode({
|
||||
id: "n-price",
|
||||
label: "Target price point",
|
||||
description:
|
||||
"Need a target price point because revenue assumptions depend on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
dependsOn: ["n-value"],
|
||||
}),
|
||||
makeNode({
|
||||
id: "n-decision",
|
||||
label: "Build Confidence Engine decision",
|
||||
description: "Decision introduced by the answer.",
|
||||
kind: "state",
|
||||
status: "supported",
|
||||
confidence: "medium",
|
||||
}),
|
||||
],
|
||||
addedEdges: [
|
||||
{
|
||||
id: "e-decision-value",
|
||||
fromNodeId: "n-decision",
|
||||
toNodeId: "n-value",
|
||||
relationship: "depends_on",
|
||||
confidence: "medium",
|
||||
description: "The decision depends on customer value.",
|
||||
},
|
||||
{
|
||||
id: "e-value-price",
|
||||
fromNodeId: "n-value",
|
||||
toNodeId: "n-price",
|
||||
relationship: "depends_on",
|
||||
confidence: "medium",
|
||||
description: "Pricing depends on customer value.",
|
||||
},
|
||||
{
|
||||
id: "e-decision-price",
|
||||
fromNodeId: "n-decision",
|
||||
toNodeId: "n-price",
|
||||
relationship: "depends_on",
|
||||
confidence: "low",
|
||||
description: "The decision references pricing assumptions.",
|
||||
},
|
||||
],
|
||||
selectedQuestion: {
|
||||
nodeId: "n-price",
|
||||
question: "What is the price point?",
|
||||
reason: "Model chose pricing.",
|
||||
},
|
||||
}),
|
||||
),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
applyProposal: true,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.selectedQuestion?.nodeId).toBe("n-value");
|
||||
});
|
||||
|
||||
it("defaults to proposal-only mode", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const applyValidatedProposal = vi.fn();
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue(makeProposal()),
|
||||
};
|
||||
|
||||
const result = await updateCase(makeUpdateRequest(), {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
applyValidatedProposal,
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.stage).toBe("proposal_ready");
|
||||
expect(applyValidatedProposal).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("applies the proposal only when explicitly enabled", async () => {
|
||||
const { updateCase } = await import("@/lib/graph/orchestrator.js");
|
||||
const request = makeUpdateRequest({
|
||||
situationGraph: makeGraph({
|
||||
centralStatement:
|
||||
"Complaint counts increased while production also increased.",
|
||||
nodes: [
|
||||
makeNode({
|
||||
id: "n-rate",
|
||||
label: "Complaint rate",
|
||||
description: "Need complaint rate",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
affects: ["n-conclusion"],
|
||||
}),
|
||||
makeNode({
|
||||
id: "n-other-unknown",
|
||||
label: "Other unknown",
|
||||
description: "Another unresolved unknown",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
}),
|
||||
makeNode({
|
||||
id: "n-conclusion",
|
||||
label: "Quality deterioration",
|
||||
description: "Quality conclusion",
|
||||
kind: "conclusion",
|
||||
status: "supported",
|
||||
confidence: "medium",
|
||||
dependsOn: ["n-rate"],
|
||||
}),
|
||||
],
|
||||
edges: [],
|
||||
activeUnknownNodeId: "n-rate",
|
||||
resolvedNodeIds: [],
|
||||
currentSummary: "Initial summary",
|
||||
}),
|
||||
});
|
||||
const provider = {
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
addedNodes: [],
|
||||
updatedNodes: [
|
||||
{
|
||||
nodeId: "n-rate",
|
||||
previousStatus: "unknown",
|
||||
newStatus: "resolved",
|
||||
previousValue: "2.0 complaints per 100 units",
|
||||
newValue: "1.9 complaints per 100 units",
|
||||
reason: "The answer provides the updated rate.",
|
||||
},
|
||||
{
|
||||
nodeId: "n-conclusion",
|
||||
previousStatus: "supported",
|
||||
newStatus: "weakened",
|
||||
previousValue: null,
|
||||
newValue: null,
|
||||
reason: "The updated rate weakens the conclusion.",
|
||||
},
|
||||
],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: ["n-rate"],
|
||||
affectedNodeIds: ["n-conclusion"],
|
||||
}),
|
||||
};
|
||||
|
||||
const result = await updateCase(request, {
|
||||
provider,
|
||||
config: MOCK_CONFIG,
|
||||
applyProposal: true,
|
||||
});
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: true,
|
||||
stage: "update_applied",
|
||||
affectedNodeIds: expect.arrayContaining(["n-rate", "n-conclusion"]),
|
||||
resolvedUnknownNodeIds: ["n-rate"],
|
||||
previousActiveUnknownNodeId: "n-rate",
|
||||
newActiveUnknownNodeId: "n-other-unknown",
|
||||
});
|
||||
expect(validateGraphReferences(result.updatedSituationGraph)).toEqual({
|
||||
valid: true,
|
||||
errors: [],
|
||||
});
|
||||
});
|
||||
|
||||
it("startCase behaviour remains unchanged", async () => {
|
||||
mockAnalyseScenario.mockResolvedValue(makeAnalysisResult());
|
||||
const { startCase } = await import("@/lib/graph/orchestrator.js");
|
||||
|
||||
const result = await startCase({ scenario: "Scenario text" });
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.selectedQuestion).toEqual({
|
||||
id: "q-1",
|
||||
question: "What denominator is being used for the complaint rate?",
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,114 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { buildGraphUpdatePrompt } from "@/lib/graph/prompt-builder.js";
|
||||
import { makeEdge, makeGraph, makeNode } from "@/lib/graph/schema.js";
|
||||
|
||||
function makeContext() {
|
||||
const unknown = makeNode({
|
||||
id: "n-unknown",
|
||||
label: "Complaint rate denominator",
|
||||
description: "Need the denominator to compare complaint rates",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
});
|
||||
const observation = makeNode({
|
||||
id: "n-obs",
|
||||
label: "Complaints up 35%",
|
||||
description: "Complaints increased by 35%",
|
||||
kind: "observation",
|
||||
status: "supported",
|
||||
confidence: "high",
|
||||
});
|
||||
|
||||
return {
|
||||
situationGraph: makeGraph({
|
||||
centralStatement: "Complaints increased while production increased.",
|
||||
nodes: [unknown, observation],
|
||||
edges: [
|
||||
makeEdge({
|
||||
id: "e1",
|
||||
fromNodeId: observation.id,
|
||||
toNodeId: unknown.id,
|
||||
relationship: "supports",
|
||||
confidence: "high",
|
||||
description: "Observation informs the unknown",
|
||||
}),
|
||||
],
|
||||
activeUnknownNodeId: unknown.id,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary:
|
||||
"Nodes: 1 observation, 1 unknown | Edges: 1 total | Unknowns: 1 unresolved",
|
||||
}),
|
||||
previousQuestion: "What denominator is being used for the complaint rate?",
|
||||
answer:
|
||||
"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.",
|
||||
};
|
||||
}
|
||||
|
||||
describe("buildGraphUpdatePrompt", () => {
|
||||
it("includes the current graph", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain(
|
||||
"Complaints increased while production increased.",
|
||||
);
|
||||
expect(prompt).toContain("Complaint rate denominator");
|
||||
});
|
||||
|
||||
it("includes previous question and answer", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain(
|
||||
"What denominator is being used for the complaint rate?",
|
||||
);
|
||||
expect(prompt).toContain(
|
||||
"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.",
|
||||
);
|
||||
});
|
||||
|
||||
it("contains exact schema keys", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain("addedNodes");
|
||||
expect(prompt).toContain("updatedNodes");
|
||||
expect(prompt).toContain("addedEdges");
|
||||
expect(prompt).toContain("removedEdgeIds");
|
||||
expect(prompt).toContain("resolvedUnknownNodeIds");
|
||||
expect(prompt).toContain("affectedNodeIds");
|
||||
expect(prompt).toContain("selectedQuestion");
|
||||
});
|
||||
|
||||
it("lists enum values", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain(
|
||||
"observation | reported_claim | metric | state | transition | relationship | assumption | unknown | conclusion",
|
||||
);
|
||||
expect(prompt).toContain(
|
||||
"known | unknown | provisional | supported | weakened | contradicted | resolved",
|
||||
);
|
||||
expect(prompt).toContain(
|
||||
"supports | weakens | contradicts | depends_on | causes | may_cause | measures | compares_with | updates | other",
|
||||
);
|
||||
});
|
||||
|
||||
it("forbids full-graph replacement", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain("Never return a replacement graph");
|
||||
expect(prompt).toContain("Propose changes only");
|
||||
});
|
||||
|
||||
it("requires JSON only", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain("Return JSON only");
|
||||
expect(prompt).toContain("Return one JSON object only");
|
||||
});
|
||||
|
||||
it("describes controlled emergent unknown rules", () => {
|
||||
const prompt = buildGraphUpdatePrompt(makeContext());
|
||||
expect(prompt).toContain("Add at most 3 new unknown nodes");
|
||||
expect(prompt).toContain("Resolve the answered unknown first");
|
||||
expect(prompt).toContain(
|
||||
"selectedQuestion.question must be one narrow non-compound question",
|
||||
);
|
||||
expect(prompt).toContain(
|
||||
"the engine will deterministically choose final priority after validation",
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,209 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { formulateQuestion } from "@/lib/graph/question-formulator.js";
|
||||
import { makeEdge, makeGraph, makeNode } from "@/lib/graph/schema.js";
|
||||
|
||||
function makeGraphFor(node, extra = {}) {
|
||||
return makeGraph({
|
||||
centralStatement: extra.centralStatement || "Decision context",
|
||||
nodes: [node, ...(extra.nodes || [])],
|
||||
edges: extra.edges || [],
|
||||
activeUnknownNodeId: node.id,
|
||||
resolvedNodeIds: extra.resolvedNodeIds || [],
|
||||
currentSummary: "Test summary",
|
||||
});
|
||||
}
|
||||
|
||||
describe("formulateQuestion", () => {
|
||||
it("commercial viability plus build decision produces a decision-criterion question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-commercial",
|
||||
label: "Uncertainty regarding the commercial value of the product",
|
||||
description:
|
||||
"Commercial justification remains unclear because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
parentId: "n-decision",
|
||||
});
|
||||
const decision = makeNode({
|
||||
id: "n-decision",
|
||||
label: "Build decision",
|
||||
description: "Decision introduced by the answer.",
|
||||
kind: "state",
|
||||
status: "known",
|
||||
confidence: "medium",
|
||||
childIds: [unknown.id],
|
||||
value: "Deciding whether to build the product",
|
||||
});
|
||||
const graph = makeGraphFor(unknown, {
|
||||
nodes: [decision],
|
||||
resolvedNodeIds: [decision.id],
|
||||
});
|
||||
|
||||
const result = formulateQuestion({ node: unknown, graph });
|
||||
|
||||
expect(result.strategy).toBe("decision criterion");
|
||||
expect(result.question).toContain("What outcome");
|
||||
expect(result.question.toLowerCase()).toContain("justify");
|
||||
});
|
||||
|
||||
it("commercial viability does not produce a pricing-first question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-commercial",
|
||||
label: "Commercial viability",
|
||||
description:
|
||||
"Commercial viability remains unresolved because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
});
|
||||
const graph = makeGraphFor(unknown);
|
||||
|
||||
const result = formulateQuestion({ node: unknown, graph });
|
||||
|
||||
expect(result.question.toLowerCase()).not.toContain("price");
|
||||
expect(result.question.toLowerCase()).not.toContain("pricing");
|
||||
});
|
||||
|
||||
it("undefined term produces a definition question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-term",
|
||||
label: "Success criteria definition",
|
||||
description:
|
||||
"Need a definition of the term because the team uses it inconsistently.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.strategy).toBe("definition");
|
||||
expect(result.question).toMatch(/^What does /);
|
||||
});
|
||||
|
||||
it("unsupported claim produces an evidence question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-claim",
|
||||
label: "Demand claim",
|
||||
description: "Need evidence because the claim has not been validated.",
|
||||
kind: "reported_claim",
|
||||
status: "provisional",
|
||||
confidence: "low",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.strategy).toBe("evidence");
|
||||
expect(result.question).toContain("What evidence");
|
||||
});
|
||||
|
||||
it("missing previous state produces a baseline question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-baseline",
|
||||
label: "Baseline conversion rate",
|
||||
description:
|
||||
"Need the previous baseline because the change cannot be assessed without it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.strategy).toBe("baseline");
|
||||
expect(result.question).toContain("What was the comparable state before");
|
||||
});
|
||||
|
||||
it("unknown customer produces an actor/customer question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-customer",
|
||||
label: "Target customer",
|
||||
description:
|
||||
"Need to know the customer because value depends on who receives it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.strategy).toBe("actor/customer");
|
||||
expect(result.question).toContain(
|
||||
"Who experiences the problem or receives the value",
|
||||
);
|
||||
});
|
||||
|
||||
it("constraint unknown produces a constraint question", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-constraint",
|
||||
label: "Budget constraint",
|
||||
description:
|
||||
"Need the main budget constraint because it limits the available options.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.strategy).toBe("constraint");
|
||||
expect(result.question).toContain("What constraint most limits");
|
||||
});
|
||||
|
||||
it("question is singular and answerable", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-evidence",
|
||||
label: "Evidence of demand",
|
||||
description:
|
||||
"Need evidence of demand because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.question.match(/\?/g) || []).toHaveLength(1);
|
||||
expect(result.question.toLowerCase()).not.toContain(" and ");
|
||||
});
|
||||
|
||||
it("awkward uncertainty phrasing is rejected via fallback", () => {
|
||||
const unknown = makeNode({
|
||||
id: "n-weird",
|
||||
label: "Uncertainty regarding service reliability",
|
||||
description: "Unknown service reliability.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
});
|
||||
|
||||
const result = formulateQuestion({
|
||||
node: unknown,
|
||||
graph: makeGraphFor(unknown),
|
||||
});
|
||||
|
||||
expect(result.question).not.toContain("How should uncertainty regarding");
|
||||
expect(result.question).not.toContain(
|
||||
"What would resolve uncertainty regarding",
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,155 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { applyValidatedProposal } from "@/lib/graph/apply-proposal.js";
|
||||
import { formulateQuestion } from "@/lib/graph/question-formulator.js";
|
||||
import { selectActiveUnknownCandidate } from "@/lib/graph/utils.js";
|
||||
import { questionPriorityGeneralisationFixtures } from "@/tests/fixtures/question-priority-generalisation.js";
|
||||
|
||||
function clone(value) {
|
||||
return JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
function buildResolutionProposal(graph) {
|
||||
const activeNode = graph.nodes.find(
|
||||
(node) => node.id === graph.activeUnknownNodeId,
|
||||
);
|
||||
const placeholderCandidate = graph.nodes.find(
|
||||
(node) => node.kind === "unknown" && node.id !== activeNode.id,
|
||||
);
|
||||
|
||||
return {
|
||||
addedNodes: [],
|
||||
updatedNodes: [
|
||||
{
|
||||
nodeId: activeNode.id,
|
||||
previousStatus: activeNode.status,
|
||||
newStatus: "resolved",
|
||||
previousValue: activeNode.value ?? null,
|
||||
newValue: activeNode.value ?? "Resolved context answer",
|
||||
reason:
|
||||
"The resolved context unknown is treated as answered for fixture progression.",
|
||||
},
|
||||
],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: [activeNode.id],
|
||||
affectedNodeIds: [],
|
||||
selectedQuestion: {
|
||||
nodeId: placeholderCandidate?.id,
|
||||
question: "Placeholder candidate question?",
|
||||
reason: "Candidate only; deterministic selector should override it.",
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function assertQuestionStructure(question) {
|
||||
expect(question.match(/\?/g) || []).toHaveLength(1);
|
||||
expect(question).not.toMatch(/\?\s*(and|or)\b/i);
|
||||
expect(question).not.toMatch(/^What is\s+/i);
|
||||
expect(question).toMatch(/^(What|Who|When)\b/);
|
||||
}
|
||||
|
||||
describe("question priority generalisation", () => {
|
||||
for (const fixture of questionPriorityGeneralisationFixtures) {
|
||||
it(`${fixture.scenario} selects a foundational unknown and singular answerable strategy`, () => {
|
||||
const originalGraph = clone(fixture.graph);
|
||||
const deterministicSelection = selectActiveUnknownCandidate(
|
||||
fixture.graph,
|
||||
[fixture.graph.activeUnknownNodeId],
|
||||
);
|
||||
|
||||
const result = applyValidatedProposal({
|
||||
situationGraph: fixture.graph,
|
||||
proposal: buildResolutionProposal(fixture.graph),
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(fixture.graph).toEqual(originalGraph);
|
||||
expect(result.graphUpdate.selectedQuestion?.question).toBe(
|
||||
"Placeholder candidate question?",
|
||||
);
|
||||
|
||||
expect(deterministicSelection.nodeId).toBe(
|
||||
result.selectedQuestion.nodeId,
|
||||
);
|
||||
expect(fixture.acceptableFoundationalUnknownNodeIds).toContain(
|
||||
result.selectedQuestion.nodeId,
|
||||
);
|
||||
expect(result.selectedQuestion.nodeId).not.toBe(
|
||||
fixture.graph.nodes[fixture.graph.nodes.length - 1].id,
|
||||
);
|
||||
|
||||
expect(fixture.acceptableQuestionStrategies).toContain(
|
||||
result.selectedQuestion.strategy,
|
||||
);
|
||||
assertQuestionStructure(result.selectedQuestion.question);
|
||||
|
||||
const lowerQuestion = result.selectedQuestion.question.toLowerCase();
|
||||
for (const topic of fixture.prohibitedFirstTopics) {
|
||||
expect(lowerQuestion).not.toContain(topic.toLowerCase());
|
||||
}
|
||||
|
||||
const selectedNode = result.updatedSituationGraph.nodes.find(
|
||||
(node) => node.id === result.selectedQuestion.nodeId,
|
||||
);
|
||||
const reformulated = formulateQuestion({
|
||||
node: selectedNode,
|
||||
graph: result.updatedSituationGraph,
|
||||
context: {
|
||||
resolvedValues: ["Resolved context answer"],
|
||||
},
|
||||
});
|
||||
|
||||
expect(reformulated.question).toBe(result.selectedQuestion.question);
|
||||
expect(clone(result.updatedSituationGraph)).toEqual(
|
||||
result.updatedSituationGraph,
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
it("reports all five selected unknowns and strategies", () => {
|
||||
const summary = questionPriorityGeneralisationFixtures.map((fixture) => {
|
||||
const result = applyValidatedProposal({
|
||||
situationGraph: fixture.graph,
|
||||
proposal: buildResolutionProposal(fixture.graph),
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
|
||||
return {
|
||||
scenario: fixture.scenario,
|
||||
nodeId: result.selectedQuestion.nodeId,
|
||||
strategy: result.selectedQuestion.strategy,
|
||||
};
|
||||
});
|
||||
|
||||
expect(summary).toMatchInlineSnapshot(`
|
||||
[
|
||||
{
|
||||
"nodeId": "hire-success-criteria",
|
||||
"scenario": "Should we hire another engineer?",
|
||||
"strategy": "decision criterion",
|
||||
},
|
||||
{
|
||||
"nodeId": "van-reliability-threshold",
|
||||
"scenario": "Should we replace the delivery vans?",
|
||||
"strategy": "decision criterion",
|
||||
},
|
||||
{
|
||||
"nodeId": "country-value-threshold",
|
||||
"scenario": "Should we launch in another country?",
|
||||
"strategy": "actor/customer",
|
||||
},
|
||||
{
|
||||
"nodeId": "project-benefit-threshold",
|
||||
"scenario": "Should we continue a project that is over budget?",
|
||||
"strategy": "decision criterion",
|
||||
},
|
||||
{
|
||||
"nodeId": "support-value-threshold",
|
||||
"scenario": "Should we introduce a paid support tier?",
|
||||
"strategy": "actor/customer",
|
||||
},
|
||||
]
|
||||
`);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,442 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import {
|
||||
SituationKind,
|
||||
SituationStatus,
|
||||
ConfidenceLevel,
|
||||
SituationRelationship,
|
||||
situationNodeSchema,
|
||||
situationEdgeSchema,
|
||||
situationGraphSchema,
|
||||
graphUpdateSchema,
|
||||
startCaseRequestSchema,
|
||||
updateCaseRequestSchema,
|
||||
makeNodeId,
|
||||
makeNode,
|
||||
makeEdge,
|
||||
makeGraph,
|
||||
} from "@/lib/graph/schema.js";
|
||||
|
||||
describe("situationNodeSchema", () => {
|
||||
const validNode = {
|
||||
id: "n1",
|
||||
label: "Test Node",
|
||||
description: "A test node",
|
||||
kind: "observation",
|
||||
status: "known",
|
||||
confidence: "high",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
};
|
||||
|
||||
it("validates a complete valid node", () => {
|
||||
const result = situationNodeSchema.safeParse(validNode);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("requires id", () => {
|
||||
const invalid = { ...validNode, id: "" };
|
||||
const result = situationNodeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("requires label", () => {
|
||||
const invalid = { ...validNode, label: "" };
|
||||
const result = situationNodeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects invalid kind", () => {
|
||||
const invalid = { ...validNode, kind: "nonexistent" };
|
||||
const result = situationNodeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects invalid status", () => {
|
||||
const invalid = { ...validNode, status: "unknown_status" };
|
||||
const result = situationNodeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects invalid confidence", () => {
|
||||
const invalid = { ...validNode, confidence: "extreme" };
|
||||
const result = situationNodeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("allows numeric value", () => {
|
||||
const node = { ...validNode, value: 42 };
|
||||
const result = situationNodeSchema.safeParse(node);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("allows string value", () => {
|
||||
const node = { ...validNode, value: "active" };
|
||||
const result = situationNodeSchema.safeParse(node);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("situationEdgeSchema", () => {
|
||||
const validEdge = {
|
||||
id: "e1",
|
||||
fromNodeId: "n1",
|
||||
toNodeId: "n2",
|
||||
relationship: "supports",
|
||||
confidence: "medium",
|
||||
description: "Edge between nodes",
|
||||
};
|
||||
|
||||
it("validates a complete valid edge", () => {
|
||||
const result = situationEdgeSchema.safeParse(validEdge);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects invalid relationship type", () => {
|
||||
const invalid = { ...validEdge, relationship: "invalid_rel" };
|
||||
const result = situationEdgeSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("validates all relationship types", () => {
|
||||
for (const rel of Object.values(SituationRelationship)) {
|
||||
const edge = { ...validEdge, relationship: rel };
|
||||
const result = situationEdgeSchema.safeParse(edge);
|
||||
expect(result.success).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects self-referencing edges", () => {
|
||||
// Self-refs are structurally valid but semantically questionable
|
||||
const edge = { ...validEdge, fromNodeId: "n1", toNodeId: "n1" };
|
||||
const result = situationEdgeSchema.safeParse(edge);
|
||||
expect(result.success).toBe(true); // Structure is valid; semantics checked elsewhere
|
||||
});
|
||||
});
|
||||
|
||||
describe("situationGraphSchema", () => {
|
||||
const validGraph = {
|
||||
centralStatement: "Test graph summary",
|
||||
nodes: [makeNode({ id: "n1", label: "Node 1" })],
|
||||
edges: [],
|
||||
activeUnknownNodeId: null,
|
||||
resolvedNodeIds: [],
|
||||
currentSummary: "Initial summary",
|
||||
};
|
||||
|
||||
it("validates a complete valid graph", () => {
|
||||
const result = situationGraphSchema.safeParse(validGraph);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("requires at least one node", () => {
|
||||
const invalid = { ...validGraph, nodes: [] };
|
||||
const result = situationGraphSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("allows empty edges array", () => {
|
||||
const graph = { ...validGraph, edges: [] };
|
||||
const result = situationGraphSchema.safeParse(graph);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects missing centralStatement", () => {
|
||||
const invalid = { ...validGraph, centralStatement: "" };
|
||||
const result = situationGraphSchema.safeParse(invalid);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("graphUpdateSchema", () => {
|
||||
it("validates empty update (no-op proposal)", () => {
|
||||
const result = graphUpdateSchema.safeParse({});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("validates a complete update", () => {
|
||||
const node = makeNode({ id: "n2", label: "New Node" });
|
||||
const edge = makeEdge({ fromNodeId: "n1", toNodeId: "n2" });
|
||||
|
||||
const result = graphUpdateSchema.safeParse({
|
||||
addedNodes: [node],
|
||||
updatedNodes: [
|
||||
{
|
||||
nodeId: "n1",
|
||||
newStatus: "resolved",
|
||||
previousStatus: "unknown",
|
||||
reason: "Question answered",
|
||||
},
|
||||
],
|
||||
addedEdges: [edge],
|
||||
removedEdgeIds: ["e-old"],
|
||||
resolvedUnknownNodeIds: ["n2"],
|
||||
affectedNodeIds: ["n3"],
|
||||
selectedQuestion: {
|
||||
nodeId: "n2",
|
||||
question: "What does this new node mean?",
|
||||
reason: "A follow-up unknown remains.",
|
||||
},
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("allows null selectedQuestion", () => {
|
||||
const result = graphUpdateSchema.safeParse({
|
||||
selectedQuestion: null,
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects update with invalid node kind in addedNodes", () => {
|
||||
const invalid = graphUpdateSchema.safeParse({
|
||||
addedNodes: [
|
||||
{
|
||||
id: "x",
|
||||
label: "Test",
|
||||
kind: "invalid_kind",
|
||||
description: "test",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(invalid.success).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("API request schemas", () => {
|
||||
describe("startCaseRequestSchema", () => {
|
||||
it("validates scenario field", () => {
|
||||
const result = startCaseRequestSchema.safeParse({
|
||||
scenario: "Test scenario",
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects empty scenario", () => {
|
||||
const result = startCaseRequestSchema.safeParse({ scenario: "" });
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects scenario over 10000 chars", () => {
|
||||
const longScenario = "a".repeat(10001);
|
||||
const result = startCaseRequestSchema.safeParse({
|
||||
scenario: longScenario,
|
||||
});
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("accepts optional promptVersion", () => {
|
||||
const result = startCaseRequestSchema.safeParse({
|
||||
scenario: "Test",
|
||||
promptVersion: "v0.3",
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("updateCaseRequestSchema", () => {
|
||||
it("validates complete update request", () => {
|
||||
const graph = makeGraph({
|
||||
centralStatement: "Test scenario",
|
||||
nodes: [makeNode({ id: "n1", label: "N" })],
|
||||
currentSummary: "Current state of situation",
|
||||
});
|
||||
const result = updateCaseRequestSchema.safeParse({
|
||||
situationGraph: graph,
|
||||
previousQuestion: "What happened?",
|
||||
answer: "This is the answer",
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects missing situationGraph", () => {
|
||||
const result = updateCaseRequestSchema.safeParse({
|
||||
previousQuestion: "Q?",
|
||||
answer: "A",
|
||||
});
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects answer over 5000 chars", () => {
|
||||
const graph = makeGraph({
|
||||
centralStatement: "Test",
|
||||
nodes: [makeNode({ id: "n1", label: "N" })],
|
||||
currentSummary: "Test summary",
|
||||
});
|
||||
const result = updateCaseRequestSchema.safeParse({
|
||||
situationGraph: graph,
|
||||
previousQuestion: "Q?",
|
||||
answer: "x".repeat(5001),
|
||||
});
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("deterministic ID generation", () => {
|
||||
it("generate consistent IDs for same label", () => {
|
||||
const id1 = makeNodeId("Same Label");
|
||||
const id2 = makeNodeId("Same Label");
|
||||
expect(id1).toBe(id2);
|
||||
});
|
||||
|
||||
it("generates different IDs for different labels", () => {
|
||||
const id1 = makeNodeId("Label A");
|
||||
const id2 = makeNodeId("Label B");
|
||||
expect(id1).not.toBe(id2);
|
||||
});
|
||||
|
||||
it("IDs are prefixed with 'n' and short", () => {
|
||||
const id = makeNodeId(
|
||||
"A very long label that would produce a longer hash if not truncated",
|
||||
);
|
||||
expect(id.startsWith("n")).toBe(true);
|
||||
expect(id.length).toBeLessThan(15);
|
||||
});
|
||||
|
||||
it("same kind of nodes get deterministic IDs", () => {
|
||||
for (let i = 0; i < 10; i++) {
|
||||
expect(makeNodeId("Test Node")).toBe(makeNodeId("Test Node"));
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("helper functions", () => {
|
||||
describe("makeNode", () => {
|
||||
it("creates a minimal node with defaults", () => {
|
||||
const node = makeNode({ label: "Minimal" });
|
||||
const result = situationNodeSchema.safeParse(node);
|
||||
expect(result.success).toBe(true);
|
||||
expect(node.kind).toBe("observation");
|
||||
expect(node.status).toBe("unknown");
|
||||
expect(node.confidence).toBe("medium");
|
||||
});
|
||||
|
||||
it("creates a node with custom kind/status", () => {
|
||||
const node = makeNode({
|
||||
label: "Custom",
|
||||
kind: "metric",
|
||||
status: "known",
|
||||
confidence: "high",
|
||||
value: 42,
|
||||
unit: "count",
|
||||
});
|
||||
expect(node.kind).toBe("metric");
|
||||
expect(node.status).toBe("known");
|
||||
expect(node.confidence).toBe("high");
|
||||
expect(node.value).toBe(42);
|
||||
expect(node.unit).toBe("count");
|
||||
});
|
||||
|
||||
it("generates ID from label if none provided", () => {
|
||||
const node = makeNode({ label: "Auto-ID" });
|
||||
expect(node.id.startsWith("n")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("makeEdge", () => {
|
||||
it("creates a minimal edge with defaults", () => {
|
||||
const edge = makeEdge({ fromNodeId: "n1", toNodeId: "n2" });
|
||||
const result = situationEdgeSchema.safeParse(edge);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("generates description from node ids if not provided", () => {
|
||||
const edge = makeEdge({ fromNodeId: "n-alpha", toNodeId: "n-beta" });
|
||||
expect(edge.description).toContain("alpha");
|
||||
expect(edge.description).toContain("beta");
|
||||
});
|
||||
});
|
||||
|
||||
describe("makeGraph", () => {
|
||||
it("creates a minimal graph with defaults", () => {
|
||||
const graph = makeGraph({
|
||||
centralStatement: "Test",
|
||||
currentSummary: "Default summary",
|
||||
nodes: [makeNode({ id: "n1", label: "Placeholder" })],
|
||||
});
|
||||
const result = situationGraphSchema.safeParse(graph);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("allows specifying nodes and edges", () => {
|
||||
const graph = makeGraph({
|
||||
centralStatement: "Full Graph",
|
||||
currentSummary: "Full summary",
|
||||
nodes: [makeNode({ id: "n1", label: "N1" })],
|
||||
edges: [makeEdge({ fromNodeId: "n1", toNodeId: "n2" })],
|
||||
});
|
||||
expect(graph.nodes.length).toBe(1);
|
||||
expect(graph.edges.length).toBe(1);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("enum values completeness", () => {
|
||||
it("SituationKind has all expected values", () => {
|
||||
const expected = [
|
||||
"observation",
|
||||
"reported_claim",
|
||||
"metric",
|
||||
"state",
|
||||
"transition",
|
||||
"relationship",
|
||||
"assumption",
|
||||
"unknown",
|
||||
"conclusion",
|
||||
];
|
||||
const actual = Object.values(SituationKind);
|
||||
expect(actual).toEqual(expect.arrayContaining(expected));
|
||||
});
|
||||
|
||||
it("SituationStatus has all expected values", () => {
|
||||
const expected = [
|
||||
"known",
|
||||
"unknown",
|
||||
"provisional",
|
||||
"supported",
|
||||
"weakened",
|
||||
"contradicted",
|
||||
"resolved",
|
||||
];
|
||||
const actual = Object.values(SituationStatus);
|
||||
expect(actual).toEqual(expect.arrayContaining(expected));
|
||||
});
|
||||
|
||||
it("SituationRelationship has all expected values", () => {
|
||||
const expected = [
|
||||
"supports",
|
||||
"weakens",
|
||||
"contradicts",
|
||||
"depends_on",
|
||||
"causes",
|
||||
"may_cause",
|
||||
"measures",
|
||||
"compares_with",
|
||||
"updates",
|
||||
"other",
|
||||
];
|
||||
const actual = Object.values(SituationRelationship);
|
||||
expect(actual).toEqual(expect.arrayContaining(expected));
|
||||
});
|
||||
|
||||
it("ConfidenceLevel has all expected values", () => {
|
||||
const actual = Object.values(ConfidenceLevel);
|
||||
expect(actual).toContain("low");
|
||||
expect(actual).toContain("medium");
|
||||
expect(actual).toContain("high");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,152 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { parseGraphUpdateProposal } from "@/lib/graph/update-proposal.js";
|
||||
|
||||
function makeValidProposal(overrides = {}) {
|
||||
return {
|
||||
addedNodes: [],
|
||||
updatedNodes: [
|
||||
{
|
||||
nodeId: "n-unknown",
|
||||
previousStatus: "unknown",
|
||||
newStatus: "resolved",
|
||||
previousValue: null,
|
||||
newValue: "1.9 complaints per 100 units",
|
||||
reason: "The answer directly provides the normalized complaint rate.",
|
||||
},
|
||||
],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: ["n-unknown"],
|
||||
affectedNodeIds: [],
|
||||
selectedQuestion: null,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("parseGraphUpdateProposal", () => {
|
||||
it("parses a valid proposal", () => {
|
||||
const result = parseGraphUpdateProposal(makeValidProposal());
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.updatedNodes).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("fails on malformed JSON", () => {
|
||||
const result = parseGraphUpdateProposal("{not json");
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("fails when required update content is invalid", () => {
|
||||
const result = parseGraphUpdateProposal({
|
||||
updatedNodes: [{ nodeId: "n-unknown" }],
|
||||
});
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("removes null array entries and logs them", () => {
|
||||
const result = parseGraphUpdateProposal(
|
||||
JSON.stringify({
|
||||
...makeValidProposal(),
|
||||
addedNodes: [null],
|
||||
}),
|
||||
);
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.addedNodes).toEqual([]);
|
||||
expect(result.normalisationsApplied).toEqual(
|
||||
expect.arrayContaining([
|
||||
expect.objectContaining({ change: "Removed null array entry" }),
|
||||
]),
|
||||
);
|
||||
});
|
||||
|
||||
it("fills missing optional arrays with empty arrays", () => {
|
||||
const result = parseGraphUpdateProposal({
|
||||
updatedNodes: [],
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.addedNodes).toEqual([]);
|
||||
expect(result.proposal.addedEdges).toEqual([]);
|
||||
expect(result.normalisationsApplied.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("normalises confirmed enum alias and preserves IDs", () => {
|
||||
const result = parseGraphUpdateProposal({
|
||||
...makeValidProposal(),
|
||||
addedNodes: [
|
||||
{
|
||||
id: "n-new",
|
||||
label: "Reported update",
|
||||
description: "A new reported claim",
|
||||
kind: "reported_statement",
|
||||
status: "supported",
|
||||
confidence: "medium",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.addedNodes[0].kind).toBe("reported_claim");
|
||||
expect(result.proposal.addedNodes[0].id).toBe("n-new");
|
||||
});
|
||||
|
||||
it("unknown enum values still fail", () => {
|
||||
const result = parseGraphUpdateProposal({
|
||||
...makeValidProposal(),
|
||||
addedNodes: [
|
||||
{
|
||||
id: "n-new",
|
||||
label: "Bad node",
|
||||
description: "Bad node",
|
||||
kind: "unsupported_kind",
|
||||
status: "supported",
|
||||
confidence: "medium",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("defaults missing selectedQuestion to null", () => {
|
||||
const result = parseGraphUpdateProposal({
|
||||
addedNodes: [],
|
||||
updatedNodes: [],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: [],
|
||||
affectedNodeIds: [],
|
||||
});
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.selectedQuestion).toBeNull();
|
||||
});
|
||||
|
||||
it("parses a valid selectedQuestion", () => {
|
||||
const result = parseGraphUpdateProposal(
|
||||
makeValidProposal({
|
||||
selectedQuestion: {
|
||||
nodeId: "n-follow-up",
|
||||
question: "How should commercial value be defined for this decision?",
|
||||
reason: "A consequential unknown remains unresolved.",
|
||||
},
|
||||
}),
|
||||
);
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.proposal.selectedQuestion?.nodeId).toBe("n-follow-up");
|
||||
});
|
||||
|
||||
it("does not invent a next question field outside the contract", () => {
|
||||
const result = parseGraphUpdateProposal(makeValidProposal());
|
||||
expect(result.proposal.nextQuestion).toBeUndefined();
|
||||
});
|
||||
});
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,179 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { normaliseAnalysisResponse } from "@/lib/reconstruction/compatibility.js";
|
||||
|
||||
const mockGenerateReconstruction = vi.fn();
|
||||
|
||||
vi.mock("@/lib/config.js", () => ({
|
||||
getConfig: () => ({
|
||||
ok: true,
|
||||
config: {
|
||||
OLLAMA_BASE_URL: "http://example.test",
|
||||
OLLAMA_MODEL: "test-model",
|
||||
},
|
||||
}),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/llm/provider.js", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: (...args) => mockGenerateReconstruction(...args),
|
||||
}),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/reconstruction/prompt.js", () => ({
|
||||
buildPrompt: async () => ({ prompt: "prompt", version: "v0.3" }),
|
||||
PROMPT_VERSIONS: ["v0.1", "v0.2", "v0.3"],
|
||||
DEFAULT_PROMPT_VERSION: "v0.3",
|
||||
}));
|
||||
|
||||
describe("normaliseAnalysisResponse", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("leaves already-valid responses unchanged", () => {
|
||||
const input = {
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "x",
|
||||
evidenceType: "reported_statement",
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
source: "report",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = normaliseAnalysisResponse(input);
|
||||
|
||||
expect(result.normalised).toEqual(input);
|
||||
expect(result.changesApplied).toEqual([]);
|
||||
});
|
||||
|
||||
it("normalises null evidence source deterministically", () => {
|
||||
const input = {
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "x",
|
||||
evidenceType: "reported_statement",
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
source: null,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const result = normaliseAnalysisResponse(input);
|
||||
|
||||
expect(result.normalised.evidence[0]).not.toHaveProperty("source");
|
||||
expect(result.changesApplied).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("does not invent a next question", () => {
|
||||
const input = { evidence: [] };
|
||||
const result = normaliseAnalysisResponse(input);
|
||||
expect(result.normalised.nextQuestion).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not repair missing reasoning content", () => {
|
||||
const input = { evidence: [{ source: null }] };
|
||||
const result = normaliseAnalysisResponse(input);
|
||||
expect(result.normalised.reconstruction).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("analyseScenario compatibility", () => {
|
||||
it("succeeds when the only mismatch is null evidence source", async () => {
|
||||
mockGenerateReconstruction.mockResolvedValue({
|
||||
inputClassification: {
|
||||
primaryType: "unexplained_change",
|
||||
secondaryTypes: [],
|
||||
reasoningModes: ["validate_measurement"],
|
||||
classificationReason: "reason",
|
||||
confidence: "medium",
|
||||
},
|
||||
reconstruction: {
|
||||
summary: "summary",
|
||||
actors: [],
|
||||
systemsOrObjects: [],
|
||||
expectedStates: [],
|
||||
observedStates: [],
|
||||
differences: [],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [],
|
||||
importantUnknowns: [],
|
||||
plausibleInterpretations: [],
|
||||
},
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "desc",
|
||||
evidenceType: "reported_statement",
|
||||
source: null,
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What denominator?",
|
||||
targets: ["observedStates"],
|
||||
reason: "reason",
|
||||
expectedInformationValue: "high",
|
||||
reasoningMode: "validate_measurement",
|
||||
},
|
||||
});
|
||||
|
||||
const { analyseScenario } = await import("@/lib/analysis.js");
|
||||
const result = await analyseScenario("Scenario text", {
|
||||
promptVersion: "v0.3",
|
||||
});
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(result.compatibilityApplied).toBe(true);
|
||||
expect(result.compatibilityChanges).toHaveLength(1);
|
||||
expect(result.evidence[0]).not.toHaveProperty("source");
|
||||
expect(result.nextQuestion.question).toBe("What denominator?");
|
||||
});
|
||||
|
||||
it("still fails when required reasoning content is missing", async () => {
|
||||
mockGenerateReconstruction.mockResolvedValue({
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "desc",
|
||||
evidenceType: "reported_statement",
|
||||
source: null,
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
const { analyseScenario } = await import("@/lib/analysis.js");
|
||||
const result = await analyseScenario("Scenario text", {
|
||||
promptVersion: "v0.3",
|
||||
});
|
||||
|
||||
expect(result.success).toBe(false);
|
||||
expect(result.compatibilityApplied).toBe(true);
|
||||
expect(result.nextQuestion).toBeUndefined();
|
||||
});
|
||||
|
||||
it("malformed JSON still fails", async () => {
|
||||
mockGenerateReconstruction.mockResolvedValue("{not valid json");
|
||||
|
||||
const { analyseScenario } = await import("@/lib/analysis.js");
|
||||
const result = await analyseScenario("Scenario text", {
|
||||
promptVersion: "v0.3",
|
||||
});
|
||||
|
||||
expect(result.success).toBe(false);
|
||||
expect(result.compatibilityApplied).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,107 @@
|
||||
import { test, expect } from "@playwright/test";
|
||||
|
||||
const BASE_URL = process.env.PLAYWRIGHT_BASE_URL || "http://localhost:3000";
|
||||
|
||||
test.setTimeout(300000);
|
||||
|
||||
test("graph-backed one-turn update smoke test", async ({ page }) => {
|
||||
await page.goto(BASE_URL);
|
||||
|
||||
// Page should load without error
|
||||
await expect(page.getByText(/Confidence Engine/i)).toBeVisible();
|
||||
|
||||
// Type the scenario
|
||||
const textarea = page.locator("textarea[placeholder*='Describe']");
|
||||
await textarea.fill(
|
||||
"Complaints increased by 35% while production increased by 40%.",
|
||||
);
|
||||
await expect(textarea).toHaveValue(
|
||||
"Complaints increased by 35% while production increased by 40%.",
|
||||
);
|
||||
|
||||
// Button should be enabled
|
||||
await expect(page.getByRole("button", { name: /Analyse/i })).toBeEnabled();
|
||||
|
||||
// Click Analyse and wait for graph-backed result
|
||||
await page.getByRole("button", { name: /Analyse/i }).click();
|
||||
|
||||
await expect(
|
||||
page
|
||||
.locator("section")
|
||||
.filter({ hasText: /Selected Question/i })
|
||||
.last()
|
||||
.getByRole("heading", { name: /Selected Question/i }),
|
||||
).toBeVisible({ timeout: 180000 });
|
||||
await expect(
|
||||
page.getByRole("heading", { name: /Situation Graph/i }),
|
||||
).toBeVisible({ timeout: 180000 });
|
||||
await expect(page.getByText(/Central statement/i)).toBeVisible();
|
||||
await expect(page.getByText(/Active unknown/i)).toBeVisible();
|
||||
await expect(page.getByText(/Error:/i)).toHaveCount(0);
|
||||
|
||||
const rawJsonToggle = page.getByText(/Raw graph JSON/i);
|
||||
await expect(rawJsonToggle).toBeVisible();
|
||||
await rawJsonToggle.click();
|
||||
await expect(page.getByText(/centralStatement/i)).toBeVisible();
|
||||
|
||||
const selectedQuestionSections = page
|
||||
.locator("section")
|
||||
.filter({ hasText: "Selected Question" });
|
||||
await expect(selectedQuestionSections).toHaveCount(1);
|
||||
const questionText = await selectedQuestionSections.first().innerText();
|
||||
expect(questionText.length).toBeGreaterThan(25);
|
||||
|
||||
const answerTextarea = page.locator(
|
||||
"textarea[placeholder*='Enter the answer']",
|
||||
);
|
||||
await expect(answerTextarea).toBeVisible();
|
||||
await answerTextarea.fill(
|
||||
"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.",
|
||||
);
|
||||
await page.getByRole("button", { name: /Update situation/i }).click();
|
||||
|
||||
await expect(page.getByText(/Graph update applied/i)).toBeVisible({
|
||||
timeout: 240000,
|
||||
});
|
||||
await expect(page.getByText(/Resolved unknowns/i)).toBeVisible();
|
||||
await expect(page.getByText(/Newly surfaced unknowns/i)).toBeVisible();
|
||||
await expect(page.getByText(/Affected nodes/i)).toBeVisible();
|
||||
await expect(
|
||||
page.getByText(/Selected Question|Next question:/i),
|
||||
).toBeVisible();
|
||||
await expect(
|
||||
page.getByText(
|
||||
/Additional submission is disabled in this one-update prototype\./i,
|
||||
),
|
||||
).toBeVisible();
|
||||
await expect(page.getByText(/Error:/i)).toHaveCount(0);
|
||||
await expect(page.getByText(/Update error:/i)).toHaveCount(0);
|
||||
await expect(answerTextarea).toHaveValue("");
|
||||
|
||||
const proposalToggle = page.getByText(/Proposal details/i);
|
||||
await expect(proposalToggle).toBeVisible();
|
||||
|
||||
await rawJsonToggle.click();
|
||||
await expect(page.getByText(/resolvedNodeIds/i)).toBeVisible();
|
||||
|
||||
// Get full body text for verification
|
||||
const bodyText = await page.locator("body").innerText();
|
||||
|
||||
// Check key content indicators
|
||||
const hasSelectedQuestion = bodyText.includes("Selected Question");
|
||||
const hasComplaints =
|
||||
bodyText.includes("Complaint") || bodyText.includes("complaint");
|
||||
const hasProduction =
|
||||
bodyText.includes("production") || bodyText.includes("Production");
|
||||
const hasRateContext =
|
||||
bodyText.toLowerCase().includes("rate") ||
|
||||
bodyText.toLowerCase().includes("unit") ||
|
||||
bodyText.toLowerCase().includes("denominator") ||
|
||||
bodyText.toLowerCase().includes("per-unit");
|
||||
|
||||
// Basic structural checks
|
||||
expect(bodyText.length).toBeGreaterThan(400);
|
||||
expect(hasSelectedQuestion).toBe(true);
|
||||
expect(bodyText.includes("Resolved unknowns")).toBe(true);
|
||||
expect(bodyText.includes("Affected nodes")).toBe(true);
|
||||
});
|
||||
@@ -0,0 +1,568 @@
|
||||
import React from "react";
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import { renderToStaticMarkup } from "react-dom/server";
|
||||
import DiagnosticsView from "@/components/diagnostics-view.jsx";
|
||||
import GraphUpdateView from "@/components/graph-update-view.jsx";
|
||||
import SituationGraphView from "@/components/situation-graph-view.jsx";
|
||||
import {
|
||||
ScenarioResultPanels,
|
||||
UpdateErrorPanel,
|
||||
submitAnswerForUpdateCase,
|
||||
submitScenarioForStartCase,
|
||||
} from "@/components/scenario-form.jsx";
|
||||
|
||||
function makeGraphResult(overrides = {}) {
|
||||
return {
|
||||
success: true,
|
||||
situationGraph: {
|
||||
centralStatement: "Complaints increased while production increased.",
|
||||
currentSummary:
|
||||
"Nodes: 2 observation, 1 unknown | Edges: 2 total | Unknowns: 1 unresolved",
|
||||
activeUnknownNodeId: "n-unknown",
|
||||
resolvedNodeIds: [],
|
||||
nodes: [
|
||||
{
|
||||
id: "n-1",
|
||||
label: "Complaints up 35%",
|
||||
description: "Complaints increased by 35%",
|
||||
kind: "observation",
|
||||
status: "supported",
|
||||
confidence: "high",
|
||||
value: 35,
|
||||
unit: "%",
|
||||
},
|
||||
{
|
||||
id: "n-2",
|
||||
label: "Production up 40%",
|
||||
description: "Production increased by 40%",
|
||||
kind: "observation",
|
||||
status: "supported",
|
||||
confidence: "high",
|
||||
value: 40,
|
||||
unit: "%",
|
||||
},
|
||||
{
|
||||
id: "n-unknown",
|
||||
label: "Complaint rate denominator",
|
||||
description: "Need the denominator for complaint rate",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "medium",
|
||||
value: null,
|
||||
unit: null,
|
||||
},
|
||||
],
|
||||
edges: [
|
||||
{ id: "e1", fromNodeId: "n-1", toNodeId: "n-unknown" },
|
||||
{ id: "e2", fromNodeId: "n-2", toNodeId: "n-unknown" },
|
||||
],
|
||||
},
|
||||
selectedQuestion: {
|
||||
question: "What denominator is being used for the complaint rate?",
|
||||
},
|
||||
diagnostics: {
|
||||
modelName: "test",
|
||||
responseDurationMs: 1234,
|
||||
validationStatus: "valid",
|
||||
promptVersion: "test-prompt",
|
||||
nodeCount: 3,
|
||||
edgeCount: 2,
|
||||
graphReferenceValidation: { valid: true, errors: [] },
|
||||
},
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function makeUpdateSuccess(overrides = {}) {
|
||||
return {
|
||||
success: true,
|
||||
stage: "update_applied",
|
||||
updatedSituationGraph: {
|
||||
centralStatement: "Complaints increased while production increased.",
|
||||
currentSummary: "Updated summary",
|
||||
activeUnknownNodeId: "n-next-unknown",
|
||||
resolvedNodeIds: ["n-unknown"],
|
||||
nodes: [
|
||||
{
|
||||
id: "n-1",
|
||||
label: "Complaints up 35%",
|
||||
description: "Complaints increased by 35%",
|
||||
kind: "observation",
|
||||
status: "supported",
|
||||
confidence: "high",
|
||||
value: 35,
|
||||
unit: "%",
|
||||
},
|
||||
{
|
||||
id: "n-conclusion",
|
||||
label: "Quality deterioration",
|
||||
description: "Quality deterioration conclusion",
|
||||
kind: "conclusion",
|
||||
status: "weakened",
|
||||
confidence: "medium",
|
||||
value: null,
|
||||
unit: null,
|
||||
},
|
||||
{
|
||||
id: "n-unknown",
|
||||
label: "Complaint rate denominator",
|
||||
description: "Need the denominator for complaint rate",
|
||||
kind: "unknown",
|
||||
status: "resolved",
|
||||
confidence: "medium",
|
||||
value: "1.9 complaints per 100 units",
|
||||
unit: null,
|
||||
},
|
||||
{
|
||||
id: "n-next-unknown",
|
||||
label: "Commercial value definition",
|
||||
description: "Need a definition because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value: null,
|
||||
unit: null,
|
||||
},
|
||||
],
|
||||
edges: [],
|
||||
},
|
||||
proposal: {
|
||||
addedNodes: [
|
||||
{
|
||||
id: "n-next-unknown",
|
||||
label: "Commercial value definition",
|
||||
description: "Need a definition because the decision depends on it.",
|
||||
kind: "unknown",
|
||||
status: "unknown",
|
||||
confidence: "high",
|
||||
value: null,
|
||||
unit: null,
|
||||
evidenceIds: [],
|
||||
dependsOn: [],
|
||||
affects: [],
|
||||
parentId: null,
|
||||
childIds: [],
|
||||
},
|
||||
],
|
||||
updatedNodes: [
|
||||
{ nodeId: "n-unknown", newStatus: "resolved", reason: "answered" },
|
||||
],
|
||||
addedEdges: [],
|
||||
removedEdgeIds: [],
|
||||
resolvedUnknownNodeIds: ["n-unknown"],
|
||||
affectedNodeIds: ["n-conclusion"],
|
||||
selectedQuestion: {
|
||||
nodeId: "n-next-unknown",
|
||||
question: "How should commercial value be defined for this decision?",
|
||||
reason: "A narrower consequential uncertainty remains.",
|
||||
},
|
||||
},
|
||||
selectedQuestion: {
|
||||
nodeId: "n-next-unknown",
|
||||
question: "How should commercial value be defined for this decision?",
|
||||
reason: "A narrower consequential uncertainty remains.",
|
||||
},
|
||||
affectedNodeIds: ["n-conclusion"],
|
||||
resolvedUnknownNodeIds: ["n-unknown"],
|
||||
previousActiveUnknownNodeId: "n-unknown",
|
||||
newActiveUnknownNodeId: "n-next-unknown",
|
||||
changesApplied: {
|
||||
updatedNodeCount: 2,
|
||||
resolvedUnknownCount: 1,
|
||||
affectedNodeCount: 1,
|
||||
},
|
||||
diagnostics: { responseDurationMs: 100 },
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("scenario-form UI helpers", () => {
|
||||
it("submits to /api/cases/start", async () => {
|
||||
const fetchImpl = vi.fn().mockResolvedValue({ ok: true });
|
||||
|
||||
await submitScenarioForStartCase(fetchImpl, "Scenario text");
|
||||
|
||||
expect(fetchImpl).toHaveBeenCalledWith(
|
||||
"/api/cases/start",
|
||||
expect.objectContaining({
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("empty answer is rejected without fetch", async () => {
|
||||
const fetchImpl = vi.fn();
|
||||
|
||||
const result = await submitAnswerForUpdateCase(fetchImpl, {
|
||||
situationGraph: { nodes: [] },
|
||||
previousQuestion: "What changed?",
|
||||
answer: " ",
|
||||
});
|
||||
|
||||
expect(result.skipped).toBe(true);
|
||||
expect(fetchImpl).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("update request body contains graph, previousQuestion and answer", async () => {
|
||||
const fetchImpl = vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: async () => ({ success: true }),
|
||||
});
|
||||
const graph = { nodes: [{ id: "n1" }], edges: [] };
|
||||
|
||||
await submitAnswerForUpdateCase(fetchImpl, {
|
||||
situationGraph: graph,
|
||||
previousQuestion: "What changed?",
|
||||
answer: "The rate fell.",
|
||||
});
|
||||
|
||||
expect(fetchImpl).toHaveBeenCalledWith(
|
||||
"/api/cases/update",
|
||||
expect.objectContaining({
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
situationGraph: graph,
|
||||
previousQuestion: "What changed?",
|
||||
answer: "The rate fell.",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("graph-backed UI rendering", () => {
|
||||
it("renders central statement from successful graph response", () => {
|
||||
const data = makeGraphResult();
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={data.situationGraph}
|
||||
selectedQuestion={data.selectedQuestion}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Central statement");
|
||||
expect(html).toContain("Complaints increased while production increased.");
|
||||
});
|
||||
|
||||
it("renders active unknown", () => {
|
||||
const data = makeGraphResult();
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={data.situationGraph}
|
||||
selectedQuestion={data.selectedQuestion}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Active unknown");
|
||||
expect(html).toContain("Complaint rate denominator");
|
||||
});
|
||||
|
||||
it("renders selected question exactly once", () => {
|
||||
const data = makeGraphResult();
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={data.situationGraph}
|
||||
selectedQuestion={data.selectedQuestion}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(
|
||||
html.match(/What denominator is being used for the complaint rate\?/g) ||
|
||||
[],
|
||||
).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("no answer form appears when selectedQuestion is null", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={makeGraphResult().situationGraph}
|
||||
selectedQuestion={null}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).not.toContain("Update situation");
|
||||
});
|
||||
|
||||
it("renders diagnostics", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<DiagnosticsView result={makeGraphResult()} />,
|
||||
);
|
||||
|
||||
expect(html).toContain("Diagnostics");
|
||||
expect(html).toContain("test");
|
||||
expect(html).toContain("1234ms");
|
||||
expect(html).toContain("Node count");
|
||||
expect(html).toContain("Edge count");
|
||||
expect(html).toContain("Graph references");
|
||||
});
|
||||
|
||||
it("hides empty sections", () => {
|
||||
const base = makeGraphResult();
|
||||
const result = makeGraphResult({
|
||||
selectedQuestion: null,
|
||||
situationGraph: {
|
||||
...base.situationGraph,
|
||||
activeUnknownNodeId: null,
|
||||
nodes: [base.situationGraph.nodes[0]],
|
||||
edges: [],
|
||||
},
|
||||
});
|
||||
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={result.situationGraph}
|
||||
selectedQuestion={result.selectedQuestion}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).not.toContain("Selected Question");
|
||||
expect(html).not.toContain("Active unknown");
|
||||
});
|
||||
|
||||
it("displays API error clearly", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<ScenarioResultPanels
|
||||
status="error"
|
||||
result={{ error: "Invalid start-case request" }}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Error: Invalid start-case request");
|
||||
});
|
||||
|
||||
it("renders expandable raw graph JSON", () => {
|
||||
const data = makeGraphResult();
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={data.situationGraph}
|
||||
selectedQuestion={data.selectedQuestion}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Raw graph JSON");
|
||||
expect(html).toContain(""centralStatement"");
|
||||
});
|
||||
|
||||
it("resolved unknowns render", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Resolved unknowns");
|
||||
expect(html).toContain("Complaint rate denominator");
|
||||
});
|
||||
|
||||
it("newly surfaced unknowns render", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Newly surfaced unknowns");
|
||||
expect(html).toContain("Commercial value definition");
|
||||
});
|
||||
|
||||
it("affected nodes render", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Affected nodes");
|
||||
expect(html).toContain("Quality deterioration");
|
||||
});
|
||||
|
||||
it("renders validated next question when present", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain(
|
||||
"How should commercial value be defined for this decision?",
|
||||
);
|
||||
});
|
||||
|
||||
it("no fake next question appears when there is none", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess({
|
||||
newActiveUnknownNodeId: null,
|
||||
selectedQuestion: null,
|
||||
proposal: {
|
||||
...makeUpdateSuccess().proposal,
|
||||
selectedQuestion: null,
|
||||
},
|
||||
}),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("No next question selected yet.");
|
||||
});
|
||||
|
||||
it("previous and new active unknowns render labels", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Previous active unknown");
|
||||
expect(html).toContain("Complaint rate denominator");
|
||||
expect(html).toContain("New active unknown");
|
||||
expect(html).toContain("Commercial value definition");
|
||||
});
|
||||
|
||||
it("successful update renders prior and new state together", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Previous active unknown");
|
||||
expect(html).toContain("Resolved unknowns");
|
||||
expect(html).toContain("Newly surfaced unknowns");
|
||||
expect(html).toContain("New active unknown");
|
||||
expect(html).toContain("Next question");
|
||||
expect(html).toContain(
|
||||
"How should commercial value be defined for this decision?",
|
||||
);
|
||||
});
|
||||
|
||||
it("situation graph marks newly surfaced and active unknowns", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<SituationGraphView
|
||||
situationGraph={makeUpdateSuccess().updatedSituationGraph}
|
||||
selectedQuestion={makeUpdateSuccess().selectedQuestion}
|
||||
newlySurfacedNodeIds={["n-next-unknown"]}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("newly surfaced unknown");
|
||||
expect(html).toContain("active unknown");
|
||||
expect(html).toContain("resolved unknown");
|
||||
});
|
||||
|
||||
it("disabled follow-up form is shown only as prototype limitation", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={makeUpdateSuccess()}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("How should commercial value be defined for this decision?");
|
||||
});
|
||||
|
||||
it("raw ids remain only in collapsed proposal details", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView
|
||||
updateResult={{
|
||||
...makeUpdateSuccess(),
|
||||
previousSituationGraph: makeGraphResult().situationGraph,
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Proposal details");
|
||||
expect(html).toContain(""resolvedUnknownNodeIds"");
|
||||
});
|
||||
|
||||
it("update diagnostics render valid values", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<DiagnosticsView
|
||||
result={{
|
||||
diagnostics: {
|
||||
promptVersion: "v0.4",
|
||||
modelName: "configured-model",
|
||||
responseDurationMs: 456,
|
||||
validationStatus: "valid",
|
||||
nodeCount: 7,
|
||||
edgeCount: 3,
|
||||
graphReferenceValidation: { valid: true, errors: [] },
|
||||
},
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("v0.4");
|
||||
expect(html).toContain("456ms");
|
||||
expect(html).toContain("7");
|
||||
expect(html).toContain("3");
|
||||
expect(html).toContain("✅ valid");
|
||||
});
|
||||
|
||||
it("structured update error renders", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<UpdateErrorPanel
|
||||
updateError={{
|
||||
error: "Invalid graph update proposal",
|
||||
proposalErrors: [{ message: "bad proposal" }],
|
||||
}}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Update error: Invalid graph update proposal");
|
||||
expect(html).toContain("bad proposal");
|
||||
});
|
||||
|
||||
it("failed update does not fabricate history", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<>
|
||||
<UpdateErrorPanel
|
||||
updateError={{
|
||||
error: "Update case failed",
|
||||
errors: [
|
||||
'New unknown must be explicitly related to an answer-derived node: "nu_commercial_val"',
|
||||
],
|
||||
}}
|
||||
/>
|
||||
<GraphUpdateView updateResult={null} />
|
||||
</>,
|
||||
);
|
||||
|
||||
expect(html).toContain("Update error: Update case failed");
|
||||
expect(html).toContain("nu_commercial_val");
|
||||
expect(html).not.toContain("Previous active unknown");
|
||||
expect(html).not.toContain("Resolved unknowns");
|
||||
expect(html).not.toContain("Newly surfaced unknowns");
|
||||
expect(html).not.toContain("New active unknown");
|
||||
expect(html).not.toContain("Proposal details");
|
||||
});
|
||||
|
||||
it("proposal details remain collapsible", () => {
|
||||
const html = renderToStaticMarkup(
|
||||
<GraphUpdateView updateResult={makeUpdateSuccess()} />,
|
||||
);
|
||||
|
||||
expect(html).toContain("Proposal details");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,598 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { promises as fs } from "node:fs";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { dirname, join } from "node:path";
|
||||
import {
|
||||
PROMPT_VERSIONS,
|
||||
buildPrompt,
|
||||
DEFAULT_PROMPT_VERSION,
|
||||
} from "@/lib/reconstruction/prompt.js";
|
||||
import {
|
||||
reconstructionV2Schema,
|
||||
parseReconstructionV2,
|
||||
} from "@/lib/reconstruction/schema.js";
|
||||
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = dirname(__filename);
|
||||
const PROMPTS_DIR = join(__dirname, "../prompts");
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// v0.3 prompt loading tests
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("v0.3 prompt", () => {
|
||||
it("v0.3 is in PROMPT_VERSIONS", () => {
|
||||
expect(PROMPT_VERSIONS).toContain("v0.3");
|
||||
});
|
||||
|
||||
it("DEFAULT_PROMPT_VERSION is v0.3 on this branch", () => {
|
||||
expect(DEFAULT_PROMPT_VERSION).toBe("v0.3");
|
||||
});
|
||||
|
||||
it("v0.2 remains available in PROMPT_VERSIONS", () => {
|
||||
expect(PROMPT_VERSIONS).toContain("v0.2");
|
||||
});
|
||||
|
||||
it("v0.3 prompt file loads from disk", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(typeof content).toBe("string");
|
||||
expect(content.length).toBeGreaterThan(500);
|
||||
});
|
||||
|
||||
it("v0.3 prompt contains normalisation guidance", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content.toLowerCase()).toContain("normalise");
|
||||
expect(content.toLowerCase()).toContain("rate");
|
||||
expect(content.toLowerCase()).toContain("denominator") ||
|
||||
expect(content.toLowerCase()).toContain("exposure");
|
||||
});
|
||||
|
||||
it("v0.3 prompt contains discipline guidance", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
// Should mention not generating speculative interpretations
|
||||
expect(content).toMatch(/interpretation/i);
|
||||
// Should mention one question discipline
|
||||
expect(content).toMatch(/exactly.*one.*question|one.*only.*question|single.*question/i) ||
|
||||
expect(content).toMatch(/Do NOT combine/i);
|
||||
});
|
||||
|
||||
it("buildPrompt returns v0.3 prompt with scenario substituted", async () => {
|
||||
const result = await buildPrompt("Test scenario text", "v0.3");
|
||||
expect(result.version).toBe("v0.3");
|
||||
expect(result.prompt).toContain("Test scenario text");
|
||||
// Should contain the normalisation section guidance
|
||||
expect(result.prompt.toLowerCase()).toContain("normalise");
|
||||
});
|
||||
|
||||
it("buildPrompt returns v0.2 prompt when requested", async () => {
|
||||
const result = await buildPrompt("Test scenario text", "v0.2");
|
||||
expect(result.version).toBe("v0.2");
|
||||
expect(result.prompt).toContain("Test scenario text");
|
||||
});
|
||||
|
||||
it("buildPrompt default is v0.3", async () => {
|
||||
const result = await buildPrompt("Test scenario text");
|
||||
expect(result.version).toBe("v0.3");
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// v0.2 prompt still works
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("v0.2 backward compatibility", () => {
|
||||
it("v0.2 prompt file exists and loads", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.2.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(typeof content).toBe("string");
|
||||
expect(content.length).toBeGreaterThan(500);
|
||||
});
|
||||
|
||||
it("buildPrompt returns v0.2 version string", async () => {
|
||||
const result = await buildPrompt("test", "v0.2");
|
||||
expect(result.version).toBe("v0.2");
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// Schema validation tests for v0.3-shaped output
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("v0.3 schema validation", () => {
|
||||
it("validates a complete valid reconstruction with empty interpretations", () => {
|
||||
const input = {
|
||||
inputClassification: {
|
||||
primaryType: "unexplained_change",
|
||||
secondaryTypes: ["reported_claim"],
|
||||
reasoningModes: ["identify_difference"],
|
||||
classificationReason: "Two metrics changed without explanation.",
|
||||
confidence: "medium",
|
||||
},
|
||||
reconstruction: {
|
||||
summary: "Both complaints and production increased.",
|
||||
actors: [],
|
||||
systemsOrObjects: [
|
||||
{ id: "complaints_metric", description: "Volume of complaints", confidence: "high" },
|
||||
],
|
||||
expectedStates: [],
|
||||
observedStates: [
|
||||
{ id: "obs1", description: "Complaint volume rose by 35%", confidence: "medium" },
|
||||
{ id: "obs2", description: "Production volume rose by 40%", confidence: "medium" },
|
||||
],
|
||||
differences: [
|
||||
{
|
||||
id: "diff1",
|
||||
description:
|
||||
"Production grew faster than complaints, so the complaint-to-production ratio may have improved.",
|
||||
confidence: "medium",
|
||||
},
|
||||
],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [
|
||||
{
|
||||
id: "trans1",
|
||||
description: "Complaint volume shifted to a higher level without explained cause",
|
||||
confidence: "medium",
|
||||
entity: "complaints_metric",
|
||||
previousState: "Baseline volume (unknown)",
|
||||
currentState: "+35% increase",
|
||||
},
|
||||
],
|
||||
contradictions: [],
|
||||
importantUnknowns: [
|
||||
{
|
||||
id: "unk1",
|
||||
description:
|
||||
"Absolute baseline volumes and time period needed to compute complaint rate per unit",
|
||||
confidence: "low",
|
||||
},
|
||||
],
|
||||
plausibleInterpretations: [], // intentionally empty — evidence too thin
|
||||
},
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "Complaints increased by 35%",
|
||||
evidenceType: "reported_statement",
|
||||
source: "User input",
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
{
|
||||
id: "ev2",
|
||||
description: "Production increased by 40%",
|
||||
evidenceType: "reported_statement",
|
||||
source: "User input",
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
{
|
||||
id: "ev3",
|
||||
description:
|
||||
"Production growth rate (40%) exceeded complaint growth rate (35%), implying the denominator may have grown faster than complaints.",
|
||||
evidenceType: "inferred_relationship",
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What was the complaint rate per unit before and after the production increase?",
|
||||
targets: ["system"],
|
||||
reason:
|
||||
"Without normalising complaints by production volume, the absolute complaint count change is misleading. The rate per unit determines whether the situation improved, stayed stable, or worsened.",
|
||||
expectedInformationValue: "high",
|
||||
reasoningMode: "decompose_aggregate",
|
||||
},
|
||||
};
|
||||
|
||||
const result = reconstructionV2Schema.safeParse(input);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects output missing required fields", () => {
|
||||
const input = {
|
||||
inputClassification: { primaryType: "other" },
|
||||
reconstruction: {},
|
||||
evidence: [],
|
||||
nextQuestion: { id: "q1" },
|
||||
};
|
||||
|
||||
const result = reconstructionV2Schema.safeParse(input);
|
||||
expect(result.success).toBe(false);
|
||||
});
|
||||
|
||||
it("validates empty arrays for all reconstruction categories", () => {
|
||||
const input = {
|
||||
inputClassification: {
|
||||
primaryType: "other",
|
||||
classificationReason: "test",
|
||||
confidence: "low",
|
||||
},
|
||||
reconstruction: {
|
||||
summary: "empty test",
|
||||
actors: [],
|
||||
systemsOrObjects: [],
|
||||
expectedStates: [],
|
||||
observedStates: [],
|
||||
differences: [],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [],
|
||||
importantUnknowns: [],
|
||||
plausibleInterpretations: [],
|
||||
},
|
||||
evidence: [],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What is the production volume?",
|
||||
targets: ["system"],
|
||||
reason: "need baseline",
|
||||
expectedInformationValue: "medium",
|
||||
},
|
||||
};
|
||||
|
||||
const result = reconstructionV2Schema.safeParse(input);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("validates evidence distinguishing direct_observation from inferred_relationship", () => {
|
||||
const input = {
|
||||
inputClassification: {
|
||||
primaryType: "unexplained_change",
|
||||
classificationReason: "test",
|
||||
confidence: "low",
|
||||
},
|
||||
reconstruction: {
|
||||
summary: "test summary",
|
||||
actors: [],
|
||||
systemsOrObjects: [],
|
||||
expectedStates: [],
|
||||
observedStates: [{ id: "o1", description: "x", confidence: "high" }],
|
||||
differences: [],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [],
|
||||
importantUnknowns: [],
|
||||
plausibleInterpretations: [],
|
||||
},
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "Observed fact",
|
||||
evidenceType: "direct_observation",
|
||||
confidence: "high",
|
||||
importance: "critical",
|
||||
},
|
||||
{
|
||||
id: "ev2",
|
||||
description: "Derived relationship",
|
||||
evidenceType: "inferred_relationship",
|
||||
confidence: "medium",
|
||||
importance: "supporting",
|
||||
},
|
||||
],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What is the denominator?",
|
||||
targets: ["system"],
|
||||
reason: "need context",
|
||||
expectedInformationValue: "high",
|
||||
},
|
||||
};
|
||||
|
||||
const result = reconstructionV2Schema.safeParse(input);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// parseReconstructionV2 helper tests
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("parseReconstructionV2", () => {
|
||||
it("parses a valid v0.3-shaped JSON string", async () => {
|
||||
const fixture = {
|
||||
inputClassification: {
|
||||
primaryType: "unexplained_change",
|
||||
classificationReason: "test",
|
||||
confidence: "medium",
|
||||
},
|
||||
reconstruction: {
|
||||
summary: "both increased",
|
||||
actors: [],
|
||||
systemsOrObjects: [],
|
||||
expectedStates: [],
|
||||
observedStates: [
|
||||
{ id: "o1", description: "x rose 35%", confidence: "high" },
|
||||
{ id: "o2", description: "y rose 40%", confidence: "high" },
|
||||
],
|
||||
differences: [{ id: "d1", description: "y grew faster", confidence: "medium" }],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [],
|
||||
contradictions: [],
|
||||
importantUnknowns: [],
|
||||
plausibleInterpretations: [],
|
||||
},
|
||||
evidence: [
|
||||
{ id: "e1", description: "x rose 35%", evidenceType: "reported_statement", confidence: "medium", importance: "important" },
|
||||
{ id: "e2", description: "y rose 40%", evidenceType: "reported_statement", confidence: "medium", importance: "important" },
|
||||
],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What is the denominator?",
|
||||
targets: ["system"],
|
||||
reason: "need rate context",
|
||||
expectedInformationValue: "high",
|
||||
},
|
||||
};
|
||||
|
||||
const raw = JSON.stringify(fixture);
|
||||
const parsed = parseReconstructionV2(raw);
|
||||
|
||||
expect(parsed.inputClassification.primaryType).toBe("unexplained_change");
|
||||
expect(parsed.reconstruction.summary).toBe("both increased");
|
||||
expect(parsed.nextQuestion.question).toBe("What is the denominator?");
|
||||
});
|
||||
|
||||
it("rejects non-JSON string", () => {
|
||||
expect(() => parseReconstructionV2("{not valid json")).toThrow(SyntaxError);
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// v0.3 prompt contains required guidance text
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("v0.3 prompt guidance completeness", () => {
|
||||
it("mentions normalise counts when scale changed", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content.toLowerCase()).toMatch(/normali[sz]e|normalis[ei]ng/);
|
||||
});
|
||||
|
||||
it("mentions distinguishing total count from rate", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content.toLowerCase()).toContain("rate");
|
||||
expect(content.toLowerCase()).toMatch(/count.*not.*caus|correlation.*caus|distinguish.*count/);
|
||||
});
|
||||
|
||||
it("mentions avoiding correlation-as-causation", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content.toLowerCase()).toMatch(/correlation.*caus|treating.*correlation.*caus/);
|
||||
});
|
||||
|
||||
it("mentions prefer one narrow next question over compound", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
// Should mention single vs compound
|
||||
expect(content).toMatch(/exactly.*one|single.*question|Do NOT combine|combine.*multiple/i);
|
||||
});
|
||||
|
||||
it("mentions leaving empty interpretations when evidence is thin", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content).toMatch(/empty.*array|do not generate.*interpretation|fill a list/i);
|
||||
});
|
||||
|
||||
it("mentions identifying the denominator or exposure metric", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
expect(content.toLowerCase()).toMatch(/denominator|exposure/);
|
||||
});
|
||||
|
||||
it("uses the exact scenario text as a reference example only (not in rules)", async () => {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.3.md"),
|
||||
"utf-8",
|
||||
);
|
||||
// The prompt should be domain-independent — it should not mention specific industries as rules
|
||||
// but may have an example section. We verify the prompt does not hard-code a specific question text.
|
||||
expect(content).not.toMatch(/What was the complaint rate per unit before and after/);
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// Fixture: expected good structure for target scenario
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("target scenario fixture validation", () => {
|
||||
const goodFixture = JSON.parse(JSON.stringify({
|
||||
inputClassification: {
|
||||
primaryType: "unexplained_change",
|
||||
secondaryTypes: ["reported_claim"],
|
||||
reasoningModes: ["identify_difference", "decompose_aggregate"],
|
||||
classificationReason:
|
||||
"Two operational quantities changed at different percentages without a shared baseline or denominator.",
|
||||
confidence: "medium",
|
||||
},
|
||||
reconstruction: {
|
||||
summary:
|
||||
"Both complaint counts and production volumes increased, but production grew slightly faster than complaints — without absolute baselines the per-unit complaint rate cannot be determined.",
|
||||
actors: [],
|
||||
systemsOrObjects: [
|
||||
{ id: "so1", description: "Production system or output volume", confidence: "high" },
|
||||
{ id: "so2", description: "Complaint reporting mechanism", confidence: "high" },
|
||||
],
|
||||
expectedStates: [],
|
||||
observedStates: [
|
||||
{ id: "obs1", description: "Complaint count increased by 35%", confidence: "high" },
|
||||
{ id: "obs2", description: "Production volume increased by 40%", confidence: "high" },
|
||||
],
|
||||
differences: [
|
||||
{
|
||||
id: "diff1",
|
||||
description:
|
||||
"Production grew faster than complaints (+40% vs +35%), so the ratio of complaints per unit may have decreased or remained stable. The absolute complaint count alone is not a reliable indicator of whether conditions have changed.",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
knownTransitions: [],
|
||||
unexplainedTransitions: [
|
||||
{
|
||||
id: "ut1",
|
||||
description: "Complaint volume shifted to a higher level without explained cause",
|
||||
confidence: "medium",
|
||||
entity: "complaints_metric",
|
||||
previousState: "unknown baseline",
|
||||
currentState: "+35%",
|
||||
},
|
||||
],
|
||||
contradictions: [],
|
||||
importantUnknowns: [
|
||||
{
|
||||
id: "unk1",
|
||||
description:
|
||||
"Absolute complaint count and production volume baselines needed to compute the per-unit rate",
|
||||
confidence: "low",
|
||||
},
|
||||
{
|
||||
id: "unk2",
|
||||
description: "Time period over which these changes occurred",
|
||||
confidence: "low",
|
||||
},
|
||||
],
|
||||
plausibleInterpretations: [], // intentionally empty — no sufficient evidence for interpretations
|
||||
},
|
||||
evidence: [
|
||||
{
|
||||
id: "ev1",
|
||||
description: "Complaints increased by 35%",
|
||||
evidenceType: "reported_statement",
|
||||
source: "Scenario input",
|
||||
attribution: null,
|
||||
confidence: "high",
|
||||
importance: "important",
|
||||
},
|
||||
{
|
||||
id: "ev2",
|
||||
description: "Production increased by 40%",
|
||||
evidenceType: "reported_statement",
|
||||
source: "Scenario input",
|
||||
attribution: null,
|
||||
confidence: "high",
|
||||
importance: "important",
|
||||
},
|
||||
{
|
||||
id: "ev3",
|
||||
description: "Complaint count grew more slowly than production volume, suggesting per-unit rates may have improved or stayed stable.",
|
||||
evidenceType: "inferred_relationship",
|
||||
attribution: null,
|
||||
confidence: "medium",
|
||||
importance: "important",
|
||||
},
|
||||
],
|
||||
nextQuestion: {
|
||||
id: "q1",
|
||||
question: "What was the absolute complaint volume and production volume (or baseline) before these percentage changes?",
|
||||
targets: ["system", "measurement"],
|
||||
reason:
|
||||
"Without baseline counts to compute a rate per unit, we cannot determine whether conditions have worsened, stayed stable, or improved. The rate comparison is the smallest unresolved comparison needed to evaluate the situation.",
|
||||
expectedInformationValue: "high",
|
||||
reasoningMode: "decompose_aggregate",
|
||||
},
|
||||
}));
|
||||
|
||||
it("fixture validates against v0.3 schema", () => {
|
||||
const result = reconstructionV2Schema.safeParse(goodFixture);
|
||||
expect(result.success).toBe(true);
|
||||
});
|
||||
|
||||
it("fixture has exactly one next question with non-empty text", () => {
|
||||
expect(goodFixture.nextQuestion.question.length).toBeGreaterThan(10);
|
||||
expect(goodFixture.nextQuestion.reason.length).toBeGreaterThan(10);
|
||||
expect(goodFixture.nextQuestion.expectedInformationValue).toBe("high");
|
||||
});
|
||||
|
||||
it("fixture has empty plausibleInterpretations (evidence too thin)", () => {
|
||||
expect(goodFixture.reconstruction.plausibleInterpretations).toEqual([]);
|
||||
});
|
||||
|
||||
it("fixture evidence includes both direct observations and one inferred relationship", () => {
|
||||
const types = goodFixture.evidence.map((e) => e.evidenceType);
|
||||
expect(types).toContain("reported_statement");
|
||||
expect(types).toContain("inferred_relationship");
|
||||
});
|
||||
|
||||
it("fixture relationship notes complaint count grew more slowly than production", () => {
|
||||
const diffDescs = goodFixture.reconstruction.differences.map((d) => d.description);
|
||||
const found = diffDescs.some(
|
||||
(d) =>
|
||||
d.toLowerCase().includes("fast") ||
|
||||
d.toLowerCase().includes("slower") ||
|
||||
d.toLowerCase().includes("ratio") ||
|
||||
d.toLowerCase().includes("per-unit") ||
|
||||
d.toLowerCase().includes("per unit"),
|
||||
);
|
||||
expect(found).toBe(true);
|
||||
});
|
||||
|
||||
it("fixture does not assert quality deterioration", () => {
|
||||
const allText = [
|
||||
goodFixture.reconstruction.summary,
|
||||
...goodFixture.reconstruction.differences.map((d) => d.description),
|
||||
goodFixture.nextQuestion.reason,
|
||||
].join(" ").toLowerCase();
|
||||
// Should not contain strong deterioration language without caveats
|
||||
expect(allText).not.toMatch(/quality.*deteriorat|quality.*worsen|definitely.*bad/);
|
||||
});
|
||||
|
||||
it("fixture includes relationship that production grew faster", () => {
|
||||
const allText = [
|
||||
goodFixture.reconstruction.summary,
|
||||
...goodFixture.reconstruction.differences.map((d) => d.description),
|
||||
].join(" ").toLowerCase();
|
||||
expect(allText).toMatch(/produ.*grow|ratio|per-unit|per unit|\+40.*\+35/);
|
||||
});
|
||||
});
|
||||
|
||||
// ──────────────────────────────────────────────
|
||||
// Diagnostics: prompt version tracking
|
||||
// ──────────────────────────────────────────────
|
||||
|
||||
describe("diagnostics prompt version", () => {
|
||||
it("DEFAULT_PROMPT_VERSION is exported correctly", () => {
|
||||
expect(DEFAULT_PROMPT_VERSION).toBe("v0.3");
|
||||
});
|
||||
|
||||
it("PROMPT_VERSIONS includes both v0.2 and v0.3", () => {
|
||||
const hasV2 = PROMPT_VERSIONS.includes("v0.2");
|
||||
const hasV3 = PROMPT_VERSIONS.includes("v0.3");
|
||||
expect(hasV2).toBe(true);
|
||||
expect(hasV3).toBe(true);
|
||||
});
|
||||
|
||||
it("RECONSTRUCTION_PROMPT_VERSION env var overrides default", async () => {
|
||||
// The actual override happens at module load time, so we can't easily test this
|
||||
// in isolation. Instead, verify the constant reflects env or defaults to v0.3.
|
||||
expect(PROMPT_VERSIONS).toContain("v0.2");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user