From 79ea2f68241215da5c032c6860d158bb3786a2c1 Mon Sep 17 00:00:00 2001 From: robbond Date: Sat, 1 Aug 2026 15:39:30 +0100 Subject: [PATCH 01/12] feat: add v0.3 normalised comparison reasoning Add explicit reasoning guidance for normalising counts by exposure/denominator, distinguishing total count from rate, and avoiding correlation-as-causation errors. Changes: - prompts/reconstruct-v0.3.md: new prompt with normalisation discipline - lib/reconstruction/prompt.js: v0.3 loader + env var override support - lib/analysis.js: defer DEFAULT_PROMPT_VERSION to prompt module (defaults to v0.3) - PROMPT_VERSIONS extended to [v0.1, v0.2, v0.3] - tests/v03-reasoning.test.js: 34 focused tests covering prompt loading, schema validation, guidance completeness, and target scenario fixture - playwright.config.js + tests/smoke.test.js: minimal UI smoke test for browser rendering - package.json: add @playwright/test as devDependency Default switches to v0.3; v0.2 selectable via promptVersion or RECONSTRUCTION_PROMPT_VERSION env var. --- lib/analysis.js | 3 +- lib/reconstruction/prompt.js | 32 +- package-lock.json | 68 +++- package.json | 1 + playwright.config.js | 5 + prompts/reconstruct-v0.3.md | 160 ++++++++++ tests/smoke.test.js | 55 ++++ tests/v03-reasoning.test.js | 598 +++++++++++++++++++++++++++++++++++ 8 files changed, 914 insertions(+), 8 deletions(-) create mode 100644 playwright.config.js create mode 100644 prompts/reconstruct-v0.3.md create mode 100644 tests/smoke.test.js create mode 100644 tests/v03-reasoning.test.js diff --git a/lib/analysis.js b/lib/analysis.js index 8741cf1..d214dca 100644 --- a/lib/analysis.js +++ b/lib/analysis.js @@ -5,14 +5,13 @@ import { getConfig } from "../lib/config.js"; import { getProvider } from "../lib/llm/provider.js"; -import { buildPrompt, PROMPT_VERSIONS } from "../lib/reconstruction/prompt.js"; +import { buildPrompt, PROMPT_VERSIONS, DEFAULT_PROMPT_VERSION } from "../lib/reconstruction/prompt.js"; import { reconstructionV2Schema, reconstructionSchema as reconstructionV1Schema, } from "../lib/reconstruction/schema.js"; const MAX_SCENARIO_LENGTH = 10000; -const DEFAULT_PROMPT_VERSION = "v0.2"; /** * Analyse a scenario string through the full pipeline. diff --git a/lib/reconstruction/prompt.js b/lib/reconstruction/prompt.js index 66edeab..5de4609 100644 --- a/lib/reconstruction/prompt.js +++ b/lib/reconstruction/prompt.js @@ -7,7 +7,14 @@ const __dirname = dirname(__filename); const PROMPTS_DIR = join(__dirname, "../../prompts"); /** Available prompt versions */ -export const PROMPT_VERSIONS = ["v0.1", "v0.2"]; +export const PROMPT_VERSIONS = ["v0.1", "v0.2", "v0.3"]; + +/** Default prompt version (override via RECONSTRUCTION_PROMPT_VERSION env var) */ +const defaultVersionFromEnv = process.env.RECONSTRUCTION_PROMPT_VERSION; +export const DEFAULT_PROMPT_VERSION = + defaultVersionFromEnv && PROMPT_VERSIONS.includes(defaultVersionFromEnv) + ? defaultVersionFromEnv + : "v0.3"; /** Build a v0.1 (extraction-only) prompt inline for backward compatibility */ function buildV1Prompt(scenario) { @@ -56,20 +63,37 @@ async function buildV2Prompt(scenario) { } } +/** Load a versioned prompt from disk and substitute {{SCENARIO}} */ +async function buildV3Prompt(scenario) { + try { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + return content.replace("{{SCENARIO}}", scenario); + } catch { + // Fall back to v0.2 prompt if v0.3 file is missing + return buildV2Prompt(scenario); + } +} + /** * Build an analysis prompt for the given version. - * @param {"v0.1" | "v0.2"} [version="v0.2"] + * @param {"v0.1" | "v0.2" | "v0.3"} [version="v0.3"] * @returns {Promise<{prompt: string, version: string}>} */ -export async function buildPrompt(scenario, version = "v0.2") { +export async function buildPrompt(scenario, version = "v0.3") { let prompt; switch (version) { case "v0.1": prompt = buildV1Prompt(scenario); break; - default: // v0.2 + case "v0.2": prompt = await buildV2Prompt(scenario); break; + default: // v0.3 + prompt = await buildV3Prompt(scenario); + break; } const strongJsonHint = diff --git a/package-lock.json b/package-lock.json index 9ec9502..f089f7d 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "confidence-engine", - "version": "0.1.0", + "version": "0.2.0-experimental", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "confidence-engine", - "version": "0.1.0", + "version": "0.2.0-experimental", "dependencies": { "next": "^14.2.0", "react": "^18.3.0", @@ -14,6 +14,7 @@ "zod": "^3.23.0" }, "devDependencies": { + "@playwright/test": "^1.62.1", "@types/node": "^20.14.0", "@types/react": "^18.3.0", "@types/react-dom": "^18.3.0", @@ -888,6 +889,22 @@ "node": ">=14" } }, + "node_modules/@playwright/test": { + "version": "1.62.1", + "resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.62.1.tgz", + "integrity": "sha512-DTcUc8qii+cpHvtOwggMtBRMjKZHXYWdw8syRYu2vtzuq4Wxphqq4NfCs5Zt44L6mA8rfDfj+PHnxFc/FeK6mQ==", + "devOptional": true, + "license": "Apache-2.0", + "dependencies": { + "playwright": "1.62.1" + }, + "bin": { + "playwright": "cli.js" + }, + "engines": { + "node": ">=20" + } + }, "node_modules/@rollup/rollup-android-arm-eabi": { "version": "4.62.3", "resolved": "https://registry.npmjs.org/@rollup/rollup-android-arm-eabi/-/rollup-android-arm-eabi-4.62.3.tgz", @@ -5601,6 +5618,53 @@ "node": ">= 6" } }, + "node_modules/playwright": { + "version": "1.62.1", + "resolved": "https://registry.npmjs.org/playwright/-/playwright-1.62.1.tgz", + "integrity": "sha512-0M+L3LAD8/nm554LOla9Ayx0j0tmFZ0FBcoQ7F1VuVHpM/XpiC8RcDzBQB8W5+hA8L22THxELzeF+2WcUzvcLg==", + "devOptional": true, + "license": "Apache-2.0", + "dependencies": { + "playwright-core": "1.62.1" + }, + "bin": { + "playwright": "cli.js" + }, + "engines": { + "node": ">=20" + }, + "optionalDependencies": { + "fsevents": "2.3.2" + } + }, + "node_modules/playwright-core": { + "version": "1.62.1", + "resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.62.1.tgz", + "integrity": "sha512-wPYSwEBJY9GHraISXqyqtx0na0LpO3XEX7jNDhntbex7tzUS7kLnZsOlFruFJB4Hi/rhDMjXGqHewDZ68nYZVw==", + "devOptional": true, + "license": "Apache-2.0", + "bin": { + "playwright-core": "cli.js" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/playwright/node_modules/fsevents": { + "version": "2.3.2", + "resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz", + "integrity": "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==", + "dev": true, + "hasInstallScript": true, + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^8.16.0 || ^10.6.0 || >=11.0.0" + } + }, "node_modules/possible-typed-array-names": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/possible-typed-array-names/-/possible-typed-array-names-1.1.0.tgz", diff --git a/package.json b/package.json index 4f1c4e6..6dca8bf 100644 --- a/package.json +++ b/package.json @@ -19,6 +19,7 @@ "zod": "^3.23.0" }, "devDependencies": { + "@playwright/test": "^1.62.1", "@types/node": "^20.14.0", "@types/react": "^18.3.0", "@types/react-dom": "^18.3.0", diff --git a/playwright.config.js b/playwright.config.js new file mode 100644 index 0000000..03dfb3a --- /dev/null +++ b/playwright.config.js @@ -0,0 +1,5 @@ +import { defineConfig } from "@playwright/test"; +export default defineConfig({ + use: { headless: true, screenshot: "only-on-failure", actionTimeout: 120000 }, + testMatch: "**/tests/smoke.test.js", +}); diff --git a/prompts/reconstruct-v0.3.md b/prompts/reconstruct-v0.3.md new file mode 100644 index 0000000..0763025 --- /dev/null +++ b/prompts/reconstruct-v0.3.md @@ -0,0 +1,160 @@ +You are a neutral analyst performing evidence-based situation reconstruction. + +## Rules + +1. Do NOT invent facts, context or causes. Only include information present in the scenario or clearly implied. +2. First determine what kind of input has been supplied. Use only these classification types: + observed_problem, unexplained_change, contradiction, decision_request, causal_claim, + reported_claim, fault_report, ambiguous_statement, question, desired_outcome, + insufficient_context, other +3. Choose reasoning modes from: + establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate, + validate_measurement, validate_claim, investigate_contradiction, clarify_meaning, + decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other +4. Look for anchors: actor, system or object, expected outcome, observed outcome, + previous state, current state, difference between groups, change over time, measurement, + evidence source, proposed action. +5. Identify meaningful differences (e.g., some succeed while others fail; revenue rises while cash falls). +6. Keep multiple plausible interpretations separate where the evidence does not distinguish them. +7. Distinguish: what was said / what it may mean / why it may have been said. +8. If input is too ambiguous or contains no useful operational anchors, say so and ask for + the single piece of context that would best distinguish plausible interpretations. + +## Normalisation and rate reasoning (apply whenever applicable) + +When the scenario mentions counts, totals, frequencies, or volumes alongside changes in +scale, volume, exposure, time, population, or output: + +- ALWAYS consider whether a denominator or exposure metric is needed to normalise the count. +- Distinguish between absolute count (total number observed) and rate (count per unit of exposure). +- Two metrics rising at similar percentages does NOT imply that quality, performance, or safety + has worsened — production growth may outpace complaint growth, meaning the per-unit rate + could be stable or even improved. +- Identify the possible denominator explicitly (e.g., "per unit produced", "per customer served", + "per hour of operation"). +- State clearly: "The absolute count changed by X%, but without knowing the denominator we cannot + determine whether the rate per unit has worsened, stayed stable, or improved." +- Avoid treating correlation between two rising counts as evidence of a causal relationship. + +## Interpretation discipline + +- Do NOT generate plausible interpretations merely to fill a list. If the evidence does not + support useful, distinct interpretations, return an empty array []. +- Only include an interpretation when there is specific evidence that makes it distinguishable + from alternatives and worth evaluating further. +- Rank all reconstruction details by importance: + - critical: essential to resolving the situation; without it conclusions cannot be drawn + - important: materially affects understanding of the situation + - supporting: adds context but not critical + - incidental: minor detail, unlikely to affect conclusions + +## Next question discipline + +- Generate exactly ONE next question. Do NOT combine multiple questions. +- The first and only question should target the single most useful missing comparison or data point. +- Prefer narrow, specific questions over broad compound questions. +- When counts have changed alongside scale/exposure, the highest-value question typically targets + the rate-per-unit or equivalent normalised metric. +- Do NOT generate speculative interpretations merely to justify a question. + +## Confidence scale + +- low — weak evidence, speculation, or missing information +- medium — reasonable inference from available evidence +- high — strong evidence, direct observation, or confirmed fact + +## Importance scale (evidence records) + +- incidental — minor detail, unlikely to affect conclusions +- supporting — adds context but not critical +- important — materially affects understanding of the situation +- critical — essential to resolving the situation; without it conclusions cannot be drawn + +## Expected information value (next question) + +- low — marginally useful even if answered +- medium — meaningfully clarifies the situation +- high — would significantly distinguish between plausible explanations or fill a gap in understanding + +## Next question selection criteria + +Prefer questions that: +- clarify a major difference +- establish a baseline +- explain an important transition +- test an unsupported claim +- distinguish between plausible explanations +- request measurable evidence +- identify who or what is affected +- establish timing + +Avoid questions that: +- have already been answered +- assume a cause +- jump to a solution +- ask about motive before the observable situation is understood +- focus on incidental wording +- are too broad to produce useful information +- combine many unrelated questions + +## Output format — return this exact JSON structure + +Return a JSON object with exactly these four top-level keys (use **camelCase**): + +```json +{ + "inputClassification": { + "primaryType": "", + "secondaryTypes": [""], + "reasoningModes": [""], + "classificationReason": "", + "confidence": "" + }, + "reconstruction": { + "summary": "", + "actors": [{"id": "", "description": "...", "confidence": ""}], + "systemsOrObjects": [{"id": "", "description": "...", "confidence": ""}], + "expectedStates": [{"id": "...", "description": "...", "confidence": ""}], + "observedStates": [{"id": "...", "description": "...", "confidence": ""}], + "differences": [{"id": "...", "description": "...", "confidence": ""}], + "knownTransitions": [{"id": "...", "description": "...", "confidence": "", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}], + "unexplainedTransitions": [{"id": "...", "description": "...", "confidence": "", "entity": "...", "previousState": "...", "currentState": "..."}], + "contradictions": [{"id": "...", "description": "...", "confidence": ""}], + "importantUnknowns": [{"id": "...", "description": "...", "confidence": ""}], + "plausibleInterpretations": [{"id": "...", "description": "...", "supportingEvidenceIds": [""], "assumptionsRequired": [], "confidence": ""}] + }, + "evidence": [ + { + "id": "", + "description": "...", + "evidenceType": "", + "source": "", + "attribution": null, + "confidence": "", + "importance": "" + } + ], + "nextQuestion": { + "id": "", + "question": "", + "targets": [""], + "reason": "", + "expectedInformationValue": "", + "reasoningMode": "" + } +} +``` + +CRITICAL RULES for JSON output: +1. Use **exactly** the key names shown above (camelCase, no snake_case). +2. The four top-level keys must be: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`. +3. Do NOT invent new top-level keys (no `anchors`, `confidence` at top level, `meaningful_differences`, etc.). +4. Keep `actors`, `systemsOrObjects`, `expectedStates`, `observedStates`, `differences`, `contradictions`, `importantUnknowns` as arrays even if empty: []. +5. Keep `plausibleInterpretations` as an array (can be []), same for `knownTransitions` and `unexplainedTransitions`. +6. Each object in arrays must have at least `id`, `description`, `confidence`. +7. **evidenceType**: classify each evidence item clearly as either a direct observation, a reported statement, an interpretation, an assumption, or an inferred relationship. Do not treat raw counts as proof of causal relationships — they may be inferred relationships only when supported by explicit reasoning about denominators or rates. + +Scenario: +{{SCENARIO}} + +Return ONLY the JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks. diff --git a/tests/smoke.test.js b/tests/smoke.test.js new file mode 100644 index 0000000..f5cef5e --- /dev/null +++ b/tests/smoke.test.js @@ -0,0 +1,55 @@ +import { test, expect } from "@playwright/test"; + +test("v0.3 UI smoke test with live model response", async ({ page }) => { + await page.goto("http://localhost:3000"); + + // Page should load without error + await expect(page.getByText(/Confidence Engine/i)).toBeVisible(); + + // Type the scenario + const textarea = page.locator("textarea[placeholder*='Describe']"); + await textarea.fill("Complaints increased by 35% while production increased by 40%."); + + // Button should be enabled + await expect(page.getByRole("button", { name: /Analyse/i })).toBeEnabled(); + + // Click Analyse and wait for diagnostics panel + await page.getByRole("button", { name: /Analyse/i }).click(); + + // Wait for result section (ReconstructionView rendered) + await expect(page.getByRole("heading", { name: /Next Question/i })).toBeVisible({ timeout: 180000 }); + + // Take screenshot of result page + await page.screenshot({ path: "tests-results/smoke-v0.3.png", fullPage: true }); + + // Verify diagnostics panel exists and contains relevant info + const diagPanel = page.locator('details summary').first(); + if (await diagPanel.isVisible()) { + console.log("Raw response viewer:", await diagPanel.innerText().catch(() => "not visible")); + } + + // Get full body text for verification + const bodyText = await page.locator("body").innerText(); + + console.log("\n=== UI Smoke Test Results ==="); + console.log("Page title:", await page.title()); + console.log("Body content length:", bodyText.length); + + // Check key content indicators + const hasNextQ = bodyText.includes("Next Question"); + const hasComplaints = bodyText.includes("Complaint") || bodyText.includes("complaint"); + const hasProduction = bodyText.includes("production") || bodyText.includes("Production"); + const hasRateContext = bodyText.toLowerCase().includes("rate") || + bodyText.toLowerCase().includes("unit") || + bodyText.toLowerCase().includes("denominator") || + bodyText.toLowerCase().includes("per-unit"); + + console.log("Has Next Question heading:", hasNextQ); + console.log("Has complaints reference:", hasComplaints); + console.log("Has production reference:", hasProduction); + console.log("Has rate context (rate/unit/denominator):", hasRateContext); + + // Basic structural checks + expect(bodyText.length).toBeGreaterThan(200); + expect(hasNextQ).toBe(true); +}, { timeout: 300000 }); diff --git a/tests/v03-reasoning.test.js b/tests/v03-reasoning.test.js new file mode 100644 index 0000000..de40c6f --- /dev/null +++ b/tests/v03-reasoning.test.js @@ -0,0 +1,598 @@ +import { describe, it, expect } from "vitest"; +import { promises as fs } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; +import { + PROMPT_VERSIONS, + buildPrompt, + DEFAULT_PROMPT_VERSION, +} from "@/lib/reconstruction/prompt.js"; +import { + reconstructionV2Schema, + parseReconstructionV2, +} from "@/lib/reconstruction/schema.js"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = dirname(__filename); +const PROMPTS_DIR = join(__dirname, "../prompts"); + +// ────────────────────────────────────────────── +// v0.3 prompt loading tests +// ────────────────────────────────────────────── + +describe("v0.3 prompt", () => { + it("v0.3 is in PROMPT_VERSIONS", () => { + expect(PROMPT_VERSIONS).toContain("v0.3"); + }); + + it("DEFAULT_PROMPT_VERSION is v0.3 on this branch", () => { + expect(DEFAULT_PROMPT_VERSION).toBe("v0.3"); + }); + + it("v0.2 remains available in PROMPT_VERSIONS", () => { + expect(PROMPT_VERSIONS).toContain("v0.2"); + }); + + it("v0.3 prompt file loads from disk", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(typeof content).toBe("string"); + expect(content.length).toBeGreaterThan(500); + }); + + it("v0.3 prompt contains normalisation guidance", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content.toLowerCase()).toContain("normalise"); + expect(content.toLowerCase()).toContain("rate"); + expect(content.toLowerCase()).toContain("denominator") || + expect(content.toLowerCase()).toContain("exposure"); + }); + + it("v0.3 prompt contains discipline guidance", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + // Should mention not generating speculative interpretations + expect(content).toMatch(/interpretation/i); + // Should mention one question discipline + expect(content).toMatch(/exactly.*one.*question|one.*only.*question|single.*question/i) || + expect(content).toMatch(/Do NOT combine/i); + }); + + it("buildPrompt returns v0.3 prompt with scenario substituted", async () => { + const result = await buildPrompt("Test scenario text", "v0.3"); + expect(result.version).toBe("v0.3"); + expect(result.prompt).toContain("Test scenario text"); + // Should contain the normalisation section guidance + expect(result.prompt.toLowerCase()).toContain("normalise"); + }); + + it("buildPrompt returns v0.2 prompt when requested", async () => { + const result = await buildPrompt("Test scenario text", "v0.2"); + expect(result.version).toBe("v0.2"); + expect(result.prompt).toContain("Test scenario text"); + }); + + it("buildPrompt default is v0.3", async () => { + const result = await buildPrompt("Test scenario text"); + expect(result.version).toBe("v0.3"); + }); +}); + +// ────────────────────────────────────────────── +// v0.2 prompt still works +// ────────────────────────────────────────────── + +describe("v0.2 backward compatibility", () => { + it("v0.2 prompt file exists and loads", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.2.md"), + "utf-8", + ); + expect(typeof content).toBe("string"); + expect(content.length).toBeGreaterThan(500); + }); + + it("buildPrompt returns v0.2 version string", async () => { + const result = await buildPrompt("test", "v0.2"); + expect(result.version).toBe("v0.2"); + }); +}); + +// ────────────────────────────────────────────── +// Schema validation tests for v0.3-shaped output +// ────────────────────────────────────────────── + +describe("v0.3 schema validation", () => { + it("validates a complete valid reconstruction with empty interpretations", () => { + const input = { + inputClassification: { + primaryType: "unexplained_change", + secondaryTypes: ["reported_claim"], + reasoningModes: ["identify_difference"], + classificationReason: "Two metrics changed without explanation.", + confidence: "medium", + }, + reconstruction: { + summary: "Both complaints and production increased.", + actors: [], + systemsOrObjects: [ + { id: "complaints_metric", description: "Volume of complaints", confidence: "high" }, + ], + expectedStates: [], + observedStates: [ + { id: "obs1", description: "Complaint volume rose by 35%", confidence: "medium" }, + { id: "obs2", description: "Production volume rose by 40%", confidence: "medium" }, + ], + differences: [ + { + id: "diff1", + description: + "Production grew faster than complaints, so the complaint-to-production ratio may have improved.", + confidence: "medium", + }, + ], + knownTransitions: [], + unexplainedTransitions: [ + { + id: "trans1", + description: "Complaint volume shifted to a higher level without explained cause", + confidence: "medium", + entity: "complaints_metric", + previousState: "Baseline volume (unknown)", + currentState: "+35% increase", + }, + ], + contradictions: [], + importantUnknowns: [ + { + id: "unk1", + description: + "Absolute baseline volumes and time period needed to compute complaint rate per unit", + confidence: "low", + }, + ], + plausibleInterpretations: [], // intentionally empty — evidence too thin + }, + evidence: [ + { + id: "ev1", + description: "Complaints increased by 35%", + evidenceType: "reported_statement", + source: "User input", + attribution: null, + confidence: "medium", + importance: "important", + }, + { + id: "ev2", + description: "Production increased by 40%", + evidenceType: "reported_statement", + source: "User input", + attribution: null, + confidence: "medium", + importance: "important", + }, + { + id: "ev3", + description: + "Production growth rate (40%) exceeded complaint growth rate (35%), implying the denominator may have grown faster than complaints.", + evidenceType: "inferred_relationship", + attribution: null, + confidence: "medium", + importance: "important", + }, + ], + nextQuestion: { + id: "q1", + question: "What was the complaint rate per unit before and after the production increase?", + targets: ["system"], + reason: + "Without normalising complaints by production volume, the absolute complaint count change is misleading. The rate per unit determines whether the situation improved, stayed stable, or worsened.", + expectedInformationValue: "high", + reasoningMode: "decompose_aggregate", + }, + }; + + const result = reconstructionV2Schema.safeParse(input); + expect(result.success).toBe(true); + }); + + it("rejects output missing required fields", () => { + const input = { + inputClassification: { primaryType: "other" }, + reconstruction: {}, + evidence: [], + nextQuestion: { id: "q1" }, + }; + + const result = reconstructionV2Schema.safeParse(input); + expect(result.success).toBe(false); + }); + + it("validates empty arrays for all reconstruction categories", () => { + const input = { + inputClassification: { + primaryType: "other", + classificationReason: "test", + confidence: "low", + }, + reconstruction: { + summary: "empty test", + actors: [], + systemsOrObjects: [], + expectedStates: [], + observedStates: [], + differences: [], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [], + importantUnknowns: [], + plausibleInterpretations: [], + }, + evidence: [], + nextQuestion: { + id: "q1", + question: "What is the production volume?", + targets: ["system"], + reason: "need baseline", + expectedInformationValue: "medium", + }, + }; + + const result = reconstructionV2Schema.safeParse(input); + expect(result.success).toBe(true); + }); + + it("validates evidence distinguishing direct_observation from inferred_relationship", () => { + const input = { + inputClassification: { + primaryType: "unexplained_change", + classificationReason: "test", + confidence: "low", + }, + reconstruction: { + summary: "test summary", + actors: [], + systemsOrObjects: [], + expectedStates: [], + observedStates: [{ id: "o1", description: "x", confidence: "high" }], + differences: [], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [], + importantUnknowns: [], + plausibleInterpretations: [], + }, + evidence: [ + { + id: "ev1", + description: "Observed fact", + evidenceType: "direct_observation", + confidence: "high", + importance: "critical", + }, + { + id: "ev2", + description: "Derived relationship", + evidenceType: "inferred_relationship", + confidence: "medium", + importance: "supporting", + }, + ], + nextQuestion: { + id: "q1", + question: "What is the denominator?", + targets: ["system"], + reason: "need context", + expectedInformationValue: "high", + }, + }; + + const result = reconstructionV2Schema.safeParse(input); + expect(result.success).toBe(true); + }); +}); + +// ────────────────────────────────────────────── +// parseReconstructionV2 helper tests +// ────────────────────────────────────────────── + +describe("parseReconstructionV2", () => { + it("parses a valid v0.3-shaped JSON string", async () => { + const fixture = { + inputClassification: { + primaryType: "unexplained_change", + classificationReason: "test", + confidence: "medium", + }, + reconstruction: { + summary: "both increased", + actors: [], + systemsOrObjects: [], + expectedStates: [], + observedStates: [ + { id: "o1", description: "x rose 35%", confidence: "high" }, + { id: "o2", description: "y rose 40%", confidence: "high" }, + ], + differences: [{ id: "d1", description: "y grew faster", confidence: "medium" }], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [], + importantUnknowns: [], + plausibleInterpretations: [], + }, + evidence: [ + { id: "e1", description: "x rose 35%", evidenceType: "reported_statement", confidence: "medium", importance: "important" }, + { id: "e2", description: "y rose 40%", evidenceType: "reported_statement", confidence: "medium", importance: "important" }, + ], + nextQuestion: { + id: "q1", + question: "What is the denominator?", + targets: ["system"], + reason: "need rate context", + expectedInformationValue: "high", + }, + }; + + const raw = JSON.stringify(fixture); + const parsed = parseReconstructionV2(raw); + + expect(parsed.inputClassification.primaryType).toBe("unexplained_change"); + expect(parsed.reconstruction.summary).toBe("both increased"); + expect(parsed.nextQuestion.question).toBe("What is the denominator?"); + }); + + it("rejects non-JSON string", () => { + expect(() => parseReconstructionV2("{not valid json")).toThrow(SyntaxError); + }); +}); + +// ────────────────────────────────────────────── +// v0.3 prompt contains required guidance text +// ────────────────────────────────────────────── + +describe("v0.3 prompt guidance completeness", () => { + it("mentions normalise counts when scale changed", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content.toLowerCase()).toMatch(/normali[sz]e|normalis[ei]ng/); + }); + + it("mentions distinguishing total count from rate", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content.toLowerCase()).toContain("rate"); + expect(content.toLowerCase()).toMatch(/count.*not.*caus|correlation.*caus|distinguish.*count/); + }); + + it("mentions avoiding correlation-as-causation", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content.toLowerCase()).toMatch(/correlation.*caus|treating.*correlation.*caus/); + }); + + it("mentions prefer one narrow next question over compound", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + // Should mention single vs compound + expect(content).toMatch(/exactly.*one|single.*question|Do NOT combine|combine.*multiple/i); + }); + + it("mentions leaving empty interpretations when evidence is thin", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content).toMatch(/empty.*array|do not generate.*interpretation|fill a list/i); + }); + + it("mentions identifying the denominator or exposure metric", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + expect(content.toLowerCase()).toMatch(/denominator|exposure/); + }); + + it("uses the exact scenario text as a reference example only (not in rules)", async () => { + const content = await fs.readFile( + join(PROMPTS_DIR, "reconstruct-v0.3.md"), + "utf-8", + ); + // The prompt should be domain-independent — it should not mention specific industries as rules + // but may have an example section. We verify the prompt does not hard-code a specific question text. + expect(content).not.toMatch(/What was the complaint rate per unit before and after/); + }); +}); + +// ────────────────────────────────────────────── +// Fixture: expected good structure for target scenario +// ────────────────────────────────────────────── + +describe("target scenario fixture validation", () => { + const goodFixture = JSON.parse(JSON.stringify({ + inputClassification: { + primaryType: "unexplained_change", + secondaryTypes: ["reported_claim"], + reasoningModes: ["identify_difference", "decompose_aggregate"], + classificationReason: + "Two operational quantities changed at different percentages without a shared baseline or denominator.", + confidence: "medium", + }, + reconstruction: { + summary: + "Both complaint counts and production volumes increased, but production grew slightly faster than complaints — without absolute baselines the per-unit complaint rate cannot be determined.", + actors: [], + systemsOrObjects: [ + { id: "so1", description: "Production system or output volume", confidence: "high" }, + { id: "so2", description: "Complaint reporting mechanism", confidence: "high" }, + ], + expectedStates: [], + observedStates: [ + { id: "obs1", description: "Complaint count increased by 35%", confidence: "high" }, + { id: "obs2", description: "Production volume increased by 40%", confidence: "high" }, + ], + differences: [ + { + id: "diff1", + description: + "Production grew faster than complaints (+40% vs +35%), so the ratio of complaints per unit may have decreased or remained stable. The absolute complaint count alone is not a reliable indicator of whether conditions have changed.", + confidence: "high", + }, + ], + knownTransitions: [], + unexplainedTransitions: [ + { + id: "ut1", + description: "Complaint volume shifted to a higher level without explained cause", + confidence: "medium", + entity: "complaints_metric", + previousState: "unknown baseline", + currentState: "+35%", + }, + ], + contradictions: [], + importantUnknowns: [ + { + id: "unk1", + description: + "Absolute complaint count and production volume baselines needed to compute the per-unit rate", + confidence: "low", + }, + { + id: "unk2", + description: "Time period over which these changes occurred", + confidence: "low", + }, + ], + plausibleInterpretations: [], // intentionally empty — no sufficient evidence for interpretations + }, + evidence: [ + { + id: "ev1", + description: "Complaints increased by 35%", + evidenceType: "reported_statement", + source: "Scenario input", + attribution: null, + confidence: "high", + importance: "important", + }, + { + id: "ev2", + description: "Production increased by 40%", + evidenceType: "reported_statement", + source: "Scenario input", + attribution: null, + confidence: "high", + importance: "important", + }, + { + id: "ev3", + description: "Complaint count grew more slowly than production volume, suggesting per-unit rates may have improved or stayed stable.", + evidenceType: "inferred_relationship", + attribution: null, + confidence: "medium", + importance: "important", + }, + ], + nextQuestion: { + id: "q1", + question: "What was the absolute complaint volume and production volume (or baseline) before these percentage changes?", + targets: ["system", "measurement"], + reason: + "Without baseline counts to compute a rate per unit, we cannot determine whether conditions have worsened, stayed stable, or improved. The rate comparison is the smallest unresolved comparison needed to evaluate the situation.", + expectedInformationValue: "high", + reasoningMode: "decompose_aggregate", + }, + })); + + it("fixture validates against v0.3 schema", () => { + const result = reconstructionV2Schema.safeParse(goodFixture); + expect(result.success).toBe(true); + }); + + it("fixture has exactly one next question with non-empty text", () => { + expect(goodFixture.nextQuestion.question.length).toBeGreaterThan(10); + expect(goodFixture.nextQuestion.reason.length).toBeGreaterThan(10); + expect(goodFixture.nextQuestion.expectedInformationValue).toBe("high"); + }); + + it("fixture has empty plausibleInterpretations (evidence too thin)", () => { + expect(goodFixture.reconstruction.plausibleInterpretations).toEqual([]); + }); + + it("fixture evidence includes both direct observations and one inferred relationship", () => { + const types = goodFixture.evidence.map((e) => e.evidenceType); + expect(types).toContain("reported_statement"); + expect(types).toContain("inferred_relationship"); + }); + + it("fixture relationship notes complaint count grew more slowly than production", () => { + const diffDescs = goodFixture.reconstruction.differences.map((d) => d.description); + const found = diffDescs.some( + (d) => + d.toLowerCase().includes("fast") || + d.toLowerCase().includes("slower") || + d.toLowerCase().includes("ratio") || + d.toLowerCase().includes("per-unit") || + d.toLowerCase().includes("per unit"), + ); + expect(found).toBe(true); + }); + + it("fixture does not assert quality deterioration", () => { + const allText = [ + goodFixture.reconstruction.summary, + ...goodFixture.reconstruction.differences.map((d) => d.description), + goodFixture.nextQuestion.reason, + ].join(" ").toLowerCase(); + // Should not contain strong deterioration language without caveats + expect(allText).not.toMatch(/quality.*deteriorat|quality.*worsen|definitely.*bad/); + }); + + it("fixture includes relationship that production grew faster", () => { + const allText = [ + goodFixture.reconstruction.summary, + ...goodFixture.reconstruction.differences.map((d) => d.description), + ].join(" ").toLowerCase(); + expect(allText).toMatch(/produ.*grow|ratio|per-unit|per unit|\+40.*\+35/); + }); +}); + +// ────────────────────────────────────────────── +// Diagnostics: prompt version tracking +// ────────────────────────────────────────────── + +describe("diagnostics prompt version", () => { + it("DEFAULT_PROMPT_VERSION is exported correctly", () => { + expect(DEFAULT_PROMPT_VERSION).toBe("v0.3"); + }); + + it("PROMPT_VERSIONS includes both v0.2 and v0.3", () => { + const hasV2 = PROMPT_VERSIONS.includes("v0.2"); + const hasV3 = PROMPT_VERSIONS.includes("v0.3"); + expect(hasV2).toBe(true); + expect(hasV3).toBe(true); + }); + + it("RECONSTRUCTION_PROMPT_VERSION env var overrides default", async () => { + // The actual override happens at module load time, so we can't easily test this + // in isolation. Instead, verify the constant reflects env or defaults to v0.3. + expect(PROMPT_VERSIONS).toContain("v0.2"); + }); +}); From 3c1362d8a12f4b642f7f2951c26f356d97186653 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 06:52:26 +0100 Subject: [PATCH 02/12] feat: add initial situation graph orchestration --- app/api/cases/start/route.js | 38 +++++ lib/graph/orchestrator.js | 125 ++++++++++++++++ tests/app/api/cases-start-route.test.js | 114 +++++++++++++++ tests/graph/orchestrator.test.js | 181 ++++++++++++++++++++++++ 4 files changed, 458 insertions(+) create mode 100644 app/api/cases/start/route.js create mode 100644 lib/graph/orchestrator.js create mode 100644 tests/app/api/cases-start-route.test.js create mode 100644 tests/graph/orchestrator.test.js diff --git a/app/api/cases/start/route.js b/app/api/cases/start/route.js new file mode 100644 index 0000000..ac8ad6b --- /dev/null +++ b/app/api/cases/start/route.js @@ -0,0 +1,38 @@ +import { startCase } from "@/lib/graph/orchestrator.js"; + +export async function POST(request) { + try { + const body = await request.json(); + const result = await startCase(body); + + if (result.success) { + return Response.json(result, { status: 200 }); + } + + const status = + result.statusCode === 400 + ? 400 + : result.statusCode >= 500 + ? result.statusCode + : 500; + + return Response.json( + { + success: false, + error: result.error ?? "Start case failed", + validationErrors: result.validationErrors, + diagnostics: result.diagnostics, + analysisErrors: result.analysisErrors, + }, + { status }, + ); + } catch { + return Response.json( + { + success: false, + error: "Internal server error", + }, + { status: 500 }, + ); + } +} diff --git a/lib/graph/orchestrator.js b/lib/graph/orchestrator.js new file mode 100644 index 0000000..abdb0aa --- /dev/null +++ b/lib/graph/orchestrator.js @@ -0,0 +1,125 @@ +/** + * Situation Graph Case Orchestrator — manages the lifecycle of a case. + * startCase builds initial graph from analysis; updateCase applies answers. + */ + +import { analyseScenario } from "../analysis.js"; +import { + makeGraph, + startCaseRequestSchema, + situationGraphSchema, +} from "./schema.js"; +import { buildInitialGraph, describeGraph } from "./builder.js"; +import { + selectActiveUnknownCandidate, + validateGraphReferences, +} from "./utils.js"; + +function toValidationErrors(error) { + return ( + error?.errors?.map((issue) => ({ + path: issue.path, + message: issue.message, + code: issue.code, + })) ?? [{ message: "Validation failed" }] + ); +} + +function buildDiagnostics({ analysis, graph, graphReferenceValidation }) { + return { + promptVersion: analysis?.promptVersion ?? null, + modelName: analysis?.modelName ?? null, + responseDurationMs: analysis?.responseDurationMs ?? null, + validationStatus: analysis?.validationStatus ?? "invalid", + nodeCount: graph?.nodes?.length ?? 0, + edgeCount: graph?.edges?.length ?? 0, + graphReferenceValidation, + }; +} + +export async function startCase(body) { + const parsedRequest = startCaseRequestSchema.safeParse(body); + + if (!parsedRequest.success) { + return { + success: false, + error: "Invalid start-case request", + validationErrors: toValidationErrors(parsedRequest.error), + statusCode: 400, + }; + } + + const { scenario, promptVersion } = parsedRequest.data; + const analysis = await analyseScenario(scenario, { promptVersion }); + + if (!analysis.success) { + return { + success: false, + error: analysis.error ?? "Scenario analysis failed", + diagnostics: buildDiagnostics({ + analysis, + graph: null, + graphReferenceValidation: null, + }), + analysisErrors: analysis.errors ?? undefined, + rawResponse: analysis.rawResponse ?? undefined, + statusCode: Number(analysis.statusCode) || 502, + }; + } + + const initialGraph = buildInitialGraph({ + reconstruction: analysis.reconstruction, + evidence: analysis.evidence, + }); + + const currentSummary = describeGraph(initialGraph); + const activeUnknownNodeId = + selectActiveUnknownCandidate( + { + ...initialGraph, + resolvedNodeIds: [], + }, + [], + )?.nodeId ?? null; + + const situationGraph = makeGraph({ + centralStatement: scenario, + nodes: initialGraph.nodes, + edges: initialGraph.edges, + activeUnknownNodeId, + resolvedNodeIds: [], + currentSummary, + }); + + situationGraphSchema.parse(situationGraph); + + const graphReferenceValidation = validateGraphReferences(situationGraph); + if (!graphReferenceValidation.valid) { + return { + success: false, + error: "Situation graph reference validation failed", + diagnostics: buildDiagnostics({ + analysis, + graph: situationGraph, + graphReferenceValidation, + }), + validationErrors: graphReferenceValidation.errors, + statusCode: 500, + }; + } + + return { + success: true, + situationGraph, + selectedQuestion: analysis.nextQuestion ?? null, + diagnostics: buildDiagnostics({ + analysis, + graph: situationGraph, + graphReferenceValidation, + }), + }; +} + +export async function updateCase() { + throw new Error("updateCase is not implemented yet"); +} diff --git a/tests/app/api/cases-start-route.test.js b/tests/app/api/cases-start-route.test.js new file mode 100644 index 0000000..76937d7 --- /dev/null +++ b/tests/app/api/cases-start-route.test.js @@ -0,0 +1,114 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const mockStartCase = vi.fn(); + +vi.mock("@/lib/graph/orchestrator.js", () => ({ + startCase: (...args) => mockStartCase(...args), +})); + +describe("app/api/cases/start route", () => { + beforeEach(() => { + vi.resetModules(); + vi.clearAllMocks(); + }); + + it("delegates request body to the orchestrator", async () => { + mockStartCase.mockResolvedValue({ + success: true, + situationGraph: { nodes: [{ id: "n1" }], edges: [] }, + selectedQuestion: null, + diagnostics: {}, + }); + + const { POST } = await import("@/app/api/cases/start/route.js"); + const request = new Request("http://localhost/api/cases/start", { + method: "POST", + body: JSON.stringify({ scenario: "Scenario text" }), + headers: { "content-type": "application/json" }, + }); + + await POST(request); + + expect(mockStartCase).toHaveBeenCalledWith({ scenario: "Scenario text" }); + }); + + it("returns 200 on success", async () => { + mockStartCase.mockResolvedValue({ + success: true, + situationGraph: { nodes: [{ id: "n1" }], edges: [] }, + selectedQuestion: null, + diagnostics: {}, + }); + + const { POST } = await import("@/app/api/cases/start/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/start", { + method: "POST", + body: JSON.stringify({ scenario: "Scenario text" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(200); + }); + + it("returns 400 for invalid request input", async () => { + mockStartCase.mockResolvedValue({ + success: false, + error: "Invalid start-case request", + validationErrors: [{ message: "Required" }], + statusCode: 400, + }); + + const { POST } = await import("@/app/api/cases/start/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/start", { + method: "POST", + body: JSON.stringify({}), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(400); + await expect(response.json()).resolves.toMatchObject({ + success: false, + error: "Invalid start-case request", + }); + }); + + it("returns provider/internal failures as 5xx without stack traces", async () => { + mockStartCase.mockResolvedValue({ + success: false, + error: "Provider unavailable", + diagnostics: { modelName: "llama3" }, + statusCode: 502, + }); + + const { POST } = await import("@/app/api/cases/start/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/start", { + method: "POST", + body: JSON.stringify({ scenario: "Scenario text" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(502); + await expect(response.json()).resolves.not.toHaveProperty("stack"); + }); + + it("returns structured 500 on malformed JSON", async () => { + const { POST } = await import("@/app/api/cases/start/route.js"); + const request = { + json: vi.fn().mockRejectedValue(new Error("Unexpected token")), + }; + + const response = await POST(request); + + expect(response.status).toBe(500); + await expect(response.json()).resolves.toMatchObject({ + success: false, + error: "Internal server error", + }); + }); +}); diff --git a/tests/graph/orchestrator.test.js b/tests/graph/orchestrator.test.js new file mode 100644 index 0000000..a574821 --- /dev/null +++ b/tests/graph/orchestrator.test.js @@ -0,0 +1,181 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const mockAnalyseScenario = vi.fn(); + +vi.mock("@/lib/analysis.js", () => ({ + analyseScenario: (...args) => mockAnalyseScenario(...args), +})); + +function makeAnalysisResult(overrides = {}) { + return { + success: true, + validationStatus: "valid", + modelName: "llama3", + responseDurationMs: 321, + rawResponse: "{}", + promptVersion: "v0.3", + reconstruction: { + summary: "Revenue and complaints diverge", + actors: [], + systemsOrObjects: [], + expectedStates: [], + observedStates: [ + { + id: "obs-1", + label: "Revenue up", + description: "Revenue up 15%", + confidence: "high", + }, + ], + differences: [], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [], + importantUnknowns: [ + { + id: "unk-1", + label: "Complaint rate denominator", + description: "Need the denominator for complaint rate", + confidence: "high", + }, + ], + plausibleInterpretations: [], + }, + evidence: [], + nextQuestion: { + id: "q-1", + question: "What denominator is being used for the complaint rate?", + }, + ...overrides, + }; +} + +describe("lib/graph/orchestrator startCase", () => { + beforeEach(() => { + vi.resetModules(); + vi.clearAllMocks(); + }); + + it("passes a valid request through to analyseScenario", async () => { + mockAnalyseScenario.mockResolvedValue(makeAnalysisResult()); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ + scenario: "Revenue increased while complaint counts rose faster.", + promptVersion: "v0.3", + }); + + expect(result.success).toBe(true); + expect(mockAnalyseScenario).toHaveBeenCalledWith( + "Revenue increased while complaint counts rose faster.", + { promptVersion: "v0.3" }, + ); + }); + + it("rejects invalid request input without throwing", async () => { + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "" }); + + expect(result).toMatchObject({ + success: false, + error: "Invalid start-case request", + statusCode: 400, + }); + expect(result.validationErrors).toBeInstanceOf(Array); + expect(mockAnalyseScenario).not.toHaveBeenCalled(); + }); + + it("builds a valid graph on successful analysis", async () => { + mockAnalyseScenario.mockResolvedValue(makeAnalysisResult()); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result.success).toBe(true); + expect(result.situationGraph.centralStatement).toBe("Scenario text"); + expect(result.situationGraph.currentSummary).toContain("Nodes:"); + expect(result.diagnostics).toMatchObject({ + validationStatus: "valid", + modelName: "llama3", + graphReferenceValidation: { valid: true, errors: [] }, + }); + }); + + it("applies active unknown selection to the graph", async () => { + mockAnalyseScenario.mockResolvedValue(makeAnalysisResult()); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result.success).toBe(true); + expect(result.situationGraph.activeUnknownNodeId).toBeTruthy(); + }); + + it("returns structured failure when graph reference validation fails", async () => { + mockAnalyseScenario.mockResolvedValue(makeAnalysisResult()); + const utils = await import("@/lib/graph/utils.js"); + const validateSpy = vi + .spyOn(utils, "validateGraphReferences") + .mockReturnValue({ + valid: false, + errors: ['Edge references non-existent toNodeId "missing"'], + }); + + const { startCase } = await import("@/lib/graph/orchestrator.js"); + const result = await startCase({ scenario: "Scenario text" }); + + expect(result).toMatchObject({ + success: false, + error: "Situation graph reference validation failed", + validationErrors: ['Edge references non-existent toNodeId "missing"'], + statusCode: 500, + }); + expect(result.diagnostics.graphReferenceValidation.valid).toBe(false); + validateSpy.mockRestore(); + }); + + it("preserves analysis/provider failure details", async () => { + mockAnalyseScenario.mockResolvedValue({ + success: false, + error: "Provider unavailable", + errors: ["socket hang up"], + rawResponse: null, + modelName: "llama3", + responseDurationMs: 99, + promptVersion: "v0.3", + validationStatus: "invalid", + statusCode: 502, + }); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result).toMatchObject({ + success: false, + error: "Provider unavailable", + analysisErrors: ["socket hang up"], + statusCode: 502, + }); + }); + + it("returns null selectedQuestion when analysis has no nextQuestion", async () => { + mockAnalyseScenario.mockResolvedValue( + makeAnalysisResult({ nextQuestion: undefined }), + ); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result.success).toBe(true); + expect(result.selectedQuestion).toBeNull(); + }); + + it("exports placeholder updateCase", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + + await expect(updateCase()).rejects.toThrow( + "updateCase is not implemented yet", + ); + }); +}); From 0ccc03c111f642af9582e1f3a08e9f42c10e0311 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 06:59:23 +0100 Subject: [PATCH 03/12] feat: add situation graph foundation --- lib/graph/builder.js | 305 +++++++++++++ lib/graph/schema.js | 191 +++++++++ lib/graph/utils.js | 348 +++++++++++++++ tests/graph/builder.test.js | 489 +++++++++++++++++++++ tests/graph/schema.test.js | 372 ++++++++++++++++ tests/graph/utils.test.js | 825 ++++++++++++++++++++++++++++++++++++ 6 files changed, 2530 insertions(+) create mode 100644 lib/graph/builder.js create mode 100644 lib/graph/schema.js create mode 100644 lib/graph/utils.js create mode 100644 tests/graph/builder.test.js create mode 100644 tests/graph/schema.test.js create mode 100644 tests/graph/utils.test.js diff --git a/lib/graph/builder.js b/lib/graph/builder.js new file mode 100644 index 0000000..314d1c6 --- /dev/null +++ b/lib/graph/builder.js @@ -0,0 +1,305 @@ +/** + * Deterministic situation graph builder — builds initial graph from scenario text. + * Takes v0.2/v0.3 analysis output (from analyseScenario) and constructs a SituationGraph. + */ + +import { + situationNodeSchema, + situationEdgeSchema, + makeNodeId, +} from "./schema.js"; + +/** + * Build an initial situation graph from a v0.3 reconstruction result. + * @param {{ reconstruction: object, evidence: object[] | undefined }} analysisData + * @returns {{ nodes: import("./schema.js").SituationNode[], edges: import("./schema.js").SituationEdge[] }} + */ +export function buildInitialGraph(analysisData) { + const { reconstruction, evidence = [] } = analysisData; + + if (!reconstruction || !reconstruction.summary) { + return { nodes: [], edges: [] }; + } + + const nodeMap = new Map(); // label -> node + + // ── Helper: register or get a node by label ──────────── + + function ensureNode( + label, + kind, + status, + description, + value, + unit, + confidence, + ) { + if (nodeMap.has(label)) return nodeMap.get(label); + + const id = makeNodeId(label); + const node = situationNodeSchema.parse({ + id, + label, + description: description ?? label, + kind, + status, + confidence, + value: value ?? null, + unit: unit ?? null, + evidenceIds: [], + dependsOn: [], + affects: [], + parentId: null, + childIds: [], + }); + nodeMap.set(label, node); + return node; + } + + // ── Evidence lookup ──────────────────────────────────── + + const evidenceMap = new Map(); + for (const ev of evidence) { + if (ev.id) evidenceMap.set(ev.id, ev); + } + + function addEvidenceToNode(nodeId, evidenceId) { + const node = Object.values(nodeMap).find((n) => n.id === nodeId); + if (node && !node.evidenceIds.includes(evidenceId)) { + node.evidenceIds.push(evidenceId); + } + } + + // ── Extract observed states as nodes ──────────────────── + + const summaryNode = ensureNode( + reconstruction.summary || "Situation Summary", + "state", + "provisional", + "Summary of the situation from the scenario text", + null, + null, + "medium", + ); + + // Collect all observable quantities as metric nodes + const metrics = new Map(); + + if (reconstruction.observedStates) { + for (const obs of reconstruction.observedStates) { + const node = ensureNode( + obs.description || obs.label, + "observation", + "supported", + obs.description || obs.label, + null, + null, + obs.confidence || "medium", + ); + + if (obs.id) node.evidenceIds.push(obs.id); + } + } + + // Actors as states/nodes + if (reconstruction.actors) { + for (const actor of reconstruction.actors) { + ensureNode( + actor.description || actor.label, + "observation", + "supported", + actor.description || actor.label, + null, + null, + actor.confidence || "medium", + ); + } + } + + if (reconstruction.systemsOrObjects) { + for (const sys of reconstruction.systemsOrObjects) { + ensureNode( + sys.description || sys.label, + "metric", + "known", + sys.description || sys.label, + null, + null, + sys.confidence || "medium", + ); + } + } + + // Differences as relationship nodes + if (reconstruction.differences) { + for (const diff of reconstruction.differences) { + const node = ensureNode( + diff.description || "Difference", + "relationship", + "supported", + diff.description || "Difference", + null, + null, + diff.confidence || "medium", + ); + } + } + + // Contradictions as nodes + if (reconstruction.contradictions) { + for (const c of reconstruction.contradictions) { + const node = ensureNode( + c.description || c.label, + "relationship", + "supported", + c.description || c.label, + null, + null, + c.confidence || "medium", + ); + } + } + + // Important unknowns as unknown nodes + const unknownNodes = []; + if (reconstruction.importantUnknowns) { + for (const unk of reconstruction.importantUnknowns) { + const node = ensureNode( + unk.description || unk.label, + "unknown", + "unknown", + unk.description || "Unknown factor in the situation", + null, + null, + unk.confidence || "low", + ); + unknownNodes.push(node); + } + } + + // Plausible interpretations + if (reconstruction.plausibleInterpretations) { + for (const interp of reconstruction.plausibleInterpretations) { + ensureNode( + interp.description || interp.label, + "assumption", + "provisional", + interp.description || "Plausible interpretation", + null, + null, + interp.confidence || "low", + ); + } + } + + // Known transitions + if (reconstruction.knownTransitions) { + for (const trans of reconstruction.knownTransitions) { + ensureNode( + `${trans.entity}: ${trans.previousState} → ${trans.currentState}`, + "transition", + trans.explanationStatus === "confirmed" ? "known" : "provisional", + trans.description || + `Transition: ${trans.entity} from ${trans.previousState} to ${trans.currentState}`, + null, + null, + trans.confidence || "medium", + ); + } + } + + // ── Build edges between nodes ──────────────────────── + + const nodeArr = Array.from(nodeMap.values()); + const edges = []; + + // Link actors → observed states as measures relationships + let actorNodes = []; + let metricNodes = []; + let unknownNodeIds = []; + + for (const n of nodeArr) { + if (n.kind === "observation" && n.status === "supported") { + // These are observations — link to summary + edges.push( + situationEdgeSchema.parse({ + id: `e-sum-${n.id}`, + fromNodeId: n.id, + toNodeId: summaryNode.id, + relationship: "supports", + confidence: n.confidence || "medium", + description: `${n.label} supports the summary`, + }), + ); + } + if (n.kind === "unknown") { + unknownNodeIds.push(n.id); + edges.push( + situationEdgeSchema.parse({ + id: `e-unk-${n.id}`, + fromNodeId: n.id, + toNodeId: summaryNode.id, + relationship: "depends_on", + confidence: n.confidence || "low", + description: `${n.label} is an unresolved factor for this situation`, + }), + ); + } + } + + return { nodes: nodeArr, edges }; +} + +/** + * Build a minimal starting graph for any scenario. + * Used when analysis has no reconstruction data (e.g., error state). + */ +export function buildMinimalGraph(scenario) { + const shortLabel = scenario.slice(0, 80); + + return { + nodes: [ + situationNodeSchema.parse({ + id: "n0", + label: shortLabel, + description: `Initial situation from: "${scenario.slice(0, 200)}"`, + kind: "state", + status: "provisional", + confidence: "low", + value: null, + unit: null, + evidenceIds: [], + dependsOn: [], + affects: [], + parentId: null, + childIds: [], + }), + ], + edges: [], + }; +} + +/** + * Convert graph nodes/edges to a human-readable summary for display. + */ +export function describeGraph(graph) { + const parts = []; + + // Count by kind + const byKind = {}; + for (const n of graph.nodes) { + byKind[n.kind] = (byKind[n.kind] || 0) + 1; + } + + parts.push( + `Nodes: ${Object.entries(byKind) + .map(([k, v]) => `${v} ${k}`) + .join(", ")}`, + ); + parts.push(`Edges: ${graph.edges.length} total`); + parts.push( + `Unknowns: ${graph.nodes.filter((n) => n.status === "unknown").length} unresolved`, + ); + + return parts.join(" | "); +} diff --git a/lib/graph/schema.js b/lib/graph/schema.js new file mode 100644 index 0000000..c125c98 --- /dev/null +++ b/lib/graph/schema.js @@ -0,0 +1,191 @@ +/** + * Situation Graph schema — v0.4 experiment. + * Defines types for an evolving multi-turn situation reconstruction graph. + * Plain TypeScript interfaces implemented as Zod schemas for runtime validation. + */ + +import { z } from "zod"; + +// ── Enums ──────────────────────────────────────────── + +export const SituationKind = /** @type {const} */ ({ + observation: "observation", + reported_claim: "reported_claim", + metric: "metric", + state: "state", + transition: "transition", + relationship: "relationship", + assumption: "assumption", + unknown: "unknown", + conclusion: "conclusion", +}); + +export const SituationStatus = /** @type {const} */ ({ + known: "known", + unknown: "unknown", + provisional: "provisional", + supported: "supported", + weakened: "weakened", + contradicted: "contradicted", + resolved: "resolved", +}); + +export const ConfidenceLevel = /** @type {const} */ ({ + low: "low", + medium: "medium", + high: "high", +}); + +// ── SituationNode ──────────────────────────────────── + +export const situationNodeSchema = z.object({ + id: z.string().min(1), + label: z.string().min(1), + description: z.string().min(1), + kind: z.enum(Object.values(SituationKind)), + status: z.enum(Object.values(SituationStatus)), + confidence: z.enum(Object.values(ConfidenceLevel)), + value: z.union([z.string(), z.number(), z.null()]).nullable().optional(), + unit: z.string().nullable().optional(), + evidenceIds: z.array(z.string()).default([]), + dependsOn: z.array(z.string()).default([]), + affects: z.array(z.string()).default([]), + parentId: z.string().nullable().optional(), + childIds: z.array(z.string()).default([]), +}); + +/** @typedef {z.infer} SituationNode */ + +// ── SituationEdge ──────────────────────────────────── + +export const SituationRelationship = /** @type {const} */ ({ + supports: "supports", + weakens: "weakens", + contradicts: "contradicts", + depends_on: "depends_on", + causes: "causes", + may_cause: "may_cause", + measures: "measures", + compares_with: "compares_with", + updates: "updates", + other: "other", +}); + +export const situationEdgeSchema = z.object({ + id: z.string().min(1), + fromNodeId: z.string().min(1), + toNodeId: z.string().min(1), + relationship: z.enum(Object.values(SituationRelationship)), + confidence: z.enum(Object.values(ConfidenceLevel)), + description: z.string().min(1), +}); + +/** @typedef {z.infer} SituationEdge */ + +// ── SituationGraph ─────────────────────────────────── + +export const situationGraphSchema = z.object({ + centralStatement: z.string().min(1), + nodes: z.array(situationNodeSchema).min(1), + edges: z.array(situationEdgeSchema).default([]), + activeUnknownNodeId: z.string().nullable(), + resolvedNodeIds: z.array(z.string()).default([]), + currentSummary: z.string().min(1), +}); + +/** @typedef {z.infer} SituationGraph */ + +// ── GraphUpdate (change set) ──────────────────────── + +const graphUpdateNodeChangeSchema = z.object({ + nodeId: z.string().min(1), + previousStatus: z.enum(Object.values(SituationStatus)).nullable().optional(), + newStatus: z.enum(Object.values(SituationStatus)).nullable().optional(), + previousValue: z.union([z.string(), z.number(), z.null()]).nullable().optional(), + newValue: z.union([z.string(), z.number(), z.null()]).nullable().optional(), + reason: z.string().min(1), +}); + +export const graphUpdateSchema = z.object({ + addedNodes: z.array(situationNodeSchema).default([]), + updatedNodes: z.array(graphUpdateNodeChangeSchema).default([]), + addedEdges: z.array(situationEdgeSchema).default([]), + removedEdgeIds: z.array(z.string()).default([]), + resolvedUnknownNodeIds: z.array(z.string()).default([]), + affectedNodeIds: z.array(z.string()).default([]), +}); + +/** @typedef {z.infer} GraphUpdate */ + +// ── API request / response schemas ─────────────────── + +export const startCaseRequestSchema = z.object({ + scenario: z.string().min(1).max(10000), + promptVersion: z.string().optional(), +}); + +export const updateCaseRequestSchema = z.object({ + situationGraph: situationGraphSchema, + previousQuestion: z.string().min(1), + answer: z.string().min(1).max(5000), + promptVersion: z.string().optional(), +}); + +// ── Helpers ────────────────────────────────────────── + +/** Generate a short deterministic ID from a label */ +export function makeNodeId(label) { + return "n" + Math.abs(hashString(label)).toString(36).slice(0, 7); +} + +function hashString(str) { + let h = 0; + for (let i = 0; i < str.length; i++) { + h = (Math.imul(31, h) + str.charCodeAt(i)) | 0; + } + return h; +} + +/** Create a minimal valid node — used in tests and fixtures */ +export function makeNode(opts) { + const id = opts.id || makeNodeId(opts.label); + return situationNodeSchema.parse({ + id, + label: opts.label, + description: opts.description ?? opts.label, + kind: opts.kind ?? "observation", + status: opts.status ?? "unknown", + confidence: opts.confidence ?? "medium", + value: opts.value ?? null, + unit: opts.unit ?? null, + evidenceIds: opts.evidenceIds ?? [], + dependsOn: opts.dependsOn ?? [], + affects: opts.affects ?? [], + parentId: opts.parentId ?? null, + childIds: opts.childIds ?? [], + }); +} + +/** Create a minimal valid edge — used in tests and fixtures */ +export function makeEdge(opts) { + return situationEdgeSchema.parse({ + id: opts.id || "e" + opts.fromNodeId.slice(0,3) + "-" + opts.toNodeId.slice(0,3), + fromNodeId: opts.fromNodeId, + toNodeId: opts.toNodeId, + relationship: opts.relationship ?? "supports", + confidence: opts.confidence ?? "medium", + description: opts.description ?? opts.fromNodeId + " -> " + opts.toNodeId, + }); +} + +/** Build a minimal valid graph structure */ +export function makeGraph(opts) { + return situationGraphSchema.parse({ + centralStatement: opts.centralStatement || "", + nodes: opts.nodes ?? [], + edges: opts.edges ?? [], + activeUnknownNodeId: opts.activeUnknownNodeId ?? null, + resolvedNodeIds: opts.resolvedNodeIds ?? [], + currentSummary: opts.currentSummary || "", + }); +} diff --git a/lib/graph/utils.js b/lib/graph/utils.js new file mode 100644 index 0000000..2ee736d --- /dev/null +++ b/lib/graph/utils.js @@ -0,0 +1,348 @@ +/** + * Deterministic graph utilities for situation graph operations. + * These functions perform safe, validated operations on the graph. + * The LLM should never directly modify the graph — it proposes changes, + * and these utilities apply them safely. + */ + +import { situationNodeSchema, situationEdgeSchema, situationGraphSchema } from "./schema.js"; + +// ── Validate that all edge references point to existing nodes ── + +export function validateGraphReferences(graph) { + const errors = []; + const nodeIds = new Set(graph.nodes.map((n) => n.id)); + + for (const node of graph.nodes) { + if (node.parentId !== null && !nodeIds.has(node.parentId)) { + errors.push(`Node "${node.id}" references parentId "${node.parentId}" which does not exist`); + } + for (const cid of node.childIds) { + if (!nodeIds.has(cid)) { + errors.push(`Node "${node.id}" references childIds "${cid}" which does not exist`); + } + } + for (const dep of node.dependsOn) { + if (!nodeIds.has(dep)) { + errors.push(`Node "${node.id}" depends on "${dep}" which does not exist`); + } + } + for (const aff of node.affects) { + if (!nodeIds.has(aff)) { + errors.push(`Node "${node.id}" affects "${aff}" which does not exist`); + } + } + } + + for (const edge of graph.edges) { + if (!nodeIds.has(edge.fromNodeId)) { + errors.push(`Edge "${edge.id}" references non-existent fromNodeId "${edge.fromNodeId}"`); + } + if (!nodeIds.has(edge.toNodeId)) { + errors.push(`Edge "${edge.id}" references non-existent toNodeId "${edge.toNodeId}"`); + } + } + + return { valid: errors.length === 0, errors }; +} + +// ── Detect duplicate node IDs ── + +export function detectDuplicateNodeIds(nodes) { + const countMap = new Map(); + const seen = new Set(); + + for (const node of nodes) { + if (countMap.has(node.id)) { + countMap.set(node.id, countMap.get(node.id) + 1); + } else { + countMap.set(node.id, 1); + } + } + + const duplicates = []; + for (const [id, count] of countMap.entries()) { + if (count > 1 && !seen.has(id)) { + duplicates.push({ nodeId: id, count }); + seen.add(id); + } + } + + return duplicates; +} + +// ── Detect duplicate edges ── + +export function detectDuplicateEdges(edges) { + const seen = new Set(); + const duplicates = []; + + for (const edge of edges) { + const key = `${edge.fromNodeId}->${edge.toNodeId}:${edge.relationship}`; + if (seen.has(key)) { + duplicates.push({ edgeId: edge.id, fromNodeId: edge.fromNodeId, toNodeId: edge.toNodeId, relationship: edge.relationship }); + } + seen.add(key); + } + + return duplicates; +} + +// ── Find all nodes that depend on a given node (transitive) ── + +export function findDependentNodes(graph, nodeId) { + const direct = graph.nodes.filter((n) => n.dependsOn.includes(nodeId)).map((n) => n.id); + const affected = new Set(direct); + + // Also propagate through edges where the relationship is depends_on + for (const edge of graph.edges) { + if (edge.toNodeId === nodeId && !affected.has(edge.fromNodeId)) { + direct.push(edge.fromNodeId); + affected.add(edge.fromNodeId); + } + } + + // Transitive propagation — BFS + const queue = [...direct]; + while (queue.length > 0) { + const current = queue.shift(); + if (!current || !affected.has(current)) continue; + + for (const node of graph.nodes) { + if (node.dependsOn.includes(current) && !affected.has(node.id)) { + affected.add(node.id); + queue.push(node.id); + } + } + } + + return [...affected]; +} + +// ── Find all nodes that are directly or indirectly affected by a change in nodeId ── + +export function findAffectedNodes(graph, nodeId) { + // Direct effects: two sources + // 1. Nodes that depend on this node (they list it in their dependsOn) + const directFromDepends = graph.nodes.filter((n) => n.id !== nodeId && n.dependsOn.includes(nodeId)).map((n) => n.id); + + // 2. Targets of the node's affects relationships (this node directly affects them) + const myAffectedTargets = new Set(graph.nodes.find((n) => n.id === nodeId)?.affects || []); + + // Merge: also add edge targets where this node is the source + for (const edge of graph.edges) { + if (edge.fromNodeId === nodeId && !myAffectedTargets.has(edge.toNodeId)) { + myAffectedTargets.add(edge.toNodeId); + } + } + + // Combine both sources + const direct = [...new Set([...directFromDepends, ...myAffectedTargets])]; + + // Transitive propagation — BFS through dependsOn and affects of affected nodes + const affected = new Set(direct); + const queue = [...direct]; + while (queue.length > 0) { + const current = queue.shift(); + if (!current || !affected.has(current)) continue; + + for (const node of graph.nodes) { + if (node.id !== nodeId && !affected.has(node.id) && (node.dependsOn.includes(current) || node.affects.includes(current))) { + affected.add(node.id); + queue.push(node.id); + } + } + } + + return [...affected]; +} + +// ── Resolve an unknown node ── + +export function resolveUnknownNode(graph, nodeId, newStatus, newValue, reason) { + const nodeIdx = graph.nodes.findIndex((n) => n.id === nodeId); + if (nodeIdx === -1) { + return { success: false, error: `Node "${nodeId}" not found in graph` }; + } + + const previousStatus = graph.nodes[nodeIdx].status; + const previousValue = graph.nodes[nodeIdx].value; + + return { + success: true, + previousStatus, + newStatus, + previousValue, + newValue, + reason, + affectedNodes: findAffectedNodes(graph, nodeId), + }; +} + +// ── Select the next highest-value active unknown candidate ── + +export function selectActiveUnknownCandidate(graph, resolvedNodeIds) { + // Skip already resolved nodes + const unresolved = graph.nodes.filter( + (n) => n.kind === "unknown" && !resolvedNodeIds.includes(n.id) + ); + + if (unresolved.length === 0) return null; + + // Prioritise: critical unknowns first, then those that are depended upon most + const dependencyCount = unresolved.map((n) => { + const deps = findDependentNodes(graph, n.id).length; + const importanceOrder = { critical: 3, important: 2, supporting: 1, incidental: 0 }; + const impScore = importanceOrder[n.confidence] || 0; + return { node: n, score: deps * 2 + impScore }; + }); + + dependencyCount.sort((a, b) => b.score - a.score); + + // Return the highest-scoring unresolved unknown + const best = dependencyCount[0]; + if (!best) return null; + + return { nodeId: best.node.id, label: best.node.label, score: best.score }; +} + +// ── Apply a graph update deterministically ── + +export function applyGraphUpdate(graph, update) { + const errors = []; + const updatedNodesMap = new Map(); + + // Validate that update references existing nodes or newly added ones + const allNodeIds = new Set(graph.nodes.map((n) => n.id)); + for (const added of update.addedNodes) { + if (allNodeIds.has(added.id)) { + errors.push(`Cannot add node with duplicate ID: "${added.id}"`); + continue; + } + allNodeIds.add(added.id); + } + + // Validate updated nodes exist + for (const upd of update.updatedNodes) { + if (!allNodeIds.has(upd.nodeId)) { + errors.push(`Cannot update non-existent node: "${upd.nodeId}"`); + } + } + + // Validate added edges reference existing or new nodes + for (const edge of update.addedEdges) { + if (!allNodeIds.has(edge.fromNodeId)) { + errors.push(`Added edge references non-existent fromNodeId: "${edge.fromNodeId}"`); + } + if (!allNodeIds.has(edge.toNodeId)) { + errors.push(`Added edge references non-existent toNodeId: "${edge.toNodeId}"`); + } + } + + if (errors.length > 0) return { success: false, errors }; + + // Build the new nodes list — start with a deep copy of existing + const newNodes = graph.nodes.map((n) => ({ ...n })); + + // Apply updated nodes + for (const upd of update.updatedNodes) { + const idx = newNodes.findIndex((n) => n.id === upd.nodeId); + if (idx === -1) continue; // already validated above + + if (upd.newStatus !== undefined && upd.newStatus !== null) { + newNodes[idx].status = upd.newStatus; + } + if (upd.newValue !== undefined) { + newNodes[idx].value = upd.newValue; + } + updatedNodesMap.set(upd.nodeId, newNodes[idx]); + } + + // Add new nodes + for (const newNode of update.addedNodes) { + if (!allNodeIds.has(newNode.id)) continue; + allNodeIds.add(newNode.id); + newNodes.push({ ...newNode }); + } + + // Remove edges if requested + const removedEdgeSet = new Set(update.removedEdgeIds); + const newEdges = graph.edges.filter((e) => !removedEdgeSet.has(e.id)); + + // Add new edges + for (const newEdge of update.addedEdges) { + newEdges.push({ ...newEdge }); + + // Update dependsOn / affects on the nodes + const fromNode = newNodes.find((n) => n.id === newEdge.fromNodeId); + const toNode = newNodes.find((n) => n.id === newEdge.toNodeId); + if (fromNode && !fromNode.childIds.includes(newEdge.toNodeId)) { + fromNode.childIds.push(newEdge.toNodeId); + } + if (toNode && !toNode.dependsOn.includes(newEdge.fromNodeId)) { + toNode.dependsOn.push(newEdge.fromNodeId); + } + } + + // Add resolved node IDs + const newResolved = [...new Set([...graph.resolvedNodeIds, ...update.resolvedUnknownNodeIds])]; + + return { + success: true, + nodes: newNodes, + edges: newEdges, + resolvedNodeIds: newResolved, + }; +} + +// ── Validate a proposed graph update before application ── + +export function validateGraphUpdate(graph, update) { + const errors = []; + + // Check for duplicate node IDs against existing and newly added nodes + const extendedIds = new Set(graph.nodes.map((n) => n.id)); + for (const newNode of update.addedNodes) { + if (extendedIds.has(newNode.id)) { + errors.push(`Cannot add node with duplicate ID: "${newNode.id}"`); + } else { + extendedIds.add(newNode.id); + } + } + + // Check updated nodes exist (in original graph, not newly added ones) + const existingIds = new Set(graph.nodes.map((n) => n.id)); + for (const upd of update.updatedNodes) { + if (!existingIds.has(upd.nodeId)) { + errors.push(`Cannot update non-existent node: "${upd.nodeId}"`); + } + } + + // Reject updates with no meaningful change + const statusChanged = update.updatedNodes.some( + (u) => u.previousStatus !== null && u.newStatus !== u.previousStatus + ); + const valueChanged = update.updatedNodes.some( + (u) => u.previousValue !== null && u.newValue !== u.previousValue + ); + + const hasMeaningfulChange = + update.addedNodes.length > 0 || + statusChanged || + valueChanged || + update.addedEdges.length > 0 || + update.removedEdgeIds.length > 0; + + if (!hasMeaningfulChange) { + errors.push("Update contains no meaningful change"); + } + + // Reject oversized input + const totalSize = JSON.stringify(update).length; + if (totalSize > 100000) { + errors.push(`Proposed graph update exceeds 100KB (${totalSize} bytes)`); + } + + return { valid: errors.length === 0, errors }; +} + diff --git a/tests/graph/builder.test.js b/tests/graph/builder.test.js new file mode 100644 index 0000000..a85564a --- /dev/null +++ b/tests/graph/builder.test.js @@ -0,0 +1,489 @@ +import { describe, it, expect } from "vitest"; +import { + buildInitialGraph, + buildMinimalGraph, + describeGraph, +} from "@/lib/graph/builder.js"; +import { + makeNode, + situationEdgeSchema, + situationGraphSchema, + situationNodeSchema, +} from "@/lib/graph/schema.js"; + +// ── Helper: create a v0.3-style reconstruction fixture ─────────── + +function makeReconstructionFixture() { + return { + summary: "Company X reports revenue growth but increasing complaints", + actors: [ + { id: "actor-1", description: "Customer Base", confidence: "high" }, + { + id: "actor-2", + description: "Product Engineering Team", + confidence: "high", + }, + ], + systemsOrObjects: [ + { id: "sys-1", description: "Production Line A", confidence: "high" }, + { + id: "sys-2", + description: "Quality Control System", + confidence: "medium", + }, + ], + expectedStates: [], + observedStates: [ + { + id: "obs-1", + description: "Revenue up 15% year-over-year", + confidence: "high", + }, + { + id: "obs-2", + description: "Customer complaints up 40% year-over-year", + confidence: "high", + }, + ], + differences: [ + { + id: "diff-1", + description: "Complaint count grew faster than revenue", + confidence: "medium", + }, + ], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [ + { + id: "con-1", + description: "Revenue growth vs complaint growth inconsistency", + confidence: "high", + }, + ], + importantUnknowns: [ + { + id: "unk-1", + description: "Denominator for complaint rate (customers served)", + confidence: "high", + }, + { + id: "unk-2", + description: "Root cause of complaint increase", + confidence: "medium", + }, + ], + plausibleInterpretations: [], + }; +} + +function makeEvidenceFixture() { + return [ + { + id: "ev-1", + description: "Annual report data", + evidenceType: "direct_observation", + confidence: "high", + importance: "critical", + }, + { + id: "ev-2", + description: "Customer survey results", + evidenceType: "reported_statement", + confidence: "medium", + importance: "important", + }, + ]; +} + +describe("buildInitialGraph", () => { + it("builds nodes from reconstruction data", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: makeEvidenceFixture(), + }); + + expect(result.nodes.length).toBeGreaterThan(0); + expect(result.edges.length).toBeGreaterThan(0); + }); + + it("creates a summary node", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const summaryNode = result.nodes.find((n) => n.kind === "state"); + expect(summaryNode).toBeDefined(); + expect(summaryNode.label).toBe( + "Company X reports revenue growth but increasing complaints", + ); + }); + + it("creates observation nodes from observedStates", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const observations = result.nodes.filter((n) => n.kind === "observation"); + expect(observations.length).toBeGreaterThan(0); + }); + + it("creates unknown nodes from importantUnknowns", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const unknowns = result.nodes.filter((n) => n.kind === "unknown"); + expect(unknowns.length).toBe(2); // unk-1 and unk-2 + }); + + it("creates metric nodes from systemsOrObjects", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const metrics = result.nodes.filter((n) => n.kind === "metric"); + expect(metrics.length).toBe(2); // sys-1 and sys-2 + }); + + it("creates actor nodes as observations", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const actors = result.nodes.filter( + (n) => + n.label.includes("Customer Base") || + n.label.includes("Product Engineering"), + ); + expect(actors.length).toBe(2); + }); + + it("creates difference nodes", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const differenceNode = result.nodes.find((n) => + n.label.includes("Complaint count grew faster than revenue"), + ); + + expect(differenceNode).toBeDefined(); + expect(differenceNode.kind).toBe("relationship"); + }); + + it("creates contradiction nodes", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const contradictionNode = result.nodes.find((n) => + n.label.includes("inconsistency"), + ); + + expect(contradictionNode).toBeDefined(); + expect(contradictionNode.kind).toBe("relationship"); + }); + + it("creates edges linking observations to summary", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const supportEdges = result.edges.filter( + (e) => e.relationship === "supports", + ); + expect(supportEdges.length).toBeGreaterThan(0); + }); + + it("creates edges linking unknowns to summary as depends_on", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const depEdges = result.edges.filter( + (e) => e.relationship === "depends_on", + ); + expect(depEdges.length).toBe(2); // Two unknown nodes + }); + + it("handles empty observedStates gracefully", () => { + const reconstruction = { + ...makeReconstructionFixture(), + observedStates: [], + }; + const result = buildInitialGraph({ reconstruction, evidence: [] }); + + expect(result.nodes.length).toBeGreaterThan(0); // Summary + actors + systems still created + }); + + it("handles missing reconstruction fields gracefully", () => { + const result = buildInitialGraph({ + reconstruction: { summary: "Minimal" }, + evidence: [], + }); + + expect(result.nodes.length).toBeGreaterThan(0); + }); + + it("handles null/undefined reconstruction", () => { + const result = buildInitialGraph({ reconstruction: null, evidence: [] }); + expect(result.nodes.length).toBe(0); + expect(result.edges.length).toBe(0); + }); + + it("handles missing evidence array", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + }); + + expect(result.nodes.length).toBeGreaterThan(0); + expect(result.edges.length).toBeGreaterThan(0); + }); + + it("generates deterministic node IDs for same labels", () => { + const r1 = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + const r2 = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: [], + }); + + const ids1 = r1.nodes.map((n) => n.id).sort(); + const ids2 = r2.nodes.map((n) => n.id).sort(); + expect(ids1).toEqual(ids2); + }); + + it("produces valid schema output (no parse errors)", () => { + const result = buildInitialGraph({ + reconstruction: makeReconstructionFixture(), + evidence: makeEvidenceFixture(), + }); + + for (const node of result.nodes) { + const parsed = situationNodeSchema.safeParse(node); + if (!parsed.success) { + console.error(`Invalid node: ${node.id}`, node, parsed.error.message); + } + expect(parsed.success).toBe(true); + } + + for (const edge of result.edges) { + const parsed = situationEdgeSchema.safeParse(edge); + if (!parsed.success) { + console.error(`Invalid edge: ${edge.id}`, edge, parsed.error.message); + } + expect(parsed.success).toBe(true); + } + }); + + it("creates edges for knownTransitions as transition nodes", () => { + const reconstruction = { + ...makeReconstructionFixture(), + knownTransitions: [ + { + id: "trans-1", + description: "Product shipped v2.0", + entity: "Product", + previousState: "v1.x", + currentState: "v2.0", + explanationStatus: "confirmed", + confidence: "high", + }, + ], + }; + + const result = buildInitialGraph({ reconstruction, evidence: [] }); + const transitions = result.nodes.filter((n) => n.kind === "transition"); + expect(transitions.length).toBe(1); + }); + + it("creates nodes for unexplainedTransitions", () => { + const reconstruction = { + ...makeReconstructionFixture(), + unexplainedTransitions: [ + { + id: "ut-1", + description: "Support wait time increased", + entity: "Support", + previousState: "2hr", + currentState: "8hr", + confidence: "medium", + }, + ], + }; + + const result = buildInitialGraph({ reconstruction, evidence: [] }); + expect(result.nodes.length).toBeGreaterThan(0); + }); + + it("creates nodes for plausibleInterpretations as assumptions", () => { + const reconstruction = { + ...makeReconstructionFixture(), + plausibleInterpretations: [ + { + id: "interp-1", + description: "Quality degradation hypothesis", + supportingEvidenceIds: ["ev-2"], + assumptionsRequired: [], + confidence: "medium", + }, + ], + }; + + const result = buildInitialGraph({ reconstruction, evidence: [] }); + const assumptions = result.nodes.filter((n) => n.kind === "assumption"); + expect(assumptions.length).toBe(1); + }); + + it("links evidence to observation nodes", () => { + const reconstruction = makeReconstructionFixture(); + const evidence = [{ id: "ev-1", description: "Test evidence" }]; + + // Add a mapping from observed states to evidence IDs would require modification + // For now, just verify the nodes have empty evidenceIds (as per current implementation) + const result = buildInitialGraph({ reconstruction, evidence }); + for (const node of result.nodes) { + expect(Array.isArray(node.evidenceIds)).toBe(true); + } + }); + + it("handles very large reconstruction without errors", () => { + const actors = Array.from({ length: 20 }, (_, i) => ({ + id: `actor-${i}`, + description: `Actor ${i}`, + confidence: "high", + })); + + const result = buildInitialGraph({ + reconstruction: { ...makeReconstructionFixture(), actors }, + evidence: [], + }); + + expect(result.nodes.length).toBeGreaterThan(10); + }); + + it("handles transition with confirmed explanation", () => { + const reconstruction = { + ...makeReconstructionFixture(), + knownTransitions: [ + { + id: "t-confirmed", + description: "Confirmed event", + entity: "E1", + previousState: "s1", + currentState: "s2", + explanationStatus: "confirmed", + confidence: "high", + }, + ], + }; + + const result = buildInitialGraph({ reconstruction, evidence: [] }); + const confirmedTransitions = result.nodes.filter( + (n) => n.kind === "transition" && n.status === "known", + ); + expect(confirmedTransitions.length).toBe(1); + }); +}); + +describe("buildMinimalGraph", () => { + it("creates a single node with scenario text as label", () => { + const graph = buildMinimalGraph( + "This is a test scenario for minimal graph creation", + ); + expect(graph.nodes.length).toBe(1); + expect(graph.edges.length).toBe(0); + }); + + it("truncates label to 80 chars", () => { + const longScenario = "a".repeat(200); + const graph = buildMinimalGraph(longScenario); + expect(graph.nodes[0].label.length).toBeLessThanOrEqual(80); + }); + + it("creates provisional state node", () => { + const graph = buildMinimalGraph("Test scenario"); + expect(graph.nodes[0].kind).toBe("state"); + expect(graph.nodes[0].status).toBe("provisional"); + expect(graph.nodes[0].confidence).toBe("low"); + }); + + it("uses first 200 chars of scenario for description", () => { + const graph = buildMinimalGraph( + "This is a test scenario for minimal graph creation", + ); + expect(graph.nodes[0].description).toContain("Initial situation from:"); + }); + + it("creates deterministic ID via situationNodeSchema.parse", () => { + const graph = buildMinimalGraph("Test scenario"); + // Node has explicit id "n0" from the builder, not makeNodeId + expect(graph.nodes[0].id).toBe("n0"); + }); + + it("creates minimal valid structure", () => { + const graph = buildMinimalGraph("Test"); + expect(graph.nodes).toHaveLength(1); + expect(graph.edges).toHaveLength(0); + expect(graph.nodes[0].evidenceIds).toEqual([]); + expect(graph.nodes[0].dependsOn).toEqual([]); + expect(graph.nodes[0].affects).toEqual([]); + }); +}); + +describe("describeGraph", () => { + it("returns summary string with node count by kind", () => { + const graph = buildMinimalGraph("Test"); + const description = describeGraph(graph); + + expect(description).toContain("Nodes:"); + expect(description).toContain("Edges:"); + expect(description).toContain("Unknowns:"); + }); + + it("shows correct edge count", () => { + const graph = buildMinimalGraph("Test"); + const description = describeGraph(graph); + + expect(description).toContain("Edges: 0 total"); + }); + + it("counts unresolved unknowns", () => { + const n1 = makeNode({ + id: "n-unk", + label: "Unknown", + kind: "unknown", + status: "unknown", + }); + const graph = situationGraphSchema.parse({ + centralStatement: "Test", + nodes: [n1], + edges: [], + activeUnknownNodeId: n1.id, + resolvedNodeIds: [], + currentSummary: "Test", + }); + + const description = describeGraph(graph); + expect(description).toContain("1"); // One unresolved unknown + }); + + it("groups nodes by kind in output", () => { + const graph = buildMinimalGraph("Test"); + const description = describeGraph(graph); + + expect(description).toContain("1 state"); + }); +}); diff --git a/tests/graph/schema.test.js b/tests/graph/schema.test.js new file mode 100644 index 0000000..e8539c6 --- /dev/null +++ b/tests/graph/schema.test.js @@ -0,0 +1,372 @@ +import { describe, it, expect } from "vitest"; +import { + SituationKind, + SituationStatus, + ConfidenceLevel, + SituationRelationship, + situationNodeSchema, + situationEdgeSchema, + situationGraphSchema, + graphUpdateSchema, + startCaseRequestSchema, + updateCaseRequestSchema, + makeNodeId, + makeNode, + makeEdge, + makeGraph, +} from "@/lib/graph/schema.js"; + +describe("situationNodeSchema", () => { + const validNode = { + id: "n1", + label: "Test Node", + description: "A test node", + kind: "observation", + status: "known", + confidence: "high", + value: null, + unit: null, + evidenceIds: [], + dependsOn: [], + affects: [], + parentId: null, + childIds: [], + }; + + it("validates a complete valid node", () => { + const result = situationNodeSchema.safeParse(validNode); + expect(result.success).toBe(true); + }); + + it("requires id", () => { + const invalid = { ...validNode, id: "" }; + const result = situationNodeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("requires label", () => { + const invalid = { ...validNode, label: "" }; + const result = situationNodeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("rejects invalid kind", () => { + const invalid = { ...validNode, kind: "nonexistent" }; + const result = situationNodeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("rejects invalid status", () => { + const invalid = { ...validNode, status: "unknown_status" }; + const result = situationNodeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("rejects invalid confidence", () => { + const invalid = { ...validNode, confidence: "extreme" }; + const result = situationNodeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("allows numeric value", () => { + const node = { ...validNode, value: 42 }; + const result = situationNodeSchema.safeParse(node); + expect(result.success).toBe(true); + }); + + it("allows string value", () => { + const node = { ...validNode, value: "active" }; + const result = situationNodeSchema.safeParse(node); + expect(result.success).toBe(true); + }); +}); + +describe("situationEdgeSchema", () => { + const validEdge = { + id: "e1", + fromNodeId: "n1", + toNodeId: "n2", + relationship: "supports", + confidence: "medium", + description: "Edge between nodes", + }; + + it("validates a complete valid edge", () => { + const result = situationEdgeSchema.safeParse(validEdge); + expect(result.success).toBe(true); + }); + + it("rejects invalid relationship type", () => { + const invalid = { ...validEdge, relationship: "invalid_rel" }; + const result = situationEdgeSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("validates all relationship types", () => { + for (const rel of Object.values(SituationRelationship)) { + const edge = { ...validEdge, relationship: rel }; + const result = situationEdgeSchema.safeParse(edge); + expect(result.success).toBe(true); + } + }); + + it("rejects self-referencing edges", () => { + // Self-refs are structurally valid but semantically questionable + const edge = { ...validEdge, fromNodeId: "n1", toNodeId: "n1" }; + const result = situationEdgeSchema.safeParse(edge); + expect(result.success).toBe(true); // Structure is valid; semantics checked elsewhere + }); +}); + +describe("situationGraphSchema", () => { + const validGraph = { + centralStatement: "Test graph summary", + nodes: [makeNode({ id: "n1", label: "Node 1" })], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Initial summary", + }; + + it("validates a complete valid graph", () => { + const result = situationGraphSchema.safeParse(validGraph); + expect(result.success).toBe(true); + }); + + it("requires at least one node", () => { + const invalid = { ...validGraph, nodes: [] }; + const result = situationGraphSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); + + it("allows empty edges array", () => { + const graph = { ...validGraph, edges: [] }; + const result = situationGraphSchema.safeParse(graph); + expect(result.success).toBe(true); + }); + + it("rejects missing centralStatement", () => { + const invalid = { ...validGraph, centralStatement: "" }; + const result = situationGraphSchema.safeParse(invalid); + expect(result.success).toBe(false); + }); +}); + +describe("graphUpdateSchema", () => { + it("validates empty update (no-op proposal)", () => { + const result = graphUpdateSchema.safeParse({}); + expect(result.success).toBe(true); + }); + + it("validates a complete update", () => { + const node = makeNode({ id: "n2", label: "New Node" }); + const edge = makeEdge({ fromNodeId: "n1", toNodeId: "n2" }); + + const result = graphUpdateSchema.safeParse({ + addedNodes: [node], + updatedNodes: [{ nodeId: "n1", newStatus: "resolved", previousStatus: "unknown", reason: "Question answered" }], + addedEdges: [edge], + removedEdgeIds: ["e-old"], + resolvedUnknownNodeIds: ["n2"], + affectedNodeIds: ["n3"], + }); + expect(result.success).toBe(true); + }); + + it("rejects update with invalid node kind in addedNodes", () => { + const invalid = graphUpdateSchema.safeParse({ + addedNodes: [{ id: "x", label: "Test", kind: "invalid_kind", description: "test", status: "unknown", confidence: "medium", value: null, unit: null, evidenceIds: [], dependsOn: [], affects: [], parentId: null, childIds: [] }], + }); + expect(invalid.success).toBe(false); + }); +}); + +describe("API request schemas", () => { + describe("startCaseRequestSchema", () => { + it("validates scenario field", () => { + const result = startCaseRequestSchema.safeParse({ scenario: "Test scenario" }); + expect(result.success).toBe(true); + }); + + it("rejects empty scenario", () => { + const result = startCaseRequestSchema.safeParse({ scenario: "" }); + expect(result.success).toBe(false); + }); + + it("rejects scenario over 10000 chars", () => { + const longScenario = "a".repeat(10001); + const result = startCaseRequestSchema.safeParse({ scenario: longScenario }); + expect(result.success).toBe(false); + }); + + it("accepts optional promptVersion", () => { + const result = startCaseRequestSchema.safeParse({ + scenario: "Test", + promptVersion: "v0.3" + }); + expect(result.success).toBe(true); + }); + }); + + describe("updateCaseRequestSchema", () => { + it("validates complete update request", () => { + const graph = makeGraph({ + centralStatement: "Test scenario", + nodes: [makeNode({ id: "n1", label: "N" })], + currentSummary: "Current state of situation" + }); + const result = updateCaseRequestSchema.safeParse({ + situationGraph: graph, + previousQuestion: "What happened?", + answer: "This is the answer", + }); + expect(result.success).toBe(true); + }); + + it("rejects missing situationGraph", () => { + const result = updateCaseRequestSchema.safeParse({ + previousQuestion: "Q?", + answer: "A", + }); + expect(result.success).toBe(false); + }); + + it("rejects answer over 5000 chars", () => { + const graph = makeGraph({ + centralStatement: "Test", + nodes: [makeNode({ id: "n1", label: "N" })], + currentSummary: "Test summary" + }); + const result = updateCaseRequestSchema.safeParse({ + situationGraph: graph, + previousQuestion: "Q?", + answer: "x".repeat(5001), + }); + expect(result.success).toBe(false); + }); + }); +}); + +describe("deterministic ID generation", () => { + it("generate consistent IDs for same label", () => { + const id1 = makeNodeId("Same Label"); + const id2 = makeNodeId("Same Label"); + expect(id1).toBe(id2); + }); + + it("generates different IDs for different labels", () => { + const id1 = makeNodeId("Label A"); + const id2 = makeNodeId("Label B"); + expect(id1).not.toBe(id2); + }); + + it("IDs are prefixed with 'n' and short", () => { + const id = makeNodeId("A very long label that would produce a longer hash if not truncated"); + expect(id.startsWith("n")).toBe(true); + expect(id.length).toBeLessThan(15); + }); + + it("same kind of nodes get deterministic IDs", () => { + for (let i = 0; i < 10; i++) { + expect(makeNodeId("Test Node")).toBe(makeNodeId("Test Node")); + } + }); +}); + +describe("helper functions", () => { + describe("makeNode", () => { + it("creates a minimal node with defaults", () => { + const node = makeNode({ label: "Minimal" }); + const result = situationNodeSchema.safeParse(node); + expect(result.success).toBe(true); + expect(node.kind).toBe("observation"); + expect(node.status).toBe("unknown"); + expect(node.confidence).toBe("medium"); + }); + + it("creates a node with custom kind/status", () => { + const node = makeNode({ + label: "Custom", + kind: "metric", + status: "known", + confidence: "high", + value: 42, + unit: "count", + }); + expect(node.kind).toBe("metric"); + expect(node.status).toBe("known"); + expect(node.confidence).toBe("high"); + expect(node.value).toBe(42); + expect(node.unit).toBe("count"); + }); + + it("generates ID from label if none provided", () => { + const node = makeNode({ label: "Auto-ID" }); + expect(node.id.startsWith("n")).toBe(true); + }); + }); + + describe("makeEdge", () => { + it("creates a minimal edge with defaults", () => { + const edge = makeEdge({ fromNodeId: "n1", toNodeId: "n2" }); + const result = situationEdgeSchema.safeParse(edge); + expect(result.success).toBe(true); + }); + + it("generates description from node ids if not provided", () => { + const edge = makeEdge({ fromNodeId: "n-alpha", toNodeId: "n-beta" }); + expect(edge.description).toContain("alpha"); + expect(edge.description).toContain("beta"); + }); + }); + + describe("makeGraph", () => { + it("creates a minimal graph with defaults", () => { + const graph = makeGraph({ + centralStatement: "Test", + currentSummary: "Default summary", + nodes: [makeNode({ id: "n1", label: "Placeholder" })] + }); + const result = situationGraphSchema.safeParse(graph); + expect(result.success).toBe(true); + }); + + it("allows specifying nodes and edges", () => { + const graph = makeGraph({ + centralStatement: "Full Graph", + currentSummary: "Full summary", + nodes: [makeNode({ id: "n1", label: "N1" })], + edges: [makeEdge({ fromNodeId: "n1", toNodeId: "n2" })], + }); + expect(graph.nodes.length).toBe(1); + expect(graph.edges.length).toBe(1); + }); + }); +}); + +describe("enum values completeness", () => { + it("SituationKind has all expected values", () => { + const expected = ["observation", "reported_claim", "metric", "state", "transition", "relationship", "assumption", "unknown", "conclusion"]; + const actual = Object.values(SituationKind); + expect(actual).toEqual(expect.arrayContaining(expected)); + }); + + it("SituationStatus has all expected values", () => { + const expected = ["known", "unknown", "provisional", "supported", "weakened", "contradicted", "resolved"]; + const actual = Object.values(SituationStatus); + expect(actual).toEqual(expect.arrayContaining(expected)); + }); + + it("SituationRelationship has all expected values", () => { + const expected = ["supports", "weakens", "contradicts", "depends_on", "causes", "may_cause", "measures", "compares_with", "updates", "other"]; + const actual = Object.values(SituationRelationship); + expect(actual).toEqual(expect.arrayContaining(expected)); + }); + + it("ConfidenceLevel has all expected values", () => { + const actual = Object.values(ConfidenceLevel); + expect(actual).toContain("low"); + expect(actual).toContain("medium"); + expect(actual).toContain("high"); + }); +}); diff --git a/tests/graph/utils.test.js b/tests/graph/utils.test.js new file mode 100644 index 0000000..0deba4e --- /dev/null +++ b/tests/graph/utils.test.js @@ -0,0 +1,825 @@ +import { describe, it, expect } from "vitest"; +import { + validateGraphReferences, + detectDuplicateNodeIds, + detectDuplicateEdges, + findDependentNodes, + findAffectedNodes, + resolveUnknownNode, + selectActiveUnknownCandidate, + applyGraphUpdate, + validateGraphUpdate, +} from "@/lib/graph/utils.js"; +import { makeNode, makeEdge, makeGraph } from "@/lib/graph/schema.js"; + +// ── Helper: build a minimal graph for tests ─────────── + +function makeTestGraph() { + const n1 = makeNode({ id: "n1", label: "Actor A" }); + const n2 = makeNode({ id: "n2", label: "State B" }); + const n3 = makeNode({ id: "n3", label: "Transition C" }); + const n4 = makeNode({ id: "n4", label: "Unknown D" }); + const n5 = makeNode({ id: "n5", label: "Unknown E" }); + + // n2 depends on n1; n3 depends on n2 (transitive depends on n1) + n2.dependsOn.push(n1.id); + n3.dependsOn.push(n2.id); + + // n4 is an unknown not depended on + // n5 is an unknown depended upon by n3 indirectly + + const e1 = makeEdge({ id: "e1", fromNodeId: n1.id, toNodeId: n2.id, relationship: "depends_on" }); + const e2 = makeEdge({ id: "e2", fromNodeId: n3.id, toNodeId: n1.id, relationship: "supports" }); + + return makeGraph({ + centralStatement: "Test graph", + nodes: [n1, n2, n3, n4, n5], + edges: [e1, e2], + activeUnknownNodeId: n4.id, + resolvedNodeIds: [], + currentSummary: "Test", + }); +} + +describe("validateGraphReferences", () => { + it("accepts valid graph with all self-consistent references", () => { + const graph = makeTestGraph(); + const result = validateGraphReferences(graph); + expect(result.valid).toBe(true); + expect(result.errors.length).toBe(0); + }); + + it("detects invalid parentId reference", () => { + const graph = makeTestGraph(); + // n1 has no parentId, so this won't trigger; let's add one manually + graph.nodes[0].parentId = "nonexistent-parent"; + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + expect(result.errors.some(e => e.includes("nonexistent-parent"))).toBe(true); + }); + + it("detects invalid childIds reference", () => { + const graph = makeTestGraph(); + graph.nodes[0].childIds.push("ghost-node"); + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + }); + + it("detects invalid dependsOn reference", () => { + const graph = makeTestGraph(); + graph.nodes[0].dependsOn.push("phantom-dep"); + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + }); + + it("detects invalid affects reference", () => { + const graph = makeTestGraph(); + graph.nodes[0].affects.push("void-node"); + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + }); + + it("detects edge referencing non-existent fromNodeId", () => { + const graph = makeTestGraph(); + graph.edges[0].fromNodeId = "ghost-node"; + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + expect(result.errors.some(e => e.includes("ghost-node"))).toBe(true); + }); + + it("detects edge referencing non-existent toNodeId", () => { + const graph = makeTestGraph(); + graph.edges[0].toNodeId = "void-node"; + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + }); + + it("allows mixed valid and invalid references", () => { + const graph = makeTestGraph(); + graph.nodes[0].parentId = "missing"; + graph.nodes[1].parentId = "also-missing"; + + const result = validateGraphReferences(graph); + expect(result.valid).toBe(false); + expect(result.errors.length).toBe(2); + }); +}); + +describe("detectDuplicateNodeIds", () => { + it("returns empty for unique nodes", () => { + const graph = makeTestGraph(); + const dups = detectDuplicateNodeIds(graph.nodes); + expect(dups.length).toBe(0); + }); + + it("detects exact duplicate IDs", () => { + const n1 = makeNode({ id: "dup", label: "First" }); + const n2 = makeNode({ id: "dup", label: "Second" }); + const dups = detectDuplicateNodeIds([n1, n2]); + expect(dups.length).toBe(1); + expect(dups[0].nodeId).toBe("dup"); + expect(dups[0].count).toBe(2); + }); + + it("detects multiple duplicate groups", () => { + const nodes = [ + makeNode({ id: "dup", label: "A" }), + makeNode({ id: "dup", label: "B" }), + makeNode({ id: "dup", label: "C" }), + makeNode({ id: "dup2", label: "D" }), + makeNode({ id: "dup2", label: "E" }), + ]; + const dups = detectDuplicateNodeIds(nodes); + expect(dups.length).toBe(2); + }); + + it("reports correct count for triple duplicates", () => { + const nodes = [ + makeNode({ id: "trip", label: "1" }), + makeNode({ id: "trip", label: "2" }), + makeNode({ id: "trip", label: "3" }), + ]; + const dups = detectDuplicateNodeIds(nodes); + expect(dups[0].count).toBe(3); + }); +}); + +describe("detectDuplicateEdges", () => { + it("returns empty for unique edges", () => { + const graph = makeTestGraph(); + const dups = detectDuplicateEdges(graph.edges); + expect(dups.length).toBe(0); + }); + + it("detects duplicate edge (same from, to, relationship)", () => { + const n1 = makeNode({ id: "n1", label: "A" }); + const n2 = makeNode({ id: "n2", label: "B" }); + const e1 = makeEdge({ id: "e1", fromNodeId: n1.id, toNodeId: n2.id, relationship: "supports" }); + const e2 = makeEdge({ id: "e2", fromNodeId: n1.id, toNodeId: n2.id, relationship: "supports" }); + + const dups = detectDuplicateEdges([e1, e2]); + expect(dups.length).toBe(1); + }); + + it("allows same nodes with different relationship types", () => { + const n1 = makeNode({ id: "n1", label: "A" }); + const n2 = makeNode({ id: "n2", label: "B" }); + const e1 = makeEdge({ id: "e1", fromNodeId: n1.id, toNodeId: n2.id, relationship: "supports" }); + const e2 = makeEdge({ id: "e2", fromNodeId: n1.id, toNodeId: n2.id, relationship: "weakens" }); + + const dups = detectDuplicateEdges([e1, e2]); + expect(dups.length).toBe(0); + }); + + it("detects reversed direction as different edge", () => { + const n1 = makeNode({ id: "n1", label: "A" }); + const n2 = makeNode({ id: "n2", label: "B" }); + const e1 = makeEdge({ id: "e1", fromNodeId: n1.id, toNodeId: n2.id, relationship: "supports" }); + const e2 = makeEdge({ id: "e2", fromNodeId: n2.id, toNodeId: n1.id, relationship: "supports" }); + + const dups = detectDuplicateEdges([e1, e2]); + expect(dups.length).toBe(0); + }); +}); + +describe("findDependentNodes (transitive)", () => { + it("returns empty for node with no dependents", () => { + const graph = makeTestGraph(); + // n5 has nothing depending on it + const deps = findDependentNodes(graph, "n5"); + expect(deps.length).toBe(0); + }); + + it("finds direct dependents via dependsOn", () => { + const graph = makeTestGraph(); + // n2 depends on n1 + const deps = findDependentNodes(graph, "n1"); + expect(deps).toContain("n2"); + }); + + it("finds transitive dependents via dependsOn chain", () => { + const graph = makeTestGraph(); + // n3 depends on n2 depends on n1 — so both n2 and n3 depend on n1 + const deps = findDependentNodes(graph, "n1"); + expect(deps).toContain("n2"); + expect(deps).toContain("n3"); + }); + + it("finds dependents via edge relationship too", () => { + const graph = makeTestGraph(); + // e2: n3 -> n1 (supports), so if we query for nodes depending on n1 + // the function also looks at edges where toNodeId === queriedId + const deps = findDependentNodes(graph, "n1"); + expect(deps).toContain("n2"); + }); + + it("returns self if node depends on itself", () => { + const graph = makeTestGraph(); + graph.nodes[0].dependsOn.push("n1"); // n1 depends on n1 (circular) + const deps = findDependentNodes(graph, "n1"); + expect(deps).toContain("n1"); + }); + + it("handles deep dependency chains", () => { + const nodes = []; + for (let i = 1; i <= 10; i++) { + nodes.push(makeNode({ id: `n${i}`, label: `N${i}` })); + } + // Chain: n2 depends on n1, n3 depends on n2, ..., n10 depends on n9 + for (let i = 2; i <= 10; i++) { + nodes[i - 1].dependsOn.push(nodes[0].id); // All depend on n1 + } + + const graph = makeGraph({ + centralStatement: "Chain", + nodes, + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Test", + }); + + const deps = findDependentNodes(graph, "n1"); + expect(deps.length).toBe(9); // All other nodes depend on n1 + }); +}); + +describe("findAffectedNodes (transitive)", () => { + it("returns empty for node that affects nothing", () => { + const graph = makeTestGraph(); + const affected = findAffectedNodes(graph, "n5"); + expect(affected.length).toBe(0); + }); + + it("finds nodes listed in affects array", () => { + // Set up: n2 has n3 in its affects list + const graph = makeTestGraph(); + graph.nodes[1].affects.push("n3"); + const affected = findAffectedNodes(graph, "n2"); + expect(affected).toContain("n3"); + }); + + it("propagates through dependsOn transitive chain", () => { + // n3 depends on n2, and n2's affects includes some node that depends on n3 + const graph = makeTestGraph(); + // If n2 is changed and n3 depends on n2, then n3 should be affected + graph.nodes[2].dependsOn.push("n2"); // Explicit dependency + const affected = findAffectedNodes(graph, "n2"); + expect(affected).toContain("n3"); + }); + + it("handles empty graph", () => { + // build a minimal graph without triggering schema validation for this edge case + const graph = { centralStatement: "Empty", nodes: [], edges: [], resolvedNodeIds: [], currentSummary: "", activeUnknownNodeId: null }; + const affected = findAffectedNodes(graph, "any-node"); + expect(affected.length).toBe(0); + }); +}); + +describe("resolveUnknownNode", () => { + it("returns success for valid node id", () => { + const graph = makeTestGraph(); + const result = resolveUnknownNode(graph, "n4", "resolved", "Confirmed", "User confirmed"); + expect(result.success).toBe(true); + expect(result.newStatus).toBe("resolved"); + expect(result.reason).toBe("User confirmed"); + }); + + it("returns error for non-existent node", () => { + const graph = makeTestGraph(); + const result = resolveUnknownNode(graph, "ghost-node", "resolved", null, "reason"); + expect(result.success).toBe(false); + expect(result.error).toContain("not found"); + }); + + it("reports affectedNodes in result", () => { + const graph = makeTestGraph(); + // n5 depends on... actually let's set up properly + graph.nodes[3].affects.push("n1"); // Unknown depends on Actor A + graph.nodes[3].dependsOn.push("n2"); // Unknown depends on State B + const result = resolveUnknownNode(graph, "n4", "resolved", "Yes", "Clarified"); + expect(result.success).toBe(true); + }); + + it("tracks previous status and value", () => { + const graph = makeTestGraph(); + const result = resolveUnknownNode(graph, "n4", "known", "confirmed_value", "Evidence found"); + expect(result.previousStatus).toBe("unknown"); + expect(result.newValue).toBe("confirmed_value"); + }); +}); + +describe("selectActiveUnknownCandidate", () => { + it("returns null when no unresolved unknowns", () => { + // makeTestGraph nodes default to kind "observation", not "unknown" + // Create explicit unknown-kind nodes for this test + const nUnknown = makeNode({ id: "n-unk-x", label: "Unknown X", kind: "unknown" }); + const graph = makeGraph({ + centralStatement: "Test", + nodes: [nUnknown], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Test", + }); + // Mark it as resolved so no unresolved unknowns remain + const result = selectActiveUnknownCandidate(graph, ["n-unk-x"]); + expect(result).toBeNull(); + }); + + it("skips already-resolved nodes and returns remaining unknown", () => { + const n1 = makeNode({ id: "n1", label: "A", kind: "observation" }); + const n2 = makeNode({ id: "n-unk-b", label: "Unknown B", kind: "unknown" }); + const graph = makeGraph({ + centralStatement: "Test", + nodes: [n1, n2], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Test", + }); + + // Skip n2 by passing it as resolved; no unknown-kind nodes remain + const result = selectActiveUnknownCandidate(graph, ["n-unk-b"]); + expect(result).toBeNull(); + + // Without skipping, should return n2 + const result2 = selectActiveUnknownCandidate(graph, []); + expect(result2.nodeId).toBe("n-unk-b"); + }); + + it("prioritises nodes with more dependents", () => { + const unknownA = makeNode({ id: "unknown-a", label: "Unknown A", kind: "unknown" }); + const unknownB = makeNode({ id: "unknown-b", label: "Unknown B", kind: "unknown" }); + const dependent = makeNode({ id: "dep", label: "Dependent", kind: "state" }); + + dependent.dependsOn.push("unknown-a"); + + const graph = makeGraph({ + centralStatement: "Priority test", + nodes: [unknownA, unknownB, dependent], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Test", + }); + + const result = selectActiveUnknownCandidate(graph, []); + expect(result.nodeId).toBe("unknown-a"); // Has more dependents (score 2 vs 0) + }); + + it("returns one candidate (not array)", () => { + const n1 = makeNode({ id: "n1", label: "A", kind: "observation" }); + const nUnknown = makeNode({ id: "n-unk", label: "Pending", kind: "unknown" }); + const graph = makeGraph({ + centralStatement: "Test", + nodes: [n1, nUnknown], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: [], + currentSummary: "Test", + }); + + const result = selectActiveUnknownCandidate(graph, []); + expect(typeof result).toBe("object"); + expect(result.nodeId).toBeDefined(); + expect(result.label).toBeDefined(); + expect(result.score).toBeDefined(); + }); +}); + +describe("applyGraphUpdate", () => { + it("applies node additions correctly", () => { + const graph = makeTestGraph(); + const newNode = makeNode({ id: "n-new", label: "New Node" }); + + const update = { + addedNodes: [newNode], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + expect(result.nodes.length).toBe(graph.nodes.length + 1); + expect(result.nodes.some(n => n.id === "n-new")).toBe(true); + }); + + it("applies status updates correctly", () => { + const graph = makeTestGraph(); + + const update = { + addedNodes: [], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "resolved", + previousValue: null, + newValue: "confirmed", + reason: "Answered by user", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + + const updatedNode = result.nodes.find(n => n.id === "n4"); + expect(updatedNode.status).toBe("resolved"); + }); + + it("rejects update with non-existent nodeId in updatedNodes", () => { + const graph = makeTestGraph(); + + const update = { + addedNodes: [], + updatedNodes: [{ + nodeId: "ghost-node", + previousStatus: null, + newStatus: "known", + reason: "test", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(false); + expect(result.errors.some(e => e.includes("ghost-node"))).toBe(true); + }); + + it("removes requested edges", () => { + const graph = makeTestGraph(); + const edgeIdToRemove = graph.edges[0].id; + + const update = { + addedNodes: [], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [edgeIdToRemove], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + expect(result.edges.length).toBe(graph.edges.length - 1); + expect(result.edges.some(e => e.id === edgeIdToRemove)).toBe(false); + }); + + it("adds edges and updates node dependsOn/affects", () => { + const graph = makeTestGraph(); + const newEdge = makeEdge({ fromNodeId: "n1", toNodeId: "n4", relationship: "supports" }); + + const update = { + addedNodes: [], + updatedNodes: [], + addedEdges: [newEdge], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + + // Check the edge was added + expect(result.edges.some(e => e.id === newEdge.id)).toBe(true); + + // Check node relationship arrays updated + const fromNode = result.nodes.find(n => n.id === "n1"); + const toNode = result.nodes.find(n => n.id === "n4"); + expect(fromNode.childIds).toContain("n4"); + expect(toNode.dependsOn).toContain("n1"); + }); + + it("accumulates resolved node IDs", () => { + const graph = makeTestGraph(); + + const update = { + addedNodes: [], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + expect(result.resolvedNodeIds).toContain("n4"); + }); + + it("rejects adding duplicate node IDs", () => { + const graph = makeTestGraph(); + const existingNode = graph.nodes[0]; // id: "n1" + + // Use the exact same ID as an existing node to create a real duplicate + const update = { + addedNodes: [{ ...existingNode, id: "n1", label: "Dup Node" }], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(false); + }); + + it("rejects edges referencing non-existent nodes", () => { + const graph = makeTestGraph(); + + const update = { + addedNodes: [], + updatedNodes: [], + addedEdges: [{ + id: "e-new", + fromNodeId: "missing-node", + toNodeId: "n1", + relationship: "supports", + confidence: "medium", + description: "bad edge", + }], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(false); + }); + + it("preserves nodes not mentioned in the update", () => { + const graph = makeTestGraph(); + const unchangedCount = graph.nodes.length; + + const update = { + addedNodes: [], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + expect(result.nodes.length).toBe(unchangedCount); + }); + + it("applies multiple operations in one update", () => { + const graph = makeTestGraph(); + const newNode = makeNode({ id: "n-multi", label: "Multi" }); + + const update = { + addedNodes: [newNode], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "resolved", + reason: "Multiple ops test", + }], + addedEdges: [makeEdge({ fromNodeId: "n-multi", toNodeId: "n1" })], + removedEdgeIds: [graph.edges[0]?.id || ""], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }; + + const result = applyGraphUpdate(graph, update); + expect(result.success).toBe(true); + }); +}); + +describe("validateGraphUpdate", () => { + it("accepts a no-op update with added nodes", () => { + const graph = makeTestGraph(); + const newNode = makeNode({ id: "n-new", label: "New" }); + + const result = validateGraphUpdate(graph, { + addedNodes: [newNode], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(true); + }); + + it("rejects update with no meaningful change", () => { + const graph = makeTestGraph(); + + const result = validateGraphUpdate(graph, { + addedNodes: [], + updatedNodes: [{ + nodeId: "n1", + previousStatus: null, + newStatus: null, + previousValue: null, + newValue: null, + reason: "No change test", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(false); + expect(result.errors.some(e => e.includes("no meaningful"))).toBe(true); + }); + + it("rejects duplicate node IDs in additions", () => { + const graph = makeTestGraph(); + const existingNode = graph.nodes[0]; + + const result = validateGraphUpdate(graph, { + addedNodes: [existingNode], // Duplicate ID + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(false); + }); + + it("rejects update to non-existent node", () => { + const graph = makeTestGraph(); + + const result = validateGraphUpdate(graph, { + addedNodes: [], + updatedNodes: [{ + nodeId: "ghost-node", + previousStatus: null, + newStatus: "known", + reason: "test", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(false); + }); + + it("accepts valid status change as meaningful", () => { + const graph = makeTestGraph(); + + const result = validateGraphUpdate(graph, { + addedNodes: [], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "known", + reason: "Confirmed", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(true); + }); + + it("rejects oversized update (>100KB)", () => { + const graph = makeTestGraph(); + const largeDescription = "x".repeat(150000); + + const result = validateGraphUpdate(graph, { + addedNodes: [{ label: largeDescription }], // Will create huge JSON + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(false); + expect(result.errors.some(e => e.includes("100KB") || e.includes("exceeds"))).toBe(true); + }); + + it("returns empty errors array for valid update", () => { + const graph = makeTestGraph(); + + const result = validateGraphUpdate(graph, { + addedNodes: [makeNode({ id: "n-valid", label: "Valid" })], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }); + + expect(result.valid).toBe(true); + expect(result.errors.length).toBe(0); + }); +}); + +// ── Integration: full update lifecycle ─────────────────── + +describe("update lifecycle integration", () => { + it("complete update cycle: validate → apply → verify", () => { + const graph = makeTestGraph(); + + // Create a meaningful update + const newNode = makeNode({ id: "n-new", label: "New Discovery" }); + const newEdge = makeEdge({ fromNodeId: "n1", toNodeId: "n-new", relationship: "supports" }); + + // Validate first + const validationResult = validateGraphUpdate(graph, { + addedNodes: [newNode], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "resolved", + reason: "Answered via follow-up question", + }], + addedEdges: [newEdge], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }); + expect(validationResult.valid).toBe(true); + + // Apply + const applyResult = applyGraphUpdate(graph, { + addedNodes: [newNode], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "resolved", + reason: "Answered via follow-up question", + }], + addedEdges: [newEdge], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }); + + expect(applyResult.success).toBe(true); + expect(applyResult.nodes.length).toBe(graph.nodes.length + 1); + expect(applyResult.edges.length).toBe(graph.edges.length + 1); + expect(applyResult.resolvedNodeIds).toContain("n4"); + + // Verify post-apply integrity + const postValidation = validateGraphReferences(applyResult); + expect(postValidation.valid).toBe(true); + }); + + it("reject and retry: invalid update should be caught", () => { + const graph = makeTestGraph(); + + const invalidUpdate = { + addedNodes: [], + updatedNodes: [{ nodeId: "ghost-node", newStatus: "known", reason: "test" }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }; + + // Validation should catch it + expect(validateGraphUpdate(graph, invalidUpdate).valid).toBe(false); + + // Apply should also catch it + expect(applyGraphUpdate(graph, invalidUpdate).success).toBe(false); + }); + + it("preserve unchanged nodes during update", () => { + const graph = makeTestGraph(); + const originalNode1 = JSON.parse(JSON.stringify(graph.nodes[0])); + + applyGraphUpdate(graph, { + addedNodes: [], + updatedNodes: [{ + nodeId: "n4", + previousStatus: "unknown", + newStatus: "resolved", + reason: "Test preserve", + }], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n4"], + affectedNodeIds: [], + }); + + // Re-read the graph and check n1 wasn't modified + expect(graph.nodes[0].id).toBe("n1"); + expect(graph.nodes[0].status).toBe("unknown"); // unchanged + }); +}); From 84858107b752ccc8839d9ad5169e67ce0a5a69f4 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 06:59:23 +0100 Subject: [PATCH 04/12] chore: document v0.4 route and test status --- docs/v0.4-route-status.md | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 docs/v0.4-route-status.md diff --git a/docs/v0.4-route-status.md b/docs/v0.4-route-status.md new file mode 100644 index 0000000..ff52c37 --- /dev/null +++ b/docs/v0.4-route-status.md @@ -0,0 +1,20 @@ +# v0.4 Route Status + +- `app/api/cases/start/route.js` + - Current tracked start-case route for the v0.4 graph orchestration path. + - Covered by `tests/app/api/cases-start-route.test.js`. + +- `app/api/start-case/route.js` + - Untracked earlier experiment / duplicate start route. + - Not referenced by the current UI. + - Still mentioned in untracked handoff docs. + - Leave untracked for now; recommend deletion once route migration is explicitly confirmed. + +- `app/api/update-case/route.js` + - Untracked future `updateCase` work. + - Not referenced by the current UI. + - Leave untracked for the current milestone. + +- Current UI status + - `components/scenario-form.jsx` still calls `/api/analyse`. + - No active UI path currently calls `/api/cases/start`, `/api/start-case`, or `/api/update-case`. From 575b8fd9713564ff16d9af77ec4264c39d13798b Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 07:21:40 +0100 Subject: [PATCH 05/12] feat: connect UI to situation graph start flow --- components/diagnostics-view.jsx | 50 ++++++-- components/scenario-form.jsx | 109 ++++++++-------- components/situation-graph-view.jsx | 121 ++++++++++++++++++ docs/v0.4-route-status.md | 12 +- tests/smoke.test.js | 69 ++++++---- tests/ui/scenario-form.test.jsx | 190 ++++++++++++++++++++++++++++ 6 files changed, 450 insertions(+), 101 deletions(-) create mode 100644 components/situation-graph-view.jsx create mode 100644 tests/ui/scenario-form.test.jsx diff --git a/components/diagnostics-view.jsx b/components/diagnostics-view.jsx index 60a5d05..07433ea 100644 --- a/components/diagnostics-view.jsx +++ b/components/diagnostics-view.jsx @@ -1,3 +1,5 @@ +import React from "react"; + const ValidationIndicator = ({ status }) => { const styles = { valid: "text-green-600", @@ -27,23 +29,53 @@ const validationIcons = { export default function DiagnosticsView({ result }) { if (!result) return null; + const diagnostics = result.diagnostics || result; + const metrics = [ - { label: "Model", value: result.modelName || "?" }, + { label: "Model", value: diagnostics.modelName || result.modelName || "?" }, { label: "Provider", value: "Ollama" }, - { label: "Prompt version", value: result.promptVersion || "?" }, + { + label: "Prompt version", + value: diagnostics.promptVersion || result.promptVersion || "?", + }, { label: "Duration", value: - result.responseDurationMs != null - ? `${result.responseDurationMs}ms` + diagnostics.responseDurationMs != null + ? `${diagnostics.responseDurationMs}ms` : "?", }, { label: "Validation", value: ( - + ), }, + { + label: "Node count", + value: diagnostics.nodeCount != null ? diagnostics.nodeCount : "?", + }, + { + label: "Edge count", + value: diagnostics.edgeCount != null ? diagnostics.edgeCount : "?", + }, + { + label: "Graph references", + value: + diagnostics.graphReferenceValidation == null + ? "?" + : diagnostics.graphReferenceValidation.valid + ? `${validationIcons.valid} valid` + : `${validationIcons.invalid} invalid`, + }, + ]; + + const errors = [ + ...(result.errors || []), + ...(result.validationErrors || []), + ...(result.analysisErrors || []), ]; return ( @@ -72,14 +104,14 @@ export default function DiagnosticsView({ result }) { )} {/* Errors if present */} - {result.errors && result.errors.length > 0 && ( + {errors.length > 0 && (
- Validation errors ({result.errors.length}) + Validation errors ({errors.length})
    - {result.errors.map((err, i) => ( -
  • {err}
  • + {errors.map((err, i) => ( +
  • {typeof err === "string" ? err : err?.message || JSON.stringify(err)}
  • ))}
diff --git a/components/scenario-form.jsx b/components/scenario-form.jsx index fb2e7d8..e71bc28 100644 --- a/components/scenario-form.jsx +++ b/components/scenario-form.jsx @@ -1,14 +1,59 @@ "use client"; +import React from "react"; import { useState, useRef } from "react"; -import ReconstructionView from "@/components/reconstruction-view"; import DiagnosticsView from "@/components/diagnostics-view"; +import SituationGraphView from "@/components/situation-graph-view"; const MAX_LENGTH = 10000; +export async function submitScenarioForStartCase(fetchImpl, scenario) { + return fetchImpl("/api/cases/start", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ scenario }), + }); +} + +export function ScenarioResultPanels({ status, result }) { + if (!result) return null; + + const hasGraph = Boolean(result.situationGraph); + const hasQuestion = Boolean(result.selectedQuestion?.question); + const hasDiagnostics = Boolean(result.diagnostics); + + return ( + <> + {status === "error" && ( +
+ {result.error && ( +
+ Error: {result.error} +
+ )} + {!hasGraph && !hasQuestion && ( +
+ Validation failed — no structured graph output was produced. +
+ )} +
+ )} + + {(status === "success" || hasGraph || hasQuestion) && ( + + )} + + {hasDiagnostics && } + + ); +} + export default function ScenarioForm() { const [scenario, setScenario] = useState(""); - const [status, setStatus] = useState("idle"); // idle | loading | error | success | partial + const [status, setStatus] = useState("idle"); // idle | loading | error | success const [result, setResult] = useState(null); const textareaRef = useRef(null); @@ -18,19 +63,11 @@ export default function ScenarioForm() { setResult(null); try { - const res = await fetch("/api/analyse", { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ scenario }), - }); + const res = await submitScenarioForStartCase(fetch, scenario); const data = await res.json(); - if (res.ok && data.validationStatus === "valid") { - setStatus("success"); - setResult(data); - } else if (data.success) { - // Success in analysis but validation may be partial + if (res.ok && data.success) { setStatus("success"); setResult(data); } else { @@ -43,14 +80,6 @@ export default function ScenarioForm() { } }; - // Determine if we have meaningful content to display - const hasClassification = result?.inputClassification; - const hasReconstruction = result?.reconstruction; - const hasNextQuestion = result?.nextQuestion; - const hasEvidence = result?.evidence && result.evidence.length > 0; - const hasMeaningfulContent = - hasClassification || hasReconstruction || hasNextQuestion || hasEvidence; - return (
@@ -76,38 +105,7 @@ export default function ScenarioForm() {
- {/* Error state */} - {status === "error" && ( -
- {result?.error && ( -
- Error: {result.error} -
- )} - {/* Show partial content even on validation failure */} - {(hasClassification || hasReconstruction) && ( -
- ⚠ Partial result — some fields failed validation. Showing what was - accepted. -
- )} - {hasReconstruction && ( - - )} -
- )} - - {/* Success state */} - {status === "success" && hasMeaningfulContent && ( -
- -
- )} - - {/* Always show diagnostics when we have any result */} - {(hasClassification || hasReconstruction || hasNextQuestion) && ( - - )} + {status === "loading" && (
@@ -123,13 +121,6 @@ export default function ScenarioForm() {

)} - - {/* Invalid result with no partial data */} - {status === "error" && !result?.error && !hasMeaningfulContent && ( -
- Validation failed — no structured output was produced. -
- )} ); } diff --git a/components/situation-graph-view.jsx b/components/situation-graph-view.jsx new file mode 100644 index 0000000..81bd33a --- /dev/null +++ b/components/situation-graph-view.jsx @@ -0,0 +1,121 @@ +"use client"; + +import React from "react"; + +function NodeBadge({ children, tone = "gray" }) { + const tones = { + gray: "border-gray-200 bg-gray-50 text-gray-700", + blue: "border-blue-200 bg-blue-50 text-blue-700", + green: "border-green-200 bg-green-50 text-green-700", + yellow: "border-yellow-200 bg-yellow-50 text-yellow-700", + }; + + return ( + + {children} + + ); +} + +function NodeGroup({ title, nodes }) { + if (!nodes?.length) return null; + + return ( +
+

+ {title} ({nodes.length}) +

+
    + {nodes.map((node) => ( +
  • +
    + {node.label} + {node.status} + {node.confidence} + {node.value != null && ( + + {node.value} + {node.unit ? ` ${node.unit}` : ""} + + )} +
    + {node.description && node.description !== node.label && ( +

    {node.description}

    + )} +
  • + ))} +
+
+ ); +} + +export default function SituationGraphView({ + situationGraph, + selectedQuestion, +}) { + if (!situationGraph) return null; + + const activeUnknown = situationGraph.activeUnknownNodeId + ? situationGraph.nodes.find((node) => node.id === situationGraph.activeUnknownNodeId) + : null; + + const nodesByKind = situationGraph.nodes.reduce((acc, node) => { + if (!acc[node.kind]) acc[node.kind] = []; + acc[node.kind].push(node); + return acc; + }, {}); + + return ( +
+ {selectedQuestion?.question && ( +
+

Selected Question

+

{selectedQuestion.question}

+
+ )} + +
+

Situation Graph

+
+
+
Central statement
+
{situationGraph.centralStatement}
+
+ {situationGraph.currentSummary && ( +
+
Current summary
+
{situationGraph.currentSummary}
+
+ )} + {activeUnknown && ( +
+
Active unknown
+
{activeUnknown.label}
+
+ )} +
+
Edge count
+
{situationGraph.edges.length}
+
+
+
+ + {Object.entries(nodesByKind).map(([kind, nodes]) => ( + + ))} + +
+ + Raw graph JSON + +
+          {JSON.stringify(situationGraph, null, 2)}
+        
+
+
+ ); +} \ No newline at end of file diff --git a/docs/v0.4-route-status.md b/docs/v0.4-route-status.md index ff52c37..f423afa 100644 --- a/docs/v0.4-route-status.md +++ b/docs/v0.4-route-status.md @@ -5,10 +5,9 @@ - Covered by `tests/app/api/cases-start-route.test.js`. - `app/api/start-case/route.js` - - Untracked earlier experiment / duplicate start route. - - Not referenced by the current UI. - - Still mentioned in untracked handoff docs. - - Leave untracked for now; recommend deletion once route migration is explicitly confirmed. + - Earlier experiment / duplicate start route. + - No repository UI/test references were found. + - Deleted from the working tree during UI connection cleanup. - `app/api/update-case/route.js` - Untracked future `updateCase` work. @@ -16,5 +15,6 @@ - Leave untracked for the current milestone. - Current UI status - - `components/scenario-form.jsx` still calls `/api/analyse`. - - No active UI path currently calls `/api/cases/start`, `/api/start-case`, or `/api/update-case`. + - `components/scenario-form.jsx` now calls `/api/cases/start` for the main experimental flow. + - `/api/analyse` remains available for compatibility. + - No active UI path currently calls `/api/update-case`. diff --git a/tests/smoke.test.js b/tests/smoke.test.js index f5cef5e..a15f642 100644 --- a/tests/smoke.test.js +++ b/tests/smoke.test.js @@ -1,55 +1,70 @@ import { test, expect } from "@playwright/test"; -test("v0.3 UI smoke test with live model response", async ({ page }) => { - await page.goto("http://localhost:3000"); +const BASE_URL = process.env.PLAYWRIGHT_BASE_URL || "http://localhost:3000"; + +test.setTimeout(300000); + +test("graph-backed start flow smoke test", async ({ page }) => { + await page.goto(BASE_URL); // Page should load without error await expect(page.getByText(/Confidence Engine/i)).toBeVisible(); // Type the scenario const textarea = page.locator("textarea[placeholder*='Describe']"); - await textarea.fill("Complaints increased by 35% while production increased by 40%."); + await textarea.fill( + "Complaints increased by 35% while production increased by 40%.", + ); + await expect(textarea).toHaveValue( + "Complaints increased by 35% while production increased by 40%.", + ); // Button should be enabled await expect(page.getByRole("button", { name: /Analyse/i })).toBeEnabled(); - // Click Analyse and wait for diagnostics panel + // Click Analyse and wait for graph-backed result await page.getByRole("button", { name: /Analyse/i }).click(); - - // Wait for result section (ReconstructionView rendered) - await expect(page.getByRole("heading", { name: /Next Question/i })).toBeVisible({ timeout: 180000 }); - // Take screenshot of result page - await page.screenshot({ path: "tests-results/smoke-v0.3.png", fullPage: true }); + await expect( + page.getByRole("heading", { name: /Selected Question/i }), + ).toBeVisible({ timeout: 180000 }); + await expect( + page.getByRole("heading", { name: /Situation Graph/i }), + ).toBeVisible({ timeout: 180000 }); + await expect(page.getByText(/Central statement/i)).toBeVisible(); + await expect(page.getByText(/Active unknown/i)).toBeVisible(); + await expect(page.getByText(/Error:/i)).toHaveCount(0); - // Verify diagnostics panel exists and contains relevant info - const diagPanel = page.locator('details summary').first(); - if (await diagPanel.isVisible()) { - console.log("Raw response viewer:", await diagPanel.innerText().catch(() => "not visible")); - } + const rawJsonToggle = page.getByText(/Raw graph JSON/i); + await expect(rawJsonToggle).toBeVisible(); + await rawJsonToggle.click(); + await expect(page.getByText(/centralStatement/i)).toBeVisible(); // Get full body text for verification const bodyText = await page.locator("body").innerText(); - + console.log("\n=== UI Smoke Test Results ==="); console.log("Page title:", await page.title()); console.log("Body content length:", bodyText.length); - + // Check key content indicators - const hasNextQ = bodyText.includes("Next Question"); - const hasComplaints = bodyText.includes("Complaint") || bodyText.includes("complaint"); - const hasProduction = bodyText.includes("production") || bodyText.includes("Production"); - const hasRateContext = bodyText.toLowerCase().includes("rate") || - bodyText.toLowerCase().includes("unit") || - bodyText.toLowerCase().includes("denominator") || - bodyText.toLowerCase().includes("per-unit"); - - console.log("Has Next Question heading:", hasNextQ); + const hasSelectedQuestion = bodyText.includes("Selected Question"); + const hasComplaints = + bodyText.includes("Complaint") || bodyText.includes("complaint"); + const hasProduction = + bodyText.includes("production") || bodyText.includes("Production"); + const hasRateContext = + bodyText.toLowerCase().includes("rate") || + bodyText.toLowerCase().includes("unit") || + bodyText.toLowerCase().includes("denominator") || + bodyText.toLowerCase().includes("per-unit"); + + console.log("Has selected question heading:", hasSelectedQuestion); console.log("Has complaints reference:", hasComplaints); console.log("Has production reference:", hasProduction); console.log("Has rate context (rate/unit/denominator):", hasRateContext); // Basic structural checks expect(bodyText.length).toBeGreaterThan(200); - expect(hasNextQ).toBe(true); -}, { timeout: 300000 }); + expect(hasSelectedQuestion).toBe(true); +}); diff --git a/tests/ui/scenario-form.test.jsx b/tests/ui/scenario-form.test.jsx new file mode 100644 index 0000000..e08dcbc --- /dev/null +++ b/tests/ui/scenario-form.test.jsx @@ -0,0 +1,190 @@ +import React from "react"; +import { describe, expect, it, vi } from "vitest"; +import { renderToStaticMarkup } from "react-dom/server"; +import DiagnosticsView from "@/components/diagnostics-view.jsx"; +import SituationGraphView from "@/components/situation-graph-view.jsx"; +import { + ScenarioResultPanels, + submitScenarioForStartCase, +} from "@/components/scenario-form.jsx"; + +function makeGraphResult(overrides = {}) { + return { + success: true, + situationGraph: { + centralStatement: "Complaints increased while production increased.", + currentSummary: + "Nodes: 2 observation, 1 unknown | Edges: 2 total | Unknowns: 1 unresolved", + activeUnknownNodeId: "n-unknown", + resolvedNodeIds: [], + nodes: [ + { + id: "n-1", + label: "Complaints up 35%", + description: "Complaints increased by 35%", + kind: "observation", + status: "supported", + confidence: "high", + value: 35, + unit: "%", + }, + { + id: "n-2", + label: "Production up 40%", + description: "Production increased by 40%", + kind: "observation", + status: "supported", + confidence: "high", + value: 40, + unit: "%", + }, + { + id: "n-unknown", + label: "Complaint rate denominator", + description: "Need the denominator for complaint rate", + kind: "unknown", + status: "unknown", + confidence: "medium", + value: null, + unit: null, + }, + ], + edges: [ + { id: "e1", fromNodeId: "n-1", toNodeId: "n-unknown" }, + { id: "e2", fromNodeId: "n-2", toNodeId: "n-unknown" }, + ], + }, + selectedQuestion: { + question: "What denominator is being used for the complaint rate?", + }, + diagnostics: { + modelName: "test", + responseDurationMs: 1234, + validationStatus: "valid", + promptVersion: "test-prompt", + nodeCount: 3, + edgeCount: 2, + graphReferenceValidation: { valid: true, errors: [] }, + }, + ...overrides, + }; +} + +describe("scenario-form UI helpers", () => { + it("submits to /api/cases/start", async () => { + const fetchImpl = vi.fn().mockResolvedValue({ ok: true }); + + await submitScenarioForStartCase(fetchImpl, "Scenario text"); + + expect(fetchImpl).toHaveBeenCalledWith( + "/api/cases/start", + expect.objectContaining({ + method: "POST", + headers: { "Content-Type": "application/json" }, + }), + ); + }); +}); + +describe("graph-backed UI rendering", () => { + it("renders central statement from successful graph response", () => { + const data = makeGraphResult(); + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Central statement"); + expect(html).toContain("Complaints increased while production increased."); + }); + + it("renders active unknown", () => { + const data = makeGraphResult(); + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Active unknown"); + expect(html).toContain("Complaint rate denominator"); + }); + + it("renders selected question exactly once", () => { + const data = makeGraphResult(); + const html = renderToStaticMarkup( + , + ); + + expect( + html.match(/What denominator is being used for the complaint rate\?/g) || + [], + ).toHaveLength(1); + }); + + it("renders diagnostics", () => { + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Diagnostics"); + expect(html).toContain("test"); + expect(html).toContain("1234ms"); + expect(html).toContain("Node count"); + expect(html).toContain("Edge count"); + expect(html).toContain("Graph references"); + }); + + it("hides empty sections", () => { + const base = makeGraphResult(); + const result = makeGraphResult({ + selectedQuestion: null, + situationGraph: { + ...base.situationGraph, + activeUnknownNodeId: null, + nodes: [base.situationGraph.nodes[0]], + edges: [], + }, + }); + + const html = renderToStaticMarkup( + , + ); + + expect(html).not.toContain("Selected Question"); + expect(html).not.toContain("Active unknown"); + }); + + it("displays API error clearly", () => { + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Error: Invalid start-case request"); + }); + + it("renders expandable raw graph JSON", () => { + const data = makeGraphResult(); + const html = renderToStaticMarkup( + , + ); + + expect(html).toContain("Raw graph JSON"); + expect(html).toContain(""centralStatement""); + }); +}); \ No newline at end of file From 02a6ecd0da02ab3c830b526880d1be7688aaa881 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 07:50:02 +0100 Subject: [PATCH 06/12] fix: normalise compatible live reconstruction responses --- lib/analysis.js | 93 +++++++++-- lib/graph/orchestrator.js | 3 + lib/reconstruction/compatibility.js | 40 +++++ tests/graph/orchestrator.test.js | 26 +++ tests/reconstruction/compatibility.test.js | 179 +++++++++++++++++++++ 5 files changed, 325 insertions(+), 16 deletions(-) create mode 100644 lib/reconstruction/compatibility.js create mode 100644 tests/reconstruction/compatibility.test.js diff --git a/lib/analysis.js b/lib/analysis.js index d214dca..3d964c1 100644 --- a/lib/analysis.js +++ b/lib/analysis.js @@ -5,7 +5,12 @@ import { getConfig } from "../lib/config.js"; import { getProvider } from "../lib/llm/provider.js"; -import { buildPrompt, PROMPT_VERSIONS, DEFAULT_PROMPT_VERSION } from "../lib/reconstruction/prompt.js"; +import { + buildPrompt, + PROMPT_VERSIONS, + DEFAULT_PROMPT_VERSION, +} from "../lib/reconstruction/prompt.js"; +import { normaliseAnalysisResponse } from "../lib/reconstruction/compatibility.js"; import { reconstructionV2Schema, reconstructionSchema as reconstructionV1Schema, @@ -33,7 +38,10 @@ export async function analyseScenario(scenario, opts = {}) { return buildErrorResponse("Scenario cannot be empty", startTime); } if (trimmed.length > MAX_SCENARIO_LENGTH) { - return buildErrorResponse(`Scenario must be under ${MAX_SCENARIO_LENGTH} characters`, startTime); + return buildErrorResponse( + `Scenario must be under ${MAX_SCENARIO_LENGTH} characters`, + startTime, + ); } // ── Configuration check ──────────────────────────── @@ -50,18 +58,24 @@ export async function analyseScenario(scenario, opts = {}) { try { promptObj = await buildPrompt(trimmed, promptVersion); } catch (e) { - return buildErrorResponse(`Failed to build prompt: ${e.message}`, startTime); + return buildErrorResponse( + `Failed to build prompt: ${e.message}`, + startTime, + ); } // ── Call provider ────────────────────────────────── const provider = getProvider(); let rawResponse; try { - rawResponse = await provider.generateReconstruction(promptObj.prompt, OLLAMA_MODEL); + rawResponse = await provider.generateReconstruction( + promptObj.prompt, + OLLAMA_MODEL, + ); } catch (e) { return buildErrorResponse( e.message || "Provider error during analysis", - Date.now() - startTime + Date.now() - startTime, ); } @@ -75,16 +89,37 @@ export async function analyseScenario(scenario, opts = {}) { rawResponseStr = String(rawResponse).slice(0, 2000); } + const compatibility = normaliseAnalysisResponse(rawResponse); + const candidateResponse = compatibility.normalised; + // ── Validate against v0.2 schema (preferred) ────── - const resultV2 = tryValidateAgainstSchema(rawResponse, reconstructionV2Schema); + const resultV2 = tryValidateAgainstSchema( + candidateResponse, + reconstructionV2Schema, + ); if (resultV2.valid) { - return buildSuccessResultV2(resultV2.data, OLLAMA_MODEL, duration, promptVersion); + return buildSuccessResultV2( + resultV2.data, + OLLAMA_MODEL, + duration, + promptVersion, + compatibility, + ); } // ── Fallback to v0.1 schema ──────────────────────── - const resultV1 = tryValidateAgainstSchema(rawResponse, reconstructionV1Schema); + const resultV1 = tryValidateAgainstSchema( + candidateResponse, + reconstructionV1Schema, + ); if (resultV1.valid) { - return buildSuccessResultV1(resultV1.data, OLLAMA_MODEL, duration, promptVersion); + return buildSuccessResultV1( + resultV1.data, + OLLAMA_MODEL, + duration, + promptVersion, + compatibility, + ); } // ── Neither schema matched — partial failure ─────── @@ -93,17 +128,23 @@ export async function analyseScenario(scenario, opts = {}) { resultV2.error ?? resultV1.error, OLLAMA_MODEL, duration, - promptVersion + promptVersion, + compatibility, ); } /** Attempt validation against a Zod schema */ function tryValidateAgainstSchema(data, schema) { if (!schema.safeParse) { - return { valid: false, error: new Error("Schema does not support safeParse") }; + return { + valid: false, + error: new Error("Schema does not support safeParse"), + }; } const result = schema.safeParse(data); - return result.success ? { valid: true, data: result.data } : { valid: false, error: result.error }; + return result.success + ? { valid: true, data: result.data } + : { valid: false, error: result.error }; } // ── Result builders ────────────────────────────────── @@ -121,7 +162,15 @@ function buildErrorResponse(message, elapsed, statusCode = 500) { }; } -function buildSuccessResultV2(data, model, duration, version) { +function buildCompatibilityDiagnostics(compatibility) { + return { + compatibilityApplied: compatibility.changesApplied.length > 0, + compatibilityChanges: compatibility.changesApplied, + compatibilityWarnings: compatibility.warnings, + }; +} + +function buildSuccessResultV2(data, model, duration, version, compatibility) { return { success: true, validationStatus: "valid", @@ -134,10 +183,11 @@ function buildSuccessResultV2(data, model, duration, version) { evidence: data.evidence, nextQuestion: data.nextQuestion, errors: undefined, + ...buildCompatibilityDiagnostics(compatibility), }; } -function buildSuccessResultV1(data, model, duration, version) { +function buildSuccessResultV1(data, model, duration, version, compatibility) { return { success: true, validationStatus: "valid", @@ -150,14 +200,24 @@ function buildSuccessResultV1(data, model, duration, version) { evidence: undefined, nextQuestion: undefined, errors: undefined, + ...buildCompatibilityDiagnostics(compatibility), }; } -function buildPartialResult(rawResp, error, model, duration, version) { +function buildPartialResult( + rawResp, + error, + model, + duration, + version, + compatibility, +) { let errors = []; if (error && typeof error.flatten === "function") { errors = error.flatten().fieldErrors - ? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [`${k}: ${v.join(", ")}`]) + ? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [ + `${k}: ${v.join(", ")}`, + ]) : [String(error)]; } else if (error) { errors = [String(error).slice(0, 500)]; @@ -175,6 +235,7 @@ function buildPartialResult(rawResp, error, model, duration, version) { evidence: undefined, nextQuestion: undefined, errors, + ...buildCompatibilityDiagnostics(compatibility), }; } diff --git a/lib/graph/orchestrator.js b/lib/graph/orchestrator.js index abdb0aa..f041af1 100644 --- a/lib/graph/orchestrator.js +++ b/lib/graph/orchestrator.js @@ -34,6 +34,9 @@ function buildDiagnostics({ analysis, graph, graphReferenceValidation }) { nodeCount: graph?.nodes?.length ?? 0, edgeCount: graph?.edges?.length ?? 0, graphReferenceValidation, + compatibilityApplied: analysis?.compatibilityApplied ?? false, + compatibilityChanges: analysis?.compatibilityChanges ?? [], + compatibilityWarnings: analysis?.compatibilityWarnings ?? [], }; } diff --git a/lib/reconstruction/compatibility.js b/lib/reconstruction/compatibility.js new file mode 100644 index 0000000..efb5de2 --- /dev/null +++ b/lib/reconstruction/compatibility.js @@ -0,0 +1,40 @@ +function cloneJsonSafe(value) { + if (value == null) return value; + return JSON.parse(JSON.stringify(value)); +} + +export function normaliseAnalysisResponse(input) { + const normalised = cloneJsonSafe(input); + const changesApplied = []; + const warnings = []; + + if (!normalised || typeof normalised !== "object") { + return { normalised: input, changesApplied, warnings }; + } + + if (Array.isArray(normalised.evidence)) { + normalised.evidence = normalised.evidence.map((record, index) => { + if (!record || typeof record !== "object") return record; + + if (record.source === null) { + changesApplied.push({ + path: ["evidence", index, "source"], + change: "Converted null source to undefined", + }); + + const { source: _removed, ...rest } = record; + return rest; + } + + return record; + }); + } + + if (changesApplied.length > 0) { + warnings.push( + "Applied deterministic reconstruction compatibility normalisation", + ); + } + + return { normalised, changesApplied, warnings }; +} diff --git a/tests/graph/orchestrator.test.js b/tests/graph/orchestrator.test.js index a574821..682a270 100644 --- a/tests/graph/orchestrator.test.js +++ b/tests/graph/orchestrator.test.js @@ -46,6 +46,9 @@ function makeAnalysisResult(overrides = {}) { id: "q-1", question: "What denominator is being used for the complaint rate?", }, + compatibilityApplied: false, + compatibilityChanges: [], + compatibilityWarnings: [], ...overrides, }; } @@ -171,6 +174,29 @@ describe("lib/graph/orchestrator startCase", () => { expect(result.selectedQuestion).toBeNull(); }); + it("includes compatibility diagnostics when provided by analysis", async () => { + mockAnalyseScenario.mockResolvedValue( + makeAnalysisResult({ + compatibilityApplied: true, + compatibilityChanges: [ + { + path: ["evidence", 0, "source"], + change: "Converted null source to undefined", + }, + ], + compatibilityWarnings: [ + "Applied deterministic reconstruction compatibility normalisation", + ], + }), + ); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result.diagnostics.compatibilityApplied).toBe(true); + expect(result.diagnostics.compatibilityChanges).toHaveLength(1); + }); + it("exports placeholder updateCase", async () => { const { updateCase } = await import("@/lib/graph/orchestrator.js"); diff --git a/tests/reconstruction/compatibility.test.js b/tests/reconstruction/compatibility.test.js new file mode 100644 index 0000000..15fc5e5 --- /dev/null +++ b/tests/reconstruction/compatibility.test.js @@ -0,0 +1,179 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; +import { normaliseAnalysisResponse } from "@/lib/reconstruction/compatibility.js"; + +const mockGenerateReconstruction = vi.fn(); + +vi.mock("@/lib/config.js", () => ({ + getConfig: () => ({ + ok: true, + config: { + OLLAMA_BASE_URL: "http://example.test", + OLLAMA_MODEL: "test-model", + }, + }), +})); + +vi.mock("@/lib/llm/provider.js", () => ({ + getProvider: () => ({ + generateReconstruction: (...args) => mockGenerateReconstruction(...args), + }), +})); + +vi.mock("@/lib/reconstruction/prompt.js", () => ({ + buildPrompt: async () => ({ prompt: "prompt", version: "v0.3" }), + PROMPT_VERSIONS: ["v0.1", "v0.2", "v0.3"], + DEFAULT_PROMPT_VERSION: "v0.3", +})); + +describe("normaliseAnalysisResponse", () => { + beforeEach(() => { + vi.resetModules(); + vi.clearAllMocks(); + }); + + it("leaves already-valid responses unchanged", () => { + const input = { + evidence: [ + { + id: "ev1", + description: "x", + evidenceType: "reported_statement", + confidence: "medium", + importance: "important", + source: "report", + }, + ], + }; + + const result = normaliseAnalysisResponse(input); + + expect(result.normalised).toEqual(input); + expect(result.changesApplied).toEqual([]); + }); + + it("normalises null evidence source deterministically", () => { + const input = { + evidence: [ + { + id: "ev1", + description: "x", + evidenceType: "reported_statement", + confidence: "medium", + importance: "important", + source: null, + }, + ], + }; + + const result = normaliseAnalysisResponse(input); + + expect(result.normalised.evidence[0]).not.toHaveProperty("source"); + expect(result.changesApplied).toHaveLength(1); + }); + + it("does not invent a next question", () => { + const input = { evidence: [] }; + const result = normaliseAnalysisResponse(input); + expect(result.normalised.nextQuestion).toBeUndefined(); + }); + + it("does not repair missing reasoning content", () => { + const input = { evidence: [{ source: null }] }; + const result = normaliseAnalysisResponse(input); + expect(result.normalised.reconstruction).toBeUndefined(); + }); +}); + +describe("analyseScenario compatibility", () => { + it("succeeds when the only mismatch is null evidence source", async () => { + mockGenerateReconstruction.mockResolvedValue({ + inputClassification: { + primaryType: "unexplained_change", + secondaryTypes: [], + reasoningModes: ["validate_measurement"], + classificationReason: "reason", + confidence: "medium", + }, + reconstruction: { + summary: "summary", + actors: [], + systemsOrObjects: [], + expectedStates: [], + observedStates: [], + differences: [], + knownTransitions: [], + unexplainedTransitions: [], + contradictions: [], + importantUnknowns: [], + plausibleInterpretations: [], + }, + evidence: [ + { + id: "ev1", + description: "desc", + evidenceType: "reported_statement", + source: null, + attribution: null, + confidence: "medium", + importance: "important", + }, + ], + nextQuestion: { + id: "q1", + question: "What denominator?", + targets: ["observedStates"], + reason: "reason", + expectedInformationValue: "high", + reasoningMode: "validate_measurement", + }, + }); + + const { analyseScenario } = await import("@/lib/analysis.js"); + const result = await analyseScenario("Scenario text", { + promptVersion: "v0.3", + }); + + expect(result.success).toBe(true); + expect(result.compatibilityApplied).toBe(true); + expect(result.compatibilityChanges).toHaveLength(1); + expect(result.evidence[0]).not.toHaveProperty("source"); + expect(result.nextQuestion.question).toBe("What denominator?"); + }); + + it("still fails when required reasoning content is missing", async () => { + mockGenerateReconstruction.mockResolvedValue({ + evidence: [ + { + id: "ev1", + description: "desc", + evidenceType: "reported_statement", + source: null, + attribution: null, + confidence: "medium", + importance: "important", + }, + ], + }); + + const { analyseScenario } = await import("@/lib/analysis.js"); + const result = await analyseScenario("Scenario text", { + promptVersion: "v0.3", + }); + + expect(result.success).toBe(false); + expect(result.compatibilityApplied).toBe(true); + expect(result.nextQuestion).toBeUndefined(); + }); + + it("malformed JSON still fails", async () => { + mockGenerateReconstruction.mockResolvedValue("{not valid json"); + + const { analyseScenario } = await import("@/lib/analysis.js"); + const result = await analyseScenario("Scenario text", { + promptVersion: "v0.3", + }); + + expect(result.success).toBe(false); + expect(result.compatibilityApplied).toBe(false); + }); +}); From b38a6a9f2e23d83e3e666878f279f6680b718f29 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 08:00:37 +0100 Subject: [PATCH 07/12] feat: define graph update proposal contract --- lib/graph/prompt-builder.js | 114 +++++++++++++++++++++++ lib/graph/update-proposal.js | 138 ++++++++++++++++++++++++++++ tests/graph/prompt-builder.test.js | 101 ++++++++++++++++++++ tests/graph/update-proposal.test.js | 124 +++++++++++++++++++++++++ 4 files changed, 477 insertions(+) create mode 100644 lib/graph/prompt-builder.js create mode 100644 lib/graph/update-proposal.js create mode 100644 tests/graph/prompt-builder.test.js create mode 100644 tests/graph/update-proposal.test.js diff --git a/lib/graph/prompt-builder.js b/lib/graph/prompt-builder.js new file mode 100644 index 0000000..87b101d --- /dev/null +++ b/lib/graph/prompt-builder.js @@ -0,0 +1,114 @@ +import { + ConfidenceLevel, + SituationKind, + SituationRelationship, + SituationStatus, +} from "./schema.js"; + +const DEFAULT_PROMPT_VERSION = "v0.4"; + +function formatEnumValues(values) { + return Object.values(values).join(" | "); +} + +function formatGraph(graph) { + return JSON.stringify(graph, null, 2); +} + +function formatExampleAnswerBlock() { + return [ + "Example answer the model must be able to handle without hard-coding output:", + '"The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units."', + "This may justify resolving a rate-related unknown or updating a metric node, but only if the current graph and answer support that proposal.", + ].join("\n"); +} + +export function buildGraphUpdatePrompt({ + situationGraph, + previousQuestion, + answer, + promptVersion = DEFAULT_PROMPT_VERSION, +}) { + const nodeKinds = formatEnumValues(SituationKind); + const nodeStatuses = formatEnumValues(SituationStatus); + const edgeRelationships = formatEnumValues(SituationRelationship); + const confidenceLevels = formatEnumValues(ConfidenceLevel); + + return `You are proposing a graph update for Confidence Engine ${promptVersion}. + +Return exactly one JSON object matching the GraphUpdate contract. +Return JSON only. Do not include markdown, explanation, or any text before or after the JSON object. + +## Current Situation Graph +${formatGraph(situationGraph)} + +## Previous Selected Question +${previousQuestion} + +## User Answer +${answer} + +## Allowed Node Kinds +${nodeKinds} + +## Allowed Node Statuses +${nodeStatuses} + +## Allowed Edge Relationships +${edgeRelationships} + +## Allowed Confidence Values +${confidenceLevels} + +## Required JSON Field Names +The JSON object must contain exactly these top-level fields: +- addedNodes +- updatedNodes +- addedEdges +- removedEdgeIds +- resolvedUnknownNodeIds +- affectedNodeIds + +## Required Shapes +- addedNodes: array of nodes using these exact keys: + id, label, description, kind, status, confidence, value, unit, evidenceIds, dependsOn, affects, parentId, childIds +- updatedNodes: array of node updates using these exact keys: + nodeId, previousStatus, newStatus, previousValue, newValue, reason +- addedEdges: array of edges using these exact keys: + id, fromNodeId, toNodeId, relationship, confidence, description +- removedEdgeIds: array of strings +- resolvedUnknownNodeIds: array of strings +- affectedNodeIds: array of strings + +## Proposal Rules +1. Propose changes only. Never return a replacement graph. +2. Preserve unrelated nodes and edges by omitting them from the proposal. +3. Reference existing node IDs when updating an existing concept. +4. Use addedNodes only for genuinely new concepts. +5. Resolve the active unknown when the answer supports it. +6. Propagate only through explicit dependencies or relationships already present in the graph. +7. Do not invent evidence. +8. Do not create unsupported causal edges. +9. Do not ask more than one next question. In this contract you are not returning any next-question field at all. +10. Use empty arrays when there are no changes in a category. +11. Never return null array entries. +12. Never use unknown enum values. +13. Do not change existing IDs. +14. Do not replace the whole graph, and do not restate unchanged graph content inside the proposal. + +## Additional Guidance +- If the answer only clarifies an existing unknown, prefer updatedNodes and resolvedUnknownNodeIds over creating duplicate nodes. +- If a new metric or observation is necessary, add the smallest set of nodes and edges needed. +- If the answer does not justify a change, return empty arrays for every category. + +## Example Constraint Reminder +${formatExampleAnswerBlock()} + +## Output Contract Reminder +Return one JSON object only, with exact field names and exact enum values. +Never include a full graph. +Never include a nextQuestion field. +`; +} + +export const buildUpdatePrompt = buildGraphUpdatePrompt; diff --git a/lib/graph/update-proposal.js b/lib/graph/update-proposal.js new file mode 100644 index 0000000..f44a581 --- /dev/null +++ b/lib/graph/update-proposal.js @@ -0,0 +1,138 @@ +import { graphUpdateSchema } from "./schema.js"; + +const TOP_LEVEL_ARRAY_FIELDS = [ + "addedNodes", + "updatedNodes", + "addedEdges", + "removedEdgeIds", + "resolvedUnknownNodeIds", + "affectedNodeIds", +]; + +function cloneJsonSafe(value) { + if (value == null) return value; + return JSON.parse(JSON.stringify(value)); +} + +function removeNullArrayEntries(value, path = [], normalisationsApplied = []) { + if (Array.isArray(value)) { + const filtered = []; + value.forEach((item, index) => { + if (item === null) { + normalisationsApplied.push({ + path: [...path, index], + change: "Removed null array entry", + }); + return; + } + filtered.push( + removeNullArrayEntries(item, [...path, index], normalisationsApplied), + ); + }); + return filtered; + } + + if (value && typeof value === "object") { + return Object.fromEntries( + Object.entries(value).map(([key, child]) => [ + key, + removeNullArrayEntries(child, [...path, key], normalisationsApplied), + ]), + ); + } + + return value; +} + +function applyKnownEnumAliases(proposal, normalisationsApplied) { + if (!proposal || typeof proposal !== "object") return proposal; + + if (Array.isArray(proposal.addedNodes)) { + proposal.addedNodes = proposal.addedNodes.map((node, index) => { + if (node?.kind === "reported_statement") { + normalisationsApplied.push({ + path: ["addedNodes", index, "kind"], + change: "Converted reported_statement to reported_claim", + }); + return { ...node, kind: "reported_claim" }; + } + return node; + }); + } + + return proposal; +} + +function fillMissingOptionalArrays(proposal, normalisationsApplied) { + if (!proposal || typeof proposal !== "object") return proposal; + + for (const field of TOP_LEVEL_ARRAY_FIELDS) { + if (!(field in proposal)) { + proposal[field] = []; + normalisationsApplied.push({ + path: [field], + change: "Filled missing optional array with []", + }); + } + } + + return proposal; +} + +export function parseGraphUpdateProposal(rawResponse) { + const raw = rawResponse; + let parsed; + + if (typeof rawResponse === "string") { + try { + parsed = JSON.parse(rawResponse); + } catch (error) { + return { + success: false, + proposal: null, + raw, + normalisationsApplied: [], + errors: [error.message || "Model response is not valid JSON"], + }; + } + } else if (rawResponse && typeof rawResponse === "object") { + parsed = cloneJsonSafe(rawResponse); + } else { + return { + success: false, + proposal: null, + raw, + normalisationsApplied: [], + errors: ["Graph update proposal must be a JSON object or JSON string"], + }; + } + + const normalisationsApplied = []; + let normalised = removeNullArrayEntries(parsed, [], normalisationsApplied); + normalised = applyKnownEnumAliases(normalised, normalisationsApplied); + normalised = fillMissingOptionalArrays(normalised, normalisationsApplied); + + const parsedProposal = graphUpdateSchema.safeParse(normalised); + + if (!parsedProposal.success) { + return { + success: false, + proposal: null, + raw, + normalisationsApplied, + errors: parsedProposal.error.issues.map((issue) => ({ + path: issue.path, + message: issue.message, + code: issue.code, + })), + }; + } + + return { + success: true, + proposal: parsedProposal.data, + raw, + normalisationsApplied, + errors: [], + }; +} diff --git a/tests/graph/prompt-builder.test.js b/tests/graph/prompt-builder.test.js new file mode 100644 index 0000000..7ea795a --- /dev/null +++ b/tests/graph/prompt-builder.test.js @@ -0,0 +1,101 @@ +import { describe, expect, it } from "vitest"; +import { buildGraphUpdatePrompt } from "@/lib/graph/prompt-builder.js"; +import { makeEdge, makeGraph, makeNode } from "@/lib/graph/schema.js"; + +function makeContext() { + const unknown = makeNode({ + id: "n-unknown", + label: "Complaint rate denominator", + description: "Need the denominator to compare complaint rates", + kind: "unknown", + status: "unknown", + confidence: "high", + }); + const observation = makeNode({ + id: "n-obs", + label: "Complaints up 35%", + description: "Complaints increased by 35%", + kind: "observation", + status: "supported", + confidence: "high", + }); + + return { + situationGraph: makeGraph({ + centralStatement: "Complaints increased while production increased.", + nodes: [unknown, observation], + edges: [ + makeEdge({ + id: "e1", + fromNodeId: observation.id, + toNodeId: unknown.id, + relationship: "supports", + confidence: "high", + description: "Observation informs the unknown", + }), + ], + activeUnknownNodeId: unknown.id, + resolvedNodeIds: [], + currentSummary: + "Nodes: 1 observation, 1 unknown | Edges: 1 total | Unknowns: 1 unresolved", + }), + previousQuestion: "What denominator is being used for the complaint rate?", + answer: + "The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.", + }; +} + +describe("buildGraphUpdatePrompt", () => { + it("includes the current graph", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain( + "Complaints increased while production increased.", + ); + expect(prompt).toContain("Complaint rate denominator"); + }); + + it("includes previous question and answer", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain( + "What denominator is being used for the complaint rate?", + ); + expect(prompt).toContain( + "The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.", + ); + }); + + it("contains exact schema keys", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain("addedNodes"); + expect(prompt).toContain("updatedNodes"); + expect(prompt).toContain("addedEdges"); + expect(prompt).toContain("removedEdgeIds"); + expect(prompt).toContain("resolvedUnknownNodeIds"); + expect(prompt).toContain("affectedNodeIds"); + }); + + it("lists enum values", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain( + "observation | reported_claim | metric | state | transition | relationship | assumption | unknown | conclusion", + ); + expect(prompt).toContain( + "known | unknown | provisional | supported | weakened | contradicted | resolved", + ); + expect(prompt).toContain( + "supports | weakens | contradicts | depends_on | causes | may_cause | measures | compares_with | updates | other", + ); + }); + + it("forbids full-graph replacement", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain("Never return a replacement graph"); + expect(prompt).toContain("Propose changes only"); + }); + + it("requires JSON only", () => { + const prompt = buildGraphUpdatePrompt(makeContext()); + expect(prompt).toContain("Return JSON only"); + expect(prompt).toContain("Return one JSON object only"); + }); +}); diff --git a/tests/graph/update-proposal.test.js b/tests/graph/update-proposal.test.js new file mode 100644 index 0000000..6bba205 --- /dev/null +++ b/tests/graph/update-proposal.test.js @@ -0,0 +1,124 @@ +import { describe, expect, it } from "vitest"; +import { parseGraphUpdateProposal } from "@/lib/graph/update-proposal.js"; + +function makeValidProposal(overrides = {}) { + return { + addedNodes: [], + updatedNodes: [ + { + nodeId: "n-unknown", + previousStatus: "unknown", + newStatus: "resolved", + previousValue: null, + newValue: "1.9 complaints per 100 units", + reason: "The answer directly provides the normalized complaint rate.", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n-unknown"], + affectedNodeIds: [], + ...overrides, + }; +} + +describe("parseGraphUpdateProposal", () => { + it("parses a valid proposal", () => { + const result = parseGraphUpdateProposal(makeValidProposal()); + expect(result.success).toBe(true); + expect(result.proposal.updatedNodes).toHaveLength(1); + }); + + it("fails on malformed JSON", () => { + const result = parseGraphUpdateProposal("{not json"); + expect(result.success).toBe(false); + }); + + it("fails when required update content is invalid", () => { + const result = parseGraphUpdateProposal({ + updatedNodes: [{ nodeId: "n-unknown" }], + }); + expect(result.success).toBe(false); + }); + + it("removes null array entries and logs them", () => { + const result = parseGraphUpdateProposal( + JSON.stringify({ + ...makeValidProposal(), + addedNodes: [null], + }), + ); + expect(result.success).toBe(true); + expect(result.proposal.addedNodes).toEqual([]); + expect(result.normalisationsApplied).toEqual( + expect.arrayContaining([ + expect.objectContaining({ change: "Removed null array entry" }), + ]), + ); + }); + + it("fills missing optional arrays with empty arrays", () => { + const result = parseGraphUpdateProposal({ + updatedNodes: [], + }); + expect(result.success).toBe(true); + expect(result.proposal.addedNodes).toEqual([]); + expect(result.proposal.addedEdges).toEqual([]); + expect(result.normalisationsApplied.length).toBeGreaterThan(0); + }); + + it("normalises confirmed enum alias and preserves IDs", () => { + const result = parseGraphUpdateProposal({ + ...makeValidProposal(), + addedNodes: [ + { + id: "n-new", + label: "Reported update", + description: "A new reported claim", + kind: "reported_statement", + status: "supported", + confidence: "medium", + value: null, + unit: null, + evidenceIds: [], + dependsOn: [], + affects: [], + parentId: null, + childIds: [], + }, + ], + }); + expect(result.success).toBe(true); + expect(result.proposal.addedNodes[0].kind).toBe("reported_claim"); + expect(result.proposal.addedNodes[0].id).toBe("n-new"); + }); + + it("unknown enum values still fail", () => { + const result = parseGraphUpdateProposal({ + ...makeValidProposal(), + addedNodes: [ + { + id: "n-new", + label: "Bad node", + description: "Bad node", + kind: "unsupported_kind", + status: "supported", + confidence: "medium", + value: null, + unit: null, + evidenceIds: [], + dependsOn: [], + affects: [], + parentId: null, + childIds: [], + }, + ], + }); + expect(result.success).toBe(false); + }); + + it("does not invent a next question", () => { + const result = parseGraphUpdateProposal(makeValidProposal()); + expect(result.proposal.nextQuestion).toBeUndefined(); + }); +}); From f3cdfce0b0cebe814152f2243f6fefc22d0aeb95 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 08:15:54 +0100 Subject: [PATCH 08/12] feat: add graph update proposal orchestration --- lib/graph/orchestrator.js | 128 +++++++++++- tests/graph/orchestrator.test.js | 328 ++++++++++++++++++++++++++++++- 2 files changed, 452 insertions(+), 4 deletions(-) diff --git a/lib/graph/orchestrator.js b/lib/graph/orchestrator.js index f041af1..62038a8 100644 --- a/lib/graph/orchestrator.js +++ b/lib/graph/orchestrator.js @@ -4,12 +4,17 @@ */ import { analyseScenario } from "../analysis.js"; +import { assertConfig } from "../config.js"; +import { getProvider } from "../llm/provider.js"; import { makeGraph, startCaseRequestSchema, situationGraphSchema, + updateCaseRequestSchema, } from "./schema.js"; import { buildInitialGraph, describeGraph } from "./builder.js"; +import { buildGraphUpdatePrompt } from "./prompt-builder.js"; +import { parseGraphUpdateProposal } from "./update-proposal.js"; import { selectActiveUnknownCandidate, validateGraphReferences, @@ -124,5 +129,126 @@ export async function startCase(body) { } export async function updateCase() { - throw new Error("updateCase is not implemented yet"); + return updateCaseWithDependencies(...arguments); +} + +function sanitiseErrorMessage(error, fallbackMessage) { + if (typeof error?.message === "string" && error.message.trim().length > 0) { + return error.message; + } + + return fallbackMessage; +} + +async function updateCaseWithDependencies(body, dependencies = {}) { + const parsedRequest = updateCaseRequestSchema.safeParse(body); + + if (!parsedRequest.success) { + return { + success: false, + stage: "request_validation", + error: "Invalid update-case request", + validationErrors: toValidationErrors(parsedRequest.error), + statusCode: 400, + }; + } + + const { situationGraph, previousQuestion, answer, promptVersion } = + parsedRequest.data; + + const graphSchemaValidation = situationGraphSchema.safeParse(situationGraph); + const graphReferenceValidation = validateGraphReferences(situationGraph); + + if (!graphSchemaValidation.success || !graphReferenceValidation.valid) { + return { + success: false, + stage: "graph_validation", + error: "Invalid situation graph", + graphValidationErrors: [ + ...(!graphSchemaValidation.success + ? toValidationErrors(graphSchemaValidation.error) + : []), + ...(!graphReferenceValidation.valid + ? graphReferenceValidation.errors + : []), + ], + statusCode: 400, + }; + } + + const buildPrompt = + dependencies.buildGraphUpdatePrompt ?? buildGraphUpdatePrompt; + const parseProposal = + dependencies.parseGraphUpdateProposal ?? parseGraphUpdateProposal; + + let modelName = null; + let rawResponse; + const startedAt = Date.now(); + + try { + const config = dependencies.config ?? assertConfig(); + modelName = config.OLLAMA_MODEL; + + const prompt = buildPrompt({ + situationGraph, + previousQuestion, + answer, + promptVersion, + }); + + const provider = dependencies.provider ?? getProvider(); + rawResponse = await provider.generateReconstruction(prompt, modelName); + } catch (error) { + return { + success: false, + stage: "provider", + error: "Graph update proposal generation failed", + providerErrors: [ + sanitiseErrorMessage( + error, + "Provider failed to generate graph update proposal", + ), + ], + diagnostics: { + promptVersion: promptVersion ?? null, + modelName, + responseDurationMs: Date.now() - startedAt, + normalisationsApplied: [], + }, + statusCode: 502, + }; + } + + const parsedProposal = parseProposal(rawResponse); + const responseDurationMs = Date.now() - startedAt; + + if (!parsedProposal.success) { + return { + success: false, + stage: "proposal_validation", + error: "Invalid graph update proposal", + proposalErrors: parsedProposal.errors, + diagnostics: { + promptVersion: promptVersion ?? null, + modelName, + responseDurationMs, + normalisationsApplied: parsedProposal.normalisationsApplied, + }, + statusCode: 502, + }; + } + + return { + success: true, + stage: "proposal_ready", + proposal: parsedProposal.proposal, + diagnostics: { + promptVersion: promptVersion ?? null, + modelName, + responseDurationMs, + normalisationsApplied: parsedProposal.normalisationsApplied, + graphNodeCount: situationGraph.nodes.length, + graphEdgeCount: situationGraph.edges.length, + }, + }; } diff --git a/tests/graph/orchestrator.test.js b/tests/graph/orchestrator.test.js index 682a270..e863fe8 100644 --- a/tests/graph/orchestrator.test.js +++ b/tests/graph/orchestrator.test.js @@ -1,6 +1,8 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; +import { makeGraph, makeNode } from "@/lib/graph/schema.js"; const mockAnalyseScenario = vi.fn(); +const MOCK_CONFIG = { OLLAMA_MODEL: "configured" }; vi.mock("@/lib/analysis.js", () => ({ analyseScenario: (...args) => mockAnalyseScenario(...args), @@ -53,6 +55,67 @@ function makeAnalysisResult(overrides = {}) { }; } +function makeUpdateGraph() { + const unknown = makeNode({ + id: "n-unknown", + label: "Complaint rate denominator", + description: "Need the denominator for the complaint rate", + kind: "unknown", + status: "unknown", + confidence: "high", + }); + const observation = makeNode({ + id: "n-observation", + label: "Complaint count rose", + description: "Complaint count rose faster than output", + kind: "observation", + status: "supported", + confidence: "high", + }); + + return makeGraph({ + centralStatement: + "Complaint counts increased while production also increased.", + nodes: [unknown, observation], + edges: [], + activeUnknownNodeId: unknown.id, + resolvedNodeIds: [], + currentSummary: "Nodes: 1 unknown, 1 observation | Edges: 0 total", + }); +} + +function makeUpdateRequest(overrides = {}) { + return { + situationGraph: makeUpdateGraph(), + previousQuestion: "What denominator is being used for the complaint rate?", + answer: + "The complaint rate fell from 2.0 complaints per 100 units to 1.9 complaints per 100 units.", + promptVersion: "v0.4", + ...overrides, + }; +} + +function makeProposal(overrides = {}) { + return { + addedNodes: [], + updatedNodes: [ + { + nodeId: "n-unknown", + previousStatus: "unknown", + newStatus: "resolved", + previousValue: null, + newValue: "1.9 complaints per 100 units", + reason: "The answer directly provides the normalized rate.", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n-unknown"], + affectedNodeIds: [], + ...overrides, + }; +} + describe("lib/graph/orchestrator startCase", () => { beforeEach(() => { vi.resetModules(); @@ -197,11 +260,270 @@ describe("lib/graph/orchestrator startCase", () => { expect(result.diagnostics.compatibilityChanges).toHaveLength(1); }); - it("exports placeholder updateCase", async () => { + it("produces a validated update proposal for a valid request", async () => { const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; - await expect(updateCase()).rejects.toThrow( - "updateCase is not implemented yet", + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result).toMatchObject({ + success: true, + stage: "proposal_ready", + proposal: makeProposal(), + diagnostics: { + promptVersion: "v0.4", + modelName: "configured", + graphNodeCount: 2, + graphEdgeCount: 0, + }, + }); + expect(provider.generateReconstruction).toHaveBeenCalledTimes(1); + }); + + it("valid request reaches prompt builder", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const buildGraphUpdatePrompt = vi.fn().mockReturnValue("PROMPT"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + + const request = makeUpdateRequest(); + const result = await updateCase(request, { + buildGraphUpdatePrompt, + provider, + config: MOCK_CONFIG, + }); + + expect(result.success).toBe(true); + expect(buildGraphUpdatePrompt).toHaveBeenCalledWith({ + situationGraph: request.situationGraph, + previousQuestion: request.previousQuestion, + answer: request.answer, + promptVersion: request.promptVersion, + }); + expect(provider.generateReconstruction).toHaveBeenCalledWith( + "PROMPT", + "configured", ); }); + + it("invalid request prevents provider call", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn(), + }; + + const result = await updateCase( + { previousQuestion: "Q?", answer: "A" }, + { + provider, + config: MOCK_CONFIG, + }, + ); + + expect(result).toMatchObject({ + success: false, + stage: "request_validation", + error: "Invalid update-case request", + statusCode: 400, + }); + expect(result.validationErrors).toBeInstanceOf(Array); + expect(provider.generateReconstruction).not.toHaveBeenCalled(); + }); + + it("invalid graph prevents provider call", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn(), + }; + + const graph = makeUpdateGraph(); + graph.nodes[0].dependsOn.push("missing-node"); + + const result = await updateCase( + makeUpdateRequest({ situationGraph: graph }), + { + provider, + config: MOCK_CONFIG, + }, + ); + + expect(result).toMatchObject({ + success: false, + stage: "graph_validation", + error: "Invalid situation graph", + statusCode: 400, + }); + expect(result.graphValidationErrors).toEqual( + expect.arrayContaining([ + expect.stringContaining('depends on "missing-node"'), + ]), + ); + expect(provider.generateReconstruction).not.toHaveBeenCalled(); + }); + + it("prompt includes previous question and answer", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + const request = makeUpdateRequest(); + + await updateCase(request, { + provider, + config: MOCK_CONFIG, + }); + + const prompt = provider.generateReconstruction.mock.calls[0][0]; + expect(prompt).toContain(request.previousQuestion); + expect(prompt).toContain(request.answer); + }); + + it("returns proposal validation failure for malformed JSON", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue("{not json"), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result).toMatchObject({ + success: false, + stage: "proposal_validation", + error: "Invalid graph update proposal", + diagnostics: { + promptVersion: "v0.4", + modelName: "configured", + }, + statusCode: 502, + }); + expect(result.proposalErrors).toBeInstanceOf(Array); + }); + + it("returns structured errors for schema-invalid proposal", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue({ + updatedNodes: [{ nodeId: "n-unknown" }], + }), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result.success).toBe(false); + expect(result.stage).toBe("proposal_validation"); + expect(result.proposalErrors).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + path: expect.any(Array), + message: expect.any(String), + }), + ]), + ); + }); + + it("includes parser normalisations in diagnostics", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue({ + updatedNodes: [], + }), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result.success).toBe(true); + expect(result.diagnostics.normalisationsApplied).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + change: "Filled missing optional array with []", + }), + ]), + ); + }); + + it("returns structured provider-stage failure", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi + .fn() + .mockRejectedValue(new Error("provider offline")), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result).toMatchObject({ + success: false, + stage: "provider", + error: "Graph update proposal generation failed", + providerErrors: ["provider offline"], + statusCode: 502, + }); + }); + + it("does not mutate the input graph", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + const request = makeUpdateRequest(); + const originalGraph = JSON.parse(JSON.stringify(request.situationGraph)); + + await updateCase(request, { + provider, + config: MOCK_CONFIG, + }); + + expect(request.situationGraph).toEqual(originalGraph); + }); + + it("does not call applyGraphUpdate", async () => { + const utils = await import("@/lib/graph/utils.js"); + const applySpy = vi.spyOn(utils, "applyGraphUpdate"); + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + + await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(applySpy).not.toHaveBeenCalled(); + applySpy.mockRestore(); + }); + + it("does not invent a next question outside the proposal", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + }); + + expect(result.selectedQuestion).toBeUndefined(); + expect(result.nextQuestion).toBeUndefined(); + expect(result.proposal.nextQuestion).toBeUndefined(); + }); }); From cb77f955edb7ff9c71bd9e41cf70958b2c6fdb96 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 08:24:56 +0100 Subject: [PATCH 09/12] feat: apply validated graph update proposals --- lib/graph/apply-proposal.js | 292 +++++++++++++++++++++ lib/graph/orchestrator.js | 53 ++++ tests/graph/apply-proposal.test.js | 404 +++++++++++++++++++++++++++++ tests/graph/orchestrator.test.js | 120 +++++++++ 4 files changed, 869 insertions(+) create mode 100644 lib/graph/apply-proposal.js create mode 100644 tests/graph/apply-proposal.test.js diff --git a/lib/graph/apply-proposal.js b/lib/graph/apply-proposal.js new file mode 100644 index 0000000..403b311 --- /dev/null +++ b/lib/graph/apply-proposal.js @@ -0,0 +1,292 @@ +import { describeGraph } from "./builder.js"; +import { graphUpdateSchema, situationGraphSchema } from "./schema.js"; +import { + applyGraphUpdate, + detectDuplicateNodeIds, + findAffectedNodes, + selectActiveUnknownCandidate, + validateGraphReferences, + validateGraphUpdate, +} from "./utils.js"; + +function cloneJsonSafe(value) { + return JSON.parse(JSON.stringify(value)); +} + +function zodIssuesToErrors(error) { + return ( + error?.issues?.map((issue) => { + const path = issue.path?.length ? `${issue.path.join(".")}: ` : ""; + return `${path}${issue.message}`; + }) ?? ["Validation failed"] + ); +} + +function collectDuplicateEdgeIds(edges) { + const counts = new Map(); + + for (const edge of edges) { + counts.set(edge.id, (counts.get(edge.id) ?? 0) + 1); + } + + return [...counts.entries()] + .filter(([, count]) => count > 1) + .map(([edgeId, count]) => ({ edgeId, count })); +} + +function buildAffectedNodeIds(graph, proposal) { + const affected = new Set(proposal.affectedNodeIds ?? []); + + for (const update of proposal.updatedNodes ?? []) { + affected.add(update.nodeId); + for (const nodeId of findAffectedNodes(graph, update.nodeId)) { + affected.add(nodeId); + } + } + + for (const nodeId of proposal.resolvedUnknownNodeIds ?? []) { + affected.add(nodeId); + for (const affectedNodeId of findAffectedNodes(graph, nodeId)) { + affected.add(affectedNodeId); + } + } + + return [...affected]; +} + +function buildChangesApplied(proposal, affectedNodeIds) { + return { + addedNodeCount: proposal.addedNodes.length, + updatedNodeCount: proposal.updatedNodes.length, + addedEdgeCount: proposal.addedEdges.length, + removedEdgeCount: proposal.removedEdgeIds.length, + resolvedUnknownCount: proposal.resolvedUnknownNodeIds.length, + affectedNodeCount: affectedNodeIds.length, + }; +} + +export function applyValidatedProposal({ situationGraph, proposal }) { + const graphValidation = situationGraphSchema.safeParse(situationGraph); + const proposalValidation = graphUpdateSchema.safeParse(proposal); + + const existingGraphReferenceValidation = graphValidation.success + ? validateGraphReferences(situationGraph) + : null; + + const existingDuplicateNodeIds = graphValidation.success + ? detectDuplicateNodeIds(situationGraph.nodes) + : []; + const existingDuplicateEdgeIds = graphValidation.success + ? collectDuplicateEdgeIds(situationGraph.edges) + : []; + + if ( + !graphValidation.success || + !existingGraphReferenceValidation?.valid || + existingDuplicateNodeIds.length > 0 || + existingDuplicateEdgeIds.length > 0 + ) { + return { + success: false, + stage: "graph_validation", + errors: [ + ...(!graphValidation.success + ? zodIssuesToErrors(graphValidation.error) + : []), + ...(!existingGraphReferenceValidation?.valid + ? existingGraphReferenceValidation.errors + : []), + ...existingDuplicateNodeIds.map( + ({ nodeId, count }) => + `Graph contains duplicate node ID: "${nodeId}" (${count} occurrences)`, + ), + ...existingDuplicateEdgeIds.map( + ({ edgeId, count }) => + `Graph contains duplicate edge ID: "${edgeId}" (${count} occurrences)`, + ), + ], + }; + } + + if (!proposalValidation.success) { + return { + success: false, + stage: "proposal_compatibility", + errors: zodIssuesToErrors(proposalValidation.error), + }; + } + + const validatedProposal = proposalValidation.data; + const proposalCompatibilityErrors = []; + const proposalGraphValidation = validateGraphUpdate( + situationGraph, + validatedProposal, + ); + + if (!proposalGraphValidation.valid) { + proposalCompatibilityErrors.push(...proposalGraphValidation.errors); + } + + const existingEdgeIds = new Set(situationGraph.edges.map((edge) => edge.id)); + const reachableNodeIds = new Set([ + ...situationGraph.nodes.map((node) => node.id), + ...validatedProposal.addedNodes.map((node) => node.id), + ]); + const addedEdgeDuplicateIds = collectDuplicateEdgeIds( + validatedProposal.addedEdges, + ); + proposalCompatibilityErrors.push( + ...addedEdgeDuplicateIds.map( + ({ edgeId, count }) => + `Proposal contains duplicate added edge ID: "${edgeId}" (${count} occurrences)`, + ), + ); + + for (const edge of validatedProposal.addedEdges) { + if (existingEdgeIds.has(edge.id)) { + proposalCompatibilityErrors.push( + `Cannot add edge with duplicate ID: "${edge.id}"`, + ); + } + if (!reachableNodeIds.has(edge.fromNodeId)) { + proposalCompatibilityErrors.push( + `Added edge references non-existent fromNodeId: "${edge.fromNodeId}"`, + ); + } + if (!reachableNodeIds.has(edge.toNodeId)) { + proposalCompatibilityErrors.push( + `Added edge references non-existent toNodeId: "${edge.toNodeId}"`, + ); + } + } + + const removedEdgeIds = new Set(validatedProposal.removedEdgeIds); + for (const edgeId of removedEdgeIds) { + if (!existingEdgeIds.has(edgeId)) { + proposalCompatibilityErrors.push( + `Cannot remove non-existent edge: "${edgeId}"`, + ); + } + } + + const combinedNodeDuplicates = detectDuplicateNodeIds([ + ...situationGraph.nodes, + ...validatedProposal.addedNodes, + ]); + proposalCompatibilityErrors.push( + ...combinedNodeDuplicates.map( + ({ nodeId, count }) => + `Proposal would produce duplicate node ID: "${nodeId}" (${count} occurrences)`, + ), + ); + + if (proposalCompatibilityErrors.length > 0) { + return { + success: false, + stage: "proposal_compatibility", + errors: proposalCompatibilityErrors, + }; + } + + const graphSnapshot = cloneJsonSafe(situationGraph); + const proposalSnapshot = cloneJsonSafe(validatedProposal); + const previousActiveUnknownNodeId = graphSnapshot.activeUnknownNodeId ?? null; + const affectedNodeIds = buildAffectedNodeIds(graphSnapshot, proposalSnapshot); + + const applied = applyGraphUpdate(graphSnapshot, proposalSnapshot); + if (!applied.success) { + return { + success: false, + stage: "application", + errors: applied.errors, + }; + } + + const updatedSituationGraph = { + ...graphSnapshot, + nodes: applied.nodes, + edges: applied.edges, + resolvedNodeIds: applied.resolvedNodeIds, + }; + + const activeUnknownWasResolved = + previousActiveUnknownNodeId != null && + updatedSituationGraph.resolvedNodeIds.includes(previousActiveUnknownNodeId); + + let newActiveUnknownNodeId = previousActiveUnknownNodeId; + if (activeUnknownWasResolved) { + newActiveUnknownNodeId = null; + } + + const remainingUnknownExists = + newActiveUnknownNodeId != null && + updatedSituationGraph.nodes.some( + (node) => + node.id === newActiveUnknownNodeId && + node.kind === "unknown" && + !updatedSituationGraph.resolvedNodeIds.includes(node.id), + ); + + if (!remainingUnknownExists) { + newActiveUnknownNodeId = + selectActiveUnknownCandidate( + updatedSituationGraph, + updatedSituationGraph.resolvedNodeIds, + )?.nodeId ?? null; + } + + updatedSituationGraph.activeUnknownNodeId = newActiveUnknownNodeId; + updatedSituationGraph.currentSummary = describeGraph(updatedSituationGraph); + + const resultGraphValidation = situationGraphSchema.safeParse( + updatedSituationGraph, + ); + const resultReferenceValidation = resultGraphValidation.success + ? validateGraphReferences(updatedSituationGraph) + : null; + const resultDuplicateNodeIds = resultGraphValidation.success + ? detectDuplicateNodeIds(updatedSituationGraph.nodes) + : []; + const resultDuplicateEdgeIds = resultGraphValidation.success + ? collectDuplicateEdgeIds(updatedSituationGraph.edges) + : []; + + if ( + !resultGraphValidation.success || + !resultReferenceValidation?.valid || + resultDuplicateNodeIds.length > 0 || + resultDuplicateEdgeIds.length > 0 + ) { + return { + success: false, + stage: "result_validation", + errors: [ + ...(!resultGraphValidation.success + ? zodIssuesToErrors(resultGraphValidation.error) + : []), + ...(!resultReferenceValidation?.valid + ? resultReferenceValidation.errors + : []), + ...resultDuplicateNodeIds.map( + ({ nodeId, count }) => + `Updated graph contains duplicate node ID: "${nodeId}" (${count} occurrences)`, + ), + ...resultDuplicateEdgeIds.map( + ({ edgeId, count }) => + `Updated graph contains duplicate edge ID: "${edgeId}" (${count} occurrences)`, + ), + ], + }; + } + + return { + success: true, + updatedSituationGraph, + graphUpdate: validatedProposal, + affectedNodeIds, + resolvedUnknownNodeIds: validatedProposal.resolvedUnknownNodeIds, + previousActiveUnknownNodeId, + newActiveUnknownNodeId, + changesApplied: buildChangesApplied(validatedProposal, affectedNodeIds), + }; +} diff --git a/lib/graph/orchestrator.js b/lib/graph/orchestrator.js index 62038a8..967be9b 100644 --- a/lib/graph/orchestrator.js +++ b/lib/graph/orchestrator.js @@ -13,6 +13,7 @@ import { updateCaseRequestSchema, } from "./schema.js"; import { buildInitialGraph, describeGraph } from "./builder.js"; +import { applyValidatedProposal } from "./apply-proposal.js"; import { buildGraphUpdatePrompt } from "./prompt-builder.js"; import { parseGraphUpdateProposal } from "./update-proposal.js"; import { @@ -180,6 +181,9 @@ async function updateCaseWithDependencies(body, dependencies = {}) { dependencies.buildGraphUpdatePrompt ?? buildGraphUpdatePrompt; const parseProposal = dependencies.parseGraphUpdateProposal ?? parseGraphUpdateProposal; + const applyProposalUpdate = + dependencies.applyValidatedProposal ?? applyValidatedProposal; + const shouldApplyProposal = dependencies.applyProposal === true; let modelName = null; let rawResponse; @@ -238,6 +242,55 @@ async function updateCaseWithDependencies(body, dependencies = {}) { }; } + if (shouldApplyProposal) { + const applicationResult = applyProposalUpdate({ + situationGraph, + proposal: parsedProposal.proposal, + }); + + if (!applicationResult.success) { + return { + success: false, + stage: applicationResult.stage, + errors: applicationResult.errors, + diagnostics: { + promptVersion: promptVersion ?? null, + modelName, + responseDurationMs, + normalisationsApplied: parsedProposal.normalisationsApplied, + graphNodeCount: situationGraph.nodes.length, + graphEdgeCount: situationGraph.edges.length, + }, + statusCode: + applicationResult.stage === "application" || + applicationResult.stage === "result_validation" + ? 500 + : 400, + }; + } + + return { + success: true, + stage: "update_applied", + updatedSituationGraph: applicationResult.updatedSituationGraph, + proposal: applicationResult.graphUpdate, + affectedNodeIds: applicationResult.affectedNodeIds, + resolvedUnknownNodeIds: applicationResult.resolvedUnknownNodeIds, + previousActiveUnknownNodeId: + applicationResult.previousActiveUnknownNodeId, + newActiveUnknownNodeId: applicationResult.newActiveUnknownNodeId, + changesApplied: applicationResult.changesApplied, + diagnostics: { + promptVersion: promptVersion ?? null, + modelName, + responseDurationMs, + normalisationsApplied: parsedProposal.normalisationsApplied, + graphNodeCount: applicationResult.updatedSituationGraph.nodes.length, + graphEdgeCount: applicationResult.updatedSituationGraph.edges.length, + }, + }; + } + return { success: true, stage: "proposal_ready", diff --git a/tests/graph/apply-proposal.test.js b/tests/graph/apply-proposal.test.js new file mode 100644 index 0000000..781f3eb --- /dev/null +++ b/tests/graph/apply-proposal.test.js @@ -0,0 +1,404 @@ +import { describe, expect, it } from "vitest"; +import { applyValidatedProposal } from "@/lib/graph/apply-proposal.js"; +import { makeEdge, makeGraph, makeNode } from "@/lib/graph/schema.js"; +import { validateGraphReferences } from "@/lib/graph/utils.js"; + +function makeApplicationFixture() { + const complaintRateUnknown = makeNode({ + id: "n-complaint-rate-unknown", + label: "Complaint rate", + description: "Need the complaint rate per 100 units", + kind: "unknown", + status: "unknown", + confidence: "high", + affects: ["n-quality-deterioration"], + }); + const staffingUnknown = makeNode({ + id: "n-staffing-unknown", + label: "Staffing change", + description: "Need to know if staffing changed", + kind: "unknown", + status: "unknown", + confidence: "medium", + }); + const qualityDeterioration = makeNode({ + id: "n-quality-deterioration", + label: "Quality deterioration conclusion", + description: "Conclusion that quality deteriorated", + kind: "conclusion", + status: "supported", + confidence: "medium", + dependsOn: ["n-complaint-rate-unknown"], + }); + const complaintCount = makeNode({ + id: "n-complaint-count", + label: "Complaint count observation", + description: "Complaint count increased", + kind: "observation", + status: "supported", + confidence: "high", + value: 135, + unit: "count", + }); + const productionCount = makeNode({ + id: "n-production-count", + label: "Production count observation", + description: "Production increased", + kind: "observation", + status: "supported", + confidence: "high", + value: 7100, + unit: "units", + }); + + const graph = makeGraph({ + centralStatement: "Complaints rose while production also rose.", + nodes: [ + complaintRateUnknown, + staffingUnknown, + qualityDeterioration, + complaintCount, + productionCount, + ], + edges: [ + makeEdge({ + id: "e-quality-depends-rate", + fromNodeId: complaintRateUnknown.id, + toNodeId: qualityDeterioration.id, + relationship: "supports", + confidence: "medium", + description: "The rate informs the quality conclusion", + }), + ], + activeUnknownNodeId: complaintRateUnknown.id, + resolvedNodeIds: [], + currentSummary: "Initial summary", + }); + + const proposal = { + addedNodes: [], + updatedNodes: [ + { + nodeId: complaintRateUnknown.id, + previousStatus: "unknown", + newStatus: "resolved", + previousValue: "2.0 complaints per 100 units", + newValue: "1.9 complaints per 100 units", + reason: "The answer provides the updated normalized complaint rate.", + }, + { + nodeId: qualityDeterioration.id, + previousStatus: "supported", + newStatus: "weakened", + previousValue: null, + newValue: null, + reason: "The improved rate weakens the deterioration conclusion.", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [complaintRateUnknown.id], + affectedNodeIds: [qualityDeterioration.id], + }; + + return { + graph, + proposal, + ids: { + complaintRateUnknown: complaintRateUnknown.id, + staffingUnknown: staffingUnknown.id, + qualityDeterioration: qualityDeterioration.id, + complaintCount: complaintCount.id, + productionCount: productionCount.id, + }, + }; +} + +describe("applyValidatedProposal", () => { + it("applies a valid proposal successfully", () => { + const { graph, proposal, ids } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result).toMatchObject({ + success: true, + graphUpdate: proposal, + resolvedUnknownNodeIds: [ids.complaintRateUnknown], + previousActiveUnknownNodeId: ids.complaintRateUnknown, + newActiveUnknownNodeId: ids.staffingUnknown, + }); + expect( + result.updatedSituationGraph.nodes.find( + (node) => node.id === ids.complaintRateUnknown, + )?.status, + ).toBe("resolved"); + expect( + result.updatedSituationGraph.nodes.find( + (node) => node.id === ids.qualityDeterioration, + )?.status, + ).toBe("weakened"); + }); + + it("rejects an invalid graph before application", () => { + const { graph, proposal } = makeApplicationFixture(); + graph.nodes[0].dependsOn.push("missing-node"); + + const original = JSON.parse(JSON.stringify(graph)); + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(false); + expect(result.stage).toBe("graph_validation"); + expect(graph).toEqual(original); + }); + + it("rejects updates referencing nonexistent nodes", () => { + const { graph, proposal } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal: { + ...proposal, + updatedNodes: [ + ...proposal.updatedNodes, + { + nodeId: "ghost-node", + previousStatus: "unknown", + newStatus: "resolved", + previousValue: null, + newValue: null, + reason: "Invalid reference", + }, + ], + }, + }); + + expect(result).toMatchObject({ + success: false, + stage: "proposal_compatibility", + }); + expect(result.errors).toEqual( + expect.arrayContaining([ + expect.stringContaining( + 'Cannot update non-existent node: "ghost-node"', + ), + ]), + ); + }); + + it("rejects added edges with invalid references", () => { + const { graph, proposal } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal: { + ...proposal, + addedEdges: [ + makeEdge({ + id: "e-invalid", + fromNodeId: "missing-node", + toNodeId: "n-quality-deterioration", + relationship: "supports", + confidence: "medium", + description: "Invalid edge", + }), + ], + }, + }); + + expect(result.success).toBe(false); + expect(result.stage).toBe("proposal_compatibility"); + }); + + it("rejects duplicate IDs", () => { + const { graph, proposal, ids } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal: { + ...proposal, + addedNodes: [ + makeNode({ + id: ids.qualityDeterioration, + label: "Duplicate", + description: "Duplicate node id", + }), + ], + }, + }); + + expect(result.success).toBe(false); + expect(result.stage).toBe("proposal_compatibility"); + expect(result.errors.join(" ")).toContain("duplicate node ID"); + }); + + it("preserves unrelated nodes byte-for-byte", () => { + const { graph, proposal, ids } = makeApplicationFixture(); + const originalComplaintCount = JSON.stringify( + graph.nodes.find((node) => node.id === ids.complaintCount), + ); + const originalProductionCount = JSON.stringify( + graph.nodes.find((node) => node.id === ids.productionCount), + ); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(true); + expect( + JSON.stringify( + result.updatedSituationGraph.nodes.find( + (node) => node.id === ids.complaintCount, + ), + ), + ).toBe(originalComplaintCount); + expect( + JSON.stringify( + result.updatedSituationGraph.nodes.find( + (node) => node.id === ids.productionCount, + ), + ), + ).toBe(originalProductionCount); + }); + + it("adds resolved unknowns to resolvedNodeIds", () => { + const { graph, proposal, ids } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(true); + expect(result.updatedSituationGraph.resolvedNodeIds).toContain( + ids.complaintRateUnknown, + ); + }); + + it("keeps the active unknown when it remains unresolved", () => { + const { graph, ids } = makeApplicationFixture(); + const proposal = { + addedNodes: [], + updatedNodes: [ + { + nodeId: ids.qualityDeterioration, + previousStatus: "supported", + newStatus: "weakened", + previousValue: null, + newValue: null, + reason: "Only the conclusion changes", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [ids.qualityDeterioration], + }; + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(true); + expect(result.previousActiveUnknownNodeId).toBe(ids.complaintRateUnknown); + expect(result.newActiveUnknownNodeId).toBe(ids.complaintRateUnknown); + }); + + it("reports affected node ids", () => { + const { graph, proposal, ids } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(true); + expect(result.affectedNodeIds).toEqual( + expect.arrayContaining([ + ids.complaintRateUnknown, + ids.qualityDeterioration, + ]), + ); + }); + + it("revalidates the completed graph references", () => { + const { graph, proposal } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal, + }); + + expect(result.success).toBe(true); + expect(validateGraphReferences(result.updatedSituationGraph)).toEqual({ + valid: true, + errors: [], + }); + }); + + it("is atomic on failure", () => { + const { graph, proposal } = makeApplicationFixture(); + const originalGraph = JSON.parse(JSON.stringify(graph)); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal: { + ...proposal, + addedEdges: [ + makeEdge({ + id: "e-bad", + fromNodeId: "missing-node", + toNodeId: "n-quality-deterioration", + relationship: "supports", + confidence: "medium", + description: "Invalid edge", + }), + ], + }, + }); + + expect(result.success).toBe(false); + expect(graph).toEqual(originalGraph); + }); + + it("rejects a proposal with no meaningful change", () => { + const { graph } = makeApplicationFixture(); + + const result = applyValidatedProposal({ + situationGraph: graph, + proposal: { + addedNodes: [], + updatedNodes: [ + { + nodeId: "n-quality-deterioration", + previousStatus: null, + newStatus: null, + previousValue: null, + newValue: null, + reason: "No change", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: [], + affectedNodeIds: [], + }, + }); + + expect(result).toMatchObject({ + success: false, + stage: "proposal_compatibility", + }); + expect(result.errors).toEqual( + expect.arrayContaining([expect.stringContaining("no meaningful change")]), + ); + }); +}); diff --git a/tests/graph/orchestrator.test.js b/tests/graph/orchestrator.test.js index e863fe8..18c48b2 100644 --- a/tests/graph/orchestrator.test.js +++ b/tests/graph/orchestrator.test.js @@ -1,4 +1,5 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; +import { validateGraphReferences } from "@/lib/graph/utils.js"; import { makeGraph, makeNode } from "@/lib/graph/schema.js"; const mockAnalyseScenario = vi.fn(); @@ -526,4 +527,123 @@ describe("lib/graph/orchestrator startCase", () => { expect(result.nextQuestion).toBeUndefined(); expect(result.proposal.nextQuestion).toBeUndefined(); }); + + it("defaults to proposal-only mode", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const applyValidatedProposal = vi.fn(); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue(makeProposal()), + }; + + const result = await updateCase(makeUpdateRequest(), { + provider, + config: MOCK_CONFIG, + applyValidatedProposal, + }); + + expect(result.success).toBe(true); + expect(result.stage).toBe("proposal_ready"); + expect(applyValidatedProposal).not.toHaveBeenCalled(); + }); + + it("applies the proposal only when explicitly enabled", async () => { + const { updateCase } = await import("@/lib/graph/orchestrator.js"); + const request = makeUpdateRequest({ + situationGraph: makeGraph({ + centralStatement: + "Complaint counts increased while production also increased.", + nodes: [ + makeNode({ + id: "n-rate", + label: "Complaint rate", + description: "Need complaint rate", + kind: "unknown", + status: "unknown", + confidence: "high", + affects: ["n-conclusion"], + }), + makeNode({ + id: "n-other-unknown", + label: "Other unknown", + description: "Another unresolved unknown", + kind: "unknown", + status: "unknown", + confidence: "medium", + }), + makeNode({ + id: "n-conclusion", + label: "Quality deterioration", + description: "Quality conclusion", + kind: "conclusion", + status: "supported", + confidence: "medium", + dependsOn: ["n-rate"], + }), + ], + edges: [], + activeUnknownNodeId: "n-rate", + resolvedNodeIds: [], + currentSummary: "Initial summary", + }), + }); + const provider = { + generateReconstruction: vi.fn().mockResolvedValue({ + addedNodes: [], + updatedNodes: [ + { + nodeId: "n-rate", + previousStatus: "unknown", + newStatus: "resolved", + previousValue: "2.0 complaints per 100 units", + newValue: "1.9 complaints per 100 units", + reason: "The answer provides the updated rate.", + }, + { + nodeId: "n-conclusion", + previousStatus: "supported", + newStatus: "weakened", + previousValue: null, + newValue: null, + reason: "The updated rate weakens the conclusion.", + }, + ], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n-rate"], + affectedNodeIds: ["n-conclusion"], + }), + }; + + const result = await updateCase(request, { + provider, + config: MOCK_CONFIG, + applyProposal: true, + }); + + expect(result).toMatchObject({ + success: true, + stage: "update_applied", + affectedNodeIds: expect.arrayContaining(["n-rate", "n-conclusion"]), + resolvedUnknownNodeIds: ["n-rate"], + previousActiveUnknownNodeId: "n-rate", + newActiveUnknownNodeId: "n-other-unknown", + }); + expect(validateGraphReferences(result.updatedSituationGraph)).toEqual({ + valid: true, + errors: [], + }); + }); + + it("startCase behaviour remains unchanged", async () => { + mockAnalyseScenario.mockResolvedValue(makeAnalysisResult()); + const { startCase } = await import("@/lib/graph/orchestrator.js"); + + const result = await startCase({ scenario: "Scenario text" }); + + expect(result.success).toBe(true); + expect(result.selectedQuestion).toEqual({ + id: "q-1", + question: "What denominator is being used for the complaint rate?", + }); + }); }); From a948910ba82ee4d61cd9057fd6059f88c4f19cc0 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 08:32:18 +0100 Subject: [PATCH 10/12] feat: add situation graph update API route --- app/api/cases/update/route.js | 68 +++++ docs/v0.4-route-status.md | 15 +- tests/app/api/cases-update-route.test.js | 301 +++++++++++++++++++++++ 3 files changed, 379 insertions(+), 5 deletions(-) create mode 100644 app/api/cases/update/route.js create mode 100644 tests/app/api/cases-update-route.test.js diff --git a/app/api/cases/update/route.js b/app/api/cases/update/route.js new file mode 100644 index 0000000..adcc5af --- /dev/null +++ b/app/api/cases/update/route.js @@ -0,0 +1,68 @@ +import { updateCase } from "@/lib/graph/orchestrator.js"; + +function mapFailureStatus(result) { + switch (result?.stage) { + case "request_validation": + case "graph_validation": + return 400; + case "provider": + return 502; + case "proposal_validation": + case "proposal_compatibility": + case "application": + return 422; + case "result_validation": + return 500; + default: + return 500; + } +} + +function buildFailureResponse(result) { + return { + success: false, + stage: result?.stage ?? "internal", + error: result?.error ?? "Update case failed", + validationErrors: result?.validationErrors, + graphValidationErrors: result?.graphValidationErrors, + proposalErrors: result?.proposalErrors, + providerErrors: result?.providerErrors, + errors: result?.errors, + diagnostics: result?.diagnostics, + }; +} + +export async function POST(request) { + try { + const body = await request.json(); + const result = await updateCase(body, { applyProposal: true }); + + if (result.success) { + return Response.json(result, { status: 200 }); + } + + return Response.json(buildFailureResponse(result), { + status: mapFailureStatus(result), + }); + } catch (error) { + if (error instanceof SyntaxError) { + return Response.json( + { + success: false, + stage: "request_validation", + error: "Invalid JSON request body", + }, + { status: 400 }, + ); + } + + return Response.json( + { + success: false, + stage: "internal", + error: "Internal server error", + }, + { status: 500 }, + ); + } +} diff --git a/docs/v0.4-route-status.md b/docs/v0.4-route-status.md index f423afa..eff9c34 100644 --- a/docs/v0.4-route-status.md +++ b/docs/v0.4-route-status.md @@ -4,17 +4,22 @@ - Current tracked start-case route for the v0.4 graph orchestration path. - Covered by `tests/app/api/cases-start-route.test.js`. +- `app/api/cases/update/route.js` + - Current tracked update-case route for the v0.4 graph orchestration path. + - Delegates to `updateCase(body, { applyProposal: true })`. + - Covered by `tests/app/api/cases-update-route.test.js`. + - `app/api/start-case/route.js` - Earlier experiment / duplicate start route. - No repository UI/test references were found. - Deleted from the working tree during UI connection cleanup. - `app/api/update-case/route.js` - - Untracked future `updateCase` work. - - Not referenced by the current UI. - - Leave untracked for the current milestone. + - Earlier experimental duplicate update route. + - Removed from the working tree during route consolidation. - Current UI status - `components/scenario-form.jsx` now calls `/api/cases/start` for the main experimental flow. - - `/api/analyse` remains available for compatibility. - - No active UI path currently calls `/api/update-case`. + - `/api/cases/update` is the active tracked update route. + - `/api/analyse` remains available for legacy one-shot analysis. + - No UI changes were required for this route milestone. diff --git a/tests/app/api/cases-update-route.test.js b/tests/app/api/cases-update-route.test.js new file mode 100644 index 0000000..bcca0ad --- /dev/null +++ b/tests/app/api/cases-update-route.test.js @@ -0,0 +1,301 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const mockUpdateCase = vi.fn(); + +vi.mock("@/lib/graph/orchestrator.js", () => ({ + updateCase: (...args) => mockUpdateCase(...args), +})); + +function makeSuccessResult() { + return { + success: true, + stage: "update_applied", + updatedSituationGraph: { + centralStatement: "Scenario", + nodes: [{ id: "n1" }], + edges: [], + activeUnknownNodeId: null, + resolvedNodeIds: ["n1"], + currentSummary: "Updated summary", + }, + proposal: { + addedNodes: [], + updatedNodes: [], + addedEdges: [], + removedEdgeIds: [], + resolvedUnknownNodeIds: ["n1"], + affectedNodeIds: ["n1"], + }, + affectedNodeIds: ["n1"], + resolvedUnknownNodeIds: ["n1"], + previousActiveUnknownNodeId: "n0", + newActiveUnknownNodeId: null, + changesApplied: { updatedNodeCount: 1 }, + diagnostics: { promptVersion: "v0.4" }, + }; +} + +describe("app/api/cases/update route", () => { + beforeEach(() => { + vi.resetModules(); + vi.clearAllMocks(); + }); + + it("valid update returns HTTP 200", async () => { + mockUpdateCase.mockResolvedValue(makeSuccessResult()); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(200); + }); + + it("route calls updateCase with applyProposal: true", async () => { + mockUpdateCase.mockResolvedValue(makeSuccessResult()); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const body = { situationGraph: {}, previousQuestion: "Q", answer: "A" }; + await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify(body), + headers: { "content-type": "application/json" }, + }), + ); + + expect(mockUpdateCase).toHaveBeenCalledWith(body, { applyProposal: true }); + }); + + it("invalid JSON returns 400", async () => { + const { POST } = await import("@/app/api/cases/update/route.js"); + const request = { + json: vi.fn().mockRejectedValue(new SyntaxError("Unexpected token")), + }; + + const response = await POST(request); + + expect(response.status).toBe(400); + await expect(response.json()).resolves.toMatchObject({ + success: false, + stage: "request_validation", + error: "Invalid JSON request body", + }); + }); + + it("request validation failure returns 400", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "request_validation", + error: "Invalid update-case request", + validationErrors: [{ message: "Required" }], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({}), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(400); + }); + + it("graph validation failure returns 400", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "graph_validation", + error: "Invalid situation graph", + graphValidationErrors: ["bad graph"], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(400); + }); + + it("provider failure returns 502", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "provider", + error: "Graph update proposal generation failed", + providerErrors: ["provider offline"], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(502); + }); + + it("proposal validation failure returns 422", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "proposal_validation", + error: "Invalid graph update proposal", + proposalErrors: [{ message: "bad proposal" }], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(422); + }); + + it("proposal compatibility failure returns 422", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "proposal_compatibility", + error: "Update case failed", + errors: ["incompatible proposal"], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(422); + }); + + it("application failure returns 422", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "application", + error: "Update case failed", + errors: ["could not apply"], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(422); + }); + + it("result validation failure returns 500", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "result_validation", + error: "Update case failed", + errors: ["invalid result"], + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(500); + }); + + it("unknown failure returns 500", async () => { + mockUpdateCase.mockRejectedValue(new Error("boom")); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + expect(response.status).toBe(500); + await expect(response.json()).resolves.toMatchObject({ + success: false, + stage: "internal", + error: "Internal server error", + }); + }); + + it("success response preserves updated graph fields", async () => { + const success = makeSuccessResult(); + mockUpdateCase.mockResolvedValue(success); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + await expect(response.json()).resolves.toMatchObject({ + updatedSituationGraph: success.updatedSituationGraph, + proposal: success.proposal, + affectedNodeIds: success.affectedNodeIds, + resolvedUnknownNodeIds: success.resolvedUnknownNodeIds, + previousActiveUnknownNodeId: success.previousActiveUnknownNodeId, + newActiveUnknownNodeId: success.newActiveUnknownNodeId, + changesApplied: success.changesApplied, + diagnostics: success.diagnostics, + }); + }); + + it("stack traces and raw provider output are not exposed", async () => { + mockUpdateCase.mockResolvedValue({ + success: false, + stage: "provider", + error: "Graph update proposal generation failed", + providerErrors: ["provider offline"], + rawResponse: "secret", + stack: "trace", + diagnostics: {}, + }); + + const { POST } = await import("@/app/api/cases/update/route.js"); + const response = await POST( + new Request("http://localhost/api/cases/update", { + method: "POST", + body: JSON.stringify({ answer: "A" }), + headers: { "content-type": "application/json" }, + }), + ); + + const payload = await response.json(); + + expect(payload).not.toHaveProperty("stack"); + expect(payload).not.toHaveProperty("rawResponse"); + }); +}); From a9bce79658359cad68cb54bf4506e1e4ae2b27e6 Mon Sep 17 00:00:00 2001 From: robbond Date: Sun, 2 Aug 2026 09:43:42 +0100 Subject: [PATCH 11/12] feat: add one-turn situation graph update UI --- components/diagnostics-view.jsx | 17 +- components/graph-update-view.jsx | 169 ++++++++++++++++++++ components/scenario-form.jsx | 184 ++++++++++++++++++++- components/situation-graph-view.jsx | 9 +- lib/graph/apply-proposal.js | 146 ++++++++++++++++- lib/graph/orchestrator.js | 57 +++++-- lib/graph/prompt-builder.js | 1 + tests/graph/apply-proposal.test.js | 54 +++++++ tests/graph/orchestrator.test.js | 13 +- tests/smoke.test.js | 55 +++++-- tests/ui/scenario-form.test.jsx | 238 ++++++++++++++++++++++++++++ 11 files changed, 901 insertions(+), 42 deletions(-) create mode 100644 components/graph-update-view.jsx diff --git a/components/diagnostics-view.jsx b/components/diagnostics-view.jsx index 07433ea..8049c26 100644 --- a/components/diagnostics-view.jsx +++ b/components/diagnostics-view.jsx @@ -55,11 +55,21 @@ export default function DiagnosticsView({ result }) { }, { label: "Node count", - value: diagnostics.nodeCount != null ? diagnostics.nodeCount : "?", + value: + diagnostics.nodeCount != null + ? diagnostics.nodeCount + : diagnostics.graphNodeCount != null + ? diagnostics.graphNodeCount + : "?", }, { label: "Edge count", - value: diagnostics.edgeCount != null ? diagnostics.edgeCount : "?", + value: + diagnostics.edgeCount != null + ? diagnostics.edgeCount + : diagnostics.graphEdgeCount != null + ? diagnostics.graphEdgeCount + : "?", }, { label: "Graph references", @@ -75,6 +85,9 @@ export default function DiagnosticsView({ result }) { const errors = [ ...(result.errors || []), ...(result.validationErrors || []), + ...(result.graphValidationErrors || []), + ...(result.proposalErrors || []), + ...(result.providerErrors || []), ...(result.analysisErrors || []), ]; diff --git a/components/graph-update-view.jsx b/components/graph-update-view.jsx new file mode 100644 index 0000000..78f9bcd --- /dev/null +++ b/components/graph-update-view.jsx @@ -0,0 +1,169 @@ +import React from "react"; + +function ListSection({ title, items, renderItem = (item) => item }) { + if (!items?.length) return null; + + return ( +
+

{title}

+
    + {items.map((item, index) => ( +
  • {renderItem(item)}
  • + ))} +
+
+ ); +} + +export default function GraphUpdateView({ updateResult }) { + if (!updateResult?.proposal) return null; + + const { + resolvedUnknownNodeIds, + affectedNodeIds, + previousActiveUnknownNodeId, + newActiveUnknownNodeId, + changesApplied, + proposal, + previousSituationGraph, + updatedSituationGraph, + } = updateResult; + + const previousNodesById = new Map( + (previousSituationGraph?.nodes || []).map((node) => [node.id, node]), + ); + const updatedNodesById = new Map( + (updatedSituationGraph?.nodes || []).map((node) => [node.id, node]), + ); + const proposalUpdatesByNodeId = new Map( + (proposal.updatedNodes || []).map((update) => [update.nodeId, update]), + ); + + function resolveNodePresentation(nodeId) { + const previousNode = previousNodesById.get(nodeId) || null; + const updatedNode = updatedNodesById.get(nodeId) || null; + const node = updatedNode || previousNode; + const update = proposalUpdatesByNodeId.get(nodeId) || null; + + if (!node) { + return ( +
+
Unknown node (ID: {nodeId})
+
+ ); + } + + return ( +
+
{node.label}
+
+ {node.kind} · {node.confidence} +
+ {(update?.previousStatus || update?.newStatus || node.status) && ( +
+ {update?.previousStatus ? `Previous status: ${update.previousStatus}` : null} + {update?.previousStatus && update?.newStatus ? " → " : null} + {update?.newStatus + ? `New status: ${update.newStatus}` + : !update?.previousStatus + ? `Status: ${node.status}` + : null} +
+ )} + {update?.reason &&
{update.reason}
} +
+ ); + } + + function resolveActiveUnknown(nodeId) { + if (!nodeId) return null; + + const node = updatedNodesById.get(nodeId) || previousNodesById.get(nodeId); + if (!node) { + return `Unknown node (ID: ${nodeId})`; + } + + return `${node.label} · ${node.status} · ${node.confidence}`; + } + + const changeItems = [ + changesApplied?.addedNodeCount + ? `${changesApplied.addedNodeCount} node(s) added` + : null, + changesApplied?.updatedNodeCount + ? `${changesApplied.updatedNodeCount} node(s) updated` + : null, + changesApplied?.addedEdgeCount + ? `${changesApplied.addedEdgeCount} edge(s) added` + : null, + changesApplied?.removedEdgeCount + ? `${changesApplied.removedEdgeCount} edge(s) removed` + : null, + changesApplied?.resolvedUnknownCount + ? `${changesApplied.resolvedUnknownCount} unknown(s) resolved` + : null, + ].filter(Boolean); + + return ( +
+
+

+ Graph update applied +

+
+ {previousActiveUnknownNodeId && ( +
+ Previous active unknown:{" "} + {resolveActiveUnknown(previousActiveUnknownNodeId)} +
+ )} + {newActiveUnknownNodeId && ( +
+ New active unknown:{" "} + {resolveActiveUnknown(newActiveUnknownNodeId)} +
+ )} + {!newActiveUnknownNodeId && previousActiveUnknownNodeId && ( +
+ Next question status: No next + question selected yet. +
+ )} +
+
+ + + + + +
+ + Proposal details + +
+          {JSON.stringify(proposal, null, 2)}
+        
+
+          {JSON.stringify(
+            {
+              previousActiveUnknownNodeId,
+              newActiveUnknownNodeId,
+              resolvedUnknownNodeIds,
+              affectedNodeIds,
+            },
+            null,
+            2,
+          )}
+        
+
+
+ ); +} \ No newline at end of file diff --git a/components/scenario-form.jsx b/components/scenario-form.jsx index e71bc28..58e6cc6 100644 --- a/components/scenario-form.jsx +++ b/components/scenario-form.jsx @@ -3,6 +3,7 @@ import React from "react"; import { useState, useRef } from "react"; import DiagnosticsView from "@/components/diagnostics-view"; +import GraphUpdateView from "@/components/graph-update-view"; import SituationGraphView from "@/components/situation-graph-view"; const MAX_LENGTH = 10000; @@ -15,6 +16,45 @@ export async function submitScenarioForStartCase(fetchImpl, scenario) { }); } +export async function submitAnswerForUpdateCase( + fetchImpl, + { situationGraph, previousQuestion, answer }, +) { + if (!answer?.trim()) { + return { + ok: false, + skipped: true, + data: { + success: false, + stage: "request_validation", + error: "Please enter an answer before updating.", + }, + }; + } + + const response = await fetchImpl("/api/cases/update", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ situationGraph, previousQuestion, answer }), + }); + + return { + ok: response.ok, + skipped: false, + data: await response.json(), + }; +} + +function normaliseStartResult(data) { + return { + ...data, + selectedQuestion: + typeof data?.selectedQuestion === "string" + ? data.selectedQuestion + : data?.selectedQuestion?.question ?? null, + }; +} + export function ScenarioResultPanels({ status, result }) { if (!result) return null; @@ -51,16 +91,58 @@ export function ScenarioResultPanels({ status, result }) { ); } +export function UpdateErrorPanel({ updateError }) { + if (!updateError) return null; + + const errors = [ + ...(updateError.errors || []), + ...(updateError.validationErrors || []), + ...(updateError.graphValidationErrors || []), + ...(updateError.proposalErrors || []), + ...(updateError.providerErrors || []), + ]; + + return ( +
+
+ Update error: {updateError.error} +
+ {errors.length > 0 && ( +
+ + Update details ({errors.length}) + +
    + {errors.map((item, index) => ( +
  • + {typeof item === "string" ? item : item?.message || JSON.stringify(item)} +
  • + ))} +
+
+ )} +
+ ); +} + export default function ScenarioForm() { const [scenario, setScenario] = useState(""); const [status, setStatus] = useState("idle"); // idle | loading | error | success const [result, setResult] = useState(null); + const [answer, setAnswer] = useState(""); + const [updateStatus, setUpdateStatus] = useState("idle"); // idle | loading | error | success + const [updateError, setUpdateError] = useState(null); + const [updateResult, setUpdateResult] = useState(null); const textareaRef = useRef(null); const handleSubmit = async (e) => { e.preventDefault(); setStatus("loading"); setResult(null); + setAnswer(""); + setUpdateStatus("idle"); + setUpdateError(null); + setUpdateResult(null); try { const res = await submitScenarioForStartCase(fetch, scenario); @@ -69,10 +151,10 @@ export default function ScenarioForm() { if (res.ok && data.success) { setStatus("success"); - setResult(data); + setResult(normaliseStartResult(data)); } else { setStatus("error"); - setResult(data); + setResult(normaliseStartResult(data)); } } catch (err) { setStatus("error"); @@ -80,6 +162,55 @@ export default function ScenarioForm() { } }; + const handleUpdate = async (e) => { + e.preventDefault(); + + const submission = await submitAnswerForUpdateCase(fetch, { + situationGraph: result?.situationGraph, + previousQuestion: result?.selectedQuestion, + answer, + }); + + if (submission.skipped) { + setUpdateStatus("error"); + setUpdateError(submission.data); + return; + } + + setUpdateStatus("loading"); + setUpdateError(null); + + try { + const outcome = submission.data; + + if (submission.ok && outcome.success) { + setUpdateStatus("success"); + setUpdateResult({ + ...outcome, + previousSituationGraph: result?.situationGraph ?? null, + }); + setResult((current) => ({ + ...current, + situationGraph: outcome.updatedSituationGraph, + selectedQuestion: null, + diagnostics: outcome.diagnostics, + })); + setAnswer(""); + } else { + setUpdateStatus("error"); + setUpdateError(outcome); + } + } catch (err) { + setUpdateStatus("error"); + setUpdateError({ error: err.message || "Network request failed" }); + } + }; + + const canRenderAnswerForm = + status === "success" && + Boolean(result?.situationGraph) && + Boolean(result?.selectedQuestion); + return (
@@ -105,9 +236,56 @@ export default function ScenarioForm() {
+ {canRenderAnswerForm && ( +
+
+

Selected Question

+

{result.selectedQuestion}

+
+
+ +