import { describe, it, expect } from "vitest"; import assessInvestigationState from "@/lib/assessment/investigation-state-assessor.js"; import selectBehaviour, { BEHAVIOUR_OPTIONS } from "@/lib/behaviour-selection/behaviour-selector.js"; /* ── Helper: build scenario fixture data inline ─────────── */ function mkN(id, label, opts = {}) { const kind = opts.kind || "unknown"; const status = opts.status || (kind === "unknown" ? "unknown" : "known"); const confidence = opts.confidence || (kind === "unknown" ? "low" : "high"); return { id, label, description: label, kind, status, confidence, evidenceIds: [], dependsOn: [], affects: [], childIds: [] }; } /* ── Scenario fixtures used in this experiment ─────────── */ function getEvaluationScenarios() { return { // SCENARIO A: Long / developing investigation (4-turn arc) "long-investigation": [ { // Turn 0 — early, single observation centralStatement: "Should we enter the European market with our SaaS analytics platform?", nodes: [ mkN("obs-1", "Current revenue is $2M ARR in the US market only", { kind: "observation", status: "known", confidence: "high" }), mkN("state-1", "Evaluating European market entry", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether there is genuine demand for our category in Europe") ], resolvedNodeIds: [], activeUnknownNodeId: "u-1", selectedQuestion: { nodeId: "u-1", question: "How large and mature is the analytics SaaS market in Europe?", reason: "market_validity" }, currentSummary: "We are US-based. The first question before any expansion is whether demand exists.", diagnosticReasoningPattern: "market_validity" }, { // Turn 3 — deepening, many resolved centralStatement: "Should we enter the European market with our SaaS analytics platform?", nodes: [ mkN("obs-1", "Current revenue is $2M ARR in the US market only", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "European analytics SaaS market valued at approximately €8B and growing 15% annually", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-3", "Our platform does not currently support EU data residency requirements", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-4", "Achieving compliance would require approximately 6 months and $500K engineering investment", { kind: "observation", status: "known", confidence: "medium" }), mkN("state-1", "Evaluating European market entry", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether there is genuine demand for our category in Europe", { status: "resolved", confidence: "medium" }), mkN("u-2", "Whether our product is suitable for European compliance requirements", { status: "resolved", confidence: "high" }), mkN("u-3", "Whether the cost of achieving compliance is justified by the market size", { status: "resolved", confidence: "medium" }), mkN("u-4", "Whether we have competitive differentiation against existing European players") ], resolvedNodeIds: ["u-1", "u-2", "u-3"], activeUnknownNodeId: "u-4", selectedQuestion: { nodeId: "u-4", question: "What differentiates our platform against established European competitors?", reason: "competitive_analysis" }, currentSummary: "Compliance is feasible. The remaining question is competitive edge.", diagnosticReasoningPattern: "competitive_analysis" }, { // Turn 4 — complete, terminal centralStatement: "Should we enter the European market with our SaaS analytics platform?", nodes: [ mkN("obs-1", "Current revenue is $2M ARR in the US market only", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "European analytics SaaS market valued at approximately €8B and growing 15% annually", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-3", "Our platform does not currently support EU data residency requirements", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-4", "Achieving compliance would require approximately 6 months and $500K engineering investment", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-5", "Our real-time collaboration feature has no direct European equivalent", { kind: "observation", status: "provisional", confidence: "medium" }), mkN("state-1", "Evaluating European market entry", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether there is genuine demand for our category in Europe", { status: "resolved", confidence: "medium" }), mkN("u-2", "Whether our product is suitable for European compliance requirements", { status: "resolved", confidence: "high" }), mkN("u-3", "Whether the cost of achieving compliance is justified by the market size", { status: "resolved", confidence: "medium" }), mkN("u-4", "Whether we have competitive differentiation against existing European players", { status: "resolved", confidence: "medium" }) ], resolvedNodeIds: ["u-1", "u-2", "u-3", "u-4"], activeUnknownNodeId: null, selectedQuestion: null, noQuestionReason: "All investigation areas resolved.", currentSummary: "European market entry is justified if compliance is achieved and the real-time collaboration feature is positioned as differentiator.", diagnosticReasoningPattern: null } ], // SCENARIO B: Contradictory / difficult investigation — competing evidence paths "contradictory-evidence": [ { // Turn 0 — early, broad contradictory signals centralStatement: "Two consultants give opposite recommendations on which supplier to choose for a $2M procurement.", nodes: [ mkN("obs-1", "Consultant A recommends Supplier X: lower cost, proven track record", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "Consultant B recommends Supplier Y: better integration capability, higher risk but long-term upside", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-3", "Supplier X has 15+ years in the sector; Supplier Y has 2 years and mixed client reviews", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-4", "Our current infrastructure is compatible with neither supplier out of the box", { kind: "observation", status: "known", confidence: "high" }), mkN("state-1", "Evaluating $2M procurement against conflicting expert advice", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether the conflict is genuine or reflects different evaluation criteria") ], resolvedNodeIds: [], activeUnknownNodeId: "u-1", selectedQuestion: { nodeId: "u-1", question: "Are the consultants evaluating the same criteria, or are they measuring different things?", reason: "comparability_check" }, currentSummary: "Conflicting recommendations from two experts. The first uncertainty is whether we are comparing the same dimensions.", diagnosticReasoningPattern: "comparability_check" }, { // Turn 1 — one resolved, progress slow centralStatement: "Two consultants give opposite recommendations on which supplier to choose for a $2M procurement.", nodes: [ mkN("obs-1", "Consultant A recommends Supplier X: lower cost, proven track record", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "Consultant B recommends Supplier Y: better integration capability, higher risk but long-term upside", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-3", "Supplier X has 15+ years in the sector; Supplier Y has 2 years and mixed client reviews", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-4", "Our current infrastructure is compatible with neither supplier out of the box", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-5", "The consultants used different evaluation weights: cost 60% vs integration 60%", { kind: "observation", status: "known", confidence: "medium" }), mkN("state-1", "Evaluating $2M procurement against conflicting expert advice", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether the conflict is genuine or reflects different evaluation criteria", { status: "resolved", confidence: "high" }), mkN("u-2", "Which supplier's strengths align with our strategic priorities") ], resolvedNodeIds: ["u-1"], activeUnknownNodeId: "u-2", selectedQuestion: { nodeId: "u-2", question: "Does cost or integration capability matter more to the organisation over a 3-year horizon?", reason: "evidence_quality" }, currentSummary: "The conflict reflects different evaluation weights. The next uncertainty is strategic alignment.", diagnosticReasoningPattern: "evidence_quality" }, { // Turn 2 — two resolved, single remaining unknown centralStatement: "Two consultants give opposite recommendations on which supplier to choose for a $2M procurement.", nodes: [ mkN("obs-1", "Consultant A recommends Supplier X: lower cost, proven track record", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "Consultant B recommends Supplier Y: better integration capability, higher risk but long-term upside", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-3", "Supplier X has 15+ years in the sector; Supplier Y has 2 years and mixed client reviews", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-4", "Our current infrastructure is compatible with neither supplier out of the box", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-5", "The consultants used different evaluation weights: cost 60% vs integration 60%", { kind: "observation", status: "known", confidence: "medium" }), mkN("obs-6", "Our strategic plan prioritises long-term capability over short-term cost savings", { kind: "observation", status: "known", confidence: "high" }), mkN("state-1", "Evaluating $2M procurement against conflicting expert advice", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether the conflict is genuine or reflects different evaluation criteria", { status: "resolved", confidence: "high" }), mkN("u-2", "Which supplier's strengths align with our strategic priorities", { status: "resolved", confidence: "medium" }), mkN("u-3", "Whether the integration risk of Supplier Y is manageable with internal resources") ], resolvedNodeIds: ["u-1", "u-2"], activeUnknownNodeId: "u-3", selectedQuestion: { nodeId: "u-3", question: "Do we have the internal capacity to manage Supplier Y's integration risk?", reason: "alternative_explanation" }, currentSummary: "Strategic priorities favour integration capability. The remaining uncertainty is operational feasibility.", diagnosticReasoningPattern: "alternative_explanation" } ], // SCENARIO C: Short, early investigation — only 2 observations, unresolved "short-early": [ { // Turn 0 — two observations, first unknown centralStatement: "A manufacturing company reports complaints increased by 35% while production increased by 40%.", nodes: [ mkN("obs-1", "Complaints increased by 35%", { kind: "observation", status: "known", confidence: "high" }), mkN("obs-2", "Production increased by 40%", { kind: "observation", status: "known", confidence: "high" }), mkN("state-1", "Current situation", { kind: "state", status: "provisional", confidence: "medium" }), mkN("u-1", "Whether the two figures cover the same period") ], resolvedNodeIds: [], activeUnknownNodeId: "u-1", selectedQuestion: { nodeId: "u-1", question: "Were the complaint and production figures measured over the same period?", reason: "comparability_check" }, currentSummary: "Two changes have been reported, but we do not yet know whether the figures are directly comparable.", diagnosticReasoningPattern: "comparability_check" } ] }; } /* ── Build assessment input from scenario turn ─────────── */ function buildAssessmentInput(turn) { return { situationGraph: { centralStatement: turn.centralStatement, currentSummary: turn.currentSummary, nodes: turn.nodes, edges: turn.edges || [], activeUnknownNodeId: turn.activeUnknownNodeId, resolvedNodeIds: turn.resolvedNodeIds }, selectedQuestion: turn.selectedQuestion, noQuestionReason: turn.noQuestionReason, diagnostics: { promptVersion: "v0.4", modelName: "mock-ollama", responseDurationMs: 0, validationStatus: "valid", nodeCount: turn.nodes.length, edgeCount: (turn.edges || []).length, reasoningPattern: turn.diagnosticReasoningPattern || null } }; } /* ── Run assessor → selector pipeline on a single turn ─── */ function evaluateTurn(scenarioName, scenarioTurns, turnIndex) { const turn = scenarioTurns[turnIndex]; const input = buildAssessmentInput(turn); const assessment = assessInvestigationState(input); const behaviourResult = selectBehaviour(assessment); return { scenario: scenarioName, turnNumber: turnIndex, centralStatement: turn.centralStatement.substring(0, 60) + (turn.centralStatement.length > 60 ? "…" : ""), assessmentPhase: assessment.phase.value, assessmentPhaseConfidence: assessment.phase.confidence, assessmentProgress: assessment.progress.value, assessmentHealth: assessment.conversationHealth.value, selectedBehaviour: behaviourResult.behaviour, behaviourConfidence: behaviourResult.confidence, behaviourReason: behaviourResult.reason }; } /* ── Classify selection quality ───────────────────────── */ function classifySelection(scenarioName, turnIndex, result) { const { assessmentPhase, assessmentProgress, assessmentHealth, selectedBehaviour, assessmentPhaseConfidence } = result; if (selectedBehaviour === "acknowledge") { // Acknowledge fires when health is healthy AND phase confidence is not low if (assessmentHealth === "healthy" && assessmentPhaseConfidence !== "low") { return { classification: "sensible", explanation: "Healthy conversation with established context — acknowledge fires first per priority ordering." }; } return { classification: "questionable", explanation: `Acknowledge fired despite health=${assessmentHealth}, phaseConf=${assessmentPhaseConfidence}.` }; } if (selectedBehaviour === "clarify") { if (assessmentHealth === "too_broad") { return { classification: "sensible", explanation: "Too broad health justifies clarification." }; } if (assessmentPhase === "orienting" && result.behaviourReason?.includes("insufficient observations")) { return { classification: "sensible", explanation: "Orienting phase with < 3 observations — clarify anchors the investigation." }; } return { classification: "questionable", explanation: `Clarify selected with phase=${assessmentPhase}, health=${assessmentHealth}. Verify which rule condition matched.` }; } if (selectedBehaviour === "summarise") { if (["synthesising", "concluding"].includes(assessmentPhase)) { return { classification: "sensible", explanation: `${assessmentPhase} phase justifies a summary pass.` }; } return { classification: "sensible", explanation: `Summarise in ${assessmentPhase} with accumulated understanding.` }; } if (selectedBehaviour === "pause") { if ((assessmentPhase === "focusing" && assessmentProgress === "stalled") || assessmentHealth === "user_overloaded") { return { classification: "sensible", explanation: `${assessmentPhase}+${assessmentProgress} or overloaded — pausing is correct.` }; } return { classification: "questionable", explanation: `Pause selected with phase=${assessmentPhase}, progress=${assessmentProgress}, health=${assessmentHealth}.` }; } if (selectedBehaviour === "continue") { // Continue is the default when no specific rule fires if (assessmentPhase === "cannot_determine" || assessmentPhase === "exploring" || assessmentPhase === "deepening") { return { classification: "sensible", explanation: `No specific rule matched for ${assessmentPhase} — continue to ask next question is appropriate.` }; } // Check if acknowledge conditions appear met but didn't fire (shouldn't happen — ack has highest priority) if (assessmentHealth === "healthy" && assessmentPhaseConfidence !== "low") { return { classification: "questionable", explanation: "Continue selected but acknowledge conditions appear met (healthy + confident phase). Contract mismatch?" }; } return { classification: "sensible", explanation: `Continue as default for ${assessmentPhase} state with no matching specific rule.` }; } return { classification: "cannot determine", explanation: `Unknown behaviour: ${selectedBehaviour}` }; } /* ── Test: assessor output can be passed directly into selector ─ */ describe("Experiment 39 — Behaviour Selection against real assessment outputs", () => { describe("Pipeline contract", () => { it("assessor output can be passed directly into the selector without transformation", () => { const input = buildAssessmentInput( getEvaluationScenarios()["short-early"][0] ); const assessment = assessInvestigationState(input); const result = selectBehaviour(assessment); expect(result.behaviour).toBeDefined(); expect(result.confidence).toBeDefined(); expect(result.reason).toBeDefined(); }); it("every assessed turn receives one valid behaviour", () => { const scenarios = getEvaluationScenarios(); for (const [name, turns] of Object.entries(scenarios)) { for (let i = 0; i < turns.length; i++) { const input = buildAssessmentInput(turns[i]); const assessment = assessInvestigationState(input); const result = selectBehaviour(assessment); expect(BEHAVIOUR_OPTIONS).toContain(result.behaviour, `${name} turn ${i}: behaviour "${result.behaviour}" not in options`); } } }); it("repeated inputs remain deterministic", () => { const scenarios = getEvaluationScenarios(); for (const [name, turns] of Object.entries(scenarios)) { for (let i = 0; i < turns.length; i++) { const input = buildAssessmentInput(turns[i]); const assessment = assessInvestigationState(input); const results = Array.from({ length: 5 }, () => selectBehaviour(assessment)); for (const r of results) { expect(r.behaviour).toBe(results[0].behaviour); expect(r.reason).toBe(results[0].reason); } } } }); it("inputs are not mutated by assessor or selector", () => { const scenarios = getEvaluationScenarios(); for (const [name, turns] of Object.entries(scenarios)) { for (let i = 0; i < turns.length; i++) { const input = buildAssessmentInput(turns[i]); const snapshot = JSON.stringify(input); assessInvestigationState(input); selectBehaviour(assessInvestigationState(JSON.parse(snapshot))); expect(JSON.stringify(input)).toBe(snapshot); } } }); }); describe("Three-scenario evaluation", () => { it("evaluates all turns in long-investigation scenario", () => { const scenarioName = "long-investigation"; const turns = getEvaluationScenarios()[scenarioName]; const results = []; for (let i = 0; i < turns.length; i++) { const r = evaluateTurn(scenarioName, turns, i); results.push(r); } // Record full results via structured assertions expect(results).toHaveLength(3); // Store for later inspection in summary tests global._exp39_longResults = results; return results; }); it("evaluates all turns in contradictory-evidence scenario", () => { const scenarioName = "contradictory-evidence"; const turns = getEvaluationScenarios()[scenarioName]; const results = []; for (let i = 0; i < turns.length; i++) { const r = evaluateTurn(scenarioName, turns, i); results.push(r); } expect(results).toHaveLength(3); global._exp39_contradictoryResults = results; return results; }); it("evaluates all turns in short-early scenario", () => { const scenarioName = "short-early"; const turns = getEvaluationScenarios()[scenarioName]; const results = []; for (let i = 0; i < turns.length; i++) { const r = evaluateTurn(scenarioName, turns, i); results.push(r); } expect(results).toHaveLength(1); global._exp39_shortResults = results; return results; }); }); describe("Behaviour distribution", () => { it("records distribution across all scenarios", () => { const dist = { acknowledge: 0, clarify: 0, summarise: 0, pause: 0, continue: 0 }; const allResults = [ ...(global._exp39_longResults || []), ...(global._exp39_contradictoryResults || []), ...(global._exp39_shortResults || []) ]; for (const r of allResults) { dist[r.selectedBehaviour]++; } expect(allResults.length).toBeGreaterThan(0); // Log distribution console.log("\n=== Experiment 39 — Behaviour Distribution ==="); console.log(`Total evaluations: ${allResults.length}`); for (const [beh, count] of Object.entries(dist)) { console.log(` ${beh}: ${count} (${((count/allResults.length)*100).toFixed(0)}%)`); } // Store for later summary tests global._exp39_distribution = dist; global._exp39_allResults = allResults; return dist; }); it("every behaviour that appears does so in a sensible context", () => { const results = global._exp39_allResults || []; const classifications = {}; for (const r of results) { const key = `${r.scenario}-t${r.turnNumber}`; classifications[key] = classifySelection(r.scenario, r.turnNumber, r); } let questionableCount = 0; let sensibleCount = 0; for (const [key, c] of Object.entries(classifications)) { if (c.classification === "questionable") { questionableCount++; console.log(`\n[QUESTIONABLE] ${key}: behaviour=${r => r.selectedBehaviour}, reason: ${c.explanation}`); } else { sensibleCount++; } } // Log classification summary console.log("\n=== Experiment 39 — Selection Classifications ==="); console.log(`Sensible: ${sensibleCount}, Questionable: ${questionableCount}`); global._exp39_classifications = classifications; // Allow some questionables — we record them, don't silently correct return classifications; }); }); describe("Scenario-level behaviour sequences", () => { it("long-investigation produces: continue → acknowledge/summarise → summarise", () => { const results = global._exp39_longResults || []; // Early state should default to Continue (no rules match) expect(results[0].selectedBehaviour).toBeDefined(); console.log(`\nLong turn 0: phase=${results[0].assessmentPhase}, progress=${results[0].assessmentProgress}, health=${results[0].assessmentHealth} → ${results[0].selectedBehaviour}`); // Deep state with resolved nodes may trigger acknowledge or summarise expect(results[1].selectedBehaviour).toBeDefined(); console.log(`Long turn 3: phase=${results[1].assessmentPhase}, progress=${results[1].assessmentProgress}, health=${results[1].assessmentHealth} → ${results[1].selectedBehaviour}`); // Terminal state should trigger summarise (concluding) expect(results[2].selectedBehaviour).toBeDefined(); console.log(`Long turn 4: phase=${results[2].assessmentPhase}, progress=${results[2].assessmentProgress}, health=${results[2].assessmentHealth} → ${results[2].selectedBehaviour}`); }); it("contradictory-evidence produces meaningful variation", () => { const results = global._exp39_contradictoryResults || []; for (let i = 0; i < results.length; i++) { expect(results[i].selectedBehaviour).toBeDefined(); console.log(`Contradictory turn ${i}: phase=${results[i].assessmentPhase}, progress=${results[i].assessmentProgress}, health=${results[i].assessmentHealth} → ${results[i].selectedBehaviour}`); } }); it("short-early produces appropriate single selection", () => { const results = global._exp39_shortResults || []; expect(results.length).toBe(1); console.log(`Short turn 0: phase=${results[0].assessmentPhase}, progress=${results[0].assessmentProgress}, health=${results[0].assessmentHealth} → ${results[0].selectedBehaviour}`); }); }); describe("Questionable selections preserved", () => { it("does not silently correct questionable selections", () => { const classifications = global._exp39_classifications || {}; // The test passes if no assertions throw — questionable results are recorded, not fixed expect(Object.keys(classifications).length).toBeGreaterThan(0); }); }); }); /* ── Regression: existing assessor and selector tests pass ─ */ describe("Regression check", () => { it("assessor returns sensible defaults for null input", () => { const r = assessInvestigationState(null); expect(r.phase.value).toBe("cannot_determine"); expect(r.progress.value).toBe("cannot_determine"); expect(r.conversationHealth.value).toBe("cannot_determine"); }); it("selector returns continue for null input", () => { const r = selectBehaviour(null); expect(r.behaviour).toBe("continue"); }); it("selector only outputs valid behaviours", () => { const scenarios = getEvaluationScenarios(); for (const [name, turns] of Object.entries(scenarios)) { for (const turn of turns) { const input = buildAssessmentInput(turn); const assessment = assessInvestigationState(input); const result = selectBehaviour(assessment); expect(BEHAVIOUR_OPTIONS).toContain(result.behaviour); } } }); });