import { describe, it, expect } from "vitest"; import { config } from "dotenv"; import path from "path"; import { fileURLToPath } from "url"; const __filename = fileURLToPath(import.meta.url); const __dirname = path.dirname(__filename); config({ path: path.resolve(__dirname, "../../.env.local") }); const OLLAMA_BASE_URL = process.env.OLLAMA_BASE_URL; const OLLAMA_MODEL = process.env.OLLAMA_MODEL; if (!OLLAMA_BASE_URL || !OLLAMA_MODEL) { throw new Error("OLLAMA_BASE_URL and OLLAMA_MODEL must be set in .env.local"); } /** * Make one live Ollama chat call: identify the specific user-owned * distinction that remains unresolved when clarification is required. * Uses the exact Experiment 54S instruction (no examples). */ async function callClarificationTarget(source, disagreement, requiresUserClarification) { const instruction = `Identify the specific unresolved distinction that only the user can clarify. If clarification is required (requiresUserClarification: true), return the smallest statement of the missing user-owned meaning, preference, priority, constraint, definition, or private fact. If clarification is not required (requiresUserClarification: false), return null. Do not write a question. Do not add evidence needs. Do not select a preferred interpretation. Return valid JSON only in this shape: { "clarificationTarget": "short statement" | null }`; const messages = [ { role: "system", content: instruction.trim() }, { role: "user", content: `Source: ${JSON.stringify(source)} Disagreement: ${disagreement.map((d, i) => `${i + 1}. ${d}`).join("\n")} requiresUserClarification: ${requiresUserClarification}`, }, ]; const res = await fetch(`${OLLAMA_BASE_URL}/api/chat`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: OLLAMA_MODEL, messages, format: "json", stream: false, }), }); if (!res.ok) { throw new Error(`Ollama API error: ${res.status} ${res.statusText}`); } const data = await res.json(); const rawContent = data.message?.content ?? ""; const cleaned = rawContent.replace(/```(?:json)?\s*/g, "").replace(/```\s*/g, ""); return JSON.parse(cleaned.trim()); } // ────────────────────────────────────────────── // Three fixed cases — exactly as specified in the brief // ────────────────────────────────────────────── const CASES = [ { id: "Case 1", label: "Preference Versus Hard Constraint", source: "I want the business to grow, but I don't want to take on more risk.", disagreement: [ "growth should be prioritised even if some additional risk is unavoidable", "avoiding additional risk is a hard constraint even if growth is slower", ], requiresUserClarification: true, humanTarget: "whether avoiding additional risk is a preference/trade-off or a hard constraint", }, { id: "Case 2", label: "Definition Ambiguity", source: "I want to replace the system, but the new option needs to be affordable.", disagreement: [ "affordable means keeping upfront cost low", "affordable means keeping total long-term cost low", ], requiresUserClarification: true, humanTarget: 'whether "affordable" means low upfront cost or low overall/long-term cost', }, { id: "Case 3", label: "Private Factual Boundary", source: "I could move the project forward next month, depending on whether I actually have enough time.", disagreement: [ "the user has enough available time next month", "the user does not have enough available time next month", ], requiresUserClarification: true, humanTarget: "whether the user has enough available time next month to take on the project", }, ]; // ────────────────────────────────────────────── // Manual semantic classification helper // ────────────────────────────────────────────── function classifySpecificity(modelResult, caseRef) { const target = modelResult.clarificationTarget; if (target == null || typeof target !== "string" || !target.trim()) { return { classification: "target_wrong", reason: `Returned null or non-string` }; } const t = target.trim(); const lower = t.toLowerCase(); // Must not be a question if (t.endsWith("?")) { return { classification: "target_wrong", reason: `Target is worded as a question` }; } // Check for unsupported meaning const forbiddenConcepts = ["timeline", "deadline", "budget", "resources"]; for (const fc of forbiddenConcepts) { if (lower.includes(fc)) { return { classification: "target_wrong", reason: `Target introduces unsupported concept "${fc}"` }; } } // Check specificity preservation per case const c1Specific = lower.includes("preference") || lower.includes("trade.?off") || lower.includes("constraint") || lower.includes("hard"); const c2Specific = lower.includes("upfront") || lower.includes("long.?term") || lower.includes("overall"); const c3Specific = (lower.includes("time") && lower.includes("month")) || (lower.includes("capacity") && lower.includes("month")); let specific = false; if (caseRef.id === "Case 1" && c1Specific) specific = true; if (caseRef.id === "Case 2" && c2Specific) specific = true; if (caseRef.id === "Case 3" && c3Specific) specific = true; // Check for broadening patterns const c1Broad = /priority.*between/.test(lower) || /which.*matters.*more/.test(lower); const c2Broad = /^what.*affordab/i.test(lower); const c3Broad = /(can move forward|project.*forward|progress)/.test(lower); if (caseRef.id === "Case 1" && !specific && c1Broad) return { classification: "target_broadened", reason: `On right topic but broadened from preference/constraint to priority ordering` }; if (caseRef.id === "Case 2" && !specific && c2Broad) return { classification: "target_broadened", reason: `On right topic but broadened definition to general affordability meaning` }; if (caseRef.id === "Case 3" && !specific && c3Broad) return { classification: "target_broadened", reason: `On right topic but broadened from time availability to project forward movement` }; if (specific) { return { classification: "target_specific", reason: `Preserves the material distinction present in the human reference` }; } return { classification: "target_wrong", reason: `Target does not identify the user-owned ambiguity correctly` }; } // ────────────────────────────────────────────── // Test suite // ────────────────────────────────────────────── describe("Experiment 54X — Clarification Target Specificity", () => { const results = []; const timings = []; for (const c of CASES) { it(c.id + " (" + c.label + ")", async () => { const start = Date.now(); const result = await callClarificationTarget(c.source, c.disagreement, c.requiresUserClarification); const elapsed = Date.now() - start; timings.push({ caseId: c.id, ms: elapsed }); const classification = classifySpecificity(result, c); results.push({ case: c, modelResult: result, classification, timingMs: elapsed, }); // Structural contract: must have clarificationTarget field with a string value expect(result.clarificationTarget).toBeDefined(); expect(typeof result.clarificationTarget).toBe("string"); expect(result.clarificationTarget.trim()).not.toMatch(/\?$/); // Specificity assertion — each case expects target_specific expect(classification.classification).toBe("target_specific"); }, 120000); } it("54X — aggregate results", () => { const specific = results.filter(r => r.classification.classification === "target_specific").length; const broadened = results.filter(r => r.classification.classification === "target_broadened").length; const wrong = results.filter(r => r.classification.classification === "target_wrong").length; console.log("\n=== Experiment 54X Summary ==="); console.log(`Total live calls: ${results.length}`); const totalTime = timings.reduce((s, t) => s + t.ms, 0); console.log(`Total time: ${totalTime}ms`); console.log(`Average: ${(totalTime / timings.length).toFixed(1)}ms per call`); console.log(`Fastest: ${Math.min(...timings.map(t => t.ms))}ms`); console.log(`Slowest: ${Math.max(...timings.map(t => t.ms))}ms`); for (const r of results) { console.log(`\n--- ${r.case.id} (${r.case.label}) ---`); console.log("Actual target:", JSON.stringify(r.modelResult.clarificationTarget)); console.log("Human reference:", r.case.humanTarget); console.log("Classification:", r.classification.classification, "—", r.classification.reason); } console.log(`\nTarget-specific: ${specific}/${results.length}`); console.log(`Target-broadened: ${broadened}/${results.length}`); console.log(`Target-wrong: ${wrong}/${results.length}`); expect(specific + broadened + wrong).toBe(results.length); }); });