Core fix: For cases with expectedBehaviours, reasoningQuality.status is now set exclusively from behaviour evaluation results (required behaviour pass/fail). Legacy concept checks remain visible as diagnostic-only metrics and do not influence the authoritative result. Key changes: - Behaviour-based scoring determines reasoning status (passed/failed) instead of legacy concept literal matching - Schema failure correctly forces not_evaluated (no vacuous truth) - Saved live results re-evaluator preserves provenance metadata - Classification tolerance map works bidirectionally for interchangeable types - normalise() treats underscores as word characters, hyphens as spaces Tests: 74 passing across both evaluator test suites - tests/evaluator-behaviour-authoritative.test.mjs (47 tests, new) - tests/evaluator-semantic.test.mjs (27 tests)
79 lines
3.2 KiB
JavaScript
79 lines
3.2 KiB
JavaScript
import { promises as fs } from "node:fs";
|
|
import { fileURLToPath } from "node:url";
|
|
import { dirname, join } from "node:path";
|
|
|
|
const __filename = fileURLToPath(import.meta.url);
|
|
const __dirname = dirname(__filename);
|
|
const PROMPTS_DIR = join(__dirname, "../../prompts");
|
|
|
|
/** Available prompt versions */
|
|
export const PROMPT_VERSIONS = ["v0.1", "v0.2"];
|
|
|
|
/** Build a v0.1 (extraction-only) prompt inline for backward compatibility */
|
|
function buildV1Prompt(scenario) {
|
|
return `You are a neutral analyst performing an evidence-based reconstruction of the following scenario.
|
|
|
|
Rules:
|
|
1. Do NOT invent facts. Only include information present in the scenario or clearly implied.
|
|
2. Distinguish carefully between:
|
|
- Direct observations (you witnessed directly)
|
|
- Reported claims (statements made by another person/entity)
|
|
- Interpretations (your analysis of what something means)
|
|
- Unsupported assumptions (things you are guessing without evidence)
|
|
3. If information is unknown, place it under "openUncertainties" — never guess.
|
|
4. Be precise, concise, and grounded in the text.
|
|
|
|
Scenario:
|
|
${scenario}
|
|
|
|
Return valid JSON matching this structure exactly:
|
|
{
|
|
"observations": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"reportedClaims": [{"id": "...", "description": "...", "confidence": "low|medium|high", "attributedTo": "person/entity or null"}],
|
|
"assumptions": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"entities": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"transitions": [{"id": "...", "description": "...", "confidence": "low|medium|high", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}],
|
|
"expectedButMissing": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"presentButUnexpected": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"contradictions": [{"id": "...", "description": "...", "confidence": "low|medium|high"}],
|
|
"openUncertainties": [{"id": "...", "description": "...", "confidence": "low|medium|high"}]
|
|
}
|
|
|
|
Return ONLY the JSON object. No markdown, no explanation, no preamble.`;
|
|
}
|
|
|
|
/** Load a versioned prompt from disk and substitute {{SCENARIO}} */
|
|
async function buildV2Prompt(scenario) {
|
|
try {
|
|
const content = await fs.readFile(
|
|
join(PROMPTS_DIR, "reconstruct-v0.2.md"),
|
|
"utf-8",
|
|
);
|
|
return content.replace("{{SCENARIO}}", scenario);
|
|
} catch {
|
|
// Fall back to v0.1 prompt if v0.2 file is missing
|
|
return buildV1Prompt(scenario);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Build an analysis prompt for the given version.
|
|
* @param {"v0.1" | "v0.2"} [version="v0.2"]
|
|
* @returns {Promise<{prompt: string, version: string}>}
|
|
*/
|
|
export async function buildPrompt(scenario, version = "v0.2") {
|
|
let prompt;
|
|
switch (version) {
|
|
case "v0.1":
|
|
prompt = buildV1Prompt(scenario);
|
|
break;
|
|
default: // v0.2
|
|
prompt = await buildV2Prompt(scenario);
|
|
break;
|
|
}
|
|
|
|
const strongJsonHint =
|
|
"\n\nReturn ONLY a valid JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.";
|
|
return { prompt: prompt + strongJsonHint, version };
|
|
}
|