Two bugs were causing the model to return {"status":"ok"} / {"status":"ready"}
instead of structured reconstruction data, resulting in POST /api/analyse 500:
1. DOUBLE-WRAPPING BUG (lib/llm/provider.js):
generateReconstruction() called buildPrompt(scenario) on input that was
already a fully-built prompt string from analyseScenario(). This wrapped the
v0.1 prompt (~5000+ chars) in another template layer, producing incomprehensible
output that the model could not parse as structured JSON.
Fix: Pass scenario through directly (it is ALREADY a built prompt).
2. MISSING JSON SPEC (prompts/reconstruct-v0.2.md):
The v0.2 prompt template said 'matching the structure exactly' but never
defined what that structure was. The model invented its own field names
(input_classification, reasoning_mode, anchors) with snake_case instead of
camelCase, which failed Zod validation -> 500 errors.
Fix: Added explicit JSON schema section with exact key names, enum values,
and nested structure matching the Zod validation layer.
Additionally:
- Refactored route to use analyseScenario from lib/analysis (centralized)
- Added lib/analysis.js with shared analysis logic
- Updated components to display promptVersion and validation errors
- Added lib/reconstruction/prompt.js v0.1/v0.2 versioning
- Added lib/reconstruction/schema.js v0.2 Zod schemas
- Added debug tool scripts, evaluation results, and comparison findings
93 lines
5.5 KiB
JSON
93 lines
5.5 KiB
JSON
[
|
|
{
|
|
"id": "diag-01",
|
|
"input": "We've seen a spike in complaints from our warehouse team this month compared to last month.",
|
|
"expectedPrimaryTypes": ["unexplained_change"],
|
|
"expectedReasoningModes": ["establish_baseline", "identify_difference"],
|
|
"shouldIdentify": ["complaints", "warehouse", "baseline comparison"],
|
|
"shouldNotInfer": ["quality issue", "staff turnover", "training gap"],
|
|
"description": "Baseline comparison — change without context. Should NOT jump to conclusions about quality or staff issues."
|
|
},
|
|
{
|
|
"id": "diag-02",
|
|
"input": "Some customers reported that the new app crashes when uploading photos.",
|
|
"expectedPrimaryTypes": ["observed_problem"],
|
|
"expectedReasoningModes": ["identify_difference", "establish_baseline"],
|
|
"shouldIdentify": ["app crashes", "photo upload", "some customers"],
|
|
"shouldNotInfer": ["all users affected", "server-side bug", "Android only"],
|
|
"description": "Subset modifier — 'some customers' means not universal. Should distinguish from blanket claims."
|
|
},
|
|
{
|
|
"id": "diag-03",
|
|
"input": "Sales fell by 15% last month after we increased prices, but the CFO says revenue is still up 2%.",
|
|
"expectedPrimaryTypes": ["contradiction"],
|
|
"expectedReasoningModes": ["investigate_contradiction", "establish_baseline"],
|
|
"shouldIdentify": ["sales decline", "price increase", "revenue increase", "CFO report"],
|
|
"shouldNotInfer": ["price was set too high", "competitors gained market share", "revenue data is wrong"],
|
|
"description": "Apparent contradiction — sales down but revenue up after price change. Distinguishes volume vs value."
|
|
},
|
|
{
|
|
"id": "diag-04",
|
|
"input": "We need to launch a marketplace app in Southeast Asia to capture the gap our competitors are exploiting.",
|
|
"expectedPrimaryTypes": ["decision_request"],
|
|
"expectedReasoningModes": ["decision_support", "identify_missing_information"],
|
|
"shouldIdentify": ["marketplace app", "Southeast Asia", "competitor gap"],
|
|
"shouldNotInfer": ["this will definitely succeed", "we have the resources", "competitors are struggling"],
|
|
"description": "Decision request — forward-looking, needs missing info identification."
|
|
},
|
|
{
|
|
"id": "diag-05",
|
|
"input": "Our production line changed suppliers three months ago but still delivers the same defect rate as before.",
|
|
"expectedPrimaryTypes": ["unexplained_change"],
|
|
"expectedReasoningModes": ["establish_baseline", "identify_difference"],
|
|
"shouldIdentify": ["supplier change", "three months ago", "same defect rate"],
|
|
"shouldNotInfer": ["new supplier is worse", "old supplier was better", "quality process is broken"],
|
|
"description": "Unexpected continuity — changed context but no outcome change."
|
|
},
|
|
{
|
|
"id": "diag-06",
|
|
"input": "From 45% to 62%, the completion rate for our onboarding flow improved significantly.",
|
|
"expectedPrimaryTypes": ["unexplained_change"],
|
|
"expectedReasoningModes": ["establish_baseline", "validate_measurement"],
|
|
"shouldIdentify": ["completion rate", "45%", "62%", "onboarding"],
|
|
"shouldNotInfer": ["all improvements are due to the redesign", "the old flow was bad", "users prefer the new design"],
|
|
"description": "Quantified improvement — needs context about measurement period and baseline conditions."
|
|
},
|
|
{
|
|
"id": "diag-07",
|
|
"input": "A user claimed that our pricing model is too complex for small businesses.",
|
|
"expectedPrimaryTypes": ["reported_claim"],
|
|
"expectedReasoningModes": ["validate_claim", "identify_difference"],
|
|
"shouldIdentify": ["pricing complexity", "small business", "user claim"],
|
|
"shouldNotInfer": ["the pricing is actually complex", "other small businesses agree", "we should simplify pricing"],
|
|
"description": "Single reported claim — needs validation, not acceptance as fact."
|
|
},
|
|
{
|
|
"id": "diag-08",
|
|
"input": "I used the phrase 'philosophical difference' in a meeting and my colleague said it meant nothing. Is that fair?",
|
|
"expectedPrimaryTypes": ["ambiguous_statement"],
|
|
"expectedReasoningModes": ["clarify_meaning"],
|
|
"shouldIdentify": ["philosophical", "ambiguous", "meaning clarification"],
|
|
"shouldNotInfer": ["the phrase was wrong", "the colleague is hostile", "we should avoid philosophical language"],
|
|
"description": "Meta-test — self-referential ambiguous statement. Should trigger clarification mode."
|
|
},
|
|
{
|
|
"id": "diag-09",
|
|
"input": "After the deployment last week, our complaint volume tripled to 47 cases per day.",
|
|
"expectedPrimaryTypes": ["causal_claim"],
|
|
"expectedReasoningModes": ["investigate_contradiction", "establish_baseline"],
|
|
"shouldIdentify": ["deployment", "complaint volume increase", "tripled", "47 cases"],
|
|
"shouldNotInfer": ["the deployment caused the complaints", "the bug report was insufficient", "rollback is needed"],
|
|
"description": "Post-event spike — presents correlation as potential causation. Must resist jumping to causal conclusion."
|
|
},
|
|
{
|
|
"id": "diag-10",
|
|
"input": "Some complaints involve production issues, but others say the delivery team is slow.",
|
|
"expectedPrimaryTypes": ["observed_problem"],
|
|
"expectedReasoningModes": ["identify_difference", "decompose_aggregate"],
|
|
"shouldIdentify": ["production issues", "delivery speed", "complaint types"],
|
|
"shouldNotInfer": ["production is worse than delivery", "the delivery team needs training", "both teams are underperforming equally"],
|
|
"description": "Paired with diag-01 — distinguishes subset complaints from aggregate claims."
|
|
}
|
|
]
|