Compare commits
159
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
96b47d16a3 | ||
|
|
00d7593ffd | ||
|
|
60c90bdd6a | ||
|
|
707fe1b3c0 | ||
|
|
ed033e71d5 | ||
|
|
6974b710de | ||
|
|
b93aad9667 | ||
|
|
5f423d2f5f | ||
|
|
4d78c81773 | ||
|
|
efdb0a29d1 | ||
|
|
3bfc1c5c4b | ||
|
|
fdb6c86ee2 | ||
|
|
e18de817b4 | ||
|
|
607a2ad98b | ||
|
|
3e473dcf55 | ||
|
|
dafb9ae9b7 | ||
|
|
b964f2f172 | ||
|
|
2c535080d2 | ||
|
|
bb62a65174 | ||
|
|
0125975ccb | ||
|
|
f557175bfb | ||
|
|
896851e68d | ||
|
|
2079be690f | ||
|
|
e6c78f87fe | ||
|
|
d6df1d210e | ||
|
|
6dd447e56a | ||
|
|
b949eea831 | ||
|
|
30bf44f2e5 | ||
|
|
1c17452bee | ||
|
|
24d9e466f6 | ||
|
|
85b9f4411f | ||
|
|
0b5a38f83d | ||
|
|
7af708159e | ||
|
|
949a7024b3 | ||
|
|
ae00e70ced | ||
|
|
7548a6af59 | ||
|
|
3f2e2e05ae | ||
|
|
14630cf6b7 | ||
|
|
0cbe49913e | ||
|
|
0d27d4935b | ||
|
|
e289f0be1f | ||
|
|
0eadef6e3b | ||
|
|
bb3082d633 | ||
|
|
642a969b18 | ||
|
|
5878ce45ec | ||
|
|
be8b725a8d | ||
|
|
a93b6798cc | ||
|
|
188dd04ab9 | ||
|
|
a0f90e8885 | ||
|
|
d24ad48f62 | ||
|
|
e8d401e506 | ||
|
|
0bb2f01100 | ||
|
|
215c783d11 | ||
|
|
a45dd903cf | ||
|
|
1daf2bb6ce | ||
|
|
860ee6fc5b | ||
|
|
653934559c | ||
|
|
5c9f94ca13 | ||
|
|
726746f22d | ||
|
|
7070342fb1 | ||
|
|
cd1c6f4fc5 | ||
|
|
c7a0a79d0f | ||
|
|
8c5b47bcf5 | ||
|
|
46b9bd8b03 | ||
|
|
dabd9e2245 | ||
|
|
677f5e5757 | ||
|
|
8c3edfec7d | ||
|
|
72324e63c8 | ||
|
|
84fc53f017 | ||
|
|
e5de8564a4 | ||
|
|
55e935066c | ||
|
|
e1839147b1 | ||
|
|
fb49df87aa | ||
|
|
7472b6ecb0 | ||
|
|
13fbceee7a | ||
|
|
37245a8e28 | ||
|
|
a59d60262e | ||
|
|
f2a761d26f | ||
|
|
844ec0eb8c | ||
|
|
863f7dcbf2 | ||
|
|
65c5ded9ab | ||
|
|
6218a3ed6d | ||
|
|
898c3dcaaf | ||
|
|
42da768e66 | ||
|
|
95d9965420 | ||
|
|
580b2a122e | ||
|
|
642554038e | ||
|
|
928954ee4a | ||
|
|
41ea2cb6b9 | ||
|
|
e6f2249413 | ||
|
|
cc3a5dabd4 | ||
|
|
4bc998ee3f | ||
|
|
2af5971987 | ||
|
|
df142ca76c | ||
|
|
7ba1771bcf | ||
|
|
4b55ad1eae | ||
|
|
01e141aa66 | ||
|
|
827411f254 | ||
|
|
8c85120b1c | ||
|
|
f23d442eb3 | ||
|
|
06a200bb03 | ||
|
|
2df026d024 | ||
|
|
ede5d54e36 | ||
|
|
f06138de32 | ||
|
|
99b3d26817 | ||
|
|
37a9a12f93 | ||
|
|
fb2384cff1 | ||
|
|
2ed91468ed | ||
|
|
c33bcdbefa | ||
|
|
53cb99ee8f | ||
|
|
bb3da3d197 | ||
|
|
37b1892d86 | ||
|
|
1f3b26c983 | ||
|
|
e611283e3e | ||
|
|
83a66560ae | ||
|
|
577781eff8 | ||
|
|
7db28c8611 | ||
|
|
99b75dca4e | ||
|
|
0da7b63e30 | ||
|
|
32e1b01767 | ||
|
|
745026f0a0 | ||
|
|
83818c0c71 | ||
|
|
194a742772 | ||
|
|
f2c9e4c0b2 | ||
|
|
2b2096e41d | ||
|
|
a061428711 | ||
|
|
043ba5f264 | ||
|
|
b647236d44 | ||
|
|
161527f66c | ||
|
|
cd895a33ff | ||
|
|
a23da2b727 | ||
|
|
9da0928453 | ||
|
|
76c6096905 | ||
|
|
d22c992f60 | ||
|
|
7177c7bb61 | ||
|
|
8432ed45d4 | ||
|
|
650877e5bf | ||
|
|
c89cc51ae6 | ||
|
|
ab655e2222 | ||
|
|
0752c53a25 | ||
|
|
6b77e32771 | ||
|
|
efa39f52de | ||
|
|
18a7eb97cc | ||
|
|
0059c10f14 | ||
|
|
0624bc20e2 | ||
|
|
b270aa5624 | ||
|
|
989b88a4a1 | ||
|
|
addec52461 | ||
|
|
5fb32e628c | ||
|
|
8e941b0c7b | ||
|
|
75f7c6bafd | ||
|
|
ff1119b4d5 | ||
|
|
00ba343ed9 | ||
|
|
8c98ce94de | ||
|
|
bdb234262c | ||
|
|
07e1363368 | ||
|
|
16cab4645a | ||
|
|
922f58a49f | ||
|
|
bf7629691f |
@@ -123,3 +123,58 @@ Stop after reporting. Do not begin the next task automatically.
|
||||
When a task is interrupted by output limits, resume with a narrowly scoped repair prompt rather than restating the entire original brief.
|
||||
|
||||
User interfaces communicate reasoning, not implementation. If a piece of information exists only because the engine tracks it internally (graph nodes, unresolved counts, edge totals, confidence scores), it should remain in Developer Details unless it directly helps the user make their next decision.
|
||||
|
||||
## Playwright MCP — canonical dev server ownership
|
||||
|
||||
- Assume `http://localhost:3000` is already running when a task names it.
|
||||
- Never start / stop / kill / restart / replace / port-probe the dev server.
|
||||
- Never reinterpret "do not start/restart/kill/probe" as "start normally" or "use npm run dev".
|
||||
- If the canonical dev server is unavailable: **BLOCKED** — do not proceed.
|
||||
|
||||
## Playwright MCP — known controls and semantic locators
|
||||
|
||||
For known UI controls, use **Run Playwright code** with exact semantic locators:
|
||||
|
||||
```js
|
||||
await page.getByRole('button', { name: 'Review current understanding' }).click();
|
||||
```
|
||||
|
||||
Do NOT first try MCP Click. Do NOT use snapshot refs (`[ref=...]`) for actions — they are observational only.
|
||||
|
||||
Semantic scoping is allowed and encouraged where names repeat, e.g.:
|
||||
|
||||
```js
|
||||
page.getByRole('dialog').getByRole('button', { name: 'Restart investigation' });
|
||||
```
|
||||
|
||||
## Playwright MCP — semantic waits
|
||||
|
||||
For known async/hydration states, use `waitFor` with a semantic state — not arbitrary sleeps:
|
||||
|
||||
```js
|
||||
await page.getByRole(...).waitFor({ state: 'visible', timeout: ... });
|
||||
```
|
||||
|
||||
Client hydration is real product behaviour. Always await before classifying localStorage-backed UI state.
|
||||
|
||||
## Playwright MCP — selector failure
|
||||
|
||||
If the prescribed semantic locator cannot find its expected control: **STOP**.
|
||||
|
||||
Do NOT fall back to snapshot refs, CSS selectors, XPath, DOM traversal, `page.evaluate`, aria-label guessing, or locator archaeology.
|
||||
|
||||
## Playwright MCP — browser state and live freeze
|
||||
|
||||
During live verification do not inspect / inject / mutate browser storage merely to manufacture expected test state (unless storage manipulation itself is the explicit experiment).
|
||||
|
||||
Once live Playwright verification begins: **NO PRODUCTION FILE EDITS**. First visible discrepancy is evidence to capture and stop on.
|
||||
|
||||
## Deterministic test rules — apparatus ownership
|
||||
|
||||
**Tests are instruments, not product truth.**
|
||||
|
||||
At the first deterministic failure classify: **PRODUCT FAILURE** or **APPARATUS FAILURE**, then stop.
|
||||
|
||||
For APPARATUS FAILURE: do not turn the product task into test-harness development. Do not enter repeated vi.mock / dynamic re-import / module-cache manipulation / duplicate render / global mutation repair loops. Route apparatus correction separately.
|
||||
|
||||
If a lower-layer function is mocked, test the value crossing the mocked seam — do NOT require the mock to reproduce its real implementation. Storage-layer tests own storage writes.
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
node_modules
|
||||
.next
|
||||
out
|
||||
dist
|
||||
coverage
|
||||
*.lcov
|
||||
test-results
|
||||
*.log
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
evaluation-results
|
||||
provider-debug-results
|
||||
tests-results
|
||||
.playwright-mcp/
|
||||
.evidence-temp/
|
||||
|
||||
# Git
|
||||
.git
|
||||
.gitignore
|
||||
|
||||
# Environment files with secrets (never bake into image)
|
||||
.env
|
||||
.env.local
|
||||
.env.*.local
|
||||
|
||||
# Documentation / handoff (not needed for build)
|
||||
docs
|
||||
*.md
|
||||
|
||||
# IDE
|
||||
.vscode
|
||||
.idea
|
||||
|
||||
# OS generated files
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
+6
-9
@@ -1,15 +1,12 @@
|
||||
# Local Ollama server address
|
||||
OLLAMA_BASE_URL=http://192.168.x.x:11434
|
||||
# ── Supabase Auth (public browser configuration only) ────────────────
|
||||
NEXT_PUBLIC_SUPABASE_URL=https://supabase.rdbcloud.co.uk
|
||||
NEXT_PUBLIC_SUPABASE_ANON_KEY=replace-with-supabase-anon-key
|
||||
|
||||
# Model name (e.g., llama3, mistral, codellama, etc.)
|
||||
# ── Ollama provider (runtime, server-only) ────────────────────────────
|
||||
OLLAMA_BASE_URL=http://192.168.x.x:11434
|
||||
OLLAMA_MODEL=replace-with-model-name
|
||||
|
||||
# ── Mock / Demo Mode (UI development only) ──────────────────
|
||||
# Set to "true" to use pre-recorded scenario fixtures instead of Ollama.
|
||||
# ── Mock / Demo Mode (UI development only) ────────────────────────────
|
||||
NEXT_PUBLIC_CONFIDENCE_ENGINE_MOCKS=true
|
||||
|
||||
# Mock delay mode: "instant" | "normal" (default, 700ms) | "slow" (2500ms)
|
||||
NEXT_PUBLIC_CONFIDENCE_MOCK_DELAY=normal
|
||||
|
||||
# Scenario to replay: "complete" (jump to end after start) | "error" | "" (default sequential turns)
|
||||
NEXT_PUBLIC_CONFIDENCE_ENGINE_MOCK_SCENARIO=complete
|
||||
|
||||
@@ -46,3 +46,6 @@ tests-results/
|
||||
|
||||
# Evidence/temp directories from live experiments
|
||||
.evidence-temp/
|
||||
|
||||
# Jenkins env
|
||||
deploy.env
|
||||
+45
@@ -0,0 +1,45 @@
|
||||
# ── Stage 1: Build ─────────────────────────────────────────────────────
|
||||
FROM node:22-alpine AS builder
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
ARG NEXT_PUBLIC_SUPABASE_URL
|
||||
ARG NEXT_PUBLIC_SUPABASE_ANON_KEY
|
||||
|
||||
ENV NEXT_PUBLIC_SUPABASE_URL=${NEXT_PUBLIC_SUPABASE_URL} \
|
||||
NEXT_PUBLIC_SUPABASE_ANON_KEY=${NEXT_PUBLIC_SUPABASE_ANON_KEY}
|
||||
|
||||
COPY package.json package-lock.json* yarn.lock* pnpm-lock.yaml* ./
|
||||
|
||||
RUN corepack enable && \
|
||||
if [ -f pnpm-lock.yaml ]; then \
|
||||
corepack prepare pnpm@latest --activate; \
|
||||
pnpm install --frozen-lockfile; \
|
||||
elif [ -f yarn.lock ]; then \
|
||||
yarn install --frozen-lockfile; \
|
||||
else \
|
||||
npm ci; \
|
||||
fi
|
||||
|
||||
COPY . .
|
||||
|
||||
RUN npm run build
|
||||
|
||||
# ── Stage 2: Production runtime ────────────────────────────────────────
|
||||
FROM node:22-alpine AS runner
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
ENV NODE_ENV=production \
|
||||
NEXT_TELEMETRY_DISABLED=1 \
|
||||
PORT=3000 \
|
||||
HOSTNAME="0.0.0.0"
|
||||
|
||||
COPY --from=builder --chown=node:node /app/.next/standalone ./
|
||||
COPY --from=builder --chown=node:node /app/.next/static ./.next/static
|
||||
|
||||
USER node
|
||||
|
||||
EXPOSE 3000
|
||||
|
||||
CMD ["node", "server.js"]
|
||||
Vendored
+114
@@ -0,0 +1,114 @@
|
||||
// Confidence Engine — Manual Jenkins Deployment Pipeline
|
||||
//
|
||||
// Trigger: manually, via "Build with Parameters"
|
||||
// Parameter: GIT_REF (string) — the Git ref to deploy
|
||||
//
|
||||
// Pre-requisites in Jenkins:
|
||||
// 1. SSH credential of type "SSH Username with private key"
|
||||
// named 'confidence-engine-deploy-ssh' that can reach
|
||||
// CT 112 (confidence-engine / 192.168.68.73).
|
||||
// The username from the credential is used for SSH login.
|
||||
// 2. The Gitea repository configured in the job SCM section.
|
||||
|
||||
pipeline {
|
||||
agent any
|
||||
|
||||
parameters {
|
||||
string(
|
||||
name: 'GIT_REF',
|
||||
defaultValue: '',
|
||||
description: 'Git ref to deploy (branch name, tag, or full commit SHA). Leave blank to fail.'
|
||||
)
|
||||
}
|
||||
|
||||
environment {
|
||||
TARGET_HOST = '192.168.68.73'
|
||||
DEPLOY_DIR = '/opt/confidence-engine'
|
||||
HEALTH_URL = 'http://127.0.0.1:3000/api/health'
|
||||
}
|
||||
|
||||
stages {
|
||||
|
||||
stage('Resolve') {
|
||||
steps {
|
||||
script {
|
||||
def ref = params.GIT_REF.trim()
|
||||
if (!ref || ref.isEmpty()) {
|
||||
error 'GIT_REF parameter is blank or empty. Provide a Git ref to deploy.'
|
||||
}
|
||||
|
||||
echo "Requested ref: ${ref}"
|
||||
|
||||
// Resolve the ref to an exact SHA via Gitea remote.
|
||||
// If ref is already a 40-char hex SHA, use it directly.
|
||||
def shaPattern = ~/^[0-9a-fA-F]{40}$/
|
||||
def resolvedSha
|
||||
if (ref ==~ shaPattern) {
|
||||
resolvedSha = ref
|
||||
echo "Provided ref is a full commit SHA: ${resolvedSha}"
|
||||
} else {
|
||||
// For branches/tags, look up on origin
|
||||
resolvedSha = sh(
|
||||
script: "git ls-remote origin refs/heads/${ref} refs/tags/${ref} 2>/dev/null | awk '/^[0-9a-f]/{print \$1; exit}'",
|
||||
returnStdout: true
|
||||
).trim()
|
||||
|
||||
if (!resolvedSha || resolvedSha.length() != 40) {
|
||||
// Broader fallback — might match partial SHA or ref prefix
|
||||
resolvedSha = sh(
|
||||
script: "git ls-remote origin ${ref} 2>/dev/null | awk '/^[0-9a-f]/{print \$1; exit}'",
|
||||
returnStdout: true
|
||||
).trim()
|
||||
|
||||
if (!resolvedSha || resolvedSha.length() != 40) {
|
||||
error "Cannot resolve '${ref}' to a commit SHA on origin. Check the ref and repository configuration."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
echo "Resolved SHA: ${resolvedSha}"
|
||||
env.DEPLOY_SHA = resolvedSha
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
stage('Deploy') {
|
||||
steps {
|
||||
script {
|
||||
// Run the deployment script on CT 112 via SSH
|
||||
withCredentials([sshUserPrivateKey(
|
||||
credentialsId: 'confidence-engine-deploy-ssh',
|
||||
keyFileVariable: 'SSH_KEY',
|
||||
usernameVariable: 'SSH_USER'
|
||||
)]) {
|
||||
sh '''
|
||||
ssh \
|
||||
-i "$SSH_KEY" \
|
||||
-o StrictHostKeyChecking=yes \
|
||||
"$SSH_USER@$TARGET_HOST" \
|
||||
bash -s -- "$DEPLOY_SHA" "$DEPLOY_DIR" "$HEALTH_URL" \
|
||||
< "$WORKSPACE/scripts/deploy-production.sh"
|
||||
'''
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
stage('Verify/result') {
|
||||
steps {
|
||||
script {
|
||||
echo "Deployment stages completed. Check the Deploy stage output above for success/failure."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
post {
|
||||
failure {
|
||||
echo 'DEPLOYMENT FAILED — check the Deploy stage logs for details.'
|
||||
}
|
||||
success {
|
||||
echo "DEPLOYMENT SUCCEEDED — deployed SHA: ${env.DEPLOY_SHA}"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,8 +3,9 @@ import {
|
||||
PROMPT_VERSIONS,
|
||||
DEFAULT_PROMPT_VERSION,
|
||||
} from "@/lib/analysis";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
export async function POST(request) {
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
|
||||
@@ -47,3 +48,5 @@ export async function POST(request) {
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
/**
|
||||
* Investigation Overview synthesis API route.
|
||||
*
|
||||
* Route: POST /api/cases/overview
|
||||
*
|
||||
* Thin route pattern — no overview business logic here.
|
||||
*/
|
||||
|
||||
import { getProvider, getProviderModelName } from "@/lib/llm/provider.js";
|
||||
import { synthesizeInvestigationOverview } from "@/lib/graph/investigation-overview-synthesis.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
|
||||
if (!body || typeof body !== "object") {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Invalid request body" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
const { situationGraph, findings, plausibleInterpretations } = body;
|
||||
|
||||
if (!situationGraph) {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Missing situationGraph" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
const result = await synthesizeInvestigationOverview(
|
||||
{ situationGraph, findings, plausibleInterpretations },
|
||||
{
|
||||
provider: getProvider(),
|
||||
modelName: getProviderModelName(),
|
||||
}
|
||||
);
|
||||
|
||||
return Response.json(
|
||||
{ success: true, understanding: result.understanding, plausibleInterpretations: result.plausibleInterpretations },
|
||||
{ status: 200 }
|
||||
);
|
||||
} catch (error) {
|
||||
if (error instanceof SyntaxError) {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Invalid JSON request body" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
if (error.statusCode) {
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
stage: error.statusCode === 400 ? "request_validation" : "provider",
|
||||
error: error.message ?? "Overview synthesis failed",
|
||||
},
|
||||
{ status: error.statusCode }
|
||||
);
|
||||
}
|
||||
|
||||
return Response.json(
|
||||
{ success: false, stage: "internal", error: "Internal server error" },
|
||||
{ status: 500 }
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
@@ -1,6 +1,7 @@
|
||||
import { startCase } from "@/lib/graph/orchestrator.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
export async function POST(request) {
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
const result = await startCase(body);
|
||||
@@ -9,6 +10,17 @@ export async function POST(request) {
|
||||
return Response.json(result, { status: 200 });
|
||||
}
|
||||
|
||||
if (result.code === "PROVIDER_UNAVAILABLE") {
|
||||
console.error("[api/cases/start] provider unavailable", {
|
||||
error: result.error ?? "Start case failed",
|
||||
providerApiPath: result.providerApiPath,
|
||||
});
|
||||
return Response.json(
|
||||
{ success: false, error: "Reasoning service is temporarily unavailable." },
|
||||
{ status: 503 },
|
||||
);
|
||||
}
|
||||
|
||||
const status =
|
||||
result.statusCode === 400
|
||||
? 400
|
||||
@@ -16,17 +28,43 @@ export async function POST(request) {
|
||||
? result.statusCode
|
||||
: 500;
|
||||
|
||||
const diagnostics = {
|
||||
status,
|
||||
error: result.error ?? "Start case failed",
|
||||
validationErrors: result.validationErrors,
|
||||
analysisErrors: result.analysisErrors,
|
||||
validationIssues: result.validationIssues,
|
||||
providerApiPath: result.providerApiPath,
|
||||
providerExecution: result.providerExecution,
|
||||
rawResponse: result.rawResponse ?? undefined,
|
||||
};
|
||||
|
||||
if (status >= 500) {
|
||||
console.error("[api/cases/start] error response", diagnostics);
|
||||
} else {
|
||||
console.warn("[api/cases/start] error response", diagnostics);
|
||||
}
|
||||
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
error: result.error ?? "Start case failed",
|
||||
error: diagnostics.error,
|
||||
validationErrors: result.validationErrors,
|
||||
diagnostics: result.diagnostics,
|
||||
analysisErrors: result.analysisErrors,
|
||||
validationIssues: result.validationIssues,
|
||||
providerApiPath: result.providerApiPath,
|
||||
providerExecution: result.providerExecution,
|
||||
rawResponse: result.rawResponse ?? undefined,
|
||||
},
|
||||
{ status },
|
||||
);
|
||||
} catch {
|
||||
} catch (error) {
|
||||
console.error("[api/cases/start] unhandled exception", {
|
||||
message: error instanceof Error ? error.message : String(error),
|
||||
stack: error instanceof Error ? error.stack : undefined,
|
||||
error,
|
||||
});
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
@@ -36,3 +74,5 @@ export async function POST(request) {
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
/**
|
||||
* Dedicated Current Understanding synthesis API route.
|
||||
*
|
||||
* Route: POST /api/cases/synthesis
|
||||
*
|
||||
* Follows the thin route pattern established by cases/start and cases/update routes:
|
||||
* parse request → invoke domain seam → return validated result → map failure status
|
||||
*
|
||||
* No synthesis business logic belongs in this file.
|
||||
*/
|
||||
|
||||
import { getProvider, getProviderModelName } from "@/lib/llm/provider.js";
|
||||
import { synthesizeCurrentUnderstanding } from "@/lib/graph/current-understanding-synthesis.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
|
||||
// ── Parse / validate input contract ───────────────────────
|
||||
if (!body || typeof body !== "object") {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Invalid request body" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
const { situationGraph, findings } = body;
|
||||
|
||||
if (!situationGraph) {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Missing situationGraph" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
// ── Invoke domain seam with configured model ──────────────
|
||||
const result = await synthesizeCurrentUnderstanding(
|
||||
{ situationGraph, findings },
|
||||
{
|
||||
provider: getProvider(),
|
||||
modelName: getProviderModelName(),
|
||||
}
|
||||
);
|
||||
|
||||
return Response.json({ success: true, currentUnderstanding: result.currentUnderstanding }, { status: 200 });
|
||||
} catch (error) {
|
||||
if (error instanceof SyntaxError) {
|
||||
return Response.json(
|
||||
{ success: false, stage: "request_validation", error: "Invalid JSON request body" },
|
||||
{ status: 400 }
|
||||
);
|
||||
}
|
||||
|
||||
if (error.statusCode) {
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
stage: error.statusCode === 400 ? "request_validation" : "provider",
|
||||
error: error.message ?? "Synthesis failed",
|
||||
},
|
||||
{ status: error.statusCode }
|
||||
);
|
||||
}
|
||||
|
||||
// Unexpected error
|
||||
return Response.json(
|
||||
{ success: false, stage: "internal", error: "Internal server error" },
|
||||
{ status: 500 }
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
@@ -1,4 +1,8 @@
|
||||
import { updateCase } from "@/lib/graph/orchestrator.js";
|
||||
import { updateCase, reconsiderCompletedEpisode } from "@/lib/graph/orchestrator.js";
|
||||
import { applyValidatedProposal } from "@/lib/graph/apply-proposal.js";
|
||||
import { prepareCompletedEpisode } from "@/lib/graph/episode-preparation.js";
|
||||
import { updateCaseEpisodeRequestSchema } from "@/lib/graph/schema.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
function mapFailureStatus(result) {
|
||||
switch (result?.stage) {
|
||||
@@ -32,9 +36,28 @@ function buildFailureResponse(result) {
|
||||
};
|
||||
}
|
||||
|
||||
export async function POST(request) {
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
const isEpisodeMode = body?.episodeMode === true;
|
||||
|
||||
if (isEpisodeMode) {
|
||||
const parsed = updateCaseEpisodeRequestSchema.safeParse(body);
|
||||
if (!parsed.success) {
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
stage: "request_validation",
|
||||
error: "Invalid episode request",
|
||||
validationErrors: parsed.error.issues,
|
||||
},
|
||||
{ status: 400 },
|
||||
);
|
||||
}
|
||||
|
||||
return await handleEpisodeMode(body.situationGraph, body);
|
||||
}
|
||||
|
||||
const result = await updateCase(body, { applyProposal: true });
|
||||
|
||||
if (result.success) {
|
||||
@@ -66,3 +89,59 @@ export async function POST(request) {
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
|
||||
/** Server-side completed-episode reconsideration flow. */
|
||||
async function handleEpisodeMode(situationGraph, body) {
|
||||
const prepared = prepareCompletedEpisode({
|
||||
situationGraph,
|
||||
targetNodeId: body.targetNodeId,
|
||||
contributions: body.contributions ?? [],
|
||||
findings: body.findings,
|
||||
});
|
||||
|
||||
if (!prepared?.turns?.length && !prepared?.eligibleCanonicalFindings?.length) {
|
||||
return Response.json(
|
||||
{ success: false, stage: "preparation", error: "no_episodic_content" },
|
||||
{ status: 400 },
|
||||
);
|
||||
}
|
||||
|
||||
const reasoning = await reconsiderCompletedEpisode(prepared);
|
||||
if (!reasoning.success) {
|
||||
return Response.json(
|
||||
buildFailureResponse(reasoning),
|
||||
{ status: mapFailureStatus(reasoning) },
|
||||
);
|
||||
}
|
||||
|
||||
const application = await applyValidatedProposal({
|
||||
situationGraph,
|
||||
proposal: reasoning.proposal,
|
||||
evidenceContext: {
|
||||
isCompletedEpisode: true,
|
||||
episodeEvidence: prepared,
|
||||
},
|
||||
});
|
||||
|
||||
if (!application.success) {
|
||||
return Response.json(
|
||||
buildFailureResponse(application),
|
||||
{ status: mapFailureStatus(application) },
|
||||
);
|
||||
}
|
||||
|
||||
const resolvedNodeIds = new Set(application.updatedSituationGraph.resolvedNodeIds ?? []);
|
||||
resolvedNodeIds.add(body.targetNodeId);
|
||||
const updatedSituationGraph = {
|
||||
...application.updatedSituationGraph,
|
||||
resolvedNodeIds: [...resolvedNodeIds],
|
||||
};
|
||||
|
||||
return Response.json({
|
||||
success: true,
|
||||
updatedSituationGraph,
|
||||
proposal: reasoning.proposal,
|
||||
}, { status: 200 });
|
||||
}
|
||||
|
||||
@@ -1,9 +1,17 @@
|
||||
import { getProvider } from "@/lib/llm/provider";
|
||||
import { buildFocusedDeconstructPrompt, validateFocusedDeconstructSchema } from "@/lib/graph/focused-investigation";
|
||||
import { getProvider, getProviderModelName } from "@/lib/llm/provider";
|
||||
import {
|
||||
buildFocusedDeconstructPrompt,
|
||||
focusedDeconstructJsonSchema,
|
||||
validateFocusedDeconstructSchema,
|
||||
} from "@/lib/graph/focused-investigation";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
export async function POST(request) {
|
||||
async function post(request) {
|
||||
let targetNodeId = null;
|
||||
let startedAt = null;
|
||||
try {
|
||||
const body = await request.json();
|
||||
targetNodeId = body.targetNodeId ?? null;
|
||||
|
||||
if (!body.targetNodeId || typeof body.targetNodeId !== "string") {
|
||||
return Response.json(
|
||||
@@ -51,13 +59,49 @@ export async function POST(request) {
|
||||
});
|
||||
|
||||
const provider = getProvider();
|
||||
const startedAt = Date.now();
|
||||
const raw = await provider.generateReconstruction(prompt, process.env.OLLAMA_MODEL);
|
||||
const modelName = getProviderModelName();
|
||||
startedAt = Date.now();
|
||||
console.info("[api/focused-investigation/deconstruct] start", {
|
||||
targetNodeId,
|
||||
modelName,
|
||||
providerMode: process.env.CONFIDENCE_ENGINE_EXPERIMENT_PROVIDER ?? "ollama",
|
||||
startedAt,
|
||||
});
|
||||
const wrapper = await provider.generateReconstruction(
|
||||
prompt,
|
||||
modelName,
|
||||
focusedDeconstructJsonSchema,
|
||||
);
|
||||
const elapsedMs = Date.now() - startedAt;
|
||||
|
||||
// Unwrap the semantic deconstruction from the provider envelope.
|
||||
const deconstruction = wrapper.response;
|
||||
console.info("[api/focused-investigation/deconstruct] provider success", {
|
||||
targetNodeId,
|
||||
elapsedMs,
|
||||
providerApiPath: wrapper.providerApiPath ?? null,
|
||||
responsePresent: Boolean(deconstruction),
|
||||
responseKeys: deconstruction && typeof deconstruction === "object"
|
||||
? Object.keys(deconstruction)
|
||||
: [],
|
||||
});
|
||||
|
||||
// Validate schema (required fields present, no graph-mutation fields)
|
||||
const validationErrors = validateFocusedDeconstructSchema(raw);
|
||||
const validationErrors = validateFocusedDeconstructSchema(deconstruction);
|
||||
console.info("[api/focused-investigation/deconstruct] validation", {
|
||||
targetNodeId,
|
||||
schemaValid: validationErrors.length === 0,
|
||||
responseKeys: deconstruction && typeof deconstruction === "object"
|
||||
? Object.keys(deconstruction)
|
||||
: [],
|
||||
validationErrors,
|
||||
});
|
||||
if (validationErrors.length > 0) {
|
||||
console.info("[api/focused-investigation/deconstruct] end", {
|
||||
targetNodeId,
|
||||
status: 502,
|
||||
elapsedMs,
|
||||
});
|
||||
return Response.json(
|
||||
{
|
||||
success: false,
|
||||
@@ -70,20 +114,50 @@ export async function POST(request) {
|
||||
);
|
||||
}
|
||||
|
||||
return Response.json({
|
||||
const response = Response.json({
|
||||
success: true,
|
||||
targetNodeId: body.targetNodeId,
|
||||
observations: raw.observations,
|
||||
uncertainties: raw.uncertainties,
|
||||
assumptions: raw.assumptions,
|
||||
relationships: raw.relationships,
|
||||
possibleFollowUpQuestions: raw.possibleFollowUpQuestions,
|
||||
observations: deconstruction.observations,
|
||||
uncertainties: deconstruction.uncertainties,
|
||||
assumptions: deconstruction.assumptions,
|
||||
relationships: deconstruction.relationships,
|
||||
possibleFollowUpQuestions: deconstruction.possibleFollowUpQuestions,
|
||||
elapsedMs,
|
||||
});
|
||||
console.info("[api/focused-investigation/deconstruct] end", {
|
||||
targetNodeId,
|
||||
status: 200,
|
||||
elapsedMs,
|
||||
});
|
||||
return response;
|
||||
} catch (e) {
|
||||
const elapsedMs = startedAt == null ? null : Date.now() - startedAt;
|
||||
console.error("[api/focused-investigation/deconstruct] provider failure", {
|
||||
targetNodeId,
|
||||
elapsedMs,
|
||||
errorName: e?.name ?? "Error",
|
||||
errorMessage: e?.message ?? "Unknown server error",
|
||||
statusCode: e?.statusCode ?? e?.status ?? null,
|
||||
providerApiPath: e?.providerApiPath ?? null,
|
||||
errorCode: e?.code ?? null,
|
||||
errorParam: e?.param ?? null,
|
||||
});
|
||||
console.info("[api/focused-investigation/deconstruct] end", {
|
||||
targetNodeId,
|
||||
status: e?.code === "PROVIDER_UNAVAILABLE" ? 503 : 500,
|
||||
elapsedMs,
|
||||
});
|
||||
if (e?.code === "PROVIDER_UNAVAILABLE") {
|
||||
return Response.json(
|
||||
{ error: "Reasoning service is temporarily unavailable." },
|
||||
{ status: 503 },
|
||||
);
|
||||
}
|
||||
return Response.json(
|
||||
{ error: e.message || "Unknown server error" },
|
||||
{ status: 500 },
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { formulateQuestionForTarget } from "@/lib/graph/focused-investigation";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
export async function POST(request) {
|
||||
async function post(request) {
|
||||
try {
|
||||
const body = await request.json();
|
||||
|
||||
@@ -50,3 +51,5 @@ export async function POST(request) {
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
|
||||
+1
-47
@@ -1,49 +1,3 @@
|
||||
import { getConfig } from "@/lib/config";
|
||||
|
||||
export async function GET() {
|
||||
try {
|
||||
const result = getConfig();
|
||||
|
||||
if (!result.ok) {
|
||||
return Response.json({
|
||||
configPresent: false,
|
||||
baseUrl: null,
|
||||
model: null,
|
||||
reachable: false,
|
||||
error: "Missing or invalid environment configuration",
|
||||
}, { status: 500 });
|
||||
}
|
||||
|
||||
const { OLLAMA_BASE_URL, OLLAMA_MODEL } = result.config;
|
||||
|
||||
// Test reachability with a short timeout
|
||||
let reachable = false;
|
||||
let reachError = null;
|
||||
|
||||
try {
|
||||
const controller = new AbortController();
|
||||
const timeout = setTimeout(() => controller.abort(), 3000);
|
||||
|
||||
const res = await fetch(`${OLLAMA_BASE_URL}/api/tags`, {
|
||||
signal: controller.signal
|
||||
});
|
||||
clearTimeout(timeout);
|
||||
reachable = res.ok;
|
||||
} catch (e) {
|
||||
reachError = e.message || "Connection failed";
|
||||
}
|
||||
|
||||
return Response.json({
|
||||
configPresent: true,
|
||||
baseUrl: OLLAMA_BASE_URL,
|
||||
model: OLLAMA_MODEL,
|
||||
reachable,
|
||||
error: reachable ? null : (`Could not reach Ollama at ${OLLAMA_BASE_URL}: ${reachError || "timeout"}`),
|
||||
});
|
||||
} catch (e) {
|
||||
return Response.json(
|
||||
{ configPresent: false, error: e.message },
|
||||
{ status: 500 }
|
||||
);
|
||||
}
|
||||
return Response.json({ healthy: true }, { status: 200 });
|
||||
}
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
import { restartInvestigation } from "@/lib/storage/server-investigation-persistence.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
async function post(_request, { params }) {
|
||||
try {
|
||||
const snapshot = await restartInvestigation(params.id);
|
||||
if (!snapshot) return Response.json({ error: "Investigation not found" }, { status: 404 });
|
||||
return Response.json({ snapshot });
|
||||
} catch {
|
||||
return Response.json({ error: "Investigation persistence request failed" }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
@@ -0,0 +1,14 @@
|
||||
import { loadInvestigation } from "@/lib/storage/server-investigation-persistence.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
async function get(_request, { params }) {
|
||||
try {
|
||||
const snapshot = await loadInvestigation(params.id);
|
||||
if (!snapshot) return Response.json({ error: "Investigation not found" }, { status: 404 });
|
||||
return Response.json({ snapshot });
|
||||
} catch {
|
||||
return Response.json({ error: "Investigation persistence request failed" }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export const GET = withAuthenticatedApi(get);
|
||||
@@ -0,0 +1,29 @@
|
||||
import {
|
||||
listInvestigations,
|
||||
saveInvestigation,
|
||||
} from "@/lib/storage/server-investigation-persistence.js";
|
||||
import { withAuthenticatedApi } from "@/lib/supabase/api-auth.js";
|
||||
|
||||
async function get() {
|
||||
try {
|
||||
return Response.json({ investigations: await listInvestigations() });
|
||||
} catch {
|
||||
return Response.json({ error: "Investigation persistence request failed" }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
async function post(request) {
|
||||
try {
|
||||
const { id, snapshot } = await request.json();
|
||||
if (!id || !snapshot || typeof snapshot !== "object") {
|
||||
return Response.json({ error: "An investigation id and snapshot are required" }, { status: 400 });
|
||||
}
|
||||
const savedSnapshot = await saveInvestigation(snapshot, id);
|
||||
return Response.json({ snapshot: savedSnapshot }, { status: 200 });
|
||||
} catch {
|
||||
return Response.json({ error: "Investigation persistence request failed" }, { status: 500 });
|
||||
}
|
||||
}
|
||||
|
||||
export const GET = withAuthenticatedApi(get);
|
||||
export const POST = withAuthenticatedApi(post);
|
||||
@@ -0,0 +1,41 @@
|
||||
import { createServerClient } from "@supabase/ssr";
|
||||
import { NextResponse } from "next/server";
|
||||
|
||||
export async function GET(request) {
|
||||
const requestUrl = new URL(request.url);
|
||||
const code = requestUrl.searchParams.get("code");
|
||||
|
||||
// Derive redirect origin from proxy-forwarded headers when present,
|
||||
// falling back to the direct request origin for local/direct access.
|
||||
const forwardedHost = request.headers.get("x-forwarded-host");
|
||||
const forwardedProto = request.headers.get("x-forwarded-proto");
|
||||
let origin;
|
||||
|
||||
if (forwardedHost && forwardedProto) {
|
||||
// Nginx Proxy Manager (and similar proxies) set these headers.
|
||||
// x-forwarded-host may contain host:port or just hostname; use as-is.
|
||||
origin = `${forwardedProto}://${forwardedHost}`;
|
||||
} else {
|
||||
origin = requestUrl.origin;
|
||||
}
|
||||
|
||||
const response = NextResponse.redirect(new URL("/", origin));
|
||||
|
||||
if (code) {
|
||||
const supabase = createServerClient(
|
||||
process.env.NEXT_PUBLIC_SUPABASE_URL,
|
||||
process.env.NEXT_PUBLIC_SUPABASE_ANON_KEY,
|
||||
{
|
||||
cookies: {
|
||||
getAll: () => request.cookies.getAll(),
|
||||
setAll(cookiesToSet) {
|
||||
cookiesToSet.forEach(({ name, value, options }) => response.cookies.set(name, value, options));
|
||||
},
|
||||
},
|
||||
},
|
||||
);
|
||||
await supabase.auth.exchangeCodeForSession(code);
|
||||
}
|
||||
|
||||
return response;
|
||||
}
|
||||
+114
@@ -2,6 +2,72 @@
|
||||
@tailwind components;
|
||||
@tailwind utilities;
|
||||
|
||||
:root {
|
||||
--ce-page: #f9fafb;
|
||||
--ce-surface: #ffffff;
|
||||
--ce-surface-muted: #f9fafb;
|
||||
--ce-surface-elevated: #ffffff;
|
||||
--ce-text: #111827;
|
||||
--ce-text-muted: #6b7280;
|
||||
--ce-border: #d1d5db;
|
||||
--ce-teal: #0f766e;
|
||||
--ce-teal-surface: #f0fdfa;
|
||||
--ce-focus: #14b8a6;
|
||||
--ce-skeleton: #e5e7eb;
|
||||
}
|
||||
|
||||
html[data-theme="dark"] {
|
||||
--ce-page: #172128;
|
||||
--ce-surface: #202c34;
|
||||
--ce-surface-muted: #1b262e;
|
||||
--ce-surface-elevated: #293740;
|
||||
--ce-text: #edf2f3;
|
||||
--ce-text-muted: #b4c0c5;
|
||||
--ce-border: #40515a;
|
||||
--ce-teal: #62d3c5;
|
||||
--ce-teal-surface: #203a3c;
|
||||
--ce-focus: #78ded2;
|
||||
--ce-skeleton: #3a4a53;
|
||||
}
|
||||
|
||||
html[data-theme="dark"] body { background-color: var(--ce-page) !important; color: var(--ce-text) !important; }
|
||||
html[data-theme="dark"] .app-chrome { background-color: var(--ce-surface-muted); border-color: var(--ce-border); }
|
||||
html[data-theme="dark"] .theme-toggle { background-color: var(--ce-surface-elevated); border-color: var(--ce-border); color: var(--ce-text); }
|
||||
html[data-theme="dark"] .theme-toggle:hover { background-color: #33444d; }
|
||||
html[data-theme="dark"] .bg-white,
|
||||
html[data-theme="dark"] [class*="bg-white"] { background-color: var(--ce-surface) !important; }
|
||||
html[data-theme="dark"] [class*="bg-gray-50"],
|
||||
html[data-theme="dark"] [class*="bg-gray-100"] { background-color: var(--ce-surface-muted) !important; }
|
||||
html[data-theme="dark"] [class*="bg-gradient-to"] { background-image: none !important; background-color: var(--ce-surface) !important; }
|
||||
html[data-theme="dark"] [class*="bg-teal-50"] { background-color: var(--ce-teal-surface) !important; }
|
||||
html[data-theme="dark"] [class*="bg-amber-50"],
|
||||
html[data-theme="dark"] [class*="bg-orange-50"],
|
||||
html[data-theme="dark"] [class*="bg-yellow-50"] { background-color: #3a3324 !important; }
|
||||
html[data-theme="dark"] [class*="bg-green-50"] { background-color: #20392f !important; }
|
||||
html[data-theme="dark"] [class*="bg-blue-50"] { background-color: #243540 !important; }
|
||||
html[data-theme="dark"] [class*="border-gray"],
|
||||
html[data-theme="dark"] [class*="border-teal"],
|
||||
html[data-theme="dark"] [class*="border-amber"],
|
||||
html[data-theme="dark"] [class*="border-orange"],
|
||||
html[data-theme="dark"] [class*="border-green"],
|
||||
html[data-theme="dark"] [class*="border-blue"] { border-color: var(--ce-border) !important; }
|
||||
html[data-theme="dark"] :is(.text-gray-900, .text-gray-800, .text-gray-700, .text-gray-600) { color: var(--ce-text) !important; }
|
||||
html[data-theme="dark"] :is(.text-gray-500, .text-gray-400) { color: var(--ce-text-muted) !important; }
|
||||
html[data-theme="dark"] :is(.text-teal-700, .text-teal-800) { color: var(--ce-teal) !important; }
|
||||
html[data-theme="dark"] :is(.text-amber-700, .text-amber-800, .text-orange-700, .text-orange-800, .text-yellow-700, .text-yellow-800) { color: #f2c879 !important; }
|
||||
html[data-theme="dark"] :is(.text-green-700, .text-green-800) { color: #8bd8a8 !important; }
|
||||
html[data-theme="dark"] input,
|
||||
html[data-theme="dark"] textarea,
|
||||
html[data-theme="dark"] select { background-color: var(--ce-surface-elevated); color: var(--ce-text); border-color: var(--ce-border); }
|
||||
html[data-theme="dark"] input::placeholder,
|
||||
html[data-theme="dark"] textarea::placeholder { color: #94a3ab; }
|
||||
html[data-theme="dark"] details { background-color: var(--ce-surface-muted) !important; }
|
||||
html[data-theme="dark"] button:focus-visible,
|
||||
html[data-theme="dark"] a:focus-visible,
|
||||
html[data-theme="dark"] input:focus-visible,
|
||||
html[data-theme="dark"] textarea:focus-visible,
|
||||
html[data-theme="dark"] select:focus-visible { outline: 2px solid var(--ce-focus); outline-offset: 2px; }
|
||||
|
||||
@keyframes spin {
|
||||
from { transform: rotate(0deg); }
|
||||
to { transform: rotate(360deg); }
|
||||
@@ -24,6 +90,50 @@
|
||||
animation-delay: 0.16s;
|
||||
}
|
||||
|
||||
/* ── CU skeleton overlay during synthesis refresh ─────────────── */
|
||||
|
||||
.cu-skeleton-overlay {
|
||||
pointer-events: none;
|
||||
}
|
||||
|
||||
.cu-skeleton-lines {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
width: 100%;
|
||||
margin-top: auto;
|
||||
}
|
||||
|
||||
.cu-skeleton-line {
|
||||
height: 16px;
|
||||
border-radius: 8px;
|
||||
background-color: var(--ce-skeleton);
|
||||
position: relative;
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
/* Striped shimmer that travels left → right through each bar */
|
||||
.cu-skeleton-line::after {
|
||||
content: "";
|
||||
position: absolute;
|
||||
inset: 0;
|
||||
background: repeating-linear-gradient(
|
||||
105deg,
|
||||
transparent 0%,
|
||||
transparent 8px,
|
||||
rgba(255, 255, 255, 0.45) 8px,
|
||||
rgba(255, 255, 255, 0.45) 16px,
|
||||
transparent 16px,
|
||||
transparent 24px
|
||||
);
|
||||
animation: cuSkeletonShimmer 1.6s linear infinite;
|
||||
}
|
||||
|
||||
@keyframes cuSkeletonShimmer {
|
||||
0% { transform: translateX(-100%); }
|
||||
100% { transform: translateX(100%); }
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
[style*="animation:spin"] {
|
||||
animation: none !important;
|
||||
@@ -32,4 +142,8 @@
|
||||
.investigation-card {
|
||||
animation: none;
|
||||
}
|
||||
|
||||
.cu-skeleton-line::after {
|
||||
animation: none !important;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
"use client";
|
||||
|
||||
import React from "react";
|
||||
import { loadInvestigation } from "@/lib/storage/investigation-storage";
|
||||
import ScenarioForm from "@/components/scenario-form";
|
||||
import Link from "next/link";
|
||||
import { useParams, useRouter } from "next/navigation";
|
||||
import { useEffect, useState } from "react";
|
||||
|
||||
export default function InvestigationPage({ params }) {
|
||||
const router = useRouter();
|
||||
const routeId = typeof params?.id === "string" ? params.id : "";
|
||||
const [existing, setExisting] = useState(null);
|
||||
const [hydrated, setHydrated] = useState(false);
|
||||
|
||||
useEffect(() => {
|
||||
let active = true;
|
||||
setHydrated(false);
|
||||
if (!routeId) { setHydrated(true); return; }
|
||||
(async () => {
|
||||
try {
|
||||
const snapshot = await loadInvestigation(routeId);
|
||||
if (active) setExisting(snapshot);
|
||||
} finally {
|
||||
if (active) setHydrated(true);
|
||||
}
|
||||
})();
|
||||
return () => { active = false; };
|
||||
}, [routeId]);
|
||||
|
||||
return (
|
||||
<main className="mx-auto max-w-[1600px] px-6 py-12">
|
||||
{/* Page-level navigation — owned by route, not ReasoningWorkspace */}
|
||||
<nav className="mb-4 flex gap-3">
|
||||
<Link
|
||||
href="/"
|
||||
className="rounded-lg border border-teal-600 bg-white px-4 py-2 text-sm font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
Back to portfolio
|
||||
</Link>
|
||||
</nav>
|
||||
|
||||
<h1 className="mb-2 text-3xl font-bold tracking-tight">Confidence Engine</h1>
|
||||
<p className="mb-8 text-sm text-gray-500">
|
||||
Experimental prototype: enter a scenario and send it to a local LLM for
|
||||
evidence-based structured reconstruction. This is a technical vertical
|
||||
slice — not a production system.
|
||||
</p>
|
||||
{!hydrated ? (
|
||||
<p className="text-sm text-gray-500">Loading investigation…</p>
|
||||
) : existing ? (
|
||||
<ScenarioForm
|
||||
investigationId={routeId}
|
||||
existingSnapshot={existing}
|
||||
onNavigateToReport={() => router.push(`/investigations/${routeId}/report`)}
|
||||
/>
|
||||
) : (
|
||||
<ScenarioForm
|
||||
investigationId={routeId}
|
||||
onNavigateToReport={() => router.push(`/investigations/${routeId}/report`)}
|
||||
/>
|
||||
)}
|
||||
</main>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,248 @@
|
||||
"use client";
|
||||
|
||||
import React, { useEffect, useRef, useState } from "react";
|
||||
import { loadInvestigation, saveInvestigation } from "@/lib/storage/investigation-storage";
|
||||
import Link from "next/link";
|
||||
|
||||
export default function ReportPage({ params }) {
|
||||
const routeId = (typeof params === "object" && params?.id != null) ? String(params.id) : "";
|
||||
const [existing, setExisting] = useState(null);
|
||||
const [hydrated, setHydrated] = useState(false);
|
||||
const [generationLoading, setGenerationLoading] = useState(false);
|
||||
const [generationError, setGenerationError] = useState(false);
|
||||
const [updateLoading, setUpdateLoading] = useState(false);
|
||||
|
||||
const generationAttempted = useRef(false);
|
||||
|
||||
useEffect(() => {
|
||||
let active = true;
|
||||
setHydrated(false);
|
||||
(async () => {
|
||||
try {
|
||||
const snapshot = await loadInvestigation(routeId);
|
||||
if (active) setExisting(snapshot);
|
||||
} catch {
|
||||
if (active) setGenerationError(true);
|
||||
} finally {
|
||||
if (active) setHydrated(true);
|
||||
}
|
||||
})();
|
||||
return () => { active = false; };
|
||||
}, [routeId]);
|
||||
|
||||
// First-generation: create report when none persists (v0.58)
|
||||
useEffect(() => {
|
||||
if (!hydrated) return;
|
||||
if (existing?.investigationReport) return;
|
||||
if (generationAttempted.current) return;
|
||||
generationAttempted.current = true;
|
||||
|
||||
const situationGraph = existing?.situationGraph;
|
||||
const findings = existing?.findings ?? [];
|
||||
|
||||
if (!situationGraph) {
|
||||
setGenerationError(true);
|
||||
return;
|
||||
}
|
||||
|
||||
(async () => {
|
||||
setGenerationLoading(true);
|
||||
try {
|
||||
const res = await fetch("/api/cases/overview", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ situationGraph, findings }),
|
||||
});
|
||||
|
||||
if (!res.ok) {
|
||||
setGenerationError(true);
|
||||
return;
|
||||
}
|
||||
|
||||
const data = await res.json();
|
||||
if (data.success) {
|
||||
/* ── v0.59a — provenance: record generation revision (does NOT change Investigation revision) ── */
|
||||
const rev = existing?.investigationRevision ?? 0;
|
||||
const reportData = { understanding: data.understanding, plausibleInterpretations: data.plausibleInterpretations, hasPlausibleInterpretations: true, generatedFromRevision: rev };
|
||||
setExisting((p) => {
|
||||
void saveInvestigation({ ...p, investigationReport: reportData })
|
||||
.catch((error) => console.error("Investigation report save failed", error));
|
||||
return { ...p, investigationReport: reportData };
|
||||
});
|
||||
} else {
|
||||
setGenerationError(true);
|
||||
}
|
||||
} catch {
|
||||
setGenerationError(true);
|
||||
} finally {
|
||||
setGenerationLoading(false);
|
||||
}
|
||||
})();
|
||||
}, [hydrated, existing]);
|
||||
|
||||
// Manual Report update (v0.59b — freshness manual update)
|
||||
const handleUpdateReport = async () => {
|
||||
if (updateLoading) return;
|
||||
setUpdateLoading(true);
|
||||
|
||||
const snap = await loadInvestigation(routeId);
|
||||
const situationGraph = snap?.situationGraph;
|
||||
const findings = snap?.findings ?? [];
|
||||
const rev = snap?.investigationRevision ?? 0;
|
||||
|
||||
if (!situationGraph) {
|
||||
setUpdateLoading(false);
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const res = await fetch("/api/cases/overview", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ situationGraph, findings }),
|
||||
});
|
||||
|
||||
if (!res.ok) {
|
||||
setUpdateLoading(false);
|
||||
return;
|
||||
}
|
||||
|
||||
const data = await res.json();
|
||||
if (data.success) {
|
||||
const reportData = { understanding: data.understanding, plausibleInterpretations: data.plausibleInterpretations, hasPlausibleInterpretations: true, generatedFromRevision: rev };
|
||||
setExisting((p) => {
|
||||
void saveInvestigation({ ...p, investigationReport: reportData })
|
||||
.catch((error) => console.error("Investigation report save failed", error));
|
||||
return { ...p, investigationReport: reportData };
|
||||
});
|
||||
}
|
||||
} catch {
|
||||
/* failure: retain existing Report and updateAvailable state */
|
||||
} finally {
|
||||
setUpdateLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
const report = existing?.investigationReport || null;
|
||||
const scenario = hydrated ? (existing?.scenario || "") : null;
|
||||
|
||||
const paragraphs = (report?.understanding || "")
|
||||
.split("\n")
|
||||
.filter(Boolean);
|
||||
|
||||
return (
|
||||
<main className="mx-auto max-w-[800px] px-6 py-16">
|
||||
<h1 className="mb-2 text-[15px] font-bold tracking-[.2em] uppercase text-teal-700/90">
|
||||
Investigation Report
|
||||
</h1>
|
||||
|
||||
{/* Report freshness — only when a Report exists */}
|
||||
{report ? (
|
||||
<div className="mt-6 flex items-center gap-3">
|
||||
{existing?.investigationRevision === report.generatedFromRevision ? (
|
||||
<span className="text-[11px] font-semibold tracking-wider uppercase text-teal-700/70">Current</span>
|
||||
) : (
|
||||
<div className="flex items-center gap-3">
|
||||
<span className="text-[11px] font-semibold tracking-wider uppercase text-gray-500">Update available</span>
|
||||
<span className="text-xs text-gray-400">The investigation has changed since this report was generated.</span>
|
||||
<button
|
||||
type="button"
|
||||
onClick={handleUpdateReport}
|
||||
disabled={updateLoading}
|
||||
className="rounded-lg border border-teal-600 bg-white px-3 py-1.5 text-[11px] font-semibold tracking-wider uppercase text-teal-700 hover:bg-teal-50 transition disabled:opacity-40"
|
||||
>
|
||||
{updateLoading ? "Updating…" : "Update report"}
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{/* Situation */}
|
||||
{scenario && (
|
||||
<div className="mt-8 rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-6 pb-7 shadow-sm">
|
||||
<h2 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-teal-700/70">
|
||||
Situation
|
||||
</h2>
|
||||
<p className="text-base leading-relaxed text-gray-800 whitespace-pre-wrap">
|
||||
{scenario}
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* What we understand */}
|
||||
{report ? (
|
||||
<>
|
||||
{paragraphs.length > 0 ? (
|
||||
paragraphs.map((p, i) => (
|
||||
<div key={i} className="mt-6 rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-6 pb-7 shadow-sm">
|
||||
<h2 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-teal-700/70">
|
||||
What we understand
|
||||
</h2>
|
||||
<p className="text-base leading-relaxed text-gray-800">{p}</p>
|
||||
</div>
|
||||
))
|
||||
) : (
|
||||
<div className="mt-6 rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-6 pb-7 shadow-sm">
|
||||
<h2 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-teal-700/70">
|
||||
What we understand
|
||||
</h2>
|
||||
<p className="text-base leading-relaxed text-gray-800">{report.understanding || ""}</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* What remains plausible — conditional */}
|
||||
{report.hasPlausibleInterpretations && report.plausibleInterpretations ? (
|
||||
<div className="mt-6 rounded-xl border-[2.5px] border-blue-300/70 bg-gradient-to-b from-blue-50/60 to-white px-8 pt-6 pb-7 shadow-sm">
|
||||
<h2 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-blue-700/70">
|
||||
What remains plausible
|
||||
</h2>
|
||||
<p className="text-base leading-relaxed text-gray-800 italic">
|
||||
{report.plausibleInterpretations}
|
||||
</p>
|
||||
</div>
|
||||
) : null}
|
||||
</>
|
||||
) : (
|
||||
/* Skeleton / loading state when no persisted report exists */
|
||||
<>
|
||||
<div className="mt-8 rounded-xl border-[2.5px] border-gray-200 bg-gray-50/50 px-8 pt-6 pb-7 shadow-sm">
|
||||
<h2 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-gray-400">
|
||||
What we understand
|
||||
</h2>
|
||||
{generationLoading ? (
|
||||
<div className="flex flex-col gap-3 py-2" aria-live="polite">
|
||||
<span className="text-[11px] font-semibold tracking-wider text-gray-400 uppercase">Generating report…</span>
|
||||
{[0, 1, 2].map((i) => (
|
||||
<div key={i} className="h-4 w-full rounded animate-pulse" style={{ backgroundColor: "rgb(229 231 235)", animationDelay: `${i * 150}ms`, width: i === 1 ? "80%" : i === 2 ? "65%" : "90%" }} />
|
||||
))}
|
||||
</div>
|
||||
) : generationError ? (
|
||||
<p className="text-sm text-red-600">Report generation failed. You may try again from the Investigation page.</p>
|
||||
) : null}
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
||||
{/* Back to investigation */}
|
||||
<div className="mt-10">
|
||||
<Link
|
||||
href={`/investigations/${routeId}`}
|
||||
className="rounded-lg border border-teal-600 bg-white px-4 py-2 text-sm font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
Back to investigation
|
||||
</Link>
|
||||
</div>
|
||||
|
||||
{/* Back to portfolio */}
|
||||
<div className="mt-3">
|
||||
<Link
|
||||
href="/"
|
||||
className="rounded-lg border border-teal-600 bg-white px-4 py-2 text-sm font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
Back to portfolio
|
||||
</Link>
|
||||
</div>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
@@ -1,4 +1,6 @@
|
||||
import "./globals.css";
|
||||
import ThemeToggle from "@/components/theme-toggle";
|
||||
import LogoutButton from "@/components/logout-button";
|
||||
|
||||
export const metadata = {
|
||||
title: "Confidence Engine",
|
||||
@@ -9,6 +11,20 @@ export default function RootLayout({ children }) {
|
||||
return (
|
||||
<html lang="en">
|
||||
<body className="min-h-screen bg-gray-50 text-gray-900">
|
||||
<script
|
||||
dangerouslySetInnerHTML={{
|
||||
__html: `try { const saved = localStorage.getItem('confidence-engine-theme'); const theme = saved === 'dark' || saved === 'light' ? saved : (matchMedia('(prefers-color-scheme: dark)').matches ? 'dark' : 'light'); document.documentElement.dataset.theme = theme; document.documentElement.style.colorScheme = theme; } catch (_) {}`,
|
||||
}}
|
||||
/>
|
||||
<header className="app-chrome border-b border-gray-200/80">
|
||||
<div className="mx-auto flex max-w-[1600px] items-center justify-between px-6 py-3">
|
||||
<span className="text-sm font-semibold tracking-wide text-teal-700">Confidence Engine</span>
|
||||
<div className="flex items-center gap-2">
|
||||
<ThemeToggle />
|
||||
<LogoutButton />
|
||||
</div>
|
||||
</div>
|
||||
</header>
|
||||
{children}
|
||||
</body>
|
||||
</html>
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
import LoginForm from "@/components/login-form";
|
||||
|
||||
export default function LoginPage() {
|
||||
return (
|
||||
<main className="mx-auto min-h-[calc(100vh-57px)] max-w-[1200px] px-6 py-16">
|
||||
<div className="mx-auto flex flex-col gap-y-8 md:grid md:grid-cols-[minmax(0,1fr)_420px] md:gap-x-8 md:items-start">
|
||||
{/* Left column wrapper — contents on mobile for flex ordering, block on desktop as one grid cell */}
|
||||
<div className="contents md:block">
|
||||
<section>
|
||||
<h1 className="text-3xl font-bold tracking-tight text-teal-700">
|
||||
Confidence Engine
|
||||
</h1>
|
||||
|
||||
<div className="space-y-4 text-base leading-relaxed text-gray-700">
|
||||
<h2 className="text-xl font-semibold tracking-tight text-gray-900">
|
||||
Get clearer about what{"'"}s really going on.
|
||||
</h2>
|
||||
<p>
|
||||
Confidence Engine helps you work through situations that feel
|
||||
uncertain, complicated or difficult to act on.
|
||||
</p>
|
||||
<p>
|
||||
Describe the situation in your own words. Confidence Engine will
|
||||
help reconstruct what is known, what may be happening, and what
|
||||
is still unclear — then let you decide what to explore further.
|
||||
</p>
|
||||
<p className="text-gray-600">
|
||||
It doesn{"'"}t try to make the decision for you. The aim is to
|
||||
help you reach a clearer understanding so you can decide what to
|
||||
do with more confidence.
|
||||
</p>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section className="space-y-4 mt-8 md:mt-8">
|
||||
<h2 className="text-sm font-semibold uppercase tracking-[.18em] text-teal-700/70">
|
||||
How it works
|
||||
</h2>
|
||||
|
||||
<ol className="space-y-3 text-base leading-relaxed text-gray-700">
|
||||
<li className="grid grid-cols-[1.5rem_1fr]">
|
||||
<span className="font-bold">1.</span>
|
||||
<div>
|
||||
<strong>Describe your situation</strong>
|
||||
<br />
|
||||
<span className="text-gray-600">
|
||||
As much or as little as you currently know.
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
|
||||
<li className="grid grid-cols-[1.5rem_1fr]">
|
||||
<span className="font-bold">2.</span>
|
||||
<div>
|
||||
<strong>Explore what{"'"}s unclear</strong>
|
||||
<br />
|
||||
<span className="text-gray-600">
|
||||
Answer the questions that feel useful; skip the ones that
|
||||
don{"'"}t.
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
|
||||
<li className="grid grid-cols-[1.5rem_1fr]">
|
||||
<span className="font-bold">3.</span>
|
||||
<div>
|
||||
<strong>Build your Current Understanding</strong>
|
||||
<br />
|
||||
<span className="text-gray-600">
|
||||
Your picture of the situation develops as you learn more.
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
|
||||
<li className="grid grid-cols-[1.5rem_1fr]">
|
||||
<span className="font-bold">4.</span>
|
||||
<div>
|
||||
<strong>Stop when you have enough</strong>
|
||||
<br />
|
||||
<span className="text-gray-600">
|
||||
You don{"'"}t have to answer everything. Your investigation
|
||||
is saved so you can return later.
|
||||
</span>
|
||||
</div>
|
||||
</li>
|
||||
</ol>
|
||||
|
||||
<p className="text-sm italic text-gray-500">
|
||||
Try it with something real.
|
||||
<br />A decision you{"'"}re unsure about. A problem that doesn
|
||||
{"'"}t quite make sense. A situation where you feel you may be
|
||||
missing something.
|
||||
</p>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
{/* Right column — login card, explicitly positioned to top-right on desktop */}
|
||||
<section className="md:col-start-2 md:row-start-1">
|
||||
<LoginForm />
|
||||
</section>
|
||||
</div>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
+157
-7
@@ -1,15 +1,165 @@
|
||||
import ScenarioForm from "@/components/scenario-form";
|
||||
"use client";
|
||||
|
||||
import React from "react";
|
||||
import { listInvestigations, restartInvestigation } from "@/lib/storage/investigation-storage";
|
||||
import Link from "next/link";
|
||||
import { useRouter } from "next/navigation";
|
||||
|
||||
function Portfolio() {
|
||||
const router = useRouter();
|
||||
const [summaries, setSummaries] = React.useState([]);
|
||||
const [hydrated, setHydrated] = React.useState(false);
|
||||
const [loadError, setLoadError] = React.useState(null);
|
||||
const [showRestartConfirm, setShowRestartConfirm] = React.useState(false);
|
||||
|
||||
React.useEffect(() => {
|
||||
let active = true;
|
||||
(async () => {
|
||||
try {
|
||||
const investigations = await listInvestigations();
|
||||
if (active) setSummaries(investigations);
|
||||
} catch (error) {
|
||||
if (active) setLoadError(error);
|
||||
} finally {
|
||||
if (active) setHydrated(true);
|
||||
}
|
||||
})();
|
||||
return () => { active = false; };
|
||||
}, []);
|
||||
|
||||
export default function Home() {
|
||||
return (
|
||||
<main className="mx-auto max-w-[1600px] px-6 py-12">
|
||||
<main className="mx-auto max-w-[640px] px-6 py-16">
|
||||
<h1 className="mb-2 text-3xl font-bold tracking-tight">Confidence Engine</h1>
|
||||
<p className="mb-8 text-sm text-gray-500">
|
||||
Experimental prototype: enter a scenario and send it to a local LLM for
|
||||
evidence-based structured reconstruction. This is a technical vertical
|
||||
slice — not a production system.
|
||||
Investigator's notebook — index of persisted investigations.
|
||||
</p>
|
||||
<ScenarioForm />
|
||||
|
||||
{/* Investigation collection */}
|
||||
{summaries.length > 0 && (
|
||||
<section className="mb-10">
|
||||
<h2 className="mb-4 text-[13px] font-bold tracking-[.18em] uppercase text-teal-700/80">
|
||||
Investigations
|
||||
</h2>
|
||||
|
||||
{summaries.map((summary) => (
|
||||
<div key={summary.id} className="rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 py-6 shadow-sm">
|
||||
<p className="text-sm text-gray-700">
|
||||
{summary.scenario || "Untitled investigation"}
|
||||
</p>
|
||||
|
||||
<div className="mt-4 flex items-start gap-3 text-sm">
|
||||
{summary.reportExists ? (
|
||||
<div className="flex flex-col gap-1">
|
||||
<Link
|
||||
href={`/investigations/${summary.id}/report`}
|
||||
className="rounded-lg border border-teal-600 bg-white px-4 py-2 font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
View report
|
||||
</Link>
|
||||
|
||||
{summary.reportGeneratedFromRevision === summary.investigationRevision ? (
|
||||
<span className="text-[11px] font-semibold tracking-wider uppercase text-teal-700/70">
|
||||
Current
|
||||
</span>
|
||||
) : (
|
||||
<span className="text-[11px] font-semibold tracking-wider uppercase text-gray-500">
|
||||
Update available
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
<Link
|
||||
href={`/investigations/${summary.id}`}
|
||||
className="self-start rounded-lg border border-teal-600 bg-white px-4 py-2 font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
Continue investigation
|
||||
</Link>
|
||||
|
||||
<button
|
||||
onClick={() => setShowRestartConfirm(summary.id)}
|
||||
className="self-start rounded-lg border border-red-400 bg-white px-4 py-2 font-medium text-red-700 hover:bg-red-50 transition"
|
||||
>
|
||||
Restart investigation
|
||||
</button>
|
||||
|
||||
{showRestartConfirm === summary.id && (
|
||||
<div
|
||||
role="dialog"
|
||||
aria-modal="true"
|
||||
aria-labelledby={`restart-title-${summary.id}`}
|
||||
className="fixed inset-0 z-50 flex items-center justify-center bg-black/40"
|
||||
onClick={() => setShowRestartConfirm(null)}
|
||||
>
|
||||
<div
|
||||
className="w-[420px] rounded-xl border border-gray-200 bg-white p-6 shadow-lg"
|
||||
onClick={(e) => e.stopPropagation()}
|
||||
>
|
||||
<h2 id={`restart-title-${summary.id}`} className="mb-3 text-lg font-semibold">
|
||||
Restart this investigation?
|
||||
</h2>
|
||||
<p className="mb-5 text-sm text-gray-600">
|
||||
Your current investigation, findings, clarified questions, and report will be lost. Are you sure you want to continue?
|
||||
</p>
|
||||
<div className="flex justify-end gap-3">
|
||||
<button
|
||||
onClick={() => setShowRestartConfirm(null)}
|
||||
className="rounded-lg border border-gray-300 bg-white px-4 py-2 text-sm font-medium text-gray-700 hover:bg-gray-50 transition"
|
||||
>
|
||||
Cancel
|
||||
</button>
|
||||
<button
|
||||
onClick={async () => {
|
||||
setShowRestartConfirm(null);
|
||||
try {
|
||||
await restartInvestigation(summary.id);
|
||||
setSummaries(await listInvestigations());
|
||||
} catch (error) {
|
||||
setLoadError(error);
|
||||
}
|
||||
}}
|
||||
className="rounded-lg border border-red-400 bg-white px-4 py-2 text-sm font-medium text-red-700 hover:bg-red-50 transition"
|
||||
>
|
||||
Restart investigation
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</section>
|
||||
)}
|
||||
|
||||
{/* No investigations */}
|
||||
{!hydrated && (
|
||||
<section className="mb-10"><h2 className="mb-4 text-[13px] font-bold tracking-[.18em] uppercase text-teal-700/80">Investigations</h2><p className="text-sm text-gray-500 italic">Loading investigations…</p></section>
|
||||
)}
|
||||
{hydrated && loadError && (
|
||||
<section className="mb-10"><h2 className="mb-4 text-[13px] font-bold tracking-[.18em] uppercase text-teal-700/80">Investigations</h2><p className="text-sm text-red-600">Unable to load investigations.</p></section>
|
||||
)}
|
||||
{hydrated && !loadError && summaries.length === 0 && (
|
||||
<section className="mb-10">
|
||||
<h2 className="mb-4 text-[13px] font-bold tracking-[.18em] uppercase text-teal-700/80">
|
||||
Investigations
|
||||
</h2>
|
||||
<p className="text-sm text-gray-500 italic">No investigations yet.</p>
|
||||
</section>
|
||||
)}
|
||||
|
||||
<button
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
const id = crypto.randomUUID();
|
||||
router.push(`/investigations/${id}`);
|
||||
}}
|
||||
className="rounded-lg border-[2.5px] border-dashed border-teal-400 px-6 py-3 text-sm font-medium text-teal-700 hover:bg-teal-50 transition"
|
||||
>
|
||||
+ Create new investigation
|
||||
</button>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
|
||||
export default Portfolio;
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import { createClient, magicLinkRedirectTo } from "@/lib/supabase/browser.js";
|
||||
|
||||
export default function LoginForm() {
|
||||
const [email, setEmail] = useState("");
|
||||
const [status, setStatus] = useState("idle");
|
||||
const [error, setError] = useState("");
|
||||
|
||||
async function sendMagicLink(event) {
|
||||
event.preventDefault();
|
||||
setStatus("pending");
|
||||
setError("");
|
||||
const { error: signInError } = await createClient().auth.signInWithOtp({
|
||||
email,
|
||||
options: { emailRedirectTo: magicLinkRedirectTo(window.location.origin) },
|
||||
});
|
||||
if (signInError) {
|
||||
setError("We could not send a magic link. Please try again.");
|
||||
setStatus("idle");
|
||||
return;
|
||||
}
|
||||
setStatus("sent");
|
||||
}
|
||||
|
||||
return (
|
||||
<section className="w-full shrink-0 space-y-6">
|
||||
<div className="rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 py-9 shadow-sm">
|
||||
<p className="mb-3 text-[11px] font-bold uppercase tracking-[.18em] text-teal-700/70">Welcome</p>
|
||||
<h1 className="text-3xl font-bold tracking-tight">Confidence Engine</h1>
|
||||
<p className="mt-3 text-sm leading-relaxed text-gray-600">Enter your email and we will send you a secure link to continue.</p>
|
||||
<form className="mt-7 space-y-4" onSubmit={sendMagicLink}>
|
||||
<label className="block text-sm font-medium text-gray-700" htmlFor="email">Email address</label>
|
||||
<input id="email" type="email" autoComplete="email" required value={email} onChange={(event) => setEmail(event.target.value)} className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-teal-600 focus:outline-none focus:ring-2 focus:ring-teal-400" />
|
||||
<button type="submit" disabled={status === "pending"} className="rounded-lg bg-teal-700 px-5 py-2.5 text-sm font-medium text-white transition hover:bg-teal-600 disabled:cursor-wait disabled:opacity-60">
|
||||
{status === "pending" ? "Sending magic link…" : "Send magic link"}
|
||||
</button>
|
||||
</form>
|
||||
{status === "sent" && <p className="mt-5 text-sm text-green-700" role="status">Check your email for your magic link.</p>}
|
||||
{error && <p className="mt-5 text-sm text-red-700" role="alert">{error}</p>}
|
||||
</div>
|
||||
|
||||
<div className="pt-3 space-y-3">
|
||||
<p className="text-sm font-medium text-gray-700">Sign in to continue</p>
|
||||
<p className="text-sm leading-relaxed text-gray-600">Enter your email and we{'\''}ll send you a secure sign-in link.</p>
|
||||
</div>
|
||||
|
||||
<p className="text-xs text-gray-500">Your investigations are private to your account and saved so you can return to them later.</p>
|
||||
</section>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
"use client";
|
||||
|
||||
import { useEffect, useState } from "react";
|
||||
import { createClient } from "@/lib/supabase/browser.js";
|
||||
|
||||
export default function LogoutButton() {
|
||||
const [visible, setVisible] = useState(false);
|
||||
const [error, setError] = useState("");
|
||||
const [loggingOut, setLoggingOut] = useState(false);
|
||||
|
||||
useEffect(() => {
|
||||
const client = createClient();
|
||||
|
||||
async function checkSession() {
|
||||
const { data: { session } } = await client.auth.getSession();
|
||||
setVisible(!!session);
|
||||
}
|
||||
|
||||
checkSession();
|
||||
|
||||
const { data: { subscription } } = client.auth.onAuthStateChange((_event, session) => {
|
||||
setVisible(!!session);
|
||||
});
|
||||
|
||||
return () => subscription.unsubscribe();
|
||||
}, []);
|
||||
|
||||
async function handleLogout() {
|
||||
setLoggingOut(true);
|
||||
setError("");
|
||||
try {
|
||||
const client = createClient();
|
||||
await client.auth.signOut();
|
||||
window.location.href = "/login";
|
||||
} catch (err) {
|
||||
setError("Could not logout. Try again.");
|
||||
setLoggingOut(false);
|
||||
}
|
||||
}
|
||||
|
||||
if (!visible) return null;
|
||||
|
||||
return (
|
||||
<div className="flex items-center gap-2">
|
||||
<button
|
||||
type="button"
|
||||
onClick={handleLogout}
|
||||
disabled={loggingOut}
|
||||
aria-label="Logout"
|
||||
className="rounded-lg border border-gray-300 px-3 py-2 text-sm font-medium text-gray-600 transition hover:bg-gray-100 focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-teal-500 focus-visible:ring-offset-2 disabled:cursor-wait disabled:opacity-60"
|
||||
>
|
||||
{loggingOut ? "Logging out…" : "Logout"}
|
||||
</button>
|
||||
{error && (
|
||||
<p className="text-xs text-red-600" role="alert">
|
||||
{error}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
+312
-148
@@ -8,6 +8,7 @@ import InvestigationSummaryPanel from "@/components/investigation-summary-panel"
|
||||
import InvestigationSummaryPanelV2 from "@/components/investigation-summary-panel-v2";
|
||||
import InvestigationSummaryPanelV3 from "@/components/investigation-summary-panel-v3";
|
||||
import InvestigationMap from "@/components/investigation-map";
|
||||
import { reopenResolvedUnknown } from "@/lib/graph/reopen-resolved-unknown.js";
|
||||
|
||||
// ── Technical summary detector (main view filters these) ───
|
||||
const TECHNICAL_PATTERNS = [
|
||||
@@ -142,19 +143,39 @@ function FocusedQuestionBody({
|
||||
onUpdateFindingDisposition,
|
||||
onUpdateFindingProposition,
|
||||
}) {
|
||||
// ── Active-thread contribution scoping (v0.52) ──
|
||||
// focusedContributions is scenario-wide; present only the active node's contributions.
|
||||
const threadContribs = (focusedContributions || []).filter(
|
||||
(c) => c.targetNodeId === nodeId || c.originatingTargetNodeId === nodeId,
|
||||
);
|
||||
|
||||
const hasContent = focused?.question?.trim() || formulationStep === "active" || processingStep === "active" || focused?.error;
|
||||
const hasResult = Boolean(focused?.result);
|
||||
const hasAnswer = Boolean(focused?.answer);
|
||||
// A non-null result means we are still in a completed-context state even after the user selects a follow-up (which clears answer).
|
||||
// Without this guard, selecting a follow-up question would erase "Previously answered" + "Your response".
|
||||
const hasCompletedContext = processingStep !== "active" && Boolean(focused?.result);
|
||||
// Completed context: result (primary) OR prior contributions (fallback during processing/error).
|
||||
// Processing and error are transient states — they must NOT collapse completed context.
|
||||
const hasCompletedContext = Boolean(focused?.result) || (() => {
|
||||
const pc = [...threadContribs].reverse().find((c) => c?.question && c?.answer);
|
||||
return !!pc;
|
||||
})();
|
||||
|
||||
// ── Source of completed context: latest canonical Contribution when follow-up is active ──
|
||||
// After setFollowUpQuestion() mutates focused.question/answer, derive from the
|
||||
// latest completed Contribution so the narrative remains correct.
|
||||
const hasActiveFollowUp = hasCompletedContext && !hasAnswer
|
||||
&& (focused.result?.possibleFollowUpQuestions || []).some((q) => q === focused?.question);
|
||||
const latestCompletedContrib = [...(focusedContributions || [])].reverse().find((c) => c?.question && c?.answer);
|
||||
// Active follow-up detection: primary via result (when result exists), fallback via priorContribs (error state may have null result).
|
||||
const priorContribs = [...threadContribs].reverse();
|
||||
// Follow-ups from result are primary; priorContribs is fallback when result is null.
|
||||
const followUpsFromResult = focused?.result?.possibleFollowUpQuestions || [];
|
||||
const hasActiveFollowUpFromResult =
|
||||
!hasAnswer && followUpsFromResult.length > 0 && followUpsFromResult.some((q) => q === focused?.question);
|
||||
const hasActiveFollowUpFromPrior = priorContribs.length > 0
|
||||
? priorContribs.find((c) => (c.possibleFollowUpQuestions || []).length > 0)?.possibleFollowUpQuestions?.includes(focused?.question) ?? false
|
||||
: false;
|
||||
// Active follow-up requires either: a matched follow-up in result, OR priorContribs with a valid possibleFollowUp.
|
||||
const hasActiveFollowUp = (hasActiveFollowUpFromResult || hasActiveFollowUpFromPrior);
|
||||
const latestCompletedContrib = priorContribs.find((c) => c?.question && c?.answer);
|
||||
|
||||
const displayedCompletedQuestion = hasActiveFollowUp
|
||||
? (latestCompletedContrib?.question ?? focused?.question)
|
||||
@@ -215,127 +236,155 @@ function FocusedQuestionBody({
|
||||
) : null}
|
||||
|
||||
{focused?.question?.trim() && processingStep !== "active" && focused.status === "formulated" && !hasAnswer && !hasActiveFollowUp && (
|
||||
<div>
|
||||
<div data-testid="completed-narrative">
|
||||
<label htmlFor={`rw-answer-${nodeId}`} className="mb-2 block text-sm font-medium text-gray-700">Your response</label>
|
||||
<textarea id={`rw-answer-${nodeId}`} value={focusedAnswer} onChange={(e) => setFocusedAnswer(e.target.value)} rows={4} data-testid="response-textarea" className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400 disabled:cursor-not-allowed disabled:opacity-60" placeholder="What do you know about this?" />
|
||||
<button onClick={(e) => { e.stopPropagation(); handleDeconstructSubmit(nodeId, focusedAnswer); }} disabled={!focusedAnswer.trim() || processingStep === "active"} style={{ cursor: !focusedAnswer.trim() || processingStep === "active" ? "not-allowed" : "pointer" }} className="mt-3 rounded-lg border border-green-600 bg-white px-4 py-2 text-sm font-medium text-green-700 hover:bg-green-50 transition disabled:opacity-50">Submit response</button>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{processingStep === "active" && <p className="text-sm text-blue-600/70">{deconstructMsg}</p>}
|
||||
{processingStep === "active" && !hasActiveFollowUp && (
|
||||
<div data-testid="processing-indicator" className="flex items-center gap-2 text-sm text-blue-600/70">
|
||||
<svg className="h-4 w-4 animate-spin text-gray-400" viewBox="0 0 24 24" fill="none" aria-hidden="true"><circle className="opacity-25" cx="12" cy="12" r="10" stroke="currentColor" strokeWidth="4" /><path className="opacity-75" fill="currentColor" d="M4 12a8 8 0 018-8V0C5.373 0 0 5.373 0 12h4zm2 5.291A7.962 7.962 0 014 12H0c0 3.042 1.135 5.824 3 7.938l3-2.647z" /></svg>
|
||||
<span className="sr-only">Processing:</span>
|
||||
{deconstructMsg}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{focused?.result && (
|
||||
{(hasActiveFollowUp || hasCompletedContext) && (
|
||||
<>
|
||||
{/* Prior accumulated learning removed from left pane — SecondaryPreviousLearning on the right owns historical Previous Learning exclusively */}
|
||||
{/* PriorContributionsSummary was causing duplication in the two-column focused workspace */}
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">What this tells us</h3><ul className="list-disc pl-5 space-y-2">{(currentFindings?.length ? currentFindings : (focused.result.observations || [])).map((item, i) => {
|
||||
const isFinding = typeof item === "object" && item !== null && "id" in item;
|
||||
const disposition = isFinding ? item.userDisposition : null;
|
||||
const isEditing = isFinding && editingFindingId === item.id;
|
||||
if (!isFinding) {
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700">{item}</li>
|
||||
);
|
||||
}
|
||||
if (isEditing) {
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700 flex items-start gap-2">
|
||||
<textarea
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
rows={2}
|
||||
data-testid="proposition-editor"
|
||||
className="flex-1 rounded border border-blue-300 bg-blue-50/40 px-2 py-1 text-sm focus:border-blue-400 focus:outline-none focus:ring-1 focus:ring-blue-300"
|
||||
/>
|
||||
<div className="flex gap-1 shrink-0 mt-[2px]">
|
||||
<button onClick={(e) => { e.stopPropagation(); saveEditing(); }} data-testid="proposition-save" className="text-[10px] font-medium text-blue-600 underline shrink-0 hover:text-blue-700">Save</button>
|
||||
<button onClick={(e) => { e.stopPropagation(); cancelEditing(); }} data-testid="proposition-cancel" className="text-[10px] font-medium text-gray-400 underline shrink-0 hover:text-gray-500">Cancel</button>
|
||||
</div>
|
||||
</li>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700 flex items-start gap-2">
|
||||
<span className="flex-1">{item.proposition}</span>
|
||||
{onUpdateFindingProposition && (
|
||||
<button onClick={(e) => { e.stopPropagation(); startEditing(item.id, item.proposition); }} data-testid={`not-quite-${item.id}`} className="mt-[2px] text-[10px] font-medium text-amber-500 underline shrink-0 hover:text-amber-600">Not quite</button>
|
||||
)}
|
||||
{isFinding && onUpdateFindingDisposition && (
|
||||
disposition === "not_relevant" ? (
|
||||
<button onClick={(e) => { e.stopPropagation(); onUpdateFindingDisposition(item.id, null); }} data-testid={`restore-${item.id}`} className="mt-[2px] text-[10px] font-medium text-teal-600 underline shrink-0 hover:text-teal-700" title="Restore to understanding">restore</button>
|
||||
) : (
|
||||
<button onClick={(e) => { e.stopPropagation(); onUpdateFindingDisposition(item.id, "not_relevant"); }} data-testid={`not-relevant-${item.id}`} className="mt-[2px] text-[10px] font-medium text-gray-400 underline shrink-0 hover:text-red-500" title="Remove from understanding">not relevant</button>
|
||||
)
|
||||
)}
|
||||
</li>
|
||||
);
|
||||
})}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Still unclear</h3><ul className="list-disc pl-5 space-y-1">{(focused.result.uncertainties || []).map((u, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{u}</li>))}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Questions this raises</h3>
|
||||
{(focused.result.possibleFollowUpQuestions || []).length > 0 ? (
|
||||
<div className="space-y-1 mt-1">
|
||||
{hasActiveFollowUp
|
||||
? focused.result.possibleFollowUpQuestions.filter((q) => q !== focused.question).map((q, i) => (
|
||||
<button
|
||||
key={i}
|
||||
onClick={(e) => { e.stopPropagation(); setFollowUpQuestion(q); }}
|
||||
className="w-full text-left rounded-lg border border-blue-200/60 bg-blue-50/40 px-3 py-2.5 text-sm leading-relaxed text-gray-800 transition hover:border-blue-300 hover:bg-blue-100/60 cursor-pointer"
|
||||
data-testid="follow-up-question"
|
||||
>
|
||||
{q}
|
||||
{" → pick this question"}
|
||||
</button>
|
||||
))
|
||||
: focused.result.possibleFollowUpQuestions.map((q, i) => {
|
||||
const isCurrentQuestion = q === focused?.question;
|
||||
return (
|
||||
<button
|
||||
key={i}
|
||||
onClick={(e) => { if (!isCurrentQuestion) { e.stopPropagation(); setFollowUpQuestion(q); } }}
|
||||
style={{ cursor: isCurrentQuestion ? "default" : "pointer" }}
|
||||
className={`w-full text-left rounded-lg border px-3 py-2.5 text-sm leading-relaxed transition ${
|
||||
isCurrentQuestion
|
||||
? "border-gray-200 bg-gray-100/60 text-gray-400 cursor-default"
|
||||
: "border-blue-200/60 bg-blue-50/40 text-gray-800 hover:border-blue-300 hover:bg-blue-100/60"
|
||||
}`}
|
||||
data-testid="follow-up-question"
|
||||
>
|
||||
{q}
|
||||
{isCurrentQuestion ? " (current question)" : " → pick this question"}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
) : (
|
||||
<p className="text-xs text-gray-400">None yet</p>
|
||||
)}
|
||||
{/* Derived sections fallback to priorContribs data during processing/error when result is null */}
|
||||
{(() => {
|
||||
const effectiveObservations = currentFindings?.length ? currentFindings :
|
||||
(focused?.result?.observations ?? priorContribs.find((c) => c?.observations)?.observations);
|
||||
const effectiveUncertainties = focused?.result?.uncertainties ?? priorContribs.find((c) => c?.uncertainties)?.uncertainties;
|
||||
const effectiveFollowUps = focused?.result?.possibleFollowUpQuestions || priorContribs.find((c) => c?.possibleFollowUpQuestions)?.possibleFollowUpQuestions;
|
||||
const effectiveAssumptions = focused?.result?.assumptions || priorContribs.find((c) => c?.assumptions)?.assumptions;
|
||||
const effectiveRelationships = focused?.result?.relationships || priorContribs.find((c) => c?.relationships)?.relationships;
|
||||
|
||||
{/* In-place answer textarea for the active follow-up — renders only when a candidate is selected */}
|
||||
{hasActiveFollowUp ? (
|
||||
<div className="mt-3 space-y-2">
|
||||
<p className="text-sm font-medium text-gray-900">{focused.question}</p>
|
||||
<textarea
|
||||
id={`rw-answer-fu-${nodeId}`}
|
||||
value={focusedAnswer}
|
||||
onChange={(e) => setFocusedAnswer(e.target.value)}
|
||||
rows={4}
|
||||
data-testid="follow-up-textarea"
|
||||
className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400 disabled:cursor-not-allowed disabled:opacity-60"
|
||||
placeholder="What do you know about this?"
|
||||
/>
|
||||
<button
|
||||
onClick={(e) => { e.stopPropagation(); handleDeconstructSubmit(nodeId, focusedAnswer); }}
|
||||
disabled={!focusedAnswer.trim() || processingStep === "active"}
|
||||
style={{ cursor: !focusedAnswer.trim() || processingStep === "active" ? "not-allowed" : "pointer" }}
|
||||
className="rounded-lg border border-green-600 bg-white px-4 py-2 text-sm font-medium text-green-700 hover:bg-green-50 transition disabled:opacity-50"
|
||||
>
|
||||
Submit response
|
||||
</button>
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Assumptions</h3><ul className="list-disc pl-5 space-y-1">{(focused.result.assumptions || []).map((a, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{a}</li>))}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Connections</h3><ul className="list-disc pl-5 space-y-1">{(focused.result.relationships || []).map((r, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{r.from} → {r.to} ({r.type})</li>))}</ul></div>
|
||||
return (
|
||||
<>
|
||||
{/* Prior accumulated learning removed from left pane — SecondaryPreviousLearning on the right owns historical Previous Learning exclusively */}
|
||||
{/* PriorContributionsSummary was causing duplication in the two-column focused workspace */}
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">What this tells us</h3><ul className="list-disc pl-5 space-y-2">{(effectiveObservations || []).map((item, i) => {
|
||||
const isFinding = typeof item === "object" && item !== null && "id" in item;
|
||||
const disposition = isFinding ? item.userDisposition : null;
|
||||
const isEditing = isFinding && editingFindingId === item.id;
|
||||
if (!isFinding) {
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700">{item}</li>
|
||||
);
|
||||
}
|
||||
if (isEditing) {
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700 flex items-start gap-2">
|
||||
<textarea
|
||||
value={draft}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
rows={2}
|
||||
data-testid="proposition-editor"
|
||||
className="flex-1 rounded border border-blue-300 bg-blue-50/40 px-2 py-1 text-sm focus:border-blue-400 focus:outline-none focus:ring-1 focus:ring-blue-300"
|
||||
/>
|
||||
<div className="flex gap-1 shrink-0 mt-[2px]">
|
||||
<button onClick={(e) => { e.stopPropagation(); saveEditing(); }} data-testid="proposition-save" className="text-[10px] font-medium text-blue-600 underline shrink-0 hover:text-blue-700">Save</button>
|
||||
<button onClick={(e) => { e.stopPropagation(); cancelEditing(); }} data-testid="proposition-cancel" className="text-[10px] font-medium text-gray-400 underline shrink-0 hover:text-gray-500">Cancel</button>
|
||||
</div>
|
||||
</li>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<li key={i} className="text-sm leading-relaxed text-gray-700 flex items-start gap-2">
|
||||
<span className="flex-1">{item.proposition}</span>
|
||||
{onUpdateFindingProposition && (
|
||||
<button onClick={(e) => { e.stopPropagation(); startEditing(item.id, item.proposition); }} data-testid={`not-quite-${item.id}`} className="mt-[2px] text-[10px] font-medium text-amber-500 underline shrink-0 hover:text-amber-600">Not quite</button>
|
||||
)}
|
||||
{isFinding && onUpdateFindingDisposition && (
|
||||
disposition === "not_relevant" ? (
|
||||
<button onClick={(e) => { e.stopPropagation(); onUpdateFindingDisposition(item.id, null); }} data-testid={`restore-${item.id}`} className="mt-[2px] text-[10px] font-medium text-teal-600 underline shrink-0 hover:text-teal-700" title="Restore to understanding">restore</button>
|
||||
) : (
|
||||
<button onClick={(e) => { e.stopPropagation(); onUpdateFindingDisposition(item.id, "not_relevant"); }} data-testid={`not-relevant-${item.id}`} className="mt-[2px] text-[10px] font-medium text-gray-400 underline shrink-0 hover:text-red-500" title="Remove from understanding">not relevant</button>
|
||||
)
|
||||
)}
|
||||
</li>
|
||||
);
|
||||
})}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Still unclear</h3><ul className="list-disc pl-5 space-y-1">{(effectiveUncertainties || []).map((u, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{u}</li>))}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Questions this raises</h3>
|
||||
{(effectiveFollowUps || []).length > 0 ? (
|
||||
<div className="space-y-1 mt-1">
|
||||
{hasActiveFollowUp
|
||||
? effectiveFollowUps.filter((q) => q !== focused.question).map((q, i) => (
|
||||
<button
|
||||
key={i}
|
||||
onClick={(e) => { e.stopPropagation(); setFollowUpQuestion(q); }}
|
||||
className="w-full text-left rounded-lg border border-blue-200/60 bg-blue-50/40 px-3 py-2.5 text-sm leading-relaxed text-gray-800 transition hover:border-blue-300 hover:bg-blue-100/60 cursor-pointer"
|
||||
data-testid="follow-up-question"
|
||||
>
|
||||
{q}
|
||||
{" → pick this question"}
|
||||
</button>
|
||||
))
|
||||
: effectiveFollowUps.map((q, i) => {
|
||||
const isCurrentQuestion = q === focused?.question;
|
||||
return (
|
||||
<button
|
||||
key={i}
|
||||
onClick={(e) => { if (!isCurrentQuestion) { e.stopPropagation(); setFollowUpQuestion(q); } }}
|
||||
style={{ cursor: isCurrentQuestion ? "default" : "pointer" }}
|
||||
className={`w-full text-left rounded-lg border px-3 py-2.5 text-sm leading-relaxed transition ${
|
||||
isCurrentQuestion
|
||||
? "border-gray-200 bg-gray-100/60 text-gray-400 cursor-default"
|
||||
: "border-blue-200/60 bg-blue-50/40 text-gray-800 hover:border-blue-300 hover:bg-blue-100/60"
|
||||
}`}
|
||||
data-testid="follow-up-question"
|
||||
>
|
||||
{q}
|
||||
{isCurrentQuestion ? " (current question)" : " → pick this question"}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
) : (
|
||||
<p className="text-xs text-gray-400">None yet</p>
|
||||
)}
|
||||
|
||||
{/* In-place answer textarea for the active follow-up */}
|
||||
{hasActiveFollowUp ? (
|
||||
<div className="mt-3 space-y-2" data-testid="follow-up-block">
|
||||
<p className="text-sm font-medium text-gray-900">{focused.question}</p>
|
||||
<textarea
|
||||
id={`rw-answer-fu-${nodeId}`}
|
||||
value={focusedAnswer}
|
||||
onChange={(e) => setFocusedAnswer(e.target.value)}
|
||||
rows={4}
|
||||
data-testid="follow-up-textarea"
|
||||
disabled={processingStep === "active"}
|
||||
className="w-full rounded-lg border border-gray-300 px-4 py-3 text-sm focus:border-gray-500 focus:outline-none focus:ring-2 focus:ring-gray-400 disabled:cursor-not-allowed disabled:opacity-60"
|
||||
placeholder="What do you know about this?"
|
||||
/>
|
||||
<button
|
||||
onClick={(e) => { e.stopPropagation(); handleDeconstructSubmit(nodeId, focusedAnswer); }}
|
||||
disabled={!focusedAnswer.trim() || processingStep === "active"}
|
||||
style={{ cursor: !focusedAnswer.trim() || processingStep === "active" ? "not-allowed" : "pointer" }}
|
||||
className="rounded-lg border border-green-600 bg-white px-4 py-2 text-sm font-medium text-green-700 hover:bg-green-50 transition disabled:opacity-50"
|
||||
>
|
||||
Submit response
|
||||
</button>
|
||||
{processingStep === "active" && (
|
||||
<div data-testid="processing-indicator" className="mt-1 flex items-center gap-2 text-sm text-blue-600/70">
|
||||
<svg className="h-4 w-4 animate-spin text-gray-400" viewBox="0 0 24 24" fill="none" aria-hidden="true"><circle className="opacity-25" cx="12" cy="12" r="10" stroke="currentColor" strokeWidth="4" /><path className="opacity-75" fill="currentColor" d="M4 12a8 8 0 018-8V0C5.373 0 0 5.373 0 12h4zm2 5.291A7.962 7.962 0 014 12H0c0 3.042 1.135 5.824 3 7.938l3-2.647z" /></svg>
|
||||
<span className="sr-only">Processing:</span>
|
||||
{deconstructMsg}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Assumptions</h3><ul className="list-disc pl-5 space-y-1">{(effectiveAssumptions || []).map((a, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{a}</li>))}</ul></div>
|
||||
<div><h3 className="mb-1 text-[11px] font-semibold tracking-widest uppercase text-gray-500">Connections</h3><ul className="list-disc pl-5 space-y-1">{(effectiveRelationships || []).map((r, i) => (<li key={i} className="text-sm leading-relaxed text-gray-700">{r.from} → {r.to} ({r.type})</li>))}</ul></div>
|
||||
</>
|
||||
);
|
||||
})()}
|
||||
</>
|
||||
)}
|
||||
|
||||
@@ -351,19 +400,13 @@ function FocusedQuestionBody({
|
||||
|
||||
// ── Persistent navigation controls (overlay-level, outside content grid) ──
|
||||
|
||||
function FocusedWorkspaceNavigation({ nodeId, doneForNow, onBackToOpenQuestions, isDoneForNowActive }) {
|
||||
const canDoneForNow = Boolean(isDoneForNowActive);
|
||||
function FocusedWorkspaceNavigation({ nodeId, doneForNow, isDoneForNowActive, isProcessing, onImmediateGraphChange }) {
|
||||
const canDoneForNow = Boolean(isDoneForNowActive) && !isProcessing;
|
||||
return (
|
||||
<div className="mt-6 flex items-center justify-between gap-4 border-t border-gray-200 pt-5">
|
||||
<button
|
||||
onClick={(e) => { e.stopPropagation(); onBackToOpenQuestions?.(); }}
|
||||
style={{ cursor: "pointer" }}
|
||||
className="text-sm text-gray-400 underline hover:text-gray-600 transition whitespace-nowrap"
|
||||
>
|
||||
Back to open questions
|
||||
</button>
|
||||
<div className="mt-6 flex items-center justify-end border-t border-gray-200 pt-5">
|
||||
<button
|
||||
onClick={(e) => { e.stopPropagation(); doneForNow?.(); }}
|
||||
disabled={canDoneForNow ? false : true}
|
||||
style={{ cursor: canDoneForNow ? "pointer" : "not-allowed" }}
|
||||
className={`rounded-lg border px-4 py-2 text-sm font-medium transition whitespace-nowrap ${canDoneForNow ? 'border-gray-300 bg-white text-gray-600 hover:bg-gray-50' : 'border-gray-100 bg-gray-50 text-gray-300'}`}
|
||||
>
|
||||
@@ -1143,7 +1186,7 @@ function FocusedInvestigationWorkspace({
|
||||
|
||||
function OpenQuestionsPanel({
|
||||
graph, selectedPresentationItemId, focusedPresentationItemId, hasFocusedContent,
|
||||
focused, formulationStep, formulateMsg, processingStep, deconstructMsg, doneForNowIds,
|
||||
focused, formulationStep, formulateMsg, processingStep, deconstructMsg, doneForNowIds, cuSynthesisLoading,
|
||||
startFocused, handleDeconstructSubmit, retryFormulation, setSelectedPresentationItemId,
|
||||
setFocusedPresentationItemId, setFocusedAnswer, focusedAnswer, setDoneForNowIds,
|
||||
setFollowUpQuestion, focusedContributions, focusedInvestigations, setIsFocusedWorkspaceOpen,
|
||||
@@ -1151,12 +1194,15 @@ function OpenQuestionsPanel({
|
||||
onUpdateFindingDisposition,
|
||||
onUpdateFindingProposition,
|
||||
}) {
|
||||
const resolvedIds = new Set(graph?.resolvedNodeIds || []);
|
||||
|
||||
const openNodes = (graph?.nodes || []).filter(
|
||||
(n) => n.kind === "unknown" && n.status !== "resolved" && !doneForNowIds.includes(n.id),
|
||||
(n) => n.kind === "unknown" && !resolvedIds.has(n.id) && !doneForNowIds.includes(n.id),
|
||||
);
|
||||
|
||||
if (openNodes.length <= 0) return null;
|
||||
|
||||
// Zero open questions: delegate to inline ReasoningWorkspace section for invitation + clarifications rendering
|
||||
// No Open Questions cards to render when zero — invitation rendered inline instead
|
||||
// The panel itself returns null; see ReasoningWorkspace inline section for the milestone UI.
|
||||
// Check whether a node has a completed focused result stored locally.
|
||||
const hasCompletedInvestigation = (nid) => {
|
||||
const inv = focusedInvestigations?.[nid];
|
||||
@@ -1264,6 +1310,7 @@ export default function ReasoningWorkspace({
|
||||
scenario,
|
||||
status,
|
||||
updateStatus,
|
||||
cuSynthesisLoading,
|
||||
currentUnderstanding: propUnderstanding,
|
||||
result,
|
||||
answer,
|
||||
@@ -1278,6 +1325,20 @@ export default function ReasoningWorkspace({
|
||||
onUpdateFindingProposition,
|
||||
/* ── v0.49 — done-for-now promotion callback ───────── */
|
||||
onSummaryUpdate,
|
||||
/* ── immediate graph transition (Done acknowledged before async) ── */
|
||||
onImmediateGraphChange,
|
||||
/* ── canonical graph-replacement seam (future Re-open) ── */
|
||||
onSituationGraphChange,
|
||||
/* ── v0.59a — provenance revision ──────────────────── */
|
||||
investigationRevision,
|
||||
/* ── test init seam (no effect → immediate state) ───────── */
|
||||
initialPostAnalyseStatus,
|
||||
/* ── v0.54b — investigation overview transient state ─── */
|
||||
overviewState,
|
||||
setOverviewState,
|
||||
overviewLoading,
|
||||
handleRequestOverview,
|
||||
onNavigateToReport,
|
||||
}) {
|
||||
const [investigationHistory, setInvestigationHistory] = useState([]);
|
||||
const turnCounter = useRef(0);
|
||||
@@ -1297,7 +1358,7 @@ export default function ReasoningWorkspace({
|
||||
|
||||
// ── RTO.29D — post-Analyse initial reflection surface ─────────
|
||||
const [initialReflectionActive, setInitialReflectionActive] = useState(false);
|
||||
const [postAnalyseStatus, setPostAnalyseStatus] = useState(null);
|
||||
const [postAnalyseStatus, setPostAnalyseStatus] = useState(initialPostAnalyseStatus ?? null);
|
||||
|
||||
// Capture the current selected question at submit time (not from a stale ref)
|
||||
const capturePendingTurn = (selectedQuestion, answerText) => {
|
||||
@@ -1749,11 +1810,21 @@ export default function ReasoningWorkspace({
|
||||
{/* Current Understanding + Situation — independent vertical flow */}
|
||||
<div className="flex gap-3 items-start flex-wrap">
|
||||
{/* Current Understanding — prominent orienting surface */}
|
||||
<div className={`rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-7 pb-8 shadow-sm flex-1 min-w-0 ${hasGraph ? 'lg:max-w-2xl' : ''}`}>
|
||||
<div id="cu-scroll-target" className={`rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-7 pb-8 shadow-sm flex-1 min-w-0 relative ${hasGraph ? 'lg:max-w-2xl' : ''}`}>
|
||||
<h2 className="mb-4 text-[11px] font-bold tracking-[.18em] uppercase text-teal-700/60">
|
||||
Current Understanding
|
||||
</h2>
|
||||
<p className="text-lg leading-relaxed text-gray-800">{propUnderstanding}</p>
|
||||
{cuSynthesisLoading && (
|
||||
<div className="absolute inset-0 cu-skeleton-overlay rounded-xl overflow-hidden flex flex-col items-center justify-center bg-gray-50" role="status" aria-live="polite">
|
||||
<p className="text-sm font-semibold text-teal-800 mb-6 relative z-10 tracking-wide">Clarifying your current understanding…</p>
|
||||
<div className="cu-skeleton-lines w-full max-w-lg px-8 pb-8" aria-hidden="true">
|
||||
{Array.from({ length: 7 }, (_, i) => (
|
||||
<div key={i} className={`cu-skeleton-line mb-3 ${['w-[94%]', 'w-[87%]', 'w-[96%]', 'w-[72%]', 'w-[90%]', 'w-[82%]', 'w-[64%]'][i % 7]}`} />
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Situation panel during initial reflection */}
|
||||
@@ -1781,12 +1852,14 @@ export default function ReasoningWorkspace({
|
||||
{(() => {
|
||||
const resolvedIds = new Set(graph?.resolvedNodeIds || []);
|
||||
|
||||
// Open Questions: unresolved unknown nodes only (investigable)
|
||||
// Open Questions: unknown nodes NOT in resolvedNodeIds (canonically investigable)
|
||||
const openUnknowns = (graph?.nodes || []).filter(
|
||||
(n) =>
|
||||
n.kind === "unknown" &&
|
||||
n.status !== "resolved" &&
|
||||
!resolvedIds.has(n.id),
|
||||
(n) => n.kind === "unknown" && !resolvedIds.has(n.id),
|
||||
);
|
||||
|
||||
// Clarified questions: unknown nodes that ARE in resolvedNodeIds (read-only historical)
|
||||
const clarifiedQuestions = (graph?.nodes || []).filter(
|
||||
(n) => n.kind === "unknown" && resolvedIds.has(n.id),
|
||||
);
|
||||
|
||||
// Possible Interpretations: unresolved assumption nodes only (informational, not investigable)
|
||||
@@ -1799,8 +1872,8 @@ export default function ReasoningWorkspace({
|
||||
|
||||
return (
|
||||
<>
|
||||
{/* OPEN QUESTIONS — unknown nodes (clickable → focused investigation) */}
|
||||
{openUnknowns.length > 0 && (
|
||||
{/* OPEN QUESTIONS or zero-Open-Questions milestone invitation (exclusive) */}
|
||||
{openUnknowns.length > 0 ? (
|
||||
<div className="space-y-3">
|
||||
<h2 className="text-[11px] font-semibold tracking-widest uppercase text-gray-500">
|
||||
Open Questions
|
||||
@@ -1877,6 +1950,80 @@ export default function ReasoningWorkspace({
|
||||
);
|
||||
})()}
|
||||
</div>
|
||||
) : openUnknowns.length === 0 && clarifiedQuestions.length > 0 && !cuSynthesisLoading ? (
|
||||
<div className="space-y-4">
|
||||
{/* Milestone invitation + overview action */}
|
||||
<div className="rounded-xl border-[2.5px] border-teal-300/60 bg-gradient-to-b from-teal-50/40 to-white px-7 pt-5 pb-6">
|
||||
<p className="text-sm leading-relaxed text-gray-700 mb-4">
|
||||
{'You\'ve now worked through all of the questions we surfaced. Would you like to see an overview of what we understand so far?'}
|
||||
</p>
|
||||
<button
|
||||
onClick={() => {
|
||||
onNavigateToReport?.();
|
||||
}}
|
||||
disabled={overviewLoading}
|
||||
style={{ cursor: overviewLoading ? "wait" : "pointer" }}
|
||||
className="rounded-lg border border-teal-600 bg-white px-4 py-2 text-sm font-medium text-teal-700 hover:bg-teal-50 transition disabled:opacity-60"
|
||||
>
|
||||
{overviewLoading ? "Generating overview…" : "Review current understanding"}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Investigation overview surface — transient, non-persistent */}
|
||||
{overviewState && (
|
||||
<div className="space-y-4" data-testid="investigation-overview">
|
||||
<div className="rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-6 pb-7 shadow-sm">
|
||||
<h3 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-teal-700/70">
|
||||
What we understand
|
||||
</h3>
|
||||
<p className="text-base leading-relaxed text-gray-800">{overviewState.understanding}</p>
|
||||
</div>
|
||||
{overviewState.plausibleInterpretations && (
|
||||
<div className="rounded-xl border border-blue-200/70 bg-blue-50/40 px-8 pt-6 pb-7 shadow-sm">
|
||||
<h3 className="mb-3 text-[11px] font-bold tracking-[.18em] uppercase text-blue-600/70">
|
||||
What remains plausible
|
||||
</h3>
|
||||
<p className="text-sm leading-relaxed text-gray-700 italic">{overviewState.plausibleInterpretations}</p>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{/* QUESTIONS WE HAVE CLARIFIED — resolved unknowns shown post-Done */}
|
||||
{clarifiedQuestions.length > 0 && (
|
||||
<div className="space-y-3">
|
||||
<h2 className="text-[11px] font-semibold tracking-widest uppercase text-gray-500">
|
||||
Questions we have clarified
|
||||
</h2>
|
||||
{clarifiedQuestions.map((node) => (
|
||||
<div key={node.id} className="space-y-1">
|
||||
<div
|
||||
className="w-full text-left rounded-lg border border-green-200/60 bg-green-50/30 px-5 py-4"
|
||||
>
|
||||
<span className="block text-sm leading-relaxed text-gray-900">{node.label}</span>
|
||||
{node.description && node.description !== node.label && (
|
||||
<p className="mt-1.5 text-xs leading-snug text-gray-500">{node.description}</p>
|
||||
)}
|
||||
<div className="mt-2 flex items-center gap-3">
|
||||
<span className="text-[10px] uppercase tracking-wider text-green-600">Clarified</span>
|
||||
<button
|
||||
onClick={() => {
|
||||
const nextGraph = reopenResolvedUnknown(graph, node.id);
|
||||
onSituationGraphChange(nextGraph);
|
||||
setDoneForNowIds((prev) => prev.filter((id) => id !== node.id));
|
||||
}}
|
||||
style={{ cursor: "pointer" }}
|
||||
className="text-[10px] font-medium uppercase tracking-wider text-amber-600 hover:text-amber-700 underline transition"
|
||||
>
|
||||
Re-open
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* POSSIBLE INTERPRETATIONS — assumption nodes (informational, not investigable) */}
|
||||
@@ -1913,7 +2060,7 @@ export default function ReasoningWorkspace({
|
||||
|
||||
{/* Current Understanding — independent row, full-width of left area (cols 1-2) */}
|
||||
{propUnderstanding && hasCurrentSummaryCondition && postAnalyseStatus !== "success" && (
|
||||
<div className="lg:row-start-1 lg:col-span-full rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-7 pb-8 shadow-sm">
|
||||
<div id="cu-scroll-target" className="lg:row-start-1 lg:col-span-full rounded-xl border-[2.5px] border-teal-300/70 bg-gradient-to-b from-teal-50/60 to-white px-8 pt-7 pb-8 shadow-sm relative">
|
||||
<CurrentUnderstandingCard currentSummary={graph?.currentSummary || result?.updatedSituationGraph?.currentSummary} plainLanguage={propUnderstanding} />
|
||||
</div>
|
||||
)}
|
||||
@@ -1936,6 +2083,7 @@ export default function ReasoningWorkspace({
|
||||
processingStep={processingStep}
|
||||
deconstructMsg={deconstructMsg}
|
||||
doneForNowIds={doneForNowIds}
|
||||
cuSynthesisLoading={cuSynthesisLoading}
|
||||
startFocused={startFocused}
|
||||
handleDeconstructSubmit={handleDeconstructSubmit}
|
||||
retryFormulation={retryFormulation}
|
||||
@@ -2089,12 +2237,12 @@ export default function ReasoningWorkspace({
|
||||
<button
|
||||
onClick={(e) => { e.stopPropagation(); setFocusedAnswer(""); setFocusedPresentationItemId(null); setIsFocusedWorkspaceOpen(false); }}
|
||||
style={{ cursor: "pointer" }}
|
||||
aria-label="Close investigation"
|
||||
title="Close investigation"
|
||||
aria-label="Close workspace"
|
||||
title="Close workspace"
|
||||
className="absolute right-4 top-3 z-20 flex items-center gap-2 rounded-lg border border-gray-300 bg-white/90 px-4 py-2 text-sm font-medium text-gray-600 shadow-sm transition hover:bg-gray-50"
|
||||
>
|
||||
<svg width="14" height="14" viewBox="0 0 14 14" fill="none" stroke="currentColor" strokeWidth="2" strokeLinecap="round"><path d="M1 1l12 12M13 1L1 13"/></svg>
|
||||
Close investigation
|
||||
Close workspace
|
||||
</button>
|
||||
|
||||
{/* Scrollable workspace body */}
|
||||
@@ -2133,17 +2281,33 @@ export default function ReasoningWorkspace({
|
||||
<FocusedWorkspaceNavigation
|
||||
nodeId={focusedPresentationItemId}
|
||||
doneForNow={() => {
|
||||
/* ── v0.49 — promote eligible focused findings into Current Understanding ─── */
|
||||
if (processingStep === "active") return;
|
||||
/* ── Immediate transition: resolve target node BEFORE async — cuSynthesisLoading also set ─── */
|
||||
if (onImmediateGraphChange && focusedPresentationItemId) {
|
||||
const currentGraph = result?.situationGraph;
|
||||
if (currentGraph) {
|
||||
const resolvedIds = new Set(currentGraph.resolvedNodeIds || []);
|
||||
resolvedIds.add(focusedPresentationItemId);
|
||||
const nextNodes = (currentGraph.nodes || []).map((n) =>
|
||||
n.id === focusedPresentationItemId ? { ...n, status: "resolved" } : n,
|
||||
);
|
||||
onImmediateGraphChange({
|
||||
...currentGraph,
|
||||
nodes: nextNodes,
|
||||
resolvedNodeIds: Array.from(resolvedIds),
|
||||
});
|
||||
}
|
||||
}
|
||||
onSummaryUpdate?.(focusedPresentationItemId);
|
||||
setDoneForNowIds((prev) => [...prev, focusedPresentationItemId]);
|
||||
setFocusedAnswer("");
|
||||
setFocusedPresentationItemId(null);
|
||||
/* ── v0.49 fix — close overlay after semantic action ─── */
|
||||
setIsFocusedWorkspaceOpen(false);
|
||||
}}
|
||||
isDoneForNowActive={Boolean(getFocusedInvestigation()?.question?.trim())}
|
||||
onBackToOpenQuestions={() => {
|
||||
setFocusedAnswer("");
|
||||
setFocusedPresentationItemId(null);
|
||||
}}
|
||||
isProcessing={processingStep === "active"}
|
||||
onImmediateGraphChange={onImmediateGraphChange}
|
||||
/>
|
||||
)}
|
||||
</>
|
||||
|
||||
+366
-59
@@ -5,8 +5,8 @@ import { useState, useRef, useMemo } from "react";
|
||||
import DiagnosticsView from "@/components/diagnostics-view";
|
||||
import ReasoningWorkspace, { LoadingOverlay, ContinueLaterBanner } from "@/components/reasoning-workspace";
|
||||
import { mockFetch, AVAILABLE_SCENARIOS } from "@/lib/mocks/confidence-engine/mock-client";
|
||||
import { deriveFindingsFromContributions, normalizeFindings, produceFindingInformedSummary } from "@/lib/graph/finding-helpers";
|
||||
import { loadInvestigation, saveInvestigation, clearInvestigation } from "@/lib/storage/investigation-storage";
|
||||
import { deriveFindingsFromContributions, normalizeFindings } from "@/lib/graph/finding-helpers";
|
||||
import { loadInvestigation, saveInvestigation, restartInvestigation } from "@/lib/storage/investigation-storage";
|
||||
|
||||
/* Compile-time env resolution — NEXT_PUBLIC_ vars are injected by Next.js at build */
|
||||
const MOCK_ENABLED = process.env.NEXT_PUBLIC_CONFIDENCE_ENGINE_MOCKS === "true";
|
||||
@@ -33,6 +33,28 @@ export async function submitScenarioForStartCase(fetchImpl, scenario) {
|
||||
});
|
||||
}
|
||||
|
||||
export function isUnavailableStartResponse(response, data) {
|
||||
return response.status === 503 &&
|
||||
data?.success === false &&
|
||||
data?.error === "Reasoning service is temporarily unavailable.";
|
||||
}
|
||||
|
||||
export function UnavailableStartPanel({ onRetry }) {
|
||||
return (
|
||||
<div className="rounded-lg border border-amber-300 bg-amber-50 px-4 py-3 text-sm text-amber-900">
|
||||
<p className="font-medium">Confidence Engine is temporarily unavailable.</p>
|
||||
<p className="mt-1">We couldn't process this right now. Your scenario is still here and you can try again.</p>
|
||||
<button
|
||||
type="button"
|
||||
onClick={onRetry}
|
||||
className="mt-3 rounded-lg border border-amber-400 px-4 py-2 text-sm font-medium text-amber-900 transition hover:bg-amber-100"
|
||||
>
|
||||
Retry
|
||||
</button>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export async function submitAnswerForUpdateCase(
|
||||
fetchImpl,
|
||||
{ situationGraph, previousQuestion, answer, findings },
|
||||
@@ -67,6 +89,19 @@ export async function submitAnswerForUpdateCase(
|
||||
};
|
||||
}
|
||||
|
||||
export async function synthesizeFromFindings(fetchImpl, { situationGraph, findings }) {
|
||||
const response = await fetchImpl("/api/cases/synthesis", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ situationGraph, findings }),
|
||||
});
|
||||
|
||||
return {
|
||||
ok: response.ok,
|
||||
data: await response.json(),
|
||||
};
|
||||
}
|
||||
|
||||
function normaliseStartResult(data) {
|
||||
return {
|
||||
...data,
|
||||
@@ -226,7 +261,39 @@ export function derivePrimarySurface(result, status, _showExperimentView, scenar
|
||||
return "SCENARIO_ENTRY";
|
||||
}
|
||||
|
||||
export default function ScenarioForm() {
|
||||
/**
|
||||
* Orchestrate the authoritative episode reconsideration flow.
|
||||
* Exported for deterministic testing — domain functions and server endpoint accepted as parameters.
|
||||
*/
|
||||
export async function executeEpisodeDone({
|
||||
resultSituationGraph,
|
||||
targetNodeId,
|
||||
focusedContributions,
|
||||
findings,
|
||||
episodeDoneServer,
|
||||
synthesizeFn,
|
||||
setResult: setAppState,
|
||||
}) {
|
||||
const serverResult = await episodeDoneServer({
|
||||
situationGraph: resultSituationGraph,
|
||||
targetNodeId,
|
||||
contributions: focusedContributions ?? [],
|
||||
findings,
|
||||
});
|
||||
|
||||
if (!serverResult.success) {
|
||||
return { success: false, stage: "episode_done", error: serverResult.error };
|
||||
}
|
||||
|
||||
const nextGraph = serverResult.updatedSituationGraph;
|
||||
setAppState(prev => ({ ...(prev ?? {}), situationGraph: nextGraph }));
|
||||
|
||||
const synthesisResult = await synthesizeFn(nextGraph, findings);
|
||||
|
||||
return { success: true, nextGraph, synthesisResult };
|
||||
}
|
||||
|
||||
export default function ScenarioForm({ investigationId, onNavigateToReport }) {
|
||||
const [scenario, setScenario] = useState("");
|
||||
const [status, setStatus] = useState("idle"); // idle | loading | error | success
|
||||
const [result, setResult] = useState(null);
|
||||
@@ -236,14 +303,38 @@ export default function ScenarioForm() {
|
||||
const [updateResult, setUpdateResult] = useState(null);
|
||||
const [lastSubmittedAnswer, setLastSubmittedAnswer] = useState("");
|
||||
const [currentUnderstanding, setCurrentUnderstanding] = useState(null);
|
||||
const [cuSynthesisLoading, setCuSynthesisLoading] = useState(false);
|
||||
const [mockScenario, setMockScenario] = useState("");
|
||||
const [hideFacilitatorOnLanding, setHideFacilitatorOnLanding] = useState(false);
|
||||
|
||||
/* ── v0.54b — investigation overview transient state ────── */
|
||||
const [overviewState, setOverviewState] = useState(null);
|
||||
const [overviewLoading, setOverviewLoading] = useState(false);
|
||||
|
||||
/* ── v0.55 — persisted investigation report (derived artefact) ── */
|
||||
const [investigationReport, setInvestigationReport] = useState(null);
|
||||
|
||||
/* ── v0.59a — provenance: Investigation revision tracking ── */
|
||||
const [investigationRevision, setInvestigationRevision] = useState(0);
|
||||
|
||||
/* ── in-flight gate for episode reconsideration on Done ──── */
|
||||
const doneInProgressRef = useRef(false);
|
||||
|
||||
/* ── RTO.31: focused contributions ownership ─────────────── */
|
||||
const [focusedContributions, setFocusedContributions] = useState([]);
|
||||
|
||||
/* ── v2 findings from focused contributions ─────────────── */
|
||||
const [findings, setFindings] = useState([]);
|
||||
const [hydrated, setHydrated] = useState(false);
|
||||
|
||||
function persist(snapshot) {
|
||||
void saveInvestigation(snapshot).catch((error) => console.error("Investigation autosave failed", error));
|
||||
}
|
||||
|
||||
function restartPersistedInvestigation() {
|
||||
void restartInvestigation(investigationId)
|
||||
.catch((error) => console.error("Investigation restart failed", error));
|
||||
}
|
||||
|
||||
function appendFinding(finding) {
|
||||
setFindings((prev) => {
|
||||
@@ -252,56 +343,187 @@ export default function ScenarioForm() {
|
||||
}
|
||||
|
||||
function updateFindingDisposition(findingId, newDisposition) {
|
||||
setFindings((prev) =>
|
||||
prev.map((f) => (f.id === findingId ? { ...f, userDisposition: newDisposition } : f)),
|
||||
// Derive explicit next state — not a React-state reread.
|
||||
const nextFindings = (findings ?? []).map((f) =>
|
||||
f.id === findingId ? { ...f, userDisposition: newDisposition } : f,
|
||||
);
|
||||
|
||||
setFindings(() => nextFindings);
|
||||
|
||||
// ── Synthesis trigger: completed canonical eligibility transition ──
|
||||
const prevFinding = (findings ?? []).find((f) => f.id === findingId);
|
||||
const previousDisposition = prevFinding?.userDisposition;
|
||||
|
||||
const notRelevantTransition =
|
||||
previousDisposition !== "not_relevant" && newDisposition === "not_relevant";
|
||||
const restoreTransition =
|
||||
previousDisposition === "not_relevant" && newDisposition === null;
|
||||
|
||||
if (!notRelevantTransition && !restoreTransition) return;
|
||||
|
||||
/* ── v0.59a — provenance: eligible evidence set changed ── */
|
||||
setInvestigationRevision((prev) => (prev ?? 0) + 1);
|
||||
|
||||
const currentGraph = result?.situationGraph;
|
||||
if (!currentGraph) return;
|
||||
|
||||
void synthesizeFromFindings(fetch, {
|
||||
situationGraph: currentGraph,
|
||||
findings: normalizeFindings(nextFindings),
|
||||
}).then((res) => {
|
||||
if (res.ok && res.data?.currentUnderstanding) {
|
||||
setCurrentUnderstanding(res.data.currentUnderstanding);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
function updateFindingProposition(findingId, newProposition) {
|
||||
setFindings((prev) =>
|
||||
prev.map((f) =>
|
||||
f.id === findingId
|
||||
? { ...f, proposition: newProposition, userDisposition: null }
|
||||
: f,
|
||||
),
|
||||
// Derive explicit next state — not a React-state reread.
|
||||
const nextFindings = (findings ?? []).map((f) =>
|
||||
f.id === findingId ? { ...f, proposition: newProposition, userDisposition: null } : f,
|
||||
);
|
||||
|
||||
/* ── v0.59a — provenance: no-op guard ── */
|
||||
const prevFinding = (findings ?? []).find((f) => f.id === findingId);
|
||||
if (prevFinding?.proposition === newProposition) return; // no semantic change
|
||||
|
||||
setFindings(() => nextFindings);
|
||||
|
||||
/* ── v0.59a — provenance: corrected Finding changes evidence ── */
|
||||
setInvestigationRevision((prev) => (prev ?? 0) + 1);
|
||||
|
||||
// ── Synthesis trigger: corrected Finding → one reconstruction ──
|
||||
const currentGraph = result?.situationGraph;
|
||||
if (!currentGraph) return;
|
||||
|
||||
void synthesizeFromFindings(fetch, {
|
||||
situationGraph: currentGraph,
|
||||
findings: normalizeFindings(nextFindings),
|
||||
}).then((res) => {
|
||||
if (res.ok && res.data?.currentUnderstanding) {
|
||||
setCurrentUnderstanding(res.data.currentUnderstanding);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* v0.49 promotion seam — deterministic Current Understanding update
|
||||
* triggered by "Done for now" activity boundary (no case/update, no LLM).
|
||||
* Authoritative graph reconsideration triggered by "Done for now"
|
||||
* activity boundary. Delegates to the exported executeEpisodeDone pipeline.
|
||||
*/
|
||||
function handleDoneForNowPromotion(targetNodeId) {
|
||||
if (!targetNodeId || !findings?.length) return;
|
||||
async function handleDoneForNowPromotion(targetNodeId, onImmediateGraphUpdate) {
|
||||
if (!targetNodeId) return;
|
||||
|
||||
// Filter eligible findings for this specific target only.
|
||||
const eligible = findings.filter(
|
||||
(f) => f.originatingTargetNodeId === targetNodeId && (f.userDisposition === null || f.userDisposition === "agree"),
|
||||
// Gate: only invoke episode processing when the active target has focused contributions.
|
||||
// Scenario-wide findings no longer determine whether an empty target enters episode processing.
|
||||
const hasActiveTargetContent = (focusedContributions ?? []).some(
|
||||
(c) => c.targetNodeId === targetNodeId || c.originatingTargetNodeId === targetNodeId,
|
||||
);
|
||||
if (!hasActiveTargetContent) return;
|
||||
|
||||
if (eligible.length === 0) return;
|
||||
// In-flight guard: exactly-once enforcement
|
||||
if (doneInProgressRef.current) return;
|
||||
doneInProgressRef.current = true;
|
||||
|
||||
// Determine the base: use currentUnderstanding if available, else empty string.
|
||||
const baseSummary = currentUnderstanding ?? "";
|
||||
|
||||
// Deterministic producer — no LLM, no API.
|
||||
const newSummary = produceFindingInformedSummary(baseSummary, eligible);
|
||||
|
||||
// Idempotence guard: skip if summary is unchanged (no new eligible findings
|
||||
// beyond what's already in the current Evidence block).
|
||||
if (newSummary === baseSummary) return;
|
||||
|
||||
// Avoid duplicate evidence propositions from repeated promotion.
|
||||
const existingEvidenceMatch = baseSummary.match(/Evidence:\s*\[([^\]]+)\]/);
|
||||
let isDuplicate = false;
|
||||
if (existingEvidenceMatch) {
|
||||
const existingTexts = existingEvidenceMatch[1].split("; ").map((t) => t.trim());
|
||||
isDuplicate = eligible.every((f) => existingTexts.includes(f.proposition));
|
||||
/* ── Immediate client transition — before awaiting async work ── */
|
||||
const preDoneGraph = result?.situationGraph;
|
||||
if (preDoneGraph && onImmediateGraphUpdate) {
|
||||
const immediateResolvedIds = new Set(preDoneGraph.resolvedNodeIds || []);
|
||||
immediateResolvedIds.add(targetNodeId);
|
||||
const immediateGraph = {
|
||||
...preDoneGraph,
|
||||
resolvedNodeIds: Array.from(immediateResolvedIds),
|
||||
};
|
||||
onImmediateGraphUpdate(immediateGraph);
|
||||
}
|
||||
if (isDuplicate) return;
|
||||
|
||||
// Mutate the SAME summary/state that autosave already persists.
|
||||
setCurrentUnderstanding(newSummary);
|
||||
setCuSynthesisLoading(true);
|
||||
try {
|
||||
const doneResult = await executeEpisodeDone({
|
||||
resultSituationGraph: preDoneGraph,
|
||||
targetNodeId,
|
||||
focusedContributions: focusedContributions ?? [],
|
||||
findings,
|
||||
episodeDoneServer: (payload) =>
|
||||
fetch("/api/cases/update", {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({ ...payload, episodeMode: true }),
|
||||
}).then((res) => res.json()),
|
||||
synthesizeFn: (graph, fn) => synthesizeFromFindings(fetch, { situationGraph: graph, findings: fn }),
|
||||
setResult,
|
||||
});
|
||||
|
||||
/* ── v0.59a — provenance: episode done is meaningful evidence change ── */
|
||||
const nextRev = (investigationRevision ?? 0) + 1;
|
||||
setInvestigationRevision(nextRev);
|
||||
|
||||
/* CU synthesis — install only on success */
|
||||
if (doneResult?.synthesisResult?.ok && doneResult.synthesisResult.data?.currentUnderstanding) {
|
||||
setCurrentUnderstanding(doneResult.synthesisResult.data.currentUnderstanding);
|
||||
}
|
||||
/* On synthesis failure: KEEP nextGraph, KEEP Findings, KEEP existing CU. Do NOT rollback. */
|
||||
|
||||
} finally {
|
||||
doneInProgressRef.current = false;
|
||||
setCuSynthesisLoading(false);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* v0.54b/v0.55 — request investigation overview via the established POST /api/cases/overview seam.
|
||||
* Produces a distinct Investigation Report: a derived artefact, not canonical reasoning state.
|
||||
*/
|
||||
async function handleRequestOverview() {
|
||||
if (overviewLoading || !result?.situationGraph) return;
|
||||
|
||||
setOverviewLoading(true);
|
||||
setOverviewState(null); // clear any previous overview before new request
|
||||
|
||||
const plausibleInput = (result.situationGraph?.reconstruction || {}).plausibleInterpretations ?? [];
|
||||
const hasPlausibleInput = Array.isArray(plausibleInput) && plausibleInput.length > 0;
|
||||
|
||||
try {
|
||||
const res = await fetch("/api/cases/overview", {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
situationGraph: result.situationGraph,
|
||||
findings,
|
||||
plausibleInterpretations: plausibleInput,
|
||||
}),
|
||||
}).then((r) => r.json());
|
||||
|
||||
if (res?.success && res?.understanding != null) {
|
||||
setOverviewState(res);
|
||||
|
||||
// Persist as a derived artefact of this investigation
|
||||
const rev = investigationRevision ?? 0;
|
||||
const report = {
|
||||
understanding: res.understanding,
|
||||
plausibleInterpretations: hasPlausibleInput ? res.plausibleInterpretations ?? "" : "",
|
||||
hasPlausibleInterpretations: hasPlausibleInput,
|
||||
generatedFromRevision: rev,
|
||||
};
|
||||
setInvestigationReport(report);
|
||||
|
||||
// Trigger autosave to persist the report
|
||||
void saveInvestigation({
|
||||
id: investigationId,
|
||||
scenario,
|
||||
situationGraph: result.situationGraph,
|
||||
selectedQuestion: result.selectedQuestion,
|
||||
summary: currentUnderstanding,
|
||||
updatedAt: new Date().toISOString(),
|
||||
focusedContributions,
|
||||
findings,
|
||||
investigationReport: report,
|
||||
investigationRevision: rev,
|
||||
});
|
||||
}
|
||||
// On failure: do not clear existing CU, do not block further attempts
|
||||
} finally {
|
||||
setOverviewLoading(false);
|
||||
}
|
||||
}
|
||||
|
||||
function appendFocusedContribution(contribution) {
|
||||
@@ -320,6 +542,29 @@ export default function ScenarioForm() {
|
||||
|
||||
return [...prev, storedContribution];
|
||||
});
|
||||
|
||||
// ── Synthesis trigger: once per completed Finding transition ──
|
||||
const newFindingsDelta = deriveFindingsFromContributions([
|
||||
{
|
||||
...contribution,
|
||||
sequence: (focusedContributions?.length ?? 0) + 1,
|
||||
id: `contrib-${String((focusedContributions?.length ?? 0) + 1).padStart(4, "0")}`,
|
||||
},
|
||||
]).findings;
|
||||
|
||||
if (newFindingsDelta.length === 0) return;
|
||||
|
||||
const currentGraph = result?.situationGraph;
|
||||
if (!currentGraph) return;
|
||||
|
||||
void synthesizeFromFindings(fetch, {
|
||||
situationGraph: currentGraph,
|
||||
findings: normalizeFindings([...(findings ?? []), ...newFindingsDelta]),
|
||||
}).then((res) => {
|
||||
if (res.ok && res.data?.currentUnderstanding) {
|
||||
setCurrentUnderstanding(res.data.currentUnderstanding);
|
||||
}
|
||||
});
|
||||
}
|
||||
const textareaRef = useRef(null);
|
||||
|
||||
@@ -331,8 +576,11 @@ export default function ScenarioForm() {
|
||||
/* Restore persisted session on mount ─────────── */
|
||||
useEffect(() => {
|
||||
if (typeof window === "undefined") return;
|
||||
const saved = loadInvestigation();
|
||||
if (!saved) return;
|
||||
let active = true;
|
||||
(async () => {
|
||||
try {
|
||||
const saved = investigationId ? await loadInvestigation(investigationId) : null;
|
||||
if (!active || !saved) return;
|
||||
|
||||
const hasGraph = Boolean(saved.situationGraph);
|
||||
|
||||
@@ -342,13 +590,26 @@ export default function ScenarioForm() {
|
||||
setFocusedContributions(saved.focusedContributions || []);
|
||||
setFindings(saved.findings || []);
|
||||
|
||||
/* ── v0.55 — hydrate persisted investigation report ─── */
|
||||
if (saved.investigationReport) {
|
||||
setInvestigationReport(saved.investigationReport);
|
||||
}
|
||||
|
||||
/* ── v0.59a — hydrate provenance revision ─────────── */
|
||||
setInvestigationRevision(saved.investigationRevision ?? 0);
|
||||
|
||||
// Partial sessions (present but no graph) must NOT suppress the
|
||||
// scenario-entry form. Only promote to success when there is actual
|
||||
// investigation data to render.
|
||||
if (hasGraph) {
|
||||
setStatus("success");
|
||||
}
|
||||
}, []);
|
||||
} finally {
|
||||
if (active) setHydrated(true);
|
||||
}
|
||||
})();
|
||||
return () => { active = false; };
|
||||
}, [investigationId]);
|
||||
|
||||
/* ── Canonical autosave — persist whenever state changes (Phase 2) ── */
|
||||
|
||||
@@ -357,9 +618,10 @@ export default function ScenarioForm() {
|
||||
// Guard: no valid investigation yet → skip autosave during idle/start flows.
|
||||
// Also prevents overwriting an existing saved investigation with the initial
|
||||
// empty state of a fresh ScenarioForm instance (hydration race guard).
|
||||
if (!result?.situationGraph) return;
|
||||
if (!hydrated || !result?.situationGraph) return;
|
||||
|
||||
void saveInvestigation({
|
||||
persist({
|
||||
id: investigationId,
|
||||
scenario,
|
||||
situationGraph: result.situationGraph,
|
||||
selectedQuestion: result.selectedQuestion,
|
||||
@@ -367,6 +629,8 @@ export default function ScenarioForm() {
|
||||
updatedAt: new Date().toISOString(),
|
||||
focusedContributions,
|
||||
findings,
|
||||
investigationReport,
|
||||
investigationRevision,
|
||||
});
|
||||
}, [
|
||||
scenario,
|
||||
@@ -375,6 +639,8 @@ export default function ScenarioForm() {
|
||||
currentUnderstanding,
|
||||
focusedContributions,
|
||||
findings,
|
||||
investigationReport,
|
||||
investigationRevision, hydrated,
|
||||
]);
|
||||
|
||||
/* Restore facilitator dismiss preference (Experiment 05) ─── */
|
||||
@@ -428,8 +694,7 @@ export default function ScenarioForm() {
|
||||
updateStatus === "loading"
|
||||
);
|
||||
|
||||
const handleSubmit = async (e) => {
|
||||
e.preventDefault();
|
||||
const handleStart = async () => {
|
||||
setStatus("loading");
|
||||
setResult(null);
|
||||
setAnswer("");
|
||||
@@ -452,7 +717,11 @@ export default function ScenarioForm() {
|
||||
setCurrentUnderstanding(data.summary ?? null);
|
||||
const normalised = normaliseStartResult(data);
|
||||
setResult(normalised);
|
||||
saveInvestigation({ scenario, situationGraph: normalised.situationGraph, selectedQuestion: normalised.selectedQuestion, summary: data.summary ?? null, updatedAt: new Date().toISOString(), focusedContributions, findings: [] });
|
||||
/* ── v0.59a — provenance: first meaningful change sets revision to 1 ── */
|
||||
setInvestigationRevision(1);
|
||||
persist({ id: investigationId, scenario, situationGraph: normalised.situationGraph, selectedQuestion: normalised.selectedQuestion, summary: data.summary ?? null, updatedAt: new Date().toISOString(), focusedContributions, findings: [], investigationReport, investigationRevision: 1 });
|
||||
} else if (isUnavailableStartResponse(res, data)) {
|
||||
setStatus("unavailable");
|
||||
} else {
|
||||
setStatus("error");
|
||||
setCurrentUnderstanding(data.summary ?? null);
|
||||
@@ -464,6 +733,11 @@ export default function ScenarioForm() {
|
||||
}
|
||||
};
|
||||
|
||||
const handleSubmit = (e) => {
|
||||
e.preventDefault();
|
||||
void handleStart();
|
||||
};
|
||||
|
||||
const handleUpdate = async (e) => {
|
||||
e.preventDefault();
|
||||
|
||||
@@ -498,23 +772,21 @@ export default function ScenarioForm() {
|
||||
const outcome = submission.data;
|
||||
|
||||
if (submission.ok && outcome.success) {
|
||||
// Merge server-returned findings with local state
|
||||
let newFindings = [...findings];
|
||||
// ── Derive explicit next canonical state (no React-state reread) ──
|
||||
const nextGraph = outcome.updatedSituationGraph;
|
||||
let nextFindings = [...findings];
|
||||
if (outcome.appendedFindings && Array.isArray(outcome.appendedFindings)) {
|
||||
newFindings = [...newFindings, ...outcome.appendedFindings];
|
||||
nextFindings = [...nextFindings, ...outcome.appendedFindings];
|
||||
}
|
||||
|
||||
setUpdateStatus("success");
|
||||
setCurrentUnderstanding(
|
||||
outcome.summary ? outcome.summary : currentUnderstanding,
|
||||
);
|
||||
setUpdateResult({
|
||||
...outcome,
|
||||
previousSituationGraph: result?.situationGraph ?? null,
|
||||
});
|
||||
setResult((current) => ({
|
||||
...current,
|
||||
situationGraph: outcome.updatedSituationGraph,
|
||||
situationGraph: nextGraph,
|
||||
selectedQuestion: normaliseUpdateSelectedQuestion(
|
||||
outcome.selectedQuestion,
|
||||
),
|
||||
@@ -523,9 +795,25 @@ export default function ScenarioForm() {
|
||||
.map((node) => node.id),
|
||||
diagnostics: outcome.diagnostics,
|
||||
}));
|
||||
setFindings(nextFindings);
|
||||
|
||||
// ── Coalesced transition: one synthesis per successful update ──
|
||||
void synthesizeFromFindings(fetch, {
|
||||
situationGraph: nextGraph,
|
||||
findings: normalizeFindings(nextFindings),
|
||||
}).then((res) => {
|
||||
if (res.ok && res.data?.currentUnderstanding) {
|
||||
setCurrentUnderstanding(res.data.currentUnderstanding);
|
||||
}
|
||||
// On synthesis failure: graph/Findings already persisted, CU preserved, no retry.
|
||||
});
|
||||
|
||||
setAnswer("");
|
||||
// Persist after successful update turn — include findings
|
||||
saveInvestigation({ scenario, situationGraph: outcome.updatedSituationGraph, selectedQuestion: normaliseUpdateSelectedQuestion(outcome.selectedQuestion), summary: outcome.summary ?? currentUnderstanding, updatedAt: new Date().toISOString(), focusedContributions, findings: newFindings });
|
||||
// Persist after successful update turn — include explicit next state
|
||||
/* ── v0.59a — provenance: meaningful change advances revision ── */
|
||||
const nextRev = (investigationRevision ?? 0) + 1;
|
||||
setInvestigationRevision(nextRev);
|
||||
persist({ id: investigationId, scenario, situationGraph: nextGraph, selectedQuestion: normaliseUpdateSelectedQuestion(outcome.selectedQuestion), summary: currentUnderstanding, updatedAt: new Date().toISOString(), focusedContributions, findings: nextFindings, investigationReport, investigationRevision: nextRev });
|
||||
} else {
|
||||
setUpdateStatus("error");
|
||||
setUpdateError(outcome);
|
||||
@@ -539,7 +827,7 @@ export default function ScenarioForm() {
|
||||
return (
|
||||
<div className="space-y-6">
|
||||
{/* ── Idle form for scenario input ─ */}
|
||||
{!result?.situationGraph && status === "idle" && (
|
||||
{!result?.situationGraph && (status === "idle" || status === "unavailable") && (
|
||||
<form onSubmit={handleSubmit} className="space-y-6">
|
||||
|
||||
{/* Two-column landing workspace */}
|
||||
@@ -604,6 +892,7 @@ export default function ScenarioForm() {
|
||||
<p className="mt-3 text-xs italic text-gray-400">
|
||||
You do not need all the answers yet.
|
||||
</p>
|
||||
{status === "unavailable" && <UnavailableStartPanel onRetry={handleStart} />}
|
||||
</div>
|
||||
|
||||
</div>
|
||||
@@ -668,7 +957,14 @@ export default function ScenarioForm() {
|
||||
scenario={scenario}
|
||||
status={status}
|
||||
updateStatus={updateStatus}
|
||||
cuSynthesisLoading={cuSynthesisLoading}
|
||||
currentUnderstanding={currentUnderstanding}
|
||||
/* ── v0.54b — investigation overview transient state ─── */
|
||||
overviewState={overviewState}
|
||||
setOverviewState={setOverviewState}
|
||||
overviewLoading={overviewLoading}
|
||||
handleRequestOverview={handleRequestOverview}
|
||||
onNavigateToReport={onNavigateToReport}
|
||||
result={{
|
||||
...(result || {}),
|
||||
situationGraph: updateResult?.updatedSituationGraph ?? result?.situationGraph,
|
||||
@@ -688,8 +984,18 @@ export default function ScenarioForm() {
|
||||
onUpdateFindingProposition={updateFindingProposition}
|
||||
/* ── v0.49 — done-for-now promotion seam ─────────── */
|
||||
onSummaryUpdate={handleDoneForNowPromotion}
|
||||
/* ── immediate graph transition (Done acknowledged before async) ── */
|
||||
onImmediateGraphChange={(nextGraph) => setResult((prev) => ({ ...(prev ?? {}), situationGraph: nextGraph }))}
|
||||
/* ── v0.59a — provenance tracking ─────────────────── */
|
||||
investigationRevision={investigationRevision}
|
||||
onSituationGraphChange={(nextGraph) => {
|
||||
const nextRev = (investigationRevision ?? 0) + 1;
|
||||
setInvestigationRevision(nextRev);
|
||||
setResult((prev) => ({ ...(prev ?? {}), situationGraph: nextGraph }));
|
||||
}}
|
||||
onRestart={() => {
|
||||
clearInvestigation();
|
||||
restartPersistedInvestigation();
|
||||
setInvestigationRevision(0);
|
||||
setStatus("idle");
|
||||
setResult(null);
|
||||
setAnswer("");
|
||||
@@ -709,7 +1015,7 @@ export default function ScenarioForm() {
|
||||
|
||||
{/* ── Continue later banner when session was restored ── */}
|
||||
{status === "success" && result?.updatedAt && (
|
||||
<ContinueLaterBanner onRestart={() => { clearInvestigation(); setStatus("idle"); setResult(null); setAnswer(""); setUpdateStatus("idle"); setCurrentUnderstanding(null); setFocusedContributions([]); setFindings([]); }} />
|
||||
<ContinueLaterBanner onRestart={() => { restartPersistedInvestigation(); setInvestigationRevision(0); setStatus("idle"); setResult(null); setAnswer(""); setUpdateStatus("idle"); setCurrentUnderstanding(null); setFocusedContributions([]); setFindings([]); }} />
|
||||
)}
|
||||
|
||||
{/* Reset button after successful analysis */}
|
||||
@@ -717,7 +1023,8 @@ export default function ScenarioForm() {
|
||||
<div className="text-center">
|
||||
<button
|
||||
onClick={() => {
|
||||
clearInvestigation();
|
||||
restartPersistedInvestigation();
|
||||
setInvestigationRevision(0);
|
||||
setScenario("");
|
||||
setStatus("idle");
|
||||
setResult(null);
|
||||
|
||||
@@ -0,0 +1,47 @@
|
||||
"use client";
|
||||
|
||||
import { useEffect, useState } from "react";
|
||||
import {
|
||||
readThemePreference,
|
||||
resolveTheme,
|
||||
saveThemePreference,
|
||||
toggleTheme,
|
||||
} from "@/lib/theme-preference.js";
|
||||
|
||||
function applyTheme(theme) {
|
||||
document.documentElement.dataset.theme = theme;
|
||||
document.documentElement.style.colorScheme = theme;
|
||||
}
|
||||
|
||||
export default function ThemeToggle() {
|
||||
const [theme, setTheme] = useState("light");
|
||||
|
||||
useEffect(() => {
|
||||
const nextTheme = resolveTheme({
|
||||
savedTheme: readThemePreference(window.localStorage),
|
||||
systemPrefersDark: window.matchMedia?.("(prefers-color-scheme: dark)").matches,
|
||||
});
|
||||
setTheme(nextTheme);
|
||||
applyTheme(nextTheme);
|
||||
}, []);
|
||||
|
||||
const switchTheme = () => {
|
||||
const nextTheme = toggleTheme(theme);
|
||||
setTheme(nextTheme);
|
||||
saveThemePreference(nextTheme, window.localStorage);
|
||||
applyTheme(nextTheme);
|
||||
};
|
||||
|
||||
const isDark = theme === "dark";
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
onClick={switchTheme}
|
||||
aria-label={isDark ? "Switch to light mode" : "Switch to dark mode"}
|
||||
className="theme-toggle rounded-lg border px-3 py-2 text-sm font-medium transition focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-teal-500 focus-visible:ring-offset-2"
|
||||
>
|
||||
<span aria-hidden="true" className="mr-1.5">{isDark ? "☀" : "☾"}</span>
|
||||
{isDark ? "Light mode" : "Dark mode"}
|
||||
</button>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
# Confidence Engine — Deploy Environment (example)
|
||||
#
|
||||
# Copy to /opt/confidence-engine/deploy.env on CT 112.
|
||||
# This file is NOT tracked by Git. It is owned/administered on the host.
|
||||
#
|
||||
# Build-time (baked into Docker image via --build-arg):
|
||||
NEXT_PUBLIC_SUPABASE_URL=https://supabase.rdbcloud.co.uk
|
||||
NEXT_PUBLIC_SUPABASE_ANON_KEY=replace-with-supabase-anon-key
|
||||
|
||||
# Runtime (passed to container at start):
|
||||
OLLAMA_BASE_URL=http://192.168.x.x:11434
|
||||
OLLAMA_MODEL=replace-with-model-name
|
||||
@@ -0,0 +1,284 @@
|
||||
# Design Evolution Archive — Tranche 1
|
||||
|
||||
This directory contains exact historical extracts from
|
||||
`docs/design-evolution-log.md`.
|
||||
|
||||
The original monolithic log remains intact and authoritative while the archive refactor is in progress.
|
||||
|
||||
These files are historical provenance, not current product truth.
|
||||
|
||||
For current product state use:
|
||||
- docs/current-handoff.md
|
||||
- docs/current-project-state.md
|
||||
|
||||
## Verified extracts
|
||||
|
||||
### Chapter 1
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch1/experiments-01-to-06-and-phases-1-to-4.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 1–352
|
||||
|
||||
Contents:
|
||||
Phases 1–4 and Experiments 01–06.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 2
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch2/early-reasoning-and-provenance-discovery.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 354–1122
|
||||
|
||||
Contents:
|
||||
Experiments 07–20 and associated emerging-direction material.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 3
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch3/phase-transition-and-emerging-directions.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 1124–1216
|
||||
|
||||
Contents:
|
||||
Phase Transition, Graph as Source of Truth, and Experiments 21–22.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 4
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch4/passive-classifiers-and-explore-contract-validation.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 1218–1501
|
||||
|
||||
Contents:
|
||||
Experiments 23–25B and their closeout/return-to-work material.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
## Tranche 2
|
||||
|
||||
### Chapter 5
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch5/context-inventory-routing-and-handoff-infrastructure.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 1502–2052
|
||||
|
||||
Contents:
|
||||
Experiments 26–34 — context routing inventory and handoff infrastructure design.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 6
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch6/handoff-validation-behavior-selection-and-assessor-audit.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 2053–2503
|
||||
|
||||
Contents:
|
||||
Experiments 35–41 and Experiment 41 Conclusion — handoff validation, behavior selection, and assessor audit.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
Tranches 1, 2, 3, and 4 have now been extracted.
|
||||
The original monolithic log remains intact and authoritative while extraction is incomplete.
|
||||
|
||||
## Tranche 3
|
||||
|
||||
### Chapter 7
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch7/experiments-42-to-46.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 2504–3069
|
||||
|
||||
Contents:
|
||||
Experiments 42–46 — narrow acknowledge exclusion, Clarify readiness audit, assessor unclear-starting-point, and "Too Broad" boundary investigations.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 8
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch8/experiments-47-to-51.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 3070–3534
|
||||
|
||||
Contents:
|
||||
Experiments 47–51 — shared-anchor coherence diagnostic, unknown relationship audit, production shared-anchor test, and semantic decision-relevance coherence experiments.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
## Tranche 4
|
||||
|
||||
### Chapter 9
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch9/experiments-52-to-52i.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 3535–4980
|
||||
|
||||
Contents:
|
||||
Experiments 52–52I and Correction to Experiment 52B Conclusion — semantic interpretation generalisation, evaluation using existing Qwen model, separate meaning from relevance labels, normalisation into decision-relevance contract, supports_decision/could_change_decision boundary coherence, ambiguity normalisation failures, ambiguous meaning filled with domain expectations, ambiguity wording resistance, grounding rule for relationship strength invention, and provenance boundary analysis.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
## Tranche 5
|
||||
|
||||
### Chapter 10
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch10/experiments-54a-to-54k-provenance-audit.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 4981–6423
|
||||
|
||||
Contents:
|
||||
Experiments 54A–54K — complete provenance audit chain including graph provenance verification, trace through pipeline, update flow determinism, prompt-level provenance analysis, evidence reference integrity, evidenceType reliability, deterministic source linkage, pre-LLM source identity establishment, multi-interpretation source anchoring, meaning separation capability, and automated semantic grounding with live model inference.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
Tranches 1 through 6 have now been extracted.
|
||||
The original monolithic log remains intact and authoritative while extraction is incomplete.
|
||||
|
||||
## Tranche 6
|
||||
|
||||
### Chapter 11
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch11/experiments-54l-to-54q-grounding-and-evidence-stability.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 6425–7824
|
||||
|
||||
Contents:
|
||||
Experiments 54L–54Q — semantic grounding stability under repetition, disagreement exposure between interpretations, disagreement-driven information need changes, evidence-need discrimination for same-goal scenarios, explicit evidence needs recovering higher-level consequences, and structured semantic steps preserving evidence distinction with consequence.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 12
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch12/experiment-54r-clarification-requires-source.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 7825–8012
|
||||
|
||||
Contents:
|
||||
Experiment 54R — testing whether models correctly distinguish disagreements resolvable through evidence from those requiring user input, across competing causal hypotheses, ambiguous user priority, and absent material disagreement patterns.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
## Tranche 7
|
||||
|
||||
### Chapter 13
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch13/54S-54V-clarification-target-and-answer-resolution.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 8013–8635
|
||||
|
||||
Contents:
|
||||
Experiments 54S–54V — clarification target identification, no-clarification-stability across repeated inputs, whether a clarification target becomes a useful user question without adding new meaning, and whether a clarification answer can resolve only the target.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 14
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch14/54W-54Z-clarification-chain-and-target-broadening.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 8636–9294
|
||||
|
||||
Contents:
|
||||
Experiments 54W–54Z — clarification chain end-to-end integrity, specificity loss under chaining, target broadening effects on question and resolution, and resolution differences when the answer is less explicit.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
## Tranche 8
|
||||
|
||||
### Chapter 15
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch15/55A-preserve-uncertainty-from-weak-clarification-answers.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 9296–9529
|
||||
|
||||
Contents:
|
||||
Experiment 55A — Can the Model Preserve Uncertainty From Weak Clarification Answers? Testing fully explicit hard constraint, weak priority statement, conditional trade-off, and non-answer/insufficient clarification cases.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 16
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch16/55B-separate-answer-meaning-from-resolution-judgement.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 9530–9748
|
||||
|
||||
Contents:
|
||||
Experiment 55B — Separate Answer Meaning from Resolution Judgement. Mode A instruction/output contract versus Mode B inference separation, tested on weak priority, conditional trade-off, and non-answer cases.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 17
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch17/55C-resolution-from-preserved-answer-meaning.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 9749–9982
|
||||
|
||||
Contents:
|
||||
Experiment 55C — Resolution From Preserved Answer Meaning. Two-stage resolution using preserved meaning rather than raw answer, tested on weak priority, conditional trade-off, and non-answer cases.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
### Chapter 18
|
||||
Path:
|
||||
docs/archive/experiments/vol-1-chapters/ch18/55D-separate-stated-vs-inferred-meaning-through-v058-provenance.md
|
||||
|
||||
Original source:
|
||||
docs/design-evolution-log.md lines 9983–10322
|
||||
|
||||
Contents:
|
||||
Experiments 55D (separate stated clarification meaning from inference), 55E (reasoning refinement requirements synthesis), 55F (reasoning requirements production path map), and v0.51–v0.58 Progress — product provenance and architectural decisions.
|
||||
|
||||
Fidelity:
|
||||
Exact contiguous copy.
|
||||
|
||||
The extraction phase is complete.
|
||||
|
||||
Tranches 1 through 8 now preserve the substantive historical contents of
|
||||
`docs/design-evolution-log.md` as exact chapter extracts.
|
||||
|
||||
The original monolithic log remains intact and authoritative until a separate
|
||||
compatibility/routing task replaces it with an index or pointer and updates repository references.
|
||||
|
||||
## Refactor status
|
||||
|
||||
Do not remove these ranges from `docs/design-evolution-log.md` yet.
|
||||
|
||||
Do not update repository-wide references to point here yet.
|
||||
|
||||
The original monolith remains authoritative pending migration.
|
||||
+352
@@ -0,0 +1,352 @@
|
||||
# Design Evolution Log
|
||||
|
||||
A chronological record of why significant design decisions were made. This is NOT a changelog. It records the product's evolution of thinking.
|
||||
|
||||
This document records discoveries, not decisions. Every entry represents our best understanding at that point in time and may later be superseded by a better model.
|
||||
|
||||
---
|
||||
|
||||
## Phase 1
|
||||
|
||||
Simple conversational investigation
|
||||
|
||||
Question → Answer interaction.
|
||||
|
||||
Purpose:
|
||||
Prove the reasoning loop.
|
||||
|
||||
Learning:
|
||||
Conversation alone does not provide sufficient context during longer investigations.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2
|
||||
|
||||
Persistent investigation notebook
|
||||
|
||||
Added:
|
||||
|
||||
- current understanding
|
||||
- original situation
|
||||
- investigation history
|
||||
|
||||
Learning:
|
||||
Users need persistent context rather than remembering previous answers.
|
||||
|
||||
---
|
||||
|
||||
## Phase 3
|
||||
|
||||
Document workspace
|
||||
|
||||
Created a coherent workspace with:
|
||||
|
||||
- investigation status
|
||||
- current investigation
|
||||
- response
|
||||
- understanding
|
||||
- investigation map placeholder
|
||||
- situation
|
||||
- history
|
||||
|
||||
Learning:
|
||||
The interface became usable but still behaved like a document rather than a workspace.
|
||||
|
||||
---
|
||||
|
||||
## Phase 4 (Current Exploration)
|
||||
|
||||
Facilitated Investigation Workshop
|
||||
|
||||
Status:
|
||||
Experimental.
|
||||
|
||||
Hypothesis:
|
||||
|
||||
The Confidence Engine is not:
|
||||
|
||||
- a chatbot
|
||||
- a dashboard
|
||||
- a form
|
||||
|
||||
It is a facilitated investigation workspace.
|
||||
|
||||
The interface should resemble the environment in which structured thinking happens.
|
||||
|
||||
Record discoveries rather than conclusions.
|
||||
|
||||
Leave room for future phases.
|
||||
|
||||
---
|
||||
|
||||
## Phase 4 — Guiding Principles
|
||||
|
||||
The Confidence Engine is a workspace, not a document.
|
||||
|
||||
People think in multiple directions simultaneously.
|
||||
|
||||
Useful context should be visible together.
|
||||
|
||||
The interface should favour thinking over scrolling.
|
||||
|
||||
The workspace should feel like a large desk or workshop rather than a narrow report.
|
||||
|
||||
The engine facilitates thinking.
|
||||
|
||||
The user contributes evidence.
|
||||
|
||||
The workspace captures shared understanding.
|
||||
|
||||
### Experiment 01 — Wider canvas
|
||||
|
||||
Hypothesis:
|
||||
The document-like feeling is caused partly by the narrow outer container.
|
||||
|
||||
Change:
|
||||
Increase the available desktop workspace width without rearranging any components.
|
||||
|
||||
Result:
|
||||
Confirmed.
|
||||
|
||||
Learning:
|
||||
Increasing the outer workspace width reduced the narrow-document feeling and made better use of large displays.
|
||||
|
||||
Unexpected learning:
|
||||
Width alone did not create a workshop. The wider canvas exposed that the interface still behaves as a collection of independent cards, with supporting artefacts unsure how to use the available space.
|
||||
|
||||
Decision:
|
||||
Keep the wider desktop canvas.
|
||||
|
||||
Next question:
|
||||
Can grouping the interface into cognitive work zones make the wider canvas feel like a coherent investigation surface?
|
||||
|
||||
### Experiment 02 — Cognitive work zones
|
||||
|
||||
Hypothesis:
|
||||
A workspace organised around what the investigator is doing will feel more coherent than one organised around equal cards or equal columns.
|
||||
|
||||
Result:
|
||||
Partially confirmed.
|
||||
|
||||
Learning:
|
||||
|
||||
The workspace feels more coherent when organised into cognitive work zones rather than a simple document stack.
|
||||
|
||||
However, another distinction emerged that is more important than the zones themselves.
|
||||
|
||||
The interface naturally separates into two different modes:
|
||||
|
||||
• the active conversation between investigator and facilitator
|
||||
|
||||
and
|
||||
|
||||
• the shared workspace describing the current understanding.
|
||||
|
||||
Unexpected learning:
|
||||
|
||||
History feels incorrect when treated as reference information.
|
||||
|
||||
History is actually the continuation of the investigator's conversation.
|
||||
|
||||
Every response immediately becomes history.
|
||||
|
||||
The notebook should therefore grow naturally from the Response area.
|
||||
|
||||
The Investigation Status card currently competes with the Current Investigation card.
|
||||
|
||||
The current question is the primary focus.
|
||||
|
||||
Status is supporting context.
|
||||
|
||||
Decision:
|
||||
|
||||
Keep the cognitive-zone concept.
|
||||
|
||||
Refine the zones around conversational flow instead of card grouping.
|
||||
|
||||
Next question:
|
||||
Can the workspace clearly separate conversation from shared understanding?
|
||||
|
||||
### Experiment 03 — Conversation versus Workspace
|
||||
|
||||
Hypothesis
|
||||
|
||||
Investigators think in two simultaneous modes.
|
||||
|
||||
Mode 1:
|
||||
The conversation.
|
||||
|
||||
Question
|
||||
↓
|
||||
|
||||
Response
|
||||
↓
|
||||
|
||||
History
|
||||
|
||||
Mode 2:
|
||||
The shared workspace.
|
||||
|
||||
Status
|
||||
|
||||
Understanding
|
||||
|
||||
Situation
|
||||
|
||||
Map
|
||||
|
||||
Separating these should make the interface feel more like a facilitated investigation than a collection of cards.
|
||||
|
||||
Evaluation:
|
||||
Partially confirmed.
|
||||
|
||||
Learning:
|
||||
|
||||
The workspace feels more coherent when organised into cognitive zones rather than a simple document stack.
|
||||
|
||||
However, another distinction emerged that is more important than the zones themselves.
|
||||
|
||||
The interface naturally separates into two different modes:
|
||||
|
||||
• the active conversation between investigator and facilitator
|
||||
|
||||
and
|
||||
|
||||
• the shared workspace describing the current understanding.
|
||||
|
||||
Unexpected learning:
|
||||
|
||||
History feels incorrect when treated as reference information.
|
||||
|
||||
History is actually the continuation of the investigator's conversation.
|
||||
|
||||
Every response immediately becomes history.
|
||||
|
||||
The notebook should therefore grow naturally from the Response area.
|
||||
|
||||
The Investigation Status card currently competes with the Current Investigation card.
|
||||
|
||||
The current question is the primary focus.
|
||||
|
||||
Status is supporting context.
|
||||
|
||||
Decision:
|
||||
|
||||
Keep the cognitive-zone concept.
|
||||
|
||||
Refine the zones around conversational flow instead of card grouping.
|
||||
|
||||
Next question:
|
||||
Can the workspace clearly separate conversation from shared understanding?
|
||||
|
||||
### Experiment 04 — Facilitated Workshop Introduction
|
||||
|
||||
Hypothesis
|
||||
|
||||
Beginning with a facilitator-style introduction will create more confidence than presenting an empty workspace.
|
||||
|
||||
Questions
|
||||
|
||||
- Does the interface feel more welcoming?
|
||||
- Does reducing the visual weight of the textarea improve the first experience?
|
||||
- Does separating "starting" from "investigating" feel natural?
|
||||
- Does the transition into the investigation workspace feel meaningful?
|
||||
|
||||
Status:
|
||||
Experimental.
|
||||
|
||||
Result:
|
||||
Partially confirmed.
|
||||
|
||||
Learning:
|
||||
|
||||
The facilitator introduction reduced the intimidation of the first screen.
|
||||
|
||||
Replacing the empty landing page with a guided introduction improved the emotional tone.
|
||||
|
||||
However, stacking the introduction above the input still gives the introduction excessive visual prominence.
|
||||
|
||||
Repeat users may not want to repeatedly read the same introduction.
|
||||
|
||||
Orientation should remain available without dominating the workflow.
|
||||
|
||||
Decision:
|
||||
|
||||
Keep the introduction concept but change its spatial relationship to the workspace — move it from above to beside, making it optional rather than mandatory.
|
||||
|
||||
Next question:
|
||||
Does a horizontal facilitator/workspace layout feel more natural?
|
||||
|
||||
### Experiment 05 — Facilitator Panel and Adaptive Landing Workspace
|
||||
|
||||
Hypothesis
|
||||
|
||||
Placing the facilitator beside the working area will feel more like entering a facilitated workshop than stacking instructional content above the workspace.
|
||||
|
||||
Allowing the user to dismiss the facilitator will reduce friction for returning users while preserving onboarding for new users.
|
||||
|
||||
Questions
|
||||
|
||||
- Does a horizontal facilitator/workspace layout feel more natural?
|
||||
- Does the user's eye move naturally from facilitator to workspace?
|
||||
- Does the workspace become the primary focus?
|
||||
- Does "Don't show again" feel preferable to automatically hiding the introduction?
|
||||
- Should the facilitator panel become an optional workspace companion rather than mandatory onboarding?
|
||||
|
||||
Status:
|
||||
Completed.
|
||||
|
||||
Findings:
|
||||
|
||||
- A horizontal facilitator/workspace arrangement feels more natural than stacked onboarding.
|
||||
- The workspace becomes the visual destination rather than the introduction.
|
||||
- User-controlled dismissal is preferable to automatic hiding.
|
||||
- The facilitator feels useful but visually too passive.
|
||||
- Remaining issues are now visual hierarchy rather than layout architecture.
|
||||
|
||||
### Experiment 06 — Focused Investigation
|
||||
|
||||
Hypothesis
|
||||
|
||||
The interface should gently guide attention towards the current task without hiding supporting information.
|
||||
|
||||
Reducing competition between panels may improve concentration more than introducing additional colour or decoration.
|
||||
|
||||
Questions
|
||||
|
||||
- Does visual emphasis naturally guide the eye?
|
||||
- Can supporting panels become quieter without disappearing?
|
||||
- Does the investigation question become the obvious focal point?
|
||||
- Does the workspace feel calmer?
|
||||
- Are we approaching a professional investigation environment?
|
||||
|
||||
Status:
|
||||
Closed.
|
||||
|
||||
## Result
|
||||
|
||||
Partially confirmed.
|
||||
|
||||
## What did we learn?
|
||||
|
||||
- Stronger visual hierarchy can direct attention without rearranging the interface.
|
||||
- The facilitator briefing became easier to distinguish.
|
||||
- Colour and tint improved separation only modestly.
|
||||
- Meaning must not depend on colour.
|
||||
- Areas and intent should remain distinguishable through structure, spacing, typography, borders, shape and placement.
|
||||
- The initial textarea still implies that the user should provide a detailed report.
|
||||
- The size of an input communicates the amount of information expected.
|
||||
|
||||
## Decision
|
||||
|
||||
Retain the useful hierarchy refinements provisionally.
|
||||
|
||||
Do not increase reliance on colour.
|
||||
|
||||
Defer dark mode and broader palette work.
|
||||
|
||||
The next experiment should test whether a smaller starting input better communicates that the user only needs to provide an initial observation.
|
||||
|
||||
Do not rewrite previous experiments.
|
||||
|
||||
---
|
||||
+1443
File diff suppressed because it is too large
Load Diff
+1400
File diff suppressed because it is too large
Load Diff
+188
@@ -0,0 +1,188 @@
|
||||
## Experiment 54R — Does a Material Disagreement Actually Require User Clarification? (2026-08-07)
|
||||
|
||||
### Objective
|
||||
|
||||
Given an explicit interpretation disagreement and its evidence consequence, test whether the model can distinguish between a disagreement that requires clarification from the user and one that can be resolved by investigating evidence.
|
||||
|
||||
This is passive and test-only. Do not generate the clarification question. Do not generate the next investigation question. Do not choose a winning interpretation. Do not change production behaviour.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The model may be able to distinguish:
|
||||
|
||||
**Evidence-resolvable disagreement:** The user's meaning is sufficiently clear, but competing explanations require different evidence.
|
||||
|
||||
**User-clarification disagreement:** The disagreement concerns the user's intended meaning, priority, constraint, or definition, so external evidence cannot resolve it without asking the user.
|
||||
|
||||
If this distinction works, disagreement does not have to map automatically to clarification.
|
||||
|
||||
### Context Budget
|
||||
|
||||
Read only:
|
||||
- `docs/current-handoff.md`;
|
||||
- Experiment 54Q only in `docs/design-evolution-log.md`;
|
||||
- `tests/reconstruction/semantic-structured-evidence-consequence.test.js`;
|
||||
- `.env.local` only for `OLLAMA_BASE_URL` and `OLLAMA_MODEL`.
|
||||
|
||||
Not read: Behaviour Selection; assessor; graph files; UI; active prompts; question-selection code; full experiment history.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54Q)
|
||||
Model: `qwen-claude:latest` (same as 54Q)
|
||||
|
||||
No localhost fallback. No experiment-specific model variable.
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **3** live Ollama calls — one per case.
|
||||
|
||||
### Input Contract
|
||||
|
||||
Each call receives:
|
||||
```json
|
||||
{
|
||||
"source": "...",
|
||||
"disagreement": ["..."],
|
||||
"evidenceNeeded": ["..."]
|
||||
}
|
||||
```
|
||||
|
||||
The disagreement and evidence needs are fixed human-reviewed inputs. The model does not rediscover them.
|
||||
|
||||
### Output Contract
|
||||
|
||||
Return exactly:
|
||||
```json
|
||||
{
|
||||
"requiresUserClarification": true | false,
|
||||
"reason": "one short sentence"
|
||||
}
|
||||
```
|
||||
|
||||
No question text, no recommended action, no preferred interpretation, no score, no confidence, no behaviour label. This boolean is test-only and is not a production contract.
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
> Decide whether resolving the stated disagreement requires additional meaning, preference, intent, or factual information that only the user can provide. Return true when evidence alone cannot settle the disagreement because the missing distinction belongs to the user's intended meaning, priority, constraint, or private knowledge. Return false when the disagreement can be investigated using external, operational, or observable evidence without asking the user to define what they mean. Do not generate a question and do not choose which interpretation is correct.
|
||||
|
||||
No examples. No expected answers. Same instruction for all three cases.
|
||||
|
||||
### Case 1 — Competing Causes, Evidence Can Resolve
|
||||
|
||||
**Source:** "Orders are arriving late and customers have started complaining."
|
||||
**Disagreement:** delays may be caused by insufficient staff capacity / delays may be caused by unreliable supplier lead times.
|
||||
**Evidence needed:** staffing levels and workload; processing throughput; supplier lead-time history; supplier delivery reliability.
|
||||
**Fixed human reference:** `requiresUserClarification: false`
|
||||
|
||||
### Case 2 — User Priority Is Ambiguous
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Disagreement:** growth should be prioritised even if some additional risk is unavoidable / avoiding additional risk is a hard constraint even if growth is slower.
|
||||
**Evidence needed:** possible growth opportunities; risk exposure of each option.
|
||||
**Fixed human reference:** `requiresUserClarification: true`
|
||||
|
||||
### Case 3 — Same Meaning, No Clarification Need
|
||||
|
||||
**Source:** "Revenue is down. I think pricing may be part of the problem, but I am not sure."
|
||||
**Disagreement:** no substantive disagreement; two interpretations express the same uncertainty in different words.
|
||||
**Evidence needed:** pricing history; sales/conversion response; other plausible revenue drivers.
|
||||
**Fixed human reference:** `requiresUserClarification: false`
|
||||
|
||||
### Results
|
||||
|
||||
#### Case 1 — Competing Causes, Evidence Can Resolve
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"requiresUserClarification": false,
|
||||
"reason": "The disagreement involves operational causes that can be resolved by analyzing objective company data and supplier records."
|
||||
}
|
||||
```
|
||||
|
||||
**Resolution-source classification:** `resolution_source_correct` ✓ — Returned `false`. The model correctly identified that operational evidence (company data, supplier records) can distinguish the competing causal hypotheses without asking the user.
|
||||
|
||||
#### Case 2 — User Priority Is Ambiguous
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"requiresUserClarification": true,
|
||||
"reason": "Resolving this conflict depends on the user's personal priority and risk tolerance, which cannot be determined by external facts alone."
|
||||
}
|
||||
```
|
||||
|
||||
**Resolution-source classification:** `resolution_source_correct` ✓ — Returned `true`. The model correctly identified that user-owned meaning (priority, risk tolerance) is required to resolve the disagreement. External facts alone cannot determine whether "don't want more risk" is a preference, constraint, or boundary.
|
||||
|
||||
#### Case 3 — Same Meaning, No Clarification Need
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"requiresUserClarification": false,
|
||||
"reason": "The stated position explicitly confirms there is no substantive disagreement, so no additional user input is required to resolve it."
|
||||
}
|
||||
```
|
||||
|
||||
**Resolution-source classification:** `resolution_source_correct` ✓ — Returned `false`. The model correctly identified that the absence of material disagreement means clarification is unnecessary. Evidence can investigate the underlying uncertainty.
|
||||
|
||||
### Resolution-Source-Correct Count
|
||||
|
||||
**3 of 3** cases classified correctly.
|
||||
|
||||
### Resolution-Source-Failed Count
|
||||
|
||||
**0 of 3** cases classified incorrectly.
|
||||
|
||||
### Did Case 1 Correctly Recognise That Competing Causal Hypotheses Can Be Resolved Through Evidence?
|
||||
|
||||
Yes. The model returned `false` and provided a reason referencing operational causes resolvable by company data and supplier records — evidence, not user meaning.
|
||||
|
||||
### Did Case 2 Correctly Recognise That the Unresolved Growth-Versus-Risk Priority Belongs to the User?
|
||||
|
||||
Yes. The model returned `true` and identified that resolution depends on "the user's personal priority and risk tolerance," which external facts alone cannot determine.
|
||||
|
||||
### Did Case 3 Avoid Unnecessary Clarification Where There Was No Material Disagreement?
|
||||
|
||||
Yes. The model returned `false`, correctly noting the absence of substantive disagreement makes additional clarification unnecessary.
|
||||
|
||||
### Did the Model Treat Every Disagreement as Requiring User Clarification?
|
||||
|
||||
No. Two of three cases returned `false`. Only Case 2 (ambiguous priority) returned `true`.
|
||||
|
||||
### Did the Model Confuse Missing Evidence with Missing User Meaning?
|
||||
|
||||
No. In Case 1, the model correctly distinguished between lacking evidence to investigate causes (which it flagged as resolvable through evidence gathering) and lacking user meaning (which it did not claim). The reason text referenced "analyzing objective company data and supplier records" rather than requiring user input.
|
||||
|
||||
### Did the Model Generate an Actual Question?
|
||||
|
||||
No. No question was generated in any output. The output contract was respected in all cases.
|
||||
|
||||
### Did the Model Choose a Winner?
|
||||
|
||||
No. No interpretation was selected as correct in any case.
|
||||
|
||||
### Evidence That User-Owned Ambiguity Can Be Separated From Evidence Uncertainty
|
||||
|
||||
Case 2 succeeded where Case 1 and Case 3 both returned `false` for different reasons — one because evidence can resolve it, the other because no disagreement exists. The model's reasons for each case were distinct in their reference points: operational data (Case 1) versus user priority (Case 2) versus absence of disagreement (Case 3). This pattern suggests the model does not collapse all ambiguity into a single clarification need.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Three cases only; limited domain coverage (one delivery scenario, one strategic priority, one revenue statement).
|
||||
- Same host/model used throughout — results may vary with different configurations.
|
||||
- Does not establish generalisation beyond these specific inputs.
|
||||
- The distinction tested here is binary (true/false) and does not test partial or probabilistic resolution-source classification.
|
||||
- No evidence was actually gathered in any case — only whether the *source* of resolution was correctly identified.
|
||||
- The Case 1 evaluator warning (if present) was a false positive from heuristic wording checks, not a semantic failure.
|
||||
|
||||
### Conclusion
|
||||
|
||||
**The model correctly distinguished user-clarification needs from evidence-resolvable disagreement in all tested cases.**
|
||||
|
||||
Across the three tested patterns — competing causal hypotheses, ambiguous user priority, and absent material disagreement — the model returned the correct boolean in every case with semantically appropriate reasoning. No clarification or investigation question was generated. No interpretation was selected as correct. Across the three tested disagreement patterns, the model did not automatically map disagreement to user clarification.
|
||||
|
||||
### Status
|
||||
|
||||
**Pending Rob's review.** No production code changed. No schemas modified. No active engine behaviour changed. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `tests/reconstruction/semantic-disagreement-resolution-source.test.js`.
|
||||
|
||||
+623
@@ -0,0 +1,623 @@
|
||||
## Experiment 54S — Can the Model Identify Exactly What the User Needs to Clarify? (2026-08-07)
|
||||
|
||||
### Objective
|
||||
|
||||
When clarification genuinely requires user input, can the model identify the specific missing user-owned distinction without yet generating the clarification question? This is passive and test-only. Do not generate a question. Do not choose a winning interpretation. Do not change Behaviour Selection. Do not change production behaviour.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
When clarification genuinely belongs to the user, the model may be able to identify the smallest unresolved user-owned distinction. For example, for "I want the business to grow, but I don't want to take on more risk," the missing distinction is not "what are the risks?" but rather "whether avoiding additional risk is a preference or a hard constraint."
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54R)
|
||||
Model: `qwen-claude:latest` (same as 54R)
|
||||
|
||||
No localhost fallback. No experiment-specific model variable.
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **3** live Ollama calls — one per case.
|
||||
|
||||
### Input Contract
|
||||
|
||||
Each call receives: `{ source, disagreement, requiresUserClarification }`. The boolean is fixed from human-reviewed Experiment 54R-style references. The model does not re-decide whether clarification is required.
|
||||
|
||||
### Output Contract
|
||||
|
||||
Return exactly: `{ "clarificationTarget": "short statement" | null }`.
|
||||
- If `requiresUserClarification` is true → smallest specific user-owned distinction;
|
||||
- If false → null.
|
||||
|
||||
No question text, no explanation, no recommendation, no preferred interpretation, no score, no confidence, no behaviour label. Test-only, not a production schema.
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
> Identify the specific unresolved distinction that only the user can clarify. If clarification is required, return the smallest statement of the missing user-owned meaning, preference, priority, constraint, definition, or private fact. Do not write a question. Do not add evidence needs. If clarification is not required, return null.
|
||||
|
||||
No examples. No expected answers. Same instruction for all three cases.
|
||||
|
||||
### Case 1 — Growth Versus Risk Priority
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Disagreement:** growth should be prioritised even if some additional risk is unavoidable / avoiding additional risk is a hard constraint even if growth is slower.
|
||||
**Fixed clarification status:** `requiresUserClarification: true`
|
||||
**Fixed human target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
### Case 2 — Evidence-Resolvable Delivery Causes
|
||||
|
||||
**Source:** "Orders are arriving late and customers have started complaining."
|
||||
**Disagreement:** delays may be caused by insufficient staff capacity / delays may be caused by unreliable supplier lead times.
|
||||
**Fixed clarification status:** `requiresUserClarification: false`
|
||||
**Fixed human target:** null (cause should be investigated through evidence).
|
||||
|
||||
### Case 3 — Ambiguous Meaning of "Affordable"
|
||||
|
||||
**Source:** "I want to replace the system, but the new option needs to be affordable."
|
||||
**Disagreement:** affordable means keeping upfront purchase cost low / affordable means keeping overall long-term cost low even if upfront cost is higher.
|
||||
**Fixed clarification status:** `requiresUserClarification: true`
|
||||
**Fixed human target:** what the user means by "affordable" — upfront cost versus overall/long-term cost.
|
||||
|
||||
### Results
|
||||
|
||||
#### Case 1 — Growth Versus Risk Priority
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"clarificationTarget": "whether avoiding additional risk is a preference/trade-off or a hard constraint"
|
||||
}
|
||||
```
|
||||
|
||||
**Classification:** `clarification_target_correct` (14 bidirectional concept matches)
|
||||
|
||||
The model correctly identified the constraint versus preference distinction — the exact human target. No evidence confusion, no question text, no winner selection.
|
||||
|
||||
#### Case 2 — Evidence-Resolvable Delivery Causes
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"clarificationTarget": null
|
||||
}
|
||||
```
|
||||
|
||||
**Classification:** `clarification_target_correct` (correctly returned null when clarification is not required)
|
||||
|
||||
The model correctly respected the false signal and returned no target for an evidence-resolvable case. Note: in earlier runs of this test, the model occasionally ignored the false signal and produced a target — this variability was observed but did not affect this final run's classification.
|
||||
|
||||
#### Case 3 — Ambiguous Meaning of "Affordable"
|
||||
|
||||
**Model output:**
|
||||
```json
|
||||
{
|
||||
"clarificationTarget": "whether affordability prioritizes low upfront cost or low long-term total cost"
|
||||
}
|
||||
```
|
||||
|
||||
**Classification:** `clarification_target_correct` (7 bidirectional concept matches)
|
||||
|
||||
The model correctly identified the definition ambiguity — upfront cost versus long-term total cost. No vendor comparison, no budget range confusion, no question text.
|
||||
|
||||
### Clarification-Target-Correct Count
|
||||
|
||||
**3 of 3** cases classified correctly.
|
||||
|
||||
### Clarification-Target-Failed Count
|
||||
|
||||
**0 of 3** cases classified incorrectly.
|
||||
|
||||
### Required Questions — Answers
|
||||
|
||||
1. Did Case 1 identify preference/trade-off versus hard constraint? **Yes**
|
||||
2. Did Case 1 avoid asking about external risk evidence instead? **Yes** (no evidence keywords present)
|
||||
3. Did Case 2 correctly return null? **Yes** (in the final run)
|
||||
4. Did Case 3 identify the meaning of "affordable" as upfront versus long-term cost? **Yes**
|
||||
5. Did the model ever generate a full question? **No**
|
||||
6. Did it confuse clarification target with evidence needed? **No**
|
||||
7. Did it choose a winner? **No**
|
||||
|
||||
### Inference Timing
|
||||
|
||||
- Total time: 55,511ms (55.5s)
|
||||
- Average: 18,503.7ms per call
|
||||
- Fastest: 17,046ms (Case 2 — evidence-resolvable)
|
||||
- Slowest: 20,957ms (Case 1 — growth-vs-risk)
|
||||
|
||||
### Limitations
|
||||
|
||||
- Three cases only; limited domain coverage (one strategic priority, one delivery scenario, one procurement definition).
|
||||
- Same host/model used throughout — results may vary with different configurations.
|
||||
- Does not establish generalisation beyond these specific inputs.
|
||||
- The model occasionally ignored the `requiresUserClarification: false` signal in earlier test runs (producing a target when null was expected), indicating the boolean gate alone may not be sufficient for robust null enforcement.
|
||||
- No clarification question was generated — this experiment establishes the target identification layer only.
|
||||
|
||||
### Conclusion
|
||||
|
||||
**The final three-case run was correct, but earlier repetitions showed instability when clarification was explicitly not required. Clarification-target identification therefore appears promising, but null enforcement is not yet stable.**
|
||||
|
||||
Across three patterns — preference/constraint ambiguity, evidence-resolvable operational causes, and definition ambiguity — the model correctly isolated the specific user-owned distinction when clarification was required, returned null when it was not, and never generated a full question or chose a winning interpretation. Concept-overlap counts were diagnostic only; manual semantic review provided stronger evidence. Case 2 instability is an observed behaviour (the model occasionally produced a target despite `requiresUserClarification: false` in earlier runs), not merely a test warning. This establishes the wording of the future clarification question is still open; this does not establish when Behaviour Selection should choose Clarify; this does not establish how the clarification answer should update the graph.
|
||||
|
||||
### Status
|
||||
|
||||
**Pending Rob's review.** No production code changed. No schemas modified. No active engine behaviour changed. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `tests/reconstruction/semantic-clarification-target.test.js`.
|
||||
## Experiment 54T — Is "No Clarification Needed" Stable Across Repeated Identical Inputs? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
First, tighten Experiment 54S so its conclusion reflects the instability observed during earlier runs.
|
||||
|
||||
Then test one narrow question:
|
||||
|
||||
> **When `requiresUserClarification` is explicitly false, does the model consistently return no clarification target across repeated identical calls?**
|
||||
|
||||
This experiment exists because 54S produced the correct final result but earlier runs sometimes generated a clarification target even when clarification was explicitly not required.
|
||||
|
||||
This is passive and test-only.
|
||||
Do not change the clarification-target instruction yet.
|
||||
Do not generate questions.
|
||||
Do not change production behaviour.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Two possibilities are plausible.
|
||||
|
||||
**Stable gating:** When `requiresUserClarification` is false, the model consistently returns `null`.
|
||||
|
||||
**Semantic override:** The model sometimes ignores the explicit false flag and invents a clarification target because it sees unresolved uncertainty in the source.
|
||||
|
||||
Either result is useful.
|
||||
Do not try to correct the behaviour in this experiment.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54S)
|
||||
Model: `qwen-claude:latest` (same as 54S)
|
||||
|
||||
No localhost fallback. No experiment-specific model variable.
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **6** live Ollama calls — three per case, repeated identical input each time.
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
Exact Experiment 54S instruction unchanged:
|
||||
|
||||
> Identify the specific unresolved distinction that only the user can clarify. If clarification is required, return the smallest statement of the missing user-owned meaning, preference, priority, constraint, definition, or private fact. Do not write a question. Do not add evidence needs. If clarification is not required, return null.
|
||||
|
||||
No examples. No expected answers. Same instruction for all six cases.
|
||||
|
||||
### Output Contract
|
||||
|
||||
Unchanged from 54S:
|
||||
```json
|
||||
{
|
||||
"clarificationTarget": "short statement" | null
|
||||
}
|
||||
```
|
||||
|
||||
### Case A — Evidence-Resolvable / False (the unstable case from 54S)
|
||||
|
||||
**Source:** "Orders are arriving late and customers have started complaining."
|
||||
**Disagreement:** delays may be caused by insufficient staff capacity / delays may be caused by unreliable supplier lead times.
|
||||
**Fixed clarification status:** `requiresUserClarification: false`
|
||||
**Expected result:** `clarificationTarget: null`
|
||||
|
||||
Run this exact case **3 times** without changing wording. This is the unstable Case 2 from 54S.
|
||||
|
||||
### Case B — User-Owned Ambiguity / True Control
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Disagreement:** growth should be prioritised even if some additional risk is unavoidable / avoiding additional risk is a hard constraint even if growth is slower.
|
||||
**Fixed clarification status:** `requiresUserClarification: true`
|
||||
**Expected semantic target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
Run this exact case **3 times**. Purpose: determine whether instability is specific to suppressing clarification or affects clarification-target generation generally.
|
||||
|
||||
### Results
|
||||
|
||||
#### Case A — Evidence-Resolvable / False
|
||||
|
||||
| Run | Result | Classification |
|
||||
|-----|--------|----------------|
|
||||
| 1 | `null` (15,566ms) | null ✓ |
|
||||
| 2 | `null` (13,800ms) | null ✓ |
|
||||
| 3 | `null` (19,087ms) | null ✓ |
|
||||
|
||||
**Null count: 3/3**
|
||||
**Non-null count: 0/3**
|
||||
|
||||
No clarification targets were invented. The model consistently returned `null` across all three repeated identical runs with `requiresUserClarification: false`.
|
||||
|
||||
#### Case B — User-Owned Ambiguity / True Control
|
||||
|
||||
| Run | Target | Classification |
|
||||
|-----|--------|----------------|
|
||||
| 1 | "The relative priority between business growth and strict risk avoidance when they conflict" (18,595ms) | target_correct |
|
||||
| 2 | "Preferred priority between accelerating business growth and strictly avoiding additional risk" (19,458ms) | target_correct |
|
||||
| 3 | "Your maximum acceptable level of additional risk relative to desired business growth." (18,976ms) | target_correct |
|
||||
|
||||
**Correct-target count: 3/3**
|
||||
**Incorrect-target count: 0/3**
|
||||
**Null count: 0/3**
|
||||
|
||||
All three runs produced semantically correct targets aligned with the human reference. No null responses observed when clarification was required.
|
||||
|
||||
### Timing
|
||||
|
||||
- Total time: 105,470ms (105.5s)
|
||||
- Average: 17,578.3ms per call
|
||||
- Fastest: 13,798ms (Case A run 2)
|
||||
- Slowest: 19,457ms (Case B run 3)
|
||||
|
||||
### Required Questions — Answers
|
||||
|
||||
1. How many Case A runs returned `null`? **3**
|
||||
2. How many Case A runs returned a non-null clarification target? **0**
|
||||
3. If Case A produced a target, what distinction did it invent? **N/A — none produced**
|
||||
4. How many Case B runs produced the correct clarification target? **3**
|
||||
5. Did Case B ever incorrectly return `null`? **No**
|
||||
6. Is false/null behaviour materially stable across the three repeated runs? **Yes** — all 3 returned null with zero variance
|
||||
7. Is true/target behaviour materially stable across the three repeated runs? **Yes** — all 3 produced semantically correct targets with zero variance
|
||||
8. Is any observed instability asymmetric: mainly false/null / mainly true/target / both / none observed? **None observed in this experiment.** Both null-gating and target generation were fully stable across these six identical repeated calls.
|
||||
9. Does this experiment establish why instability occurs? **No** — this is an observation experiment, not a diagnostic one.
|
||||
10. Does this establish how to enforce null behaviour? **No** — the current instruction and output contract produced stable null behaviour across the three repeated false-case runs tested here; broader stability remains unproven.
|
||||
11. Does this establish how Behaviour Selection should work? **No.**
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only two cases tested (one false, one true); limited domain coverage.
|
||||
- Same host/model used throughout — results may vary with different configurations.
|
||||
- This is a stability observation experiment; it does not diagnose root causes of earlier variability.
|
||||
- Does not establish generalisation beyond these specific inputs.
|
||||
- The model's behaviour in earlier unrecorded runs (when null-gating failed) remains the unknown variable.
|
||||
|
||||
### Evaluation Conclusion
|
||||
|
||||
**Clarification null-gating was stable across all tested repeats**
|
||||
|
||||
Case A returned `null` in 3 of 3 runs. Case B produced correct targets in 3 of 3 runs. No instability was observed in either direction during this experiment.
|
||||
|
||||
### Status
|
||||
|
||||
**Pending Rob's review.** No production code changed. No schemas modified. No active engine behaviour changed. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `tests/reconstruction/semantic-clarification-null-stability.test.js`.
|
||||
|
||||
## Experiment 54U — Can a Clarification Target Become a Useful User Question Without Adding New Meaning? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Given a fixed clarification target, can the model turn that target into one concise, neutral clarification question without adding assumptions, choosing a side, or broadening the issue?
|
||||
|
||||
This is test-only.
|
||||
Do not integrate anything into the UI.
|
||||
Do not change Behaviour Selection.
|
||||
Do not change production prompts.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Given a specific clarification target, the model may be able to produce a single question that:
|
||||
|
||||
- asks only about the unresolved distinction;
|
||||
- remains neutral between the interpretations;
|
||||
- does not introduce new assumptions;
|
||||
- does not ask for evidence instead;
|
||||
- does not become a multi-part interview.
|
||||
|
||||
If it broadens the question or adds new meaning, record that failure.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54S / 54T)
|
||||
Model: `qwen-claude:latest` (same as 54S / 54T)
|
||||
|
||||
No localhost fallback. No experiment-specific model variable.
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **3** live Ollama calls — one per case.
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
> Write one concise clarification question that asks only about the supplied clarification target. Keep it neutral between the possible meanings. Do not introduce new facts, assumptions, evidence requests, recommendations, or additional questions. Do not explain why you are asking.
|
||||
|
||||
No examples. No expected wording. Same instruction for all three cases.
|
||||
|
||||
### Input Contract
|
||||
|
||||
Each call receives:
|
||||
|
||||
```json
|
||||
{ "source": "...", "clarificationTarget": "..." }
|
||||
```
|
||||
|
||||
The target is fixed human-reviewed input. The model must not decide whether clarification is needed.
|
||||
|
||||
### Output Contract
|
||||
|
||||
Return exactly:
|
||||
|
||||
```json
|
||||
{ "question": "one clarification question" }
|
||||
```
|
||||
|
||||
No explanation, score, confidence, answer options, recommendation, preferred interpretation, or second question.
|
||||
|
||||
### Case 1 — Preference Versus Hard Constraint
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
**Human-reviewed intent:** A correct question should ask whether "no more risk" is an absolute boundary or something the user would trade against growth.
|
||||
|
||||
### Case 2 — Meaning of "Affordable"
|
||||
|
||||
**Source:** "I want to replace the system, but the new option needs to be affordable."
|
||||
**Clarification target:** whether affordable means low upfront cost or low overall/long-term cost.
|
||||
**Human-reviewed intent:** A correct question should clarify which meaning of affordability the user intends.
|
||||
|
||||
### Case 3 — Private Factual Constraint
|
||||
|
||||
**Source:** "I could move the project forward next month, depending on whether I actually have enough time."
|
||||
**Clarification target:** whether the user has enough available time next month to take on the project.
|
||||
**Human-reviewed intent:** A correct question should ask the user to clarify their own available capacity/time.
|
||||
|
||||
### Results
|
||||
|
||||
#### Case 1 — Preference Versus Hard Constraint
|
||||
|
||||
**Returned question:** "Do you view avoiding additional risk as a preference/trade-off or a hard constraint?"
|
||||
**Classification:** question_correct — semantically aligned with target (6 overlapping words), structurally valid, neutral.
|
||||
**Did it stay within the target?** Yes — asks exactly whether risk avoidance is a preference/trade-off or a hard constraint.
|
||||
**Did it add assumptions?** No.
|
||||
**Did it ask for evidence?** No.
|
||||
**Multiple questions?** No — one substantive question.
|
||||
**Implied preferred interpretation?** No.
|
||||
|
||||
#### Case 2 — Meaning of "Affordable"
|
||||
|
||||
**Returned question:** 'Does your use of "affordable" refer to a low upfront cost or a low overall/long-term cost?'
|
||||
**Classification:** question_correct — semantically aligned with target (5 overlapping words), structurally valid, neutral.
|
||||
**Did it stay within the target?** Yes — asks which meaning of affordability the user intends.
|
||||
**Did it add assumptions?** No.
|
||||
**Did it ask for evidence?** No.
|
||||
**Multiple questions?** No.
|
||||
**Implied preferred interpretation?** No.
|
||||
|
||||
#### Case 3 — Private Factual Constraint
|
||||
|
||||
**Returned question:** "Do you have enough available time next month to take on the project?"
|
||||
**Classification:** question_correct — semantically aligned with target (7 overlapping words), structurally valid, neutral.
|
||||
**Did it stay within the target?** Yes — asks for the user's own capacity clarification.
|
||||
**Did it add assumptions?** No — did not estimate hours or assume availability.
|
||||
**Did it ask for evidence?** No.
|
||||
**Multiple questions?** No.
|
||||
**Implied preferred interpretation?** No.
|
||||
|
||||
### Timing
|
||||
|
||||
- Total time: 71,758ms (71.8s)
|
||||
- Average: 23,919.3ms per call
|
||||
- Fastest: 13,573ms (Case 3)
|
||||
- Slowest: 33,314ms (Case 2)
|
||||
|
||||
### Required Questions — Answers
|
||||
|
||||
1. Did Case 1 ask only about preference/trade-off versus hard constraint? **Yes**
|
||||
2. Did Case 2 ask only what "affordable" means? **Yes**
|
||||
3. Did Case 3 correctly ask for the user's private factual capacity? **Yes**
|
||||
4. Did any question introduce assumptions not present in the source/target? **No**
|
||||
5. Did any question ask for evidence instead of clarification? **No**
|
||||
6. Did any question contain more than one substantive question? **No**
|
||||
7. Did any question choose or imply a preferred interpretation? **No**
|
||||
8. How many cases were question_correct / question_failed? **3 correct, 0 failed.**
|
||||
9. Does this establish when the question should be asked? **No.**
|
||||
10. Does this establish how the answer should update reasoning state? **No.**
|
||||
11. Does this establish that the question will feel good in the UI? **No.**
|
||||
|
||||
### Evaluation Conclusion
|
||||
|
||||
**The model produced a clean clarification question for every tested target.**
|
||||
|
||||
All three cases returned one neutral question addressing only the supplied clarification target, with no introduced assumptions, evidence requests, multi-part structure, or implied preferred interpretations.
|
||||
|
||||
**Corrected conclusion:** The clarification-target → question step worked cleanly across the three tested targets; broader wording quality and user experience remain untested. Word-overlap metrics are diagnostic only; manual semantic review is the stronger evidence.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only three cases tested; limited domain coverage.
|
||||
- Same host/model used throughout — results may vary with different configurations.
|
||||
- Does not establish when the question should be asked (that is a separate step).
|
||||
- Does not establish how answers should update reasoning state.
|
||||
- Semantic quality assessed through structural and overlap heuristics; manual review would strengthen confidence.
|
||||
- Does not establish that the question will feel good in the UI.
|
||||
|
||||
### Status
|
||||
|
||||
**Pending Rob's review.** No production code changed. No schemas modified. No active engine behaviour changed. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `tests/reconstruction/semantic-clarification-question.test.js`.
|
||||
|
||||
## Experiment 54V — Can a Clarification Answer Resolve Only the Target Without Rewriting Everything Else? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Given the original source, a fixed clarification target, the clarification question, and the user's answer, can the model identify what has now been resolved without adding new meaning or rewriting unrelated reasoning?
|
||||
|
||||
This is test-only.
|
||||
Do not integrate with the graph, Behaviour Selection, or UI.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A clarification answer should be able to resolve one narrow user-owned ambiguity without causing the model to:
|
||||
- reinterpret the whole source;
|
||||
- add unsupported consequences;
|
||||
- reopen unrelated uncertainty.
|
||||
|
||||
If the model cannot preserve that boundary, clarification answers may create as much ambiguity as they remove.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54S / 54T / 54U)
|
||||
Model: `qwen-claude:latest` (same as 54S / 54T / 54U)
|
||||
|
||||
No localhost fallback. No experiment-specific model variable.
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **3** live Ollama calls — one per case.
|
||||
|
||||
### Input Contract
|
||||
|
||||
Each call receives:
|
||||
```json
|
||||
{ "source": "...", "clarificationTarget": "...", "clarificationQuestion": "...", "userAnswer": "..." }
|
||||
```
|
||||
|
||||
### Output Contract
|
||||
|
||||
Return exactly:
|
||||
```json
|
||||
{ "resolvedMeaning": "short statement", "targetResolved": true, "remainingUncertainty": null }
|
||||
```
|
||||
|
||||
No next question, recommendation, confidence score, graph update, extra interpretation, or explanation.
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
> Use the user's clarification answer only to resolve the supplied clarification target. State the meaning now established by that answer. Mark targetResolved true only when the answer settles the target. Put any uncertainty that remains specifically about that target into remainingUncertainty; otherwise return null. Do not infer wider consequences, rewrite unrelated source meaning, recommend action, or generate another question.
|
||||
|
||||
No examples. No expected answers. Same instruction for all three cases.
|
||||
|
||||
### Case 1 — Hard Constraint Resolved
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
**Clarification question:** Do you view avoiding additional risk as a preference/trade-off or a hard constraint?
|
||||
**User answer:** It's a hard constraint. I don't want any increase in risk.
|
||||
|
||||
**Human reference:**
|
||||
- resolvedMeaning: "avoiding additional risk is a hard constraint"
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
Must not infer: growth is impossible; which growth option should be chosen; how much risk currently exists.
|
||||
|
||||
### Case 2 — Definition Resolved
|
||||
|
||||
**Source:** "I want to replace the system, but the new option needs to be affordable."
|
||||
**Clarification target:** whether affordable means low upfront cost or low overall/long-term cost.
|
||||
**Clarification question:** Does your use of "affordable" refer to a low upfront cost or a low overall/long-term cost?
|
||||
**User answer:** I care about the total cost over five years, not the upfront price.
|
||||
|
||||
**Human reference:**
|
||||
- resolvedMeaning: "affordability means overall/long-term cost rather than upfront cost"
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
Do not invent a budget or preferred system.
|
||||
|
||||
### Case 3 — Answer Does Not Fully Resolve Target
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
**Clarification question:** Do you view avoiding additional risk as a preference/trade-off or a hard constraint?
|
||||
**User answer:** It depends on the opportunity.
|
||||
|
||||
**Human reference:**
|
||||
- The target is **not fully resolved**.
|
||||
- targetResolved: false
|
||||
- remainingUncertainty should preserve that the acceptable trade-off still depends on circumstances.
|
||||
- resolvedMeaning may state only that the user's risk position is conditional on the opportunity.
|
||||
|
||||
Do not force the answer into either preference or hard-constraint category.
|
||||
|
||||
### Results
|
||||
|
||||
#### Case 1 — Hard Constraint Resolved
|
||||
|
||||
**Returned result:**
|
||||
- resolvedMeaning: "Avoiding additional risk is established as a hard constraint, meaning no increase in risk is acceptable."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
**Manual semantic review:**
|
||||
- Used only the user answer to resolve the target? **Yes.**
|
||||
- Preserved unresolved ambiguity where present? N/A (target was resolved).
|
||||
- Avoided wider consequences? **Yes** — did not state growth is impossible, did not choose a growth option, did not estimate current risk levels. The phrase "meaning no increase in risk is acceptable" is a direct restatement of the user answer ("I don't want any increase in risk"), not an addition from the source.
|
||||
- Avoided adding unsupported facts? **Yes.**
|
||||
- Avoided generating another question? **Yes.**
|
||||
|
||||
**Classification:** resolution_correct
|
||||
|
||||
#### Case 2 — Definition Resolved
|
||||
|
||||
**Returned result:**
|
||||
- resolvedMeaning: "Affordable refers to a low overall or long-term cost rather than a low upfront price."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
**Manual semantic review:**
|
||||
- Used only the user answer to resolve the target? **Yes.**
|
||||
- Preserved unresolved ambiguity where present? N/A (target was resolved).
|
||||
- Avoided wider consequences? **Yes** — did not invent a budget amount, did not recommend a system.
|
||||
- Avoided adding unsupported facts? **Yes.**
|
||||
- Avoided generating another question? **Yes.**
|
||||
|
||||
**Classification:** resolution_correct
|
||||
|
||||
#### Case 3 — Answer Does Not Fully Resolve Target
|
||||
|
||||
**Returned result:**
|
||||
- resolvedMeaning: "Avoiding additional risk is conditional rather than fixed, varying based on the specific opportunity."
|
||||
- targetResolved: false
|
||||
- remainingUncertainty: "It remains unclear whether avoiding additional risk is strictly a preference/trade-off or a hard constraint, as the stance depends on the opportunity."
|
||||
|
||||
**Manual semantic review:**
|
||||
- Used only the user answer to resolve the target? **Yes.** The model preserved the conditionality present in "It depends on the opportunity" without forcing into either category.
|
||||
- Preserved unresolved ambiguity where present? **Yes.** Correctly kept the target unresolved and described the remaining uncertainty specifically about that target.
|
||||
- Avoided wider consequences? **Yes.**
|
||||
- Avoided adding unsupported facts? **Yes.**
|
||||
- Avoided generating another question? **Yes.**
|
||||
|
||||
**Classification:** resolution_correct
|
||||
|
||||
### Timing
|
||||
|
||||
- Total time: 38,052ms (38.1s)
|
||||
- Average: 12,684.0ms per call
|
||||
- Fastest: 8,479ms (Case 1)
|
||||
- Slowest: 15,208ms (Case 3)
|
||||
|
||||
### Required Questions — Answers
|
||||
|
||||
1. Did Case 1 resolve the target to a hard constraint without adding wider consequences? **Yes.** The resolved meaning stays within the user answer's scope. No inference about growth feasibility, option selection, or current risk levels.
|
||||
2. Did Case 2 resolve "affordable" to long-term cost without inventing a budget? **Yes.** The model correctly captured the five-year perspective without adding any budget figure or system recommendation.
|
||||
3. Did Case 3 correctly keep the target unresolved? **Yes.** The model returned targetResolved=false, preserved conditionality in resolvedMeaning, and provided meaningful remainingUncertainty.
|
||||
4. Did any case force an ambiguous answer into a stronger meaning? **No.** Case 3's conditional answer was kept at its actual strength — neither strengthened to preference nor to hard constraint.
|
||||
5. Did any case rewrite unrelated parts of the source? **No.** In Cases 1 and 3 (same source), the model treated the "grow" portion identically to the original source meaning without reinterpreting it.
|
||||
6. Did any case generate another question? **No.** All resolvedMeaning fields are statements, not questions.
|
||||
7. How many cases were resolution_correct / resolution_failed? **3 correct, 0 failed.**
|
||||
8. Does this establish how graph state should be updated? **No.** This only tests semantic recognition of what a clarification answer resolves; it does not test any graph update mechanism.
|
||||
9. Does this establish what question should come next? **No.** The experiment tested one directional step (answer → resolved meaning) and did not test the next question generation cycle.
|
||||
10. Does this establish how Behaviour Selection should react? **No.** No behaviour selection logic was tested or involved.
|
||||
|
||||
### Evaluation Conclusion
|
||||
|
||||
**Clarification answers resolved only the intended target across all tested cases.**
|
||||
|
||||
All three cases returned correct resolution boundaries: Cases 1 and 2 settled the target cleanly; Case 3 preserved incomplete information at its actual strength without forcing it into a stronger category. The model did not widen beyond the clarification target, invent consequences, or generate new questions in any case.
|
||||
|
||||
**The individual clarification steps have each worked in their isolated fixed-case tests; end-to-end behaviour remains untested.**
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only three cases tested; limited domain coverage (risk constraint, affordability definition, conditional constraint).
|
||||
- Same host/model used throughout — results may vary with different configurations.
|
||||
- Does not establish how graph state should update from resolved meanings.
|
||||
- Does not establish what question should come next after resolution.
|
||||
- Does not establish how Behaviour Selection should react to resolved vs unresolved targets.
|
||||
- Semantic quality assessed through structural checks and manual review; broader generalisation untested.
|
||||
- Case 3's remainingUncertainty output is longer than the human reference — acceptable because it describes the uncertainty rather than adding meaning, but worth noting.
|
||||
|
||||
### Status
|
||||
|
||||
**Pending Rob's review.** No production code changed. No schemas modified. No active engine behaviour changed. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `tests/reconstruction/semantic-clarification-answer-resolution.test.js`.
|
||||
|
||||
+659
@@ -0,0 +1,659 @@
|
||||
## Experiment 54W — Does the Clarification Chain Hold Together End to End? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Tighten Experiment 54V so it does not overstate the isolated clarification-chain results. Then test the smallest end-to-end version of the clarification path: **starting from one disagreement, can the semantic steps correctly determine whether the user is needed, identify the clarification target, word one question, and use the user's answer to resolve only that target without semantic drift between steps?**
|
||||
|
||||
This is still test-only. Do not integrate with the active engine, graph, Behaviour Selection, or UI.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The individual clarification steps may remain aligned when chained together. For a genuine user-owned ambiguity, the chain should preserve: `disagreement → user required → clarification target → neutral question → answer → resolved target`. For an evidence-resolvable disagreement, the chain should stop early rather than inventing a clarification target or question. If the steps drift when connected, record exactly where the first material divergence occurs.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54R–54V)
|
||||
Model: `qwen-claude:latest` (same as 54R–54V)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **5** live Ollama calls — 4 for Scenario A + 1 for Scenario B.
|
||||
|
||||
### Context Used
|
||||
|
||||
- `docs/current-handoff.md`
|
||||
- Experiment 54V only in `docs/design-evolution-log.md`
|
||||
- `tests/reconstruction/semantic-disagreement-resolution-source.test.js`
|
||||
- `tests/reconstruction/semantic-clarification-target.test.js`
|
||||
- `tests/reconstruction/semantic-clarification-question.test.js`
|
||||
- `tests/reconstruction/semantic-clarification-answer-resolution.test.js`
|
||||
|
||||
### Scenarios
|
||||
|
||||
#### Scenario A — Genuine User-Owned Ambiguity
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
|
||||
**Fixed disagreement:**
|
||||
1. growth should be prioritised even if some additional risk is unavoidable;
|
||||
2. avoiding additional risk is a hard constraint even if growth is slower.
|
||||
|
||||
**Fixed user answer:** "It's a hard constraint. I don't want any increase in risk."
|
||||
|
||||
#### Scenario B — Evidence-Resolvable Disagreement
|
||||
|
||||
**Source:** "Orders are arriving late and customers have started complaining."
|
||||
|
||||
**Fixed disagreement:**
|
||||
1. delays may be caused by insufficient staff capacity;
|
||||
2. delays may be caused by unreliable supplier lead times.
|
||||
|
||||
### Stage-by-Stage Outputs
|
||||
|
||||
#### Scenario A — Full Chain
|
||||
|
||||
**A1 — Resolution Source (Experiment 54R instruction)**
|
||||
|
||||
- **Actual output:** `{ requiresUserClarification: true, reason: "The disagreement centers on a priority trade-off between growth and risk avoidance that depends entirely on the user's underlying constraints rather than external facts." }`
|
||||
- **Input used:** original source + fixed disagreement + evidence needs
|
||||
- **Classification:** `stage_correct` — correctly identified that user clarification is required for a genuine user-owned ambiguity. The reason references subjective constraints (user-owned) not external facts, consistent with the true classification.
|
||||
|
||||
**A2 — Clarification Target (Experiment 54S instruction)**
|
||||
|
||||
- **Input target used from A1:** `requiresUserClarification: true` (actual upstream output, NOT replaced by a human reference)
|
||||
- **Actual output:** `{ clarificationTarget: "Priority between business growth and risk avoidance when they conflict" }`
|
||||
- **Classification:** `stage_correct` — the model correctly identified the user-owned distinction. The target focuses on priority ordering between two competing interests (growth vs. risk), which is the exact user-owned ambiguity present in the source. No question language, no evidence framing, no interpretation chosen.
|
||||
|
||||
**A3 — Clarification Question (Experiment 54U instruction)**
|
||||
|
||||
- **Input target from A2:** `"Priority between business growth and risk avoidance when they conflict"` (actual upstream output)
|
||||
- **Actual output:** `{ question: "Which do you prioritize when business growth and risk avoidance come into conflict?" }`
|
||||
- **Classification:** `stage_correct` — single neutral question, ends with one question mark, no biased framing. The question uses the actual A2 target's language ("priority", "business growth", "risk avoidance") without substituting a human reference. Neutral between alternatives: does not say you should prioritize growth or risk.
|
||||
|
||||
**A4 — Answer Resolution (Experiment 54V instruction)**
|
||||
|
||||
- **Input target from actual A2:** `"Priority between business growth and risk avoidance when they conflict"`
|
||||
- **Input question from actual A3:** `"Which do you prioritize when business growth and risk avoidance come into conflict?"`
|
||||
- **Fixed user answer:** "It's a hard constraint. I don't want any increase in risk."
|
||||
- **Actual output:** `{ resolvedMeaning: "Risk avoidance takes absolute priority over business growth, as no increase in risk is acceptable regardless of growth potential.", targetResolved: true, remainingUncertainty: null }`
|
||||
- **Classification:** `stage_correct` — targetResolved=true correctly reflects that the user's answer settles the priority question. The resolved meaning interprets "hard constraint" and "no increase in risk" as absolute priority (non-negotiable boundary) over growth. This is a valid semantic interpretation of the user's answer within the A2/A3 framing. No wider consequences introduced: does not state growth is impossible, does not choose a specific growth option, does not estimate current risk levels.
|
||||
|
||||
#### Scenario B — Evidence-Resolvable Stop
|
||||
|
||||
**B1 — Resolution Source (Experiment 54R instruction)**
|
||||
|
||||
- **Actual output:** `{ requiresUserClarification: false, reason: "The disagreement can be resolved by examining operational metrics and supplier delivery records rather than requiring subjective input." }`
|
||||
- **Classification:** `correct_stop` — correctly identified that evidence (operational data) can resolve the cause of delays. The chain stopped immediately; no clarification target call was made, no question was generated, no answer-resolution was attempted.
|
||||
|
||||
### Actual Upstream Outputs Passed Between Stages
|
||||
|
||||
| Stage | Received From | Value |
|
||||
|---|---|---|
|
||||
| A2 | A1 | `requiresUserClarification: true`, reason about subjective constraints |
|
||||
| A3 | A2 | `clarificationTarget: "Priority between business growth and risk avoidance when they conflict"` |
|
||||
| A4 | A2 | Same A2 target (same as above) |
|
||||
| A4 | A3 | `question: "Which do you prioritize when business growth and risk avoidance come into conflict?"` |
|
||||
|
||||
Human reference data was used only for evaluation — not silently substituted between stages.
|
||||
|
||||
### First Drift Point in Scenario A
|
||||
|
||||
**No material chain failure occurred, although Stage A2 broadened the clarification target from preference-versus-hard-constraint to general priority ordering. That loss of specificity did not break this scenario.** The chain preserved the user-owned nature of the ambiguity from A1 through to resolution at A4 without introducing unsupported meaning or changing the interpretation of upstream results.
|
||||
|
||||
### Scenario B — Stop Verification
|
||||
|
||||
- **Did Scenario B stop after B1:** Yes
|
||||
- **Were any unnecessary clarification calls made for Scenario B:** No (0 additional calls)
|
||||
|
||||
### Question: Did any stage choose a winner?
|
||||
|
||||
**No.** None of the stages selected an interpretation as correct or better. A4's resolved meaning states what the user's answer settled (priority resolution) rather than declaring one pre-existing interpretation as the winner. The chain reports what was clarified, not which side of the original disagreement is right.
|
||||
|
||||
### Question: Did any stage introduce unsupported meaning that materially affected the next stage?
|
||||
|
||||
**No.** A2 stayed within the priority dimension present in the source. A3 preserved both competing terms ("business growth", "risk avoidance") from the A2 target. A4 interpreted the user's hard-constraint answer as absolute priority over growth — a valid reading given the answer and the A2/A3 framing. No stage added external facts or consequences that materially distorted downstream reasoning.
|
||||
|
||||
### Question: Evidence that isolated clarification steps survive under chaining
|
||||
|
||||
**Yes.** The chain_correct result demonstrates that all four individual capabilities (resolution source, target identification, question wording, answer resolution) remained usable when chained in the two tested scenarios. Each stage's output was a valid input for the next stage. No stage degraded or produced an unexpected format. The semantic proximity between A2 and A4 is worth noting: A2 framed the distinction as "priority" while the user answer used "hard constraint" — these are semantically close but not identical (a hard constraint is stronger than a priority preference). A4 correctly interpreted the hard-constraint answer within the priority framing, so this proximity was sufficient for alignment.
|
||||
|
||||
### Questionable or Unsupported Findings
|
||||
|
||||
- Only **one** instance of each scenario was tested. Chain stability across repeated runs needs verification.
|
||||
- Only the growth-versus-risk domain was tested for Scenario A. Different domains may produce different drift patterns.
|
||||
- A2's output ("Priority between business growth and risk avoidance when they conflict") lost the "preference/trade-off vs hard constraint" distinction present in the 54S human reference. This loss of granularity is not a failure per se — it is still correct within its contract — but it means downstream stages operate on a less precise target. The chain succeeded with this coarser representation, which is evidence that the steps tolerate some semantic imprecision.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**The clarification chain remained semantically aligned end to end in both tested scenarios.** For Scenario A (genuine user-owned ambiguity), all four stages produced correct outputs and each stage's actual output was a valid input for the next stage with no material drift. For Scenario B (evidence-resolvable disagreement), the model correctly stopped after the first decision without inventing unnecessary clarification steps.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
`chain_correct` for Scenario A + `correct_stop` for Scenario B. 5/5 live calls completed within budget. All stage assertions passed.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
The chain_correct result is new evidence not available in any earlier experiment (54R–54V tested isolated steps only). It demonstrates that the individual clarification capabilities remained usable when chained in the two tested scenarios, on one scenario and one model configuration. This does not extend to production integration readiness.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added 54V clarifying caveat; added full Experiment 54W entry
|
||||
- `docs/current-handoff.md` — added 54V clarifying caveat
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as 54R–54V.
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file, not production prompts.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Clarification-Chain Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No clarification-chain logic entered any active runtime path, production module, or behaviour selection output.
|
||||
|
||||
### Return-to-Work Note (Experiment 54W)
|
||||
|
||||
54R–54V tested the clarification steps individually in isolated fixed-case scenarios; each worked correctly on its own but end-to-end alignment was never verified. 54W tested the first chained journey using actual upstream model outputs rather than replacing them with human references across four stages for Scenario A and one stage for Scenario B. The growth-versus-risk chain stayed aligned through decision → target → question → answer resolution (chain_correct). The delivery-cause case correctly stopped before clarification (correct_stop). No material chain failure occurred, although Stage A2 broadened the clarification target from preference-versus-hard-constraint to general priority ordering. That loss of specificity did not break this scenario. Graph, Behaviour Selection, UI, and production integration remain untouched. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: `feature/user-workspace-ux-v0.7`. First test/file to inspect when resuming: `tests/reconstruction/semantic-clarification-chain.test.js` for the full experiment and results. Status pending Rob's review.
|
||||
## Experiment 54X — Does the Clarification Target Lose Important Specificity When Chained? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Isolate whether the clarification-target-generation step preserves the exact user-owned distinction or broadens it, using three fixed cases under the same instruction as Experiment 54S. Passive and test-only. No question generation, answer resolution, Behaviour Selection, graph, or UI integration attempted. Same host/model. Branch: `feature/user-workspace-ux-v0.7`.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The model may preserve clarification targets well when the ambiguity is simple and explicit, but broaden targets when the distinction is relational or preference-based. If broadening happens repeatedly, that may matter downstream because the question-generation step can only be as precise as the target it receives.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54R–54W)
|
||||
Model: `qwen-claude:latest` (same as 54R–54W)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **3** live Ollama calls — one per case.
|
||||
|
||||
### Context Used
|
||||
|
||||
- `docs/current-handoff.md`
|
||||
- Experiment 54W only in `docs/design-evolution-log.md` (as historical context for the broadening observation)
|
||||
- `tests/reconstruction/semantic-clarification-target.test.js` (for structural reference)
|
||||
- `tests/reconstruction/semantic-clarification-chain.test.js` (for structural reference)
|
||||
|
||||
### Experiment 54W Corrections Applied
|
||||
|
||||
Replaced "No drift was detected." with: "**No material chain failure occurred, although Stage A2 broadened the clarification target from preference-versus-hard-constraint to general priority ordering. That loss of specificity did not break this scenario.**"
|
||||
|
||||
Replaced "the individual clarification capabilities survive end-to-end chaining" with: "**The individual clarification capabilities remained usable when chained in the two tested scenarios.**"
|
||||
|
||||
### Case 1 — Preference Versus Hard Constraint
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
**Disagreement:** growth should be prioritised even if some additional risk is unavoidable / avoiding additional risk is a hard constraint even if growth is slower.
|
||||
**Human reference target:** whether avoiding additional risk is a preference/trade-off or a hard constraint
|
||||
|
||||
**Actual model output:** `"preferred priority between business growth and risk avoidance"`
|
||||
**Specificity classification:** **target_broadened** — On the right topic but broadened from the material distinction (preference/trade-off vs. hard constraint) to general priority ordering. Usable downstream but not fully specific.
|
||||
|
||||
### Case 2 — Definition Ambiguity
|
||||
|
||||
**Source:** "I want to replace the system, but the new option needs to be affordable."
|
||||
**Disagreement:** affordable means keeping upfront cost low / affordable means keeping total long-term cost low.
|
||||
**Human reference target:** whether "affordable" means low upfront cost or low overall/long-term cost
|
||||
|
||||
**Actual model output:** `"whether 'affordable' refers to upfront cost or total long-term cost"`
|
||||
**Specificity classification:** **target_specific** — Preserved the material distinction: upfront cost versus total long-term cost. The distinction is explicit and identical in meaning to the human reference.
|
||||
|
||||
### Case 3 — Private Factual Boundary
|
||||
|
||||
**Source:** "I could move the project forward next month, depending on whether I actually have enough time."
|
||||
**Disagreement:** the user has enough available time next month / the user does not have enough available time next month.
|
||||
**Human reference target:** whether the user has enough available time next month to take on the project
|
||||
|
||||
**Actual model output:** `"whether the user has enough available time next month"`
|
||||
**Specificity classification:** **target_specific** — Preserved all three required elements: time availability, next month, and the capacity question. The omission of "to take on the project" does not lose material specificity — it is implied by the source context.
|
||||
|
||||
### Inference Timing
|
||||
|
||||
| Metric | Value |
|
||||
|---|---|
|
||||
| Total live calls | 3 |
|
||||
| Total inference time | 49,220 ms |
|
||||
| Average | 16,406.7 ms per call |
|
||||
| Fastest | 11,815 ms (Case 2) |
|
||||
| Slowest | 21,303 ms (Case 1) |
|
||||
|
||||
### Results Summary
|
||||
|
||||
| Classification | Count |
|
||||
|---|---|
|
||||
| target_specific | 2/3 |
|
||||
| target_broadened | 1/3 |
|
||||
| target_wrong | 0/3 |
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. Did Case 1 preserve preference/trade-off versus hard constraint? **No** — broadened to priority ordering.
|
||||
2. Did Case 2 preserve upfront versus long-term affordability? **Yes** — preserved explicitly.
|
||||
3. Did Case 3 preserve the user's available-time boundary? **Yes** — preserved explicitly with all three required elements.
|
||||
4. How many cases were target_specific / target_broadened / target_wrong? **2 / 1 / 0**
|
||||
5. Did any target remain usable while still losing material specificity? **Yes** — Case 1 was broadly relevant and actionable but lost the preference-versus-constraint distinction.
|
||||
6. Did any target introduce unsupported meaning? **No** — no case introduced concepts not present in source or disagreement.
|
||||
7. Does this reproduce the broadening observed in 54W? **Yes** — both experiments show broadening from preference/constraint to priority framing on Case 1-style input.
|
||||
8. Does this establish why broadening happens? **No** — one isolated result per case cannot determine causality; only that it does occur for at least one ambiguity pattern.
|
||||
9. Does this establish whether a broader target is acceptable for the user journey? **No** — acceptability depends on downstream question quality and user experience, which were not tested here.
|
||||
10. Does this establish how the clarification question should be worded? **No** — no question-generation step was involved.
|
||||
|
||||
### Evidence About Clarification-Target Specificity
|
||||
|
||||
A clarification target can be broadly relevant without being precise enough. Case 1's output ("preferred priority between business growth and risk avoidance") is clearly about the right topic and usable downstream, but it does not preserve the material distinction that the user actually needs to clarify — whether avoiding additional risk is a preference or a hard constraint. Cases 2 and 3 show that the same instruction can produce fully specific targets when the ambiguity involves definition boundaries or private facts rather than preference-versus-constraint relationships.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only one model configuration was tested (qwen-claude:latest). Different models may behave differently.
|
||||
- Only one inference per case — stability across repeated runs is untested here (though 54L previously showed strong stability for other tasks).
|
||||
- The broadening pattern only emerged in Case 1; the instruction and model appear capable of specificity on other patterns.
|
||||
- No downstream question or answer-resolution step was tested — usability of a broader target cannot be fully assessed without those stages.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
Clarification targets remained usable but broadened in one of three tested cases (Case 1). The broadening reproduced the same pattern observed in 54W: preference-versus-constraint distinctions tend to become priority-ordering framings. This is not a failure — the target remains actionable — but Specificity loss occurred in one of the three tested ambiguity patterns and was absent in the other two.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
2/3 targets preserved material distinction; 1/3 broadened (matching 54W pattern). All structural assertions passed. No invariant violations detected.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
The Case 1 result reproduces the A2 output from Experiment 54W ("Priority between business growth and risk avoidance when they conflict" → "preferred priority between business growth and risk avoidance"). The same broadening pattern was reproduced across two tested runs under the same model and configuration, making it a repeatable candidate behaviour rather than a one-off observation.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 54X entry
|
||||
- `docs/current-handoff.md` — updated Return-to-Work note with 54X findings; applied 54W wording corrections
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as 54R–54X.
|
||||
|
||||
### Confirmation Semantic Instruction and Output Contract Remained Unchanged
|
||||
|
||||
The instruction was identical to Experiment 54S (no examples, no stronger coaching). The output contract remained `{ "clarificationTarget": "short statement" }` — unchanged from 54S.
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instruction defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Clarification-Target Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No clarification-target logic entered any active runtime path, production module, or behaviour selection output.
|
||||
|
||||
### Return-to-Work Note (Experiment 54X)
|
||||
|
||||
54W showed the full clarification chain worked in the tested pair but Stage A2 broadened one target; 54X isolated target specificity using three clarification cases under the same 54S instruction — preference/constraint distinction was lost to priority framing (broadened), affordability definition stayed precise (specific), and private factual capacity stayed precise (specific). The same broadening pattern was reproduced across two tested runs under the same model and configuration, making it a repeatable candidate behaviour rather than a one-off observation. No question generation, answer resolution, Behaviour Selection, graph, or UI integration was attempted. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: `feature/user-workspace-ux-v0.7`. First test/file to inspect when resuming: `tests/reconstruction/semantic-clarification-target-specificity.test.js` for the full experiment and results. Status pending Rob's review.
|
||||
|
||||
## Experiment 54Y — Does a Broader Clarification Target Actually Change the User Question or Resolution? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Test whether the specificity loss observed in Experiments 54W/54X actually matters downstream. If the clarification target shifts from the precise distinction "preference/trade-off versus hard constraint" to the broader "priority between growth and risk," does that materially change the question asked or the meaning resolved from the user's answer? Passive and test-only. No redesign of target generation, no integration with Behaviour Selection, graph, or UI.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The broader target may remain workable but could alter the user-facing distinction. Specifically:
|
||||
- the precise target may ask whether risk avoidance is a hard boundary or a trade-off;
|
||||
- the broader target may instead ask which objective has priority.
|
||||
|
||||
Those questions are related, but the user's answers need not mean exactly the same thing.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54R–54X)
|
||||
Model: `qwen-claude:latest` (same as 54R–54X)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **4** live Ollama calls — one question per variant + one answer-resolution per variant.
|
||||
|
||||
### Context Used
|
||||
|
||||
- `docs/current-handoff.md`
|
||||
- Experiment 54X only in `docs/design-evolution-log.md` (as historical context for the broadening observation)
|
||||
- `tests/reconstruction/semantic-clarification-question.test.js` (for structural reference: instruction and output contract)
|
||||
- `tests/reconstruction/semantic-clarification-answer-resolution.test.js` (for structural reference: instruction and output contract)
|
||||
|
||||
### Fixed Scenario
|
||||
|
||||
**Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
|
||||
**Fixed user answer:** "It's a hard constraint. I don't want any increase in risk."
|
||||
|
||||
### Variant A — Precise Target
|
||||
|
||||
**Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
This is the human-reviewed specific target.
|
||||
|
||||
### Variant B — Broadened Target
|
||||
|
||||
**Clarification target:** priority between business growth and risk avoidance when they conflict.
|
||||
|
||||
This mirrors the broader target observed in Experiments 54W and 54X.
|
||||
|
||||
---
|
||||
|
||||
### Stage 1 Results — Question Generation
|
||||
|
||||
| Variant | Generated Question |
|
||||
|---|---|
|
||||
| A (Precise) | Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off? |
|
||||
| B (Broadened) | When business growth and risk avoidance conflict, which do you prioritize? |
|
||||
|
||||
**Question analysis:** Variant A frames the question around whether avoiding risk is a hard constraint or a preference/trade-off — directly addressing the boundary distinction. Variant B frames it around priority ordering between growth and risk when they conflict — reframing the decision as relative importance rather than a boundary question. The model did not impose a reference wording for Variant B; it produced its natural framing from the broadened target.
|
||||
|
||||
### Stage 2 Results — Answer Resolution
|
||||
|
||||
| Variant | resolvedMeaning | targetResolved | remainingUncertainty |
|
||||
|---|---|---|---|
|
||||
| A (Precise) | Avoiding additional risk is established as a hard constraint. | true | null |
|
||||
| B (Broadened) | Risk avoidance takes absolute priority over business growth when they conflict. | true | null |
|
||||
|
||||
**Resolution analysis:** Both variants produced `targetResolved: true` with `remainingUncertainty: null`. The resolved meanings differ in wording but convey materially equivalent meaning for downstream reasoning: "avoiding additional risk is a hard constraint" and "risk avoidance takes absolute priority over business growth when they conflict" establish the same boundary — no more risk will be accepted. Neither resolution introduced unsupported wider consequences.
|
||||
|
||||
### Question Equivalence Classification
|
||||
|
||||
**`questions_materially_different`**
|
||||
|
||||
The precise target asked whether avoiding extra risk is a trade-off/preference or a hard constraint (a boundary question). The broadened target asked which objective has priority when they conflict (an ordering question). These ask the user to resolve different conceptual distinctions. The distinction was lost as predicted.
|
||||
|
||||
### Resolution Equivalence Classification
|
||||
|
||||
**`resolutions_materially_equivalent`**
|
||||
|
||||
Despite different questions, both resolved meanings from the same fixed answer establish the same downstream meaning: the user will not accept additional risk. For downstream reasoning — determining what can and cannot be done — this is equivalent. With the explicit hard-constraint answer used in this test, both target variants converged on materially equivalent resolved meaning.
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **What question did the precise target generate?** "Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?"
|
||||
2. **What question did the broadened target generate?** "When business growth and risk avoidance conflict, which do you prioritize?"
|
||||
3. **Did both questions ask the user to resolve the same underlying distinction?** No — one asked about boundary (constraint vs trade-off); the other asked about priority ordering.
|
||||
4. **Did the broader target turn preference-versus-constraint into simple priority ordering?** Yes — it reframed the distinction as relative importance rather than an absolute boundary.
|
||||
5. **What resolved meaning did Variant A produce from the fixed answer?** "Avoiding additional risk is established as a hard constraint." (targetResolved: true)
|
||||
6. **What resolved meaning did Variant B produce from the same answer?** "Risk avoidance takes absolute priority over business growth when they conflict." (targetResolved: true)
|
||||
7. **Were the two resolved meanings materially equivalent?** Yes — both establish that no additional risk will be accepted for growth.
|
||||
8. **Did either variant leave remaining uncertainty?** No — both returned null, indicating full resolution of the target from the fixed answer.
|
||||
9. **Did either variant introduce unsupported wider consequences?** No — neither inference extended beyond the source meaning and the user's explicit answer.
|
||||
10. **Does the broadening materially affect downstream clarification in this tested scenario?** No — despite different questions, the same answer produced the same downstream meaning.
|
||||
11. **Does this establish that broad targets are generally safe or unsafe?** No — only one scenario tested.
|
||||
12. **Does this establish how target generation should be changed?** No — no fix designed from these results.
|
||||
13. **Does this establish UI behaviour?** No — this is a clarification-target test only.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only one source scenario and one fixed answer were tested. Different sources may behave differently.
|
||||
- Only one ambiguity pattern (preference/constraint) was tested for downstream consequence. Other patterns not assessed.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434). Different models may behave differently.
|
||||
- The semantic equivalence classification is based on structured heuristic checks supplemented by the test output — for definitive judgment, human review of the actual resolved meanings is required.
|
||||
- One tested ambiguity pattern; broader safety/generalisation remains untested.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**The broader target changed the clarification question but not the resolved meaning for the tested explicit answer.**
|
||||
|
||||
The specificity loss (broadening) was confirmed: the precise target generated a boundary question ("constraint vs trade-off") and the broadened target generated an ordering question ("which to prioritize"). These are materially different questions. However, from the fixed user answer ("It's a hard constraint. I don't want any increase in risk."), both targets resolved to materially equivalent downstream meaning: no additional risk will be accepted. With the explicit hard-constraint answer used in this test, both target variants converged on materially equivalent resolved meaning.
|
||||
|
||||
The key finding is: **Does the distinction we lost actually matter? — In this tested scenario, it did not.** Specificity loss is not automatically a failure; it depends on whether it changes downstream meaning. Whether this holds across other scenarios and ambiguity patterns remains untested.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
Both variants produced `targetResolved: true` with zero remaining uncertainty and zero unsupported inferences. Resolutions were materially equivalent despite questions being materially different. 4/4 live inference calls completed successfully (all tests passed).
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Variant A's resolution ("Avoiding additional risk is established as a hard constraint") matches the expected outcome from Experiment 54V Case 1 and the handoff summary. Variant B's resolution ("Risk avoidance takes absolute priority over business growth when they conflict") represents a coarser framing — but not an incorrect one — for downstream use. The result confirms that the coarser representation remains workable even when it loses the preference-versus-constraint granularity.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 54Y entry; applied 54X wording corrections
|
||||
- `docs/current-handoff.md` — updated with Experiment 54Y summary and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as 54R–54X.
|
||||
|
||||
### Confirmation Semantic Instructions and Output Contracts Remained Unchanged
|
||||
|
||||
Both question and resolution instructions were identical to those defined in Experiments 54U and 54V. Output contracts unchanged from 54U (`{ "question": "..." }`) and 54V (`{ "resolvedMeaning", "targetResolved", "remainingUncertainty" }`).
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Specificity-Consequence Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No specificity-consequence logic entered any active runtime path, production module, or behaviour selection output.
|
||||
|
||||
---
|
||||
|
||||
### Return-to-Work Note (Experiment 54Y)
|
||||
|
||||
Experiments 54W/54X reproduced a broader priority framing for preference-versus-hard-constraint ambiguity; 54Y tested whether that specificity loss actually changes downstream clarification. The precise target generated a question asking whether avoiding risk is a hard constraint or trade-off; the broadened target asked which to prioritize when growth and risk conflict. The same fixed answer produced materially equivalent resolved meanings from both variants, so broadening did not matter in this scenario. Broader safety/generalisation remains untested. Behaviour Selection, graph, UI, and production integration remained untouched. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: feature/user-workspace-ux-v0.7. First test/file to inspect when resuming: tests/reconstruction/semantic-clarification-specificity-consequence.test.js for the full experiment and results. Status pending Rob's review.
|
||||
|
||||
## Experiment 54Z — Does Target Broadening Change Resolution When the Answer Is Less Explicit? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Experiment 54Y showed that with a very explicit hard-constraint answer ("It's a hard constraint. I don't want any increase in risk."), both precise and broadened clarification targets converged on materially equivalent resolved meaning — even though the generated questions were materially different.
|
||||
|
||||
This leaves one unresolved consequence:
|
||||
|
||||
> If the user's answer is less explicit, do those two different questions lead to materially different resolved meaning?
|
||||
|
||||
54Z tests that only. Passive and test-only. No redesign of target generation. No integration with Behaviour Selection, graph, or UI. No production code changes.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The precise and broadened questions may behave differently when the answer does not explicitly name the missing distinction. A weaker answer could:
|
||||
|
||||
- remain correctly unresolved under the precise question;
|
||||
- but be interpreted as a resolved priority decision under the broader question.
|
||||
|
||||
If that happens, target broadening has a real downstream consequence. If both variants preserve equivalent uncertainty, the broadening may be less consequential than expected.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as 54R–54Y)
|
||||
Model: `qwen-claude:latest` (same as 54R–54Y)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **4** live Ollama calls — one answer-resolution per variant × two weaker answers. No question-generation calls (questions are fixed from Experiment 54Y).
|
||||
|
||||
### Context Used
|
||||
|
||||
- `docs/current-handoff.md`
|
||||
- Experiment 54Y in `docs/design-evolution-log.md` (as basis for the unresolved consequence)
|
||||
- `tests/reconstruction/semantic-clarification-specificity-consequence.test.js` (structural reference: question generation and answer resolution helpers)
|
||||
- `tests/reconstruction/semantic-clarification-answer-resolution.test.js` (structural reference: instruction and output contract)
|
||||
|
||||
### Fixed Source
|
||||
|
||||
> I want the business to grow, but I don't want to take on more risk.
|
||||
|
||||
### Variant A — Precise Target
|
||||
|
||||
**Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
**Fixed question:** Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?
|
||||
|
||||
### Variant B — Broadened Target
|
||||
|
||||
**Clarification target:** priority between business growth and risk avoidance when they conflict.
|
||||
**Fixed question:** When business growth and risk avoidance conflict, which do you prioritize?
|
||||
|
||||
---
|
||||
|
||||
### Answer 1 — Priority Without Constraint Meaning
|
||||
|
||||
**User answer:** "Risk matters more to me."
|
||||
|
||||
**Expected behavior (human-reviewed):**
|
||||
This answer does not clearly establish whether risk avoidance is a hard constraint or merely a stronger preference. For Variant A, the precise target should therefore remain unresolved. For Variant B, the answer may legitimately resolve the priority target as "risk avoidance has higher priority than growth."
|
||||
|
||||
### Answer 1 Results
|
||||
|
||||
| Variant | resolvedMeaning | targetResolved | remainingUncertainty |
|
||||
|---|---|---|---|
|
||||
| A (Precise) | The user treats avoiding additional risk as a strong priority or preference rather than an absolute, non-negotiable constraint. | true | null |
|
||||
| B (Broadened) | Risk avoidance is prioritized over business growth when they conflict. | true | null |
|
||||
|
||||
### Answer 1 Analysis
|
||||
|
||||
Variant A interprets "Risk matters more to me" as meaning risk avoidance is a **strong preference/priority rather than an absolute constraint** — this maps correctly onto the precise target (preference/trade-off vs hard constraint). The model marked targetResolved=true because it interpreted the answer as settling the distinction toward "preference/trade-off."
|
||||
|
||||
Variant B interprets the same answer as meaning **risk avoidance has higher priority over growth when they conflict** — this maps correctly onto the broadened target (priority ordering).
|
||||
|
||||
**Classification: resolutions_materially_equivalent**
|
||||
|
||||
Both variants map the weak answer to a preference/priority-over-constraint interpretation. Neither resolves to "hard constraint." The resolved meanings use different framing but preserve the same downstream reasoning state: risk is not an absolute boundary, it is a prioritized consideration. For downstream use (what can/cannot be done), both produce equivalent uncertainty about whether risk could ever be accepted.
|
||||
|
||||
Both variants were flagged as `potential_erasal_of_uncertainty` because the answer was weak and both returned targetResolved=true with no remainingUncertainty — neither explicitly preserved the ambiguity about what "matters more" means in edge cases. **The precise target may have erased uncertainty by inferring that "Risk matters more to me" means risk is not a hard constraint; that conclusion was not explicitly supplied by the user.**
|
||||
|
||||
---
|
||||
|
||||
### Answer 2 — Conditional Trade-Off
|
||||
|
||||
**User answer:** "I'd normally avoid more risk, but for the right opportunity I might accept some."
|
||||
|
||||
**Expected behavior (human-reviewed):**
|
||||
This answer indicates risk avoidance is not an absolute hard constraint; some trade-off may be acceptable depending on the opportunity. For Variant A, this should resolve away from "hard constraint" while retaining conditionality. For Variant B, it may establish that risk is normally prioritized but can yield to growth in some cases.
|
||||
|
||||
### Answer 2 Results
|
||||
|
||||
| Variant | resolvedMeaning | targetResolved | remainingUncertainty |
|
||||
|---|---|---|---|
|
||||
| A (Precise) | Avoiding additional risk is a preference or trade-off rather than a hard constraint. | true | null |
|
||||
| B (Broadened) | Default priority is risk avoidance, with a conditional willingness to accept some risk for highly suitable opportunities. | false | It remains unclear how "the right opportunity" is defined and which factor strictly takes precedence when a specific growth opportunity carries significant risk. |
|
||||
|
||||
### Answer 2 Analysis
|
||||
|
||||
**This is the critical divergence.** Variant A collapses the conditional nature of the answer into a simple preference-vs-constraint resolution. The model says "preference or trade-off rather than a hard constraint" — but loses the key information that there are conditions (the right opportunity) under which even this preference could shift. This was flagged as `forced_certainty_detected` and `potential_erasal_of_uncertainty`.
|
||||
|
||||
Variant B preserves the conditionality ("conditional willingness") and correctly marks targetResolved=false because the answer does not establish a stable priority — the priority shifts depending on context. It also identifies remaining uncertainty about what constitutes "the right opportunity."
|
||||
|
||||
**Classification: resolutions_materially_different**
|
||||
|
||||
This is a material divergence. Variant A erases the conditional nature of the user's stated position and produces a flat preference-versus-constraint resolution. Variant B preserves both the default-priority-and-conditional structure AND the remaining uncertainty about when conditions change. For downstream reasoning, this matters because:
|
||||
- Under Variant A's meaning: risk avoidance = preference/trade-off → may be willing to accept risk in some cases (inferred)
|
||||
- Under Variant B's meaning: default priority risk, conditionally willing → conditional willingness is preserved explicitly
|
||||
|
||||
**However**, the divergence exists primarily in remainingUncertainty content, not in the resolvedMeaning itself. Both agree that risk avoidance is not a hard constraint. The difference is in whether the model preserves "there are conditions we don't yet understand" versus collapsing everything to "not a hard constraint."
|
||||
|
||||
---
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **For Answer 1, did Variant A correctly preserve uncertainty about preference versus hard constraint?** Partially. Variant A mapped the answer toward "preference/trade-off rather than absolute constraint" but marked it as fully resolved (targetResolved=true) with no remainingUncertainty — erasing the ambiguity that "matters more" doesn't define a boundary.
|
||||
|
||||
2. **For Answer 1, did Variant B resolve the broader priority target?** Yes. The broadened target ("priority between growth and risk") was resolved as "risk avoidance is prioritized over business growth when they conflict." This is a correct mapping from the weak answer to the priority frame.
|
||||
|
||||
3. **Did Answer 1 therefore create materially different resolution states between A and B?** No. Both map to the same downstream state: risk avoidance is not an absolute boundary but a stronger consideration than growth. The resolutions are materially equivalent for downstream reasoning about what can/cannot be done.
|
||||
|
||||
4. **For Answer 2, did Variant A correctly identify that risk avoidance is not an absolute hard constraint?** Partially correct on the outcome (not a hard constraint) but failed to preserve conditionality — the "might accept some" conditional was collapsed into a flat preference resolution with no remaining uncertainty.
|
||||
|
||||
5. **For Answer 2, did Variant B preserve the conditional nature of the priority?** Yes. Variant B preserved both the default-priority-and-conditional structure and identified remaining uncertainty about when conditions shift.
|
||||
|
||||
6. **Were the Answer 2 resolution states materially equivalent or different?** Different. Variant A erased conditionality; Variant B preserved it plus remainingUncertainty. This is a material divergence for downstream reasoning state.
|
||||
|
||||
7. **Did either variant force a weak answer into stronger meaning than the user supplied?** Yes — Variant A for Answer 2 collapsed conditional willingness ("might accept some") into a flat preference classification, erasing the conditionality layer.
|
||||
|
||||
8. **Did either variant erase uncertainty that should remain?** Yes — Variant A for both answers showed `potential_erasal_of_uncertainty`. For Answer 1, "Risk matters more to me" became a fully resolved preference (no remainingUncertainty). For Answer 2, conditionality was erased.
|
||||
|
||||
9. **Does target broadening have a material downstream consequence when answers are less explicit in these tested cases?** Yes — specifically for Answer 2 (conditional trade-off). The precise target question led the model to map to a flat preference-vs-constraint resolution and erase conditionality. The broadened target preserved conditional structure. This means target broadening has a real, asymmetrical consequence: the broadened question can actually preserve nuance that the precise question erases in this case.
|
||||
|
||||
10. **Does this establish that precise targets are always required?** No — Answer 1 showed no material divergence, and for Answer 2 the broader target preserved more nuance than the precise one. Neither is universally better.
|
||||
|
||||
11. **Does this establish how target-generation logic should be changed?** No — only two answers tested; neither variant was consistently better; no fix designed from these results.
|
||||
|
||||
12. **Does this establish UI behaviour?** No — this is a clarification-target test only.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only one source scenario and two user answers were tested. Different sources may behave differently.
|
||||
- Only one ambiguity pattern (preference/constraint) was tested with weak answers. Other patterns not assessed.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434). Different models may behave differently.
|
||||
- The asymmetric finding (broadened target preserving more nuance for Answer 2) is surprising and warrants further testing with additional answers that include explicit conditionality.
|
||||
- Two tested cases; broader generalisation remains untested.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Target broadening changed wording but not material resolution under weaker answers — except when the answer contained explicit conditionality.**
|
||||
|
||||
For Answer 1 ("Risk matters more to me."), both variants produced materially equivalent downstream meaning: risk avoidance is stronger than growth but not an absolute constraint. The broader target did not create a material divergence here.
|
||||
|
||||
For Answer 2 ("I'd normally avoid more risk, but for the right opportunity I might accept some."), the variants diverged. Variant A (precise) collapsed conditionality into a flat preference resolution and erased uncertainty. Variant B (broadened) preserved conditional structure and remaining uncertainty about what constitutes "the right opportunity."
|
||||
|
||||
**Unexpected finding:** The broadened target preserved more nuance than the precise target for the conditional answer. Neither framing was consistently superior across the two tested weaker answers.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
All 4 live inference calls completed successfully (all tests passed). Answer 1: materially equivalent resolutions from both variants. Answer 2: materially different resolutions — Variant A erased conditionality; Variant B preserved it. One forced certainty detection (Variant A on Answer 2).
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Compared to Experiment 54Y's explicit hard-constraint answer (where both variants converged), 54Z shows that convergence is fragile with weaker answers. Both tested weak answers confirmed the hypothesis: weak answers expose differences between precise and broadened targets, but only under specific content conditions. Neither variant was consistently superior across the two answers tested.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 54Z entry; applied 54Y wording corrections
|
||||
- `docs/current-handoff.md` — updated with Experiment 54Z summary and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as 54R–54Y.
|
||||
|
||||
### Confirmation Semantic Instructions and Output Contracts Remained Unchanged
|
||||
|
||||
Answer-resolution instruction identical to Experiment 54V. Output contract unchanged from 54V (`{ "resolvedMeaning", "targetResolved", "remainingUncertainty" }`).
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Weak-Answer Consequence Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No weak-answer consequence logic entered any active runtime path, production module, or behaviour selection output.
|
||||
|
||||
---
|
||||
+234
@@ -0,0 +1,234 @@
|
||||
## Experiment 55A — Can the Model Preserve Uncertainty From Weak Clarification Answers? (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
First, tighten Experiment 54Z so its conclusion does not over-generalise from two weaker answers.
|
||||
|
||||
Then isolate the strongest unresolved issue from 54Z:
|
||||
|
||||
> When a clarification answer is weak or conditional, can the model preserve uncertainty instead of forcing the answer into a stronger resolved meaning?
|
||||
|
||||
This experiment is about answer interpretation only. Do not compare precise versus broadened targets. Do not test question wording. Do not integrate with Behaviour Selection, graph, or UI.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The model may be too eager to convert weak clarification answers into resolved meaning. A better behaviour would preserve uncertainty when the answer does not actually settle the supplied target. If the model consistently keeps weak answers unresolved, the 54Z over-resolution may have been incidental. If it repeatedly forces resolution, that becomes a clearer limitation of the answer-resolution step itself.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as all prior experiments)
|
||||
Model: `qwen-claude:latest` (same as all prior experiments)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **4** live Ollama calls — one answer-resolution per case. Exactly 4 cases of varying strength against the same source/target/question.
|
||||
|
||||
### Context Used
|
||||
|
||||
- `docs/current-handoff.md`
|
||||
- Experiment 54Z in `docs/design-evolution-log.md` (as basis for isolating the unresolved issue)
|
||||
- `tests/reconstruction/semantic-clarification-weak-answer-consequence.test.js` (structural reference)
|
||||
- `tests/reconstruction/semantic-clarification-answer-resolution.test.js` (structural reference: instruction and output contract)
|
||||
|
||||
### Semantic Instruction Unchanged
|
||||
|
||||
> Use the user's clarification answer only to resolve the supplied clarification target. State the meaning now established by that answer. Mark targetResolved true only when the answer settles the target. Put any uncertainty that remains specifically about that target into remainingUncertainty; otherwise return null. Do not infer wider consequences, rewrite unrelated source meaning, recommend action, or generate another question.
|
||||
|
||||
No examples added. No stronger coaching attempted. Same instruction as Experiment 54V.
|
||||
|
||||
### Output Contract Unchanged
|
||||
|
||||
```json
|
||||
{
|
||||
"resolvedMeaning": "short statement",
|
||||
"targetResolved": true,
|
||||
"remainingUncertainty": "short statement or null"
|
||||
}
|
||||
```
|
||||
|
||||
### Fixed Source
|
||||
|
||||
> I want the business to grow, but I don't want to take on more risk.
|
||||
|
||||
### Clarification Target
|
||||
|
||||
> whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
### Clarification Question
|
||||
|
||||
> Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?
|
||||
|
||||
---
|
||||
|
||||
### Case 1 — Fully Explicit Hard Constraint
|
||||
|
||||
**User answer:** "It's a hard constraint. I don't want any increase in risk."
|
||||
|
||||
**Model output:**
|
||||
- resolvedMeaning: "Avoiding additional risk is a hard constraint."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
**Classification: resolution_correct**
|
||||
|
||||
The model correctly resolved the explicit hard constraint with appropriate strength and no spurious uncertainty. This is the positive control — it behaved as expected.
|
||||
|
||||
---
|
||||
|
||||
### Case 2 — Weak Priority Statement
|
||||
|
||||
**User answer:** "Risk matters more to me."
|
||||
|
||||
**Model output:**
|
||||
- resolvedMeaning: "Avoiding additional risk is prioritized as a strong preference or guiding principle rather than a rigid, non-negotiable constraint."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
**Classification: uncertainty_overresolved**
|
||||
|
||||
The model set `targetResolved=true` and produced a meaning stronger than the user supplied. "Risk matters more to me" establishes relative importance but does not determine whether avoiding additional risk is a hard constraint or a strong preference/trade-off. The model inferred "not a rigid, non-negotiable constraint" from a statement that only said risk was more important — which could be consistent with a strong preference that still functions as a de facto hard boundary in practice. This over-resolution matches the concern flagged in Experiment 54Z but isolated it here under a single fixed target to confirm the behaviour is real.
|
||||
|
||||
The model produced stronger meaning than the user supplied by declaring what risk avoidance is *not* (a rigid constraint) when the user only stated what it *is* relative to (more important than growth). This establishes that uncertainty should have been preserved with remainingUncertainty about whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
---
|
||||
|
||||
### Case 3 — Conditional Trade-Off
|
||||
|
||||
**User answer:** "I'd normally avoid more risk, but for the right opportunity I might accept some."
|
||||
|
||||
**Model output:**
|
||||
- resolvedMeaning: "Avoiding additional risk is a preference or trade-off rather than a hard constraint."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
|
||||
**Classification: resolution_correct with conditional qualification loss**
|
||||
|
||||
The target category was resolved correctly, but the answer's conditional qualification was lost. The user said "I might accept some for the right opportunity" — a specific conditional exception — which became "preference or trade-off rather than a hard constraint." The target (whether risk avoidance is preference/trade-off or hard constraint) was settled, but the nuance of how it functions conditionally was erased from the resolved meaning.
|
||||
|
||||
---
|
||||
|
||||
### Case 4 — Non-Answer / Insufficient Clarification
|
||||
|
||||
**User answer:** "I'm not really sure."
|
||||
|
||||
**Model output:**
|
||||
- resolvedMeaning: "The user has not yet determined if avoiding additional risk is a hard constraint or a preference/trade-off."
|
||||
- targetResolved: false
|
||||
- remainingUncertainty: "It remains unclear whether the user considers avoiding additional risk a hard constraint or a preference/trade-off."
|
||||
|
||||
**Classification: uncertainty_preserved**
|
||||
|
||||
The model correctly remained unresolved without inventing any position. The resolvedMeaning accurately describes what was supplied (nothing) and remainingUncertainty states that the distinction between hard constraint and preference/trade-off remains unresolved. This is honest behaviour.
|
||||
|
||||
---
|
||||
|
||||
### Evaluation Summary
|
||||
|
||||
| Case | Answer | Resolution Status | Classification |
|
||||
|---|---|---|---|
|
||||
| 1 | Explicit hard constraint | Correctly resolved | resolution_correct |
|
||||
| 2 | Weak priority ("Risk matters more to me.") | Over-resolved to targetResolved=true | uncertainty_overresolved |
|
||||
| 3 | Conditional trade-off | Resolved, conditionality lost | resolution_correct (with note) |
|
||||
| 4 | Non-answer ("I'm not really sure.") | Correctly unresolved | uncertainty_preserved |
|
||||
|
||||
**Classification counts:**
|
||||
- resolution_correct: 2
|
||||
- uncertainty_preserved: 1
|
||||
- uncertainty_overresolved: 1
|
||||
- resolution_failed: 0
|
||||
|
||||
---
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **Did Case 1 correctly resolve the explicit hard constraint?** Yes — targetResolved=true, appropriate meaning strength, no spurious uncertainty.
|
||||
|
||||
2. **Did Case 2 preserve uncertainty rather than infer that risk is not a hard constraint?** No — the model over-resolved, setting targetResolved=true and inferring "not a rigid, non-negotiable constraint" from a weak priority statement. This is the key finding: one tested weak answer received stronger meaning than the user supplied.
|
||||
|
||||
3. **Did Case 3 preserve the conditional nature of the trade-off?** Partially — the model correctly resolved the target (risk avoidance is not a hard constraint) but flattened the conditional qualification ("might accept some for the right opportunity") into flat "preference or trade-off" language. The conditional layer was lost even though target resolution was correct.
|
||||
|
||||
4. **Did Case 4 correctly remain unresolved?** Yes — no invented position, appropriate remainingUncertainty stating the distinction remains unresolved.
|
||||
|
||||
5. **How many cases were:**
|
||||
- resolution_correct: 2
|
||||
- uncertainty_preserved: 1
|
||||
- uncertainty_overresolved: 1
|
||||
- resolution_failed: 0
|
||||
|
||||
6. **Did any answer get stronger meaning than the user supplied?** Yes — Case 2 ("Risk matters more to me.") received a stronger interpretation than it justified. The model inferred "not a rigid, non-negotiable constraint" from a statement that only established relative priority.
|
||||
|
||||
7. **Did any unresolved answer incorrectly return targetResolved true?** No — Case 4 (the only truly unresolved case) correctly returned false. Case 2 over-resolved but did not remain unresolved.
|
||||
|
||||
8. **Did any resolved answer incorrectly retain uncertainty?** No — both resolved cases (1 and 3) correctly returned null for remainingUncertainty.
|
||||
|
||||
9. **Does the answer-resolution step appear biased toward resolution in these tested cases?** Yes — Case 2 demonstrates that a weak priority statement can be over-resolved to a definitive classification ("not a constraint") when it should have remained unresolved. One out of four cases showed this bias, but it appeared on the weakest-answer type where uncertainty preservation matters most.
|
||||
|
||||
10. **Does this establish how answer-resolution logic should be changed?** No — one weak-priority case over-resolved; this does not justify a broad change without broader testing.
|
||||
11. **Does this establish how the graph should represent unresolved clarification?** No — the unresolved representation question is separate from whether the model *should* remain unresolved.
|
||||
12. **Does this establish how Behaviour Selection should react?** No — this experiment did not integrate with Behaviour Selection.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only one source scenario and four answer types were tested. Different sources may behave differently.
|
||||
- Only one weak-answer pattern (priority without constraint meaning) over-resolved; other patterns not assessed.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434). Different models may behave differently.
|
||||
- Case 3 showed conditionality loss that is subtler than over-resolution — it resolved correctly but flattened nuance. This pattern warrants further testing with additional conditional answers.
|
||||
- Four cases tested; broader generalisation remains untested.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Clarification-answer resolution preserved uncertainty appropriately across some cases but over-resolved one weak answer.**
|
||||
|
||||
Case 1 (explicit hard constraint) and Case 4 (non-answer) behaved honestly — the explicit case resolved, the empty case remained unresolved. This is the expected baseline.
|
||||
|
||||
Case 2 (weak priority: "Risk matters more to me.") demonstrates the core limitation: the model over-resolved a weak answer, converting relative priority into a definitive negative ("not a rigid, non-negotiable constraint") when only relative importance was supplied. The model set targetResolved=true and erased uncertainty that should have been preserved.
|
||||
|
||||
Case 3 (conditional trade-off) resolved the target correctly but lost the conditional qualification in the process — "might accept some for the right opportunity" was flattened to "preference or trade-off rather than a hard constraint." This is a subtler form of meaning inflation: correct resolution with erasure of nuance.
|
||||
|
||||
Two of the four tested answers showed loss of nuance: one was over-resolved (Case 2) and one retained the correct target category while losing conditional qualification (Case 3). The remaining two cases behaved correctly (Cases 1 and 4). With qwen-claude:latest and the current semantic instruction, uncertainty preservation works for truly empty answers but is unreliable for weak-priority answers.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
All 4 live inference calls completed successfully. One over-resolution (Case 2), one conditionality loss (Case 3), two honest classifications (Cases 1 and 4). Total: ~62s, average: ~15.5s per call, fastest: 8.2s, slowest: 19.3s.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Compared to Experiment 54Z's finding that both variants over-resolved Case 1 ("Risk matters more to me."), Experiment 55A confirms this is a persistent issue under the same semantic instruction and model — not a side-effect of target framing. When a single fixed target was used, the weak priority answer still over-resolved (targetResolved=true with inferred "not a constraint" meaning). The same over-resolution reproduced with a fixed target, so target broadening is not required for the failure to occur.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 55A entry; applied 54Z wording corrections
|
||||
- `docs/current-handoff.md` — updated with Experiment 55A summary and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as all prior experiments.
|
||||
|
||||
### Confirmation Semantic Instruction and Output Contract Remained Unchanged
|
||||
|
||||
Answer-resolution instruction identical to Experiment 54V. No stronger coaching, no added examples. Output contract unchanged from 54V (`{ resolvedMeaning, targetResolved, remainingUncertainty }`).
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Uncertainty-Preservation Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No uncertainty-preservation logic entered any active runtime path, production module, or behaviour selection output.
|
||||
|
||||
---
|
||||
|
||||
### Return-to-Work Note (Experiment 55A)
|
||||
|
||||
54Z showed weaker answers can produce different downstream resolution states under different framings, suggesting target broadening matters; 55A isolated the answer-resolution step using one fixed target and four answers of varying strength (explicit hard constraint, weak priority, conditional trade-off, non-answer) to test whether uncertainty preservation holds independently of framing. The explicit case resolved correctly, the non-answer remained honestly unresolved, but the weak-priority case over-resolved by setting targetResolved=true and inferring "not a constraint" from relative importance alone. Conditional language was also flattened even when resolution was correct. Uncertainty preservation remains unreliable for weak-priority answers. Graph, Behaviour Selection, UI, and production integration remain untouched. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: feature/user-workspace-ux-v0.7. First file to inspect when resuming: tests/reconstruction/semantic-clarification-uncertainty-preservation.test.js for the full experiment and results. Status pending Rob's review.
|
||||
|
||||
---
|
||||
|
||||
+219
@@ -0,0 +1,219 @@
|
||||
## Experiment 55B — Separate Answer Meaning from Resolution Judgement (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Experiment 55A showed one weak answer was over-resolved and one conditional answer lost nuance. The remaining question: is the distortion introduced when the model restates the answer's meaning, or only when it decides whether the clarification target is resolved?
|
||||
|
||||
55B separates these two steps using independent calls per case:
|
||||
- **Mode A** (meaning-only): "State what the user's answer establishes. Preserve uncertainty and qualification exactly." No resolution decision.
|
||||
- **Mode B** (resolution): The same instruction and contract as Experiment 54V/55A.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The model may preserve weak and conditional meaning correctly when asked only to restate what the user established, and only become over-confident when asked whether the clarification target is resolved. If so, the problem lies in converting meaning into a `targetResolved` judgement, not in interpreting the answer itself. If the meaning-only output already strengthens or flattens the answer, then the problem occurs earlier.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as all prior experiments)
|
||||
Model: `qwen-claude:latest` (same as all prior experiments)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **6** live Ollama calls — one meaning-only call and one resolution call per case, three cases. The calls are independent; Mode A output is never fed into Mode B.
|
||||
|
||||
### Fixed Source / Target / Question (identical to 55A)
|
||||
|
||||
- **Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
- **Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
- **Clarification question:** Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?
|
||||
|
||||
### Mode A instruction
|
||||
|
||||
> State only what the user's answer establishes in relation to the clarification question. Preserve uncertainty, conditionality, and qualification exactly as supplied. Do not decide whether the clarification target is resolved. Do not infer what the user did not say. Do not recommend action or generate another question.
|
||||
>
|
||||
> Return valid JSON only in this shape: { "answerMeaning": "short statement" }
|
||||
|
||||
### Mode A output contract
|
||||
|
||||
```json
|
||||
{
|
||||
"answerMeaning": "short statement"
|
||||
}
|
||||
```
|
||||
|
||||
### Mode B instruction unchanged
|
||||
|
||||
Same as Experiment 54V / 55A resolution instruction. Output contract unchanged (`resolvedMeaning`, `targetResolved`, `remainingUncertainty`).
|
||||
|
||||
---
|
||||
|
||||
### Case 1 — Weak Priority
|
||||
|
||||
**Answer:** "Risk matters more to me."
|
||||
|
||||
**Mode A result:**
|
||||
- answerMeaning: "The user indicates that avoiding additional risk holds greater importance to them than other factors."
|
||||
- Classification: **meaning_preserved**
|
||||
|
||||
The meaning preserved relative importance without deciding whether it is a hard constraint or not. No strengthening or loss detected.
|
||||
|
||||
**Mode B result:**
|
||||
- resolvedMeaning: "Avoiding additional risk is established as a strongly prioritized preference over growth, rather than an explicitly stated boundary."
|
||||
- targetResolved: false
|
||||
- remainingUncertainty: "Whether this priority functions as a non-negotiable hard constraint or remains a flexible trade-off in certain situations."
|
||||
- Classification: **resolution_correct**
|
||||
|
||||
The resolution stayed unresolved with appropriate uncertainty. Interestingly, the resolved meaning is slightly stronger than the Mode A output ("strongly prioritized preference" vs "greater importance"), but `targetResolved` was correctly false — no over-resolution occurred this time.
|
||||
|
||||
---
|
||||
|
||||
### Case 2 — Conditional Trade-Off
|
||||
|
||||
**Answer:** "I'd normally avoid more risk, but for the right opportunity I might accept some."
|
||||
|
||||
**Mode A result:**
|
||||
- answerMeaning: "The user indicates that avoiding additional risk is generally a preference that may become a trade-off under specific conditions, rather than an absolute constraint."
|
||||
- Classification: **meaning_preserved**
|
||||
|
||||
Both the normal preference and the conditional exception ("under specific conditions") were preserved. The meaning captures all of what the user established without flattening.
|
||||
|
||||
**Mode B result:**
|
||||
- resolvedMeaning: "Avoiding additional risk is a preference or trade-off rather than a hard constraint."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
- Classification: **resolution_meaning_loss**
|
||||
|
||||
The target was correctly resolved (`targetResolved=true`). However, the conditional qualification ("for the right opportunity I might accept some") was **flattened** into a flat "preference or trade-off" statement. Mode A preserved "generally a preference that may become a trade-off under specific conditions"; Mode B collapsed it to "a preference or trade-off."
|
||||
|
||||
This is a direct comparison within the same case: **meaning was preserved in Mode A but lost during resolution.**
|
||||
|
||||
---
|
||||
|
||||
### Case 3 — Non-Answer
|
||||
|
||||
**Answer:** "I'm not really sure."
|
||||
|
||||
**Mode A result:**
|
||||
- answerMeaning: "The user is uncertain whether avoiding additional risk is a hard constraint or a preference or trade-off."
|
||||
- Classification: **meaning_preserved**
|
||||
|
||||
Simple uncertainty was preserved accurately. No position was invented.
|
||||
|
||||
**Mode B result:**
|
||||
- resolvedMeaning: "The user is unsure whether avoiding additional risk is a hard constraint or a preference/trade-off."
|
||||
- targetResolved: false
|
||||
- remainingUncertainty: "Whether avoiding additional risk is a hard constraint or a preference/trade-off remains unclear."
|
||||
- Classification: **resolution_correct**
|
||||
|
||||
Correctly unresolved. No meaning distortion detected in either mode.
|
||||
|
||||
---
|
||||
|
||||
### Evaluation Summary
|
||||
|
||||
| Case | Answer | Mode A | Mode A Class | Mode B | Mode B Class |
|
||||
|------|--------|--------|-------------|--------|-------------|
|
||||
| 1 | "Risk matters more to me." | Preserved relative priority | meaning_preserved | Correctly unresolved | resolution_correct |
|
||||
| 2 | "I'd normally avoid more risk, but for the right opportunity I might accept some." | Preserved conditional qualification | meaning_preserved | Resolved but flattened conditionality | resolution_meaning_loss |
|
||||
| 3 | "I'm not really sure." | Preserved uncertainty | meaning_preserved | Correctly unresolved | resolution_correct |
|
||||
|
||||
**Meaning counts (Mode A):**
|
||||
- meaning_preserved: 3
|
||||
- meaning_strengthened: 0
|
||||
- meaning_lost: 0
|
||||
|
||||
**Resolution counts (Mode B):**
|
||||
- resolution_correct: 2
|
||||
- resolution_meaning_loss: 1
|
||||
- resolution_overresolved: 0
|
||||
- resolution_underresolved: 0
|
||||
|
||||
---
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **Did Case 1 Mode A preserve only relative priority without deciding hard-constraint status?** Yes — "greater importance to them than other factors" preserves the relative priority without declaring anything about hard constraint status.
|
||||
|
||||
2. **Did Case 1 Mode B over-resolve the target again?** No — this run returned `targetResolved=false` with appropriate remaining uncertainty. The resolved meaning was slightly stronger ("strongly prioritized preference") but did not cross into definitive classification. (Note: this differs from the 55A run on the same case, which had over-resolved to `targetResolved=true`. This may indicate some instability in the resolution step across runs.)
|
||||
|
||||
3. **Did Case 2 Mode A preserve the conditional "for the right opportunity" qualification?** Yes — "generally a preference that may become a trade-off under specific conditions" preserves both the normal stance and the conditional exception.
|
||||
|
||||
4. **Did Case 2 Mode B preserve or flatten that same conditionality?** Flattened. Mode B collapsed "I'd normally avoid more risk, but for the right opportunity I might accept some" into "a preference or trade-off rather than a hard constraint." The conditional qualification ("for the right opportunity") was lost during resolution.
|
||||
|
||||
5. **Did Case 3 Mode A preserve simple uncertainty?** Yes — "The user is uncertain whether avoiding additional risk is a hard constraint or a preference or trade-off" preserves the lack of position without inventing one.
|
||||
|
||||
6. **Did Case 3 Mode B correctly remain unresolved?** Yes — `targetResolved=false` with accurate remainingUncertainty. No meaning distortion in either mode.
|
||||
|
||||
7. **In any case, was meaning already distorted before the resolution judgement?** No — all three cases preserved their meaning accurately in Mode A (meaning_preserved: 3). The first material information loss appeared only during the resolution step.
|
||||
|
||||
8. **In any case, did meaning remain accurate in Mode A but become stronger or flatter in Mode B?** Yes — Case 2 is the clearest example. Mode A preserved "generally a preference that may become a trade-off under specific conditions"; Mode B flattened it to "a preference or trade-off."
|
||||
|
||||
9. **Does this isolate the failure to the resolution judgement?** Partially yes, for the case of conditional meaning loss. The comparison within Case 2 (same answer, same model, independent calls) shows that meaning can be preserved in isolation and then lost when a resolution decision is introduced. However, only one instance of meaning-preserved-but-resolution-flattened was observed; broader generalisation requires more tested cases.
|
||||
|
||||
10. **Does this establish how production logic should be redesigned?** No — the evidence from three answers is insufficient to justify specific production changes. Further testing with additional answer types and different models would be needed before redesigning any logic.
|
||||
|
||||
11. **Does this establish graph representation or Behaviour Selection changes?** No — this experiment did not integrate with graph, Behaviour Selection, or any other engine component.
|
||||
|
||||
---
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only three answers were tested. Different answer patterns may behave differently.
|
||||
- The weak-priority case in 55B resolved correctly (unlike 55A which over-resolved it), suggesting the resolution step may have some instability across runs with the same configuration.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434).
|
||||
- The conditional trade-off answer is a specific pattern; other conditional phrasings may behave differently.
|
||||
- Six live calls total — insufficient for broader generalisation.
|
||||
|
||||
---
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Meaning-only extraction preserved all three tested answers; one conditional answer then lost qualification during the independent resolution judgement.**
|
||||
|
||||
All three tested answers preserved their meaning correctly in Mode A (meaning_preserved: 3/3). The only information loss appeared in Case 2 when transitioning from meaning-only to resolution: the conditional qualification "for the right opportunity I might accept some" was present and preserved by Mode A, then flattened to a flat "preference or trade-off" statement during resolution.
|
||||
|
||||
Additionally, Case 1 produced different resolution outcomes across experiments (55A over-resolved; 55B correctly unresolved), suggesting the resolution step exhibits some run-to-run instability under the same configuration — an observation worth monitoring but not yet actionable without more data.
|
||||
|
||||
Separating the two experimentally was useful for locating where the observed meaning loss first appeared.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
All 6 live inference calls completed successfully. Three answers tested independently through two modes each. Meaning preservation was perfect across Mode A (3/3). Resolution introduced one meaning-loss case (Case 2) and correctly handled the other two. Total: ~104s, average: ~17.3s per call, fastest: 11.5s, slowest: 24.1s.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Compared to Experiment 55A's conclusion that "the answer-resolution step appears biased toward resolution," Experiment 55B shows this bias is not universal: Case 1 did not over-resolve in the 55B run, and Case 3 was correct in both experiments. The specific loss pattern (conditional meaning preserved in isolation but flattened during resolution) appeared only in Case 2. This narrows the failure from "biased toward resolution" to a more specific pattern: conditional nuance is vulnerable to flattening specifically when the model is forced to make a target-resolution decision.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 55B entry; applied wording corrections to Experiment 55A
|
||||
- `docs/current-handoff.md` — updated with Experiment 55B summary and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as all prior experiments.
|
||||
|
||||
### Confirmation Semantic Instruction and Output Contract Remained Unchanged (for Mode B)
|
||||
|
||||
Mode B instruction and output contract identical to Experiment 54V / 55A. No production code changed.
|
||||
|
||||
Mode A used a new minimal instruction and output contract specific to this experiment only. It does not replace any existing mechanism.
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Meaning-Resolution Separation Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No meaning-resolution separation logic entered any active runtime path, production module, or behaviour selection output. Production continues using the pre-existing combined instruction and contract.
|
||||
|
||||
---
|
||||
|
||||
+234
@@ -0,0 +1,234 @@
|
||||
## Experiment 55C — Resolution From Preserved Answer Meaning (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Experiment 55B showed meaning-only extraction preserved all three tested answers while one conditional answer lost qualification during independent resolution. The remaining question: if the preserved answer meaning is explicitly carried into the resolution step, does the later judgement still flatten it?
|
||||
|
||||
This tests whether carrying semantic state forward across two calls eliminates the conditionality loss, introduces a new failure mode, or changes neither.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A two-step test-only chain may preserve meaning better because the resolution stage no longer needs to reinterpret the raw answer. Possible outcomes: preserved meaning survives resolution; the resolution stage still flattens it; some answers improve while others do not. Any outcome is useful.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as all prior experiments)
|
||||
Model: `qwen-claude:latest` (same as all prior experiments)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **6** live Ollama calls — one meaning call and one resolution call per case, three cases. Stage 2 uses the actual answerMeaning from Stage 1, not a human reference.
|
||||
|
||||
### Fixed Source / Target / Question (identical to 55A/55B)
|
||||
|
||||
- **Source:** "I want the business to grow, but I don't want to take on more risk."
|
||||
- **Clarification target:** whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
- **Clarification question:** Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?
|
||||
|
||||
### Stage 1 instruction (unchanged from 55B Mode A)
|
||||
|
||||
> State only what the user's answer establishes in relation to the clarification question. Preserve uncertainty, conditionality, and qualification exactly as supplied. Do not decide whether the clarification target is resolved. Do not infer what the user did not say. Do not recommend action or generate another question.
|
||||
>
|
||||
> Return valid JSON only in this shape: { "answerMeaning": "short statement" }
|
||||
|
||||
### Stage 1 output contract
|
||||
|
||||
```json
|
||||
{
|
||||
"answerMeaning": "short statement"
|
||||
}
|
||||
```
|
||||
|
||||
### Stage 2 instruction (new — uses preserved meaning, not raw answer)
|
||||
|
||||
> Decide whether the supplied preserved answer meaning settles the clarification target. Treat the preserved meaning as the full extent of what has been established — do not strengthen, simplify, or reinterpret it. Mark targetResolved true only when that preserved meaning settles the target. If any part of the target remains unresolved, preserve that uncertainty. Keep qualifications and conditions intact.
|
||||
>
|
||||
> Return valid JSON only in this shape: { "resolvedMeaning": "short statement", "targetResolved": true, "remainingUncertainty": null }
|
||||
|
||||
### Stage 2 output contract
|
||||
|
||||
```json
|
||||
{
|
||||
"resolvedMeaning": "short statement",
|
||||
"targetResolved": true,
|
||||
"remainingUncertainty": null
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Case 1 — Weak Priority
|
||||
|
||||
**Raw answer:** "Risk matters more to me."
|
||||
|
||||
**Stage 1 result:**
|
||||
- answerMeaning: "The user indicates that risk carries greater weight, suggesting it leans toward being a significant preference or trade-off rather than explicitly confirming an absolute hard constraint."
|
||||
- Classification: **meaning_strengthened**
|
||||
|
||||
Stage 1 went beyond relative priority by adding language about "preference or trade-off" and "absolute hard constraint" — content the user did not supply. This was also seen in 55A's Stage 2, but now it appears at Stage 1. The weak-priority case has shown run-to-run variation across Experiments 55A–55C.
|
||||
|
||||
**Stage 2 result (input = actual Stage 1 meaning):**
|
||||
- resolvedMeaning: "Avoiding additional risk leans toward being a significant preference or trade-off rather than an absolute hard constraint."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
- Classification: **resolution_overresolved**
|
||||
|
||||
Because Stage 1 already strengthened the answer, Stage 2 treated that strengthened meaning as input and over-resolved with `targetResolved=true` and no remaining uncertainty. If Stage 1 distorts the answer, Stage 2 may preserve and act on that distortion rather than correct it.
|
||||
|
||||
---
|
||||
|
||||
### Case 2 — Conditional Trade-Off
|
||||
|
||||
**Raw answer:** "I'd normally avoid more risk, but for the right opportunity I might accept some."
|
||||
|
||||
**Stage 1 result:**
|
||||
- answerMeaning: "Avoiding additional risk is treated as a normal preference that may be conditionally accepted for the right opportunity."
|
||||
- Classification: **meaning_preserved**
|
||||
|
||||
Both the normal preference and conditional exception were preserved. The wording uses "treated as" rather than directly repeating the user's phrasing but captures the full semantic content.
|
||||
|
||||
**Stage 2 result (input = actual Stage 1 meaning):**
|
||||
- resolvedMeaning: "Avoiding additional risk is a preference rather than a hard constraint, as it may be conditionally accepted for the right opportunity."
|
||||
- targetResolved: true
|
||||
- remainingUncertainty: null
|
||||
- Classification: **resolution_correct**
|
||||
|
||||
Target was correctly resolved with the conditional qualification ("as it may be conditionally accepted for the right opportunity") retained in the resolved meaning. This is the key result: the two-stage chain preserved the conditionality that 55B's Mode B flattened.
|
||||
|
||||
---
|
||||
|
||||
### Case 3 — Non-Answer
|
||||
|
||||
**Raw answer:** "I'm not really sure."
|
||||
|
||||
**Stage 1 result:**
|
||||
- answerMeaning: "The user states they are not really sure whether avoiding additional risk is a hard constraint or a preference/trade-off."
|
||||
- Classification: **meaning_preserved**
|
||||
|
||||
Uncertainty preserved accurately. The model added the clarification target context ("whether...is a hard constraint or a preference/trade-off") which is reasonable contextual framing for the non-answer.
|
||||
|
||||
**Stage 2 result (input = actual Stage 1 meaning):**
|
||||
- resolvedMeaning: "The user is uncertain whether avoiding additional risk is a preference/trade-off or a hard constraint."
|
||||
- targetResolved: false
|
||||
- remainingUncertainty: "Whether avoiding additional risk is a preference/trade-off or a hard constraint remains unresolved."
|
||||
- Classification: **resolution_correct**
|
||||
|
||||
Correctly unresolved with appropriate remaining uncertainty. No meaning distortion in either stage.
|
||||
|
||||
---
|
||||
|
||||
### Evaluation Summary
|
||||
|
||||
| Case | Answer | Stage 1 | Stage 1 Class | Stage 2 Result | Stage 2 Class |
|
||||
|------|--------|---------|--------------|----------------|---------------|
|
||||
| 1 | "Risk matters more to me." | Strengthened beyond priority | meaning_strengthened | Over-resolved (targetResolved=true) | resolution_overresolved |
|
||||
| 2 | "I'd normally avoid more risk, but for the right opportunity I might accept some." | Preserved conditionality | meaning_preserved | Resolved with qualification retained | resolution_correct |
|
||||
| 3 | "I'm not really sure." | Preserved uncertainty | meaning_preserved | Correctly unresolved | resolution_correct |
|
||||
|
||||
**Meaning counts (Stage 1):**
|
||||
- meaning_preserved: 2
|
||||
- meaning_strengthened: 1
|
||||
- meaning_lost: 0
|
||||
|
||||
**Resolution counts (Stage 2):**
|
||||
- resolution_correct: 2
|
||||
- resolution_overresolved: 1
|
||||
- resolution_meaning_loss: 0
|
||||
- resolution_underresolved: 0
|
||||
|
||||
---
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **Did Case 1 Stage 1 preserve only relative priority?** No — it strengthened beyond relative priority by introducing "preference or trade-off" and "absolute hard constraint" language not present in the user's answer.
|
||||
|
||||
2. **Did Case 1 Stage 2 remain unresolved?** No — targetResolved=true because the distorted Stage 1 input led the model to conclude resolution was achieved.
|
||||
|
||||
3. **Did Case 2 Stage 1 preserve the conditional qualification?** Yes — "normal preference that may be conditionally accepted for the right opportunity" preserved both the default stance and the exception.
|
||||
|
||||
4. **Did Case 2 Stage 2 retain that qualification while resolving the target?** Yes — resolved meaning explicitly retained "as it may be conditionally accepted for the right opportunity." This is a direct improvement over 55B's Mode B (resolution_meaning_loss → resolution_correct).
|
||||
|
||||
5. **Did Case 3 preserve uncertainty through both stages?** Yes — Stage 1 preserved uncertainty, Stage 2 correctly returned targetResolved=false with remainingUncertainty.
|
||||
|
||||
6. **Did any Stage 2 output become stronger than its actual Stage 1 input?** No — manual semantic review found no case where Stage 2 strengthened beyond the actual Stage 1 meaning. The chaining check confirmed this explicitly for all three cases.
|
||||
|
||||
7. **Did any Stage 2 output flatten a condition present in Stage 1?** No — Case 2's condition survived both stages intact. This is the key positive finding.
|
||||
|
||||
8. **Compared with 55B, did carrying preserved meaning forward remove the observed conditionality loss?** Yes — in 55B Mode B, Case 2 was resolution_meaning_loss (flattened). In 55C Stage 2, Case 2 was resolution_correct with qualification retained. The two-stage chain eliminated this specific failure mode for the tested answer.
|
||||
|
||||
9. **Compared with 55A/55B, did the weak-priority case remain honestly unresolved?** No — in 55B, Case 1 was resolution_correct (unresolved) in that run; in 55C, it over-resolved because Stage 1 distorted the meaning first. The weak-priority problem is not solved by this approach.
|
||||
|
||||
10. **Does this prove a two-stage production design is required?** No — evidence from three answers is insufficient to justify specific production changes.
|
||||
|
||||
11. **Does this establish graph representation?** No — this experiment did not integrate with graph, Behaviour Selection, or any other engine component.
|
||||
|
||||
12. **Does this establish Behaviour Selection changes?** No — Behaviour Selection was not called or referenced.
|
||||
|
||||
---
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only three answers were tested. Different answer patterns may behave differently.
|
||||
- Case 1 revealed a new failure mode: if Stage 1 distorts meaning, Stage 2 amplifies it through chaining. This is not an improvement over 55B's approach for weak answers.
|
||||
- The conditional trade-off improvement (Case 2) may not generalise to other conditional patterns.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434).
|
||||
- Six live calls total — insufficient for broader generalisation.
|
||||
- Case 3 Stage 1 added contextual framing ("whether...is a hard constraint or a preference/trade-off") to the non-answer, which could be questioned as mild interpretation even though it preserved uncertainty correctly.
|
||||
|
||||
---
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Preserved meaning improved resolution but some loss remained.**
|
||||
|
||||
The two-stage chain successfully eliminated the conditionality loss observed in 55B: Case 2's conditional qualification survived through both stages and resolved correctly (resolution_correct). Non-answer uncertainty was also preserved through both stages (resolution_correct). These are genuine improvements.
|
||||
|
||||
However, the weak-priority case revealed a new failure mode: Stage 1 strengthened "Risk matters more to me." into language about "preference or trade-off rather than absolute hard constraint," and Stage 2 then over-resolved based on that distorted input. Carrying semantic state forward means distortion propagates as well as fidelity. This does not improve the weak-priority problem relative to 55B's Mode B (which correctly left Case 1 unresolved in its run).
|
||||
|
||||
The core finding is asymmetric: preserving meaning before resolution helps for conditional answers (eliminates flattening) and non-answers (preserves uncertainty), but does not help — and may worsen outcomes — when the meaning extraction step itself distorts. The question "does the judgement stop rewriting the meaning?" is answered partially: it stops rewriting when the input to judgement already carries the full meaning, but it amplifies rewriting when that input is itself distorted.
|
||||
|
||||
Does this prove a two-stage production design is required? **No.** The weak-priority case over-resolved in 55C while remaining unresolved in the 55B run — and neither result establishes which approach is better for all cases.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
All 6 live inference calls completed successfully. Three answers tested through chained meaning→resolution stages. Stage 2 eliminated the conditionality loss from 55B (Case 2: resolution_meaning_loss → resolution_correct) but did not eliminate over-resolution for weak-priority input when Stage 1 strengthened it first (Case 1: meaning_strengthened → resolution_overresolved). Non-answer uncertainty was preserved through both stages. Total: ~117s, average: ~19.5s per call, fastest: 15.0s, slowest: 23.8s.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Compared to Experiment 55B's finding that "meaning was preserved in Mode A but lost during resolution," Experiment 55C shows that carrying preserved meaning forward eliminates that specific loss pattern for conditional answers (Case 2 resolved correctly with qualification retained) but introduces a different asymmetry: Stage 1 distortion propagates through Stage 2. The weak-priority case improved relative to 55A's over-resolution but degraded relative to the 55B run's correct unresolved result. Neither two-stage approach consistently outperforms the other across all tested answer types.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 55C entry; applied corrected wording to Experiment 55B
|
||||
- `docs/current-handoff.md` — updated with Experiment 55C summary, corrected 55B wording, and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as all prior experiments.
|
||||
|
||||
### Confirmation Semantic Instruction and Output Contract Remained Unchanged (for Stage 1)
|
||||
|
||||
Stage 1 instruction identical to Experiment 55B Mode A. No production code changed.
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Preserved-Meaning Resolution Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No preserved-meaning resolution logic entered any active runtime path, production module, or behaviour selection output. Production continues using the pre-existing combined instruction and contract.
|
||||
|
||||
---
|
||||
|
||||
### Return-to-Work Note (Experiment 55C)
|
||||
|
||||
55A showed one weak answer was over-resolved and one conditional answer lost nuance; 55B separated answer meaning from target-resolution judgement using independent calls. All three tested meanings were preserved in Mode A — the weak priority ("risk matters more"), the conditional trade-off ("for the right opportunity I might accept some"), and the non-answer uncertainty. The first material information loss appeared only when deciding target resolution: Case 2's conditional qualification was preserved by the meaning-only call but flattened during resolution. This suggests the distortion occurs in the resolution judgement step rather than the meaning extraction step, though the pattern was observed for only one case. Whether other answer types show the same pattern remains unproven. Graph, Behaviour Selection, UI and production remain untouched. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: feature/user-workspace-ux-v0.7, commit 3af623a. First test/file to inspect when resuming: tests/reconstruction/semantic-preserved-meaning-resolution.test.js for the full experiment and results. Status pending Rob's review.
|
||||
|
||||
+340
@@ -0,0 +1,340 @@
|
||||
## Experiment 55D — Separate Stated Clarification Meaning from Inference (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Experiment 55C showed that when Stage 1 preserved the user's meaning accurately, carrying that meaning forward protected conditionality during resolution. But for "Risk matters more to me.", Stage 1 itself added meaning about preference/trade-off rather than hard constraint — content not supplied by the user. The unresolved question is now one step earlier: can the first interpretation step distinguish what the user actually established from what merely seems plausible?
|
||||
|
||||
This experiment tests interpretation only. No target resolution, no question generation, no production changes.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The model may interpret weak answers more faithfully if it keeps meaning directly established by the answer and possible implications not directly established in separate fields. If the weak-priority answer remains cleanly stated while the "not a hard constraint" idea moves into a separate inference field, that would show the information can be represented without silently strengthening the user's meaning. If strengthening still appears inside stated meaning, the problem occurs even with explicit separation.
|
||||
|
||||
### Configuration
|
||||
|
||||
Host: `http://192.168.1.111:11434` (same as all prior experiments)
|
||||
Model: `qwen-claude:latest` (same as all prior experiments)
|
||||
|
||||
### Number of Live Inference Calls
|
||||
|
||||
Exactly **4** live Ollama calls — one per answer. Single call per case (no two-stage chain).
|
||||
|
||||
### Fixed Clarification Question
|
||||
|
||||
> Do you view avoiding additional risk as a hard constraint, or as a preference or trade-off?
|
||||
|
||||
Clarification target context: whether avoiding additional risk is a preference/trade-off or a hard constraint.
|
||||
|
||||
### Output Contract
|
||||
|
||||
```json
|
||||
{
|
||||
"statedMeaning": "short statement",
|
||||
"possibleInference": "short statement or null"
|
||||
}
|
||||
```
|
||||
|
||||
### Semantic Instruction Used
|
||||
|
||||
> State only what the user's answer directly establishes in statedMeaning. Preserve uncertainty, qualification, and conditionality. Do not turn relative importance into a hard boundary or the absence of one unless the user actually says so. If there is a plausible implication that goes beyond what the answer directly establishes, place it only in possibleInference. Do not decide whether the clarification target is resolved. Do not recommend action or generate another question.
|
||||
|
||||
---
|
||||
|
||||
### Case 1 — Weak Priority
|
||||
|
||||
**Raw answer:** "Risk matters more to me."
|
||||
|
||||
**statedMeaning:** "The user considers risk to be of greater importance than other factors."
|
||||
**possibleInference:** "This may suggest that avoiding additional risk is weighted as a stricter constraint rather than a flexible preference, though this remains unconfirmed."
|
||||
**Classification:** stated_meaning_preserved / inference_cleanly_separated
|
||||
|
||||
statedMeaning establishes only relative importance — no hard constraint or not-a-hard-constraint boundary. possibleInference contains the stronger implication about stricter constraint, properly qualified with "may suggest" and "remains unconfirmed". The user's relative priority language was not strengthened in statedMeaning. The model did generate a plausible inference (stricter constraint weighting) but kept it separate from what the answer directly establishes.
|
||||
|
||||
---
|
||||
|
||||
### Case 2 — Conditional Trade-Off
|
||||
|
||||
**Raw answer:** "I'd normally avoid more risk, but for the right opportunity I might accept some."
|
||||
|
||||
**statedMeaning:** "Avoiding additional risk is generally preferred but can be accepted for the right opportunity, qualifying it as a preference or trade-off rather than a hard constraint."
|
||||
**possibleInference:** "The user may weigh potential rewards against new risks to decide when increased exposure is justified."
|
||||
**Classification:** stated_meaning_preserved (semantic review) / inference_cleanly_separated
|
||||
|
||||
statedMeaning preserved both the normal preference (generally preferred to avoid risk) and the conditional exception (can be accepted for the right opportunity). It also directly established that avoiding all additional risk is not absolute — because the user explicitly said they might accept some. The additional clause "qualifying it as a preference or trade-off rather than a hard constraint" goes slightly beyond what was explicitly stated, but the core meaning (both sides of the condition) remains intact. possibleInference adds a reasonable speculative implication about weighing rewards against risks, cleanly separated.
|
||||
|
||||
---
|
||||
|
||||
### Case 3 — Explicit Hard Constraint
|
||||
|
||||
**Raw answer:** "It's a hard constraint. I don't want any increase in risk."
|
||||
|
||||
**statedMeaning:** "Avoiding additional risk is a hard constraint, with no increase in risk acceptable."
|
||||
**possibleInference:** "Proposed options will strictly exclude any that add risk, potentially limiting flexibility on other objectives."
|
||||
**Classification:** stated_meaning_preserved / unnecessary_inference
|
||||
|
||||
statedMeaning directly and correctly establishes the hard constraint. possibleInference adds a consequential implication about proposed options excluding risk — a reasonable inference but unnecessary for an explicit answer where none was lacking. This is the first case where possibleInference goes beyond what was warranted; it should normally be null here.
|
||||
|
||||
---
|
||||
|
||||
### Case 4 — Non-Answer
|
||||
|
||||
**Raw answer:** "I'm not really sure."
|
||||
|
||||
**statedMeaning:** "The user expresses uncertainty about whether avoiding additional risk is a hard constraint or a preference/trade-off."
|
||||
**possibleInference:** "The user may require further details on the potential risks or alternatives to clarify their position."
|
||||
**Classification:** stated_meaning_preserved / unnecessary_inference
|
||||
|
||||
statedMeaning preserves uncertainty without inventing a preference, constraint, or likely leaning. However, it adds contextual framing ("about whether avoiding additional risk is a hard constraint or a preference/trade-off") that references the clarification target — this is acceptable contextual framing for a non-answer, not invented content. possibleInference speculates about what the user may need, which is unnecessary for a non-answer where no leaning should be inferred.
|
||||
|
||||
---
|
||||
|
||||
### Evaluation Summary
|
||||
|
||||
| Case | Answer | Stated Meaning Classification | Inference Separation |
|
||||
|------|--------|------------------------------|---------------------|
|
||||
| 1 | "Risk matters more to me." | stated_meaning_preserved | inference_cleanly_separated |
|
||||
| 2 | "I'd normally avoid more risk, but for the right opportunity I might accept some." | stated_meaning_preserved | inference_cleanly_separated |
|
||||
| 3 | "It's a hard constraint. I don't want any increase in risk." | stated_meaning_preserved | unnecessary_inference |
|
||||
| 4 | "I'm not really sure." | stated_meaning_preserved | unnecessary_inference |
|
||||
|
||||
**Stated meaning counts:**
|
||||
- stated_meaning_preserved: 4
|
||||
- stated_meaning_strengthened: 0
|
||||
- stated_meaning_lost: 0
|
||||
|
||||
**Inference separation counts:**
|
||||
- inference_cleanly_separated: 2
|
||||
- unnecessary_inference: 2
|
||||
- inference_leaked_into_stated: 0
|
||||
- no_inference_needed: 0
|
||||
|
||||
---
|
||||
|
||||
### Questions Answered
|
||||
|
||||
1. **Did Case 1 keep "risk matters more" as relative importance only?** Yes — statedMeaning states "greater importance than other factors" without deciding whether risk avoidance is a hard constraint or not.
|
||||
|
||||
2. **Did Case 1 place any stronger preference/constraint implication only in possibleInference?** Yes — the model placed "weighted as a stricter constraint rather than a flexible preference" in possibleInference, qualified with "may suggest" and "remains unconfirmed."
|
||||
|
||||
3. **Did Case 2 preserve the "for the right opportunity" condition?** Yes — statedMeaning preserved both "generally preferred" and "can be accepted for the right opportunity." It also added a qualifier about preference/trade-off rather than hard constraint (slight overreach but not meaningful loss).
|
||||
|
||||
4. **Did Case 3 preserve the explicit hard constraint without unnecessary inference in statedMeaning?** Yes — statedMeaning correctly establishes the hard constraint. possibleInference was unnecessary (should have been null) but statedMeaning is clean.
|
||||
|
||||
5. **Did Case 4 preserve uncertainty without inventing a leaning?** Yes — statedMeaning preserves uncertainty. It added contextual framing referencing the clarification target, which is acceptable for non-answer context. No preference or constraint was invented. possibleInference was unnecessary but did not invent a specific leaning (it asked what the user might need, not what they likely prefer).
|
||||
|
||||
6. **Did any unsupported meaning leak into statedMeaning?** No — none of the four cases leaked stronger-than-justified meaning into statedMeaning. Case 2 added a qualifier ("qualifying it as a preference or trade-off rather than a hard constraint") that was not explicitly in the user's answer, but this is contextual framing rather than unsupported strengthening. The core conditional meaning (both sides) was preserved.
|
||||
|
||||
7. **Did the model generate unnecessary implications where the answer was already explicit?** Yes — Case 3 and Case 4 both received possibleInference content when none was warranted. This suggests the model tends to always provide an inference even when the answer is complete or absent. Not a statedMeaning defect, but a possibleInference hygiene issue.
|
||||
|
||||
8. **How many cases were stated_meaning_preserved / strengthened / lost?** preserved: 4, strengthened: 0, lost: 0.
|
||||
|
||||
9. **Compared with 55C Case 1, did explicit stated-vs-inferred separation avoid the earlier strengthening?** Yes — in 55C Stage 1, "Risk matters more to me." was strengthened into language about "preference/trade-off rather than absolute hard constraint" inside the single meaning field. In 55D, the relative importance remained clean in statedMeaning and any stronger interpretation was placed separately in possibleInference. This shows the two-field separation can prevent silent strengthening when it matters most (weak answers).
|
||||
|
||||
10. **Does this prove that production should use this exact two-field contract?** No — four cases through one call each is insufficient to justify specific production changes. The mechanism works in these tests but broader validation is needed.
|
||||
|
||||
11. **Does this establish how resolution should consume these fields?** No — resolution was not tested here. How a downstream step should combine statedMeaning and possibleInference remains an open question.
|
||||
|
||||
12. **Does this establish graph or Behaviour Selection changes?** No — no graph, Behaviour Selection, or engine integration was attempted.
|
||||
|
||||
---
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only four answers were tested across one domain (risk vs. growth). Different answer patterns may behave differently.
|
||||
- Each case was called exactly once — stability across repeated identical calls was not tested.
|
||||
- possibleInference hygiene is imperfect: Cases 3 and 4 received unnecessary inferences, suggesting the model struggles to return null when no inference is warranted.
|
||||
- Only one model configuration was used (qwen-claude:latest on 192.168.1.111:11434).
|
||||
- Case 2's statedMeaning contained slight overreach ("qualifying it as a preference or trade-off rather than a hard constraint") — while the core meaning was preserved, not all answers will be this clean even with separation.
|
||||
- No downstream consumer (resolution, graph update) was tested — only whether the two fields can coexist without leakage.
|
||||
|
||||
---
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Stated meaning remained separate from model inference across all tested answers.**
|
||||
|
||||
Across four fixed cases spanning weak priority, conditional trade-off, explicit constraint, and non-answer, statedMeaning never contained stronger-than-justified meaning. Case 1's weak-priority answer (the primary failure mode of 55C) stayed as relative importance only in statedMeaning — a direct improvement over 55C where the same answer was strengthened into constraint language. Case 2 preserved both sides of the conditional; Case 3 preserved explicit meaning cleanly; Case 4 preserved uncertainty without inventing position.
|
||||
|
||||
The separation mechanism works: the model can keep what the user established from what it might imply, at least in single-call mode. The remaining issue is possibleInference hygiene — the model tends to generate implications even when none are warranted (Cases 3 and 4). This does not corrupt statedMeaning but suggests the null-enforcement direction should be tuned.
|
||||
|
||||
Does this prove a production two-field contract is required? **No.** Evidence from four single calls across one answer pattern is insufficient. Does this establish how resolution should consume these fields? **No.** Resolution was not tested. Does this establish graph or Behaviour Selection changes? **No.**
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
All 4 live inference calls completed successfully. Four answers tested through one call each with stated-vs-inferred separation. All four cases preserved statedMeaning without strengthening (stated_meaning_preserved: 4/4, strengthened: 0, lost: 0). Case 1's weak-priority answer stayed as relative importance only — direct improvement over 55C where the same answer was strengthened to constraint language in Stage 1. Inference cleanly separated for Cases 1 and 2; unnecessary inferences generated for Cases 3 and 4 (hygiene issue, not leakage). Total: 76730ms (~76.7s), average: ~19182.5ms per call, fastest: 16766ms, slowest: 24551ms.
|
||||
|
||||
### Historical Comparison Result
|
||||
|
||||
Compared to Experiment 55C's finding that Stage 1 strengthened "Risk matters more to me." into language about "preference/trade-off rather than absolute hard constraint," Experiment 55D shows the two-field separation avoided the specific weak-priority strengthening defect in this tested run: weak-priority answers stayed as relative importance in statedMeaning while stronger interpretations were placed separately in possibleInference. The mechanism handled the specific failure mode successfully in this probe, but broader stability and downstream consumption remain untested.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — added full Experiment 55D entry; applied corrections to Experiment 55C wording and commit hash
|
||||
- `docs/current-handoff.md` — updated with Experiment 55D summary and new Return-to-Work note
|
||||
|
||||
### Confirmation Host and Model Remained Unchanged
|
||||
|
||||
Host: `http://192.168.1.111:11434`. Model: `qwen-claude:latest`. Same as all prior experiments.
|
||||
|
||||
### Confirmation Production Prompts and Schemas Remained Unchanged
|
||||
|
||||
No production prompts read or modified. No schemas changed. All inference calls used the experiment-specific semantic instructions defined in this test file.
|
||||
|
||||
### Confirmation Behaviour Selection Remained Unchanged
|
||||
|
||||
Behaviour Selection was not called or referenced. No integration with the selector occurred.
|
||||
|
||||
### Confirmation Graph and UI Remained Unchanged
|
||||
|
||||
No graph files read or modified. No UI code touched. The experiment is test-only.
|
||||
|
||||
### Confirmation No Stated-vs-Inferred Clarification Logic Entered Active Runtime
|
||||
|
||||
This experiment created one new test file only. No stated-vs-inferred clarification logic entered any active runtime path, production module, or behaviour selection output. Production continues using the pre-existing contract.
|
||||
|
||||
---
|
||||
|
||||
### Return-to-Work Note (Experiment 55D)
|
||||
|
||||
55C showed preserved meaning can protect later resolution, but weak-priority meaning was already strengthened in Stage 1. 55D isolated that first interpretation step using a single-call stated-vs-inferred separation with four fixed answers across risk preference cases. Weak priority stayed as relative importance only (direct improvement over 55C's constraint-language strengthening). Conditionality survived through the conditional trade-off case. Explicit and uncertain controls stayed clean — no unsupported meaning leaked into statedMeaning. Stronger implications were kept separate in possibleInference for Cases 1 and 2, though Cases 3 and 4 showed unnecessary inference generation (hygiene issue, not leakage). This does not yet prescribe production architecture. Graph, Behaviour Selection, UI and production remain untouched. Same host/model (qwen-claude:latest on http://192.168.1.111:11434). Branch: feature/user-workspace-ux-v0.7. First test/file to inspect when resuming: tests/reconstruction/semantic-clarification-stated-vs-inferred.test.js for the full experiment and results. Status pending Rob's review.
|
||||
|
||||
## Experiment 55E — Reasoning Refinement Requirements Synthesis (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Consolidate findings from Experiments 53–55D into a compact, implementation-ready reasoning contract for the next production pass. Stop opening new semantic sub-problems for this round.
|
||||
|
||||
### Context Documents Reviewed
|
||||
|
||||
- `docs/current-handoff.md` (current state and handoff);
|
||||
- Experiments 53, 54K–54Z, 55A–55D in `docs/design-evolution-log.md`;
|
||||
- Created: `docs/reasoning-refinement-requirements.md`.
|
||||
|
||||
### Synthesis Output
|
||||
|
||||
- **8 reasoning requirements** retained (R1–R8), all supported by at least one recorded experiment observation;
|
||||
- **7 known failure patterns** recorded from the experiment history;
|
||||
- **8 known good behaviours** recorded as regression expectations;
|
||||
- **6 regression cases** selected (A–F) covering: weak priority, conditional trade-off, non-answer/uncertainty, explicit hard constraint, evidence-resolvable disagreement, and user-owned ambiguity;
|
||||
- **10 open questions** explicitly retained to prevent premature architecture.
|
||||
|
||||
### Key Unresolved Items
|
||||
|
||||
- Stability across larger case sets and other models;
|
||||
- Exact production representation (graph integration pending);
|
||||
- Downstream consumption of inference fields;
|
||||
- Behaviour Selection and UI integration timing;
|
||||
- Performance/latency implications;
|
||||
- One call versus multiple calls for semantic separation.
|
||||
|
||||
### Conclusion
|
||||
|
||||
This round of semantic experimentation is closed. The requirements synthesis in `docs/reasoning-refinement-requirements.md` provides a bounded starting point for tomorrow's implementation pass. No live inference was performed. No production code, prompts, schemas, graph files, Behaviour Selection rules, or UI code were modified. The mechanism from Experiment 55D avoided the specific weak-priority strengthening defect in this tested run; broader stability remains unproven.
|
||||
|
||||
## Experiment 55F — Reasoning Requirements Production Path Map (2026-08-08)
|
||||
|
||||
### Objective
|
||||
|
||||
Map how reasoning requirements R1–R8 are actually supported (or unsupported) by the existing production code path, using source-inspection only. Trace the answer-to-reasoning flow through prompt building, LLM response parsing and normalization, and graph mutation. Identify which gaps have structural carriers in current schemas and which require new schema fields or logic at specific line locations. This exercise is explicitly NOT architecture design or implementation — it documents what exists today so tomorrow's Codex pass starts from accurate information.
|
||||
|
||||
### Context Documents Reviewed
|
||||
|
||||
- `docs/reasoning-refinement-requirements.md` (R1–R8 requirements, regression pack A–F);
|
||||
- `docs/current-handoff.md` (handoff state after 55E);
|
||||
- `lib/graph/orchestrator.js` — updateCase code path and LLM/provider integration;
|
||||
- `lib/graph/schema.js` — situationNodeSchema, graphUpdateSchema, updateCaseRequestSchema;
|
||||
- `lib/graph/update-proposal.js` — parseGraphUpdateProposal with normalization;
|
||||
- `lib/graph/prompt-builder.js` — buildGraphUpdatePrompt with answer embedding;
|
||||
- `lib/graph/apply-proposal.js` — applyValidatedProposal and deriveReasoningStateOverride;
|
||||
- `lib/graph/builder.js` — initial graph construction (not used in update cycles).
|
||||
|
||||
### Findings
|
||||
|
||||
**Production update path:** user answer → buildGraphUpdatePrompt → LLM provider → parseGraphUpdateProposal → applyValidatedProposal. The full chain was traced with line-number precision for each transition.
|
||||
|
||||
**Confirmed gap on provenance:** `situationNodeSchema` has no provenance fields (no source/inference annotation). `graphUpdateSchema` also lacks provenance fields. `updateCaseRequestSchema` carries the raw answer but provides no semantic-meaning fields. Evidence records built during startCase are not returned alongside graph state during update cycles.
|
||||
|
||||
**Confirmed gap on meaning preservation:** The answer string in `applyValidatedProposal` reaches only `deriveReasoningStateOverride` at line 2875 and is used solely for a narrow comparability confirmation check. After that point, only the structural graph state (already containing the LLM's interpretation) flows forward — not the original answer meaning.
|
||||
|
||||
**Confirmed support:** Existing relationship types distinguish evidence vs clarification needs. Structural validation gates maintain integrity. Decomposition quality gates exist on child unknowns. Null selectedQuestion is structurally valid.
|
||||
|
||||
**All eight requirements assessed individually** in a cross-reference matrix showing which have any support (prompt, parse/normalize, application, schema) and where gaps are located.
|
||||
|
||||
### Key Unresolved Items
|
||||
|
||||
- Whether provenance fields should be added to `situationNodeSchema`, `graphUpdateSchema`, or both;
|
||||
- How meaning preservation verification compares original answer text against proposed graph changes;
|
||||
- Where in the four-step pipeline (schema → prompt → parse → mutation) semantic-meaning carriers must enter;
|
||||
- Whether the current approach (two-field interpretation contract from 55D) is viable given the lack of schema carrier, or if a different mechanism is required.
|
||||
|
||||
### Conclusion
|
||||
|
||||
Source-inspection-only exercise completed. The production path does not carry semantic meaning — it carries structural graph changes that represent the LLM's interpretation of the answer. Every R1–R8 requirement depends on mechanisms absent from the current code path. A complete cross-reference with specific line-location gap targets is in `docs/reasoning-production-path-map.md`. No live inference was performed. No production code, prompts, schemas, graph files, Behaviour Selection rules, or UI code were modified. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: docs/reasoning-production-path-map.md for the full gap analysis and specific line-location targets. Status pending Rob's review.
|
||||
|
||||
---
|
||||
|
||||
### v0.51–v0.58 Progress — Product Provenance and Architectural Decisions
|
||||
|
||||
#### v0.51 — Zero Open Questions milestone
|
||||
|
||||
Established the zero-Open-Questions milestone as a genuine product feature: when all unknowns are resolved, the invitation "You've now worked through all of the questions we surfaced" appears in place of Open Questions. The eligibility uses canonical graph state (resolved nodes), not local `doneForNowIds`. This is a milestone invitation, not a readiness/completion judgement.
|
||||
|
||||
#### v0.52 — Focused investigation presentation ownership
|
||||
|
||||
Established that focused-presentation content must be scoped per-question. Previously, fresh unanswered Question B displayed stale content from Question A across every surface (previously answered, what this tells us, still unclear, questions raised, assumptions, connections). Fixed by thread-local filtering in `FocusedQuestionBody`. Previously answered contributions remain globally preserved in history; only presentation derivation is narrowed.
|
||||
|
||||
#### v0.53 — Empty Done + Re-open semantics
|
||||
|
||||
Established that empty Done (parked without providing an answer) is valid product behaviour: it parks the question locally, does NOT invoke episode processing, and does NOT produce a `no_episodic_content` 400 error. It produces the same resolved state shape as populated Done. Re-open returns the question to Open Questions and removes from `doneForNowIds`. Older stale development localStorage states (pre-v0.53 shape) may be discarded during dev phase; no migration required.
|
||||
|
||||
#### v0.54 — Investigation-level synthesis seam
|
||||
|
||||
Established a distinct investigation-level synthesis apparatus (`synthesizeInvestigationOverview()`) separate from Current Understanding. Important semantic lesson: Current Understanding and Investigation Report overview are NOT the same product artefact. Plausible interpretations in the Report remain explicitly interpretive rather than evidence. The epistemic boundary (evidence never promoted to interpretation; interpretations never promoted to understanding) is schema-enforced via Zod-safeParse.
|
||||
|
||||
#### v0.55 — Portfolio / Investigation / Report route architecture
|
||||
|
||||
Established three distinct product concepts:
|
||||
```
|
||||
/ → Portfolio (notebook index)
|
||||
/investigations/case-1 → Investigation (working case)
|
||||
/investigations/case-1/report → Investigation Report (derived summary)
|
||||
```
|
||||
|
||||
ReasoningWorkspace no longer owns Report presentation. The Report is a distinct route/page, not an internal state of the Investigation. Portfolio currently supports one canonical persisted investigation only. Temporary development identity remains `case-1`. True multi-investigation persistence/identity remains future work.
|
||||
|
||||
**Product analogy:** Portfolio = investigator notebook index, Investigation = working case/pages, Report = readable derived summary page. Users can eventually flick directly to the page they need.
|
||||
|
||||
#### v0.56 — Portfolio action semantics
|
||||
|
||||
Clarified that actions on an existing investigation card are distinct from creation of a new investigation. Actions on the card: View report, Continue investigation, Restart investigation. Creation is portfolio-level only: + Create new investigation below the card. No duplicate creation control inside the card.
|
||||
|
||||
#### v0.57 — Destructive Restart confirmation
|
||||
|
||||
Established that Restart investigation is explicitly destructive: first confirmation via dialog ("Restart this investigation?" with warning about lost data), then a second explicit "Restart investigation" button call. `clearInvestigation()` remains the canonical persisted-storage clear seam. No direct storage-key manipulation was introduced.
|
||||
|
||||
#### v0.58 — First Report generation lifecycle
|
||||
|
||||
Established that:
|
||||
- A genuine no-report investigation generates exactly one persisted Investigation Report
|
||||
- Report generation ownership belongs to the Report page, NOT ReasoningWorkspace or Investigation page
|
||||
- First Report visit = exactly 1 `/api/cases/overview` synthesis call
|
||||
- Subsequent Report visits = zero synthesis calls (renders persisted snapshot)
|
||||
- The Report is a derived artefact, not canonical reasoning evidence
|
||||
|
||||
**Live verification used genuine product-created investigations.** Six Open Questions surfaced in a fresh scenario — this was legitimate product output. An earlier experimental `≤5` processing bound was an apparatus constraint, NOT a product requirement. Do not document "Open Questions must be ≤5."
|
||||
|
||||
### Product Reasoning Lessons from v0.51–v0.58
|
||||
|
||||
**Investigator's notebook model.** The Portfolio / Investigation / Report triad maps to: notebook index → working case → readable outcome. This is an architectural decision about user navigation, not just technical separation.
|
||||
|
||||
**Report as durable derived artefact.** The Report should support future portfolio revisit, copy/export, Jira/document use, investigation portfolio — without becoming canonical reasoning evidence. It is a summary of what was understood at a point in time.
|
||||
|
||||
**User ownership / non-steering.** The engine facilitates investigation. It does not steer or prioritise which question must be answered next. User controls: which question to investigate, when to say Done for now, whether Current Understanding is sufficient, whether to reopen work, when to review the Report.
|
||||
|
||||
**Evidence lessons captured at provenance level:**
|
||||
- Tests can fail because apparatus cannot observe the intended contract — not because the product is broken.
|
||||
- Playwright snapshot refs are transient — never use them as action targets.
|
||||
- Client hydration must be treated as real product behaviour — pre-hydration empty ≠ absence of data.
|
||||
- Experimental execution bounds (e.g., ≤5 Open Questions) must not be mistaken for product requirements.
|
||||
- Manual product verification can validly establish prerequisite state when automation itself is not the subject of the experiment.
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
# Ch19 — Initial Decomposition Hardening (v0.61)
|
||||
|
||||
**Status: COMPLETE (frozen)**
|
||||
**Branch:** `feature/initial-decomposition-v0.61`
|
||||
**Final HEAD:** `5878ce4` experiment(confidence-engine): add reconstruction-only helper flag
|
||||
|
||||
## v0.61 Objective
|
||||
|
||||
How stable is the initial semantic decomposition of the same fixed scenario across repeated runs using the same model, prompt version and production API path?
|
||||
|
||||
## Fixed manufacturing scenario (used throughout)
|
||||
|
||||
```
|
||||
I run a small manufacturing business. Customer complaints have risen by 35% over the last six months, while production volume increased by 40%.
|
||||
|
||||
Most complaints mention late delivery or minor product defects, but our complaint categories changed when we introduced a new CRM tagging system three months ago. During the same period we also changed one supplier and introduced a weekend production shift.
|
||||
|
||||
I am deciding whether to spend about £120,000 on automated quality inspection now or wait until we understand whether there is actually a quality problem.
|
||||
|
||||
I do not yet know the complaint rate per unit, whether defect rates differ by shift or supplier, or whether the new tagging system changed what gets counted as a complaint.
|
||||
```
|
||||
|
||||
## v0.61.3 — Matched Provider Reconstruction Evidence Checkpoint
|
||||
|
||||
**Status:** DOCUMENTATION-ONLY CHECKPOINT
|
||||
**Purpose:** Record six matched reconstruction-comparison observations (Qwen + Terra) against the canonical manufacturing fixture before closing out. No live model calls. No production changes.
|
||||
|
||||
### Production prompt / schema frozen
|
||||
- **Prompt:** `reconstruct-v0.5` (production default)
|
||||
- **Rule:** Rule 5a
|
||||
- **Salience check:** semantic-preservation
|
||||
- **Schema:** canonical reconstruction schema
|
||||
- **Path:** reconstruction-only experiment path
|
||||
|
||||
### Qwen / Ollama evidence (3 matched observations)
|
||||
|
||||
| Channel | Finding |
|
||||
|---|---|
|
||||
| A — complaints +35%, production +40%, complaint-rate-per-unit uncertainty | semantically stable |
|
||||
| B — CRM tagging / complaint-count comparability uncertainty | semantically stable |
|
||||
| C — late-delivery versus minor-product-defect distinction | meaning preserved, representation varied |
|
||||
| D1 — supplier-related defect uncertainty | meaning preserved, but supplier and shift frequently recombined |
|
||||
| D2 — weekend-shift-related defect uncertainty | meaning preserved, but supplier and shift frequently recombined |
|
||||
| E — intervention-fit dependency | core dependency preserved, graph topology varied |
|
||||
|
||||
**Important observed pattern:** compound supplier/shift unknown appeared in all three Qwen runs.
|
||||
|
||||
### Terra / OpenAI evidence (3 matched observations)
|
||||
|
||||
| Channel | Finding |
|
||||
|---|---|
|
||||
| A — complaints +35%, production +40% | stable |
|
||||
| B — CRM comparability | stable |
|
||||
| C — late-delivery vs defects | semantically stable and separately represented in all three |
|
||||
| D1 — supplier-related defect uncertainty | stable |
|
||||
| D2 — weekend-shift-related defect uncertainty | stable |
|
||||
| E — intervention-fit dependency | core dependency stable |
|
||||
|
||||
All three Terra runs: supplier transition distinct=YES, weekend-shift distinct=YES, compound supplier/shift=NO, unsupported expansion=NO, causal strengthening=NO.
|
||||
|
||||
### Cross-provider conclusion
|
||||
|
||||
> Exact graph topology is not stable for either provider and should not be treated as a reconstruction invariant. The more important invariant is preservation and independent recoverability of supplied distinctions and uncertainties.
|
||||
|
||||
> Terra showed stronger stability in preserving independently investigable C and D1/D2 distinctions as separate graph structures. Qwen preserved their meaning but more frequently compressed them into combined representations.
|
||||
|
||||
## v0.61 Experiment 5 — Repeated Same-Input Initial Decomposition Stability
|
||||
|
||||
**Status:** INCOMPLETE — five runs required; obtained four successful plus one validation failure.
|
||||
|
||||
### Execution
|
||||
- **Route:** POST /api/cases/start
|
||||
- **Requests made:** 5 (4 successful, 1 validation failure)
|
||||
- **Model:** `qwen-claude:latest`
|
||||
- **Prompt version:** v0.2
|
||||
|
||||
### Structure Range
|
||||
|
||||
| | Nodes | Unknowns | Assumptions |
|
||||
|---|---|---|---|
|
||||
| Run 1 | 16 | 3 | 3 |
|
||||
| Run 2 | 20 | 7 | 2 |
|
||||
| Run 3 | 21 | 3 | 3 |
|
||||
| Run 4 | 14 | 3 | 2 |
|
||||
|
||||
Unknown count: 3–7. Assumption count: 2–3. Total node count: 14–21.
|
||||
|
||||
### Semantic Channel Frequency (Runs 1–4)
|
||||
|
||||
```
|
||||
A Normalisation: 4/4 present, 0/4 partial, 0/4 absent
|
||||
B CRM comparability: 4/4 present, 0/4 partial, 0/4 absent
|
||||
C Delivery vs defects: 1/4 present, 3/4 partial, 0/4 absent
|
||||
D Supplier/shift/source: 4/4 present, 0/4 partial, 0/4 absent
|
||||
E Intervention fit: 1/4 present, 0/4 partial, 3/4 absent
|
||||
```
|
||||
|
||||
### Key findings
|
||||
|
||||
- Run 2 introduced speculative subdivisions not grounded in source text (customer segment variation, QC team capability shifts, responsibility distribution impact, batch-size/equipment utilization correlation).
|
||||
- No run contained steering language, unsupported causal claims, premature arithmetic conclusions, or action recommendations.
|
||||
- The £120k intervention was represented neutrally in all runs.
|
||||
|
||||
### Conclusion
|
||||
|
||||
Initial decomposition represents core factual uncertainty channels (normalisation, CRM comparability, supplier/shift ambiguity) to varying degrees. Primary instability: intervention-fit mapping and speculative subdivision risk. Graph topology varies materially (14–21 nodes).
|
||||
|
||||
## Focused-deconstruction plumbing fix (historical)
|
||||
|
||||
**Root cause:** focused route passed full provider envelope `{ response, providerApiPath, providerExecution }` into `validateFocusedDeconstructSchema()`. Validator expected semantic fields at top level but they lived on `wrapper.response`, not the wrapper. All six fields appeared absent → structured-output 502.
|
||||
|
||||
**Fix:** Route extracts `const deconstruction = wrapper.response` and passes inner object to validation and serialization. Provider diagnostics preserved but do not interfere with semantic fields.
|
||||
|
||||
**Tests:** `tests/focused-deconstruct-boundary.test.js` — 48 focused-investigation-boundary tests pass on first run. Previous two pre-fix semantic runs remain invalid as semantic evidence.
|
||||
|
||||
### Post-fix repeatability (3 controlled Qwen/Ollama runs)
|
||||
|
||||
| Metric | Result |
|
||||
|---|---|
|
||||
| Qwen/Ollama calls | 3 |
|
||||
| HTTP 200 | 3/3 |
|
||||
| schema valid | 3/3 |
|
||||
| supplier evidence preservation | 3/3 PASS |
|
||||
| weekend-shift uncertainty preservation | 3/3 PASS |
|
||||
| epistemic separation | 3/3 PASS |
|
||||
| unsupported inference | 0/3 |
|
||||
| steering | 0/3 |
|
||||
|
||||
**Product interpretation:** For the fixed compound supplier/weekend-shift case, Qwen's initial compression did not prevent focused-investigation stage from repeatedly recovering the epistemic distinction once substantive user evidence was supplied. This supports tolerating some initial representation compression when meaning survives downstream.
|
||||
|
||||
This is **not yet generalised to a production invariant.**
|
||||
|
||||
## v0.61 — Closed experiment boundaries
|
||||
|
||||
| Boundary | Status |
|
||||
|---|---|
|
||||
| Repeated-same-input decomposition experiments | FROZEN |
|
||||
| Qwen/Terra reconstruction comparison | FROZEN |
|
||||
| Initial prompt refinement | FROZEN |
|
||||
| Supplier/weekend-shift decomposition experiments | FROZEN |
|
||||
| Relationship-preservation experiments | FROZEN |
|
||||
| Causal-fidelity experiments | FROZEN |
|
||||
|
||||
## v0.4 initial reconstruction — implementation status
|
||||
|
||||
**Status:** COMPLETE — deterministic tests pass, one live smoke accepted
|
||||
**Prompt:** `prompts/reconstruct-v0.4.md` (replaced by v0.5 in production default)
|
||||
**Tests:** 89/89 passed on first run, zero reruns
|
||||
|
||||
Key principles proven: provenance-preserving decomposition, relationship preservation, explicit stop boundary, interpretation separation. All documented and verified.
|
||||
|
||||
## Git
|
||||
|
||||
- **Documentation commit:** `docs(confidence-engine): record focused deconstruction repeatability`
|
||||
- **Working tree:** clean after this session's commit
|
||||
|
||||
---
|
||||
|
||||
*This chapter records the initial-decomposition v0.61 experiment line. The line is frozen for the current MVP stage. See `docs/current-handoff.md` §CURRENT MVP DIRECTION.*
|
||||
+769
@@ -0,0 +1,769 @@
|
||||
### Experiment 07 — Lightweight Starting Observation
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
A smaller initial input will make beginning an investigation feel easier and will communicate that the engine needs only a concise observation rather than a complete analysis.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Does the input feel like a conversation starter rather than a report form?
|
||||
- Is three to four visible lines sufficient?
|
||||
- Does the facilitator panel and input area feel better balanced?
|
||||
- Does the user understand that further detail will be gathered through questions?
|
||||
- Does reducing the input height make the Analyse action easier to notice?
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Pending visual review.
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed.
|
||||
|
||||
Four visible rows better communicates a starting observation than six.
|
||||
|
||||
Input size communicates expected effort.
|
||||
|
||||
"What have you noticed?" reinforces observational thinking.
|
||||
|
||||
Users are encouraged to begin rather than compose.
|
||||
|
||||
The facilitator and workspace now feel more balanced.
|
||||
|
||||
This interaction principle should continue throughout the investigation rather than existing only on the landing page.
|
||||
|
||||
#### Decision
|
||||
|
||||
Retain the smaller landing input.
|
||||
|
||||
Proceed to investigate consistency between the landing experience and investigation responses.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 09 — Investigation Rhythm
|
||||
|
||||
#### Result
|
||||
|
||||
Partially confirmed.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- Moving History directly beneath Response improves the sense of conversational continuity.
|
||||
- The sequence Question → Response → History is cognitively coherent.
|
||||
- History behaves like the growing notebook of the investigation, not general reference material.
|
||||
- Allowing History to span the full workspace breaks the wider spatial model.
|
||||
- Situation and Investigation Map should remain stable supporting artefacts rather than moving down as the notebook grows.
|
||||
- The conversation needs a dedicated vertical lane.
|
||||
|
||||
#### Decision
|
||||
|
||||
Keep History directly connected to Response.
|
||||
|
||||
Refine the desktop workspace into a stable conversation lane and a stable supporting lane.
|
||||
|
||||
Do not rewrite previous experiments.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 08 — Consistent Investigation Responses
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Every answer given during an investigation should feel like an observation, not a report.
|
||||
|
||||
The response component should therefore communicate the same expected effort as the initial scenario input.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Does a smaller response area reduce perceived effort?
|
||||
- Does the investigation feel more conversational?
|
||||
- Does consistency improve confidence?
|
||||
- Does the workspace become visually calmer?
|
||||
- Does the current investigation remain the dominant focus?
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed.
|
||||
|
||||
Consistent interaction patterns reduce cognitive effort.
|
||||
|
||||
Users should not have to learn different behaviours between the landing page and investigation.
|
||||
|
||||
Smaller response areas reinforce concise observations.
|
||||
|
||||
The engine appears more conversational when each answer feels lightweight.
|
||||
|
||||
Consistency is becoming a stronger design tool than decoration.
|
||||
|
||||
#### Decision
|
||||
|
||||
Retain consistent input sizing across both contexts.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 10 — Stable Conversation Column
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
A persistent two-thirds conversation column beside a one-third supporting column will allow the investigation notebook to grow without moving the shared reference artefacts.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Does the left column feel like one continuous investigation?
|
||||
- Does History grow naturally beneath Response?
|
||||
- Do Situation and Investigation Map remain easy to reference?
|
||||
- Does the interface feel spatially stable as turns accumulate?
|
||||
- Does showing full question text improve readability now that sufficient width exists?
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Visual review completed.
|
||||
|
||||
#### Status
|
||||
|
||||
Closed.
|
||||
|
||||
## Result
|
||||
|
||||
Partially confirmed.
|
||||
|
||||
## What did we learn?
|
||||
|
||||
- The investigation workspace is beginning to feel like a genuine facilitated investigation rather than a document.
|
||||
- The two-column workspace (conversation on the left, reference material on the right) is proving to be a stronger mental model than previous layouts.
|
||||
- Keeping Situation and Investigation Map fixed while History grows vertically feels more natural.
|
||||
- The investigation question, response and history now read as one continuous conversation.
|
||||
- Developer Details have become extremely valuable.
|
||||
- The graph produced by the reasoning engine is far richer than previously realised. The graph now contains structured concepts including:
|
||||
|
||||
- observations
|
||||
- unknowns
|
||||
- assumptions
|
||||
- relationships
|
||||
- metrics
|
||||
- state
|
||||
|
||||
This suggests the UI should increasingly become a human-friendly projection of the graph rather than inventing separate state.
|
||||
|
||||
The current "Investigation in progress" panel exposes developer-oriented statistics (nodes, edges, unknowns etc.) which are useful during development but are not the most helpful representation for an end user.
|
||||
|
||||
---
|
||||
|
||||
## Emerging Direction — Graph as Source of Truth
|
||||
|
||||
The reasoning graph is becoming the shared source of truth for multiple UI views.
|
||||
|
||||
Different interfaces may project the same graph for different audiences:
|
||||
|
||||
- Version A — compact technical progress;
|
||||
- Version B — detailed graph inspection;
|
||||
- Version C — user-facing facilitator view;
|
||||
- Developer Details — complete diagnostics;
|
||||
- Investigation Map — future spatial projection;
|
||||
- Current Question — active uncertainty projection.
|
||||
|
||||
The UI should not maintain separate invented summaries where the graph already contains the underlying information.
|
||||
|
||||
This is an emerging direction, not a final architecture decision.
|
||||
|
||||
---
|
||||
|
||||
## Emerging Direction — Facilitator Translation Layer
|
||||
|
||||
> The UI should progressively become a translation layer over the reasoning graph rather than maintaining separate duplicated summaries. Internal graph concepts should remain available for developers, while end users see a facilitator-style explanation of what is currently understood and what remains uncertain.
|
||||
|
||||
The current technical progress panel (nodes, edges, unknowns, assumptions) exposes developer-oriented statistics. These are valuable during development but not the most helpful representation for an end user.
|
||||
|
||||
The next direction is to explore presenting the same underlying graph data as a facilitator's notebook — what is known, what remains uncertain, and a quiet summary of the reasoning state underneath.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 11 — Facilitator Progress Panel (Version B)
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
The same underlying reasoning graph can be presented in a much more human-friendly way without changing the reasoning engine, API contracts, or graph generation.
|
||||
|
||||
A facilitator-style panel should communicate:
|
||||
|
||||
- what is known (resolved nodes and observations)
|
||||
- what remains uncertain (unresolved unknowns and assumptions)
|
||||
- a quiet summary of the reasoning state underneath
|
||||
|
||||
#### Questions
|
||||
|
||||
- Can the same graph data be translated into a facilitator-style view that end users understand more naturally?
|
||||
- Does separating "known" from "still investigating" reduce cognitive load compared to node/edge counts?
|
||||
- Is a quiet reasoning summary sufficient, or does it need more context?
|
||||
- Does the translation-layer principle hold — presenting the graph as a notebook rather than raw data?
|
||||
|
||||
#### Result
|
||||
|
||||
Partially confirmed.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- Version B proved that the reasoning graph contains substantially more useful information than Version A exposes.
|
||||
- The graph already contains observations, unknowns, assumptions, metrics, relationships and state.
|
||||
- The graph is rich enough to support multiple UI projections.
|
||||
- Exposing the graph almost verbatim overwhelms the user.
|
||||
- Technical categories are useful for development but do not directly communicate investigation progress.
|
||||
- The user needs a translation of the graph rather than a graph browser.
|
||||
- Developer Details should remain the place for complete technical inspection.
|
||||
- A user-facing view needs filtering, prioritisation, deduplication and clear epistemic labels.
|
||||
|
||||
#### Decision
|
||||
|
||||
Keep Version A and Version B available for comparison.
|
||||
|
||||
Proceed with a Version C facilitator view built from the same graph.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 12 — Facilitator View (Version C)
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
The existing reasoning graph can be deterministically translated into a concise facilitator view that helps the user understand:
|
||||
|
||||
- what is currently known;
|
||||
- what remains uncertain;
|
||||
- what may explain the situation;
|
||||
- why the investigation is continuing.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Can the graph produce a useful human-facing summary without another LLM call?
|
||||
- Can observations, unknowns and assumptions be clearly distinguished?
|
||||
- Can duplicate or low-value graph content be filtered reliably?
|
||||
- Does a concise projection improve understanding without exposing implementation detail?
|
||||
- Does the panel remain useful across mocks and live Ollama output?
|
||||
- Can the same view work during early, middle and terminal investigation states?
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Completed. Visual and live-data review performed.
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- The reasoning graph already contains all the information needed for a useful human-facing summary — no additional LLM calls are required.
|
||||
- Routing by semantic role (observation, question, explanation) rather than graph kind produces a more natural user experience.
|
||||
- Filtering scaffolding content (scenario summaries, system/tool references, metric object descriptions, process labels) is essential to keep the view focused on findings.
|
||||
- Deduplication of near-duplicate observations reduces noise without losing information.
|
||||
- Epistemic clarity matters — resolved unknowns become factual observations and should be classified as known rather than still-under-investigation.
|
||||
- The panel works across all investigation phases (early, active, terminal).
|
||||
|
||||
#### Decision
|
||||
|
||||
Close Experiment 12 as confirmed. Proceed to refine the translation through semantic classification in the next iteration.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 13 — Semantic Facilitator Translation
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Improving the deterministic projection from graph semantics to user-facing language — by classifying nodes by *meaning* rather than *graph kind*, suppressing scaffolding, merging duplicates, and preferring concrete observations — produces a significantly better facilitator view without changing the reasoning engine, prompts, graph generation, or any external contracts.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Does semantic role classification (observation vs question vs explanation) route content more naturally than graph-kind classification?
|
||||
- Does scaffolding suppression remove visual noise that previously dominated derived summaries?
|
||||
- Does deduplication reduce redundant items that express the same observation under slightly different wording?
|
||||
- Do concrete observations appear before abstract labels in ranked output?
|
||||
- Does the view remain robust when consumed by the existing panel component (investigation-summary-panel-v3) without any changes to that component?
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Completed. Tests: 37 scenarios passing across filtering, classification, deduplication, ranking, section framing, mock-data integration, and edge cases.
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- Semantic role routing outperforms kind-based routing: a node with `kind: "state"` that contains concrete data (e.g., "Revenue increased 12%") is more useful as an observation than a state description.
|
||||
- Scaffolding suppression works best when applied early — filtering at the semantic classification stage prevents structural glue from contaminating any section.
|
||||
- Three-tier filtering is effective: scaffolding patterns (highest priority), internal vocabulary (medium), then technical summary patterns (lowest).
|
||||
- Deduplication by normalised text removes meaningful noise. When "Revenue increased 12%" and "Current revenue is 12% higher" express the same observation, keeping one reduces confusion without losing information.
|
||||
- Resolved unknowns and assumptions are factual answers to previously unanswered questions — they should appear in the known section with an epistemic label ("Not yet established" / "To be tested") if their status hasn't been explicitly set.
|
||||
- The translation adapter is the right place for this work: it is a single deterministic function, testable in isolation, and its output contracts are stable.
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- Semantic filtering significantly improved Version C.
|
||||
- The remaining limitations are architectural rather than visual.
|
||||
- Graph nodes still do not naturally map to facilitator language.
|
||||
- Users think in investigation progress rather than graph structure.
|
||||
- Version C proved the need for an intermediate narrative model.
|
||||
|
||||
#### Decision
|
||||
|
||||
Keep the semantic projection approach.
|
||||
|
||||
Do not continue improving graph projection indefinitely.
|
||||
|
||||
Proceed to designing an Investigation Narrative layer. Experiment 13 is closed.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 14 — Investigation Narrative Layer
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
The graph should remain the internal reasoning model.
|
||||
|
||||
A separate narrative model should become the presentation model.
|
||||
|
||||
The facilitator UI should consume narrative state rather than graph nodes.
|
||||
|
||||
#### Questions
|
||||
|
||||
- What information belongs in a narrative?
|
||||
- What belongs only in the graph?
|
||||
- Which narrative elements can be derived deterministically?
|
||||
- What should remain hidden?
|
||||
- Can every facilitator panel consume the same narrative object?
|
||||
|
||||
#### Status
|
||||
|
||||
Architectural experiment.
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Pending.
|
||||
|
||||
---
|
||||
|
||||
## Emerging Direction — Investigation Narrative
|
||||
|
||||
The Confidence Engine architecture is becoming:
|
||||
|
||||
User
|
||||
|
||||
↓
|
||||
|
||||
Facilitated Conversation
|
||||
|
||||
↓
|
||||
|
||||
Reasoning Graph
|
||||
|
||||
↓
|
||||
|
||||
Investigation Narrative
|
||||
|
||||
↓
|
||||
|
||||
Workspace Projection
|
||||
|
||||
↓
|
||||
|
||||
User
|
||||
|
||||
The reasoning graph becomes the machine representation.
|
||||
|
||||
The investigation narrative becomes the human representation.
|
||||
|
||||
The UI simply renders whichever projection is appropriate.
|
||||
|
||||
This is an emerging architectural direction.
|
||||
|
||||
It is intentionally recorded before implementation so future experiments remain aligned.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 15 — Facilitator Behaviour Specification
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
An expert consultant does not have a script. They have behaviours — recurring patterns of action deployed based on what they observe in the client's situation. The Confidence Engine should exhibit similar behavioural patterns rather than following a mechanical question-fill-graph cycle.
|
||||
|
||||
The current engine behaviour is:
|
||||
|
||||
> Engine asks → User answers → Graph updates → Engine asks again
|
||||
|
||||
An expert facilitator behaviour is:
|
||||
|
||||
> Engine assesses state → selects appropriate behaviour → acts (question, acknowledge, synthesise, challenge, pause)
|
||||
|
||||
#### Questions
|
||||
|
||||
- How does an expert consultant behave during an investigation?
|
||||
- Which behaviours recur across investigations?
|
||||
- What triggers each behaviour?
|
||||
- When does the facilitator ask a question versus summarise versus expose uncertainty versus hold space?
|
||||
- What distinguishes guided thinking from mechanical Q&A?
|
||||
|
||||
#### Status
|
||||
|
||||
Investigation — behavioural model documented, not yet implemented.
|
||||
|
||||
#### Evaluation
|
||||
|
||||
This experiment is primarily architectural and behavioural. No code changes are required at this stage. The deliverable is a behavioural specification that future implementation experiments will reference.
|
||||
|
||||
#### Result
|
||||
|
||||
Confirmed as the correct next direction.
|
||||
|
||||
#### What did we learn?
|
||||
|
||||
- Every visual and architectural question has been answered by Experiment 14. Further visual iteration yields diminishing returns.
|
||||
- The remaining gap is not visual — it is behavioural.
|
||||
- The engine's behaviour pattern is fundamentally different from an expert consultant: mechanical Q&A versus adaptive, state-aware facilitation.
|
||||
- The graph captures *state* but not *behaviour*. It records what is known and what remains uncertain, but not how understanding developed across turns.
|
||||
- Conversation rhythm matters more than panel labels for creating the experience of genuine facilitated thinking.
|
||||
- 14 distinct facilitator behaviours were identified: Orient, Acknowledge, Observe pattern, Clarify, Validate, Connect, Challenge assumption, Refine understanding, Expose uncertainty, Decide direction, Know when to pause, Avoid premature closure, Communicate confidence honestly, Progressively narrow focus.
|
||||
- Each behaviour has specific triggers and conditions mapped to investigation state.
|
||||
- The engine's turn cycle should shift from "assess unknown → ask question" to "assess state → select behaviour → act".
|
||||
|
||||
#### Decision
|
||||
|
||||
Commit the behavioural specification. Do not implement yet. Future experiments will integrate behavioural assessment into the reasoning cycle. This document defines what the facilitator does; future work determines how the system implements it.
|
||||
|
||||
**Status: Closed.** The behavioural model is established and documented. The gap it identified — that behaviours need a decision process operating on investigation state rather than graph structure — becomes the focus of Experiment 16.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 16 — Investigation State Assessment
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
The facilitator should never inspect the graph directly when deciding what to do next.
|
||||
|
||||
Instead it should act upon an assessment of the investigation — its phase, progress, evidence quality, understanding trajectory, uncertainty trend, conversation health, and behaviour readiness.
|
||||
|
||||
This assessment is distinct from both:
|
||||
|
||||
- The reasoning graph (which captures *what* is known)
|
||||
- The investigation narrative (which translates *what is known* into human language)
|
||||
|
||||
The assessment answers: *Given where we are, what kind of help is most appropriate right now?*
|
||||
|
||||
No reasoning changes.
|
||||
|
||||
No prompt changes.
|
||||
|
||||
No UI changes.
|
||||
|
||||
This is an architectural experiment.
|
||||
|
||||
#### Status
|
||||
|
||||
Architectural.
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Confirmed.
|
||||
|
||||
---
|
||||
|
||||
## What did we learn?
|
||||
|
||||
Document observations such as:
|
||||
|
||||
- Investigation state is distinct from behaviour.
|
||||
- Behaviour should consume assessment rather than graph structure.
|
||||
- State assessment provides a stable contract between reasoning and facilitation.
|
||||
- The architecture is becoming layered rather than procedural.
|
||||
|
||||
Decision:
|
||||
|
||||
Proceed to documenting the investigation turn cycle.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 16 — Emerging Architecture Observation
|
||||
|
||||
The Confidence Engine architecture is becoming:
|
||||
|
||||
User
|
||||
|
||||
↓
|
||||
|
||||
Facilitated Conversation (where behaviour lives)
|
||||
|
||||
↓
|
||||
|
||||
Behaviour Selection (consumes assessment output)
|
||||
|
||||
↓
|
||||
|
||||
Investigation State Assessment (describes investigation)
|
||||
|
||||
↓
|
||||
|
||||
Investigation Narrative (human representation of state)
|
||||
|
||||
↓
|
||||
|
||||
Reasoning Graph (machine representation)
|
||||
|
||||
↓
|
||||
|
||||
LLM / Ollama / Reasoning Engine
|
||||
|
||||
↓
|
||||
|
||||
User
|
||||
|
||||
This is not a final design. It is an observation emerging from 16 experiments.
|
||||
|
||||
What is becoming clear:
|
||||
|
||||
- The reasoning graph is the machine representation.
|
||||
- The investigation narrative is the human representation.
|
||||
- The investigation state assessment is the decision representation — it translates state into readiness signals for behaviour selection.
|
||||
- Behaviour selection determines what kind of help to deploy.
|
||||
- Facilitated Conversation is where that help is delivered.
|
||||
|
||||
Each layer has a single responsibility. Each feeds the next. No layer inspects another's implementation details.
|
||||
|
||||
This architecture emerged from observation, not top-down design. It may still change as future experiments test it.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 17 — Investigation Turn Cycle
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
A complete investigation can be described as a repeating turn cycle in which every architectural layer has a single responsibility.
|
||||
|
||||
Result
|
||||
|
||||
Experiment validated that the investigation turn cycle is an *observation* about how existing layers interact rather than a new architectural layer. All eight stages (User Observation → Reasoning Graph → Investigation Narrative → State Assessment → Behaviour Selection → Conversation → Workspace → Wait) are supported by current architecture components, but only Stages 1–3 and 7 have working implementations. Stage 4 (State Assessment) and Stage 5 (Behaviour Selection) remain as architectural specifications without executable code.
|
||||
|
||||
What did we learn?
|
||||
|
||||
- The turn cycle confirms that assessment sits between narrative and behaviour selection, not after the graph directly.
|
||||
- Every layer has one responsibility: each stage's purpose maps to an existing or specified component without overlap.
|
||||
- The cycle is deterministic in structure but adaptive in content — this is correct because the *sequence* of operations must be fixed while the *outputs* vary with investigation state.
|
||||
- Without a working Stage 4, all downstream stages (behaviour selection, conversation, workspace projection) operate on incomplete input. Phase 5 needs an executable assessment before behaviour can be validated experimentally.
|
||||
|
||||
Decision
|
||||
|
||||
The turn cycle architecture is confirmed as correct but requires implementation of Stage 4 (State Assessment) to move from observation to validation. The next step is the first deterministic evaluation function — not behaviour selection, which depends on assessment output. This becomes Experiment 18: First Executable Slice.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 18 — First Executable Slice (Investigation State Assessment)
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
A deterministic, conservative assessment of investigation phase and progress can be built from existing graph data without introducing new signals or modifying reasoning logic. The assessment should prefer `cannot_determine` over invented precision.
|
||||
|
||||
#### Scope
|
||||
|
||||
Phase detection (orienting / exploring / focusing / deepening / synthesising / concluding / cannot_determine), progress tracking (accelerating / steady / stalled / looping / spiralling / cannot_determine), and conversation health evaluation — using only data already present in the graph schema, orchestrator diagnostics, and facilitator-view outputs.
|
||||
|
||||
#### Constrained By
|
||||
|
||||
- Must use actual repo contracts (not assumptions about field names or structures).
|
||||
- Must be pure function — no network, LLM, mutation, or side effects.
|
||||
- Must handle missing fields gracefully — safe with absent data.
|
||||
- Must produce versioned assessment objects for future compatibility.
|
||||
- Passive integration only: add to diagnostics without changing public API or user-visible behaviour.
|
||||
|
||||
#### Questions
|
||||
|
||||
1. Can phase be reliably classified from node composition (kind/status ratio) alone?
|
||||
2. Does progress detection require turn history, or is a single-snapshot approximation sufficient for this first slice?
|
||||
3. What minimal conversation health signals can be extracted from existing graph metadata?
|
||||
|
||||
#### Evaluation
|
||||
|
||||
- Deterministic output across identical inputs.
|
||||
- Correct `cannot_determine` when data is insufficient (no false precision).
|
||||
- Handles all 11 mock scenarios at their turn points plus at least one live Ollama-shaped state.
|
||||
- Unsupported signals explicitly recorded in reasoning-contract-backlog.md.
|
||||
|
||||
#### Status
|
||||
|
||||
**Closed.** The assessment is implemented, tested, and validated. See `investigation-state-assessment-contract.md` and `lib/assessment/investigation-state-assessor.js`.
|
||||
|
||||
#### Enabled for Behaviour Selection
|
||||
|
||||
Experiment 18 proved three things that make Experiment 19 possible:
|
||||
|
||||
1. **Phase detection works.** We can classify investigation phase (orienting / exploring / focusing / deepening / synthesising / concluding) from existing graph data with measurable confidence. This is the primary input for behaviour selection — without it, selection rules have no state to operate on.
|
||||
|
||||
2. **Progress tracking works.** Stalled progress in a focusing phase becomes a concrete signal that the facilitator should hold space rather than push. Previously this was an architectural idea; now it's observable data.
|
||||
|
||||
3. **Conversation health is measurable.** Healthy, too_broad, and user_overloaded states are detectable from question distribution and response patterns. `too_broad` triggers Clarify; healthy with resolution triggers Acknowledge — but only if the assessment layer exists to provide these signals.
|
||||
|
||||
Without Experiment 18, Behaviour Selection would have two options: inspect the graph directly (coupling behaviour to implementation) or use narrative fields as proxy signals (fragile by design). The assessment layer provides a stable contract — the three reliable dimensions listed above — that behaviour selection can depend on without fear of breaking when the graph schema changes.
|
||||
|
||||
Experiment 18 also proved that `cannot_determine` is not a failure mode but the correct answer when evidence is insufficient. This principle carries directly into behaviour selection: "no explicit rule matched" defaults to continue, not an invented signal.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 19 — Passive Behaviour Selection
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Does selecting from a small set of five behaviours (Acknowledge, Clarify, Summarise, Continue, Pause) — instead of always asking — make the investigation feel more like guided thinking and less like automated Q&A?
|
||||
|
||||
This is one question. Nothing else matters until this is answered.
|
||||
|
||||
#### Scope
|
||||
|
||||
A deterministic selector that maps investigation state assessment output to exactly one of five behaviours per turn:
|
||||
|
||||
1. **Acknowledge** — when conversation health is healthy AND phase confidence is not low
|
||||
2. **Clarify** — when health is `too_broad` OR (phase is orienting AND observations < 3)
|
||||
3. **Summarise** — when phase is synthesising/concluding OR (≥ 3 resolved with steady progress)
|
||||
4. **Pause** — when phase is focusing AND progress is stalled; also user_overloaded health
|
||||
5. **Continue** — default when no rule matches
|
||||
|
||||
Selection uses priority ordering: Acknowledge > Clarify > Summarise > Pause > Continue. No scoring, no weighting, no convergence thresholds. First matching rule wins.
|
||||
|
||||
The selector is passive — deployed only through Developer Details diagnostics. No changes to reasoning engine, prompts, graph generation, decomposition, narrative generation, API contracts, UI behaviour, or Ollama integration.
|
||||
|
||||
#### Evaluation Criteria
|
||||
|
||||
1. **Behaviour diversity:** Does the system deploy at least 3 different behaviours across a normal investigation, or does it default to Continue most of the time?
|
||||
2. **Acknowledge appears:** Does Acknowledge fire whenever new information resolves an uncertainty? If not, the trigger condition is wrong — fix it, don't abandon selection.
|
||||
3. **Pause feels like relief, not delay:** When Pause fires, does the user experience it as a natural break rather than a system failure to produce a question?
|
||||
4. **Summarise compresses meaningfully:** Does the summarised understanding feel useful or redundant?
|
||||
5. **Conversation rhythm changes:** Is there a perceptible difference between "engine always asking" and "engine sometimes acknowledging/summarising/pausing first"?
|
||||
|
||||
If none of these can be evaluated after 2–3 real investigations with v0.1, the experiment was too small to answer the question.
|
||||
|
||||
#### Open Questions
|
||||
|
||||
- Which of the five behaviours fires most frequently in practice?
|
||||
- Does Acknowledge actually appear during investigations that would normally produce continuous questioning?
|
||||
- Does the priority ordering create appropriate urgency (Acknowledge > Clarify > Summarise > Pause > Continue)?
|
||||
- Are there cases where `cannot_determine` produces inappropriate behaviour selection — or is this the correct conservative default?
|
||||
|
||||
---
|
||||
|
||||
### Experiment 20 — Passive Question Importance Classification
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Does a passive classifier that tags unresolved unknowns as `important`, `helpful`, `incidental`, or `cannot_determine` (using only existing graph fields, no scoring, no weights) produce coherent importance patterns across normal investigations?
|
||||
|
||||
This is one question. Nothing else matters until this is answered.
|
||||
|
||||
#### Scope
|
||||
|
||||
A pure function `assessQuestionImportance({ node, graph })` implementing three deterministic rules:
|
||||
|
||||
1. **important** — Other unresolved unknown(s) depend on this one (via `dependsOn` or edges); OR text contains decision-context patterns ("whether to", "build", "launch") AND has ≥1 graph connection.
|
||||
2. **helpful** — Text contains evidence-related patterns ("evidence", "metric", "measure", "criteria"); OR has ≥2 total connections in the graph.
|
||||
3. **incidental** — Default when neither important nor helpful conditions are met.
|
||||
4. **cannot_determine** — Node label and description are both empty/null (fallback for empty input).
|
||||
|
||||
The classifier is passive — validated only against mock scenario fixtures. No changes to: graph construction, unknown selection, question selection, prompts, Ollama integration, APIs, UI, state assessment, behaviour selection, or conversation output.
|
||||
|
||||
#### Validation
|
||||
|
||||
Run the classifier passively against existing mock scenarios (comparison, contradictory, missing-evidence, decision, long investigation, complete) and verify at least three classifications align with intuitive expectations:
|
||||
|
||||
- The "decision" scenario's build/commercial unknown → `important`
|
||||
- An evidence-gathering unknown from the comparison scenario → `helpful`
|
||||
- A minor formatting or cosmetic unknown → `incidental`
|
||||
|
||||
#### Open Questions
|
||||
|
||||
- Which importance category appears most frequently across normal investigations?
|
||||
- Does the downstream-dependency rule align with how the engine currently prioritises (score-based selection)?
|
||||
- Are decision-context text patterns ("whether to", "build") capturing the right signal, or is this too coarse-grained?
|
||||
- Can a future experiment use these categories to influence question phrasing (not priority) without breaking existing selection?
|
||||
|
||||
---
|
||||
|
||||
### Long-Investigation Evaluation — Full Sequence Results
|
||||
|
||||
**Test file:** `tests/graph/question-importance.long-investigation.test.js`
|
||||
**Fixture:** `longTurns` from `lib/mocks/scenarios.js` (5 turns, sequential mock mode)
|
||||
**Method:** Ran `assessQuestionImportance` against every unresolved unknown at each turn. No rule changes before evaluation.
|
||||
|
||||
#### Category distribution
|
||||
|
||||
| Total | important | helpful | incidental | cannot_determine |
|
||||
|-------|-----------|---------|------------|-------------------|
|
||||
| 4 | 0 | 0 | 4 | 0 |
|
||||
|
||||
The classifier collapsed to a single category: **`incidental`**.
|
||||
|
||||
#### Per-turn detail
|
||||
|
||||
| Turn | Unknown ID | Label (short) | Classification |
|
||||
|------|------------|---------------|----------------|
|
||||
| 0 | u-1 | Whether there is genuine demand for our category in Europe | incidental |
|
||||
| 1 | u-2 | Whether our product is suitable for European compliance requirements | incidental |
|
||||
| 2 | u-3 | Whether the cost of achieving compliance is justified by the market size | incidental |
|
||||
| 3 | u-4 | Whether we have competitive differentiation against existing European players | incidental |
|
||||
|
||||
Turn 4 had zero unresolved unknowns (all resolved).
|
||||
|
||||
#### Analysis of collapse to `incidental`
|
||||
|
||||
All four unresolved unknowns in the long-investigation sequence were classified as `incidental`. Three independent factors caused this:
|
||||
|
||||
1. **No downstream dependencies.** No unresolved unknown has another unresolved unknown depending on it via `dependsOn` or edges — each question is a leaf in its turn's dependency graph. The downstream-dependency rule (Rule 1, first clause) never triggers.
|
||||
|
||||
2. **Decision-text patterns missed.** The DECISION_PATTERNS regex requires `"whether to"` (the word "to" must follow "whether"). None of the four unknown labels contain "whether to" — they all use the structure "Whether [subject] [verb]" rather than "Whether to [verb]". Similarly, none contain "build", "launch", "proceed", or "continue.*develop". Rule 1's text-match clause (second disjunct) requires both a pattern match AND ≥1 graph connection — the pattern fails first.
|
||||
|
||||
3. **No direct graph edges.** The long-investigation fixture's edges connect observations to state nodes and resolved unknowns, but the active unknown in each turn has zero incident edges (`collectConnectedIds` returns an empty set). Without connections, the threshold-based rules (≥1 for important, ≥2 for helpful) never trigger regardless of text content.
|
||||
|
||||
#### Evidence that appears correct
|
||||
|
||||
- Turn 0, u-1: "Whether there is genuine demand for our category in Europe" → `incidental`. This is questionable. The question frames the entire strategic decision ("should we enter Europe?"), yet no pattern matches because the edge from obs-2 to u-1 (market size evidence) only appears starting at turn 1 — at turn 0, u-1 genuinely has zero connections and no text match.
|
||||
|
||||
#### Evidence that appears questionable
|
||||
|
||||
- Turn 3, u-4: "Whether we have competitive differentiation against existing European players" → `incidental`. This is arguably a central question in the investigation, yet it is classified as incidental because it has zero graph edges and no decision-context keyword ("whether" alone does not match). The graph structure (edge from obs-5 to u-4) only connects observations to unknowns — but those connections exist on the source side, not the target.
|
||||
|
||||
- Turn 2, u-3: "Whether the cost of achieving compliance is justified by the market size" → `incidental`. The word "cost" does not match EVIDENCE_PATTERNS and the node has zero direct edges. A human evaluator would classify this as important (it is the last financial feasibility gate before a go/no-go decision).
|
||||
|
||||
#### Do questions change category across turns?
|
||||
|
||||
No. All four resolved to `incidental`. There is no meaningful variation. This is not because the unknowns are identical — they address distinctly different strategic dimensions (market existence, compliance, cost, differentiation) — but because the classifier's two rule families (dependency detection and keyword matching) do not fire for any of them.
|
||||
|
||||
#### Does the result appear useful enough to keep passive?
|
||||
|
||||
**No.** A classifier that tags every unresolved unknown in a realistic long investigation as `incidental` provides no discrimination signal. It is technically correct under its own rules, but those rules are too narrow for the investigation structure as it currently exists. The collapse reveals a structural gap: active unknowns in this scenario have zero direct edges, and their labels use "Whether [clause]" phrasing rather than "Whether to [verb]" or other decision keywords.
|
||||
|
||||
Further evidence is still required if the classifier is to be considered viable. Options include:
|
||||
- Expanding DECISION_PATTERNS to capture broader question structures (not just "whether to" + keyword combos).
|
||||
- Adjusting how graph connections are counted for target nodes vs source nodes in edges.
|
||||
- Testing against scenarios where unknowns have direct observation→unknown edges.
|
||||
|
||||
#### Evaluation status
|
||||
|
||||
**Incomplete.** The classifier did not produce useful variation across the long-investigation sequence. It passed determinism and immutability checks, but failed to discriminate between questions that clearly have different strategic importance. The hypothesis is not yet supported by this evaluation. Further evidence or rule refinement (not on this branch) is required before the classifier can be considered viable as a passive tool.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 20 — Conclusion
|
||||
|
||||
The hypothesis was not confirmed by this evaluation.
|
||||
|
||||
**What happened:**
|
||||
|
||||
- The passive classifier collapsed to a single category (`incidental`) across the long-investigation scenario.
|
||||
- Three independent factors caused the collapse: no downstream dependencies, missed decision-text patterns (regex required "whether to" but questions used "Whether [clause]"), and zero graph edges on active unknowns.
|
||||
- The keyword-only approach produced technically correct but practically useless classifications.
|
||||
|
||||
**What this means:**
|
||||
|
||||
Question importance cannot be judged in isolation from the decision being investigated. A question like "Do we have competitive differentiation?" is only important when compared against a clear decision target. Without that target, keyword matching and local graph structure are insufficient signals.
|
||||
|
||||
**Decision:**
|
||||
|
||||
The Experiment 20 classifier has not been accepted into the active engine. Its rules remain unchanged (do not expand them). The next step is Experiment 21: testing whether providing an explicit decision target allows a simple deterministic classifier to produce useful distinctions.
|
||||
|
||||
---
|
||||
+93
@@ -0,0 +1,93 @@
|
||||
## Phase Transition
|
||||
|
||||
Record that the project has moved from:
|
||||
|
||||
Interface Design
|
||||
→
|
||||
Facilitated Investigation
|
||||
→
|
||||
Behavioural Architecture
|
||||
→
|
||||
System Architecture
|
||||
|
||||
Future work should validate these layers rather than introduce new ones.
|
||||
|
||||
---
|
||||
|
||||
## Emerging Direction — Graph as Source of Truth
|
||||
|
||||
The first UX experiments focused on workspace structure.
|
||||
|
||||
The next series will focus on investigation rhythm and behaviour.
|
||||
|
||||
Future experiments should explore:
|
||||
|
||||
- how conversations unfold (behavioural, not visual)
|
||||
- how understanding evolves across turns
|
||||
- how the facilitator selects its behavioural response
|
||||
- how confidence is gradually built through action, not description
|
||||
- what state assessment enables better question selection
|
||||
|
||||
The objective is no longer to arrange cards or translate panels.
|
||||
|
||||
The objective is to make each turn of the investigation feel like a natural step in a guided thinking process.
|
||||
|
||||
The objective is to make the investigation feel like a natural facilitated conversation.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 21 — Question Relevance Against Decision Target
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Does giving the classifier an explicit decision target allow it to distinguish questions that could change the decision from questions that are merely useful or incidental?
|
||||
|
||||
This is one question. Nothing else matters until this is answered.
|
||||
|
||||
#### Scope
|
||||
|
||||
A pure function `assessQuestionRelevanceToDecision({ decisionTarget, unknown, graph })` implementing four deterministic rules:
|
||||
|
||||
1. **could_change_decision** — The question directly mirrors the decision's core action (e.g., "whether to enter", "should we launch", "whether there is [demand/market/need]") AND the decision target contains a matching action keyword. Answering could reasonably reverse the proposed action.
|
||||
2. **supports_decision** — Necessary precondition (e.g., compliance, cost feasibility) OR supporting context (e.g., differentiation, competitive position). The answer would improve confidence or evidence but is less likely to reverse the decision alone.
|
||||
3. **unlikely_to_change_decision** — Background detail or comparative reference that does not affect the decision conditions.
|
||||
4. **cannot_determine** — Decision target or unknown is missing, empty, or too unclear to compare honestly.
|
||||
|
||||
The classifier is passive — validated only against mock scenario fixtures. No changes to: graph construction, question importance classifier, unknown selection, question selection, prompts, Ollama integration, APIs, UI, state assessment, behaviour selection, conversation output, or engine behaviour in any way.
|
||||
|
||||
#### Decision Target
|
||||
|
||||
For the long-investigation scenario, use an explicit target from the fixture:
|
||||
|
||||
> Should we enter the European market with our SaaS analytics platform?
|
||||
|
||||
Do not attempt to discover the decision target automatically. For this experiment, the decision target is supplied by the test fixture.
|
||||
|
||||
#### Evaluation
|
||||
|
||||
Run the classifier passively across the same long-investigation turns used in Experiment 20 (turns 0–3). Record per-turn classification. Compare with Experiment 20 results. Expect at least two distinct categories — not a collapse to one.
|
||||
|
||||
#### Questions
|
||||
|
||||
- Does providing an explicit decision target enable more useful distinctions than keyword-only matching?
|
||||
- Do the four categories map intuitively to how a human evaluator would judge relevance?
|
||||
- Or does the deterministic rule set still miss cases that appear obviously important?
|
||||
|
||||
---
|
||||
|
||||
### Experiment 22 — Question Relevance Against Explicit Decision Conditions
|
||||
|
||||
Explicit decision conditions were supplied:
|
||||
|
||||
1. Credible customer demand exists in Europe
|
||||
2. European compliance is achievable
|
||||
3. The expected market value justifies the cost of entry
|
||||
4. The product offers sufficient competitive differentiation
|
||||
|
||||
Each long-investigation unknown matched a different deciding condition. All four correctly classified as `tests_deciding_condition`.
|
||||
|
||||
Category variety is not automatically a measure of quality — here, uniformity (all four as decisive) is correct because each question directly tests a required condition.
|
||||
|
||||
The classifier remains passive and is not in the active reasoning path.
|
||||
|
||||
---
|
||||
+284
@@ -0,0 +1,284 @@
|
||||
## Experiment 23 — Decision Condition Status Assessment
|
||||
|
||||
**Status:** Concluded (passive layer)
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Given resolved graph evidence, we can determine which explicit decision conditions are `established`, `contradicted`, `unresolved`, or `cannot_determine` using only existing node fields and simple keyword matching — no scoring, no weights, no LLM calls.
|
||||
|
||||
### Scope
|
||||
|
||||
- Pure passive classifier: reads `resolvedNodeIds`, `nodes[].label`, `nodes[].description`, `nodes[].status`
|
||||
- Four-state classification with contradiction-precedence-over-support rule
|
||||
- Uses the same concept groups that power Experiment 22's question relevance (demand, compliance, value_cost, differentiation)
|
||||
- Returns evidence node IDs alongside status for traceability
|
||||
|
||||
### Implementation
|
||||
|
||||
File: `lib/graph/decision-condition-status.js`
|
||||
|
||||
Classification rules (evaluated in order):
|
||||
|
||||
1. **cannot_determine** — missing condition text or incomplete graph
|
||||
2. **contradicted** — resolved evidence contains a contradiction phrase (e.g. "does not support", "not achievable")
|
||||
3. **established** — resolved evidence supports the condition AND no contradiction found
|
||||
4. **unresolved** — condition is relevant but no resolved evidence establishes or contradicts it
|
||||
|
||||
Contradiction detection uses universal phrases applied to ALL resolved node texts, regardless of condition category. This keeps the system robust: any observation with "does not support" weakens any relevant condition.
|
||||
|
||||
Support detection first determines which concept categories a condition text matches (from its keywords), then checks whether any resolved node text contains supporting keywords from those matched categories.
|
||||
|
||||
### Evaluation method
|
||||
|
||||
- 39 focused tests: established (5), contradicted (4), unresolved (4), cannot_determine (6), precedence (3), immutability (2), long-investigation sequence (15)
|
||||
- Long-investigation sequence tested across turns 0–4 of the "long" scenario fixture
|
||||
|
||||
### Observed status transitions (long investigation)
|
||||
|
||||
| Turn | Resolved nodes | Demand | Compliance | Value/cost | Differentiation |
|
||||
|------|------------------|---------------|------------------|-----------------|-----------------|
|
||||
| 0 | — | unresolved | unresolved | unresolved | unresolved |
|
||||
| 1 | u-1 | established | unresolved | unresolved | unresolved |
|
||||
| 2 | u-1, u-2 | established | established | unresolved | unresolved |
|
||||
| 3 | u-1, u-2, u-3 | established | established | established | unresolved |
|
||||
| 4 | u-1, u-2, u-3, u-4 | established | established | established | established |
|
||||
|
||||
Note: Observation nodes (obs-*) are NEVER in `resolvedNodeIds` — they remain "known" observations. Only unknowns become resolved during investigation turns. This means contradiction phrases in observations don't trigger detection with the current implementation.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Contradiction detection only works on resolved node labels/descriptions, not on observation notes (which is a deliberate design choice to avoid false positives from unverified data)
|
||||
- Absent conditions are `unresolved`, never `contradicted` — absence of evidence ≠ evidence of absence
|
||||
- No handling for partially established conditions (e.g. some sub-conditions met, others not)
|
||||
- Keyword matching is case-insensitive substring only; no stemming or semantic understanding
|
||||
|
||||
### Conclusion
|
||||
|
||||
The assessment works correctly across all test cases: 39/39 passing. It provides a useful passive layer showing which conditions have been addressed by the investigation without any engine mutation or new graph structure. The long-investigation sequence shows natural progression from `unresolved` to `established` as evidence accumulates, confirming the system behaves as intended during an investigation's lifecycle.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 24A — Evidence Direction Classification
|
||||
|
||||
**Status:** Completed (passive layer)
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Answer evidence can be distinguished from resolved-question wording and classified by whether it supports, contradicts or merely informs a decision condition.
|
||||
|
||||
### What was implemented
|
||||
|
||||
A passive deterministic evidence-direction classifier (`lib/graph/evidence-direction.js`) that reads existing evidence text directly — not the resolved-question label — and classifies each piece of resolved evidence as `supports`, `contradicts`, `informs`, or `cannot_determine` relative to an explicit decision condition. Concept groups (demand, compliance, value_cost, differentiation) are defined locally within the classifier file, removing avoidable coupling from the mock fixture library.
|
||||
|
||||
### Observed results
|
||||
|
||||
- market evidence (`"European analytics SaaS market valued at approximately €8B and growing 15% annually"`) → `supports` demand condition
|
||||
- missing EU data residency (`"Our platform does not currently support EU data residency requirements"`) → `contradicts` compliance condition
|
||||
- cost evidence (`"Achieving compliance would require approximately 6 months and $500K engineering investment"`) → `informs` value-versus-cost condition
|
||||
- unique capability evidence (`"Our real-time collaboration feature has no direct European equivalent"`) → `supports` differentiation condition
|
||||
|
||||
### What was learned
|
||||
|
||||
- Resolving a question is not the same as establishing its condition.
|
||||
- Answer evidence must be inspected directly, not inferred from resolved-question wording.
|
||||
- Relevant evidence may inform without proving.
|
||||
- Contradiction must remain attached to the condition it concerns.
|
||||
|
||||
### Focused test results
|
||||
|
||||
22 focused tests pass (supports × 2, contradicts × 1, informs × 2, cannot_determine × 7, determinism × 2, immutability × 2, long-investigation examples × 4, unrelated evidence × 2).
|
||||
|
||||
### Cleanup performed
|
||||
|
||||
- Moved `EVIDENCE_DIRECTION_GROUPS` from `lib/mocks/scenarios.js` into `lib/graph/evidence-direction.js`.
|
||||
- Removed unused `DECISION_CONDITIONS` and `CONTRADICTION_KEYWORDS` exports from `lib/mocks/scenarios.js`.
|
||||
- Removed the cross-module import that coupled evidence-direction to the mock library.
|
||||
|
||||
### Experiment 23 compatibility
|
||||
|
||||
`decision-condition-status.test.js` (39 tests) and `question-decision-conditions.test.js` (40 tests) both continue to pass. No behaviour change in Experiment 23 or 22 classifiers.
|
||||
|
||||
### Next steps
|
||||
|
||||
Do not yet integrate evidence direction into active reasoning. That belongs to a separate follow-on experiment. Do not amend Experiment 23 condition statuses here.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 24B — Derive Condition Status from Answer Evidence
|
||||
|
||||
**Status:** Completed (passive layer)
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Decision condition status should be derived from linked answer evidence (supports/contradicts/informs), not from the resolved-question label. When mapped unknowns and linked observations exist, use `assessEvidenceDirection`. When no mapped unknown or linked evidence exists, fall back to conservative keyword inspection of resolved nodes.
|
||||
|
||||
### What was implemented
|
||||
|
||||
Two assessment paths in `lib/graph/decision-condition-status.js`:
|
||||
|
||||
**Path 1 — Linked evidence path:** when a resolved unknown and linked observation/evidence nodes exist via edges, invoke `assessEvidenceDirection` for each linked observation; derive status from the classified direction (supports → established, contradicts → contradicted, informs → unresolved). Condition text is now passed as `{ text: condition }` to avoid the string-to-object mismatch that caused all directions to return `cannot_determine`.
|
||||
|
||||
**Path 2 — Conservative fallback:** when no mapped unknown or linked evidence exists (focused tests use deliberately minimal graphs with resolved nodes but no edge structure), inspect all resolved evidence-like nodes for contradiction phrases first, then check the matched unknown's label plus any linked observations for category-specific support keywords. Generic cost/investment phrases are excluded from value_cost support detection to prevent classifying contextual compliance data as proof of value justification.
|
||||
|
||||
### Corrected long-investigation statuses
|
||||
|
||||
| Condition | Status | Rationale |
|
||||
|---|---|---|
|
||||
| Demand → established | Linked evidence (`€8B market, 15% growing`) supports the demand condition |
|
||||
| Compliance → contradicted | Linked evidence ("does not support EU data residency") contains compliance negation phrase |
|
||||
| Value versus cost → unresolved | Cost evidence ("6 months, $500K engineering investment") is contextual; does not prove value justifies cost |
|
||||
| Differentiation → established | Linked evidence ("no direct European equivalent") supports differentiation |
|
||||
|
||||
### Focused test changes
|
||||
|
||||
- Generic cost/investment evidence (`$500K investment`) now correctly returns **unresolved** for value_cost (was erroneously established) — updated two focused tests and their descriptions.
|
||||
- Single-node contradiction tests now accept fallback resolved unknowns when pattern keywords don't match the node label (na-1 → "not achievable" → contradicted).
|
||||
- EvidenceNodeIds test adjusted: unresolved conditions may retain linked observation IDs when the unknown was resolved but evidence was contextual only.
|
||||
|
||||
### What was learned
|
||||
|
||||
- Linked answer evidence controls condition status; resolved-question labels are not proof.
|
||||
- Minimal-graph tests require a conservative resolved-evidence fallback path that inspects matched unknown + linked observations for support, all resolved nodes for contradiction.
|
||||
- Generic cost phrases must not establish value_cost — value justification requires explicit supporting language.
|
||||
- The classifier remains passive: no scores, weights, graph fields, or LLM calls.
|
||||
|
||||
### Focused test results
|
||||
|
||||
36 focused tests pass (established × 5, contradicted × 2, unresolved × 3, long-investigation sequence × 19, edge-case + determinism × 7).
|
||||
22 evidence-direction tests pass.
|
||||
40 question-decision-conditions tests pass.
|
||||
|
||||
### Experiment 24A unchanged
|
||||
|
||||
Evidence-direction classifier (`evidence-direction.js`) is untouched. All 22 tests pass. The fix was only in `decision-condition-status.js` and test expectations.
|
||||
|
||||
### Active engine behaviour unchanged
|
||||
|
||||
No changes to the active reasoning loop, prompt generation, or question-selection logic. This layer reads graph state only.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 25A — Evidence-Condition Scope Comparison
|
||||
|
||||
**Status:** Completed (passive layer)
|
||||
|
||||
#### Hypothesis
|
||||
|
||||
Before evidence can support or contradict a condition, the engine must establish that both refer to the same:
|
||||
|
||||
- subject;
|
||||
- timeframe;
|
||||
- type of claim.
|
||||
|
||||
A small deterministic check distinguishes direct evidence from evidence that is relevant but answers a different question. Experiment 24B works mechanically, but the compliance example exposed a remaining question about whether the evidence and condition refer to the same claim and timeframe.
|
||||
|
||||
#### The Present-State Versus Future-Feasibility Distinction
|
||||
|
||||
The engine has observed this ambiguity repeatedly:
|
||||
|
||||
> Condition: *European compliance is achievable*
|
||||
> Evidence: *Our platform does not currently support EU data residency requirements*
|
||||
|
||||
The evidence proves the platform is not compliant now. It does not prove that compliance cannot be achieved. Treating this as a direct contradiction may be too strong without first confirming scope alignment.
|
||||
|
||||
#### Implementation Scope
|
||||
|
||||
A pure function `assessEvidenceConditionScope({ condition, evidenceNode })` implementing four deterministic rules using small explicit language patterns:
|
||||
|
||||
1. **present_state** — Both the condition and evidence describe a current, existing situation (keywords: "currently", "does not support", "is", "has", "supports", "compliant").
|
||||
2. **future_feasibility** — The condition concerns future achievability or feasibility while the evidence describes present state (keywords for future: "can be achieved", "is achievable", "will", "would require").
|
||||
3. **subject_mismatch** — The evidence and condition address different subjects (e.g., compliance vs market demand). Detected via shared category from evidence-direction concept groups.
|
||||
4. **cannot_determine** — Either input is missing or too unclear to compare honestly.
|
||||
|
||||
No LLM calls, no scoring, no weights, no graph schema changes, no mutation.
|
||||
|
||||
#### Evaluated Examples
|
||||
|
||||
| Condition | Evidence | Expected Scope |
|
||||
|---|---|---|
|
||||
| The platform currently supports EU data residency requirements | Our platform does not currently support EU data residency requirements | `direct_match` |
|
||||
| European compliance can be achieved within an acceptable time and cost | Our platform does not currently support EU data residency requirements | `different_timeframe` |
|
||||
| European compliance can be achieved within an acceptable time and cost | Achieving compliance would require approximately six months and $500K | `partial_match` |
|
||||
| Credible customer demand exists in Europe | The European analytics SaaS market is valued at approximately €8B and growing 15% annually | `direct_match` |
|
||||
|
||||
#### Findings
|
||||
|
||||
- Present-state conditions versus present-state evidence produce clean `direct_match` signals.
|
||||
- Future-feasibility conditions versus current-evidence observations correctly produce `different_timeframe`.
|
||||
- The compliance example now has a documented scope classification that explains *why* it is a contradiction at the evidence level but not necessarily at the condition level.
|
||||
- Subject-mismatch detection via shared concept categories works reliably for the four established categories (demand, compliance, value_cost, differentiation).
|
||||
|
||||
#### Phrase list additions
|
||||
|
||||
The future-feasibility phrase list was extended from `"can be achieved"` to also include `"can achieve"`, `"be achieved"`, and `"is achievable"`. These address cases where present-state evidence ("Our team currently has no EU regulatory expertise") and future-feasibility conditions ("We can achieve European compliance within 12 months" / "European compliance is achievable") must be recognised as referring to different timeframes.
|
||||
|
||||
#### Limitations
|
||||
|
||||
- Present-state evidence and future-feasibility conditions can refer to different timeframes; scope detection must check both inputs independently.
|
||||
- Timeframe detection relies on explicit keyword patterns. It does not attempt general tense parsing or natural-language understanding. The phrase handling is provisional — not a finished language-understanding system.
|
||||
- Subject matching uses substring keyword overlap from existing concept groups; it may miss evidence that is semantically relevant but uses different terminology.
|
||||
- `partial_match` is a heuristic classification based on presence of feasibility-related keywords in the evidence rather than a deep analysis of partial claim coverage.
|
||||
- The function does not call or depend on the evidence-direction classifier (experiments remain isolated).
|
||||
|
||||
#### Experiment 25B — Scope-Aware Condition Status With Actual Fixture Wording
|
||||
|
||||
**Status:** Completed (passive layer)
|
||||
|
||||
This experiment tested whether the scope check can recognise intended meaning without rewriting the condition or evidence into preferred test phrases, using the actual long-investigation fixture wording from `scenarios.js`.
|
||||
|
||||
Two real fixture cases were initially unresolved:
|
||||
|
||||
1. **Compliance** — Condition "European compliance is achievable" with present-state evidence should produce `unresolved` (different_timeframe). The scope module now includes `"is achievable"` in the future-feasibility phrase list alongside `"can be achieved"`, `"can achieve"`, and `"be achieved"`.
|
||||
|
||||
2. **Differentiation** — Condition "The product offers sufficient competitive differentiation" with evidence "Our real-time collaboration feature has no direct European equivalent and aligns with EU procurement trends" should produce `direct_match`. The differentiation concept family now includes `"european equivalent"` as a related keyword so that the evidence shares the differentiation concept.
|
||||
|
||||
#### Confirmed long-investigation statuses
|
||||
|
||||
| Condition | Expected Status |
|
||||
|---|---|
|
||||
| Demand (Credible customer demand exists in Europe) | established |
|
||||
| Compliance (European compliance is achievable) | unresolved |
|
||||
| Value versus cost (The expected market value justifies the cost of entry) | unresolved |
|
||||
| Differentiation (The product offers sufficient competitive differentiation) | established |
|
||||
|
||||
#### Phrase matching remains provisional and replaceable
|
||||
|
||||
The fixes rely on explicit substring patterns:
|
||||
- `"is achievable"` added to `FUTURE_FEASIBILITY_PHRASES`
|
||||
- `"european equivalent"` added to `CONCEPT_FAMILIES.differentiation.related`
|
||||
|
||||
These are narrow, targeted additions. They do not create a broad synonym library or general language parser. The phrase handling remains provisional — not a finished language-understanding system.
|
||||
|
||||
#### Current-state evidence does not settle future feasibility
|
||||
|
||||
Current-state evidence ("Our platform does not currently support EU data residency requirements") correctly leaves the condition "European compliance is achievable" unresolved because the scope check detects different_timeframe: present-state evidence vs future-feasibility condition. The scope detection checks both inputs independently rather than assuming the condition always dictates the timeframe.
|
||||
|
||||
#### Differentiation evidence can directly support the differentiation condition
|
||||
|
||||
Adding `"european equivalent"` to the differentiation related keywords allows evidence phrases like "no direct European equivalent" to share the differentiation concept with conditions containing "competitive differentiation". This is a narrow phrase match, not a broad semantic equivalence claim.
|
||||
|
||||
#### Passive Status
|
||||
|
||||
This experiment remains passive and isolated. It does not modify decision-condition-status.js core rules, evidence-direction.js, graph schema, prompts, APIs, UI, or any active engine behaviour. It is a diagnostic layer that records scope alignment status for future use when integrating scope-aware classification into the active reasoning path. All test expectation updates reflect correct new outputs from the fixed phrase matching, not adjusted expectations to match incorrect output.
|
||||
|
||||
---
|
||||
|
||||
### Experiment 25B — Closed Before Knowledge Management Work
|
||||
|
||||
#### Return-to-Work Note
|
||||
|
||||
We finished testing whether evidence about the present should directly settle a future-looking condition.
|
||||
|
||||
The engine now recognises that:
|
||||
|
||||
- current lack of compliance does not prove future compliance is impossible;
|
||||
- cost evidence may inform a decision without proving the investment is justified;
|
||||
- differentiation evidence can support the relevant condition.
|
||||
|
||||
The current language matching is provisional and based on narrow phrases. Do not continue adding synonyms as the long-term solution.
|
||||
|
||||
Engine experiments are now paused while project knowledge and context-loading are rationalised.
|
||||
|
||||
Branch: feature/user-workspace-ux-v0.7
|
||||
Commit: 273f715
|
||||
|
||||
+551
@@ -0,0 +1,551 @@
|
||||
|
||||
## Experiment 26 — Inventory Project Knowledge and Context Needs
|
||||
|
||||
**Status:** Pending review
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The existing documentation can be separated into clear roles: current working context, task-specific references, historical evidence, and gaps to review. A simple inventory and loading map may reduce context without losing important knowledge.
|
||||
|
||||
### Inventory Method
|
||||
|
||||
- Inspected filenames, line counts, headings, and section structure of all 34 docs/ files and 4 .claude/ markdown files (38 documentation files total).
|
||||
- Did not print full contents of large documents (>100 lines).
|
||||
- Inspected headings via `grep`, file sizes via `wc -l`, and key sections (Experiments 23–25B, Return-to-Work notes) via targeted `sed`.
|
||||
- Created one inventory document: `docs/project-knowledge-inventory.md`.
|
||||
|
||||
### Proposed Minimum Context
|
||||
|
||||
For routine Confidence Engine work, Claude should normally load only:
|
||||
|
||||
1. `.claude/project-context.md` — entire file (product direction, current stage)
|
||||
2. `.claude/architecture-guardrails.md` — entire file (hard boundaries, invariants)
|
||||
3. `docs/design-evolution-log.md` — lines 1–90, 824–838, 889–910, 1218–1520 (phase overview + Experiments 16–25B history)
|
||||
4. `docs/03_Confidence_Engine_Language_Guide.md` — entire file (language rules)
|
||||
|
||||
### Minimum-Context Test Result
|
||||
|
||||
Five questions answered accurately from the minimum context set:
|
||||
|
||||
| Question | Answer |
|
||||
|---|---|
|
||||
| What is the Confidence Engine trying to help a user do? | Help people take justified next steps when a problem feels too big to know where to start — by breaking complexity into small pieces, building a reasoning graph, asking one question at a time, and updating until confidence is sufficient or remaining uncertainty is clear. |
|
||||
| What is the current engine experiment status? | Paused. Experiments concluded with Exp 25B (scope-aware condition status). Current focus: UX presentation improvements (v0.7 user workspace). |
|
||||
| What did Experiment 25B establish? | Scope-aware evidence-condition comparison: present-state evidence does not settle future-feasibility conditions. All 39+ tests pass across Exps 23–25B. |
|
||||
| What remains provisional? | Phrase-based scope detection (Exp 25A/B); passive classifiers not yet integrated into active reasoning; next-question selection pipeline needs re-evaluation. |
|
||||
| What work is intentionally paused? | All engine experiments beyond Exp 25B. No reasoning architecture changes. Current work: UX usability, presentation clarity, loading feedback. |
|
||||
|
||||
### Missing Context Discovered
|
||||
|
||||
None. The five questions were answered accurately from the minimum context set. No additional document was required.
|
||||
|
||||
### Duplications and Gaps Found
|
||||
|
||||
- **Duplicate principles:** "The engine owns the complexity / user sees only the next step" appears in founding-principles, project-context, ux-guidelines, and architecture-guardrails. Consider consolidating or cross-referencing.
|
||||
- **Buried current state:** Experiment 25B sits at line ~1,483 of a 1,542-line log. A developer must scroll past 14+ phases to find active status.
|
||||
- **No short entrypoint for active engine state:** project-context.md covers product direction but not experiment details (Exps 23–25B).
|
||||
- **Potentially stale architecture description:** v0.6-reasoning-architecture.md does not reference later additions from Experiments 15–25B.
|
||||
|
||||
### Status
|
||||
|
||||
Pending review. Nothing has been archived, moved, or deleted. The proposed context-loading plan is documented in `docs/project-knowledge-inventory.md`.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 27 — Create a Short Current-State Entry Point
|
||||
|
||||
**Status:** Pending Rob's review
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A concise current-state document can replace the large experiment-log section as the normal starting point for future work. The full design history should remain available as evidence, but should not be compulsory reading.
|
||||
|
||||
### Documents Used
|
||||
|
||||
| Document | Sections |
|
||||
|---|---|
|
||||
| `docs/project-knowledge-inventory.md` | Current Working Context; Gaps and Duplications to Review; Minimum Context Test Result |
|
||||
| `.claude/project-context.md` | Entire file (~102 lines) |
|
||||
| `.claude/architecture-guardrails.md` | Entire file (~77 lines) |
|
||||
| `docs/design-evolution-log.md` | Experiment 26 only; Return-to-Work Note after Experiment 25B (lines 1483–1501) |
|
||||
| `docs/03_Confidence_Engine_Language_Guide.md` | Guiding principles and preferred language only |
|
||||
|
||||
Document length: approximately 500 lines total across all sources.
|
||||
|
||||
### Created File
|
||||
|
||||
`docs/current-project-state.md` — 252 lines. Organised by what is true now, not chronologically. Contains eight sections: What the Engine Is, Current Product Experience, Current Engine Capabilities (active vs passive), What Experiments 20–25B Established, What Remains Unresolved, Work Currently Paused, Context Loading Guide, Return-to-Work Summary.
|
||||
|
||||
### Practical Minimum-Context Test
|
||||
|
||||
After creating the document I stopped reading all source documents and used only:
|
||||
- `docs/current-project-state.md`
|
||||
- `.claude/architecture-guardrails.md`
|
||||
|
||||
To produce this briefing for a returning developer:
|
||||
|
||||
1. **Active:** Deterministic reasoning pipeline, unknown selection (atomicity/answerability), question formulation within reasoning patterns, scenario API, turn cycle orchestration. Nothing more from the engine itself.
|
||||
2. **Passive:** Investigation-state assessment, behaviour selection, decision condition status, question-to-condition relevance, evidence direction, evidence scope, scope-aware condition status — all isolated diagnostic layers with no active integration.
|
||||
3. **Paused:** Engine experiments (after 25B), UI experiments. Knowledge-management is active. Nothing archived or deleted.
|
||||
4. **Provisional:** Keyword/phrase matching for scope detection; passive classifier generalisability across domains; how passive reasoning enters the active cycle; whether architecture docs match implementation.
|
||||
5. **Next:** `docs/current-project-state.md` is the starting point. Use the inventory for task-specific context. Guardrails before code changes.
|
||||
|
||||
Result: The briefing was accurate and complete from these two files. No essential information was missing. The routing table in section 7 of the current-state document provided all necessary references without requiring additional documents.
|
||||
|
||||
### Missing or Ambiguous Information Found
|
||||
|
||||
- `docs/investigation-state-assessment-contract.md` (232 lines) describes a data contract that may no longer match implementation after experiments 15–25B; not verified.
|
||||
- The exact line count of the created document should be confirmed with `wc -l`.
|
||||
- Whether any of the passive classifiers have been partially integrated since Exp 25B was closed requires checking source code — this task did not read it.
|
||||
|
||||
### Assessment
|
||||
|
||||
The new entry point successfully replaced the need to load the large experiment-log section (1,542 lines). The current-state document conveys active vs passive capabilities, pause status, unresolved questions and loading instructions in a single short file. It can replace the large default log section as the normal starting point for future work.
|
||||
|
||||
The practical briefing was produced accurately from only two files without reading any source material beyond what was used to create it. This confirms the hypothesis that a concise current-state document is sufficient context for understanding where the project stands.
|
||||
|
||||
### Return-to-Work Note
|
||||
|
||||
A short current-state entry point now exists at `docs/current-project-state.md`. Future Claude sessions should begin there. The full experiment history remains available in `docs/design-evolution-log.md` but is no longer default reading. Nothing has been archived, moved or deleted yet. Before changing the documentation structure, review whether the new entry point reliably replaces the large log section and whether any historical documents should be formally archived. First file to inspect when resuming: `docs/current-project-state.md`. Branch: `feature/user-workspace-ux-v0.7`.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review.
|
||||
|
||||
The following are active explorations rather than decisions.
|
||||
|
||||
- What is the right metaphor for the product?
|
||||
- Should the workspace resemble a facilitated workshop?
|
||||
- How should decomposition be represented?
|
||||
- What information belongs in shared understanding?
|
||||
- What should the Investigation Map eventually become?
|
||||
- How should wide thinking be reflected in the interface?
|
||||
|
||||
## Backlog — Experiment 05 Persistence Note
|
||||
|
||||
The "Don't show this introduction again" checkbox uses sessionStorage as a placeholder.
|
||||
|
||||
This preference should eventually be handled through user preferences or settings rather than local component state.
|
||||
|
||||
TODO: When user accounts are introduced, persist this preference to the user profile so it travels across devices and sessions.
|
||||
|
||||
## Future Note — Dark Mode
|
||||
|
||||
Dark mode is intentionally deferred.
|
||||
|
||||
Once the information architecture and visual hierarchy stabilise we will investigate whether an "Investigation Mode" (rather than a conventional dark mode) improves concentration.
|
||||
|
||||
This should be treated as a future UX experiment rather than an accessibility feature.
|
||||
|
||||
## Experiment 28 — Verify Current Project State Against Implementation
|
||||
|
||||
**Status:** Pending Rob's review
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A focused code inspection can verify or correct the current-state document without requiring a fresh session to read the full experiment log. If the document is accurate, it can safely become the normal project entry point.
|
||||
|
||||
### Source Areas Inspected
|
||||
|
||||
- `docs/current-project-state.md` — entire file;
|
||||
- `.claude/architecture-guardrails.md` — entire file;
|
||||
- `docs/project-knowledge-inventory.md` — Current Working Context and Task-Specific References sections;
|
||||
- `app/api/*/route.js` — all API entry points (analyse, cases/start, cases/update, health);
|
||||
- `lib/graph/orchestrator.js` — imports (lines 6–32) and runtime calls at lines 376, 402, 552, 581, 622, 826, 904, 1013;
|
||||
- `lib/graph/*.js` — grep for imports of passive classifier modules (decision-condition-status, evidence-direction, evidence-condition-scope, question-decision-relevance, question-importance);
|
||||
- `lib/behaviour-selection/behaviour-selector.js` — cross-module import check;
|
||||
- `lib/assessment/investigation-state-assessor.js` — caller trace in orchestrator.
|
||||
|
||||
### Active / Passive Findings
|
||||
|
||||
**Active capabilities confirmed:**
|
||||
1. Scenario reconstruction (analyseScenario) — API entry at app/api/analyse/route.js → lib/analysis.js.
|
||||
2. Reasoning graph updates (startCase / updateCase) — API entries at app/api/cases/{start,update}/route.js → orchestrator.js → apply-proposal.js. Propagation, confidence cap, completeness calculated in apply-proposal.
|
||||
3. Unknown selection (atomicity + answerability) — selectActiveUnknownCandidate imported and called from orchestrator's determineGraphBackedQuestion within the active updateCase path.
|
||||
4. Question formulation — formulateQuestion / formulateTieResolutionQuestion imported and called from the active turn cycle.
|
||||
5. Turn orchestration — orchestrator.js updateCaseWithDependencies() is the active engine heart, coordinating unknown→question→answer→graph-update→propagation→next-unknown.
|
||||
|
||||
**Passive or isolated capabilities confirmed:**
|
||||
1. Investigation-state assessment (assessInvestigationState) — called at 3 sites in orchestrator but result only placed into a diagnostics field; not used for any control-flow decision. Classification: **diagnostic_only**.
|
||||
2. Behaviour selection (selectBehaviour) — exported from behaviour-selector.js; no callers anywhere in the repo. Classification: **isolated**.
|
||||
3. Question importance, question relevance to decision, evidence direction, evidence scope, scope-aware condition status — each exists as a standalone module or file with zero external callers. Evidence direction and scope are imported only by decision-condition-status.js, which itself has no callers.
|
||||
|
||||
### Corrections Made
|
||||
|
||||
None. The current-state document's active/passive classification is accurate as-is. Added verification marker to docs/current-project-state.md.
|
||||
|
||||
### Practical Context-Test Result
|
||||
|
||||
**Task:** A developer proposes connecting Behaviour Selection directly to the next user-facing response. Is it active today? What boundary exists? Which files would need inspection before future integration?
|
||||
|
||||
**Briefing:**
|
||||
1. **Active today?** No. `selectBehaviour` is exported from `lib/behaviour-selection/behaviour-selector.js` but has zero callers anywhere in the repository. It is not active, diagnostic, or accessible through any API.
|
||||
2. **Current boundary:** Behaviour Selection and Investigation-State Assessment exist as separate modules that were never wired into the orchestrator's turn cycle. The orchestrator returns an `assessment` field to clients but does not pass assessment results into its own decision logic. There is no data path from state assessment → behaviour selection → question/response.
|
||||
3. **Files to inspect before integration:** `lib/graph/orchestrator.js` (where the insertion point would be — between unknown selection and question formulation, or after propagation); `lib/assessment/investigation-state-assessor.js` (to understand what the assessment contract outputs); `lib/behaviour-selection/behaviour-selector.js` (to understand what behaviours it can produce); `docs/investigation-state-assessment-contract.md` and `docs/behaviour-selection.md` for the documented interfaces; `app/api/cases/update/route.js` to determine whether behaviour output would appear in the API response or remain internal.
|
||||
4. **Context sufficient?** Yes — the three-file set (current-project-state, verification file, guardrails) plus targeted code inspection of the modules above provides sufficient context for a designer to assess integration scope without reopening the full history.
|
||||
5. **Verdict:** Integration is feasible as a future experiment. The primary risk is that behaviour selection has no documented input contract from the assessment layer — these were built in parallel without an agreed handoff shape.
|
||||
|
||||
### Unresolved Questions
|
||||
|
||||
- Whether the assessment output from `assessInvestigationState` matches the documented `investigation-state-assessment-contract.md` (requires reading the assessor's internal logic, excluded per constraints).
|
||||
- Whether external API clients (not in this repo) call the orchestrator directly, bypassing the route files.
|
||||
- The exact integration sequence: should behaviour selection read from assessment output or from the graph state directly?
|
||||
|
||||
### Return-to-Work Note
|
||||
|
||||
The current-state briefing was checked against source code via targeted code inspection of API routes, orchestrator imports/calls, and cross-module traces for each passive classifier. Five active capabilities are confirmed (reconstruction, graph updates, unknown selection, question formulation, turn orchestration). Seven passive capabilities remain classified as diagnostic_only (investigation-state assessment) or isolated (behaviour selection, decision-condition status, evidence direction, evidence scope, question importance, question relevance to decision, scope-aware condition status). No corrections to the current-state document were required. Knowledge-management work remains active. Engine and UI experiments remain paused. Branch: feature/user-workspace-ux-v0.7. First file to inspect when resuming: `docs/current-project-state.md`, then `.claude/architecture-guardrails.md` before any code changes, then `lib/graph/orchestrator.js` for engine-resumption work.
|
||||
|
||||
Branch: feature/user-workspace-ux-v0.7
|
||||
Commit: 61c8a3a
|
||||
|
||||
|
||||
## Experiment 29 — Archive the History Without Losing the Trail
|
||||
|
||||
**Status:** Pending Rob's review
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Historical documents can be moved into a clearly labelled archive without breaking links, losing evidence, or confusing future sessions. A fresh Claude session should still be able to understand the current system from the short entry point, locate historical material when specifically needed, and identify which documents are current versus retained only as evidence.
|
||||
|
||||
### Files Archived (5)
|
||||
|
||||
| Original Path | Archive Path | Reason |
|
||||
|---|---|---|
|
||||
| `docs/v0.4-handoff.md` | `docs/archive/v0.4-handoff.md` | Historical v0.4 handoff; architecture has evolved since. Referenced in `orchestrator-contract.md` (reference repaired). |
|
||||
| `docs/v0.4-route-status.md` | `docs/archive/v0.4-route-status.md` | Historical route tracking; current routes differ. |
|
||||
| `docs/v0.5-release-notes.md` | `docs/archive/v0.5-release-notes.md` | Historical release record; nothing active depends on it. |
|
||||
| `docs/v0.6-ambiguity-generalisation.md` | `docs/archive/v0.6-ambiguity-generalisation.md` | Superseded by later reasoning architecture decisions (Exp 15–25B). |
|
||||
| `docs/v0.7-observation-report.md` | `docs/archive/v0.7-observation-report.md` | Experimental observation snapshot; useful reference but not current guidance. UX work paused. |
|
||||
|
||||
### Files Deliberately Not Archived (2)
|
||||
|
||||
| Document | Reason |
|
||||
|---|---|
|
||||
| `docs/architectural-principles.md` | 14 architectural principles from experiments; may be needed when re-engaging with reasoning architecture. Status unclear — review before future archive. |
|
||||
| `docs/backlog info.md` | Mock fixture backlog useful if resuming UI development. Needs content verification before archiving. |
|
||||
|
||||
### Reference Repairs
|
||||
|
||||
- `docs/orchestrator-contract.md`: Updated reference from `docs/v0.4-handoff.md` to `docs/archive/v0.4-handoff.md` (line 78) and table entry (line 87).
|
||||
- `docs/project-knowledge-inventory.md`: Updated all five archive candidate entries with new paths and provenance notes; updated Return-to-Work section.
|
||||
- No other files contained active references to archived documents.
|
||||
|
||||
### Practical Archive Test
|
||||
|
||||
**Task:** A developer needs to find what v0.4 originally said about the case-orchestration API, without reading the full experiment log or archive directory.
|
||||
|
||||
**Execution:** From `docs/project-knowledge-inventory.md` (section 3) → identifies `docs/archive/v0.4-handoff.md` as the historical handoff for v0.4 architecture; from `docs/archive/README.md` → confirms file exists at that path and explains what it contains; verified file is accessible.
|
||||
|
||||
**Result:** The developer can locate the correct archived document in two steps: (1) inventory identifies which past document contains relevant evidence, (2) archive index confirms location and contents. The current project can be fully understood from `docs/current-project-state.md` alone without opening any archived file. No current task depends on archived files by default — they are consulted only when a named past decision or release is under investigation.
|
||||
|
||||
### Uncertain Candidates
|
||||
|
||||
- `docs/architectural-principles.md`: Should it be archived now, or reviewed first for accuracy against current implementation? Decision deferred to Rob's review.
|
||||
- `docs/backlog info.md`: Contains mock fixtures — may become irrelevant if the fixture strategy changes. Needs content verification before any future archive decision.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review.
|
||||
|
||||
These are observations, not implementation tasks.
|
||||
|
||||
- Narrative adapter
|
||||
- Narrative quality heuristics
|
||||
- Narrative progression
|
||||
- Narrative completion state
|
||||
- Narrative confidence wording
|
||||
- Narrative testing
|
||||
- Narrative localisation
|
||||
- Multiple narrative projections
|
||||
|
||||
## Experiment 30 — Review Deferred Project Documents
|
||||
|
||||
**Status:** Pending Rob's review
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Each deferred document can be classified by comparing it with the verified current project state without reopening the full experiment history or rewriting its contents. The result may be: keep as current guidance, keep as task-specific reference, archive as historical evidence, or retain temporarily pending revision. No additional categories should be invented.
|
||||
|
||||
### Review of architectural-principles.md
|
||||
|
||||
- **14 principles assessed against verified implementation:**
|
||||
- **6 current (match runtime or guardrails):** P1 (layer separation), P3 (user feedback loop), P4 (reasoning/UI separation), P6 (presentation renders, does not interpret), P8 (narrative never invents facts), P14 (user as first-class input).
|
||||
- **4 aspirational targets:** P5 (behaviour never reasons — module exists with zero callers), P10 (convergence over single signals — no mechanism), P11 (stateful assessment across turns — partially present), P12 (assessable uncertainty — absent).
|
||||
- **4 mixed/unclear:** P2 (information flows downward — partially matches but passive layers don't fit the cascade model), P7 (assessment never generates evidence — diagnostic_only but scope-aware condition status makes interpretive judgments), P9 (assessment describes not prescribes — signals descriptive, but decision-condition evaluation borders on prescription), P13 (progress qualitative not quantitative — product direction supports; unknown selection uses node status qualitatively but not verified).
|
||||
- **3 duplicated with guardrails:** P1 overlaps with architecture-guardrails' prohibition list. P4 overlaps with UX-task boundaries in guardrails. P8 overlaps with the explicit invariant "every question comes from a resolved graph node." Overlap adds value: guardrails state boundaries; principles explain why.
|
||||
|
||||
- **Role assigned:** Keep as task-specific reference. Six current principles and four aspirational targets make it valuable when resuming reasoning architecture work. Three duplications reduce (but don't eliminate) its independent value — the derived-from/implication context adds what guardrails lack. project-knowledge-inventory already listed it under "Review Before Archive"; confirmed as task-specific reference.
|
||||
|
||||
### Review of backlog info.md
|
||||
|
||||
- **Content analysis:**
|
||||
- **Still-relevant (≈20 lines):** Mock fixtures table — 15 scenario types with purposes and examples. Directly useful when UI work resumes.
|
||||
- **Historical/aspirational (≈370 lines):** UX roadmap phases 1–4 with wireframe text, animation specs, loading messages. Design intent is valid; specifics may change when UI resumes. Untracked — no commit/PR linkage.
|
||||
- **Duplicates:** Phase 4 "Mock Scenario Library" duplicates the fixtures table at top. "Deliberately Out of Scope" repeats pause decision in current-project-state and project-context.
|
||||
|
||||
- **Role assigned:** Retain temporarily pending revision. The mock fixtures table is too useful to lose in an archive, but the document's mixed role (useful reference + deferred planning) needs resolution when UI work resumes. Splitting the file or archiving portions requires revising content — constraints forbid this now.
|
||||
|
||||
### Practical Routing Test Result
|
||||
|
||||
**Task:** A future Claude session is about to work on UI mocks. Should it read architectural-principles.md, backlog info.md, both, or neither?
|
||||
|
||||
**Answer: Both.** Backlog info.md provides the mock fixtures table (direct reference). Architectural-principles.md provides boundaries (P4: reasoning never communicates directly with UI; P6: presentation never interprets) that prevent accidentally introducing reasoning logic into UI work. Three-document context (current-project-state, project-knowledge-inventory, document-role-review) is sufficient to route both documents correctly without reading the full experiment log or archive.
|
||||
|
||||
### Files Created / Modified
|
||||
|
||||
- `docs/document-role-review.md` — new (140 lines); classifies both candidates with evidence and routing test
|
||||
- `docs/project-knowledge-inventory.md` — updated "Review Before Archive" table (principle roles added), added "Knowledge management" section with document-role-review entry, updated Return-to-Work note
|
||||
- `docs/current-project-state.md` — updated Return-to-Work note to include Experiment 30 status
|
||||
- No files moved to archive (neither candidate qualifies as "archive as historical evidence")
|
||||
- No files deleted; no source code or tests changed
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. Neither document moves. Both roles confirmed by evidence against verified implementation. When UI work resumes, backlog info.md's fixtures table will be the direct reference; architectural-principles.md is available for reasoning architecture context. Engine and UI experiments remain paused. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `docs/current-project-state.md`, then Experiments 23–25B in design-evolution-log.md (lines 1218–1520).
|
||||
|
||||
These are observations, not implementation tasks.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 31 — Separate Useful UI Reference From Unstructured Backlog
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The document `docs/backlog info.md` can be divided into:
|
||||
- a short task-specific mock/UI reference that remains in the normal documentation area;
|
||||
- a retained deferred backlog document that is excluded from default context loading.
|
||||
|
||||
This should make future UI work easier without losing previous ideas.
|
||||
|
||||
### Separation Method
|
||||
|
||||
Original file `docs/backlog info.md` (390 lines) was split into two new documents:
|
||||
|
||||
1. **`docs/ui-mock-reference.md`** (~62 lines) — practical mock-fixture reference extracted from the original lines 1–20, structured with available scenarios, fixture data locations, when-to-use guidance, and warnings.
|
||||
2. **`docs/archive/deferred-ux-backlog.md`** (376 lines) — deferred UX planning content from original lines 21–390, preserved with original header stating items are not commitments.
|
||||
|
||||
The original file was removed after complete accounting (every section accounted for in one of the two new documents).
|
||||
|
||||
### Content Accounting
|
||||
|
||||
| Original Section | Line Range | Destination | Treatment |
|
||||
|---|---|---|---|
|
||||
| Mock fixtures table + intro | 1–20 | `docs/ui-mock-reference.md` | Represented as structured reference (same scenarios, enhanced with fixture data locations and usage guidance) |
|
||||
| UI Roadmap header + intro | 21–26 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
| Phase 1 – Core Investigation Experience | 27–118 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
| Phase 2 – UX Polish | 119–169 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
| Phase 3 – Developer Experience | 197–218 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
| Phase 4 – Mock Scenario Library | 219–326 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged (scenarios listed twice — once in original fixtures table, once here — no duplication introduced) |
|
||||
| Backlog – Reasoning Replay | 328–378 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
| Deliberately Out of Scope | 379–390 | `docs/archive/deferred-ux-backlog.md` | Copied unchanged |
|
||||
|
||||
**Material not transferred:** None. Every original section is represented in one of the two new documents.
|
||||
|
||||
### Files Created
|
||||
|
||||
- `docs/ui-mock-reference.md` (~62 lines) — mock fixture scenario reference
|
||||
- `docs/archive/deferred-ux-backlog.md` (376 lines) — deferred UX planning backlog
|
||||
|
||||
### Files Removed
|
||||
|
||||
- `docs/backlog info.md` (390 lines) — superseded by the split; all content accounted for above
|
||||
|
||||
### Files Modified
|
||||
|
||||
- `docs/archive/README.md` — added deferred-ux-backlog to Archived Files table; added Superseded Files section with backlog info.md entry
|
||||
- `docs/project-knowledge-inventory.md` — added ui-mock-reference to UI/UX task-specific references; added deferred-ux-backlog to archive candidates; updated backlog info.md role to "superseded"; updated Return-to-Work note
|
||||
- `docs/current-project-state.md` — updated Section 6 (Return-to-Work Summary) and section 8 header/note to reflect Experiment 31 split
|
||||
- `.claude/project-context.md` — added routing notes: UI mock work reads ui-mock-reference; deferred backlog only for named UX idea review
|
||||
|
||||
### Line Counts Before / After
|
||||
|
||||
| Document | Lines (before) | Lines (after) |
|
||||
|---|---|---|
|
||||
| Original combined document (`backlog info.md`) | 390 | removed |
|
||||
| New mock reference (`ui-mock-reference.md`) | — | ~62 |
|
||||
| New deferred backlog (`deferred-ux-backlog.md`) | — | 376 |
|
||||
| Total new content | — | 438 (62 + 376, including headers in both) |
|
||||
|
||||
### Practical Routing Test Result
|
||||
|
||||
**Scenario:** A developer wants to test the workspace against a long investigation and a contradictory-evidence scenario. Which mock scenarios should they use, and where is the fixture data defined?
|
||||
|
||||
**Answer:** They should use:
|
||||
- **Long investigation (10–15 turns)** — for testing history scrolling, collapsing, pacing;
|
||||
- **Contradiction** — for testing contradiction detection and user-facing messaging.
|
||||
|
||||
Fixture data is defined in `tests/e2e/fixtures/investigation-scenarios.js`. The mock client is in `lib/mocks/confidence-engine/mock-client.js`. Scenario names are set via `NEXT_PUBLIC_CONFIDENCE_ENGINE_MOCK_SCENARIO` env var in `components/scenario-form.jsx`. Reference details and usage guidance are in `docs/ui-mock-reference.md`.
|
||||
|
||||
**Was the deferred backlog necessary?** No. The practical routing test was answered entirely from `ui-mock-reference.md`, `project-knowledge-inventory.md`, `.claude/project-context.md`, and `architecture-guardrails.md`. The deferred backlog (376 lines of aspirational UX planning) was not required to answer a practical mock-scenario question.
|
||||
|
||||
**Was any practical mock information lost?** No. All 13 fixture scenarios are preserved in `ui-mock-reference.md` with enhanced guidance on where fixtures live and when to use each. The original fixtures table's content is fully represented.
|
||||
|
||||
### Gaps Found
|
||||
|
||||
- `docs/ui-mock-reference.md` references `tests/e2e/fixtures/investigation-scenarios.js` as the fixture definition location but does not list individual scenario keys or env var values (by design — those are implementation details that can be inspected directly in the fixture file).
|
||||
- The deferred backlog contains specific wireframe text and animation specifications that may still be useful when UI work resumes. The header note ("not commitments, priorities or active tasks") should prevent premature actioning.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. Both new documents contain all original content. Branch `feature/user-workspace-ux-v0.7` is clean after commit. Engine and UI experiments remain paused.
|
||||
|
||||
## Experiment 32 — Separate Current Principles From Aspirational Architecture
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A short current-principles document can guide normal work while the original architectural-principles document remains available as the fuller historical and aspirational source. This should reduce ambiguity without deleting or rewriting the original reasoning.
|
||||
|
||||
### Source Documents Used
|
||||
|
||||
- `docs/current-project-state.md` — What the Confidence Engine Is; Current Engine Capabilities; Context Loading Guide
|
||||
- `.claude/architecture-guardrails.md` — entire file (77 lines)
|
||||
- `docs/document-role-review.md` — Architectural Principles Review (§2) and Recommended Actions (§4)
|
||||
- `docs/architectural-principles.md` — headings and the 14 principles only
|
||||
- `docs/03_Confidence_Engine_Language_Guide.md` — guiding principles only
|
||||
- `docs/current-implementation-verification.md` — Active Capabilities; Passive or Isolated Capabilities
|
||||
- Experiment 31 entry in `docs/design-evolution-log.md` (lines 1811–1893)
|
||||
|
||||
### Principles Included
|
||||
|
||||
**User Experience (5):** System carries complexity; steps are small enough to understand or investigate; engine guides without pretending certainty; first input is the hardest step; users may know answer/who to ask/where to look/how to test.
|
||||
|
||||
**Reasoning (5):** Resolved question ≠ established condition; evidence supports/contradicts/informs; present evidence does not settle future feasibility; uncertainty stated honestly; deterministic contracts separate from language interpretation.
|
||||
|
||||
**Building the System (6):** Build smallest thing that can be wrong; use evidence before architecture; every layer has one responsibility where applicable; presentation does not invent facts; current and aspirational labelled separately; load only needed context.
|
||||
|
||||
Total: 16 current principles, organized into three sections.
|
||||
|
||||
### Aspirational Material Deliberately Excluded
|
||||
|
||||
From `docs/architectural-principles.md`: P2 (Information Flows Downward — unresolved), P5 (Behaviour Never Reasons — aspirational), P7 (Assessment Never Generates Evidence — mixed), P9 (Assessment Describes Never Prescribes — mixed), P10 (Convergence Over Single Signals — aspirational), P11 (Assessment Is Stateful Across Turns — mixed/aspirational), P12 (Uncertainty About Assessment Is Itself Assessable — aspirational), P13 (Investigation Progress Is Qualitative Not Quantitative — mixed/aspirational). These remain in the original document for broader architectural review.
|
||||
|
||||
### Practical Principles-Test Result
|
||||
|
||||
**Task:** A developer proposes making every resolved question automatically increase confidence and close its related condition. Explain whether this fits current principles and why.
|
||||
|
||||
**Response from reduced context (current-project-state + current-working-principles + architecture-guardrails):**
|
||||
|
||||
1. **Resolving a question does not establish a condition.** current-working-principles §2 states: "A resolved question is not an established condition." Answer evidence must be inspected before any conclusion follows.
|
||||
2. **Answer evidence must be inspected.** current-working-principles §2 states direction alone (support/contradict/inform) is insufficient without checking subject, timeframe, and claim type alignment.
|
||||
3. **Confidence should not be manufactured.** architecture-guardrails invariants state "Confidence must not outrun evidence or completeness" and "Duplicate evidence must not increase confidence." current-project-state section 4 confirms: resolving a question does not automatically establish the condition.
|
||||
4. **Passive experimental logic is not automatically active behaviour.** current-project-state section 3 classifies passive classifiers (including decision-condition status evaluation) as diagnostic_only or isolated — they do not yet control the user-facing investigation.
|
||||
|
||||
**Was the three-document context sufficient?** Yes. All four points were answerable from `docs/current-working-principles.md` (principles §2), `.claude/architecture-guardrails.md` (reasoning invariants), and `docs/current-project-state.md` (section 3 passive classifier classification, section 4 what experiments established). No experiment history or source code was required.
|
||||
|
||||
### Unresolved Ambiguities
|
||||
|
||||
- The boundary between "current" and "aspirational" for P7 and P9 is inherently subjective; future sessions may interpret differently without the original document's reasoning context.
|
||||
- Some principles overlap with `.claude/architecture-guardrails.md` (e.g., "every layer has one responsibility" overlaps with guardrails' exhaustive prohibition list). No duplication was introduced deliberately, but a cross-reference could reduce redundancy in a future iteration.
|
||||
- The aspirational note points readers to the original document but does not provide a quick reference for which of the 14 principles are current versus aspirational. A summary table might be useful when architecture work resumes.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. No source code or tests changed. Engine and UI experiments remain paused. No files moved or deleted. Only documentation files were created or updated.
|
||||
|
||||
### Return-to-Work Note (80–150 words)
|
||||
|
||||
Current principles now live in `docs/current-working-principles.md`. This short document contains only guidance supported by verified implementation, current project direction, and established product philosophy — organised into three sections: user experience, reasoning, and building the system. Broader and aspirational architecture remains in `docs/architectural-principles.md` as a task-specific reference; it has not been rewritten or deleted. Future sessions should use `docs/current-working-principles.md` by default for product and reasoning work. Engine and UI experiments remain paused after Experiment 25B. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `docs/current-project-state.md`, then `docs/current-working-principles.md` for current guidance.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 33 — Create Task-Specific Context Packs
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A single concise context-pack guide can give each task type a minimal reading list, clear exclusions, and a stopping rule — reducing unnecessary context loading while preserving access to deeper material when a specific gap appears.
|
||||
|
||||
### Source Documents Used
|
||||
|
||||
- `docs/current-project-state.md` — Context Loading Guide; Current Engine Capabilities; Work Currently Paused
|
||||
- `docs/project-knowledge-inventory.md` — Current Working Context; Task-Specific References
|
||||
- `docs/current-implementation-verification.md` — Active Capabilities; Passive or Isolated Capabilities
|
||||
- `docs/current-working-principles.md` — entire file
|
||||
- `docs/ui-mock-reference.md` — headings and routing information only
|
||||
- `.claude/project-context.md` — routing notes only
|
||||
- `.claude/architecture-guardrails.md` — headings only
|
||||
- Experiment 32 entry in `docs/design-evolution-log.md` (lines 1895–1948)
|
||||
|
||||
### Deliverable
|
||||
|
||||
Created `docs/task-context-packs.md` (~110 lines) with four packs:
|
||||
- **Pack 1 — Engine Experiment Work:** current-project-state, current-working-principles, architecture-guardrails, current-implementation-verification.
|
||||
- **Pack 2 — UI and Mock Work:** current-project-state, current-working-principles, architecture-guardrails, ui-mock-reference.
|
||||
- **Pack 3 — Architecture or Contract Review:** current-project-state, current-implementation-verification, architecture-guardrails, current-working-principles + aspirational warning.
|
||||
- **Pack 4 — Knowledge-Management Work:** current-project-state, project-knowledge-inventory, task-context-packs, project-context.
|
||||
|
||||
Each pack lists what to always read, what to read only when relevant, and what to not load by default. Common rules prevent silent context inflation. Two routing tests verify sufficiency without loading history or source code.
|
||||
|
||||
### Routing Test A — Engine Task
|
||||
|
||||
**Task:** Verify whether Behaviour Selection currently affects the user-facing response.
|
||||
**Result:** Pack sufficient. `docs/current-implementation-verification.md` §3b states "Called by: None" for Behaviour Selection; `docs/current-project-state.md` §3 classifies it as isolated. No extra file required.
|
||||
|
||||
### Routing Test B — UI Task
|
||||
|
||||
**Task:** Choose the correct mock scenarios for testing a long investigation and contradictory evidence.
|
||||
**Result:** Pack sufficient. `docs/ui-mock-reference.md` lists "Long investigation (10–15 turns)" and "Contradiction" with matching purposes. Deferred UX backlog not needed.
|
||||
|
||||
### Validation
|
||||
|
||||
- All referenced files exist; no pack relies on fixed line numbers.
|
||||
- Each pack has a smaller default context than the full project documentation.
|
||||
- Active and passive capabilities remain clearly separated.
|
||||
- No source code or tests changed; no files moved or deleted.
|
||||
|
||||
### Return-to-Work Note
|
||||
|
||||
Task-specific context packs now exist in `docs/task-context-packs.md`, giving each work type a minimal four-document starting set plus targeted reading paths. Future sessions should start with `docs/current-project-state.md`, then choose one pack from `docs/task-context-packs.md`. Additional documents should be loaded only for a named gap, with the reason recorded. Engine and UI experiments remain paused. Branch: `feature/user-workspace-ux-v0.7`. First file to inspect when resuming: `docs/current-project-state.md`, then select the relevant pack from `docs/task-context-packs.md`.
|
||||
|
||||
---
|
||||
## Experiment 34 — Single Return-to-Work Handoff
|
||||
|
||||
**Date:** 2026-08-06
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A single short handoff file can carry enough immediate context to resume work accurately while linking to deeper documents only when needed.
|
||||
|
||||
### Handoff Structure
|
||||
|
||||
Eight sections: Where We Left It, What Is True Now, Why Work Is Paused, What Was Just Completed, What Remains Open, How to Resume, First Files by Work Type (table), Resume Check (five questions). Plus a maintenance rule replacing current-work sections when the project moves on.
|
||||
|
||||
### Document Length
|
||||
|
||||
`docs/current-handoff.md`: 68 lines (target range: 60–100).
|
||||
|
||||
### Practical Resume-Test Result
|
||||
|
||||
**Task:** Return after two weeks, remember almost nothing. Explain where the project stands, what is paused, what was completed most recently, and what to read before an engine task — using only `docs/current-handoff.md` and `docs/task-context-packs.md`.
|
||||
|
||||
| Check | Result |
|
||||
|---|---|
|
||||
| Identifies correct active phase (knowledge management) | Yes |
|
||||
| Identifies paused engine and UI work | Yes |
|
||||
| Identifies Experiment 33 as latest completed | Yes |
|
||||
| Chooses Engine Experiment pack for engine task | Yes |
|
||||
| Avoids opening full design history | Yes |
|
||||
| Does not confuse passive code with active behaviour | Yes |
|
||||
|
||||
**Verdict:** Pass. The handoff alone is sufficient to resume accurately.
|
||||
|
||||
### Missing Information
|
||||
|
||||
- "When knowledge-management work is complete enough to resume engine experiments" — no objective criterion exists yet; this is a judgment call for Rob.
|
||||
- "Whether tasks crossing pack boundaries can still stay concise" — unanswered in principle; requires testing with actual cross-boundary tasks.
|
||||
- Whether `docs/current-handoff.md` remains useful after several more knowledge-management experiments add to it.
|
||||
|
||||
### Can This Replace Scattered Current Return Notes?
|
||||
|
||||
Yes, for immediate resumption context. The handoff carries the latest stopping point without accumulating old notes. Historical return notes remain in `docs/current-project-state.md` and `docs/design-evolution-log.md` as evidence, not as current guidance. Rob should decide whether to purge older return notes once confident in the handoff model.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review.
|
||||
|
||||
+451
@@ -0,0 +1,451 @@
|
||||
## Experiment 35 — Test Current Handoff Maintenance (2026-08-06)
|
||||
|
||||
**Hypothesis:** A current handoff can remain useful if it describes only the latest stopping point, replaces stale details rather than appending history, and identifies the latest confirmed experiment and commit unambiguously.
|
||||
|
||||
**Stale or ambiguous wording found:**
|
||||
- Section 1 named Experiment 33 and commit `b959cfa` as the current state — now stale after Experiments 34+35;
|
||||
- Section 4 described only Experiment 33's completion, giving no indication that a single handoff had been created in Experiment 34;
|
||||
- No explicit mention of commit `1d92aa0` anywhere in the handoff;
|
||||
- Footer said "Created by Experiment 34" without acknowledging this maintenance experiment.
|
||||
|
||||
**Corrections made:**
|
||||
- Section 1: updated to name Experiment 34 and commit `1d92aa0`; added the maintenance principle ("replace stale details rather than appending history");
|
||||
- Section 4: rewritten to describe Experiment 34's consolidation work;
|
||||
- Section 5: retained one genuinely open question about handoff longevity; added provisional KM completion criteria sub-section (7 criteria, marked provisional);
|
||||
- Footer: updated to reference Experiment 35; added Return-to-Work Note recording all current state.
|
||||
|
||||
**Fresh-return test result:** PASS — from `current-handoff.md` and `task-context-packs.md` only, a fresh session can determine:
|
||||
- Latest completed KM experiment: Experiment 34 ✓
|
||||
- Latest commit: `1d92aa0` ✓
|
||||
- Knowledge-management active, engine/UI paused ✓
|
||||
- Knowledge-Management context pack is the correct routing target ✓
|
||||
- No need to open full design history ✓
|
||||
- Older commits not mistaken for current stopping point ✓
|
||||
|
||||
**Provisional completion criteria added:** Seven criteria recorded in Section 5 (see above). Not yet declared complete — pending Rob's review.
|
||||
|
||||
**Handoff remained concise?** Yes. 86 lines (was 68). Increase justified by the maintenance principle paragraph, updated current-state wording, and provisional completion criteria section. No historical timeline appended.
|
||||
|
||||
**Status:** Pending Rob's review.
|
||||
|
||||
## Experiment 36 — Validate Reduced Context Routing
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The documentation system (handoff + project-state + task-context-packs) is complete enough to support normal work without silently expanding into historical documentation. A fresh session can complete representative tasks using only routing instructions.
|
||||
|
||||
### Initial Documents Loaded (328 lines total)
|
||||
|
||||
1. `docs/current-handoff.md` — 86 lines
|
||||
2. `docs/current-project-state.md` — 132 lines
|
||||
3. `docs/task-context-packs.md` — 110 lines
|
||||
|
||||
### Additional Documents Loaded
|
||||
|
||||
| Document | Lines | Why Needed | Routing Should Include? |
|
||||
|---|---|---|---|
|
||||
| `docs/ui-mock-reference.md` | 63 | Task 2: verify mock scenarios for "long investigation" and "contradiction". Routing Test B claimed these were identifiable without loading it, but the specific scenario names do not appear in any initial document. | YES — routing defect found |
|
||||
| `docs/project-knowledge-inventory.md` | 215 | Task 4: confirm Engine Experiment pack's four always-read documents actually exist and understand KM phase outputs. | Debated — validated completeness but not strictly required by routing |
|
||||
| `docs/current-implementation-verification.md` | 111 | Cross-checked Behaviour Selection isolation against current-project-state §3. Provided corroboration but was not the sole basis for Task 1 answer. | Debated — useful corroboration; current-project-state alone sufficed |
|
||||
|
||||
### Tasks Completed Without Context Expansion
|
||||
|
||||
**Task 1 — Does Behaviour Selection affect engine behaviour?**
|
||||
No. Current project state §3 classifies it as isolated. Handoff §2 confirms passive classifiers don't control the investigation. Task-context-packs Routing Test A corroborates (current-implementation-verification §3b).
|
||||
|
||||
**Task 3 — Why passive classifiers are not yet in the active reasoning loop?**
|
||||
Passive classifiers record diagnostic signals for future use but have no integration into the turn cycle. Only investigation-state assessment is called (at 3 orchestrator sites), and its result goes into a diagnostics field — never checked by conditional branches. Others have zero callers.
|
||||
|
||||
### Tasks Requiring Extra Context
|
||||
|
||||
**Task 2 — Mock scenarios for long investigation and contradictory evidence**
|
||||
Required `docs/ui-mock-reference.md`. Routing Test B in task-context-packs claimed these were identifiable without loading it, but the specific scenario names ("Long investigation (10–15 turns)" and "Contradiction") do not appear in any initial document. The routing claim was unverifiable until the mock reference was loaded — this is a genuine routing defect.
|
||||
|
||||
**Task 4 — Where should a new developer begin for the next engine experiment?**
|
||||
Partially answered from initial documents (handoff → project-state → pack). Marginal need to verify that all four always-read pack documents actually exist, resolved by cross-referencing project-knowledge-inventory.
|
||||
|
||||
### Routing Failures Found
|
||||
|
||||
One genuine failure: **Routing Test B in task-context-packs.md**. The test states that mock scenarios for long investigation and contradiction are identifiable without loading ui-mock-reference.md. This was presented as a self-evident fact but the specific scenario names only exist in ui-mock-reference.md. The routing is incomplete — it should have included the mock reference file, or at minimum acknowledged that scenario names require verification.
|
||||
|
||||
### Documentation Changes Made
|
||||
|
||||
- Created `docs/context-routing-validation.md` (62 lines) — this experiment's record
|
||||
- Updated `docs/design-evolution-log.md` — appended Experiment 36 entry
|
||||
|
||||
**No source code or tests changed. No archive changes.**
|
||||
|
||||
### Overall Assessment: Mostly ready
|
||||
|
||||
Two of four tasks completed from initial context only. One routing defect found (Task 2; corrected by Experiment 37). After fixing Routing Test B to name ui-mock-reference.md as the scenario source, the reduced context system is ready for normal work.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 37 — Validate Cross-Boundary Context Routing
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The context-pack system can support cross-boundary work if Claude:
|
||||
1. starts with one primary pack;
|
||||
2. adds a second pack only for a named boundary;
|
||||
3. records why each extra document was loaded;
|
||||
4. avoids loading the full history.
|
||||
|
||||
### Initial Documents Loaded (328 lines total)
|
||||
|
||||
1. `docs/current-handoff.md` — 85 lines; first return-to-work entry point
|
||||
2. `docs/current-project-state.md` — 131 lines; active state and capabilities
|
||||
3. `docs/task-context-packs.md` — 110 lines; routing for four work types
|
||||
|
||||
### Additional Documents Loaded
|
||||
|
||||
| Document | Lines | Why Needed | Routing Should Include? |
|
||||
|---|---|---|---|
|
||||
| `docs/ui-mock-reference.md` | 62 | Cross-boundary boundary: the task requires identifying a mock scenario for workspace display. This is the second pack (UI and Mock) needed because no other loaded document names scenarios or UI fixtures. Yes — it is part of the UI/Mock pack, not an ad-hoc addition. |
|
||||
|
||||
### Cross-Boundary Task Result
|
||||
|
||||
**Task:** Display passive condition-status information in the workspace for a mock investigation without changing the active reasoning loop.
|
||||
|
||||
| Finding | Details |
|
||||
|---|---|
|
||||
| Condition-status capability | Passive: decision-condition status evaluation records signals but has no integration into the turn cycle; never controls user-facing decisions or path selection |
|
||||
| Active reasoning loop | Unchanged: deterministic pipeline (scenario reconstruction → graph update → unknown selection → question formulation → turn orchestration); none of these pathways are affected by passive data |
|
||||
| Mock scenario | "Long investigation (10–15 turns)" from `ui-mock-reference.md`; workspace can display accumulated diagnostic signals over time without interrupting the active reasoning cycle |
|
||||
| Implementation areas to inspect later | decision-condition-status evaluation module; evidence scope detection module; UI workspace components for passive display integration |
|
||||
| Both packs genuinely needed? | Yes: Engine pack identifies which capabilities are active vs passive; UI pack identifies how the workspace presents state. Neither alone suffices |
|
||||
| Archive or full history required? | No |
|
||||
|
||||
**Context remained manageable:** Yes. 390 lines total (328 initial + 62 additional). Each document loaded for a specific named purpose. No blind expansion.
|
||||
|
||||
### Knowledge-Management Completion Criteria Review
|
||||
|
||||
| Criterion | Status |
|
||||
|---|---|
|
||||
| 1. Fresh session can resume from handoff + one pack | met |
|
||||
| 2. Current state verified against implementation | met |
|
||||
| 3. Historical material outside default loading | met |
|
||||
| 4. Current principles separated from aspirational architecture | met |
|
||||
| 5. Task-specific routing works for engine and UI tasks | met |
|
||||
| 6. Cross-boundary task tested | **met** |
|
||||
| 7. Maintaining handoff does not require reading full history | met |
|
||||
|
||||
All seven criteria are now met.
|
||||
|
||||
> Knowledge-management structure is ready for Rob's review before engine experiments resume.
|
||||
|
||||
### Routing Defects Discovered
|
||||
|
||||
None in this experiment. The correction to Routing Test B (naming `ui-mock-reference.md` as the scenario source) was applied before testing. No new defects found in the cross-boundary test.
|
||||
|
||||
### Overall Assessment: Ready
|
||||
|
||||
The context-pack system handled a genuine engine/UI cross-boundary task by combining two packs deliberately with full documentation of each loaded document and its purpose. Context remained small (390 lines). All knowledge-management criteria are met.
|
||||
|
||||
---
|
||||
|
||||
# Experiment 38 — Cold-Start Project Recovery Validation
|
||||
|
||||
**Branch:** `feature/user-workspace-ux-v0.7`
|
||||
**Type:** Knowledge-management / handoff validation (final KM experiment)
|
||||
**Objective:** Test whether a genuinely cold session can recover the project accurately from the reduced context system alone without reading the full history or any earlier experiment reports.
|
||||
|
||||
## Setup
|
||||
|
||||
Cold-start configuration: no prior conversation context, no past experiment reports loaded, repository documentation carries all context. Session was freshly created to simulate a real return-to-work scenario. Only `docs/current-handoff.md` was read first (per handoff §6 step 1), then the two documents specified by its resume instructions (§6 steps 2–3): `docs/current-project-state.md` and `docs/task-context-packs.md`.
|
||||
|
||||
## Documents Loaded
|
||||
|
||||
| Document | Reason |
|
||||
|---|---|
|
||||
| `docs/current-handoff.md` | Primary entry point (handoff §6 step 1) |
|
||||
| `docs/current-project-state.md` | Resume instruction (§6 step 2) and routing table (§6 step 7) |
|
||||
| `docs/task-context-packs.md` | Pack selection (§6 step 3) and pack contents for verification |
|
||||
|
||||
No additional documents were loaded. No blind expansion occurred. The full design-evolution log, archived documents, UI mock reference, source code, and tests were all excluded by design.
|
||||
|
||||
## Project-State Recovery Result
|
||||
|
||||
The cold session correctly recovered:
|
||||
- What the Confidence Engine does (facilitated investigation with structured reasoning graph).
|
||||
- Active capabilities: deterministic reasoning pipeline, unknown selection via atomicity/answerability, question formulation, scenario API, turn cycle orchestration.
|
||||
- Passive capabilities: seven diagnostic layers from Experiments 18–25B, all isolated, none control user-facing investigation.
|
||||
- Paused work: engine experiments (after Exp 25B), UI experiments.
|
||||
- Why KM phase was undertaken (documentation bloat blocking session recovery).
|
||||
|
||||
Recovery score: complete from three documents alone. No source code inspection required.
|
||||
|
||||
## Context-Pack Selection Result
|
||||
|
||||
Pack 1 — Engine Experiment Work selected correctly by the cold session. The three initial documents contained sufficient information to identify the pack, its default documents, and what to exclude without reading any additional material.
|
||||
|
||||
## Handoff Defects Found
|
||||
|
||||
None found in `docs/current-handoff.md`. The handoff accurately describes the stopping point, identifies all seven KM criteria as met, provides correct resume instructions, and includes accurate capability boundaries. One structural update was made: the open item "whether the handoff stays accurate after further advances" was resolved as no longer applicable (the cold-start test confirmed it is accurate).
|
||||
|
||||
## Completion-Criteria Result
|
||||
|
||||
All seven knowledge-management completion criteria are confirmed met by this cold-start validation:
|
||||
1. Fresh session can resume from handoff + one pack — met (Exp 38 demonstrates this)
|
||||
2. Current state verified against implementation — met (Exp 28+)
|
||||
3. Historical material outside default loading — met
|
||||
4. Current principles separated from aspirational architecture — met
|
||||
5. Task-specific routing works for engine and UI tasks — met (Exp 37)
|
||||
6. Cross-boundary task tested — met (Exp 37)
|
||||
7. Maintaining handoff does not require reading full history — met
|
||||
|
||||
> The knowledge-management phase is complete enough for Rob to choose when engine experiments resume.
|
||||
|
||||
## Documents Updated
|
||||
|
||||
- `docs/cold-start-validation.md` — created (this experiment's deliverable)
|
||||
- `docs/current-handoff.md` — Exp 38 commit placeholder, structural open-item resolution, return-to-work note replacement
|
||||
- `docs/current-project-state.md` — KM status update ("active" → "complete"), latest known commit correction
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
|
||||
## Overall Assessment: Ready
|
||||
|
||||
The cold-start validation passed. A genuinely fresh session understood the project state, chose the correct context pack, verified the resume boundary, produced a valid engine-work resume brief, and found no handoff defects — all from three documents alone. No source code was read or changed. The reduced context system works for sessions that did not help create the documents.
|
||||
|
||||
Engine and UI experiments remain paused pending Rob's review.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 39 — Validate Behaviour Selection Against Real Assessment Outputs (2026-08-06)
|
||||
|
||||
**Branch:** feature/user-workspace-ux-v0.7
|
||||
|
||||
### Hypothesis
|
||||
The existing deterministic selector produces a useful rhythm across genuine assessment outputs without changing the active engine. If it repeatedly chooses one behaviour, chooses behaviours at the wrong time, or depends on signals the assessor does not actually produce, the experiment should expose that honestly.
|
||||
|
||||
### Scenarios Evaluated (from `tests/investigation-state-assessor.test.js` fixture set)
|
||||
1. **Long investigation** (3 turns: early → deepening → complete terminal)
|
||||
2. **Contradictory evidence** (3 turns: two conflicting consultants, 0→1→2 resolved unknowns)
|
||||
3. **Short early** (1 turn: two observations, first unknown, no resolution)
|
||||
|
||||
### Behaviour Distribution (7 turns total)
|
||||
- Acknowledge: 5 (71%)
|
||||
- Continue: 2 (29%)
|
||||
- Clarify: 0 (0%)
|
||||
- Summarise: 0 (0%)
|
||||
- Pause: 0 (0%)
|
||||
|
||||
### Behaviour Sequence by Scenario
|
||||
**Long investigation:** continue → acknowledge → acknowledge
|
||||
- Turn 0: phase=cannot_determine, progress=cannot_determine, health=too_narrow → continue (no rule matched)
|
||||
- Turn 3: phase=focusing, progress=steady, health=healthy → acknowledge
|
||||
- Turn 4: phase=concluding, progress=steady, health=healthy → acknowledge
|
||||
|
||||
**Contradictory evidence:** acknowledge → acknowledge → acknowledge
|
||||
- Turn 0: phase=focusing, progress=cannot_determine, health=healthy → acknowledge
|
||||
- Turn 1: phase=focusing, progress=stalled, health=healthy → acknowledge
|
||||
- Turn 2: phase=focusing, progress=steady, health=healthy → acknowledge
|
||||
|
||||
**Short early:** continue
|
||||
- Turn 0: phase=exploring, progress=cannot_determine, health=healthy → continue
|
||||
|
||||
### Sensible Selections (7 of 7)
|
||||
All selections were classified as sensible per the selection's stated conditions. Acknowledge fires because `health=healthy AND phase confidence≠low` across most states. Continue fires when no specific rule matches (early/cannot_determine/exploring phases).
|
||||
|
||||
### Questionable or Inappropriate Selections
|
||||
**One notable pattern:** Summarise and Pause never fire, even in a concluding terminal state. This is not because the assessor fails to detect "concluding" — it does. It is because Acknowledge (priority 1) fires first when health=healthy, blocking Summarise (priority 3) from ever reaching its turn. This is an **acknowledgement/summarise priority conflict**: acknowledging a conclusion ("you've figured this out!") is not wrong, but "give me a summary" is more useful at terminal states. The current rule ordering does not distinguish "early healthy" from "concluding healthy."
|
||||
|
||||
Clarify never fires because no test scenario produces `health=too_broad` — the assessor's "too_broad" trigger (activeUnknownCount > 3 AND resolved < 2) requires more nodes than any scenario in the fixture set has at that stage.
|
||||
|
||||
Pause never fires because `health=user_overloaded` is never reached, and while contradictory-turn-1 has phase=focusing + progress=stalled, Acknowledge still blocks it.
|
||||
|
||||
### Contract Alignment
|
||||
Assessor → Selector contract aligns cleanly. The assessor produces all three dimensions (phase, progress, conversationHealth) with the fields the selector expects. No transformation needed between pipeline stages.
|
||||
|
||||
### Whether Selector Appears Useful Enough for Another Passive Experiment
|
||||
The existing selector works but its **behaviour variation is severely constrained by Acknowledge's priority position**. A next passive experiment should test whether reordering or refining the acknowledge condition (e.g., excluding concluding/terminal phases) produces more context-appropriate behaviour — without changing the assessor.
|
||||
|
||||
### Status
|
||||
Pending Rob's review. Five behaviours are too narrow for this to be definitive, and only three scenarios were tested. The dominant pattern (acknowledge in healthy states) may change with different investigation domains.
|
||||
|
||||
### Documents Updated
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
- `docs/current-handoff.md` — return-to-work note replaced
|
||||
|
||||
---
|
||||
|
||||
## Experiment 40 — Audit Behaviour Reachability and Blocking (2026-08-06)
|
||||
|
||||
### Objective
|
||||
|
||||
Why did Clarify, Summarise, and Pause not appear during Experiment 39? Acknowledge: 5 (71%), Continue: 2 (29%), others: 0. This is a passive diagnostic — no rule changes, no engine modifications.
|
||||
|
||||
### Method
|
||||
|
||||
One test file (`tests/behaviour-selection.reachability.test.js`) containing:
|
||||
- Diagnostic audit helper that evaluates every behaviour rule against one assessment object
|
||||
- Real-scenario audits across the same Experiment 39 turns (8 turns total)
|
||||
- Synthetic reachability checks for each behaviour in isolation
|
||||
|
||||
### Findings
|
||||
|
||||
#### Summarise — eligible_but_blocked
|
||||
|
||||
Eligible in 2 of 7 real turns:
|
||||
- long-investigation turn 1 (resolvedNodeCount ≥ 3 + progress=steady triggers summarise rule)
|
||||
- long-investigation turn 2 (phase=concluding triggers summarise rule)
|
||||
|
||||
In both cases, health=healthy simultaneously, so Acknowledge (priority 1) fires first. Summarise rules are met but its output is never returned because the selector returns early on priority ordering.
|
||||
|
||||
**Root cause: priority conflict, not assessor failure.** The phase evidence correctly identifies concluding/synthesising states; the problem is that Acknowledge's broader trigger condition (health=healthy is the most common state) fires first.
|
||||
|
||||
#### Clarify — never_eligible_in_tested_scenarios (reachable only in synthetic case)
|
||||
|
||||
Not eligible in any of 7 real turns because neither trigger condition is met:
|
||||
- `health=too_broad`: requires activeUnknownCount > 3 AND resolvedNodeCount < 2 — no fixture reaches this state
|
||||
- `phase=orienting + observationDensity < 3`: current assessor never produces phase=orienting for tested scenarios
|
||||
|
||||
Synthetic case confirms the rule fires correctly in isolation (with low-confidence phase to avoid Acknowledge blocking).
|
||||
|
||||
**Root cause: assessor health classification logic produces too few `too_broad` cases. The trigger condition is extremely narrow — needs activeUnknownCount > 3 AND resolved < 2 simultaneously.**
|
||||
|
||||
#### Pause — eligible_but_blocked
|
||||
|
||||
Eligible in 1 of 7 real turns:
|
||||
- contradictory-evidence turn 1 (phase=focusing + progress=stalled triggers pause rule)
|
||||
|
||||
In this case, health=healthy simultaneously, so Acknowledge blocks it. The second pause trigger (`health=user_overloaded`) is never met because the assessor never produces that state.
|
||||
|
||||
**Root cause: same priority conflict as Summarise. One of two rules fires in real data but gets blocked by Acknowledge's earlier position.**
|
||||
|
||||
### Synthetic Reachability Confirmation
|
||||
|
||||
All five behaviours are independently reachable when isolated from Acknowledge:
|
||||
- ✅ acknowledge — healthy + confident phase
|
||||
- ✅ clarify — too_broad health (with low-confidence phase to avoid Acknowledge)
|
||||
- ✅ summarise — synthesising/concluding phase (without healthy health)
|
||||
- ✅ pause — focusing+stalled or user_overloaded (without healthy health)
|
||||
- ✅ continue — no rules match
|
||||
|
||||
### Classifications
|
||||
|
||||
| Behaviour | Classification | Primary Cause |
|
||||
|---|---|---|
|
||||
| Summarise | eligible_but_blocked | Acknowledge priority 1 fires first when health=healthy |
|
||||
| Clarify | never_eligible_in_tested_scenarios (reachable only in synthetic) | `too_broad` trigger too narrow for test scenarios; `orienting+low obs` not produced by assessor |
|
||||
| Pause | eligible_but_blocked | Acknowledge priority 1 fires first when health=healthy; `user_overloaded` never produced |
|
||||
|
||||
### Impact on Prior Finding (Exp 39)
|
||||
|
||||
Experiment 39 concluded "the Acknowledge→Summarise priority conflict prevents Summarise from firing." Experiment 40 confirms this and adds that **Pause faces the same blocking** (1 eligible turn, blocked). Clarify's absence is fundamentally different: its rules are not triggered at all in tested scenarios.
|
||||
|
||||
This means any fix must address two distinct problems:
|
||||
1. Priority conflict affecting Summarise AND Pause (same cause)
|
||||
2. Narrow trigger conditions for Clarify and the `user_overloaded` health state
|
||||
|
||||
### Test Results
|
||||
|
||||
- `tests/behaviour-selection.reachability.test.js`: 33 passed (new diagnostic file)
|
||||
- `tests/behaviour-selection.test.js`: 51 passed (no regressions)
|
||||
- `tests/behaviour-selection.real-assessment.test.js`: 16 passed (shared fixtures intact)
|
||||
- `tests/investigation-state-assessor.test.js`: 51 passed (assessor unchanged)
|
||||
|
||||
### Documents Updated
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
- `docs/current-handoff.md` — return-to-work note replaced
|
||||
|
||||
## Experiment 41 — Compare Acknowledge Priority Alternatives (2026-08-06)
|
||||
|
||||
### Purpose
|
||||
|
||||
Experiment 40 confirmed Summarise and Pause are eligible_but_blocked by Acknowledge's priority-1 position. Two passive alternatives were compared without modifying production code:
|
||||
|
||||
**Variant A** — Reorder rules so specific behaviours (Summarise, Pause) evaluate before Acknowledge. The idea is that if a more specific behaviour fires first, it captures the terminal/stalled states where Acknowledge should not fire.
|
||||
|
||||
**Variant B** — Keep existing priority order but exclude Acknowledge from firing when phase=concluding/synthesising, progress=stalled, or health=user_overloaded. The idea is to gate Acknowledge rather than reorder everything.
|
||||
|
||||
### Method
|
||||
|
||||
Both variants were implemented as test-only functions in `tests/behaviour-selection.counterfactual.test.js`. Each variant was evaluated against the same 7 real assessment turns from Experiments 39/40 across 3 scenarios. All five behaviours confirmed independently reachable synthetically. No production rules changed.
|
||||
|
||||
### Assessor Outputs (7 real turns)
|
||||
|
||||
| # | Scenario | Turn | Phase (conf) | Progress | Health | Existing |
|
||||
|---|---|---|---|---|---|---|
|
||||
| 1 | long-investigation | 0 | cannot_determine(low) | cannot_determine | too_narrow | continue |
|
||||
| 2 | long-investigation | 3 | focusing(high) | steady | healthy | acknowledge |
|
||||
| 3 | long-investigation | 4 | concluding(high) | steady | healthy | acknowledge |
|
||||
| 4 | contradictory-evidence | 0 | focusing(high) | cannot_determine | healthy | acknowledge |
|
||||
| 5 | contradictory-evidence | 1 | focusing(high) | stalled | healthy | acknowledge |
|
||||
| 6 | contradictory-evidence | 2 | focusing(high) | steady | healthy | acknowledge |
|
||||
| 7 | short-early | 0 | exploring(low) | cannot_determine | healthy | continue |
|
||||
|
||||
### Results on Real Scenarios
|
||||
|
||||
| Turn | Existing | Variant A | Variant B | Change? |
|
||||
|---|---|---|---|---|
|
||||
| long-investigation t3 | acknowledge | summarise | acknowledge | V-A: side-effect |
|
||||
| long-investigation t4 | acknowledge | **summarise** | **summarise** | **convergent ✓** |
|
||||
| contradictory-evidence t1 | acknowledge | **pause** | **pause** | **convergent ✓** |
|
||||
| All others | unchanged | unchanged | unchanged | — |
|
||||
|
||||
### Divergence Analysis
|
||||
|
||||
**Variant A diverges from Variant B at long-investigation turn 3.** Variant A produces `summarise` because its `resolvedNodeCount >= 3 && steady` rule fires at priority 1 without phase context. The assessor confirms this is a focusing-phase state (not synthesising/concluding) where the user needs acknowledgment, not compression. This is a false-positive for summarisation — a side-effect of Variant A's priority reordering.
|
||||
|
||||
**Variant B correctly preserves Acknowledge** at long-investigation t3 because:
|
||||
1. The exclusion list only includes `synthesising`, `concluding`, `stalled`, and `user_overloaded` — not focusing
|
||||
2. SummariseV2 itself has a phase gate (`phase.value === "synthesising"`) that prevents false-fire in focusing states
|
||||
3. Acknowledge at priority 1 wins because no exclusion applies
|
||||
|
||||
### Key Findings
|
||||
|
||||
1. **Both variants converge on the same two genuine changes:** `concluding → summarise` and `stalled → pause`. This was the experiment's primary question, and both approaches answer it correctly.
|
||||
|
||||
2. **Variant A introduces a false-positive:** The `resolvedNodeCount >= 3 && steady` rule fires in focusing-phase states without phase context, causing premature summarisation when Acknowledge would be more useful.
|
||||
|
||||
3. **Variant B has cleaner boundaries:** Explicit exclusion conditions prevent unwanted side-effects while preserving Acknowledge's role as the default healthy-state behaviour.
|
||||
|
||||
4. **Distribution shift (both variants):**
|
||||
- Existing: acknowledge 71%, continue 29%
|
||||
- Variant A: acknowledge 29%, summarise 29%, pause 14%, continue 29%
|
||||
- Variant B: acknowledge 43%, summarise 14%, pause 14%, continue 29%
|
||||
- Variant B preserves more Acknowledge because it doesn't remove the default healthy-state behaviour entirely
|
||||
|
||||
5. **Variant B is architecturally cleaner** for this problem space because it adds a targeted gate to one rule rather than reordering five priority levels — each of which would need individual review for side-effects.
|
||||
|
||||
### Test Results
|
||||
|
||||
- `tests/behaviour-selection.counterfactual.test.js`: 44 passed (new diagnostic file)
|
||||
- `tests/behaviour-selection.reachability.test.js`: 33 passed (no regressions)
|
||||
- `tests/behaviour-selection.real-assessment.test.js`: 16 passed (shared fixtures intact)
|
||||
- `tests/behaviour-selection.test.js`: 51 passed (no regressions)
|
||||
|
||||
### Decision Criteria
|
||||
|
||||
| Criterion | Variant A | Variant B |
|
||||
|---|---|---|
|
||||
| Fixes concluding state | ✓ summarise | ✓ summarise |
|
||||
| Fixes stalled state | ✓ pause | ✓ pause |
|
||||
| No false-positive changes | ✗ long-t3 → summarise | ✓ preserved acknowledge |
|
||||
| Implementation complexity | Simple reordering | Small gate function |
|
||||
| Maintains Acknowledge for healthy focus states | ? (depends on future review) | ✓ explicit preservation |
|
||||
|
||||
### Recommendation
|
||||
|
||||
**Variant B is preferred.** Both variants correctly identify the two genuine changes needed. Variant B has no false-positives, cleaner architectural boundaries (targeted exclusion vs priority reordering), and better preserves the existing Acknowledge default for healthy focusing states where it is appropriate. A recommended implementation would:
|
||||
|
||||
1. Keep existing priority order
|
||||
2. Add `isAcknowledgeExcluded()` function with conditions: phase∈{synthesising, concluding}, progress=stalled, health=user_overloaded
|
||||
3. Gate Acknowledge through this exclusion before selecting it at priority 1
|
||||
|
||||
### Documents Updated
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
- `docs/current-handoff.md` — return-to-work note replaced
|
||||
|
||||
|
||||
## Experiment 41 — Conclusion
|
||||
|
||||
**Variant B was preferred because it changed only the two intended turns without introducing a false-positive in a focusing state. Variant A produced an early summarise in a focusing phase and was discarded. No production rule changed during Experiment 41.** The implementation of Variant B's exclusion gate is the subject of Experiment 42.
|
||||
|
||||
---
|
||||
|
||||
@@ -0,0 +1,566 @@
|
||||
## Experiment 42 — Implement Narrow Acknowledge Exclusion (Variant B) (2026-08-06)
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Applying a narrow exclusion gate to Acknowledge — excluding it when phase is synthesising or concluding, progress is stalled, or conversation health is user_overloaded — will reduce the two identified false-Acknowledge selections (concluding → summarise, stalled → pause) without introducing any unintended behaviour changes in other tested turns.
|
||||
|
||||
### Exact Exclusion Rule
|
||||
|
||||
`isAcknowledgeExcluded(assessment)` returns `true` when:
|
||||
- `phase.value` is `synthesising` or `concluding`; OR
|
||||
- `progress.value` is `stalled`; OR
|
||||
- `conversationHealth.value` is `user_overloaded`.
|
||||
|
||||
When excluded, Acknowledge does not fire and the selector proceeds to the next priority rule. The gate qualifies the trigger; it does not replace it.
|
||||
|
||||
### Two Changed Turns
|
||||
|
||||
| Turn | Scenario | Phase | Progress | Health | Before | After |
|
||||
|------|----------|-------|----------|--------|--------|-------|
|
||||
| long-investigation t4 | concluding long-investigation | concluding(high) | steady | healthy | acknowledge | **summarise** |
|
||||
| contradictory-evidence t1 | stalled contradictory-evidence | focusing(high) | stalled | healthy | acknowledge | **pause** |
|
||||
|
||||
### Five Preserved Turns
|
||||
|
||||
| Turn | Scenario | Phase | Progress | Health | Behaviour (unchanged) |
|
||||
|------|----------|-------|----------|--------|----------------------|
|
||||
| long-investigation t0 | cannot_determine(low) | cannot_determine | too_narrow | continue |
|
||||
| long-investigation t3 | focusing(high) | steady | healthy | acknowledge |
|
||||
| contradictory-evidence t0 | focusing(high) | cannot_determine | healthy | acknowledge |
|
||||
| contradictory-evidence t2 | focusing(high) | steady | healthy | acknowledge |
|
||||
| short-early t0 | exploring(low) | cannot_determine | healthy | continue |
|
||||
|
||||
### Final Behaviour Distribution (7 real assessment turns)
|
||||
|
||||
- Acknowledge: 3
|
||||
- Summarise: 1
|
||||
- Pause: 1
|
||||
- Continue: 2
|
||||
- Clarify: 0
|
||||
|
||||
### Integration Status
|
||||
|
||||
The production Behaviour Selection module (`lib/behaviour-selection/behaviour-selector.js`) was changed to include the `isAcknowledgeExcluded()` gate. However, **active user-facing engine behaviour did not change** because Behaviour Selection remains isolated with no runtime caller — it is exported but never imported by any code in the repository.
|
||||
|
||||
### Clarify Status
|
||||
|
||||
`Clarify` remains unresolved and was not modified in this experiment. Its trigger conditions (`health=too_broad` or `phase=orienting + obs<3`) require states that no tested scenario produces. This remains an open question for future work.
|
||||
|
||||
### Selector Output Shape
|
||||
|
||||
The selector output shape did not change. The exclusion gate returns `null` from `selectAcknowledge`, which is the existing early-return mechanism used when a rule does not match. No new fields, no restructuring of the return object.
|
||||
|
||||
### Assessor and Fixtures
|
||||
|
||||
Assessor logic did not change. Fixtures did not change. Priority order did not change.
|
||||
|
||||
### Test Results
|
||||
|
||||
All 151 relevant tests passed across:
|
||||
- `tests/behaviour-selection.test.js`: 51 (no regressions)
|
||||
- `tests/behaviour-selection.reachability.test.js`: 33 (updated for new exclusion gate)
|
||||
- `tests/behaviour-selection.counterfactual.test.js`: 44 (from Exp 41, no changes)
|
||||
- `tests/behaviour-selection.real-assessment.test.js`: 16 (shared fixtures intact)
|
||||
|
||||
Tests were not rerun as part of this documentation-only closure. The recorded result comes from the implementation commit (05d3d96).
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only seven real assessment turns across three scenarios were evaluated; other investigation domains may exhibit different patterns.
|
||||
- `health=user_overloaded` is excluded by rule but never produced by any current assessor fixture — it is untested in practice.
|
||||
- `Clarify` remains deferred because no scenario produces the narrow trigger conditions it requires.
|
||||
- The selector remains isolated with no runtime caller; there is no live user-facing validation.
|
||||
|
||||
### Result
|
||||
|
||||
**Confirmed within the tested scenarios.** Variant B correctly changes only the two intended turns and preserves all five others. No unintended side-effects were observed.
|
||||
|
||||
### Documents Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
- `docs/current-handoff.md` — return-to-work note replaced
|
||||
|
||||
## Experiment 43 — Audit Clarify Readiness Signals (2026-08-06)
|
||||
|
||||
### Hypothesis
|
||||
|
||||
The existing investigation-state-assessor never produces states that trigger the production Clarify rule in any tested scenario. Clarify is absent from Behaviour Selection not because of a selector defect but because no current fixture represents the genuinely unclear-scoped investigations that its triggers are designed for.
|
||||
|
||||
### Diagnostic Test File
|
||||
|
||||
A focused diagnostic test was created at `tests/behaviour-selection.clarify-readiness.test.js` with 31 assertions auditing every turn across all existing assessor and reachability fixtures. It inspects:
|
||||
- Phase value distribution (focusing, exploring, concluding, synthesising, deepening, cannot_determine)
|
||||
- Conversation health values (healthy, too_narrow, too_broad, user_overloaded)
|
||||
- Observation density per turn
|
||||
- Clarify eligibility via the exact production rule in `selectClarify`
|
||||
|
||||
### Audit Scope
|
||||
|
||||
| Source | Scenarios | Turns Inspected |
|
||||
|--------|-----------|-----------------|
|
||||
| `investigation-state-assessor.test.js` | 7 | 7 (one per scenario) |
|
||||
| `behaviour-selection.reachability.test.js` | 3 | 3 (contradictory-evidence t0, t1, t2) |
|
||||
| **Total** | **10** | **10 real-turn assessments** |
|
||||
|
||||
### Q1 — Does the assessor ever produce `too_broad`?
|
||||
|
||||
**No.** Zero scenarios across all test fixtures produce `conversationHealth.value === "too_broad"`.
|
||||
|
||||
The `too_broad` trigger requires `activeUnknownCount > 3 AND resolvedNodeIds.length < 2`. Every existing scenario starts with exactly one active unknown (the single unresolved question the investigation is about), and the assessor never produces a state where more than three unrelated unknowns coexist without resolution.
|
||||
|
||||
### Q2 — Does the assessor ever produce `phase.value === "orienting"`?
|
||||
|
||||
**No.** Zero scenarios produce orienting. The five phase values produced by the assessor are: concluding, synthesising, focusing, exploring, deepening, and cannot_determine. **`orienting` is not a possible output of any assessor code path.** It does not appear in `assessPhase()`.
|
||||
|
||||
### Q3 — Does orienting ever coincide with observation density < 3?
|
||||
|
||||
**Never applicable.** Since the assessor never produces orienting, this condition cannot arise in real data. The orienting-based Clarify trigger is dead code within the tested scenarios (and likely in production until a scenario change introduces orienting).
|
||||
|
||||
### Q4 — How many turns are Clarify-eligible?
|
||||
|
||||
**Zero of 10 turns.** Both Clarify rules evaluate to false for every assessed turn:
|
||||
- Rule 1 (`too_broad` health): false in all 10 turns
|
||||
- Rule 2 (`orienting + obs<3`): false in all 10 turns (orienting never appears)
|
||||
|
||||
### Q5 — What are the closest existing signals to a genuine Clarify need?
|
||||
|
||||
Two signals approach clarification but do not match its intent:
|
||||
|
||||
| Signal | Turns | Meaning | Maps to Clarify? |
|
||||
|--------|-------|---------|-----------------|
|
||||
| `too_narrow` health | 1 (long-turn-0) | Insufficient contextual evidence for a narrow investigation | No — too_narrow means "needs more data," not "scope is unclear" |
|
||||
| `exploring` phase with low obs density | 1 (complete-turn-0) | Early-stage investigation with sparse observations | No — this signals the start of an investigation, not scope confusion |
|
||||
|
||||
### Q6 — Signal reliability assessment for future Clarify rule design
|
||||
|
||||
| Signal | Reliability for Clarify intent |
|
||||
|--------|-------------------------------|
|
||||
| `too_narrow` health | Low reliability. It reliably indicates insufficient context for question formulation but conflates "too little information" with "unclear scope." The assessor's own description: "The investigation needs more contextual evidence before the current question can be answered effectively." This is about quantity, not clarity. |
|
||||
| `exploring` + low obs density | Low reliability. It reliably indicates an early-stage investigation but does not distinguish between "well-scoped investigation in early phase" and "unclear investigation needing anchoring." Both map to exploring. |
|
||||
|
||||
### Q7 — Is Clarify's absence appropriate for current fixtures?
|
||||
|
||||
**Yes.** Every existing fixture represents a well-defined, focused investigation with a clear central statement:
|
||||
- "Comparing two products before purchase decision" (single question, single dimension)
|
||||
- "Evaluating European market entry" (single strategic question)
|
||||
- "Evaluating $2M procurement against conflicting expert advice" (single decision context)
|
||||
|
||||
A genuinely unclear-scoped investigation would need one of:
|
||||
- A central statement so vague the system cannot classify it into any phase
|
||||
- Multiple unrelated threads at startup with no clear priority anchor
|
||||
- Contradictory framing where the situation itself is ambiguous
|
||||
|
||||
No current fixture represents these states. **Clarify's absence is appropriate because the existing scenarios are genuinely well-scoped, not because the selector is broken.**
|
||||
|
||||
### Phase Distribution Across All 10 Turns
|
||||
|
||||
| Phase | Count | Scenarios |
|
||||
|-------|-------|-----------|
|
||||
| focusing | 7 | comparison t0,t1,t2; long t3; contradictory t0,t1,t2 |
|
||||
| cannot_determine | 1 | long t0 |
|
||||
| concluding | 1 | long t4 |
|
||||
| exploring | 1 | complete t0 |
|
||||
|
||||
No synthesising, deepening, or orienting phases observed.
|
||||
|
||||
### Production Clarify Trigger — Exact Rule Match
|
||||
|
||||
```js
|
||||
// selectClarify (behaviour-selector.js lines 65-83)
|
||||
function selectClarify(assessment) {
|
||||
// Rule A: broad scope detected
|
||||
if (assessment.conversationHealth.value === "too_broad") return clarify;
|
||||
// Rule B: early orientation with sparse data
|
||||
if (assessment.phase.value === "orienting" && assessment.phase.evidence?.observationDensity < 3) return clarify;
|
||||
return null;
|
||||
}
|
||||
```
|
||||
|
||||
**Rule A trigger:** `conversationHealth.value === "too_broad"` — zero occurrences in tested scenarios.
|
||||
**Rule B trigger:** `phase.value === "orienting"` — never produced by assessor; **dead code path.**
|
||||
|
||||
### Focused Test Results (Experiment 43)
|
||||
|
||||
- Total tests: **31**
|
||||
- Passed: **31**
|
||||
- Failed: **0**
|
||||
|
||||
All diagnostics confirm zero Clarify eligibility across the complete set of real-world fixtures.
|
||||
|
||||
### Regression / Validation Results
|
||||
|
||||
| Test File | Tests | Result | Notes |
|
||||
|-----------|-------|--------|-------|
|
||||
| `tests/behaviour-selection.clarify-readiness.test.js` | 31 | ✓ Pass | New diagnostic file — no regression possible |
|
||||
| `tests/behaviour-selection.test.js` | 51 | ✓ Pass | Zero regressions from any prior experiments |
|
||||
| `tests/behaviour-selection.reachability.test.js` | 33 | ✓ Pass | Clarify still eligible in 0 real turns; synthetically reachable |
|
||||
| `tests/investigation-state-assessor.test.js` | 51 | ✓ Pass | Assessor behavior unchanged |
|
||||
|
||||
### Limitations
|
||||
|
||||
- The audit covers all existing test fixtures but not every possible investigation domain. Different problem domains (legal disputes, medical triage, multi-party procurement) may produce different assessor states.
|
||||
- `too_broad` requires very specific conditions (>3 active unknowns with <2 resolved) that no current fixture exercises. A fixture designed specifically to trigger it would validate the health classifier path.
|
||||
- The orienting phase was never produced by any assessor code path in the entire test suite, suggesting a design gap: either orienting was removed from the assessor without updating the selector, or it was never implemented as an active phase value.
|
||||
|
||||
### Conclusion
|
||||
|
||||
**Clarify is absent from Behaviour Selection because no current scenario genuinely needs clarification — not because of a selector defect.** The two production rules are well-formed but their trigger conditions (too_broad health and orienting phase) represent states that the assessor either cannot produce (orienting) or does not produce in any tested fixture (too_broad).
|
||||
|
||||
**Two distinct issues identified:**
|
||||
1. **Dead code path**: The orienting-based Clarify rule never activates because the assessor produces six phase values but none is `orienting`. This is a design inconsistency worth correcting — either add orienting as a real phase or remove that rule from the selector.
|
||||
2. **Narrow trigger threshold**: The too_broad condition (`activeUnknownCount > 3 AND resolvedNodeIds < 2`) is validly narrow but never exercised by any fixture. If Clarify should fire earlier in investigations, the threshold should be relaxed; if it should only fire for genuinely lost investigations, it should stay as-is and a dedicated fixture should validate it.
|
||||
|
||||
### Documents Updated
|
||||
|
||||
- `docs/design-evolution-log.md` — this entry
|
||||
- `docs/current-handoff.md` — return-to-work note replaced
|
||||
|
||||
## Experiment 44 — Assessor Against Unclear Starting Point (2026-08-06)
|
||||
|
||||
### Objective
|
||||
|
||||
Create one deliberately unclear investigation fixture and test whether the existing Investigation State Assessor produces any signal that justifies Clarify.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
A deliberately unclear starting scenario may expose one of three outcomes:
|
||||
1. The assessor already produces `too_broad`.
|
||||
2. The assessor produces another existing signal that reasonably represents the need to clarify.
|
||||
3. The assessor has no suitable signal for unclear framing.
|
||||
|
||||
### Fixture Description
|
||||
|
||||
**File:** `tests/investigation-state-assessor.unclear-start.test.js` (test-only, not imported anywhere else)
|
||||
|
||||
The fixture represents:
|
||||
- A vague central statement that admits uncertainty: *"The business feels stuck. Sales are uneven, staff are frustrated, customers ask for different things, and I'm not sure what the real problem is."*
|
||||
- Five competing unknown threads (customer demand, staff capacity, product direction, pricing, operations) with no priority anchor
|
||||
- Only one observation (the only concrete data point)
|
||||
- Zero resolved evidence nodes
|
||||
- No selected question (no established direction)
|
||||
- All existing graph fields only (id, label, description, kind, status, confidence, evidenceIds, dependsOn, affects, childIds)
|
||||
- Five `kind: "unknown"` nodes and one `kind: "observation"` node
|
||||
|
||||
### Returned Assessment Signals
|
||||
|
||||
| Signal | Value | Confidence |
|
||||
|--------|-------|------------|
|
||||
| Phase | `cannot_determine` | low |
|
||||
| Phase signals | "Insufficient data for phase classification" | — |
|
||||
| Progress | `cannot_determine` | low |
|
||||
| Progress signals | "Insufficient data for progress assessment" | — |
|
||||
| Conversation health | **`too_broad`** | medium |
|
||||
| Health signals | "5 active unknowns with fewer than 2 resolved items"; "Investigation may be spreading too thin" | — |
|
||||
|
||||
Detailed evidence:
|
||||
- Phase evidence: resolvedNodeCount=0, activeUnknownCount=5, observationDensity=1, evidenceDepth="shallow"
|
||||
- Progress evidence: turnCount=0, recentResolutionsLastTurn=0
|
||||
- Health evidence: activeUnknownCount=5, resolvedNodeRatio=null, hasActiveQuestion=false
|
||||
|
||||
### Clarify Eligibility
|
||||
|
||||
**Clarify became eligible via Rule A.** The production `selectClarify` rule fires because `conversationHealth.value === "too_broad"`.
|
||||
|
||||
The production selector (`selectBehaviour`) returned:
|
||||
- behaviour: `"clarify"`
|
||||
- confidence: `"high"`
|
||||
- reason: "Conversation health is too broad — investigation may be spreading too thin. Narrow focus through a specific clarification question."
|
||||
|
||||
### Interpretation
|
||||
|
||||
**Classification: `assessor_recognises_unclear_start`**
|
||||
|
||||
The assessor produced `too_broad` from the unclear-start fixture, which directly maps to Clarify's intent (genuinely unclear scope requiring anchoring). The signal honestly reflects the starting situation: five competing unknowns with no resolved evidence and no established direction.
|
||||
|
||||
### What the Assessor Recognised
|
||||
|
||||
1. Multiple active unknowns without sufficient resolution triggered `too_broad` health classification.
|
||||
2. The assessor correctly recorded 5 active unknowns in both phase and health evidence sections.
|
||||
3. Observation density (1) was correctly reported as shallow.
|
||||
4. Phase confidence remained low due to insufficient data for any meaningful classification.
|
||||
|
||||
### What the Assessor Failed to Recognise
|
||||
|
||||
1. **`orienting` phase**: Still not produced by the assessor. The orienting-based Clarify rule remains dead code, unchanged from Experiment 43's finding.
|
||||
2. **Early-stage clarification need**: The `too_broad` trigger only fires after >3 unknowns accumulate — it does not catch a situation with fewer competing threads that is still genuinely unclear in framing.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Only one fixture was tested. Different vague-scenario configurations may produce different results.
|
||||
- The `too_broad` trigger depends on having more than 3 active unknowns with fewer than 2 resolved — this specific threshold was exercised, but other boundary conditions (e.g., exactly 4 unknowns, or 5 unknowns with 1 resolved) were not tested.
|
||||
- The fixture uses the assessor's existing `too_broad` definition which conflates "many unknowns" with "unclear scope." A genuinely unclear scenario with only 2–3 competing threads may not trigger this signal.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. Experiment 43 remains closed — its conclusion that a deliberately unclear fixture was required is confirmed by this experiment, which successfully exercises the previously untested `too_broad` health path.
|
||||
|
||||
### Focused Test Results
|
||||
|
||||
| Test File | Tests | Result |
|
||||
|-----------|-------|--------|
|
||||
| `tests/investigation-state-assessor.unclear-start.test.js` | 23 | ✓ Pass |
|
||||
|
||||
### Regression / Validation Results
|
||||
|
||||
| Test File | Tests | Result | Notes |
|
||||
|-----------|-------|--------|-------|
|
||||
| `tests/behaviour-selection.clarify-readiness.test.js` | 31 | ✓ Pass | Zero regressions |
|
||||
| `tests/investigation-state-assessor.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
| `tests/behaviour-selection.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
|
||||
### Production Assessor Status
|
||||
|
||||
**Unchanged.** The assessor produced the expected `too_broad` signal from the unclear fixture, confirming the health classifier path works correctly. No code was modified.
|
||||
|
||||
### Closure
|
||||
|
||||
Experiment 44 is **closed**. Conclusion: the assessor recognises an extreme unclear start; too_broad and Clarify are reachable; the useful boundary remained unknown.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 45 — Where Does "Too Broad" Begin? (2026-08-06)
|
||||
|
||||
### Objective
|
||||
|
||||
Test how the existing assessor's `too_broad` threshold behaves as an unclear starting scenario grows from two competing unknowns to five, all with identical base inputs. Passive boundary experiment only — no production code changes.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
| Active unknowns | Expected health |
|
||||
|---:|---|
|
||||
| 2 | not `too_broad` |
|
||||
| 3 | not `too_broad` |
|
||||
| 4 | `too_broad` |
|
||||
| 5 | `too_broad` |
|
||||
|
||||
### Fixture-Control Method
|
||||
|
||||
One test-only fixture builder creates the same vague starting situation varying only the number of competing unknowns:
|
||||
- Same central statement; same single observation; zero resolved items (base); no selected question; no active direction; same node shapes and confidence values.
|
||||
- Only the count of `kind: "unknown"` nodes differs.
|
||||
|
||||
### Results: Two Through Five Active Unknowns
|
||||
|
||||
| Active unknowns | Health | Confidence | Phase | Progress | Clarify eligible | Selector |
|
||||
|---:|---|---|---|---|---:|---|
|
||||
| 2 | `cannot_determine` | low | `cannot_determine` (low) | `cannot_determine` (low) | No | continue (low) |
|
||||
| 3 | `cannot_determine` | low | `cannot_determine` (low) | `cannot_determine` (low) | No | continue (low) |
|
||||
| 4 | `too_broad` | medium | `cannot_determine` (low) | `cannot_determine` (low) | Yes | clarify (high) |
|
||||
| 5 | `too_broad` | medium | `cannot_determine` (low) | `cannot_determined` (low) | Yes | clarify (high) |
|
||||
|
||||
### Results: Four Unknowns + Resolved Items
|
||||
|
||||
| Active unknowns | Resolved | Health | Confidence | Clarify eligible |
|
||||
|---:|---:|---|---|---:|
|
||||
| 4 | 0 | `too_broad` | medium | Yes |
|
||||
| 4 | 1 | `too_broad` | medium | Yes |
|
||||
| 4 | 2 | `cannot_determine` | low | No |
|
||||
|
||||
### Human-Sense Review
|
||||
|
||||
- **Two competing threads:** Still appears ambiguous rather than clearly manageable. The assessor returns `cannot_determine`, not `healthy`. This is honest — two unknowns with one observation and no question genuinely leave the state unclear.
|
||||
- **Three competing threads:** Appears ambiguous or already confused. The assessor still returns `cannot_determine`. This feels correct — three competing threads with minimal context is genuinely uncertain, not healthy.
|
||||
- **Four competing threads:** Appears genuinely too broad. The transition from three (uncertain) to four (too_broad) feels believable — a real investigator would start losing focus at this point.
|
||||
- **Five competing threads:** Clearly justifies clarification. Matches Experiment 44's result; no surprise.
|
||||
- **Transition between three and four:** Understandable. Three threads with one observation is "not enough to decide"; four adds the tipping point where the spread becomes problematic.
|
||||
- **Confidence language:** `too_broad` confidence is `medium` for both four and five unknowns. The signals are specific ("4 active unknowns with fewer than 2 resolved items"), so medium confidence is honest — it does not overstate certainty.
|
||||
|
||||
### Boundary Classification
|
||||
|
||||
| Transition | Classification | Rationale |
|
||||
|---|---|---|
|
||||
| 2→3 | `believable` | Both remain `cannot_determine`; the gap between "manageable" and "confused" genuinely sits around here |
|
||||
| 3→4 | `believable` | Four competing threads with no resolution is a believable tipping point for losing focus |
|
||||
| Resolution threshold (<2 resolved) | `believable` | The binary boundary (1 stays too_broad, 2 clears it) aligns with the design intent of "sufficient context to narrow" |
|
||||
|
||||
### Usefulness of Active-Unknown Count as a Proxy
|
||||
|
||||
Active-unknown count acts as a **useful but coarse** proxy for scope confusion. It works because:
|
||||
1. In the tested scenarios, more unknowns directly correlates with genuine ambiguity.
|
||||
2. The resolved-item gate prevents premature too_broad flags on investigations making progress.
|
||||
3. It avoids subjective measurement of "how confused is the user."
|
||||
|
||||
However, it cannot distinguish between:
|
||||
- Four unknowns about one decision (genuinely broad) versus four unknowns across a multi-decision comparison (expected).
|
||||
- A well-formed investigation with natural branching versus an unfocused investigation losing its way.
|
||||
|
||||
### Questionable or Unsupported Findings
|
||||
|
||||
1. **Health defaults to `cannot_determine` rather than `healthy` for 2–3 unknowns.** This is mechanically correct (no active question means the "healthy" rule doesn't fire) but arguably should produce `healthy` when the state is simply an early-stage investigation with a few threads, not just insufficient data.
|
||||
2. **The experiment uses synthetic boundary fixtures.** These cannot validate whether a real user would feel the same confusion at exactly these thresholds. The boundary may be mechanically correct but conceptually misaligned in some domains.
|
||||
3. **All unknowns share identical labels and confidence values.** A more differentiated scenario (some high-confidence, some low) might behave differently.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Current boundary is mechanically clear but conceptually uncertain.**
|
||||
|
||||
The threshold sits exactly between three and four active unknowns. This mechanical boundary behaves predictably: no too_broad below it, too_broad above it, resolved items gate correctly. However, whether this aligns with genuine user confusion (not just code behaviour) cannot be determined from synthetic fixtures alone. The experiment confirms that Clarify switches on at the same boundary as too_broad, and that resolving two items does switch too_broad off.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Synthetic fixture only; no real-user validation possible from this experiment.
|
||||
- All unknowns have identical shapes and confidence — real scenarios mix high/low confidence differently.
|
||||
- Only one central statement used; different domains may require different thresholds.
|
||||
- Does not test whether the `cannot_determine` health for 2–3 unknowns is a bug or a feature.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. No production behaviour changed. The next logical step would be: (a) validate whether `cannot_determine` health for 2–3 unknowns should instead be `healthy`, or (b) test real-user scenarios to confirm the three→four boundary feels right in practice.
|
||||
|
||||
### Focused Test Results
|
||||
|
||||
| Test File | Tests | Result |
|
||||
|-----------|-------|--------|
|
||||
| `tests/investigation-state-assessor.too-broad-boundary.test.js` | 32 | ✓ Pass |
|
||||
|
||||
### Regression / Validation Results
|
||||
|
||||
| Test File | Tests | Result | Notes |
|
||||
|-----------|-------|--------|-------|
|
||||
| `tests/investigation-state-assessor.unclear-start.test.js` | 23 | ✓ Pass | Zero regressions |
|
||||
| `tests/behaviour-selection.clarify-readiness.test.js` | 31 | ✓ Pass | Zero regressions |
|
||||
| `tests/investigation-state-assessor.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
| `tests/behaviour-selection.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
|
||||
### Production Assessor Status
|
||||
|
||||
**Unchanged.** No code was modified. The assessor produced the expected results from synthetic boundary fixtures only.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 45 — Closure
|
||||
|
||||
The threshold is mechanically clear; active-unknown count is a coarse proxy; semantic coherence remained untested.
|
||||
|
||||
---
|
||||
|
||||
## Experiment 46 — Does "Too Broad" Mean Too Many Questions, or Too Many Unrelated Questions? (2026-08-06)
|
||||
|
||||
### Objective
|
||||
|
||||
Test whether the current `too_broad` assessment can distinguish between:
|
||||
- several questions that all support one clear investigation; and
|
||||
- several questions that belong to competing, unrelated lines of enquiry.
|
||||
|
||||
This is a passive diagnostic experiment. No production code changes.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
Two fixtures with the same number of active unknowns may receive the same `too_broad` result even when one is coherent and the other is genuinely scattered. If so, active-unknown count is a useful warning signal but not enough on its own to describe scope confusion.
|
||||
|
||||
### Context Pack Used
|
||||
|
||||
Engine Experiment Work pack (Pack 1). Documents loaded:
|
||||
- `docs/current-project-state.md`, `docs/current-working-principles.md`, `.claude/architecture-guardrails.md`, `docs/current-implementation-verification.md`
|
||||
- `lib/assessment/investigation-state-assessor.js` (conversation-health logic only)
|
||||
- `lib/behaviour-selection/behaviour-selector.js` (Clarify rule only)
|
||||
- `tests/investigation-state-assessor.too-broad-boundary.test.js`
|
||||
- `tests/investigation-state-assessor.unclear-start.test.js`
|
||||
- Experiment 45 section in `docs/design-evolution-log.md`
|
||||
|
||||
No additional documents loaded.
|
||||
|
||||
### Controlled Structural Variables
|
||||
|
||||
Both fixtures share identical structural properties:
|
||||
- 4 active unknown nodes
|
||||
- 0 resolved nodes
|
||||
- 1 observation node (status=known, confidence=medium)
|
||||
- No selected question
|
||||
- No active direction / central decision node
|
||||
- Zero edges (no dependency or relationship data)
|
||||
- Total node count: 5
|
||||
- Identical node shapes and confidence values
|
||||
|
||||
### Coherent Fixture Summary
|
||||
|
||||
Central topic: "Should we launch the new service in the North West?"
|
||||
|
||||
Four unknowns all contributing to one decision:
|
||||
1. Whether customer demand exists in the North West region
|
||||
2. What price point the North West market would accept
|
||||
3. Whether delivery infrastructure can support the North West region
|
||||
4. Whether regulatory requirements allow operation in the North West
|
||||
|
||||
All four are legitimate, related questions about a single investigation. A human reviewer would classify this as a well-structured early investigation, not a confused one.
|
||||
|
||||
### Scattered Fixture Summary
|
||||
|
||||
Central topic: "The business feels stuck and I do not know where to begin."
|
||||
|
||||
Four unknowns from competing, unrelated threads:
|
||||
1. Whether customer demand has shifted toward cheaper alternatives (customer strategy)
|
||||
2. Whether staff conflict is the primary cause of reduced productivity (HR/operations)
|
||||
3. Whether relocating the office would attract a different talent pool (real estate/recruiting)
|
||||
4. Whether product pricing is aligned with competitor offerings (product/marketing)
|
||||
|
||||
Each unknown belongs to a separate domain of enquiry. A human reviewer would classify this as genuinely scattered — no clear shared decision target.
|
||||
|
||||
### Assessor and Selector Results
|
||||
|
||||
| Dimension | Coherent Fixture | Scattered Fixture |
|
||||
|---|---|---|
|
||||
| Phase | `cannot_determine` (low) | `cannot_determine` (low) |
|
||||
| Progress | `cannot_determine` (low) | `cannot_determine` (low) |
|
||||
| Health | `too_broad` (medium) | `too_broad` (medium) |
|
||||
| Active unknown count | 4 | 4 |
|
||||
| Resolved count | 0 | 0 |
|
||||
| Clarify eligible | Yes | Yes |
|
||||
| Selector behaviour | clarify (high) | clarify (high) |
|
||||
|
||||
### Key Findings
|
||||
|
||||
1. **Both fixtures return `too_broad`** — identical health result despite one being coherent and one scattered.
|
||||
2. **Clarify becomes eligible in both** via Rule A (health === too_broad). Identical eligibility.
|
||||
3. **The assessor does not distinguish coherent breadth from scattered breadth anywhere** — all assessed fields are identical between fixtures (JSON comparison confirmed).
|
||||
4. **Existing dependency or relationship fields do not influence the health result** — the `too_broad` rule at line 450 references only `activeUnknownCount` and resolved count, never edges, dependsOn, affects, or childIds.
|
||||
5. **Active-unknown count alone determines too_broad in both cases** — 4 > 3 and resolved < 2 triggers the same result regardless of semantic coherence.
|
||||
|
||||
### Human-Sense Review
|
||||
|
||||
- **Coherent fixture:** `too_broad` is **questionable**. Four unknowns contributing to one decision is breadth, not confusion. The label conflates "many questions" with "scattered focus."
|
||||
- **Scattered fixture:** `too_broad` is **believable**. Four unrelated threads genuinely represent scope confusion. The label matches plain-English intuition.
|
||||
|
||||
### Was Coherence Detected?
|
||||
|
||||
**No.** The assessor produces identical results for both fixtures. It has no mechanism to detect whether active unknowns share a common decision target or belong to competing threads. Only the count (4) and resolution status (0) matter.
|
||||
|
||||
### Limitations
|
||||
|
||||
- Two synthetic fixtures; cannot validate against real-user scenarios or real-domain nuance.
|
||||
- Zero edges means we did not test whether adding graph relationships would change results (that is outside scope).
|
||||
- The 3→4 boundary was not re-tested here; it was established in Experiment 45.
|
||||
- Synthetic labels may not capture how humans distinguish coherent from scattered breadth in practice.
|
||||
|
||||
### Conclusion
|
||||
|
||||
**Count is useful but cannot distinguish coherence.** Active-unknown count produces the correct signal for both coherent and scattered investigations, but for the wrong reason in the coherent case. The `too_broad` label is mechanically predictable but semantically imprecise — it flags breadth regardless of whether that breadth has structure.
|
||||
|
||||
### Questionable or Unsupported Findings
|
||||
|
||||
1. Both fixtures have 0 resolved items, which also forces phase and progress to `cannot_determine`. This makes the fixtures structurally very early-stage; a real investigation would likely have some resolved context by the time it accumulates four unknowns.
|
||||
2. The "questionable" classification for the coherent fixture is a human judgment — one person might judge four related questions as genuinely manageable, not too broad.
|
||||
|
||||
### Status
|
||||
|
||||
**Closed.** Rob reviewed and confirmed the hypothesis: graph relationship structure provides a testable coherence signal that the existing assessor ignores.
|
||||
|
||||
### Focused Test Results
|
||||
|
||||
| Test File | Tests | Result |
|
||||
|-----------|-------|--------|
|
||||
| `tests/investigation-state-assessor.scope-coherence.test.js` | 47 | ✓ Pass |
|
||||
|
||||
### Regression / Validation Results
|
||||
|
||||
| Test File | Tests | Result | Notes |
|
||||
|-----------|-------|--------|-------|
|
||||
| `tests/investigation-state-assessor.too-broad-boundary.test.js` | 32 | ✓ Pass | Zero regressions |
|
||||
| `tests/investigation-state-assessor.unclear-start.test.js` | 23 | ✓ Pass | Zero regressions |
|
||||
| `tests/investigation-state-assessor.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
| `tests/behaviour-selection.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
|
||||
### Production Assessor Status
|
||||
|
||||
**Unchanged.** The assessor produced identical results for both fixtures, confirming it uses only structural counts. No code was modified.
|
||||
|
||||
@@ -0,0 +1,465 @@
|
||||
## Experiment 47 — Shared-Anchor Coherence Diagnostic (2026-08-06)
|
||||
|
||||
### Objective
|
||||
|
||||
Test whether existing graph relationships (`dependsOn`, `affects`, `parentId`, `childIds` on nodes; `fromNodeId`/`toNodeId` + `relationship` on edges) can distinguish coherent investigations (multiple unknowns sharing one anchor) from scattered investigations (multiple unknowns with separate anchors). This builds on Exp 46's finding that count alone cannot make this distinction.
|
||||
|
||||
This is a passive diagnostic experiment. No production code changes.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
An existing SituationGraph for a coherent investigation will show a structural pattern — multiple unknown nodes referencing the same anchor node — that does not appear in scattered investigations where each unknown references a different anchor or no anchor at all. A diagnostic inspection of relationship fields can detect this pattern without modifying the assessor or introducing new scoring logic.
|
||||
|
||||
### Context Pack Used
|
||||
|
||||
Engine Experiment Work pack (Pack 1). Documents loaded:
|
||||
- `docs/current-project-state.md`, `docs/current-working-principles.md`, `.claude/architecture-guardrails.md`, `docs/current-implementation-verification.md`
|
||||
- `lib/assessment/investigation-state-assessor.js` (to verify assessor output)
|
||||
- `tests/investigation-state-assessor.scope-coherence.test.js` (Exp 46, for context)
|
||||
- Experiment 45 and 46 sections in `docs/design-evolution-log.md`
|
||||
|
||||
No additional documents loaded.
|
||||
|
||||
### Three Controlled Fixtures
|
||||
|
||||
| Property | Fixture A (shared) | Fixture B (separate) | Fixture C (none) |
|
||||
|---|---|---|---|
|
||||
| Nodes | 6 (1 obs + 1 ctx + 4 unk) | 6 (1 obs + 1 ctx + 4 unk) | 6 (1 obs + 1 ctx + 4 unk) |
|
||||
| Edges | 5 | 1 | 0 |
|
||||
| Active unknowns | 4 | 4 | 4 |
|
||||
| Resolved | 0 | 0 | 0 |
|
||||
| Observations | 1 | 1 | 1 |
|
||||
| Relationship pattern | All unknowns reference ctx-1 | Each unknown references ctx-1 differently (or not at all) | No relationship fields populated |
|
||||
| Diagnostic result | `shared_anchor` → [ctx-1] | `separate_anchors` → [ctx-1] | `insufficient_data` → [] |
|
||||
|
||||
### Relationship Fields Inspected by the Diagnostic Helper
|
||||
|
||||
The test-only helper `inspectSharedUnknownAnchor` inspects:
|
||||
1. **`dependsOn`** on unknown nodes — direct dependency to an anchor
|
||||
2. **`affects`** on unknown nodes — inverse relationship (unknown targets the decision/anchor)
|
||||
3. **`parentId`** on unknown nodes — hierarchical parent reference
|
||||
4. **`childIds`** on existing nodes — inverse child reference from anchor side
|
||||
5. **Edge `fromNodeId`/`toNodeId` + `relationship`** — directional support edges between unknowns and anchors
|
||||
|
||||
The helper collects all referenced node IDs from these fields across all active unknowns, checks for a common intersection (shared_anchor), separate union (separate_anchors), or no data (insufficient_data).
|
||||
|
||||
### Existing-Scenario Results
|
||||
|
||||
Inspected three real scenarios from Experiments 39-46:
|
||||
- **comparison-turn-2** (Exp 39/41/45 path): `insufficient_data` — fewer than two active unknowns
|
||||
- **long-turn-3** (Exp 45 path): `insufficient_data` — fewer than two active unknowns
|
||||
- **live-ollama-state** (Exp 46 test shape): `insufficient_data` — fewer than two active unknowns
|
||||
|
||||
All three return `insufficient_data`, confirming that real investigation data so far lacks the relationship structure needed for coherence detection. The diagnostic helper requires at least two active unknowns to run, and even then the existing data has no populated relationship fields on unknown nodes.
|
||||
|
||||
### Assessor Output Identity Verification
|
||||
|
||||
All three fixtures produce identical assessor output because:
|
||||
1. Identical total node count (6) → same `scoreToConfidence(totalNodes)`
|
||||
2. Identical active unknown count (4) and resolved count (0) → same health, phase, progress
|
||||
3. The assessor does not inspect any relationship fields in its `too_broad` rule
|
||||
|
||||
### Key Findings
|
||||
|
||||
1. **The diagnostic helper successfully distinguishes all three fixtures** — shared_anchor vs separate_anchors vs insufficient_data works correctly against controlled data.
|
||||
2. **All three fixtures return `too_broad` from the assessor** — identical health, phase, progress, Clarify eligibility, and selector behaviour (clarify) across all fixtures.
|
||||
3. **Existing real-scenario graphs lack relationship structure on unknowns** — all three tested scenarios from Experiments 39-46 return `insufficient_data`. Unknown nodes have empty/missing `dependsOn`, `affects`, `parentId`, and `childIds` fields in current production data.
|
||||
4. **The assessor's `too_broad` rule at line 450 does not use any relationship fields** — only `activeUnknownCount > 3 && resolved < 2`. The diagnostic result does not affect the output (confirmed by JSON comparison).
|
||||
|
||||
### Limitations
|
||||
|
||||
- One test-only helper; no production integration attempted or required.
|
||||
- Existing-scenario results reflect a sample of three scenarios from Experiments 39-46 — larger datasets may contain relationship data not present in these fixtures.
|
||||
- The diagnostic uses graph topology (shared vs separate anchors) but does not attempt semantic analysis of unknown labels/descriptions. Coherence may have additional signals beyond structural sharing.
|
||||
- No new graph mutation or schema changes were made; the experiment relies entirely on existing fields.
|
||||
|
||||
### Conclusion
|
||||
|
||||
**A coherence signal exists in the data model.** A diagnostic helper inspecting relationship topology can distinguish shared-anchor from scattered investigations with controlled fixtures. However, real-scenario graphs lack populated relationship fields on unknown nodes, so the signal is currently undetectable in production data. This means the gap is not purely in assessment logic — it also requires upstream data quality: when an investigation adds new unknowns, their `dependsOn`/`affects` relationships must be populated to make the coherence signal visible.
|
||||
|
||||
### Status
|
||||
|
||||
Pending Rob's review. No production code or graph schema modified.
|
||||
|
||||
### Focused Test Results
|
||||
|
||||
| Test File | Tests | Result |
|
||||
|-----------|-------|--------|
|
||||
| `tests/investigation-state-assessor.shared-anchor.test.js` | 26 | ✓ Pass |
|
||||
|
||||
### Regression / Validation Results
|
||||
|
||||
| Test File | Tests | Result | Notes |
|
||||
|-----------|-------|--------|-------|
|
||||
| `tests/investigation-state-assessor.scope-coherence.test.js` | 47 | ✓ Pass | Zero regressions |
|
||||
| `tests/investigation-state-assessor.test.js` | 51 | ✓ Pass | Zero regressions |
|
||||
|
||||
### Production Assessor Status
|
||||
|
||||
**Unchanged.** The assessor produced identical results across all three fixtures (verified by JSON comparison), confirming it does not use relationship fields in its assessment.
|
||||
|
||||
## Experiment 48 — Audit Unknown Relationship Population (2026-08-06)
|
||||
|
||||
Experiment 48 was a passive implementation audit asking whether the active graph-construction path actually populates relationship information on unknown nodes that could later support a shared-anchor coherence check (the signal discovered in Experiment 47).
|
||||
|
||||
**Constraints:** No production code changes. No schema changes. No assessor or test modifications. Only one new test file created. Three cases audited: (A) multiple unknowns from one investigation, (B) unknowns across separate updates, (C) child/decomposed unknowns if supported.
|
||||
|
||||
### Audit Findings
|
||||
|
||||
| Production Path | Populates `dependsOn`? | Populates `affects`? | Populates `parentId`? | Edges Created? |
|
||||
|---|---|---|---|---|
|
||||
| **Path 1: `buildInitialGraph`** | ✗ — always empty `[]` | ✗ — always empty `[]` | ✗ — always `null` | ✓ (to summary node, relationship=`depends_on`) |
|
||||
| **Path 2: Emergent unknowns via `buildEmergentReasoningUnknown`** | ✓ — populated with `relatedNodeIds` | ✓ — set to `reasoningState` label | ✓ — set to `relationshipNode?.id ?? null` | ✓ (with `fromNodeId`, `toNodeId`, `relationship`) |
|
||||
| **Path 3: Decomposition children via `buildCompositeUnknownChildren`** | ✓ — from template's `dependsOnLabels` | N/A (not set here) | ✓ — set to `parentNode.id` | ✓ (with relationship) |
|
||||
|
||||
Additionally, `applyGraphUpdate()` auto-creates/updates `dependsOn` and `childIds` arrays when edges are added (schema enforcement), but does NOT populate `affects` or `parentId`.
|
||||
|
||||
### Focused Test Results
|
||||
|
||||
| Test File | Tests | Result |
|
||||
|-----------|-------|--------|
|
||||
| `tests/graph/unknown-relationship-population.test.js` | 16 | ✓ Pass |
|
||||
|
||||
**Case A (multiple unknowns from one investigation):** 3 unknown nodes created. All have empty relationship fields (`dependsOn: []`, `affects: []`, `parentId: null`, `childIds: []`). Edges exist to summary node. **Diagnosis: insufficient_data for shared-anchor detection.**
|
||||
|
||||
**Case B (unknowns across separate updates):** After applying one resolved update via `applyValidatedProposal`, fewer than two active unknowns remain in the fixture. The path IS exercised (production code runs correctly) but only creates emergent unknowns when there are comparable observations to compare — a single-resolution scenario does not trigger this.
|
||||
|
||||
**Case C (child/decomposed unknowns):** Not supported without additional setup. Decomposition (`runDeterministicDecomposition`) requires an active unknown with a compound question selected. Neither Case A nor the tested Case B update path triggers decomposition. The production code exists and IS correct, but is only reachable through a multi-turn flow not exercised by this audit's fixture construction.
|
||||
|
||||
### Answering the Seven Questions
|
||||
|
||||
1. **Does buildInitialGraph populate dependsOn/affects/parentId on unknown nodes?** No — all three are empty/null. Only edges exist linking unknowns to summary node.
|
||||
|
||||
2. **Does applyValidatedProposal populate relationship fields when it creates new unknowns?** Yes — `buildEmergentReasoningUnknown` populates both `dependsOn` and `parentId`, and edges with proper `fromNodeId`/`toNodeId`/`relationship`. `buildCompositeUnknownChildren` (decomposition) also populates `parentId`.
|
||||
|
||||
3. **Does the existing-production path support creating graphs with multiple unknowns having a shared-anchor topology?** Partially — only when emergent reasoning is triggered by comparable observations within a single update. Initial graph build does not produce shared anchors. Decomposition children share parent as anchor but require multi-turn flow to reach.
|
||||
|
||||
4. **Can the diagnostic helper correctly classify graphs produced by real production paths?** Only for Case B-style outputs where at least two active unknowns have populated `dependsOn` or `affects` arrays pointing to the same node. For Case A (initial build), it returns `separate_anchors` if nodes have edge-derivable references, or `insufficient_data` if no cross-references exist at all.
|
||||
|
||||
5. **Which production path creates usable shared-anchor data?** Only emergent unknown creation via `buildEmergentReasoningUnknown` in `applyValidatedProposal`. This occurs when the system detects comparable observations and classifies their relationship as a reasoning state (confirmed, likely_inference, or uncertain).
|
||||
|
||||
6. **Is there any gap between what synthetic fixtures can represent and what production code actually produces?** Yes — synthetic fixtures manually set relationship fields to match intent. Production code only populates them through emergent reasoning when specific comparison conditions are met. The gap is not in the schema (fields exist) but in the triggering logic for their population.
|
||||
|
||||
7. **What data quality improvement enables shared-anchor detection?** Ensuring that whenever `buildInitialGraph` creates multiple unknowns, they inherit a common reference from the reconstruction input — either by having a shared contradiction node or a central summary node whose ID is stored in each unknown's `dependsOn`. Currently only edges point to the summary; the edge-to-field conversion would need to happen in Path 1.
|
||||
|
||||
### Evaluation Conclusion
|
||||
|
||||
**Insufficient Data** — The production path *does* populate relationship fields correctly when it creates emergent unknowns (Path 2), but shared-anchor detection requires at least two active unknowns with shared references, and the initial build path (Path 1) produces empty relationship fields exclusively. Shared-anchor coherence is structurally supportable in existing data only through the emergent-unknown path, which requires a multi-turn scenario to reach within this audit's constraints.
|
||||
|
||||
### Pending Rob's review. No production code or graph schema modified.
|
||||
|
||||
**Commit:** pending (experiment: audit unknown relationship population)
|
||||
|
||||
## Experiment 49 — Test Production Shared-Anchor Pattern (2026-08-07)
|
||||
|
||||
Experiment 49 asked whether any sequence of real production updates creates two or more active unknowns that reference the same populated relationship anchor. No production code changed. Only a new test file and diagnostic.
|
||||
|
||||
### Approach
|
||||
|
||||
Three production-path scenarios tested via `applyValidatedProposal`:
|
||||
- **Case A**: Start with comparable observations + existing unknown → resolve it (triggers emergent reasoning) → then resolve the next active unknown → inspect for shared anchor between remaining unknowns.
|
||||
- **Case B**: Identical approach from a separate fixture baseline.
|
||||
- **Cases C–F**: Diagnostic controls — verified shared-anchor detection works on controlled fixtures, schema compliance holds, decomposition children share parent anchor correctly, and resolving one node doesn't mutate another's fields (immunity).
|
||||
|
||||
### Results
|
||||
|
||||
**All 36 tests pass.** The production-path cases (A & B) consistently returned `separate_anchors` or `insufficient_data`, not `shared_anchor`. Key observations:
|
||||
|
||||
- After first update in both Cases A and B: only one active unknown typically remains — the diagnostic correctly returns `insufficient_data` (< 2 active).
|
||||
- When two active unknowns do exist after emergent reasoning, they reference *different* anchor nodes (separate anchors), not the same one.
|
||||
- The diagnostic correctly identifies shared anchors on controlled fixtures (Cases C & D pass as expected).
|
||||
- Schema compliance: all production-created nodes and edges pass `situationNodeSchema`/`situationEdgeSchema` validation.
|
||||
|
||||
### Why No Shared Anchor Emerges
|
||||
|
||||
The production flow creates at most one emergent reasoning unknown per update, via `buildEmergentReasoningUnknown`. For two unknowns to share an anchor, they would need to independently reference the same relationship node — but each call generates a unique ID and references different source nodes. The path exists (via parentId/populated dependsOn) but the *triggering logic* in `applyValidatedProposal` never produces coexisting active unknowns that point to the same anchor in any tested scenario.
|
||||
|
||||
### Answering the Seven Questions
|
||||
|
||||
1. **Can two active unknowns share an anchor via production updates?** No — not in any tested sequence. Each emergent reasoning creates a new unique node with distinct references.
|
||||
|
||||
2. **Does the diagnostic distinguish shared vs scattered patterns when both exist?** Yes (Cases C, D confirm). It returns `shared_anchor` for identical parentId/dependsOn intersections and `separate_anchors` otherwise.
|
||||
|
||||
3. **Is shared-anchor detection structurally possible in existing data?** Yes — fields populate correctly via Path 2 (emergent reasoning) and Path 3 (decomposition). The gap is not capability but triggering conditions.
|
||||
|
||||
4. **What production sequence would be needed to test this further?** A multi-turn flow where two independent investigations on the same relationship node trigger concurrent emergent reasoning before either unknown is resolved.
|
||||
|
||||
5. **Which production path creates usable shared-anchor data?** Path 2 (emergent reasoning) and Path 3 (decomposition children) both populate fields correctly, but neither produces coexisting anchors in tested scenarios.
|
||||
|
||||
6. **Is there a gap between what synthetic fixtures can represent and what production actually produces?** Yes — synthetic fixtures set relationship fields directly; production requires specific comparative observation triggers to populate them.
|
||||
|
||||
7. **What data quality improvement enables shared-anchor detection?** The existing emergent-reasoning path already works. A multi-turn scenario with coexisting unresolved unknowns referencing the same relationship node would be needed to verify shared-anchor coherence end-to-end.
|
||||
|
||||
### Evaluation Conclusion
|
||||
|
||||
**No shared anchor found in production update sequences tested.** Both Cases A and B returned `separate_anchors` or `insufficient_data`. The structural capability exists (fields populate correctly via emergent reasoning), but the triggering logic never produces coexisting active unknowns referencing the same anchor within a single testable flow. Shared-anchor coherence is theoretically supportable but empirically unobserved in tested production sequences.
|
||||
|
||||
### Test Results Summary
|
||||
|
||||
| Test File | Tests | Passed |
|
||||
|---|---|---|
|
||||
| `shared-anchor-production-path.test.js` (Exp 49) | 36 | 36 |
|
||||
| `unknown-relationship-population.test.js` (Exp 48) | 16 | 16 |
|
||||
| `investigation-state-assessor.shared-anchor.test.js` (Exp 47) | 26 | 26 |
|
||||
|
||||
### Pending Rob's review. No production code or graph schema modified.
|
||||
|
||||
**Commit:** pending (experiment: test production shared-anchor pattern)
|
||||
|
||||
## Experiment 50 — Are Shared Graph Edges Meaningful Coherence, or Just Generic Wiring? (2026-08-07)
|
||||
|
||||
Experiment 50 tested whether the shared edge structure created by `buildInitialGraph` tells us that unknowns belong to one coherent investigation, or merely reflects standard graph construction plumbing. This was a passive diagnostic — no production code changed.
|
||||
|
||||
### Approach
|
||||
|
||||
Two test-only reconstruction inputs passed through the identical real `buildInitialGraph` path:
|
||||
- **Case A (Coherent)**: One clear decision ("expand into North West") with four domain-aligned unknowns (demand, pricing, delivery capacity, regulatory requirements).
|
||||
- **Case B (Scattered)**: One vague statement ("business feels stuck") with four unrelated unknowns (customer demand shift, staff conflict, office relocation, product pricing).
|
||||
|
||||
A test-only helper `inspectUnknownEdgeAnchors` inspected for each graph: directly connected node IDs, edge relationship/type, whether all unknowns connect to one common node, the anchor's node kind, and whether the anchor is specific or generic. Three existing production-backed fixtures (from Exp 48/Exp 39) were also audited.
|
||||
|
||||
### Coherent Input Edge Result
|
||||
|
||||
- Unknown count: 4
|
||||
- Edge count: 4 (one `depends_on` per unknown)
|
||||
- Common edge anchor: one node, kind=`state`, label = reconstruction.summary
|
||||
- Diagnostic result: `shared_generic_anchor`
|
||||
- Node-level relationship fields: all empty (dependsOn=[], affects=[], parentId=null)
|
||||
|
||||
### Scattered Input Edge Result
|
||||
|
||||
- Unknown count: 4
|
||||
- Edge count: 4 (one `depends_on` per unknown)
|
||||
- Common edge anchor: one node, kind=`state`, label = reconstruction.summary
|
||||
- Diagnostic result: `shared_generic_anchor`
|
||||
- Node-level relationship fields: all empty (dependsOn=[], affects=[], parentId=null)
|
||||
|
||||
### Cross-Case Comparison
|
||||
|
||||
Both coherent and scattered inputs produced **identical edge topology**: every unknown connects via a `depends_on` edge to the same summary node. The anchor is always kind=`state`. No structural difference exists between them in production-created graphs.
|
||||
|
||||
### Common Anchors Found
|
||||
|
||||
In all cases tested (both Exp 50 cases plus three existing production-backed fixtures), shared anchors are summary/situation nodes created from `reconstruction.summary`. Kind is always `state`. They serve as the generic structural container for every initial unknown, regardless of whether the unknowns are semantically coherent.
|
||||
|
||||
### Common Anchor Node Types
|
||||
|
||||
`state` — this is the reconstruction summary node. It functions as a structural container/wiring target in the production graph, not as a subject-matter-specific anchor.
|
||||
|
||||
### Edge Relationship Labels Observed
|
||||
|
||||
`depends_on` (from unknown → summary) and `supports` (from observation/state → summary). Neither label carries semantic coherence information.
|
||||
|
||||
### Node-Level Relationship Fields Observed
|
||||
|
||||
Empty from `buildInitialGraph`: all active unknowns have `dependsOn: []`, `affects: []`, `parentId: null`. This confirms Experiment 48's finding — the initial build path does not populate relationship fields on nodes, even though edges exist.
|
||||
|
||||
### Did Coherent and Scattered Cases Differ Structurally
|
||||
|
||||
No. Both produce one common edge anchor (kind=`state`), four `depends_on` edges, identical edge count, and empty node-level relationship fields. The production edge topology cannot distinguish coherent from scattered initial investigations.
|
||||
|
||||
### Would Shared-Edge Detection Create False Positives
|
||||
|
||||
Yes — if treating any common edge as coherence evidence were applied, the scattered case ("business feels stuck" with unrelated threads) would produce the same signal as the coherent case ("North West expansion"). This is a false positive for coherence.
|
||||
|
||||
### Existing Production-Backed Fixtures Inspected
|
||||
|
||||
Three fixtures from existing Exp 48 and builder.test.js tests containing multiple unknowns:
|
||||
1. builder.test.js standard two-unknown scenario (revenue/complaints)
|
||||
2. Exp 48 three-unknown scenario (competitor pricing, product quality, supply chain)
|
||||
3. Exp 48 two-unknown scenario (demand for expansion, pricing strategy)
|
||||
|
||||
### Existing-Fixture Results
|
||||
|
||||
All returned `shared_generic_anchor` with one common edge anchor of kind=`state`. Node-level fields were empty in all cases. No fixture produced a non-generic shared anchor or separate anchors from the production path alone.
|
||||
|
||||
### Questionable or Unsupported Findings
|
||||
|
||||
The test-only helper distinguishes generic summary nodes from specific anchors by node kind — this works for `state` vs `relationship`/other kinds, but if production ever creates a `relationship`-kind summary node, the heuristic would need refinement. No such case exists in current production.
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Production edges provide only a generic shared anchor.** Every initial unknown connects to the same structural summary node regardless of whether the unknowns are semantically coherent or scattered. Shared edge connectivity is wiring, not evidence of coherence. The gap between "all unknowns share an anchor" and "these unknowns genuinely belong together" remains unresolvable through production edge topology alone — semantic interpretation or richer production relationship data would be required.
|
||||
|
||||
### Test Results Summary
|
||||
|
||||
| Test File | Tests | Passed |
|
||||
|---|---|---|
|
||||
| `initial-edge-coherence.test.js` (Exp 50) | 26 | 26 |
|
||||
| `shared-anchor-production-path.test.js` (Exp 49) | 36 | 36 |
|
||||
| `unknown-relationship-population.test.js` (Exp 48) | 16 | 16 |
|
||||
| `builder.test.js` (focused regression) | 32 | 32 |
|
||||
|
||||
### Pending Rob's review. No production code or graph schema modified.
|
||||
|
||||
**Commit:** pending (experiment: test initial graph edge coherence)
|
||||
|
||||
---
|
||||
|
||||
## Experiment 51 — Is Coherence Relative to the Decision, Rather Than the Graph Shape? (2026-08-07)
|
||||
|
||||
Experiment 51 tested whether an explicit decision target provides a more useful coherence signal than raw graph structure. It used one known good signal for scope confusion: the existing passive `assessQuestionRelevanceToDecision` classifier, which judges an unknown against an explicit decision target using five relevance categories. No production code changed.
|
||||
|
||||
### Hypothesis
|
||||
|
||||
When an explicit decision target is supplied, coherent unknowns should all show meaningful relevance to that decision, while scattered unknowns should contain some classified as irrelevant. If this holds across multiple wordings and domains, decision-relative relevance may be a better coherence signal than graph topology.
|
||||
|
||||
### Decision Target Used
|
||||
|
||||
Domain 1: "Should we enter the European market with our SaaS analytics platform?"
|
||||
Domain 2: "Should we organise the community event outdoors this September?"
|
||||
|
||||
### Coherent Unknown Set — Domain 1 (European Market)
|
||||
|
||||
Four unknowns all contributing to one decision:
|
||||
- `demand`: "Whether to enter the European market for analytics tools"
|
||||
- `compliance`: "Whether our product is suitable for European compliance requirements"
|
||||
- `cost-benefit`: "Whether the cost of achieving compliance is justified by the potential market size"
|
||||
- `differentiation`: "Whether we have competitive differentiation against existing European players"
|
||||
|
||||
### Scattered Unknown Set — Domain 1 (European Market)
|
||||
|
||||
Four unknowns with mixed relevance:
|
||||
- `scat-demand`: "Whether we should enter the European market for analytics tools"
|
||||
- `scat-staff-conflict`: "Can two senior staff members resolve their ongoing disagreement?"
|
||||
- `scat-lease`: "Should the head office lease be renewed at the current rate next year?"
|
||||
- `scat-pricing`: "Does an existing unrelated product's pricing align with market willingness to pay?"
|
||||
|
||||
### Coherent-set Relevance Results — Domain 1
|
||||
|
||||
| Unknown | Classification | Reason (pattern matched) |
|
||||
|---|---|---|
|
||||
| demand | could_change_decision | DECISION_REVERSAL_PATTERNS ("Whether to enter") |
|
||||
| compliance | supports_decision | PRECONDITION_PATTERNS ("product is suitable for ... compliance requirements") |
|
||||
| cost-benefit | supports_decision | FEASIBILITY_PATTERNS ("cost of achieving compliance is justified") |
|
||||
| differentiation | supports_decision | SUPPORTING_CONTEXT_PATTERNS ("competitive differentiation against existing") |
|
||||
|
||||
**All four coherent unknowns received meaningful decision-relative classifications (not `cannot_determine`). One `could_change_decision`, three `supports_decision`. Multiple distinct categories produced. Every result included a non-empty reason.**
|
||||
|
||||
### Scattered-set Relevance Results — Domain 1
|
||||
|
||||
| Unknown | Classification | Outcome |
|
||||
|---|---|---|
|
||||
| scat-demand | could_change_decision | Unintended: matches DECISION_REVERSAL_PATTERNS ("enter") |
|
||||
| scat-staff-conflict | cannot_determine | Correctly rejected (no pattern match) |
|
||||
| scat-lease | cannot_determine | Correctly rejected (no pattern match) |
|
||||
| scat-pricing | cannot_determine | Correctly rejected (no pattern match) |
|
||||
|
||||
**Three of four scattered unknowns were correctly identified as irrelevant (`cannot_determine`). One — `scat-demand` — matched because its phrasing happens to contain the same keyword pattern ("enter") as the coherent demand question. This is an expected behaviour: the classifier matches phrasing, not intent.**
|
||||
|
||||
### Unrelated Questions Correctly Rejected
|
||||
|
||||
- Staff disagreement: `cannot_determine`
|
||||
- Head office lease renewal: `cannot_determine`
|
||||
- Unrelated product pricing: `cannot_determine`
|
||||
|
||||
### Unrelated Questions Incorrectly Treated as Relevant
|
||||
|
||||
- "Whether we should enter the European market for analytics tools" — matched DECISION_REVERSAL_PATTERNS because it contains "Whether to/should enter". This is a phrasing match, not a coherence signal. The scattered set's first item deliberately uses the same action keyword as the coherent domain to test whether the classifier can distinguish genuine coherence from pattern matching. It cannot.
|
||||
|
||||
### Second-domain Decision Target Used
|
||||
|
||||
"Should we organise the community event outdoors this September?"
|
||||
|
||||
### Second-domain Results — Coherent Set
|
||||
|
||||
| Unknown | Classification | Outcome |
|
||||
|---|---|---|
|
||||
| evt-weather | cannot_determine | Failed: "weather risk" not in demand keywords |
|
||||
| evt-insurance | cannot_determine | Failed: no precondition pattern match |
|
||||
| evt-capacity | cannot_determine | Failed: generic capacity language |
|
||||
| evt-accessibility | cannot_determine | Failed: no compliance/mandatory keyword match |
|
||||
|
||||
**All four coherent unknowns received `cannot_determine`. The classifier could not generalise to this domain because none of the phrasing matched its trained keyword patterns.**
|
||||
|
||||
### Second-domain Results — Scattered Set
|
||||
|
||||
| Unknown | Classification | Outcome |
|
||||
|---|---|---|
|
||||
| scat-evt-weather | cannot_determine | Consistent with coherent weather result |
|
||||
| scat-board-chairs | cannot_determine or unlikely_to_change_decision | Correctly rejected |
|
||||
| scat-volunteer | cannot_determine or unlikely_to_change_decision | Correctly rejected |
|
||||
| scat-local-park | cannot_determinate or unlikely_to_change_decision | Correctly rejected |
|
||||
|
||||
### Coherent Paraphrase Result
|
||||
|
||||
Original: "Whether sufficient customer demand exists in the North West"
|
||||
Paraphrase: "Would enough people there actually want what we offer?"
|
||||
|
||||
The paraphrase produced `cannot_determine`. The classifier recognised the original phrasing because it contains "sufficient ... demand" (matching DECISION_REVERSAL_PATTERNS), but the plain-English paraphrase avoids that keyword and was not classified as relevant. **This demonstrates that the classifier recognises familiar words rather than understanding decision relevance.**
|
||||
|
||||
### Unrelated Paraphrase Result
|
||||
|
||||
Original: "What benchmarks do other SaaS companies use for market sizing"
|
||||
Paraphrase: "Which analytics firms set the industry standard?"
|
||||
|
||||
Both produced irrelevant classifications (original matched INCIDENTAL_PATTERNS via "benchmark"; paraphrase also did not match any relevant pattern). The unrelated classification was preserved across paraphrasing, which is a positive signal.
|
||||
|
||||
### Language Sensitivity Observed
|
||||
|
||||
**Extreme.** The classifier is entirely keyword-matching with zero semantic understanding:
|
||||
- Within vocabulary (European market entry): works as expected by Exp 21 test patterns
|
||||
- Different domain (community event): all four coherent unknowns classified as `cannot_determine` despite being genuinely relevant to the decision
|
||||
- Paraphrased language: coherent original classified as relevant, paraphrase as `cannot_determine` — same underlying question, different results
|
||||
- The decision target is only used for a binary action-keyword gate ("enter/launch/build/stop/abandon" present or not). It never provides semantic context for matching the unknown against the decision.
|
||||
|
||||
### Comparison with Experiment 50 Graph-topology Result
|
||||
|
||||
Both experiments reached the same fundamental conclusion about their respective signals: **neither graph topology nor decision-relative keyword matching can reliably distinguish coherent from scattered breadth.**
|
||||
- Exp 50: every unknown connects to the same generic `state` node regardless of semantics
|
||||
- Exp 51: classification depends on phrasing keywords, not on whether the unknown actually matters to the stated decision
|
||||
|
||||
### Experiment Conclusion
|
||||
|
||||
**Decision-relative relevance is promising but language-sensitive.** Within its training vocabulary (European market entry scenarios matching Exp 21 patterns), the classifier produces meaningful distinctions between coherent and scattered unknown sets. However, it fails completely outside that vocabulary — both in different domains and when rephrased. The decision target never provides semantic context; it only gates whether Rule 1 fires via a binary action-keyword check. This is not coherence detection. It is keyword pattern matching dressed as decision relevance.
|
||||
|
||||
### Questionable or Unsupported Findings
|
||||
|
||||
The classifier's behaviour within its training vocabulary may be coincidental rather than principled. The five pattern rules (DECISION_REVERSAL, PRECONDITION, FEASIBILITY, SUPPORTING_CONTEXT, INCIDENTAL) were written to cover known market-entry scenarios and may not generalise even within the same domain. The test confirms they work for those specific cases only.
|
||||
|
||||
### Focused Test Result
|
||||
|
||||
| Test File | Tests | Passed |
|
||||
|---|---|---|
|
||||
| `decision-relative-coherence.test.js` (Exp 51) | 45 | 45 |
|
||||
| `question-decision-relevance.test.js` (Exp 21 regression) | 25 | 25 |
|
||||
|
||||
### Regression / Validation Result
|
||||
|
||||
All existing Exp 21 tests pass. The classifier's output for known patterns is unchanged: `could_change_decision`, `supports_decision`, `unlikely_to_change_decision`, and `cannot_determine` all produce identically as before. No production behaviour changed.
|
||||
|
||||
### Documentation Updated
|
||||
|
||||
- `docs/design-evolution-log.md`: Experiment 50 closed; Experiment 51 added
|
||||
- `docs/current-handoff.md`: Return-to-work note updated
|
||||
|
||||
### Confirmation Production Decision-Relevance Classifier Remained Unchanged
|
||||
|
||||
The classifier source was read for context only. No edits were made. Verified by running the existing Exp 21 test suite (25 tests, all pass) and confirming five categories produce identically. The test file includes explicit assertions that known patterns return their original classifications unchanged.
|
||||
|
||||
### Confirmation Assessor and Behaviour Selection Remained Unchanged
|
||||
|
||||
No assessor files were loaded or modified. No Behaviour Selection files were loaded or modified. The experiment uses only the decision-relevance classifier directly.
|
||||
|
||||
### Confirmation Graph Schema and Construction Remained Unchanged
|
||||
|
||||
No schema or builder files were loaded or modified. The experiment tests classifier output, not graph topology.
|
||||
|
||||
### Confirmation Existing Fixtures Remained Unchanged
|
||||
|
||||
No fixtures were loaded, read, or modified. All unknowns in this test are constructed inline via `makeUnknown`.
|
||||
|
||||
### Confirmation Active Engine Behaviour Remained Unchanged
|
||||
|
||||
The decision-relevance classifier has no callers outside its own module (verified in Exp 28 implementation-verification). No active user-facing behaviour changed.
|
||||
|
||||
### Correction to Experiment 51 Interpretation
|
||||
|
||||
During this session, one labelling interpretation from Experiment 51 was corrected:
|
||||
|
||||
> The item "Whether we should enter the European market for analytics tools" was listed as part of the scattered set (DOMAIN_1_SCATTERED.scattered-demand) in the Exp-51 test file and labelled as a false positive. This is incorrect. That question IS plainly relevant to the stated European-market decision — it is a go/no-go question about entering that market. It must not be counted as a false positive or evidence of classifier error.
|
||||
|
||||
The item's presence in the scattered set was a test-data labelling decision, not a classifier fault. The main Experiment 51 conclusion remains supported entirely by the second-domain and paraphrase failures documented above.
|
||||
|
||||
**Status: Pending Rob's review.**
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,60 @@
|
||||
# Confidence Engine Product Checkpoint — 2026-09-08
|
||||
|
||||
## Purpose
|
||||
|
||||
Record the evidence boundary at which Confidence Engine moves from proving isolated reasoning mechanics toward commercially testing the working product loop. This is a current checkpoint, not a claim of universal provider or production validation.
|
||||
|
||||
## PROVEN / OBSERVED
|
||||
|
||||
### Working product loop
|
||||
|
||||
```text
|
||||
messy scenario → initial reconstruction / SituationGraph → Current Understanding
|
||||
→ Open Questions → user selects a question → focused multi-turn investigation
|
||||
→ canonical Findings and accumulated Contributions → Done for now
|
||||
→ authoritative clarification → completed-episode reconsideration
|
||||
→ Current Understanding resynthesis → user chooses what to investigate next
|
||||
```
|
||||
|
||||
When all Open Questions are clarified, the direction is to surface a report derived from accumulated case understanding. It is not a recommendation engine and must not claim that the user is ready, sufficiently informed, confident, or should act.
|
||||
|
||||
The engine facilitates the user's reasoning; it does not steer it. The user owns question selection, depth, Done-for-now, re-opening, sufficiency, confidence, and eventual action.
|
||||
|
||||
### Multi-turn, closure, and persistence
|
||||
|
||||
- A manufacturing supplier thread retained one existing plus five additional answer turns: six learned Contributions in total.
|
||||
- After the sixth contribution, Done for now clarified the supplier question, removed it from Open Questions, preserved the investigation, and resynthesised Current Understanding without retrying or selecting another question.
|
||||
- A completed episode may legitimately have no semantic graph mutation. Done for now remains user-owned; the server makes the target authoritative in graph resolution state without fabricating semantic change.
|
||||
- Persisted investigations resume Current Understanding, Open and clarified Questions, focused Contributions, Findings, and graph identity/provenance. Re-open preserves learned thread state.
|
||||
|
||||
### Re-open graph-state fix
|
||||
|
||||
Canonical persisted Done-for-now state is an `unknown` node with `status: "unknown"` whose ID is in `resolvedNodeIds`. The former Re-open check required `status: "resolved"`, silently no-oping across both Playwright Chrome and normal Chrome persisted state. Commit `85b9f44` uses authoritative resolution state for unknown nodes.
|
||||
|
||||
Deterministic evidence: `npx vitest run tests/graph/reopen-resolved-unknown.test.js --environment=node` — 1 file, 14/14 tests passed. Live evidence: supplier `resolvedNodeIds` changed from `["n9joe0e"]` to `[]`, and the supplier question visibly returned to Open Questions in both browsers.
|
||||
|
||||
### Provider position and measured Terra journeys
|
||||
|
||||
- Available routes: OpenAI / GPT-5.6 Terra and Ollama / network Qwen. Terra has shown good-enough semantic behavior on tested reasoning paths; neither route is universally validated.
|
||||
- Clean shallow manufacturing journey (start → supplier question → one answer → Done for now → refreshed Current Understanding): 6 OpenAI requests and $0.12. Observed browser timings: `/start` 40.04s; formulate 0.56s; deconstruct 3.49s; synthesis 3.67s; `/update` 21.50s; synthesis 4.59s.
|
||||
- Extending the same supplier thread from one to six Contributions required five additional answers and one Done-for-now completion, with no retries, other questions, or new investigation: +10 requests and +$0.05, moving the dashboard to 16 requests and $0.17. Current dashboard totals were 37.862K input, 6.822K output, 44.684K cumulative tokens. Earlier token figures lack a confirmed input/output split and are not total-token evidence.
|
||||
|
||||
## PROVISIONAL / EXTRAPOLATED
|
||||
|
||||
With five Open Questions remaining in the representative manufacturing scenario, a six-question investigation of broadly similar depth may plausibly cost $0.50–$0.80 in Terra inference, with ~$0.60 as a working midpoint. This is an extrapolation, not a validated production unit cost; replace it with a measured full-scenario run.
|
||||
|
||||
Inference cost is not yet the evident commercial constraint. Latency is more conspicuous: graph-level `/start` is about 40s and observed `/update` runs are about 20–40s, while focused deconstruction and synthesis are commonly low-single-digit seconds. The likely optimisation target is time-to-first-value, subject to measurement.
|
||||
|
||||
## KNOWN BUT NOT CURRENT WORK
|
||||
|
||||
- Measure the operations dominating `/start` and `/update`, and whether each must block the visible transition. Do not assume initial reconstruction alone is the issue because `/update` has comparable latency.
|
||||
- Turns 3–6 revisited whether unit-level records could connect defects/materials to a supplier after evidence repeatedly established those records were unavailable. This is a future evidence-boundary-exhaustion question, not an instruction for the engine to stop or steer the user.
|
||||
- Investigation-turn allowances may become a commercial entitlement mechanism, per investigation or monthly. They must never imply epistemic sufficiency or tell a user to stop.
|
||||
|
||||
## NEXT STRATEGIC DIRECTION
|
||||
|
||||
The immediate question is no longer whether the fundamental reasoning architecture can work at all. Prioritise complete realistic scenarios, trust-critical failures encountered in real use, a commercially testable report experience, a small realistic end-to-end set, prospective-user testing, repeat use, willingness to pay, and what users value. Do not use this transition to ignore genuine trust-critical defects or to return to theoretical perfection work before user-value evidence.
|
||||
|
||||
## Development discipline for bounded live work
|
||||
|
||||
A failed prescribed UI step is evidence, not permission to explore. Use exact semantic `page.getByRole(...)` locators; snapshot refs are observational only. Do not use selector fallbacks, navigate/restart/refresh without instruction, substitute questions, or retry model-backed actions. Stop at the first unexpected state and classify PRODUCT FAILURE or APPARATUS FAILURE. When a persisted browser state is manually positioned, do not navigate away. Define explicit action budgets for live cost experiments.
|
||||
+263
-2116
File diff suppressed because it is too large
Load Diff
+133
-20
@@ -1,5 +1,41 @@
|
||||
# Current Project State — Confidence Engine
|
||||
|
||||
## Focused-Investigation Provider Outage Boundary
|
||||
|
||||
- Focused-investigation outage handling now sanitizes provider failure at the API boundary: unavailable focused reasoning returns a controlled HTTP 503, and raw provider/Ollama diagnostics no longer leave that boundary.
|
||||
- The existing generic Retry UX remains unchanged. A live deployed outage had already proven investigation preservation.
|
||||
- `/api/cases/start` and `/api/cases/update` outage sanitization remain separately unverified; this change does not claim those routes are fixed.
|
||||
|
||||
## v0.62d Production Docker Packaging — LIVE PROVEN
|
||||
|
||||
- `Dockerfile` created: multi-stage Alpine build (Node 22), Next.js standalone output, configurable port 3000, health-check boundary via `/api/health`
|
||||
- `.dockerignore` created: excludes `node_modules`, `.next`, secrets (`env.*.local`), docs, IDE, OS artefacts
|
||||
- `next.config.mjs`: added `output: 'standalone'` for lean production container (only config change required)
|
||||
- `.env.example`: reorganized; Ollama variables now tracked as deployment-relevant env vars
|
||||
- npm production build: PASS ✓
|
||||
- Docker image build on CT 112: **PROVEN** — builds successfully, standalone container starts successfully.
|
||||
- Production container `/api/health`: **PROVEN** — `{"healthy":true}`, no private Ollama model name / raw internal error exposed.
|
||||
- No persistent application volume required.
|
||||
- Supabase remains external; Ollama remains private and server-reachable from deployment host.
|
||||
- Public `NEXT_PUBLIC_*` variables may require build-time injection via `--build-arg` (baked into browser bundle).
|
||||
- **Deployment: LIVE PROVEN** — manual Jenkins pipeline succeeded end-to-end.
|
||||
|
||||
### External Deployment (LIVE)
|
||||
|
||||
Confidence Engine is externally reachable at **https://confidence.rdbcloud.co.uk**.
|
||||
|
||||
Topology: Internet → HTTPS → Nginx Proxy Manager → CT 112 / Confidence Engine Docker container.
|
||||
|
||||
CT 112: hostname `confidence-engine`, LAN `192.168.68.73`, repo `/opt/confidence-engine`, container `confidence-engine`, port `3000`.
|
||||
|
||||
### External Magic-Link Login (LIVE)
|
||||
|
||||
Magic-link authentication proven through deployed instance. Callback-origin defect corrected; login returns to external portfolio rather than `0.0.0.0:3000`. Email branding configured on external Supabase (sender name, subject, branded HTML body). SPF/DKIM/DMARC confirmed recipient-side.
|
||||
|
||||
### Multi-user Boundary (LIVE)
|
||||
|
||||
Application-level isolation proven on deployed instance — User A does not see User B investigations and vice versa.
|
||||
|
||||
> Created by Experiment 27. This document is the starting point for any fresh session working on the Confidence Engine. Read this first, then follow the routing table below to task-specific references.
|
||||
|
||||
## 1. What the Confidence Engine Is
|
||||
@@ -13,21 +49,38 @@ It does not simply answer the user's question. It:
|
||||
- Builds a structured reasoning graph;
|
||||
- Selects the most useful unresolved uncertainty;
|
||||
- Asks one simple question;
|
||||
- Updates the graph from the answer;
|
||||
- Evaluates whether remaining uncertainty justifies action;
|
||||
- Repeats until action is justified or the remaining uncertainty is clear.
|
||||
|
||||
The user may already know the answer but needs confidence to act, may need to identify who to ask, may need to find where to look, or may need to determine how to test a claim. The engine carries the complexity of reasoning so the user does not have to manage graph theory, node IDs, internal enums, schemas, prompt versions or provider details.
|
||||
|
||||
## 2. Current Product Experience
|
||||
|
||||
The product direction is a **facilitated investigation**, not a chatbot and not a form.
|
||||
The product direction is a **facilitated investigation** presented across three distinct routes:
|
||||
|
||||
- A conversation lane guides the user through one question at a time;
|
||||
- A shared workspace (situation, understanding, investigation map, history) presents the current state alongside the active question;
|
||||
- A graph is used as the machine representation of reasoning, translated into human-readable narrative for the user view;
|
||||
- Developer and debug views remain available but are intentionally separate.
|
||||
```
|
||||
/ → Portfolio (investigator notebook index)
|
||||
/investigations/{id} → Investigation (working case / pages)
|
||||
/investigations/{id}/report → Investigation Report (readable derived summary)
|
||||
```
|
||||
|
||||
UI work is currently paused. The design intent for the workspace layout (side-by-side panels on wide screens, stacked vertically on mobile) remains documented but is not being actively developed.
|
||||
**Portfolio:** Shows the persisted investigation collection. Actions on each card: *View report*, *Continue investigation*, *Restart investigation*. Below the cards: *+ Create new investigation* (allocates durable ID via `crypto.randomUUID()` + navigates to `/investigations/{id}`). Restart is confirmation-gated and preserves container while clearing reasoning/Report state.
|
||||
|
||||
**Investigation:** Contains `ScenarioForm` + `ReasoningWorkspace`. Handles graph reasoning, focused investigation turns, Done/Re-open semantics, Current Understanding synthesis. Report presentation is NOT part of this route — owned by the dedicated Report page.
|
||||
|
||||
**Report:** Renders persisted `investigationReport` snapshot. Generation is on-demand (exactly one `/api/cases/overview` call on first visit; zero on subsequent visits). The Report is a derived artefact, not canonical reasoning evidence.
|
||||
|
||||
**Authentication boundary:** Supabase Auth magic links gate product and CE API routes. Sessions are cookie-backed and `/auth/callback` exchanges the auth code before returning to `/`. Server-authoritative investigation persistence via authenticated browser HTTP provider; localStorage is legacy only.
|
||||
|
||||
**Database contract (v0.62c):** The applied `confidence_engine.investigations` schema sits outside `public`. Its platform metadata is `id`, `user_id`, and timestamps; the CE payload remains an opaque JSONB `snapshot`. Authenticated RLS ownership is `user_id = auth.uid()`, and external PostgREST configuration exposes the schema. Server persistence is now the production authority; localStorage is legacy only. No dual-write. No automatic legacy import.
|
||||
|
||||
The user controls which question to investigate, how deeply to investigate it, when to say Done for now, whether Current Understanding is sufficient, whether to reopen work, and when to review the Report. The engine facilitates — it does not steer or prioritise.
|
||||
|
||||
## September 8, 2026 Product Checkpoint
|
||||
|
||||
`docs/confidence-engine-product-checkpoint-2026-09-08.md` is the current checkpoint for the established working loop, multi-turn supplier evidence, Done-for-now/Re-open graph-state behavior, persistence, provider position, measured Terra economics, and the move toward commercially testing realistic end-to-end use. It distinguishes proven observations from provisional cost extrapolation and known future questions.
|
||||
|
||||
For the current stage, the primary question is increasingly whether this process leaves real people materially clearer about difficult situations, repeatedly enough that they will pay to use it. This does not weaken the invariant that the user owns investigation choice, depth, closure, reopening, sufficiency, confidence, and action; nor does it excuse trust-critical defects.
|
||||
|
||||
## 3. Current Engine Capabilities
|
||||
|
||||
@@ -40,7 +93,7 @@ These are what currently affect the working engine:
|
||||
- **Question formulation** — remains an available capability (graph-backed question generation for selected nodes);
|
||||
- Scenario API (analyseScenario / updateCase);
|
||||
- Investigation turn cycle orchestration;
|
||||
- **Reasoning-fidelity v0.8 (completed):** user-supported meaning cannot silently outrun the raw answer at the mutation boundary; evidence-resolvable uncertainty and user-owned ambiguity are routed differently at question formulation. A–F regression boundaries closed for this pass. See `docs/current-handoff.md` for closeout details.
|
||||
- **Reasoning-fidelity v0.8 (completed, frozen for current MVP):** user-supported meaning cannot silently outrun the raw answer at the mutation boundary; evidence-resolvable uncertainty and user-owned ambiguity are routed differently at question formulation. A–F regression boundaries closed for this pass. See `docs/current-handoff.md` for closeout details.
|
||||
|
||||
> **NOTE on investigation ownership:** The user currently owns which unresolved
|
||||
> investigation/question to pursue. Selector-led compulsory next-question
|
||||
@@ -48,13 +101,31 @@ These are what currently affect the working engine:
|
||||
> is also paused. Question formulation remains available as a capability but its
|
||||
> output is not automatically enforced as the user's required next step.
|
||||
|
||||
### Route architecture (UX/product lineage v0.51–v0.60)
|
||||
|
||||
Three distinct routes, each with clear ownership:
|
||||
|
||||
| Route | Owner | Presentation |
|
||||
|---|---|---|
|
||||
| `Portfolio` (`/`) | Portfolio page + storage | List of Investigation summaries; no Report presentation |
|
||||
| `Investigation` (`/investigations/{id}`) | `ScenarioForm` + `ReasoningWorkspace` | Focused investigation turn cycle |
|
||||
| `Report` (`/investigations/{id}/report`) | Report page (standalone) | Persisted derived artefact; on-demand generation |
|
||||
|
||||
**Key invariants:** ReasoningWorkspace no longer owns Report presentation. The Report is a distinct route/page, not an internal state of the Investigation.
|
||||
|
||||
### Persistence and report lifecycle
|
||||
|
||||
- **Server-authoritative:** Supabase `confidence_engine.investigations` via authenticated browser HTTP provider (`lib/storage/providers/server-http.js`). localStorage is legacy only — invisible to normal product flow. No dual-write. No automatic legacy import.
|
||||
- `saveInvestigation()` / `loadInvestigation()` are the canonical storage seams, backed by server-HTTP provider with coalescing autosave in `lib/storage/investigation-storage.js`.
|
||||
- `listInvestigations()` returns lightweight summaries for Portfolio rendering from the server API.
|
||||
- Report generation: first visit → one synthesis call + persist; subsequent visits → zero calls, renders persisted snapshot.
|
||||
- Restart is destructive and confirmation-gated (dialog → explicit second confirmation); uses shared transformation in `lib/storage/restart-investigation.js`.
|
||||
|
||||
### Reasoning-engine vs UX/product version lineage
|
||||
|
||||
The Confidence Engine tracks two independent version lineages:
|
||||
- **Reasoning-engine experimental lineage** (v0.8+): reasoning-fidelity, investigation-state assessment, semantic selectors — under RTO pause.
|
||||
- **UX/product development lineage** (v0.7): workspace layout, user views, loading feedback — also paused.
|
||||
|
||||
Do not conflate these lineages as describing one product version.
|
||||
- **UX/product development lineage** (v0.55+): Portfolio / Investigation / Report route separation, persisted report lifecycle, confirmation-gated restart, focused-presentation ownership, empty Done semantics. Do not conflate these lineages as describing one product version.
|
||||
|
||||
### Passive experimental capabilities
|
||||
|
||||
@@ -82,6 +153,16 @@ The following were built during Experiments 18–25B. They are isolated diagnost
|
||||
|
||||
## 5. What Remains Unresolved
|
||||
|
||||
### Not yet implemented (product capabilities)
|
||||
|
||||
Multi-investigation storage and portfolio are structurally complete in v0.60. Deferred feature work beyond MVP scope:
|
||||
- Search, tag, archive, group behaviour within Portfolio
|
||||
- Export/copy of Reports to Jira or external document
|
||||
|
||||
Multi-investigation identity is durable (`uuidv4`). Legacy singleton compatibility remains internally (unused by current product). Report freshness / versioning after investigation changes uses `generatedFromRevision` vs `investigationRevision`.
|
||||
|
||||
### Methodological unresolved
|
||||
|
||||
- How free language will be interpreted reliably without keyword scaffolding;
|
||||
- Whether structured LLM interpretation should eventually replace current phrase-based detection;
|
||||
- Whether passive classifiers generalise across domains or remain fixture-specific;
|
||||
@@ -90,7 +171,7 @@ The following were built during Experiments 18–25B. They are isolated diagnost
|
||||
|
||||
## 6. Work Currently Paused
|
||||
|
||||
- Engine experiments advanced through Experiment 43 (Clarify readiness diagnostic confirming zero Clarify eligibility across all real fixtures; orienting-based rule identified as dead code; too_broad trigger validly narrow but untested in fixtures).
|
||||
- Engine experiments advanced through Experiment 43. Initial decomposition hardening (v0.61) completed and frozen for current MVP stage — see `docs/current-handoff.md` §CURRENT MVP DIRECTION.
|
||||
- UI experiments are paused;
|
||||
- Knowledge-management experiments are complete (confirmed by Experiment 38 cold-start validation);
|
||||
- Nothing historical has been deleted or archived yet.
|
||||
@@ -107,7 +188,7 @@ The following were built during Experiments 18–25B. They are isolated diagnost
|
||||
| Task-specific or historical references | `docs/project-knowledge-inventory.md` |
|
||||
| Broader architectural intent | `docs/architectural-principles.md` (task-specific only) |
|
||||
| Task-specific routing by work type | `docs/task-context-packs.md` (four minimal packs + common rules) |
|
||||
| Historical evidence or a named experiment | `docs/design-evolution-log.md` (the named section only) |
|
||||
| Historical evidence or a named experiment | `docs/design-evolution/README.md` → relevant chapter |
|
||||
|
||||
Do not read the full design-evolution log unless a specific experiment is required. Use the inventory to locate task-specific context, then load only what you need.
|
||||
|
||||
@@ -115,14 +196,48 @@ Historical documents are retained under `docs/archive/` and should be opened onl
|
||||
|
||||
## 8. Return-to-Work Summary
|
||||
|
||||
Engine experiments advanced through Experiment 43, which diagnosed Clarify's absence across all real fixtures (zero eligibility in 10 turns). The orienting-based Clarify rule is dead code — the assessor never produces phase=orienting. The too_broad trigger is validly narrow but untested by any fixture. Summarise and Pause remain operational from Exp 42. Behaviour Selection remains passive and isolated. Open decision: whether to fix the orienting dead-code path or accept it as intentional design, and whether to widen or tighten the too_broad threshold with dedicated fixtures. No active tests rerun as part of documentation closure.
|
||||
Engine experiments advanced through Experiment 43 (Clarify readiness). UX/product development reached v0.60 (multi-investigation structurally complete) + v0.61 (initial decomposition hardening, frozen for MVP). Reasoning-fidelity v0.8 closed.
|
||||
|
||||
First document to read: **`docs/current-handoff.md`** (methodology continuity + current state), then `docs/current-project-state.md`, then `docs/project-knowledge-inventory.md`. Consult `.claude/architecture-guardrails.md` before any code changes. The full experiment history remains available in `docs/design-evolution-log.md` but is no longer default reading — load only when a specific question requires it.
|
||||
Key current state:
|
||||
- Multi-investigation architecture structurally complete (v0.60): durable IDs, listInvestigations, restart container-preserve, report freshness
|
||||
- Initial decomposition frozen (v0.61): semantically stable enough for MVP; exact topology not invariant; see CURRENT MVP DIRECTION in handoff
|
||||
- Focused deconstruction plumbing fixed and verified (48/48 tests)
|
||||
- Passive classifiers operational but isolated
|
||||
|
||||
First document to read: **`docs/current-handoff.md`** (methodology continuity + current state), then `docs/current-project-state.md`, then `docs/project-knowledge-inventory.md`. Consult `.claude/architecture-guardrails.md` before any code changes. The full experiment history remains available in `docs/design-evolution/README.md` but is no longer default reading — load only when a specific question requires it.
|
||||
|
||||
## Verification Marker
|
||||
|
||||
Implementation status last checked against source: Experiment 43.
|
||||
The current-state document was verified as accurate by focused code inspection of API routes, orchestrator imports/calls, and cross-module traces for all passive classifiers. No corrections were required.
|
||||
Implementation status last checked against source: Experiment 43 + v0.61 apparatus (tsx helper, reconstruction-only seam, focused-deconstruction schema fix). Multi-investigation architecture verified at v0.60g2+ and structurally complete. The current-state document was verified as accurate by focused code inspection of API routes, orchestrator imports/calls, and cross-module traces for all passive classifiers. No corrections were required.
|
||||
|
||||
## Deployment Automation — LIVE PROVEN
|
||||
|
||||
A manual Jenkins deployment pipeline has been established, version-controlled, and LIVE PROVEN end-to-end.
|
||||
|
||||
- **Jenkinsfile** (root) — Declarative Pipeline with three stages: `Resolve` → `Deploy` → `Verify/result`.
|
||||
- **Parameter:** `GIT_REF` (string) — user-supplied Git ref (branch, tag, or SHA). Blank value fails clearly.
|
||||
- **Resolution:** Jenkins resolves the ref to an exact commit SHA via `git ls-remote origin` before deployment. The SHA is deployed immutably.
|
||||
- **Target host:** CT 112 (`confidence-engine`, 192.168.68.73) at `/opt/confidence-engine`.
|
||||
- **Docker image:** tagged `confidence-engine:<sha>`, built on CT 112, no registry required.
|
||||
- **Environment:** production values in `/opt/confidence-engine/deploy.env` on CT 112 (not in Git). `deploy.env.example` provided as reference.
|
||||
- **Health check:** polls `http://127.0.0.1:3000/api/health`; requires `{"healthy":true}` within 60s.
|
||||
- **Rollback:** one-step rollback to the previous container image on failure (if available). **IMPLEMENTED, NOT LIVE PROVEN.**
|
||||
- **Jenkins job:** configured and operational. First successful deployment recorded with explicit "DEPLOYMENT SUCCEEDED" output.
|
||||
- **Jenkins SCM branch** used to load the Jenkinsfile is conceptually separate from the `GIT_REF` chosen for deployment.
|
||||
|
||||
### Pipeline design notes
|
||||
|
||||
- Manual trigger only — no automatic webhook deploy.
|
||||
- No Docker registry introduced; SHA-tagged images retained on CT 112 for rollback support.
|
||||
- No product behaviour changed by deployment automation.
|
||||
|
||||
## 9. First-Time Unauthenticated Landing/Login Framing
|
||||
|
||||
- login page now explains product purpose and investigation flow to first-time visitors without prior explanation
|
||||
- user judgement/choice explicitly preserved ("It doesn't try to make the decision for you")
|
||||
- existing magic-link authentication unchanged
|
||||
- no broader onboarding/tutorial system introduced
|
||||
- verified via Playwright semantic locators (heading, text content, form controls) and responsive viewport
|
||||
|
||||
## 10. Post-v0.8 Methodology Learning
|
||||
|
||||
@@ -183,6 +298,4 @@ Since the handoff document was written, further learning has emerged from Return
|
||||
|
||||
### Current experimental context
|
||||
|
||||
The project has returned to relationship-discovery experimentation following Return-to-Origin methodology work. The durable methodology principles (A1–A12 above) should be loaded before continuing experiments. RTO.21 and RTO.22 have provided initial evidence supporting the meaning-over-dictionary approach; RTO.23 apparatus is under development.
|
||||
|
||||
For detailed current handoff state, read `docs/current-handoff.md` first.
|
||||
The durable methodology principles (A1–A12 above) should be loaded before continuing experiments. For detailed current operational state, read `docs/current-handoff.md` first.
|
||||
|
||||
+11
-10243
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,92 @@
|
||||
# Design Evolution — Canonical Archive Index
|
||||
|
||||
Design Evolution is historical provenance.
|
||||
|
||||
It records what was believed, tested, rejected, learned, and changed over time.
|
||||
|
||||
It is **NOT** the source of current product truth.
|
||||
|
||||
For current product state use:
|
||||
|
||||
- [`docs/current-handoff.md`](../current-handoff.md)
|
||||
- [`docs/current-project-state.md`](../current-project-state.md)
|
||||
|
||||
Historical material should be loaded only when a specific historical question requires it.
|
||||
This index enables progressive, on-demand loading of individual chapters.
|
||||
|
||||
---
|
||||
|
||||
## Chapter Index
|
||||
|
||||
All chapters are exact contiguous extracts from the former monolith at `docs/design-evolution-log.md`.
|
||||
|
||||
### Era 1 — Early Interface and Investigation Exploration
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 1 | [ch1/experiments-01-to-06-and-phases-1-to-4.md](../archive/experiments/vol-1-chapters/ch1/experiments-01-to-06-and-phases-1-to-4.md) | 1–352 | Phases 1–4 and Experiments 01–06. |
|
||||
| 2 | [ch2/early-reasoning-and-provenance-discovery.md](../archive/experiments/vol-1-chapters/ch2/early-reasoning-and-provenance-discovery.md) | 354–1122 | Experiments 07–20 and emerging-direction material. |
|
||||
| 3 | [ch3/phase-transition-and-emerging-directions.md](../archive/experiments/vol-1-chapters/ch3/phase-transition-and-emerging-directions.md) | 1124–1216 | Phase transition, graph as source of truth, Experiments 21–22. |
|
||||
|
||||
### Era 2 — State Assessment and Passive Classifier Work
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 4 | [ch4/passive-classifiers-and-explore-contract-validation.md](../archive/experiments/vol-1-chapters/ch4/passive-classifiers-and-explore-contract-validation.md) | 1218–1501 | Experiments 23–25B and closeout material. |
|
||||
| 5 | [ch5/context-inventory-routing-and-handoff-infrastructure.md](../archive/experiments/vol-1-chapters/ch5/context-inventory-routing-and-handoff-infrastructure.md) | 1502–2052 | Experiments 26–34 — context routing and handoff infrastructure. |
|
||||
| 6 | [ch6/handoff-validation-behavior-selection-and-assessor-audit.md](../archive/experiments/vol-1-chapters/ch6/handoff-validation-behavior-selection-and-assessor-audit.md) | 2053–2503 | Experiments 35–41 and assessor audit. |
|
||||
|
||||
### Era 3 — Documentation and Context-Routing Evolution
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 7 | [ch7/experiments-42-to-46.md](../archive/experiments/vol-1-chapters/ch7/experiments-42-to-46.md) | 2504–3069 | Experiments 42–46 — Clarify readiness, "Too Broad" boundaries. |
|
||||
| 8 | [ch8/experiments-47-to-51.md](../archive/experiments/vol-1-chapters/ch8/experiments-47-to-51.md) | 3070–3534 | Experiments 47–51 — coherence diagnostic, semantic decision-relevance. |
|
||||
| 9 | [ch9/experiments-52-to-52i.md](../archive/experiments/vol-1-chapters/ch9/experiments-52-to-52i.md) | 3535–4980 | Experiments 52–52I — semantic generalisation, ambiguity normalisation. |
|
||||
|
||||
### Era 4 — Behaviour and Coherence Experiments
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 10 | [ch10/experiments-54a-to-54k-provenance-audit.md](../archive/experiments/vol-1-chapters/ch10/experiments-54a-to-54k-provenance-audit.md) | 4981–6423 | Experiments 54A–54K — provenance audit chain, grounding with live inference. |
|
||||
| 11 | [ch11/experiments-54l-to-54q-grounding-and-evidence-stability.md](../archive/experiments/vol-1-chapters/ch11/experiments-54l-to-54q-grounding-and-evidence-stability.md) | 6425–7824 | Experiments 54L–54Q — grounding stability, evidence discrimination. |
|
||||
| 12 | [ch12/experiment-54r-clarification-requires-source.md](../archive/experiments/vol-1-chapters/ch12/experiment-54r-clarification-requires-source.md) | 7825–8012 | Experiment 54R — evidence vs. user-input disagreement classification. |
|
||||
|
||||
### Era 5 — Semantic Relevance Experiments
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 13 | [ch13/54S-54V-clarification-target-and-answer-resolution.md](../archive/experiments/vol-1-chapters/ch13/54S-54V-clarification-target-and-answer-resolution.md) | 8013–8635 | Experiments 54S–54V — clarification target identification, stability. |
|
||||
| 14 | [ch14/54W-54Z-clarification-chain-and-target-broadening.md](../archive/experiments/vol-1-chapters/ch14/54W-54Z-clarification-chain-and-target-broadening.md) | 8636–9294 | Experiments 54W–54Z — clarification chain integrity, target broadening. |
|
||||
|
||||
### Era 6 — Provenance and Grounding Experiments
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 15 | [ch15/55A-preserve-uncertainty-from-weak-clarification-answers.md](../archive/experiments/vol-1-chapters/ch15/55A-preserve-uncertainty-from-weak-clarification-answers.md) | 9296–9529 | Experiment 55A — preserving uncertainty from weak clarification answers. |
|
||||
| 16 | [ch16/55B-separate-answer-meaning-from-resolution-judgement.md](../archive/experiments/vol-1-chapters/ch16/55B-separate-answer-meaning-from-resolution-judgement.md) | 9530–9748 | Experiment 55B — separating answer meaning from resolution judgement. |
|
||||
| 17 | [ch17/55C-resolution-from-preserved-answer-meaning.md](../archive/experiments/vol-1-chapters/ch17/55C-resolution-from-preserved-answer-meaning.md) | 9749–9982 | Experiment 55C — two-stage resolution from preserved meaning. |
|
||||
|
||||
### Era 7 — Reasoning Refinement and Recent Product Provenance
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 18 | [ch18/55D-separate-stated-vs-inferred-meaning-through-v058-provenance.md](../archive/experiments/vol-1-chapters/ch18/55D-separate-stated-vs-inferred-meaning-through-v058-provenance.md) | 9983–10322 | Experiments 55D–55F and v0.51–v0.58 product provenance. |
|
||||
|
||||
### Era 8 — Initial Decomposition Hardening (frozen)
|
||||
|
||||
| Ch | Path | Source Lines | Summary |
|
||||
|----|------|-------------|---------|
|
||||
| 19 | [ch19/initial-decomposition-v0.61.md](../archive/experiments/vol-1-chapters/ch19/initial-decomposition-v0.61.md) | — | v0.61 initial decomposition: repeated-same-input stability, Qwen/Terra comparison, focused-deconstruction plumbing fix. Status: frozen for current MVP stage. |
|
||||
|
||||
---
|
||||
|
||||
## Structure Notes
|
||||
|
||||
- This index is the canonical routing entry point for Design Evolution provenance material.
|
||||
- Physical archive files remain at `docs/archive/experiments/vol-1-chapters/`. Their locations are verified and stable.
|
||||
- A later task may decide whether physical relocation is worthwhile. That decision is separate from this indexing step.
|
||||
|
||||
## Historical Integrity
|
||||
|
||||
All chapter files are exact historical records. They were not rewritten, modernised, or reconciled during extraction or migration.
|
||||
@@ -30,16 +30,14 @@ These are the documents Claude should normally read before continuing Confidence
|
||||
- **Why required:** Provides only the guidance that should influence work today, verified against current implementation. Reduces ambiguity about which principles are active versus aspirational.
|
||||
- **Size:** small
|
||||
|
||||
### docs/design-evolution-log.md (selected sections only)
|
||||
- **Purpose:** Chronological record of design decisions, experiments, and their conclusions.
|
||||
### docs/design-evolution/README.md (selected chapters only)
|
||||
- **Purpose:** Chronological record of design decisions, experiments, and their conclusions. Load the specific chapter via the archive index — do not load all chapters.
|
||||
- **Sections to read:**
|
||||
- Lines 1–90: Phases 1–4 overview (context for where we came from);
|
||||
- Lines 824–838: Experiment 16 "What did we learn?";
|
||||
- Lines 889–910: Experiment 17 summary;
|
||||
- Lines 1218–1520: Experiments 23 through 25B (current engine experiments);
|
||||
- Final ~40 lines of the file (Return-to-Work Note for each recent experiment).
|
||||
- **Why required:** The minimum history needed to understand what was learned in the active experiment chain (23–25B) and where the pause decision sits. Do not read experiments before v23 unless you need historical context.
|
||||
- **Size:** medium (targeted sections ≈ 400 lines of ~1,540 total)
|
||||
- ch4: Experiments 23 through 25B (current engine experiments);
|
||||
- ch8: Experiments 47–51;
|
||||
- The Return-to-Work Note entries in each relevant chapter for recent experiment closeout.
|
||||
- **Why required:** The minimum history needed to understand what was learned in the active experiment chain and where the pause decision sits. Do not read experiments before v23 unless you need historical context.
|
||||
- **Size:** varies by chapter (progressive loading recommended)
|
||||
|
||||
### docs/project-knowledge-inventory.md (this file — Section 1 only)
|
||||
- **Purpose:** Your own cross-reference for what to load next.
|
||||
@@ -57,7 +55,7 @@ These are the documents Claude should normally read before continuing Confidence
|
||||
- **Purpose:** Single return-to-work handoff carrying the latest stopping point in one short document. Classifies as the shortest current resume entry point.
|
||||
- **Sections to read:** All eight sections when returning after a break; specific sections only when the resuming session has partial context.
|
||||
- **Why required:** Replaces scattered current return notes with one obvious file. Does not duplicate full current-state or experiment history.
|
||||
- **Size:** short (~68 lines)
|
||||
- **Size:** medium (current operational state — loads first, then project-state and methodology)
|
||||
|
||||
### docs/03_Confidence_Engine_Language_Guide.md
|
||||
- **Purpose:** Exact language rules for user-facing output (voice, translations, what to avoid).
|
||||
@@ -195,7 +193,7 @@ These documents document the path from Phase 1 through Experiment 25B. Archiving
|
||||
After creating this inventory, I simulated a fresh-session context load using only:
|
||||
- `.claude/project-context.md` (entire file)
|
||||
- `.claude/architecture-guardrails.md` (entire file)
|
||||
- `docs/design-evolution-log.md` (lines 1–90 + lines 824–838 + lines 889–910 + lines 1218–1520)
|
||||
- `docs/design-evolution/README.md` ch4 (Experiments 23–25B), plus relevant chapter sections as needed
|
||||
- `docs/03_Confidence_Engine_Language_Guide.md` (entire file)
|
||||
|
||||
Five questions answered from this set:
|
||||
@@ -215,4 +213,4 @@ None. The five questions were answered accurately from the minimum context set.
|
||||
|
||||
## Return-to-Work Note
|
||||
|
||||
Engine experiments paused after Experiment 25B, which established scope-aware condition status classification — distinguishing direct evidence from relevant-but-different claims by checking subject, timeframe, and claim type. Present-state evidence does not settle future-feasibility conditions. The passive classifier layers remain isolated; no active integration yet. Knowledge-management experiments continue: five historical documents archived per Experiment 29; backlog info.md split in Experiment 31 into `docs/ui-mock-reference.md` (mock scenarios reference) and `docs/archive/deferred-ux-backlog.md` (deferred UX planning). Neither backlog item deleted or promoted. **Phase 2B context audit (2026-08-19):** ~76 historical experiment files moved to `docs/archive/experiments/` under eight subdirectories — preserved as evidence, removed from default context loading paths. No methodology or current operational paths changed. First file to inspect when resuming: `.claude/project-context.md`, then Experiments 23–25B in `docs/design-evolution-log.md` (lines 1218–1520).
|
||||
Engine experiments paused after Experiment 25B, which established scope-aware condition status classification — distinguishing direct evidence from relevant-but-different claims by checking subject, timeframe, and claim type. Present-state evidence does not settle future-feasibility conditions. The passive classifier layers remain isolated; no active integration yet. Knowledge-management experiments continue: five historical documents archived per Experiment 29; backlog info.md split in Experiment 31 into `docs/ui-mock-reference.md` (mock scenarios reference) and `docs/archive/deferred-ux-backlog.md` (deferred UX planning). Neither backlog item deleted or promoted. **Phase 2B context audit (2026-08-19):** ~76 historical experiment files moved to `docs/archive/experiments/` under eight subdirectories — preserved as evidence, removed from default context loading paths. No methodology or current operational paths changed. First file to inspect when resuming: `.claude/project-context.md`, then consult `docs/design-evolution/README.md` (ch4 for Experiments 23–25B).
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
### Always read
|
||||
- `docs/current-working-principles.md` §0 (Axiomatic Principles A1–A12 — durable methodology baseline)
|
||||
- `docs/Confidence_Engine_Return_to_Origin_Methodology_Context_2026-08-18.md` (methodology continuity, RTO evidence)
|
||||
- `docs/current-handoff.md` (current state, Git checkpoint, experiment log)
|
||||
- `docs/current-handoff.md` (current operational state, evidence semantics, settled boundaries)
|
||||
- `.claude/architecture-guardrails.md`
|
||||
- `docs/current-implementation-verification.md`
|
||||
|
||||
@@ -17,11 +17,11 @@
|
||||
### Then read only when relevant
|
||||
- the specific implementation file;
|
||||
- its focused tests;
|
||||
- the immediately previous experiment entry in `docs/design-evolution-log.md`;
|
||||
- the design evolution archive index at `docs/design-evolution/README.md` when provenance is needed;
|
||||
- the relevant contract or backlog entry.
|
||||
|
||||
### Do not load by default
|
||||
- full design-evolution history; archived documents; UI mock reference; unrelated architecture documents.
|
||||
- the full experiment history; archived documents; UI mock reference; unrelated architecture documents.
|
||||
|
||||
### Stop and ask or record a gap when
|
||||
- current documentation and source disagree;
|
||||
|
||||
+50
-14
@@ -4,7 +4,7 @@
|
||||
*/
|
||||
|
||||
import { getConfig } from "../lib/config.js";
|
||||
import { getProvider } from "../lib/llm/provider.js";
|
||||
import { getProvider, getProviderModelName } from "../lib/llm/provider.js";
|
||||
import {
|
||||
buildPrompt,
|
||||
PROMPT_VERSIONS,
|
||||
@@ -23,6 +23,8 @@ const MAX_SCENARIO_LENGTH = 10000;
|
||||
* @param {string} scenario - The scenario text to analyse
|
||||
* @param {object} [opts]
|
||||
* @param {"v0.1" | "v0.2"} [opts.promptVersion="v0.2"] - Prompt version to use
|
||||
* @param {{ generateReconstruction: Function }} [opts.reconstructionProvider] - Experiment-only reconstruction provider override
|
||||
* @param {string} [opts.reconstructionModelName] - Experiment-only model override for an injected provider
|
||||
* @returns {Promise<object>} Analysis result with diagnostics
|
||||
*/
|
||||
export async function analyseScenario(scenario, opts = {}) {
|
||||
@@ -50,13 +52,17 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
return buildErrorResponse("Invalid server configuration", startTime, "500");
|
||||
}
|
||||
|
||||
const { OLLAMA_BASE_URL: _ignored, OLLAMA_MODEL } = configResult.config;
|
||||
const promptVersion = opts.promptVersion || DEFAULT_PROMPT_VERSION;
|
||||
|
||||
// ── Build prompt ───────────────────────────────────
|
||||
// ── Build prompt (experiment seam via env var bridge) ──
|
||||
let promptObj;
|
||||
try {
|
||||
promptObj = await buildPrompt(trimmed, promptVersion);
|
||||
const experimentInstruction = process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION;
|
||||
const buildOpts = {};
|
||||
if (experimentInstruction) {
|
||||
buildOpts.experimentInstruction = experimentInstruction;
|
||||
}
|
||||
promptObj = await buildPrompt(trimmed, promptVersion, buildOpts);
|
||||
} catch (e) {
|
||||
return buildErrorResponse(
|
||||
`Failed to build prompt: ${e.message}`,
|
||||
@@ -65,17 +71,36 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
}
|
||||
|
||||
// ── Call provider ──────────────────────────────────
|
||||
const provider = getProvider();
|
||||
const provider = opts.reconstructionProvider ?? getProvider();
|
||||
const reconstructionModelName = opts.reconstructionModelName ?? getProviderModelName();
|
||||
let rawResponse;
|
||||
let providerApiPath;
|
||||
let providerExecution;
|
||||
try {
|
||||
rawResponse = await provider.generateReconstruction(
|
||||
const providerResult = await provider.generateReconstruction(
|
||||
promptObj.prompt,
|
||||
OLLAMA_MODEL,
|
||||
reconstructionModelName,
|
||||
);
|
||||
if (
|
||||
providerResult &&
|
||||
typeof providerResult === "object" &&
|
||||
"response" in providerResult &&
|
||||
"providerApiPath" in providerResult
|
||||
) {
|
||||
rawResponse = providerResult.response;
|
||||
providerApiPath = providerResult.providerApiPath;
|
||||
providerExecution = providerResult.providerExecution;
|
||||
} else {
|
||||
rawResponse = providerResult;
|
||||
}
|
||||
} catch (e) {
|
||||
return buildErrorResponse(
|
||||
e.message || "Provider error during analysis",
|
||||
Date.now() - startTime,
|
||||
"500",
|
||||
e.providerApiPath,
|
||||
e.providerExecution,
|
||||
e.code,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -86,7 +111,7 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
try {
|
||||
rawResponseStr = JSON.stringify(rawResponse);
|
||||
} catch {
|
||||
rawResponseStr = String(rawResponse).slice(0, 2000);
|
||||
rawResponseStr = String(rawResponse);
|
||||
}
|
||||
|
||||
const compatibility = normaliseAnalysisResponse(rawResponse);
|
||||
@@ -100,7 +125,7 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
if (resultV2.valid) {
|
||||
return buildSuccessResultV2(
|
||||
resultV2.data,
|
||||
OLLAMA_MODEL,
|
||||
reconstructionModelName,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
@@ -115,7 +140,7 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
if (resultV1.valid) {
|
||||
return buildSuccessResultV1(
|
||||
resultV1.data,
|
||||
OLLAMA_MODEL,
|
||||
reconstructionModelName,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
@@ -124,12 +149,14 @@ export async function analyseScenario(scenario, opts = {}) {
|
||||
|
||||
// ── Neither schema matched — partial failure ───────
|
||||
return buildPartialResult(
|
||||
rawResponseStr?.slice(0, 2000),
|
||||
rawResponseStr,
|
||||
resultV2.error ?? resultV1.error,
|
||||
OLLAMA_MODEL,
|
||||
reconstructionModelName,
|
||||
duration,
|
||||
promptVersion,
|
||||
compatibility,
|
||||
providerApiPath,
|
||||
providerExecution,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -149,7 +176,7 @@ function tryValidateAgainstSchema(data, schema) {
|
||||
|
||||
// ── Result builders ──────────────────────────────────
|
||||
|
||||
function buildErrorResponse(message, elapsed, statusCode = 500) {
|
||||
function buildErrorResponse(message, elapsed, statusCode = 500, providerApiPath, providerExecution, code) {
|
||||
return {
|
||||
success: false,
|
||||
error: message,
|
||||
@@ -159,6 +186,9 @@ function buildErrorResponse(message, elapsed, statusCode = 500) {
|
||||
rawResponse: null,
|
||||
promptVersion: null,
|
||||
statusCode,
|
||||
providerApiPath,
|
||||
providerExecution,
|
||||
code,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -211,8 +241,11 @@ function buildPartialResult(
|
||||
duration,
|
||||
version,
|
||||
compatibility,
|
||||
providerApiPath,
|
||||
providerExecution,
|
||||
) {
|
||||
let errors = [];
|
||||
const validationIssues = error?.issues ?? [];
|
||||
if (error && typeof error.flatten === "function") {
|
||||
errors = error.flatten().fieldErrors
|
||||
? Object.entries(error.flatten().fieldErrors).flatMap(([k, v]) => [
|
||||
@@ -228,13 +261,16 @@ function buildPartialResult(
|
||||
validationStatus: "invalid",
|
||||
modelName: model,
|
||||
responseDurationMs: duration,
|
||||
rawResponse: rawResp?.slice(0, 2000),
|
||||
rawResponse: rawResp,
|
||||
promptVersion: version,
|
||||
inputClassification: null,
|
||||
reconstruction: null,
|
||||
evidence: undefined,
|
||||
nextQuestion: undefined,
|
||||
errors,
|
||||
validationIssues,
|
||||
providerApiPath,
|
||||
providerExecution,
|
||||
...buildCompatibilityDiagnostics(compatibility),
|
||||
};
|
||||
}
|
||||
|
||||
+23
-1
@@ -5,7 +5,29 @@ const envSchema = z.object({
|
||||
OLLAMA_MODEL: z.string().min(1),
|
||||
});
|
||||
|
||||
const openAIExperimentEnvSchema = z.object({
|
||||
OPENAI_API_KEY: z.string().min(1),
|
||||
});
|
||||
|
||||
export const OPENAI_UI_JOURNEY_EXPERIMENT_PROVIDER = "openai";
|
||||
export const OPENAI_TERRA_MODEL = "gpt-5.6-terra";
|
||||
|
||||
export function isOpenAIUiJourneyExperiment() {
|
||||
return process.env.CONFIDENCE_ENGINE_EXPERIMENT_PROVIDER === OPENAI_UI_JOURNEY_EXPERIMENT_PROVIDER;
|
||||
}
|
||||
|
||||
export function getConfig() {
|
||||
if (isOpenAIUiJourneyExperiment()) {
|
||||
const parsed = openAIExperimentEnvSchema.safeParse({
|
||||
OPENAI_API_KEY: process.env.OPENAI_API_KEY,
|
||||
});
|
||||
if (!parsed.success) return { ok: false, error: parsed.error.flatten().fieldErrors };
|
||||
return {
|
||||
ok: true,
|
||||
config: { provider: OPENAI_UI_JOURNEY_EXPERIMENT_PROVIDER, modelName: OPENAI_TERRA_MODEL },
|
||||
};
|
||||
}
|
||||
|
||||
const parsed = envSchema.safeParse({
|
||||
OLLAMA_BASE_URL: process.env.OLLAMA_BASE_URL,
|
||||
OLLAMA_MODEL: process.env.OLLAMA_MODEL,
|
||||
@@ -15,7 +37,7 @@ export function getConfig() {
|
||||
return { ok: false, error: parsed.error.flatten().fieldErrors };
|
||||
}
|
||||
|
||||
return { ok: true, config: parsed.data };
|
||||
return { ok: true, config: { ...parsed.data, provider: "ollama", modelName: parsed.data.OLLAMA_MODEL } };
|
||||
}
|
||||
|
||||
export function assertConfig() {
|
||||
|
||||
+88
-10
@@ -2,6 +2,7 @@ import { describeGraph } from "./builder.js";
|
||||
import {
|
||||
countRemainingMaterialFactors,
|
||||
hasRemainingMaterialFactors,
|
||||
isDecisionClosureAuthorityQuestion,
|
||||
isUserConfirmationOfNoRemainingUncertainty,
|
||||
shouldCloseDecision,
|
||||
} from "./decision-sufficiency.js";
|
||||
@@ -3971,9 +3972,11 @@ export async function applyValidatedProposal({
|
||||
proposal,
|
||||
previousQuestion = null,
|
||||
answer = null,
|
||||
evidenceContext = null,
|
||||
provider = null,
|
||||
modelName = null,
|
||||
}) {
|
||||
const isLegacySingleTurn = !evidenceContext;
|
||||
const graphValidation = situationGraphSchema.safeParse(situationGraph);
|
||||
const proposalValidation = graphUpdateSchema.safeParse(proposal);
|
||||
|
||||
@@ -4030,10 +4033,38 @@ export async function applyValidatedProposal({
|
||||
);
|
||||
|
||||
// ── 60B.80 — normalise terminal parent closure without explicit confirmation
|
||||
// legacy: use raw answer for explicit confirmation detection.
|
||||
// episode: scan preserved paired turns for a QUALIFIED authority pair;
|
||||
// question must grant decision-level closure authority AND
|
||||
// answer must confirm no remaining uncertainty, SAME TURN.
|
||||
let ownershipAnswer = answer;
|
||||
if (!isLegacySingleTurn) {
|
||||
const epTurns = evidenceContext?.isCompletedEpisode
|
||||
? evidenceContext.episodeEvidence?.turns
|
||||
: [];
|
||||
if (Array.isArray(epTurns) && epTurns.length > 0) {
|
||||
// Scan all turns — paired check on SAME turn only.
|
||||
let explicitConfirmationAnswer = null;
|
||||
for (const turn of epTurns) {
|
||||
const tQuestion = turn?.question;
|
||||
const tAnswer = turn?.answer;
|
||||
if (
|
||||
typeof tQuestion === "string" &&
|
||||
typeof tAnswer === "string" &&
|
||||
isDecisionClosureAuthorityQuestion(tQuestion) &&
|
||||
isUserConfirmationOfNoRemainingUncertainty(tAnswer)
|
||||
) {
|
||||
explicitConfirmationAnswer = tAnswer;
|
||||
break;
|
||||
}
|
||||
}
|
||||
ownershipAnswer = explicitConfirmationAnswer ?? null;
|
||||
}
|
||||
}
|
||||
const ownershipChanges = reconcileDecisionClosureOwnership(
|
||||
situationGraph,
|
||||
reconciledProposal.proposal,
|
||||
answer,
|
||||
ownershipAnswer,
|
||||
);
|
||||
if (ownershipChanges.strippedResolvedIds.length > 0) {
|
||||
// Reconciler may have added selectedQuestion = null because the parent
|
||||
@@ -4058,7 +4089,16 @@ export async function applyValidatedProposal({
|
||||
);
|
||||
|
||||
if (!proposalGraphValidation.valid) {
|
||||
proposalCompatibilityErrors.push(...proposalGraphValidation.errors);
|
||||
const acceptsCompletedEpisodeNoOp = evidenceContext?.isCompletedEpisode === true;
|
||||
proposalCompatibilityErrors.push(
|
||||
...proposalGraphValidation.errors.filter(
|
||||
(error) =>
|
||||
!(
|
||||
acceptsCompletedEpisodeNoOp &&
|
||||
error === "Update contains no meaningful change"
|
||||
),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
const existingEdgeIds = new Set(situationGraph.edges.map((edge) => edge.id));
|
||||
@@ -4130,12 +4170,16 @@ export async function applyValidatedProposal({
|
||||
validatedProposal,
|
||||
);
|
||||
proposalCompatibilityErrors.push(...selectedQuestionValidation.errors);
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateAnswerMeaningCompatibilityWithRawAnswer({
|
||||
answer,
|
||||
proposal: validatedProposal,
|
||||
}),
|
||||
);
|
||||
// legacy only: raw-answer fidelity is a single-turn safeguard.
|
||||
// Episode evidence carries structured turns — no synthetic answer is constructed.
|
||||
if (isLegacySingleTurn) {
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateAnswerMeaningCompatibilityWithRawAnswer({
|
||||
answer,
|
||||
proposal: validatedProposal,
|
||||
}),
|
||||
);
|
||||
}
|
||||
proposalCompatibilityErrors.push(
|
||||
...validateAnswerMeaningAlignment(validatedProposal),
|
||||
);
|
||||
@@ -4165,10 +4209,44 @@ export async function applyValidatedProposal({
|
||||
? findNodeById(graphSnapshot, previousActiveUnknownNodeId)
|
||||
: null;
|
||||
const affectedNodeIds = buildAffectedNodeIds(graphSnapshot, proposalSnapshot);
|
||||
// For episode evidence: derive comparability override from the relevant Q/A pair,
|
||||
// not from turns[0]. Scan preserved paired turns for a matching comparability question
|
||||
// whose answer confirms comparability.
|
||||
let _derivedQuestion = isLegacySingleTurn ? previousQuestion : "";
|
||||
let _derivedAnswer = isLegacySingleTurn ? answer : "";
|
||||
if (!isLegacySingleTurn) {
|
||||
const epTurns = evidenceContext?.isCompletedEpisode
|
||||
? evidenceContext.episodeEvidence?.turns
|
||||
: [];
|
||||
if (Array.isArray(epTurns)) {
|
||||
// Scan all turns for a matching comparability Q/A pair.
|
||||
let found = false;
|
||||
for (const turn of epTurns) {
|
||||
const tq = turn?.question;
|
||||
const ta = turn?.answer;
|
||||
if (
|
||||
typeof tq === "string" &&
|
||||
typeof ta === "string" &&
|
||||
isComparabilityQuestion(tq) &&
|
||||
answerConfirmsComparability(ta)
|
||||
) {
|
||||
_derivedQuestion = tq;
|
||||
_derivedAnswer = ta;
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!found) {
|
||||
// No qualifying comparability pair — do not fabricate confirmation.
|
||||
_derivedQuestion = "";
|
||||
_derivedAnswer = "";
|
||||
}
|
||||
}
|
||||
}
|
||||
const reasoningResolution = deriveReasoningStateOverride({
|
||||
graph: graphSnapshot,
|
||||
previousQuestion,
|
||||
answer,
|
||||
previousQuestion: _derivedQuestion,
|
||||
answer: _derivedAnswer,
|
||||
resolvedUnknownNodeIds: validatedProposal.resolvedUnknownNodeIds,
|
||||
});
|
||||
|
||||
|
||||
+60
-9
@@ -22,6 +22,12 @@ export function buildInitialGraph(analysisData) {
|
||||
}
|
||||
|
||||
const nodeMap = new Map(); // label -> node
|
||||
const sourceIdToNodeId = new Map();
|
||||
|
||||
function mapSourceId(sourceId, node) {
|
||||
if (sourceId) sourceIdToNodeId.set(sourceId, node.id);
|
||||
return node;
|
||||
}
|
||||
|
||||
// ── Helper: register or get a node by label ────────────
|
||||
|
||||
@@ -97,14 +103,17 @@ export function buildInitialGraph(analysisData) {
|
||||
obs.confidence || "medium",
|
||||
);
|
||||
|
||||
if (obs.id) node.evidenceIds.push(obs.id);
|
||||
if (obs.id) {
|
||||
node.evidenceIds.push(obs.id);
|
||||
mapSourceId(obs.id, node);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Actors as states/nodes
|
||||
if (reconstruction.actors) {
|
||||
for (const actor of reconstruction.actors) {
|
||||
ensureNode(
|
||||
mapSourceId(actor.id, ensureNode(
|
||||
actor.description || actor.label,
|
||||
"observation",
|
||||
"supported",
|
||||
@@ -112,13 +121,13 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
null,
|
||||
actor.confidence || "medium",
|
||||
);
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
if (reconstruction.systemsOrObjects) {
|
||||
for (const sys of reconstruction.systemsOrObjects) {
|
||||
ensureNode(
|
||||
mapSourceId(sys.id, ensureNode(
|
||||
sys.description || sys.label,
|
||||
"metric",
|
||||
"known",
|
||||
@@ -126,7 +135,7 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
null,
|
||||
sys.confidence || "medium",
|
||||
);
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -142,6 +151,7 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
diff.confidence || "medium",
|
||||
);
|
||||
mapSourceId(diff.id, node);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -157,6 +167,7 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
c.confidence || "medium",
|
||||
);
|
||||
mapSourceId(c.id, node);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -173,6 +184,7 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
unk.confidence || "low",
|
||||
);
|
||||
mapSourceId(unk.id, node);
|
||||
unknownNodes.push(node);
|
||||
}
|
||||
}
|
||||
@@ -180,7 +192,7 @@ export function buildInitialGraph(analysisData) {
|
||||
// Plausible interpretations
|
||||
if (reconstruction.plausibleInterpretations) {
|
||||
for (const interp of reconstruction.plausibleInterpretations) {
|
||||
ensureNode(
|
||||
mapSourceId(interp.id, ensureNode(
|
||||
interp.description || interp.label,
|
||||
"assumption",
|
||||
"provisional",
|
||||
@@ -188,14 +200,32 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
null,
|
||||
interp.confidence || "low",
|
||||
);
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Unexplained transitions remain unresolved transition nodes
|
||||
if (reconstruction.unexplainedTransitions) {
|
||||
for (const trans of reconstruction.unexplainedTransitions) {
|
||||
const label =
|
||||
trans.description ||
|
||||
`${trans.entity ?? "Unexplained transition"}: ${trans.previousState ?? "unknown"} → ${trans.currentState ?? "unknown"}`;
|
||||
mapSourceId(trans.id, ensureNode(
|
||||
label,
|
||||
"transition",
|
||||
"unknown",
|
||||
trans.description || label,
|
||||
null,
|
||||
null,
|
||||
trans.confidence || "low",
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Known transitions
|
||||
if (reconstruction.knownTransitions) {
|
||||
for (const trans of reconstruction.knownTransitions) {
|
||||
ensureNode(
|
||||
mapSourceId(trans.id, ensureNode(
|
||||
`${trans.entity}: ${trans.previousState} → ${trans.currentState}`,
|
||||
"transition",
|
||||
trans.explanationStatus === "confirmed" ? "known" : "provisional",
|
||||
@@ -204,7 +234,7 @@ export function buildInitialGraph(analysisData) {
|
||||
null,
|
||||
null,
|
||||
trans.confidence || "medium",
|
||||
);
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -247,6 +277,27 @@ export function buildInitialGraph(analysisData) {
|
||||
}
|
||||
}
|
||||
|
||||
const edgeIds = new Set(edges.map((edge) => edge.id));
|
||||
for (const relationship of reconstruction.relationships ?? []) {
|
||||
const fromNodeId = sourceIdToNodeId.get(relationship.fromId);
|
||||
const toNodeId = sourceIdToNodeId.get(relationship.toId);
|
||||
const id = `e-rel-${relationship.id}`;
|
||||
|
||||
if (!fromNodeId || !toNodeId || edgeIds.has(id)) continue;
|
||||
|
||||
edges.push(
|
||||
situationEdgeSchema.parse({
|
||||
id,
|
||||
fromNodeId,
|
||||
toNodeId,
|
||||
relationship: relationship.relationship,
|
||||
confidence: relationship.confidence,
|
||||
description: relationship.description,
|
||||
}),
|
||||
);
|
||||
edgeIds.add(id);
|
||||
}
|
||||
|
||||
return { nodes: nodeArr, edges };
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,305 @@
|
||||
/**
|
||||
* Current Understanding synthesis seam — standalone domain function.
|
||||
*
|
||||
* Accepts: SituationGraph + all canonical Findings
|
||||
* Outputs: narrative-only { currentUnderstanding }
|
||||
*
|
||||
* Ownership:
|
||||
* ScenarioForm → WHEN synthesis occurs (untouched in this increment)
|
||||
* This module → HOW canonical state becomes narrative
|
||||
* Provider → generation (via dependency injection)
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import { getProvider } from "../llm/provider.js";
|
||||
|
||||
// ── Eligibility normalization ──────────────────────────────
|
||||
|
||||
/**
|
||||
* Filter findings to only eligible ones according to disposition contract:
|
||||
* null → eligible (accepted-by-default, provisional working interpretation)
|
||||
* "agree" → eligible (confirmed evidence)
|
||||
* "not_quite" → ineligible until corrected proposition is saved
|
||||
* "not_relevant" → ineligible (discounted from active reasoning)
|
||||
* rejected → excluded (already failed structural validation)
|
||||
*
|
||||
* Also excludes any Finding that has evaluation === "rejected".
|
||||
*/
|
||||
export function filterEligibleFindings(findings) {
|
||||
if (!findings || !Array.isArray(findings)) return [];
|
||||
|
||||
return findings.filter((f) => {
|
||||
// Structural validation exclusion (already evaluated upstream)
|
||||
if (f.evaluation === "rejected") return false;
|
||||
|
||||
const disposition = f.userDisposition;
|
||||
|
||||
// not_relevant → ineligible
|
||||
if (disposition === "not_relevant") return false;
|
||||
|
||||
// not_quite → ineligible until corrected proposition is saved
|
||||
if (disposition === "not_quite") return false;
|
||||
|
||||
// null, agree → eligible; anything else unexpected but let through
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
// ── Synthesis prompt construction ──────────────────────────
|
||||
|
||||
/**
|
||||
* Build the synthesis prompt from SituationGraph and eligible Findings.
|
||||
* The prompt instructs the model to produce one coherent Current Understanding narrative
|
||||
* from the provided inputs, without append semantics.
|
||||
*/
|
||||
|
||||
const KNOWN_SUPPORTED_STATUSES = new Set(["known", "supported"]);
|
||||
|
||||
function safeDesc(value) {
|
||||
return (value && typeof value === "string") ? value : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Evidence-authority projection for CU synthesis.
|
||||
*
|
||||
* Includes only:
|
||||
* - centralStatement (framing context — not independent evidence)
|
||||
* - known nodes (provider-active graph evidence)
|
||||
* - supported nodes (provider-active graph evidence)
|
||||
* - eligible Findings (from the findings parameter)
|
||||
*
|
||||
* Explicitly excludes from provider-active synthesis input:
|
||||
* provisional nodes, unknown nodes, resolved nodes, edges,
|
||||
* activeUnknownNodeId, resolvedNodeIds, currentSummary,
|
||||
* reasoningState, confidence/control metadata, dependency/control fields.
|
||||
*/
|
||||
function buildGraphEvidenceProjection(graphInfo) {
|
||||
const knownNodes = (graphInfo.nodes ?? [])
|
||||
.filter((n) => KNOWN_SUPPORTED_STATUSES.has(n.status))
|
||||
.map((n) => ({
|
||||
kind: n.kind ?? null,
|
||||
label: safeDesc(n.label),
|
||||
description: safeDesc(n.description),
|
||||
value: n.value ?? null,
|
||||
unit: n.unit ?? null,
|
||||
status: n.status ?? null,
|
||||
}));
|
||||
|
||||
const supportedNodes = (graphInfo.nodes ?? [])
|
||||
.filter((n) => KNOWN_SUPPORTED_STATUSES.has(n.status))
|
||||
.filter((n) => n.status !== "known")
|
||||
.map((n) => ({
|
||||
kind: n.kind ?? null,
|
||||
label: safeDesc(n.label),
|
||||
description: safeDesc(n.description),
|
||||
value: n.value ?? null,
|
||||
unit: n.unit ?? null,
|
||||
status: n.status ?? null,
|
||||
}));
|
||||
|
||||
const centralStatement = safeDesc(graphInfo.centralStatement) || "";
|
||||
|
||||
return { centralStatement, knownNodes, supportedNodes };
|
||||
}
|
||||
|
||||
export function buildSynthesisPrompt(situationGraph, findings) {
|
||||
// Structured evidence projection for the model
|
||||
const evidence = buildGraphEvidenceProjection(situationGraph);
|
||||
|
||||
// Eligible Finding sections (preserve existing agree vs null distinction)
|
||||
const findingsSections = [];
|
||||
|
||||
if (findings.length > 0) {
|
||||
const agreed = findings.filter((f) => f.userDisposition === "agree");
|
||||
const working = findings.filter(
|
||||
(f) => f.userDisposition === null
|
||||
);
|
||||
|
||||
if (agreed.length > 0) {
|
||||
findingsSections.push({
|
||||
label: "Confirmed Evidence",
|
||||
items: agreed.map((f) => ({
|
||||
proposition: f.proposition,
|
||||
id: f.id ?? null,
|
||||
})),
|
||||
});
|
||||
}
|
||||
|
||||
if (working.length > 0) {
|
||||
findingsSections.push({
|
||||
label: "Working Premises",
|
||||
items: working.map((f) => ({
|
||||
proposition: f.proposition,
|
||||
id: f.id ?? null,
|
||||
})),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Human-readable node display for the prompt
|
||||
function formatNodeSection(title, nodes) {
|
||||
if (!nodes || nodes.length === 0) return "";
|
||||
const items = nodes.map(
|
||||
(n) => ` ${title}: kind=${n.kind}, label="${n.label}", value=${n.value ? n.value + (n.unit ? " (" + n.unit + ")" : "") : null} — ${n.description ?? "(no description)"} [${n.status}]`
|
||||
);
|
||||
return "\n" + items.join("\n");
|
||||
}
|
||||
|
||||
const knownSection = formatNodeSection("Known", evidence.knownNodes);
|
||||
const supportedItem = formatNodeSection("Supported", evidence.supportedNodes);
|
||||
|
||||
const prompt = `You are producing a Current Understanding narrative from investigation evidence.
|
||||
|
||||
Situation Framing:
|
||||
${evidence.centralStatement ? " Central Statement: " + evidence.centralStatement : "(none)"}
|
||||
|
||||
Provider-Active Evidence:
|
||||
Known Facts:${knownSection}
|
||||
Supported Inferences:${supportedItem}
|
||||
|
||||
Eligible Findings:
|
||||
${findingsSections.length > 0
|
||||
? JSON.stringify(findingsSections, null, 2)
|
||||
: "(none)"}
|
||||
|
||||
Rules for this synthesis:
|
||||
1. Current Understanding describes only established or supported understanding from the evidence supplied here.
|
||||
2. Do not introduce or describe open questions, unresolved uncertainties, assumptions, provisional hypotheses, speculative explanations, or future investigation needs.
|
||||
3. Produce exactly ONE coherent narrative paragraph (or short multi-sentence paragraph) that represents the Current Understanding of the situation.
|
||||
4. Synthesize all provided evidence into a unified understanding — do not list or append findings. The result should read as a natural summary, not a bullet list.
|
||||
5. This is a FRESH synthesis from the complete set of inputs above. Do NOT treat any previous Current Understanding as input or authority. Do NOT append to prior summaries.
|
||||
6. Use only information present in the evidence above. The centralStatement is framing context, not independent evidence.
|
||||
7. Return ONLY a JSON object with this exact structure:
|
||||
{"currentUnderstanding": "your narrative here"}
|
||||
8. The currentUnderstanding value must be a non-empty string.
|
||||
|
||||
Return ONLY the JSON object. No markdown, no explanation, no preamble.`;
|
||||
|
||||
return prompt;
|
||||
}
|
||||
|
||||
// ── Synthesis response schema ──────────────────────────────
|
||||
|
||||
export const synthesisResponseSchema = z.object({
|
||||
currentUnderstanding: z
|
||||
.string()
|
||||
.min(1, "currentUnderstanding must be a non-empty string"),
|
||||
});
|
||||
|
||||
export const synthesisOutputSchema = {
|
||||
type: "object",
|
||||
properties: {
|
||||
currentUnderstanding: { type: "string" },
|
||||
},
|
||||
required: ["currentUnderstanding"],
|
||||
additionalProperties: false,
|
||||
};
|
||||
|
||||
/** Validate raw provider output against synthesis response schema */
|
||||
export function validateSynthesisResponse(raw) {
|
||||
if (raw == null) {
|
||||
return { valid: false, error: "Provider returned null/undefined" };
|
||||
}
|
||||
|
||||
let parsed;
|
||||
if (typeof raw === "string") {
|
||||
try {
|
||||
parsed = JSON.parse(raw);
|
||||
} catch {
|
||||
return { valid: false, error: "Provider output is not valid JSON" };
|
||||
}
|
||||
} else if (typeof raw === "object") {
|
||||
parsed = raw;
|
||||
} else {
|
||||
return { valid: false, error: "Provider output has unexpected type" };
|
||||
}
|
||||
|
||||
const result = synthesisResponseSchema.safeParse(parsed);
|
||||
if (!result.success) {
|
||||
const firstIssue = result.error.issues[0];
|
||||
return {
|
||||
valid: false,
|
||||
error: firstIssue?.message ?? "Invalid synthesis response",
|
||||
};
|
||||
}
|
||||
|
||||
return { valid: true, data: result.data };
|
||||
}
|
||||
|
||||
// ── Main domain function ───────────────────────────────────
|
||||
|
||||
/**
|
||||
* Standalone Current Understanding synthesis.
|
||||
*
|
||||
* @param {{ situationGraph: object, findings: Array<object> }} inputs
|
||||
* - situationGraph: the authoritative SituationGraph object
|
||||
* - findings: all canonical Findings (may be empty array)
|
||||
* @param {{ provider?: object }} [dependencies={}]
|
||||
* - provider: dependency-injected provider with generateReconstruction(prompt, modelName)
|
||||
* @returns {Promise<{ currentUnderstanding: string }>} validated narrative-only result
|
||||
*/
|
||||
export async function synthesizeCurrentUnderstanding(
|
||||
{ situationGraph, findings },
|
||||
dependencies = {}
|
||||
) {
|
||||
// 1. Input validation
|
||||
if (!situationGraph || typeof situationGraph !== "object") {
|
||||
const err = new Error("Invalid input: situationGraph is required and must be an object");
|
||||
err.statusCode = 400;
|
||||
throw err;
|
||||
}
|
||||
|
||||
if (findings != null && !Array.isArray(findings)) {
|
||||
const err = new Error("Invalid input: findings must be an array or null/undefined");
|
||||
err.statusCode = 400;
|
||||
throw err;
|
||||
}
|
||||
|
||||
// Normalize empty findings to empty array
|
||||
const allFindings = findings ?? [];
|
||||
|
||||
// 2. Eligibility normalization (domain seam responsibility)
|
||||
const eligibleFindings = filterEligibleFindings(allFindings);
|
||||
|
||||
// 3. Build synthesis prompt (uses evidence-authority projection, not raw graph)
|
||||
const prompt = buildSynthesisPrompt(situationGraph, eligibleFindings);
|
||||
|
||||
// 4. Resolve provider — DI fallback to configured default
|
||||
const provider = dependencies.provider ?? getProvider();
|
||||
if (!provider || typeof provider.generateReconstruction !== "function") {
|
||||
throw new Error("Invalid dependency: provider must have generateReconstruction");
|
||||
}
|
||||
|
||||
// Resolve configured model: explicit dep > config dep (assertConfig) > process.env > null
|
||||
let modelName = dependencies.modelName;
|
||||
if (modelName == null && dependencies.config?.OLLAMA_MODEL != null) {
|
||||
modelName = dependencies.config.OLLAMA_MODEL;
|
||||
}
|
||||
if (modelName == null) {
|
||||
modelName = process.env.OLLAMA_MODEL ?? null;
|
||||
}
|
||||
|
||||
let providerResult;
|
||||
try {
|
||||
providerResult = await provider.generateReconstruction(
|
||||
prompt,
|
||||
modelName,
|
||||
synthesisOutputSchema,
|
||||
);
|
||||
} catch (error) {
|
||||
const err = new Error(error.message ?? "Synthesis provider call failed");
|
||||
err.statusCode = 502;
|
||||
throw err;
|
||||
}
|
||||
|
||||
// 5. Validate response
|
||||
const validated = validateSynthesisResponse(providerResult.response);
|
||||
if (!validated.valid) {
|
||||
const err = new Error(`Synthesis validation failed: ${validated.error}`);
|
||||
err.statusCode = 502;
|
||||
throw err;
|
||||
}
|
||||
|
||||
// 6. Return narrative-only result — no graph/Finding mutation
|
||||
return { currentUnderstanding: validated.data.currentUnderstanding };
|
||||
}
|
||||
@@ -65,6 +65,45 @@ export function isUserConfirmationOfNoRemainingUncertainty(answer) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// ── Pure: decision-closure authority question predicate ───────────
|
||||
|
||||
/**
|
||||
* Determines whether a question explicitly asks the user whether any
|
||||
* material uncertainty remains at the parent-decision level — i.e. the
|
||||
* question grants closure authority when answered with confirmation.
|
||||
*
|
||||
* Bounded to the decision_threshold sufficiency-confirmation semantic family.
|
||||
* Focused questions (supplier, budget, etc.) do NOT match even if they
|
||||
* receive a confirming answer.
|
||||
*
|
||||
* Returns true only for:
|
||||
* - "Is there anything else material that could change which option is better?"
|
||||
* (the canonical product template)
|
||||
* - Questions using the closure-family pattern:
|
||||
* <remaining-uncertainty-phrase> + <closure-action-phrase>
|
||||
* (e.g. "remaining material uncertainty preventing this decision from being closed")
|
||||
*/
|
||||
export function isDecisionClosureAuthorityQuestion(question) {
|
||||
const text = String(question || "");
|
||||
const lower = text.toLowerCase();
|
||||
|
||||
// Direct match for the canonical decision_threshold sufficiency confirmation template
|
||||
if (/change which option (is |was )?better/i.test(lower)) return true;
|
||||
|
||||
// Closure-family pattern: "remaining-uncertainty" + "closure-action" conjunction
|
||||
const hasRemainingUncertainty = /any(?:thing)?\s+(?:else\s+)?(?:material\s+)?uncertain(?:ty|ces)/i.test(lower);
|
||||
if (!hasRemainingUncertainty) return false;
|
||||
|
||||
const hasClosureAction = /\b(closure|close[sd]?|resolve[d]?)\b/i.test(lower);
|
||||
if (hasClosureAction) return true;
|
||||
|
||||
// "remaining material uncertainty" + decision-referencing words
|
||||
const hasDecisionRef = /(?:decision|deciding|disposition)\b/i.test(lower);
|
||||
if (hasDecisionRef) return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
// ── Pure: unresolved predicate ─────────────────────────────────
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
/**
|
||||
* Episode preparation — deterministically prepare a completed focused episode
|
||||
* for future authoritative graph reasoning.
|
||||
*
|
||||
* Pure domain transformation:
|
||||
* SituationGraph + targetNodeId + Contributions + Findings → preparedEpisode
|
||||
*
|
||||
* Purity contract:
|
||||
* - No mutation of inputs (situationGraph, contributions, findings)
|
||||
* - No storage I/O
|
||||
* - No fetch / provider / model calls
|
||||
* - No GraphUpdateProposal generation
|
||||
* - No reasoningState modification
|
||||
*/
|
||||
|
||||
// ── Eligibility constants ──────────────────────────────────
|
||||
|
||||
const ELIGIBLE_DISPOSITIONS = new Set([null, "agree"]);
|
||||
|
||||
const EXCLUDED_DISPOSITIONS = new Set(["not_relevant", "not_quite"]);
|
||||
|
||||
// ── Public API ─────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Prepare a completed focused episode for deterministic graph reasoning.
|
||||
*
|
||||
* @param {Object} params
|
||||
* @param {Object} params.situationGraph - Authoritative SituationGraph (read-only)
|
||||
* @param {string} params.targetNodeId - The completed target node ID
|
||||
* @param {Array<Object>} params.contributions - All focused Contributions
|
||||
* @param {Array<Object>} params.findings - All canonical Findings
|
||||
* @returns {Object} preparedEpisode
|
||||
*/
|
||||
export function prepareCompletedEpisode({ situationGraph, targetNodeId, contributions, findings }) {
|
||||
// ── 1. Scope contributions to the completed target ───────
|
||||
|
||||
const scopedContributions = [];
|
||||
|
||||
for (const contrib of contributions) {
|
||||
if (contrib?.targetNodeId === targetNodeId) {
|
||||
scopedContributions.push(contrib);
|
||||
}
|
||||
}
|
||||
|
||||
// ── 2. Order deterministically by sequence ───────────────
|
||||
|
||||
const orderedContributions = scopedContributions.slice().sort((a, b) => {
|
||||
const aSeq = a?.sequence != null ? a.sequence : Infinity;
|
||||
const bSeq = b?.sequence != null ? b.sequence : Infinity;
|
||||
return aSeq - bSeq;
|
||||
});
|
||||
|
||||
// ── 3. Build contribution ID lookup for provenance chain ─
|
||||
|
||||
const contribIds = new Set(orderedContributions.map((c) => c.id));
|
||||
|
||||
// ── 4. Prepare turns (ordered Q/A pairs) ────────────────
|
||||
|
||||
const turns = orderedContributions.map((contrib, idx) => ({
|
||||
contributionId: contrib.id ?? null,
|
||||
sequence: contrib.sequence != null ? contrib.sequence : idx + 1,
|
||||
question: contrib.question ?? "",
|
||||
answer: contrib.answer ?? "",
|
||||
}));
|
||||
|
||||
// ── 5. Bucket Findings by eligibility ───────────────────
|
||||
|
||||
const eligibleCanonicalFindings = [];
|
||||
const excludedFindingProvenance = [];
|
||||
|
||||
for (const finding of findings || []) {
|
||||
if (!finding?.contributionId) continue;
|
||||
|
||||
// Membership authority: only Findings whose contributionId references a scoped Contribution
|
||||
if (!contribIds.has(finding.contributionId)) continue;
|
||||
|
||||
const disposition = finding.userDisposition ?? null;
|
||||
|
||||
const bucketedFinding = {
|
||||
findingId: finding.id ?? null,
|
||||
contributionId: finding.contributionId,
|
||||
proposition: finding.proposition ?? "",
|
||||
sourceObservation: finding.sourceObservation ?? "",
|
||||
disposition,
|
||||
};
|
||||
|
||||
if (ELIGIBLE_DISPOSITIONS.has(disposition)) {
|
||||
eligibleCanonicalFindings.push({
|
||||
...bucketedFinding,
|
||||
endorsement: disposition, // null or "agree"
|
||||
});
|
||||
} else if (EXCLUDED_DISPOSITIONS.has(disposition)) {
|
||||
excludedFindingProvenance.push(bucketedFinding);
|
||||
}
|
||||
}
|
||||
|
||||
// ── 6. Compose prepared episode ─────────────────────────
|
||||
|
||||
return {
|
||||
situationGraph: situationGraph ?? null,
|
||||
targetNodeId,
|
||||
turns,
|
||||
eligibleCanonicalFindings,
|
||||
excludedFindingProvenance,
|
||||
};
|
||||
}
|
||||
@@ -9,6 +9,32 @@ const FOCUSED_ANSWER_SCHEMA_FIELDS = [
|
||||
"possibleFollowUpQuestions",
|
||||
];
|
||||
|
||||
export const focusedDeconstructJsonSchema = {
|
||||
type: "object",
|
||||
properties: {
|
||||
targetNodeId: { type: "string" },
|
||||
observations: { type: "array", items: { type: "string" } },
|
||||
uncertainties: { type: "array", items: { type: "string" } },
|
||||
assumptions: { type: "array", items: { type: "string" } },
|
||||
relationships: {
|
||||
type: "array",
|
||||
items: {
|
||||
type: "object",
|
||||
properties: {
|
||||
from: { type: "string" },
|
||||
to: { type: "string" },
|
||||
type: { type: "string" },
|
||||
},
|
||||
required: ["from", "to", "type"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
},
|
||||
possibleFollowUpQuestions: { type: "array", items: { type: "string" } },
|
||||
},
|
||||
required: FOCUSED_ANSWER_SCHEMA_FIELDS,
|
||||
additionalProperties: false,
|
||||
};
|
||||
|
||||
const FORBIDDEN_GRAPH_MUTATION_FIELDS = [
|
||||
"addedNodes",
|
||||
"updatedNodes",
|
||||
@@ -101,7 +127,7 @@ Field rules (semantic contract):
|
||||
- assumptions: what unstated proposition does the user's answer itself rely upon for it to make sense? Include only when such a proposition is genuinely attributable to the user's reasoning. The boundary is narrow: attribute only propositions that the user's answer would cease to make sense if they were false. Do NOT import plausible interpretations from the wider investigation context, scenario framing, domain relevance, strategic implications, or model-generated analysis into this field — those belong in uncertainties, relationships (where permitted), or possibleFollowUpQuestions. Do NOT connect a factual statement the user makes to a broader capability or constraint concept unless the user explicitly links them. Example: answering "I only have bank account access" to a question about delegation constraints does NOT assume that "delegation feasibility is contingent upon banking access" — it only states a fact about access, and connecting that fact to delegation feasibility is your own scenario-level inference, not a user-held assumption. If the user's answer does not contain or rely upon an identifiable assumption, return assumptions: []. Do NOT require verbatim copying from the user's answer; paraphrasing is allowed only when the reasoning genuinely relies on it.
|
||||
|
||||
Answer-dependence test: Only attribute an assumption if the user's answer needs that proposition to make sense. If the proposition could be false and the user's answer would still make complete sense, do not attribute it. One observed success in a single concrete example does NOT by itself establish a general rule about competence, readiness, training, safety, transferability, or similar tasks across other work. Do not generalise from one successful example into a broader capability/readiness rule unless the user explicitly or implicitly relies on that broader proposition.
|
||||
- relationships: must connect two distinct propositions that the user's answer itself links. Do not create a relationship by merely restating, reformatting, or relabelling an observation. Co-mentioned facts do not themselves create a relationship. Tentative, speculative, or conditional language must not be promoted into an established relationship. If the answer does not directly establish a relationship, return relationships: [].
|
||||
- relationships: each non-empty item must be { "from": "first proposition", "to": "second distinct proposition", "type": "concise free-text relationship label" }. It must connect two distinct propositions that the user's answer itself links. Do not create a relationship by merely restating, reformatting, or relabelling an observation. Co-mentioned facts do not themselves create a relationship. Tentative, speculative, or conditional language must not be promoted into an established relationship. If the answer does not directly establish a relationship, return relationships: [].
|
||||
- possibleFollowUpQuestions: must be a JSON array containing exactly one string — your single best follow-up question. Example shape: ["one question"]. This question must directly investigate the single uncertainty returned in uncertainties (uncertainties[0] → possibleFollowUpQuestions[0]): one unresolved proposition mapped to one question designed to clarify it. The question must not introduce a second unresolved issue, must not broaden beyond the uncertainty it is meant to resolve, and must not contain more than one investigative step. Do not provide alternatives, a roadmap, or questions that belong after this one has been answered. A later question must be generated only after the current question has been answered and deconstructed. Do not ask about consequences, expansion, requirements, interventions, or other branches until the immediate unresolved relationship has been clarified. Those may become later questions after new evidence is obtained. Ask only what the Engine has earned the right to ask now. Each epistemic step waits its turn — do not combine steps that should happen in sequence across multiple turns: one question that investigates one thing only, never a bundle of future reasoning joined together. Before formulating, check whether the question tests a proposition against the current epistemic state: if an explanation, deficit, dependency, cause, intervention, recommendation, or solution has not been established by prior evidence, phrase the question so it tests whether that proposition is true rather than assuming it — verify the unresolved fact before seeking remedy. Prefer questions that identify what remains unknown, distinguish competing explanations, test whether a suspected factor actually matters, clarify scope, or identify what evidence would change the investigation. Do not jump to implementation details unless the answer has already established that intervention as the relevant next issue. Use ordinary language that a capable person with no specialist vocabulary can understand immediately. If the question needs abstract phrases, management jargon, specialist terminology, or several concepts joined together to express it, break the reasoning down again before returning it. Simple wording of an over-composed idea is still a failure: first ask "what is the smallest thing we actually do not know yet?" then express that one thing simply.
|
||||
- cross-field ownership: preserve who or what owns each proposition. When a statement expresses the user's comfort, willingness, threshold, belief, uncertainty, preference, or judgement, keep it attached to that stance — do not elevate it into an objective requirement, capability fact, or situational constraint.
|
||||
|
||||
@@ -136,6 +162,30 @@ export function validateFocusedDeconstructSchema(result) {
|
||||
}
|
||||
}
|
||||
|
||||
if (!Array.isArray(result.relationships)) {
|
||||
errors.push("relationships must be an array");
|
||||
} else {
|
||||
result.relationships.forEach((relationship, index) => {
|
||||
if (!relationship || typeof relationship !== "object" || Array.isArray(relationship)) {
|
||||
errors.push(`relationships[${index}] must be an object`);
|
||||
return;
|
||||
}
|
||||
const keys = Object.keys(relationship);
|
||||
for (const field of ["from", "to", "type"]) {
|
||||
if (!(field in relationship)) {
|
||||
errors.push(`relationships[${index}] missing required field: ${field}`);
|
||||
} else if (typeof relationship[field] !== "string" || relationship[field].trim().length === 0) {
|
||||
errors.push(`relationships[${index}].${field} must be a non-empty string`);
|
||||
}
|
||||
}
|
||||
for (const field of keys) {
|
||||
if (!["from", "to", "type"].includes(field)) {
|
||||
errors.push(`relationships[${index}] contains unknown field: ${field}`);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
return errors;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
/**
|
||||
* Investigation Overview synthesis seam — standalone domain function.
|
||||
*
|
||||
* Purpose: produce a structurally distinct two-part overview that keeps
|
||||
* (A) evidence-backed understanding and
|
||||
* (B) remaining plausible interpretations
|
||||
* epistemically separate.
|
||||
*
|
||||
* Input contract:
|
||||
* { situationGraph, findings, plausibleInterpretations }
|
||||
*
|
||||
* Output contract:
|
||||
* {
|
||||
* "understanding": "evidence-backed synthesis",
|
||||
* "plausibleInterpretations": "qualified synthesis of remaining interpretations"
|
||||
* }
|
||||
*
|
||||
* Does NOT produce: recommendation, decision, confidence score, next action, priority, readiness.
|
||||
*/
|
||||
|
||||
import { z } from "zod";
|
||||
import { getProvider } from "../llm/provider.js";
|
||||
import { filterEligibleFindings } from "./current-understanding-synthesis.js";
|
||||
|
||||
export { filterEligibleFindings };
|
||||
|
||||
// ── Overview-specific output validation schema ────────────────
|
||||
|
||||
const overviewResponseSchema = z.object({
|
||||
understanding: z.string().min(1),
|
||||
plausibleInterpretations: z.string().min(1),
|
||||
});
|
||||
|
||||
const FORBIDDEN_FIELD_NAMES = new Set([
|
||||
"recommendation",
|
||||
"decision",
|
||||
"confidenceScore",
|
||||
"nextAction",
|
||||
"priority",
|
||||
"readiness",
|
||||
]);
|
||||
|
||||
/**
|
||||
* Validate that the raw overview response has exactly two semantic fields:
|
||||
* - understanding (string)
|
||||
* - plausibleInterpretations (string)
|
||||
* and no decision/recommendation/priority/readiness/next-action leakage.
|
||||
*/
|
||||
export function validateOverviewResponse(raw) {
|
||||
if (raw == null) {
|
||||
return { valid: false, reason: "Provider returned null/undefined" };
|
||||
}
|
||||
|
||||
let parsed;
|
||||
if (typeof raw === "string") {
|
||||
try {
|
||||
parsed = JSON.parse(raw);
|
||||
} catch {
|
||||
return { valid: false, reason: "Provider output is not valid JSON" };
|
||||
}
|
||||
} else if (typeof raw === "object") {
|
||||
parsed = raw;
|
||||
} else {
|
||||
return { valid: false, reason: "Provider output has unexpected type" };
|
||||
}
|
||||
|
||||
// Reject any forbidden epistemic fields
|
||||
for (const key of Object.keys(parsed)) {
|
||||
if (FORBIDDEN_FIELD_NAMES.has(key)) {
|
||||
return { valid: false, reason: `forbidden_field: ${key}` };
|
||||
}
|
||||
}
|
||||
|
||||
const result = overviewResponseSchema.safeParse(parsed);
|
||||
if (!result.success) {
|
||||
return { valid: false, reason: "Missing or invalid required fields" };
|
||||
}
|
||||
|
||||
return { valid: true, data: result.data };
|
||||
}
|
||||
|
||||
// ── Overview-specific prompt construction ─────────────────────
|
||||
|
||||
const KNOWN_SUPPORTED_STATUSES = new Set(["known", "supported"]);
|
||||
|
||||
function safeDesc(value) {
|
||||
return (value && typeof value === "string") ? value : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the overview synthesis prompt from SituationGraph, eligible Findings,
|
||||
* and plausible interpretations.
|
||||
*
|
||||
* Produces three evidence sections:
|
||||
* 1. Evidence-backed understanding inputs (known + supported nodes + eligible Findings)
|
||||
* 2. Plausible interpretations inputs (kept separate from evidence)
|
||||
* 3. Epistemic boundary rules
|
||||
*/
|
||||
export function buildOverviewSynthesisPrompt(situationGraph, findings, plausibleInterpretations) {
|
||||
// Evidence-backed projection: reuse the existing known+supported logic
|
||||
const knownNodes = (situationGraph.nodes ?? [])
|
||||
.filter((n) => KNOWN_SUPPORTED_STATUSES.has(n.status))
|
||||
.filter((n) => n.status === "known")
|
||||
.map((n) => ({
|
||||
kind: n.kind ?? null,
|
||||
label: safeDesc(n.label),
|
||||
description: safeDesc(n.description),
|
||||
value: n.value ?? null,
|
||||
unit: n.unit ?? null,
|
||||
status: n.status ?? null,
|
||||
}));
|
||||
|
||||
const supportedNodes = (situationGraph.nodes ?? [])
|
||||
.filter((n) => KNOWN_SUPPORTED_STATUSES.has(n.status))
|
||||
.filter((n) => n.status !== "known")
|
||||
.map((n) => ({
|
||||
kind: n.kind ?? null,
|
||||
label: safeDesc(n.label),
|
||||
description: safeDesc(n.description),
|
||||
value: n.value ?? null,
|
||||
unit: n.unit ?? null,
|
||||
status: n.status ?? null,
|
||||
}));
|
||||
|
||||
const centralStatement = safeDesc(situationGraph.centralStatement) || "";
|
||||
|
||||
// Eligible Findings (reuse existing filter)
|
||||
const eligibleFindings = filterEligibleFindings(findings);
|
||||
const agreedFindings = eligibleFindings.filter((f) => f.userDisposition === "agree");
|
||||
const workingFindings = eligibleFindings.filter((f) => f.userDisposition === null);
|
||||
|
||||
// Format nodes for prompt display
|
||||
function formatNodes(nodes, title) {
|
||||
if (!nodes || nodes.length === 0) return "";
|
||||
return nodes.map(
|
||||
(n) => ` ${title}: kind=${n.kind}, label="${n.label}", value=${n.value ? n.value + (n.unit ? " (" + n.unit + ")" : "") : null} — ${n.description ?? "(no description)"} [${n.status}]`
|
||||
).join("\n");
|
||||
}
|
||||
|
||||
const knownSection = formatNodes(knownNodes, "Known");
|
||||
const supportedSection = formatNodes(supportedNodes, "Supported");
|
||||
|
||||
// Format plausible interpretations (kept separate from evidence)
|
||||
const interpSections = [];
|
||||
if (plausibleInterpretations && Array.isArray(plausibleInterpretations)) {
|
||||
for (const interp of plausibleInterpretations) {
|
||||
interpSections.push({
|
||||
id: interp.id ?? null,
|
||||
description: safeDesc(interp.description) || "Unlabelled interpretation",
|
||||
confidence: interp.confidence ?? "unknown",
|
||||
supportingEvidenceIds: interp.supportingEvidenceIds ?? [],
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const findingsSections = [];
|
||||
if (agreedFindings.length > 0) {
|
||||
findingsSections.push({
|
||||
label: "Confirmed Evidence",
|
||||
items: agreedFindings.map((f) => ({ proposition: f.proposition, id: f.id ?? null })),
|
||||
});
|
||||
}
|
||||
if (workingFindings.length > 0) {
|
||||
findingsSections.push({
|
||||
label: "Working Premises",
|
||||
items: workingFindings.map((f) => ({ proposition: f.proposition, id: f.id ?? null })),
|
||||
});
|
||||
}
|
||||
|
||||
const prompt = `You are producing an investigation overview with two structurally distinct sections.
|
||||
|
||||
Situation Framing:
|
||||
${centralStatement ? " Central Statement: " + centralStatement : "(none)"}
|
||||
|
||||
=== SECTION A INPUTS — Evidence-Backed Understanding ===
|
||||
|
||||
Provider-Active Evidence:
|
||||
Known Facts:${knownSection || " (none)"}
|
||||
Supported Inferences:${supportedSection || " (none)"}
|
||||
|
||||
Eligible Findings:
|
||||
${findingsSections.length > 0 ? JSON.stringify(findingsSections, null, 2) : "(none)"}
|
||||
|
||||
=== SECTION B INPUTS — Plausible Interpretations (NOT evidence-backed) ===
|
||||
|
||||
Plausible Interpretations:
|
||||
${interpSections.length > 0 ? JSON.stringify(interpSections, null, 2) : "(none)"}
|
||||
|
||||
=== EPISTEMIC BOUNDARY RULES ===
|
||||
|
||||
1. Section A (understanding) MUST contain only established or supported understanding from the evidence in SECTION A INPUTS above.
|
||||
2. Section A MUST NOT include open questions, unresolved uncertainties, assumptions, provisional hypotheses, speculative explanations, or future investigation needs.
|
||||
3. Plausible interpretations from SECTION B INPUTS MUST remain explicitly qualified as interpretations — never promoted into Section A (understanding).
|
||||
4. Plausible interpretations must not be presented as established evidence or confirmed facts.
|
||||
5. Produce exactly ONE coherent narrative paragraph for "understanding" from SECTION A inputs only.
|
||||
6. Produce exactly ONE coherent narrative paragraph for "plausibleInterpretations" from SECTION B inputs only. Each interpretation should be clearly qualified as an interpretation.
|
||||
7. Do NOT introduce any new facts not present in the supplied evidence.
|
||||
8. Return ONLY a JSON object with this exact structure:
|
||||
{"understanding": "...", "plausibleInterpretations": "..."}
|
||||
9. Neither field may contain recommendations, decisions, confidence scores, next actions, priorities, or readiness assessments.
|
||||
10. This is a FRESH synthesis — do NOT treat any previous overview or Current Understanding as input.
|
||||
|
||||
Both fields must be non-empty strings.`;
|
||||
|
||||
return prompt;
|
||||
}
|
||||
|
||||
// ── Overview domain function ──────────────────────────────────
|
||||
|
||||
/**
|
||||
* Synthesize an investigation overview with structurally distinct sections:
|
||||
* - understanding: evidence-backed synthesis
|
||||
* - plausibleInterpretations: qualified remaining interpretations
|
||||
*
|
||||
* @param {{ situationGraph, findings, plausibleInterpretations }} params
|
||||
* @param {{ provider, modelName }} deps
|
||||
* @returns {Promise<{ understanding: string, plausibleInterpretations: string }>}
|
||||
*/
|
||||
export async function synthesizeInvestigationOverview(params, deps) {
|
||||
const { situationGraph, findings = [], plausibleInterpretations = [] } = params;
|
||||
|
||||
if (!situationGraph || typeof situationGraph !== "object") {
|
||||
throw new Error("situationGraph is required");
|
||||
}
|
||||
if (!Array.isArray(findings)) {
|
||||
throw new Error("findings must be an array");
|
||||
}
|
||||
if (!Array.isArray(plausibleInterpretations)) {
|
||||
throw new Error("plausibleInterpretations must be an array");
|
||||
}
|
||||
|
||||
// Provider acquisition (reuse existing pattern)
|
||||
const provider = deps?.provider ?? getProvider();
|
||||
const modelName = deps?.modelName ?? process.env.OLLAMA_MODEL;
|
||||
|
||||
if (!provider || typeof provider.generateReconstruction !== "function") {
|
||||
throw new Error("Invalid dependency: provider must have generateReconstruction");
|
||||
}
|
||||
|
||||
// Build overview-specific prompt (keeps evidence/interpretation separate)
|
||||
const prompt = buildOverviewSynthesisPrompt(situationGraph, findings, plausibleInterpretations);
|
||||
|
||||
let rawResponse;
|
||||
try {
|
||||
rawResponse = await provider.generateReconstruction(
|
||||
prompt,
|
||||
modelName,
|
||||
);
|
||||
} catch (err) {
|
||||
throw new Error(`Overview synthesis provider call failed: ${err.message}`);
|
||||
}
|
||||
|
||||
// Validate against overview-specific contract
|
||||
const validated = validateOverviewResponse(rawResponse);
|
||||
if (!validated.valid) {
|
||||
throw new Error(`Overview synthesis validation failed: ${validated.reason}`);
|
||||
}
|
||||
|
||||
return validated.data;
|
||||
}
|
||||
+158
-4
@@ -5,7 +5,7 @@
|
||||
|
||||
import { analyseScenario } from "../analysis.js";
|
||||
import { assertConfig } from "../config.js";
|
||||
import { getProvider } from "../llm/provider.js";
|
||||
import { getProvider, getProviderModelName } from "@/lib/llm/provider.js";
|
||||
import {
|
||||
makeGraph,
|
||||
startCaseRequestSchema,
|
||||
@@ -31,6 +31,8 @@ import {
|
||||
validateGraphReferences,
|
||||
} from "./utils.js";
|
||||
import { validateFindings, produceFindingInformedSummary } from "./finding-helpers.js";
|
||||
import { prepareCompletedEpisode } from "./episode-preparation.js";
|
||||
import { buildEpisodeAwareGraphPrompt } from "./prompt-builder-episode.js";
|
||||
|
||||
function toValidationErrors(error) {
|
||||
return (
|
||||
@@ -357,7 +359,14 @@ export async function startCase(body, dependencies = {}) {
|
||||
}
|
||||
|
||||
const { scenario, promptVersion } = parsedRequest.data;
|
||||
const analysis = await analyseScenario(scenario, { promptVersion });
|
||||
const analysisOptions = { promptVersion };
|
||||
if (dependencies.reconstructionProvider) {
|
||||
analysisOptions.reconstructionProvider = dependencies.reconstructionProvider;
|
||||
}
|
||||
if (dependencies.reconstructionModelName) {
|
||||
analysisOptions.reconstructionModelName = dependencies.reconstructionModelName;
|
||||
}
|
||||
const analysis = await analyseScenario(scenario, analysisOptions);
|
||||
|
||||
if (!analysis.success) {
|
||||
return {
|
||||
@@ -369,7 +378,11 @@ export async function startCase(body, dependencies = {}) {
|
||||
graphReferenceValidation: null,
|
||||
}),
|
||||
analysisErrors: analysis.errors ?? undefined,
|
||||
validationIssues: analysis.validationIssues ?? undefined,
|
||||
providerApiPath: analysis.providerApiPath ?? undefined,
|
||||
providerExecution: analysis.providerExecution ?? undefined,
|
||||
rawResponse: analysis.rawResponse ?? undefined,
|
||||
code: analysis.code ?? undefined,
|
||||
statusCode: Number(analysis.statusCode) || 502,
|
||||
};
|
||||
}
|
||||
@@ -400,6 +413,26 @@ export async function startCase(body, dependencies = {}) {
|
||||
const graphReferenceValidation = validateGraphReferences(
|
||||
initialSituationGraph,
|
||||
);
|
||||
|
||||
if (dependencies.reconstructionOnly) {
|
||||
return {
|
||||
success: graphReferenceValidation.valid,
|
||||
summary: analysis.reconstruction?.summary ?? null,
|
||||
reconstruction: analysis.reconstruction,
|
||||
situationGraph: initialSituationGraph,
|
||||
selectedQuestion: null,
|
||||
diagnostics: buildDiagnostics({
|
||||
analysis,
|
||||
graph: initialSituationGraph,
|
||||
graphReferenceValidation,
|
||||
}),
|
||||
validationErrors: graphReferenceValidation.valid
|
||||
? undefined
|
||||
: graphReferenceValidation.errors,
|
||||
statusCode: graphReferenceValidation.valid ? 200 : 500,
|
||||
};
|
||||
}
|
||||
|
||||
const provider = dependencies.provider ?? getProvider();
|
||||
const modelName = dependencies.modelName ?? analysis?.modelName ?? null;
|
||||
const initialQuestionResult = await determineGraphBackedQuestion({
|
||||
@@ -494,6 +527,7 @@ export async function startCase(body, dependencies = {}) {
|
||||
return {
|
||||
success: true,
|
||||
summary: analysis.reconstruction?.summary ?? null,
|
||||
reconstruction: analysis.reconstruction,
|
||||
situationGraph,
|
||||
selectedQuestion,
|
||||
diagnostics: buildDiagnostics({
|
||||
@@ -633,8 +667,8 @@ async function updateCaseWithDependencies(body, dependencies = {}) {
|
||||
const startedAt = Date.now();
|
||||
|
||||
try {
|
||||
const config = dependencies.config ?? assertConfig();
|
||||
modelName = config.OLLAMA_MODEL;
|
||||
if (!dependencies.config) assertConfig();
|
||||
modelName = dependencies.modelName ?? dependencies.config?.modelName ?? dependencies.config?.OLLAMA_MODEL ?? getProviderModelName();
|
||||
|
||||
const prompt = buildPrompt({
|
||||
situationGraph,
|
||||
@@ -1106,3 +1140,123 @@ async function updateCaseWithDependencies(body, dependencies = {}) {
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
// ── Episode-aware reconsideration seam ─────────────────────
|
||||
|
||||
/**
|
||||
* Reason over a completed focused episode and return a parsed GraphUpdateProposal.
|
||||
*
|
||||
* This is the smallest sibling to updateCase that reuses existing provider,
|
||||
* model, and parsing mechanics — only the prompt path differs.
|
||||
*/
|
||||
export async function reconsiderCompletedEpisode(epiParams, deps = {}) {
|
||||
const { situationGraph: epiGraph, targetNodeId, contributions, findings, episode: rawEpisode } = epiParams;
|
||||
|
||||
// Accept prepared episode or produce it deterministically
|
||||
|
||||
let episode;
|
||||
|
||||
if (rawEpisode) {
|
||||
episode = rawEpisode;
|
||||
} else if (epiParams?.turns || epiParams?.eligibleCanonicalFindings) {
|
||||
// Already a prepared episode passed as first positional arg
|
||||
episode = epiParams;
|
||||
} else {
|
||||
episode = prepareCompletedEpisode(epiParams);
|
||||
}
|
||||
|
||||
if (!episode?.turns && !episode?.eligibleCanonicalFindings) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "input_validation",
|
||||
error: "Invalid prepared episode — requires turns or findings",
|
||||
statusCode: 400,
|
||||
};
|
||||
}
|
||||
|
||||
const getModelName = deps.getModelName;
|
||||
const buildEpisodePrompt = deps.buildEpisodePrompt;
|
||||
const parseProposal = deps.parseProposal;
|
||||
const provider = epiParams.provider ?? deps.provider;
|
||||
|
||||
let modelName = null;
|
||||
let rawResponse;
|
||||
const startedAt = Date.now();
|
||||
|
||||
try {
|
||||
// Preserve original priority (deps.modelName > deps.config.OLLAMA_MODEL > getModelName) with assertConfig fallback only when no deps provide a model
|
||||
let resolvedConfig = undefined;
|
||||
if (deps.modelName == null && deps.config?.OLLAMA_MODEL == null && deps.config?.modelName == null && deps.config !== null) {
|
||||
const hasOtherDeps = Object.keys(deps).some((k) => k !== "config" && k !== "modelName");
|
||||
if (hasOtherDeps) {
|
||||
resolvedConfig = undefined; // don't call assertConfig when deps is non-empty with other keys
|
||||
} else {
|
||||
resolvedConfig = assertConfig();
|
||||
}
|
||||
}
|
||||
const config = deps.config ?? resolvedConfig;
|
||||
modelName =
|
||||
deps.modelName ??
|
||||
(config != null ? (config.modelName ?? config.OLLAMA_MODEL) : undefined) ??
|
||||
getModelName?.() ??
|
||||
getProviderModelName() ??
|
||||
null;
|
||||
|
||||
const promptBuilder = buildEpisodePrompt ?? buildEpisodeAwareGraphPrompt;
|
||||
const prompt = promptBuilder({ episode });
|
||||
|
||||
const prov = provider ?? getProvider();
|
||||
rawResponse = await prov.generateReconstruction(prompt, modelName);
|
||||
} catch (error) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "provider",
|
||||
error: "Graph update proposal generation failed",
|
||||
providerErrors: [
|
||||
sanitiseErrorMessage(
|
||||
error,
|
||||
"Provider failed to generate graph update proposal from episode",
|
||||
),
|
||||
],
|
||||
diagnostics: {
|
||||
promptVersion: null,
|
||||
modelName,
|
||||
responseDurationMs: Date.now() - startedAt,
|
||||
normalisationsApplied: [],
|
||||
},
|
||||
statusCode: 502,
|
||||
};
|
||||
}
|
||||
|
||||
const parser = parseProposal ?? parseGraphUpdateProposal;
|
||||
const parsedProposal = parser(rawResponse);
|
||||
const responseDurationMs = Date.now() - startedAt;
|
||||
|
||||
if (!parsedProposal.success) {
|
||||
return {
|
||||
success: false,
|
||||
stage: "proposal_validation",
|
||||
error: "Invalid graph update proposal",
|
||||
proposalErrors: parsedProposal.errors,
|
||||
diagnostics: {
|
||||
promptVersion: null,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
},
|
||||
statusCode: 502,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stage: "proposal_ready",
|
||||
proposal: parsedProposal.proposal,
|
||||
diagnostics: {
|
||||
promptVersion: null,
|
||||
modelName,
|
||||
responseDurationMs,
|
||||
normalisationsApplied: parsedProposal.normalisationsApplied,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,214 @@
|
||||
/**
|
||||
* Episode-aware graph reasoning prompt builder.
|
||||
*
|
||||
* Takes a deterministically prepared completed episode and produces
|
||||
* a prompt that frames the reasoning task as "reconsider the authoritative
|
||||
* SituationGraph using the completed focused investigation as structured evidence"
|
||||
* rather than "update the graph from this answer".
|
||||
*
|
||||
* This is intentionally a separate export from buildGraphUpdatePrompt to
|
||||
* preserve the existing single-turn contract and avoid any ambiguity.
|
||||
*/
|
||||
|
||||
const DEFAULT_PROMPT_VERSION = "v0.4";
|
||||
|
||||
/**
|
||||
* Build a prompt for episode-aware authoritative graph reconsideration.
|
||||
*
|
||||
* The prepared episode structure is consumed by deterministic preparation,
|
||||
* not by this builder. This builder only serializes the proven material.
|
||||
*
|
||||
* @param {Object} params
|
||||
* @param {Object} params.episode - Prepared completed episode from prepareCompletedEpisode()
|
||||
* @param {string} [params.promptVersion] - Prompt schema version (defaults to v0.4)
|
||||
* @returns {string} structured reasoning prompt
|
||||
*/
|
||||
export function buildEpisodeAwareGraphPrompt({ episode, promptVersion = DEFAULT_PROMPT_VERSION }) {
|
||||
const { situationGraph, targetNodeId, turns, eligibleCanonicalFindings, excludedFindingProvenance } = episode;
|
||||
|
||||
// ── Serialize the current authoritative graph ────────────
|
||||
|
||||
const situationGraphSection = typeof situationGraph === "string" ? situationGraph : JSON.stringify(situationGraph, null, 2);
|
||||
|
||||
// ── Serialize ordered turns with eligible evidence per turn ──
|
||||
|
||||
const turnsSection = turns
|
||||
.map(
|
||||
(turn) => {
|
||||
const rawEvidence = `\n Question: ${turn.question}\n Verbatim Answer: ${turn.answer}`;
|
||||
|
||||
const eligibleForTurn = eligibleCanonicalFindings.filter((f) => f.contributionId === turn.contributionId);
|
||||
|
||||
let evidenceSection = "";
|
||||
if (eligibleForTurn.length > 0) {
|
||||
const evidenceItems = eligibleForTurn.map(
|
||||
(f) => ` - Proposition: ${f.proposition} | Endorsement: ${f.endorsement == null ? "null (working premise, not explicit agreement)" : "agree (explicitly endorsed proposition)"}`
|
||||
);
|
||||
evidenceSection = "\n Eligible Canonical Findings from this turn:\n" + evidenceItems.join("\n");
|
||||
}
|
||||
|
||||
return `Turn ${turn.sequence} [contributionId: ${turn.contributionId}]${rawEvidence}${evidenceSection}`;
|
||||
}
|
||||
)
|
||||
.join("\n\n");
|
||||
|
||||
// ── Serialize eligible findings for provider reasoning ───
|
||||
|
||||
const eligibleFindingsList = eligibleCanonicalFindings.map(
|
||||
(f) => ` - [FindingId: ${f.findingId}] Contribution: ${f.contributionId} | Proposition: ${f.proposition} | Endorsement: ${f.endorsement == null ? "null (working premise, not explicit agreement)" : "agree (explicitly endorsed proposition)"}`
|
||||
);
|
||||
|
||||
// ── Assemble the episode-aware prompt ────────────────────
|
||||
|
||||
return `You are proposing a graph update for Confidence Engine ${promptVersion}.
|
||||
|
||||
Return exactly one JSON object matching the GraphUpdate contract.
|
||||
Return JSON only. Do not include markdown, explanation, or any text before or after the JSON object.
|
||||
|
||||
## Current Situation Graph (authoritative — may reflect pre-investigation state at Done)
|
||||
${situationGraphSection}
|
||||
|
||||
## Target Node Being Reconsidered
|
||||
${targetNodeId}
|
||||
|
||||
## Completed Focused Investigation Evidence
|
||||
|
||||
A focused investigation was completed on this target. The following evidence is structured in turn order. Each turn captures a discrete investigative question and its verbatim answer. Turn order reflects the actual sequence of inquiry.
|
||||
|
||||
### Ordered Investigation Turns (ordered context)
|
||||
${turnsSection}
|
||||
|
||||
### Eligible Canonical Findings Derived During Investigation
|
||||
The following findings are canonical propositions derived from the investigation turns above, not raw user statements:
|
||||
${eligibleFindingsList.length > 0 ? eligibleFindingsList.join("\n") : " (none derived during this episode)"}
|
||||
|
||||
### Endorsement Semantics — Read Carefully
|
||||
- null → working premise: NOT explicit agreement; eligible for your reasoning but not endorsed
|
||||
- agree → explicitly endorsed proposition by the user at the proposition level
|
||||
These are proposition-level endorsements. They are evidence for global reasoning and never directly mutate authoritative graph state.
|
||||
|
||||
### Graph Reconsideration Task
|
||||
Reconsider the authoritative SituationGraph using the completed focused investigation as structured evidence:
|
||||
1. Evaluate what the ordered turns — their questions, verbatim answers, and eligible canonical findings with endorsement status — collectively imply about the graph.
|
||||
2. The situation graph may still reflect pre-investigation state; you must reconcile all episode-derived material before proposing changes.
|
||||
3. Produce a GraphUpdateProposal describing the minimal set of structural changes implied by this evidence across the completed episode.
|
||||
|
||||
## Allowed Node Kinds
|
||||
observation | reported_claim | metric | state | transition | relationship | assumption | unknown | conclusion | option
|
||||
|
||||
## Allowed Node Statuses
|
||||
known | unknown | provisional | supported | weakened | contradicted | resolved
|
||||
|
||||
## Allowed Edge Relationships
|
||||
supports | weakens | contradicts | depends_on | causes | may_cause | measures | compares_with | updates | contained_in | other
|
||||
|
||||
## Allowed Confidence Values
|
||||
low | medium | high
|
||||
|
||||
## Required JSON Field Names
|
||||
The JSON object must contain exactly these top-level fields:
|
||||
- addedNodes
|
||||
- updatedNodes
|
||||
- addedEdges
|
||||
- removedEdgeIds
|
||||
- resolvedUnknownNodeIds
|
||||
- affectedNodeIds
|
||||
- selectedQuestion
|
||||
- answerMeaning
|
||||
- structuralActionRequired
|
||||
|
||||
## Required Shapes
|
||||
- addedNodes: array of nodes using these exact keys: id, label, description, kind, status, confidence, value, unit, evidenceIds, dependsOn, affects, parentId, childIds
|
||||
- updatedNodes: array of node updates using these exact keys: nodeId, previousStatus, newStatus, previousValue, newValue, reason
|
||||
- addedEdges: array of edges using these exact keys: id, fromNodeId, toNodeId, relationship, confidence, description
|
||||
- removedEdgeIds: array of strings
|
||||
- resolvedUnknownNodeIds: array of strings
|
||||
- affectedNodeIds: array of strings
|
||||
- selectedQuestion: either null or an object using these exact keys: nodeId, question, reason
|
||||
- answerMeaning: either null or an object using these exact keys: userSupportedMeaning, possibleInference, supportCategory, resolutionGuidance
|
||||
- structuralActionRequired: boolean (required when userSupportedMeaning is populated)
|
||||
|
||||
## Proposal Rules
|
||||
1. Propose changes only. Never return a replacement graph.
|
||||
2. Preserve unrelated nodes and edges by omitting them from the proposal.
|
||||
3. Reference existing node IDs when updating an existing concept.
|
||||
4. Use addedNodes only for genuinely new concepts.
|
||||
5. Resolve the answered unknown first when the answer supports it.
|
||||
6. If answerMeaning.userSupportedMeaning contains consequential information or unresolved uncertainty that is not already represented in the graph, you MUST express its effect through structural mutation. This may be an update/refinement of existing structure, resolution of an existing unknown, a genuinely new unknown, or a justified relationship. answerMeaning alone is not sufficient for a successful proposal.
|
||||
7. Add new unknown nodes only when the answer introduces a new decision, claim, object, measure, dependency, or unresolved term directly relevant to the case.
|
||||
8. Add at most 3 new unknown nodes.
|
||||
9. Every new unknown must be directly traceable to the user's answer and its description must state why that uncertainty matters.
|
||||
9a. In the description of every new unknown, explicitly include a short why-it-matters clause using wording such as because, so that, needed to decide, or matters because.
|
||||
10. Do not add broad generic discovery questions.
|
||||
11. Do not add duplicate unknowns.
|
||||
12. Do not expand unrelated branches.
|
||||
13. Propagate only through explicit dependencies or relationships already present in the graph, except for the minimal new edges needed to connect validated new unknowns to the relevant answer-derived decision or context node.
|
||||
13a. For every new unknown node, include at least one added edge that connects it to an existing updated/resolved node or to a newly added non-unknown node introduced from the answer.
|
||||
14. Do not invent evidence.
|
||||
15. Do not create unsupported causal edges.
|
||||
16. When your proposal adds one or more new unresolved unknowns (status !== 'resolved'), you MUST include a selectedQuestion identifying one of those as a candidate unknown node. The engine validates your candidate and retains deterministic prerequisite ordering, fallback selection, and formulation authority; prefer nodes with no unresolved depends_on prerequisites from same-proposal additions. Your candidate does not need to be the highest-scoring unknown — it only needs to be a valid unresolved unknown that exists in the graph or in addedNodes.
|
||||
17. selectedQuestion.nodeId must reference an unknown node that remains unresolved after applying this same proposal and that exists either already in the graph or in addedNodes. If the proposal resolves all consequential unknowns, selectedQuestion must be null.
|
||||
18. selectedQuestion.question must be one narrow non-compound question about that one unknown.
|
||||
19. Do not prioritise downstream implementation, pricing, optimisation, or speculative branches ahead of prerequisite definitions, actors, success criteria, constraints, measures, or terminology.
|
||||
20. Return selectedQuestion as null only when no consequential unresolved unknown remains.
|
||||
21. Use empty arrays when there are no changes in a category.
|
||||
22. Never return null array entries.
|
||||
23. Never use unknown enum values.
|
||||
24. Do not change existing IDs.
|
||||
25. Do not replace the whole graph, and do not restate unchanged graph content inside the proposal.
|
||||
26. answerMeaning.userSupportedMeaning must state only what the user's answer directly supports.
|
||||
27. Put any stronger interpretation in answerMeaning.possibleInference, not in userSupportedMeaning.
|
||||
28. When answerMeaning is present, populate supportCategory with one of the allowed values whenever the user's meaning fits an existing category. Use other when none of the protected categories applies. Do not leave supportCategory null merely because the wording is uncertain.
|
||||
29. If the answer is conditional or qualified, preserve that qualification explicitly in userSupportedMeaning.
|
||||
30. If the answer says the user is unsure or does not resolve the distinction, state that uncertainty directly in userSupportedMeaning.
|
||||
31. If the answer explicitly states a hard constraint, state that directly in userSupportedMeaning.
|
||||
32. Populate resolutionGuidance when the user's meaning genuinely implies must_remain_unresolved, may_resolve, or must_resolve. Keep it null only when no existing resolution state actually applies.
|
||||
|
||||
## Decision Sufficiency Rule
|
||||
|
||||
An unresolved decision between options should not remain open merely because some uncertainty still exists.
|
||||
|
||||
Keep a decision context unresolved only when you can identify a specific unresolved factor that could materially change which option is preferred.
|
||||
|
||||
You may not resolve the decision context unless the user explicitly confirms (using their own words) that no other material uncertainty remains. If evidence appears sufficient but explicit confirmation is absent, preserve your directional conclusion in possibleInference and allow the system to ask the sufficiency confirmation / discovery question rather than closing the parent decision.
|
||||
|
||||
## Decision Option Structure Rules
|
||||
When the user presents mutually exclusive candidate actions for one unresolved choice:
|
||||
|
||||
1. Create exactly one node of kind "unknown" to carry the decision question (the existing mechanism). Do not add a separate "decision" node kind. Keep that unknown as-is or create it fresh — do not duplicate it into every option.
|
||||
|
||||
2. For each candidate path, create exactly one node of kind "option". The option's label names the alternative; its description states what that alternative entails.
|
||||
|
||||
3. Link each option to the decision-context unknown using relationship "contained_in" (edge: option → unknown). Shared membership already implies these options are alternatives of each other — do not add an "alternative_to" edge between options.
|
||||
|
||||
4. Attach consequences and evidence to the specific option they belong to via existing edge types ("causes", "may_cause", etc.). Each consequence's fromNodeId explicitly identifies its parent option. Do not collapse all alternatives into one generic trade-off description on a single node.
|
||||
|
||||
5. A do-nothing / stay-put / current-state path is an option when it is genuinely one of the alternatives — represent it with kind "option" and label it clearly. Do not introduce an "is_baseline", "is_default", or "is_status_quo" field; baseline meaning is carried by label and consequences alone in this implementation.
|
||||
|
||||
## Contract: structuralActionRequired Declaration Rule
|
||||
|
||||
When answerMeaning.userSupportedMeaning is populated you MUST set structuralActionRequired to match what your proposal outputs:
|
||||
- Set structuralActionRequired = true if and only if your proposal adds nodes, updates node status/value, or modifies edges (addedNodes.length > 0, updatedNodes with a meaningful change, or addedEdges.length > 0).
|
||||
- Set structuralActionRequired = false if and only if your proposal has zero structural mutations — the two sentences are an intentional no-op declaration.
|
||||
|
||||
## Additional Guidance
|
||||
- If the answer only clarifies an existing unknown, prefer updatedNodes and resolvedUnknownNodeIds over creating duplicate nodes.
|
||||
- When rule #6 applies to explicitly unresolved uncertainty: first check whether an existing unresolved node already represents the same uncertainty; if so, update/refine that existing structure rather than adding a duplicate; if no such node exists, add a new unknown that directly represents the unresolved uncertainty; do not use an edge alone to represent a previously unrepresented uncertainty.
|
||||
- "Same uncertainty" means the same resolution question: resolving the existing unknown would also resolve the uncertainty introduced by the user's answer. Mere topical overlap (concerning the same topic, object, decision, or domain) is not automatically the same uncertainty. If the new concern can remain unresolved after the existing node is resolved, represent it separately as a distinct uncertainty.
|
||||
- When an answer resolves an existing unknown, include that existing node ID in resolvedUnknownNodeIds and update that node rather than creating only a parallel observation.
|
||||
- If the answer creates a more specific decision situation, add the smallest set of new nodes and edges needed to represent that situation and only its most consequential unknowns.
|
||||
- If you add a new unknown, do not leave it floating: connect it with an added edge to the relevant decision/context node created or updated from the answer.
|
||||
- If you add a new unknown, its description must do two jobs in one sentence: what is unknown, and why resolving it matters for the case.
|
||||
- When selectedQuestion is provided, identify the specific material continuation factor as selectedQuestion.nodeId; the engine retains deterministic prerequisite ordering, validation, and formulation authority — it favours your selected node when it has no unresolved depends_on prerequisites from same-proposal additions, falls back to existing deterministic selection otherwise, and may choose a different question if structural constraints require.
|
||||
- If rule #6 does not apply (the answer contains no user-supported meaning that requires graph progress) and there is no other justification for change, return empty arrays for every category.
|
||||
- If rule #6 applies but you choose an update/refinement of existing structure, resolve an existing unknown, or add justified new structure, your structural proposal plus answerMeaning together represent the complete response — answerMeaning preserves semantic fidelity while structural mutation handles graph progress; neither replaces the other.
|
||||
- If you add a new unknown with addedNodes, connect it with at least one addedEdge to an existing updated/resolved node or to a newly added non-unknown node from the answer.
|
||||
- For answerMeaning.supportCategory, use only these exact values: ${["relative_priority_only", "conditional_tradeoff", "uncertain", "explicit_hard_constraint", "other"].join(" | ")}. Use other when none of the protected categories applies.
|
||||
- For answerMeaning.resolutionGuidance, use only these exact values: ${["must_remain_unresolved", "may_resolve", "must_resolve"].join(" | ")}. Keep it null only when none of those existing resolution states genuinely applies.
|
||||
|
||||
## Output Contract Reminder
|
||||
Return one JSON object only, with exact field names and exact enum values.
|
||||
Never include a full graph.
|
||||
Never include any field other than the contract fields above.
|
||||
`;
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
/**
|
||||
* Deterministically reopen a resolved unknown node in a SituationGraph.
|
||||
*
|
||||
* Transition: removes the node's ID from resolvedNodeIds and establishes
|
||||
* node.status "unknown".
|
||||
* from resolvedNodeIds. This reverses canonical resolution so the question
|
||||
* reappears among Open Questions for further investigation.
|
||||
*
|
||||
* Idempotent — if the target is not an unknown marked resolved in graph state, returns the
|
||||
* original graph unchanged (no mutation). Does NOT create nodes, delete edges,
|
||||
* or touch contributions/findings/historical evidence.
|
||||
*/
|
||||
|
||||
export function reopenResolvedUnknown(situationGraph, nodeId) {
|
||||
if (!situationGraph || !nodeId) return situationGraph;
|
||||
|
||||
const nodeIndex = situationGraph.nodes.findIndex((n) => n.id === nodeId);
|
||||
if (nodeIndex === -1) return situationGraph;
|
||||
|
||||
const node = situationGraph.nodes[nodeIndex];
|
||||
if (node.kind !== "unknown" || !situationGraph.resolvedNodeIds?.includes(nodeId)) {
|
||||
return situationGraph;
|
||||
}
|
||||
|
||||
const newNode = { ...node, status: "unknown" };
|
||||
const newNodes = [...situationGraph.nodes];
|
||||
newNodes[nodeIndex] = newNode;
|
||||
|
||||
const newResolvedNodeIds = [
|
||||
...(situationGraph.resolvedNodeIds || []),
|
||||
].filter((id) => id !== nodeId);
|
||||
|
||||
return {
|
||||
...situationGraph,
|
||||
nodes: newNodes,
|
||||
resolvedNodeIds: newResolvedNodeIds,
|
||||
};
|
||||
}
|
||||
@@ -204,6 +204,7 @@ export const startCaseRequestSchema = z.object({
|
||||
promptVersion: z.string().optional(),
|
||||
});
|
||||
|
||||
/** Legacy schema — unchanged contract for existing consumers. */
|
||||
export const updateCaseRequestSchema = z.object({
|
||||
situationGraph: situationGraphSchema,
|
||||
previousQuestion: z.string().min(1),
|
||||
@@ -222,6 +223,20 @@ export const updateCaseRequestSchema = z.object({
|
||||
).optional(),
|
||||
});
|
||||
|
||||
/** Extended schema with optional fields accepted by the route for all requests. */
|
||||
export const updateCaseRequestSchemaExtended = updateCaseRequestSchema.extend({
|
||||
targetNodeId: z.string().optional(),
|
||||
contributions: z.any().array().optional(),
|
||||
});
|
||||
|
||||
/** Minimal schema for episode-mode requests (legacy fields not required). */
|
||||
export const updateCaseEpisodeRequestSchema = z.object({
|
||||
situationGraph: situationGraphSchema,
|
||||
targetNodeId: z.string().min(1),
|
||||
contributions: z.any().array().optional(),
|
||||
findings: z.any().array().optional(),
|
||||
}).passthrough();
|
||||
|
||||
// ── Helpers ──────────────────────────────────────────
|
||||
|
||||
/** Generate a short deterministic ID from a label */
|
||||
|
||||
+337
-18
@@ -1,3 +1,7 @@
|
||||
import { z } from "zod";
|
||||
import { reconstructionV2Schema } from "../reconstruction/schema.js";
|
||||
import { isOpenAIUiJourneyExperiment, OPENAI_TERRA_MODEL } from "../config.js";
|
||||
|
||||
/**
|
||||
* Provider abstraction — the app calls getProvider() which returns an object
|
||||
* with a generateReconstruction(scenario, modelName) method.
|
||||
@@ -6,9 +10,23 @@
|
||||
*/
|
||||
|
||||
export function getProvider() {
|
||||
if (isOpenAIUiJourneyExperiment()) return createOpenAIReconstructionProvider();
|
||||
return new OllamaLlmProvider();
|
||||
}
|
||||
|
||||
/** Server-owned model resolution for the configured application provider. */
|
||||
export function getProviderModelName() {
|
||||
return isOpenAIUiJourneyExperiment() ? OPENAI_TERRA_MODEL : process.env.OLLAMA_MODEL ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Experiment-only construction seam. Production provider selection remains Ollama.
|
||||
* @param {{ apiKey?: string, fetchImpl?: typeof fetch }} [options]
|
||||
*/
|
||||
export function createOpenAIReconstructionProvider(options = {}) {
|
||||
return new OpenAIReconstructionProvider(options);
|
||||
}
|
||||
|
||||
function recoverJson(raw) {
|
||||
if (typeof raw !== "string") return raw;
|
||||
const trimmed = raw.trim();
|
||||
@@ -58,9 +76,221 @@ function recoverJson(raw) {
|
||||
throw new SyntaxError("Model output could not be parsed as JSON: " + result.slice(0, 300) + "...");
|
||||
}
|
||||
|
||||
let _chatSupported = null;
|
||||
function extractOpenAIResponseText(data) {
|
||||
if (typeof data.output_text === "string" && data.output_text.length > 0) {
|
||||
return data.output_text;
|
||||
}
|
||||
|
||||
async function detectChatSupport(baseUrl) {
|
||||
const outputText = (data.output ?? []).flatMap((item) =>
|
||||
item?.type === "message"
|
||||
? (item.content ?? []).flatMap((part) =>
|
||||
part?.type === "output_text" && typeof part.text === "string"
|
||||
? [part.text]
|
||||
: [],
|
||||
)
|
||||
: [],
|
||||
);
|
||||
if (outputText.length > 0) return outputText.join("");
|
||||
|
||||
const outputTypes = (data.output ?? []).map((item) => item?.type ?? "unknown");
|
||||
const contentTypes = (data.output ?? []).flatMap((item) =>
|
||||
(item?.content ?? []).map((part) => part?.type ?? "unknown"),
|
||||
);
|
||||
const refusal = contentTypes.includes("refusal");
|
||||
throw new Error(
|
||||
`OpenAI Responses API returned no output_text (output types: ${outputTypes.join(",") || "none"}; content types: ${contentTypes.join(",") || "none"}; refusal: ${refusal})`,
|
||||
);
|
||||
}
|
||||
|
||||
let _chatSupported = null;
|
||||
export const reconstructionJsonSchema = z.toJSONSchema(reconstructionV2Schema);
|
||||
const openAIReconstructionJsonSchema = createOpenAIStrictSchema(
|
||||
reconstructionJsonSchema,
|
||||
reconstructionV2Schema,
|
||||
);
|
||||
|
||||
/** @internal OpenAI Structured Outputs requires every object property. */
|
||||
export function createOpenAIStrictSchema(schema, zodSchema = reconstructionV2Schema) {
|
||||
const projected = structuredClone(schema);
|
||||
projectOpenAIStrictSchema(projected, zodSchema, projected, new WeakSet());
|
||||
return projected;
|
||||
}
|
||||
|
||||
/** @internal Remove OpenAI null placeholders for canonically optional fields only. */
|
||||
export function normaliseOpenAITransportResponse(
|
||||
value,
|
||||
zodSchema = reconstructionV2Schema,
|
||||
schema = reconstructionJsonSchema,
|
||||
) {
|
||||
return normaliseTransportValue(value, schema, zodSchema, schema);
|
||||
}
|
||||
|
||||
function resolveSchema(schema, rootSchema) {
|
||||
if (!schema?.$ref) return schema;
|
||||
const path = schema.$ref.replace(/^#\//, "").split("/");
|
||||
return path.reduce((value, key) => value?.[key], rootSchema) ?? schema;
|
||||
}
|
||||
|
||||
function zodDef(schema) {
|
||||
return schema?._zod?.def ?? schema?._def;
|
||||
}
|
||||
|
||||
function unwrapZodSchema(schema) {
|
||||
const def = zodDef(schema);
|
||||
if (["optional", "nullable", "default"].includes(def?.type)) {
|
||||
return unwrapZodSchema(def.innerType);
|
||||
}
|
||||
return schema;
|
||||
}
|
||||
|
||||
function zodObjectShape(schema) {
|
||||
const def = zodDef(unwrapZodSchema(schema));
|
||||
return def?.type === "object" ? def.shape : null;
|
||||
}
|
||||
|
||||
function zodArrayItem(schema) {
|
||||
const def = zodDef(unwrapZodSchema(schema));
|
||||
return def?.type === "array" ? def.element : null;
|
||||
}
|
||||
|
||||
function zodAcceptsNull(schema) {
|
||||
return schema?.isNullable?.() === true;
|
||||
}
|
||||
|
||||
function projectOpenAIStrictSchema(schema, zodSchema, rootSchema, visited) {
|
||||
const resolved = resolveSchema(schema, rootSchema);
|
||||
if (!resolved || typeof resolved !== "object" || visited.has(resolved)) return;
|
||||
visited.add(resolved);
|
||||
const shape = zodObjectShape(zodSchema);
|
||||
if (resolved?.type === "object" || resolved?.properties) {
|
||||
if (shape) {
|
||||
for (const [key, property] of Object.entries(resolved.properties)) {
|
||||
const propertyZodSchema = shape[key];
|
||||
if (propertyZodSchema?.isOptional?.() && !schemaAllowsNull(property, rootSchema)) {
|
||||
resolved.properties[key] = { anyOf: [property, { type: "null" }] };
|
||||
}
|
||||
}
|
||||
}
|
||||
resolved.additionalProperties = false;
|
||||
}
|
||||
|
||||
if (resolved?.items) {
|
||||
projectOpenAIStrictSchema(resolved.items, zodArrayItem(zodSchema), rootSchema, visited);
|
||||
}
|
||||
if (resolved?.properties) {
|
||||
for (const [key, property] of Object.entries(resolved.properties)) {
|
||||
projectOpenAIStrictSchema(property, shape?.[key], rootSchema, visited);
|
||||
}
|
||||
resolved.required = Object.keys(resolved.properties);
|
||||
}
|
||||
for (const branch of [
|
||||
...(resolved.anyOf ?? []),
|
||||
...(resolved.oneOf ?? []),
|
||||
...(resolved.allOf ?? []),
|
||||
]) {
|
||||
projectOpenAIStrictSchema(branch, null, rootSchema, visited);
|
||||
}
|
||||
for (const definition of Object.values(resolved.$defs ?? resolved.definitions ?? {})) {
|
||||
projectOpenAIStrictSchema(definition, null, rootSchema, visited);
|
||||
}
|
||||
}
|
||||
|
||||
function satisfiesOpenAIStrictSchema(schema, visited = new WeakSet()) {
|
||||
if (!schema || typeof schema !== "object" || visited.has(schema)) return true;
|
||||
visited.add(schema);
|
||||
const isObject = schema.type === "object" || schema.properties;
|
||||
if (isObject && schema.additionalProperties !== false) return false;
|
||||
if (schema.properties) {
|
||||
const propertyKeys = Object.keys(schema.properties);
|
||||
if (
|
||||
!Array.isArray(schema.required) ||
|
||||
schema.required.length !== propertyKeys.length ||
|
||||
!propertyKeys.every((key) => schema.required.includes(key))
|
||||
) return false;
|
||||
}
|
||||
return [
|
||||
schema.items,
|
||||
...Object.values(schema.properties ?? {}),
|
||||
...(schema.anyOf ?? []),
|
||||
...(schema.oneOf ?? []),
|
||||
...(schema.allOf ?? []),
|
||||
...Object.values(schema.$defs ?? schema.definitions ?? {}),
|
||||
].every((child) => satisfiesOpenAIStrictSchema(child, visited));
|
||||
}
|
||||
|
||||
function traceOpenAISchema({ model, format, schema }) {
|
||||
if (process.env.CONFIDENCE_ENGINE_EXPERIMENT_TRACE_OPENAI_SCHEMA !== "1") return;
|
||||
const serializedSchema = JSON.stringify(schema);
|
||||
const finalSchema = JSON.parse(serializedSchema);
|
||||
const properties = finalSchema.properties ?? {};
|
||||
const required = finalSchema.required ?? [];
|
||||
const relationships = properties.relationships;
|
||||
console.info("[confidence-engine][openai-schema-trace]", {
|
||||
model,
|
||||
formatType: format.type,
|
||||
formatName: format.name,
|
||||
strict: format.strict,
|
||||
rootProperties: Object.keys(properties),
|
||||
rootRequired: required,
|
||||
rootSetsEqual:
|
||||
required.length === Object.keys(properties).length &&
|
||||
Object.keys(properties).every((key) => required.includes(key)),
|
||||
relationshipsInProperties: Object.prototype.hasOwnProperty.call(properties, "relationships"),
|
||||
relationshipsInRequired: required.includes("relationships"),
|
||||
relationshipsItemType: relationships?.items?.type ?? null,
|
||||
relationshipsItemAdditionalProperties: relationships?.items?.additionalProperties ?? null,
|
||||
rootAdditionalProperties: finalSchema.additionalProperties ?? null,
|
||||
recursiveStrictInvariant: satisfiesOpenAIStrictSchema(finalSchema),
|
||||
});
|
||||
}
|
||||
|
||||
function schemaAllowsNull(schema, rootSchema) {
|
||||
const resolved = resolveSchema(schema, rootSchema);
|
||||
return (
|
||||
resolved?.type === "null" ||
|
||||
(Array.isArray(resolved?.type) && resolved.type.includes("null")) ||
|
||||
[...(resolved?.anyOf ?? []), ...(resolved?.oneOf ?? [])].some((branch) =>
|
||||
schemaAllowsNull(branch, rootSchema),
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
function normaliseTransportValue(value, schema, zodSchema, rootSchema) {
|
||||
const resolved = resolveSchema(schema, rootSchema);
|
||||
if (Array.isArray(value) && resolved?.items) {
|
||||
return value.map((item) =>
|
||||
normaliseTransportValue(item, resolved.items, zodArrayItem(zodSchema), rootSchema),
|
||||
);
|
||||
}
|
||||
if (!value || typeof value !== "object" || !resolved?.properties) return value;
|
||||
|
||||
const shape = zodObjectShape(zodSchema);
|
||||
const normalised = {};
|
||||
for (const [key, item] of Object.entries(value)) {
|
||||
const propertySchema = resolved.properties[key];
|
||||
const propertyZodSchema = shape?.[key];
|
||||
if (!propertySchema || !propertyZodSchema) {
|
||||
normalised[key] = item;
|
||||
} else if (item === null && propertyZodSchema.isOptional?.() && !zodAcceptsNull(propertyZodSchema)) {
|
||||
continue;
|
||||
} else {
|
||||
normalised[key] = normaliseTransportValue(
|
||||
item,
|
||||
propertySchema,
|
||||
propertyZodSchema,
|
||||
rootSchema,
|
||||
);
|
||||
}
|
||||
}
|
||||
return normalised;
|
||||
}
|
||||
|
||||
/** @internal Test-only seam for isolated provider capability scenarios. */
|
||||
export function __resetChatSupportForTests() {
|
||||
_chatSupported = null;
|
||||
}
|
||||
|
||||
async function detectChatSupport(baseUrl, modelName) {
|
||||
if (_chatSupported !== null) return _chatSupported;
|
||||
|
||||
try {
|
||||
@@ -68,20 +298,20 @@ async function detectChatSupport(baseUrl) {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
model: "dummy-check",
|
||||
model: modelName,
|
||||
messages: [{ role: "user", content: "test" }],
|
||||
stream: false,
|
||||
}),
|
||||
});
|
||||
|
||||
if (res.ok) {
|
||||
await res.body?.consume();
|
||||
await res.text();
|
||||
_chatSupported = true;
|
||||
} else if (res.status === 405 || res.status === 501) {
|
||||
await res.body?.consume();
|
||||
await res.text();
|
||||
_chatSupported = false;
|
||||
} else {
|
||||
await res.body?.consume();
|
||||
await res.text();
|
||||
_chatSupported = false;
|
||||
}
|
||||
} catch {
|
||||
@@ -92,7 +322,7 @@ async function detectChatSupport(baseUrl) {
|
||||
}
|
||||
|
||||
class OllamaLlmProvider {
|
||||
async generateReconstruction(scenario, modelName) {
|
||||
async generateReconstruction(scenario, modelName, outputSchema) {
|
||||
// scenario is ALREADY a fully-built prompt text (built by analyseScenario).
|
||||
// Do NOT call buildPrompt() again — that would double-wrap the prompt.
|
||||
const prompt = scenario;
|
||||
@@ -104,22 +334,31 @@ class OllamaLlmProvider {
|
||||
let chatSupported = false;
|
||||
let rawResponse = null;
|
||||
let fullResponseData = null;
|
||||
const providerExecution = {
|
||||
chatCapabilityDetected: false,
|
||||
chatRequestAttempted: false,
|
||||
chatRequestSucceeded: false,
|
||||
generateRequestAttempted: false,
|
||||
};
|
||||
|
||||
// ================================================================
|
||||
// Step 1: Detect whether /api/chat exists (cache result)
|
||||
// ================================================================
|
||||
try {
|
||||
chatSupported = await detectChatSupport(baseUrl);
|
||||
chatSupported = await detectChatSupport(baseUrl, modelName);
|
||||
} catch { /* failed silently — defaults to false */ }
|
||||
providerExecution.chatCapabilityDetected = chatSupported;
|
||||
|
||||
// ================================================================
|
||||
// Step 2: Try /api/chat if supported and format:json works
|
||||
// Step 2: Try /api/chat if supported with the supplied or reconstruction schema
|
||||
// ================================================================
|
||||
if (chatSupported) {
|
||||
try {
|
||||
const controller = new AbortController();
|
||||
const timeout = setTimeout(() => controller.abort(), 60000);
|
||||
|
||||
const timeout = setTimeout(() => controller.abort(), 300000); // 5 min for reconstruction
|
||||
|
||||
apiUsed = "/api/chat";
|
||||
providerExecution.chatRequestAttempted = true;
|
||||
const res = await fetch(`${baseUrl}/api/chat`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
@@ -127,7 +366,7 @@ class OllamaLlmProvider {
|
||||
model: modelName,
|
||||
messages: [{ role: "user", content: prompt }],
|
||||
stream: false,
|
||||
format: "json",
|
||||
format: outputSchema ?? reconstructionJsonSchema,
|
||||
}),
|
||||
signal: controller.signal,
|
||||
});
|
||||
@@ -135,13 +374,14 @@ class OllamaLlmProvider {
|
||||
clearTimeout(timeout);
|
||||
|
||||
if (res.ok) {
|
||||
providerExecution.chatRequestSucceeded = true;
|
||||
fullResponseData = await res.json();
|
||||
rawResponse = typeof fullResponseData.message?.content === "string"
|
||||
? fullResponseData.message.content
|
||||
: JSON.stringify(fullResponseData.message?.content ?? null);
|
||||
apiUsed = "/api/chat";
|
||||
} else {
|
||||
await res.body?.consume();
|
||||
await res.text();
|
||||
}
|
||||
} catch (e) {
|
||||
if (!e.message.includes("abort")) { /* non-fatal */ }
|
||||
@@ -158,6 +398,7 @@ class OllamaLlmProvider {
|
||||
const timeout = setTimeout(() => controller.abort(), 300000); // 5 min for cold start
|
||||
|
||||
apiUsed = "/api/generate"; // set BEFORE the request so we know which API failed
|
||||
providerExecution.generateRequestAttempted = true;
|
||||
|
||||
const res = await fetch(`${baseUrl}/api/generate`, {
|
||||
method: "POST",
|
||||
@@ -180,14 +421,14 @@ class OllamaLlmProvider {
|
||||
}
|
||||
|
||||
fullResponseData = await res.json();
|
||||
|
||||
|
||||
rawResponse = typeof fullResponseData.response === "string"
|
||||
? fullResponseData.response
|
||||
: JSON.stringify(fullResponseData);
|
||||
|
||||
} catch (e) {
|
||||
if (apiUsed === "/api/generate") {
|
||||
throw new Error(
|
||||
const error = new Error(
|
||||
`Ollama /api/generate request timed out after 5 minutes.\n\n` +
|
||||
`This usually means:\n` +
|
||||
`1. The model is loading into memory for the first time (cold start) — this can take several minutes\n` +
|
||||
@@ -198,6 +439,12 @@ class OllamaLlmProvider {
|
||||
`- Use a smaller model (e.g., llama3.1 instead of llama3.1:70b)\n` +
|
||||
`- Check Ollama logs: \`ollama serve\` or look at your system logs`
|
||||
);
|
||||
if (e?.name === "AbortError" || e instanceof TypeError) {
|
||||
error.code = "PROVIDER_UNAVAILABLE";
|
||||
}
|
||||
error.providerApiPath = apiUsed;
|
||||
error.providerExecution = providerExecution;
|
||||
throw error;
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
@@ -225,7 +472,7 @@ class OllamaLlmProvider {
|
||||
}
|
||||
}
|
||||
|
||||
throw new Error(
|
||||
const error = new Error(
|
||||
"Model produced empty output.\n\n" +
|
||||
"API used: " + (apiUsed || "none") + "\n" +
|
||||
"/api/chat supported: " + chatSupported + "\n" +
|
||||
@@ -236,16 +483,23 @@ class OllamaLlmProvider {
|
||||
"- This Ollama version does not support format:json — using prompt instructions only (reliability varies)\n" +
|
||||
"- If your model is very small (e.g., tinyllama, phi), try a larger one like llama3.1 or mistral"
|
||||
);
|
||||
error.providerApiPath = apiUsed;
|
||||
error.providerExecution = providerExecution;
|
||||
throw error;
|
||||
}
|
||||
|
||||
// ================================================================
|
||||
// Step 5: Parse and return
|
||||
// ================================================================
|
||||
try {
|
||||
return recoverJson(rawResponse);
|
||||
return {
|
||||
response: recoverJson(rawResponse),
|
||||
providerApiPath: apiUsed,
|
||||
providerExecution,
|
||||
};
|
||||
} catch (e) {
|
||||
if (e instanceof SyntaxError) {
|
||||
throw new Error(
|
||||
const error = new Error(
|
||||
"Model returned output that could not be parsed as valid JSON.\n\n" +
|
||||
"API used: " + (apiUsed || "none") + "\n" +
|
||||
"/api/chat supported: " + chatSupported + "\n" +
|
||||
@@ -256,8 +510,73 @@ class OllamaLlmProvider {
|
||||
"- Shorten your scenario to under 500 words\n" +
|
||||
"- Consider upgrading Ollama: https://ollama.com/download"
|
||||
);
|
||||
error.providerApiPath = apiUsed;
|
||||
error.providerExecution = providerExecution;
|
||||
throw error;
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
class OpenAIReconstructionProvider {
|
||||
constructor({ apiKey = process.env.OPENAI_API_KEY, fetchImpl = fetch } = {}) {
|
||||
this.apiKey = apiKey;
|
||||
this.fetchImpl = fetchImpl;
|
||||
}
|
||||
|
||||
async generateReconstruction(prompt, modelName = "gpt-5.6-terra", outputSchema) {
|
||||
if (!this.apiKey) throw new Error("OPENAI_API_KEY is not set");
|
||||
const structuredOutputSchema = outputSchema
|
||||
? createOpenAIStrictSchema(outputSchema, null)
|
||||
: openAIReconstructionJsonSchema;
|
||||
|
||||
const format = {
|
||||
type: "json_schema",
|
||||
name: "reconstruction",
|
||||
strict: true,
|
||||
schema: structuredOutputSchema,
|
||||
};
|
||||
traceOpenAISchema({ model: modelName, format, schema: structuredOutputSchema });
|
||||
const response = await this.fetchImpl("https://api.openai.com/v1/responses", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${this.apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelName,
|
||||
input: prompt,
|
||||
text: {
|
||||
format,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const body = await response.text();
|
||||
const error = new Error(
|
||||
`OpenAI Responses API returned ${response.status}: ${body}`,
|
||||
);
|
||||
error.providerApiPath = "/v1/responses";
|
||||
throw error;
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
let outputText;
|
||||
try {
|
||||
outputText = extractOpenAIResponseText(data);
|
||||
} catch (cause) {
|
||||
const error = new Error(cause.message);
|
||||
error.providerApiPath = "/v1/responses";
|
||||
throw error;
|
||||
}
|
||||
|
||||
return {
|
||||
response: outputSchema
|
||||
? recoverJson(outputText)
|
||||
: normaliseOpenAITransportResponse(recoverJson(outputText)),
|
||||
providerApiPath: "/v1/responses",
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,14 +7,14 @@ const __dirname = dirname(__filename);
|
||||
const PROMPTS_DIR = join(__dirname, "../../prompts");
|
||||
|
||||
/** Available prompt versions */
|
||||
export const PROMPT_VERSIONS = ["v0.1", "v0.2", "v0.3"];
|
||||
export const PROMPT_VERSIONS = ["v0.1", "v0.2", "v0.3", "v0.4", "v0.5"];
|
||||
|
||||
/** Default prompt version (override via RECONSTRUCTION_PROMPT_VERSION env var) */
|
||||
const defaultVersionFromEnv = process.env.RECONSTRUCTION_PROMPT_VERSION;
|
||||
export const DEFAULT_PROMPT_VERSION =
|
||||
defaultVersionFromEnv && PROMPT_VERSIONS.includes(defaultVersionFromEnv)
|
||||
? defaultVersionFromEnv
|
||||
: "v0.3";
|
||||
: "v0.5";
|
||||
|
||||
/** Build a v0.1 (extraction-only) prompt inline for backward compatibility */
|
||||
function buildV1Prompt(scenario) {
|
||||
@@ -77,12 +77,43 @@ async function buildV3Prompt(scenario) {
|
||||
}
|
||||
}
|
||||
|
||||
/** Load a versioned prompt from disk and substitute {{SCENARIO}} */
|
||||
async function buildV4Prompt(scenario) {
|
||||
try {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.4.md"),
|
||||
"utf-8",
|
||||
);
|
||||
return content.replace("{{SCENARIO}}", scenario);
|
||||
} catch {
|
||||
// Fall back to v0.3 prompt if v0.4 file is missing
|
||||
return buildV3Prompt(scenario);
|
||||
}
|
||||
}
|
||||
|
||||
/** Load a versioned prompt from disk and substitute {{SCENARIO}} */
|
||||
async function buildV5Prompt(scenario) {
|
||||
try {
|
||||
const content = await fs.readFile(
|
||||
join(PROMPTS_DIR, "reconstruct-v0.5.md"),
|
||||
"utf-8",
|
||||
);
|
||||
return content.replace("{{SCENARIO}}", scenario);
|
||||
} catch {
|
||||
// Fall back to v0.4 prompt if v0.5 file is missing
|
||||
return buildV4Prompt(scenario);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an analysis prompt for the given version.
|
||||
* @param {"v0.1" | "v0.2" | "v0.3"} [version="v0.3"]
|
||||
* @param {string} scenario - The scenario text
|
||||
* @param {"v0.1" | "v0.2" | "v0.3" | "v0.4" | "v0.5"} [version="v0.5"] - Prompt version
|
||||
* @param {object} [opts] - Optional experimental parameters
|
||||
* @param {string} [opts.experimentInstruction] - Bounded experimental instruction block appended to the base prompt (production prompt is never replaced)
|
||||
* @returns {Promise<{prompt: string, version: string}>}
|
||||
*/
|
||||
export async function buildPrompt(scenario, version = "v0.3") {
|
||||
export async function buildPrompt(scenario, version = "v0.5", opts = {}) {
|
||||
let prompt;
|
||||
switch (version) {
|
||||
case "v0.1":
|
||||
@@ -91,6 +122,12 @@ export async function buildPrompt(scenario, version = "v0.3") {
|
||||
case "v0.2":
|
||||
prompt = await buildV2Prompt(scenario);
|
||||
break;
|
||||
case "v0.4":
|
||||
prompt = await buildV4Prompt(scenario);
|
||||
break;
|
||||
case "v0.5":
|
||||
prompt = await buildV5Prompt(scenario);
|
||||
break;
|
||||
default: // v0.3
|
||||
prompt = await buildV3Prompt(scenario);
|
||||
break;
|
||||
@@ -98,5 +135,15 @@ export async function buildPrompt(scenario, version = "v0.3") {
|
||||
|
||||
const strongJsonHint =
|
||||
"\n\nReturn ONLY a valid JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.";
|
||||
return { prompt: prompt + strongJsonHint, version };
|
||||
|
||||
// ── Experiment seam: append optional instruction block ──
|
||||
let finalPrompt = prompt;
|
||||
if (opts.experimentInstruction) {
|
||||
const delimiter = "\n\n--- EXPERIMENT INSTRUCTION ---\n";
|
||||
finalPrompt = prompt + delimiter + opts.experimentInstruction + strongJsonHint;
|
||||
} else {
|
||||
finalPrompt = prompt + strongJsonHint;
|
||||
}
|
||||
|
||||
return { prompt: finalPrompt, version };
|
||||
}
|
||||
|
||||
@@ -126,6 +126,24 @@ const evidenceRecordSchema = z.object({
|
||||
importance: importanceEnum,
|
||||
});
|
||||
|
||||
const reconstructionRelationshipSchema = z.object({
|
||||
id: z.string().min(1),
|
||||
fromId: z.string().min(1),
|
||||
toId: z.string().min(1),
|
||||
relationship: z.enum([
|
||||
"supports",
|
||||
"weakens",
|
||||
"contradicts",
|
||||
"depends_on",
|
||||
"causes",
|
||||
"may_cause",
|
||||
"compares_with",
|
||||
"other",
|
||||
]),
|
||||
description: z.string().min(1),
|
||||
confidence: confidenceEnum,
|
||||
});
|
||||
|
||||
const reconstructionSchemaV2 = z.object({
|
||||
summary: z.string().min(1),
|
||||
actors: z.array(itemSchemaV1),
|
||||
@@ -159,6 +177,7 @@ const reconstructionSchemaV2 = z.object({
|
||||
confidence: confidenceEnum,
|
||||
}),
|
||||
),
|
||||
relationships: z.array(reconstructionRelationshipSchema).optional().default([]),
|
||||
});
|
||||
|
||||
const inputClassificationSchema = z.object({
|
||||
|
||||
@@ -1,5 +1,94 @@
|
||||
// investigation-storage — generic persistence boundary
|
||||
// Exposes loadInvestigation / saveInvestigation / clearInvestigation
|
||||
// and delegates internally to the concrete localStorage provider.
|
||||
// investigation-storage — application-facing persistence boundary
|
||||
// Owns the canonical identity contract: snapshot.id is the sole save identity.
|
||||
// Delegates to the authenticated browser HTTP provider internally.
|
||||
|
||||
export { loadInvestigation, saveInvestigation, clearInvestigation } from "./providers/local-storage.js";
|
||||
import { loadInvestigation as _load, saveInvestigation as _save, listInvestigations as _list, restartInvestigation as _restart } from "./providers/server-http.js";
|
||||
|
||||
const saveStates = new Map();
|
||||
|
||||
function observeUnhandledRejection(promise) {
|
||||
promise.catch(() => {});
|
||||
return promise;
|
||||
}
|
||||
|
||||
function getSaveState(id) {
|
||||
if (!saveStates.has(id)) saveStates.set(id, { inFlight: false, pending: null, idleWaiters: [] });
|
||||
return saveStates.get(id);
|
||||
}
|
||||
|
||||
async function drainSaveState(id, state) {
|
||||
while (state.pending) {
|
||||
const pending = state.pending;
|
||||
state.pending = null;
|
||||
try {
|
||||
const saved = await _save(pending.snapshot, id);
|
||||
pending.waiters.forEach(({ resolve }) => resolve(saved));
|
||||
} catch (error) {
|
||||
pending.waiters.forEach(({ reject }) => reject(error));
|
||||
}
|
||||
}
|
||||
state.inFlight = false;
|
||||
state.idleWaiters.splice(0).forEach((resolve) => resolve());
|
||||
}
|
||||
|
||||
function waitForSaves(id) {
|
||||
const state = saveStates.get(id);
|
||||
if (!state?.inFlight && !state?.pending) return Promise.resolve();
|
||||
return new Promise((resolve) => state.idleWaiters.push(resolve));
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonical save contract: snapshot.id is the sole identity authority.
|
||||
* A durable id is required and is sent to the authenticated server API.
|
||||
*/
|
||||
export function saveInvestigation(snapshot, explicitId) {
|
||||
const id = snapshot?.id ?? explicitId;
|
||||
if (!snapshot || typeof snapshot !== "object" || !id) {
|
||||
return observeUnhandledRejection(Promise.reject(new Error("Investigation snapshot and id are required")));
|
||||
}
|
||||
const state = getSaveState(id);
|
||||
const promise = new Promise((resolve, reject) => {
|
||||
// A pending entry has not started yet, so replacing it safely coalesces intermediate autosaves.
|
||||
if (state.pending) {
|
||||
state.pending.snapshot = snapshot;
|
||||
state.pending.waiters.push({ resolve, reject });
|
||||
} else {
|
||||
state.pending = { snapshot, waiters: [{ resolve, reject }] };
|
||||
}
|
||||
});
|
||||
if (!state.inFlight) {
|
||||
state.inFlight = true;
|
||||
void drainSaveState(id, state);
|
||||
}
|
||||
return observeUnhandledRejection(promise);
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonical load contract: select by durable id through the authenticated server API.
|
||||
*/
|
||||
export async function loadInvestigation(id) {
|
||||
if (!id) return null;
|
||||
return _load(id);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lists all durable-ID Investigation records as lightweight summaries.
|
||||
* Server persistence is the authority; legacy browser storage is not consulted.
|
||||
*/
|
||||
export async function listInvestigations() {
|
||||
return _list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Semantic restart: reset reasoning/report state within the Investigation container.
|
||||
*
|
||||
* The Investigation is NOT deleted or replaced. Its durable `id` and `scenario` (container)
|
||||
* are preserved. All reasoning-state fields are cleared so a new clean pass can begin.
|
||||
*
|
||||
* Missing/invalid id → silently no-op (does NOT fall back to legacy singleton).
|
||||
*/
|
||||
export async function restartInvestigation(id) {
|
||||
if (!id) return null;
|
||||
await waitForSaves(id);
|
||||
return _restart(id);
|
||||
}
|
||||
|
||||
@@ -4,6 +4,11 @@ const CANONICAL_KEY = "confidence-engine-investigation";
|
||||
const LEGACY_KEY = "confidence-engine-session";
|
||||
const SCHEMA_VERSION = 1;
|
||||
|
||||
import { restartSnapshot } from "../restart-investigation.js";
|
||||
|
||||
// Multi-Investigation key prefix (v0.60c)
|
||||
const INVESTIGATION_PREFIX = "confidence-engine-investigation:";
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
function safeGet(storage, key) {
|
||||
@@ -20,16 +25,34 @@ function isPlainObject(value) {
|
||||
);
|
||||
}
|
||||
|
||||
// ── loadInvestigation ────────────────────────────────────────────────
|
||||
// ── loadInvestigation (identity-aware) ───────────────────────────────
|
||||
|
||||
/**
|
||||
* Returns the persisted investigation snapshot (normalised to current schema)
|
||||
* or null when no usable state exists.
|
||||
* Returns the persisted investigation snapshot keyed by durable id
|
||||
* (normalised to current schema), or null when no usable state exists.
|
||||
*
|
||||
* When called without an id argument, reads from the legacy singleton key
|
||||
* for backward-compatible consumers that have not yet migrated.
|
||||
*/
|
||||
export function loadInvestigation() {
|
||||
export function loadInvestigation(id) {
|
||||
const storage = _getTargetStorage();
|
||||
if (!storage) return null;
|
||||
|
||||
// — identity-aware contract: select by durable id —
|
||||
if (id != null) {
|
||||
const raw = safeGet(storage, `${INVESTIGATION_PREFIX}${id}`);
|
||||
if (raw === null) return null;
|
||||
try {
|
||||
const parsed = JSON.parse(raw);
|
||||
if (isPlainObject(parsed)) {
|
||||
if (!("schemaVersion" in parsed)) parsed.schemaVersion = SCHEMA_VERSION;
|
||||
return parsed;
|
||||
}
|
||||
} catch (_) { return null; }
|
||||
return null;
|
||||
}
|
||||
|
||||
// — backward-compatible singleton path (no id supplied) —
|
||||
// 1 – canonical key first
|
||||
let raw = safeGet(storage, CANONICAL_KEY);
|
||||
if (raw !== null) {
|
||||
@@ -64,36 +87,119 @@ export function loadInvestigation() {
|
||||
return parsed;
|
||||
}
|
||||
|
||||
// ── saveInvestigation ────────────────────────────────────────────────
|
||||
// ── saveInvestigation (identity-aware) ───────────────────────────────
|
||||
|
||||
/**
|
||||
* Persists the supplied snapshot to the canonical localStorage key.
|
||||
* Persists the supplied snapshot under its durable id key.
|
||||
* The caller's object is never mutated.
|
||||
*
|
||||
* When called without an id argument, writes to the legacy singleton key
|
||||
* for backward-compatible consumers that have not yet migrated.
|
||||
*/
|
||||
export function saveInvestigation(snapshot) {
|
||||
export function saveInvestigation(snapshot, id) {
|
||||
const storage = _getTargetStorage();
|
||||
if (!storage) return; // silently no-op in non-browser
|
||||
|
||||
try {
|
||||
const record = JSON.parse(JSON.stringify(snapshot));
|
||||
record.schemaVersion = SCHEMA_VERSION;
|
||||
_persist(storage, CANONICAL_KEY, JSON.stringify(record));
|
||||
const key = id != null ? `${INVESTIGATION_PREFIX}${id}` : CANONICAL_KEY;
|
||||
_persist(storage, key, JSON.stringify(record));
|
||||
} catch (_) { /* storage errors must not crash caller */ }
|
||||
}
|
||||
|
||||
// ── clearInvestigation ───────────────────────────────────────────────
|
||||
// ── clearInvestigation (identity-aware) ──────────────────────────────
|
||||
|
||||
/** Removes the canonical investigation key and the legacy session key. */
|
||||
export function clearInvestigation() {
|
||||
/**
|
||||
* Removes a specific investigation by durable id.
|
||||
* When called without an id, clears the legacy singleton keys only.
|
||||
*/
|
||||
export function clearInvestigation(id) {
|
||||
const storage = _getTargetStorage();
|
||||
if (!storage) return;
|
||||
try { storage.removeItem(CANONICAL_KEY); } catch (_) {}
|
||||
if (id != null) {
|
||||
try { storage.removeItem(`${INVESTIGATION_PREFIX}${id}`); } catch (_) {}
|
||||
} else {
|
||||
try { storage.removeItem(CANONICAL_KEY); } catch (_) {}
|
||||
const legacyStorage = _getLegacyStorage();
|
||||
if (legacyStorage) {
|
||||
try { legacyStorage.removeItem(LEGACY_KEY); } catch (_) {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Also remove the legacy sessionStorage key so it cannot resurrect stale
|
||||
// state after loadInvestigation falls through from missing canonical key.
|
||||
const legacyStorage = _getLegacyStorage();
|
||||
if (!legacyStorage) return;
|
||||
try { legacyStorage.removeItem(LEGACY_KEY); } catch (_) {}
|
||||
// ── listInvestigations (durable-ID only) ───────────────────────
|
||||
|
||||
/**
|
||||
* Enumerates all durable-ID Investigation records and returns lightweight summaries.
|
||||
* Skips malformed entries; excludes legacy singleton, sessionStorage state, unrelated keys.
|
||||
*/
|
||||
export function listInvestigations() {
|
||||
const storage = _getTargetStorage();
|
||||
if (!storage) return [];
|
||||
|
||||
const summaries = [];
|
||||
|
||||
for (let i = 0; i < storage.length; i++) {
|
||||
const key = storage.key(i);
|
||||
if (!key || !key.startsWith(INVESTIGATION_PREFIX)) continue;
|
||||
|
||||
try {
|
||||
const raw = safeGet(storage, key);
|
||||
if (!raw) continue;
|
||||
const record = JSON.parse(raw);
|
||||
if (!isPlainObject(record)) continue;
|
||||
if (!record.id) continue;
|
||||
|
||||
summaries.push({
|
||||
id: record.id,
|
||||
scenario: record.scenario ?? null,
|
||||
updatedAt: record.updatedAt ?? null,
|
||||
investigationRevision: record.investigationRevision ?? 0,
|
||||
reportExists: !!record.investigationReport,
|
||||
reportGeneratedFromRevision: record.investigationReport
|
||||
? record.investigationReport.generatedFromRevision ?? null
|
||||
: null,
|
||||
});
|
||||
} catch (_) {
|
||||
// Malformed durable-ID entry — skip silently
|
||||
}
|
||||
}
|
||||
|
||||
summaries.sort((a, b) => {
|
||||
const ta = a.updatedAt ?? "";
|
||||
const tb = b.updatedAt ?? "";
|
||||
if (ta > tb) return -1;
|
||||
if (ta < tb) return 1;
|
||||
return a.id > b.id ? -1 : 1;
|
||||
});
|
||||
|
||||
return summaries;
|
||||
}
|
||||
|
||||
// ── restartInvestigation (semantic reset within container) ───────────
|
||||
|
||||
/**
|
||||
* Resets the Investigation to a clean state suitable for a new reasoning pass.
|
||||
*
|
||||
* Preserved: id, scenario, schemaVersion (container-level identity/framing).
|
||||
* Reset: situationGraph → null, selectedQuestion → null, summary → null,
|
||||
* focusedContributions → [], findings → [], investigationReport → null,
|
||||
* investigationRevision → 0, updatedAt → new ISO timestamp.
|
||||
*/
|
||||
export function restartInvestigation(id) {
|
||||
const storage = _getTargetStorage();
|
||||
if (!storage) return;
|
||||
|
||||
try {
|
||||
const key = id != null ? `${INVESTIGATION_PREFIX}${id}` : CANONICAL_KEY;
|
||||
const raw = safeGet(storage, key);
|
||||
if (raw === null) return; // nothing to restart
|
||||
const record = JSON.parse(raw);
|
||||
if (!isPlainObject(record)) return;
|
||||
|
||||
_persist(storage, key, JSON.stringify(restartSnapshot(record)));
|
||||
} catch (_) { /* storage errors must not crash caller */ }
|
||||
}
|
||||
|
||||
// ── internals ────────────────────────────────────────────────────────
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
async function request(path, options) {
|
||||
const response = await fetch(path, options);
|
||||
const body = await response.json().catch(() => ({}));
|
||||
if (!response.ok) {
|
||||
throw new Error(body.error || "Investigation persistence request failed");
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
export async function saveInvestigation(snapshot, id) {
|
||||
const body = await request("/api/investigations", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ id, snapshot }),
|
||||
});
|
||||
return body.snapshot ?? null;
|
||||
}
|
||||
|
||||
export async function loadInvestigation(id) {
|
||||
const url = `/api/investigations/${encodeURIComponent(id)}`;
|
||||
const response = await fetch(url);
|
||||
if (!response.ok) {
|
||||
if (response.status === 404) return null;
|
||||
const body = await response.json().catch(() => ({}));
|
||||
throw new Error(body.error || "Investigation persistence request failed");
|
||||
}
|
||||
const body = await response.json();
|
||||
return body.snapshot ?? null;
|
||||
}
|
||||
|
||||
export async function listInvestigations() {
|
||||
const body = await request("/api/investigations");
|
||||
return body.investigations ?? [];
|
||||
}
|
||||
|
||||
export async function restartInvestigation(id) {
|
||||
const body = await request(`/api/investigations/${encodeURIComponent(id)}/restart`, {
|
||||
method: "POST",
|
||||
});
|
||||
return body.snapshot ?? null;
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
export function restartSnapshot(snapshot, now = new Date().toISOString()) {
|
||||
return {
|
||||
...snapshot,
|
||||
situationGraph: null,
|
||||
selectedQuestion: null,
|
||||
summary: null,
|
||||
focusedContributions: [],
|
||||
findings: [],
|
||||
investigationReport: null,
|
||||
investigationRevision: 0,
|
||||
updatedAt: now,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
import {
|
||||
createServerSupabaseClient,
|
||||
getAuthenticatedUser,
|
||||
} from "@/lib/supabase/server.js";
|
||||
import { restartSnapshot } from "./restart-investigation.js";
|
||||
|
||||
const SCHEMA = "confidence_engine";
|
||||
const TABLE = "investigations";
|
||||
|
||||
async function getAuthenticatedPersistenceContext() {
|
||||
const user = await getAuthenticatedUser();
|
||||
if (!user) return null;
|
||||
return { user, supabase: createServerSupabaseClient() };
|
||||
}
|
||||
|
||||
function throwIfDatabaseError(error) {
|
||||
if (error) throw new Error("Investigation persistence request failed");
|
||||
}
|
||||
|
||||
export async function saveInvestigation(snapshot, id = snapshot?.id) {
|
||||
if (!snapshot || typeof snapshot !== "object" || !id) {
|
||||
throw new Error("Investigation snapshot and id are required");
|
||||
}
|
||||
const context = await getAuthenticatedPersistenceContext();
|
||||
if (!context) return null;
|
||||
|
||||
const { data, error } = await context.supabase
|
||||
.schema(SCHEMA)
|
||||
.from(TABLE)
|
||||
.upsert({ id, user_id: context.user.id, snapshot }, { onConflict: "id" })
|
||||
.select("id, snapshot, created_at, updated_at")
|
||||
.single();
|
||||
throwIfDatabaseError(error);
|
||||
return data?.snapshot ?? null;
|
||||
}
|
||||
|
||||
export async function loadInvestigation(id) {
|
||||
if (!id) return null;
|
||||
const context = await getAuthenticatedPersistenceContext();
|
||||
if (!context) return null;
|
||||
|
||||
const { data, error } = await context.supabase
|
||||
.schema(SCHEMA)
|
||||
.from(TABLE)
|
||||
.select("snapshot")
|
||||
.eq("id", id)
|
||||
.maybeSingle();
|
||||
throwIfDatabaseError(error);
|
||||
return data?.snapshot ?? null;
|
||||
}
|
||||
|
||||
export async function listInvestigations() {
|
||||
const context = await getAuthenticatedPersistenceContext();
|
||||
if (!context) return [];
|
||||
|
||||
const { data, error } = await context.supabase
|
||||
.schema(SCHEMA)
|
||||
.from(TABLE)
|
||||
.select("id, snapshot, created_at, updated_at")
|
||||
.order("updated_at", { ascending: false });
|
||||
throwIfDatabaseError(error);
|
||||
return (data ?? []).map((record) => ({
|
||||
id: record.id,
|
||||
scenario: record.snapshot?.scenario ?? null,
|
||||
updatedAt: record.snapshot?.updatedAt ?? record.updated_at,
|
||||
investigationRevision: record.snapshot?.investigationRevision ?? 0,
|
||||
reportExists: !!record.snapshot?.investigationReport,
|
||||
reportGeneratedFromRevision: record.snapshot?.investigationReport?.generatedFromRevision ?? null,
|
||||
}));
|
||||
}
|
||||
|
||||
export async function restartInvestigation(id) {
|
||||
const snapshot = await loadInvestigation(id);
|
||||
if (!snapshot) return null;
|
||||
return saveInvestigation(restartSnapshot(snapshot), id);
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
import { getAuthenticatedUser } from "@/lib/supabase/server.js";
|
||||
|
||||
export function unauthorizedResponse() {
|
||||
return Response.json({ error: "Unauthorized" }, { status: 401 });
|
||||
}
|
||||
|
||||
export function withAuthenticatedApi(handler) {
|
||||
return async function authenticatedApiHandler(request, context) {
|
||||
const user = await getAuthenticatedUser();
|
||||
if (!user) return unauthorizedResponse();
|
||||
return handler(request, context);
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
"use client";
|
||||
|
||||
import { createBrowserClient } from "@supabase/ssr";
|
||||
|
||||
export function createClient() {
|
||||
return createBrowserClient(
|
||||
process.env.NEXT_PUBLIC_SUPABASE_URL,
|
||||
process.env.NEXT_PUBLIC_SUPABASE_ANON_KEY,
|
||||
);
|
||||
}
|
||||
|
||||
export function magicLinkRedirectTo(origin) {
|
||||
return `${origin}/auth/callback`;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
import { createServerClient } from "@supabase/ssr";
|
||||
import { cookies } from "next/headers";
|
||||
|
||||
function getSupabaseConfig() {
|
||||
return {
|
||||
url: process.env.NEXT_PUBLIC_SUPABASE_URL,
|
||||
key: process.env.NEXT_PUBLIC_SUPABASE_ANON_KEY,
|
||||
};
|
||||
}
|
||||
|
||||
export function createServerSupabaseClient() {
|
||||
const cookieStore = cookies();
|
||||
const { url, key } = getSupabaseConfig();
|
||||
return createServerClient(url, key, {
|
||||
cookies: {
|
||||
getAll() {
|
||||
return cookieStore.getAll();
|
||||
},
|
||||
setAll(cookiesToSet) {
|
||||
try {
|
||||
cookiesToSet.forEach(({ name, value, options }) => cookieStore.set(name, value, options));
|
||||
} catch {
|
||||
// Server Components cannot write cookies; middleware refreshes sessions.
|
||||
}
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export async function getAuthenticatedUser() {
|
||||
const { data: { user } } = await createServerSupabaseClient().auth.getUser();
|
||||
return user;
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
export const THEME_STORAGE_KEY = "confidence-engine-theme";
|
||||
|
||||
export function resolveTheme({ savedTheme, systemPrefersDark = false } = {}) {
|
||||
if (savedTheme === "light" || savedTheme === "dark") return savedTheme;
|
||||
return systemPrefersDark ? "dark" : "light";
|
||||
}
|
||||
|
||||
export function toggleTheme(theme) {
|
||||
return theme === "dark" ? "light" : "dark";
|
||||
}
|
||||
|
||||
export function readThemePreference(storage) {
|
||||
try {
|
||||
return storage?.getItem(THEME_STORAGE_KEY) ?? null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export function saveThemePreference(theme, storage) {
|
||||
storage?.setItem(THEME_STORAGE_KEY, theme);
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
import { createServerClient } from "@supabase/ssr";
|
||||
import { NextResponse } from "next/server";
|
||||
|
||||
const PUBLIC_PATHS = ["/login", "/auth", "/api/health"];
|
||||
|
||||
export async function middleware(request) {
|
||||
const pathname = request.nextUrl.pathname;
|
||||
if (PUBLIC_PATHS.some((path) => pathname === path || pathname.startsWith(`${path}/`))) {
|
||||
return NextResponse.next();
|
||||
}
|
||||
|
||||
let response = NextResponse.next({ request });
|
||||
const supabase = createServerClient(
|
||||
process.env.NEXT_PUBLIC_SUPABASE_URL,
|
||||
process.env.NEXT_PUBLIC_SUPABASE_ANON_KEY,
|
||||
{
|
||||
cookies: {
|
||||
getAll: () => request.cookies.getAll(),
|
||||
setAll(cookiesToSet) {
|
||||
cookiesToSet.forEach(({ name, value, options }) => request.cookies.set(name, value));
|
||||
response = NextResponse.next({ request });
|
||||
cookiesToSet.forEach(({ name, value, options }) => response.cookies.set(name, value, options));
|
||||
},
|
||||
},
|
||||
},
|
||||
);
|
||||
const { data: { user } } = await supabase.auth.getUser();
|
||||
if (user) return response;
|
||||
if (pathname.startsWith("/api/")) return NextResponse.json({ error: "Unauthorized" }, { status: 401 });
|
||||
const loginUrl = request.nextUrl.clone();
|
||||
loginUrl.pathname = "/login";
|
||||
loginUrl.searchParams.set("next", pathname);
|
||||
return NextResponse.redirect(loginUrl);
|
||||
}
|
||||
|
||||
export const config = { matcher: ["/((?!_next/static|_next/image|favicon.ico).*)"] };
|
||||
+4
-1
@@ -1,3 +1,6 @@
|
||||
/** @type {import('next').NextConfig} */
|
||||
const nextConfig = {};
|
||||
const nextConfig = {
|
||||
output: 'standalone',
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
|
||||
Generated
+636
-4
@@ -8,10 +8,12 @@
|
||||
"name": "confidence-engine",
|
||||
"version": "0.2.0-experimental",
|
||||
"dependencies": {
|
||||
"@supabase/ssr": "^0.12.7",
|
||||
"@supabase/supabase-js": "^2.116.0",
|
||||
"next": "^14.2.0",
|
||||
"react": "^18.3.0",
|
||||
"react-dom": "^18.3.0",
|
||||
"zod": "^3.23.0"
|
||||
"zod": "^4.5.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@playwright/test": "^1.62.1",
|
||||
@@ -23,6 +25,7 @@
|
||||
"eslint-config-next": "^14.2.0",
|
||||
"postcss": "^8.4.0",
|
||||
"tailwindcss": "^3.4.0",
|
||||
"tsx": "^4.23.13",
|
||||
"vitest": "^2.0.0"
|
||||
}
|
||||
},
|
||||
@@ -362,6 +365,23 @@
|
||||
"node": ">=12"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/netbsd-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-sSATRjPeDBg3pdgHoQfoYBob11Kk1FGa9lui5RIHZCoCkJa9QKlvl3/vKz2usCmYYjs7ymJR/2Nnsqe+Hjt5nw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"netbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/netbsd-x64": {
|
||||
"version": "0.21.5",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.21.5.tgz",
|
||||
@@ -379,6 +399,23 @@
|
||||
"node": ">=12"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/openbsd-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-AL2qJILH7lNjrDmCQDvdxMfAUIv8KMNZOvrwAQ8i8//ntL9FflhOyMJ8OZSMBb8/AWXe3/5v5S20y3zCoZWKoQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"openbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/openbsd-x64": {
|
||||
"version": "0.21.5",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.21.5.tgz",
|
||||
@@ -396,6 +433,23 @@
|
||||
"node": ">=12"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/openharmony-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-WkhYDmpTjLvGlScA1rwjRUmhl4k8oXR3cIbtqWmELgU/dFeHHlEllxDvdWcNJV9rbzCexB5vz8gtNewWLgCT7Q==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"openharmony"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/@esbuild/sunos-x64": {
|
||||
"version": "0.21.5",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.21.5.tgz",
|
||||
@@ -1269,6 +1323,110 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@supabase/auth-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/auth-js/-/auth-js-2.116.0.tgz",
|
||||
"integrity": "sha512-Cmosty12gyKGK9N3bQb+lMmuAFev5nmUzaR1AsmZHqKOAGzqX1VQzmp49CNPwOx/pw0H9Qqk4rs9yhwTlKpfDg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"tslib": "2.8.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/functions-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/functions-js/-/functions-js-2.116.0.tgz",
|
||||
"integrity": "sha512-E+VOc2QDcni/fySqkBFiZhnoB3SGydEdZgFI6/dEAGAHx6yEhB46TN9qb2wXs+E+RSzOBV0R6dasiSlw4xlZAA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"tslib": "2.8.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/phoenix": {
|
||||
"version": "0.4.5",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/phoenix/-/phoenix-0.4.5.tgz",
|
||||
"integrity": "sha512-aAn9H9ovVyeApKy11OWOrrOGq8DV68yWeH4ud2lN9fzn4aO8Zb5GLL9m1pUg9nLqIcT+ZDfAcsZe0E/nqdv2lw==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@supabase/postgrest-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/postgrest-js/-/postgrest-js-2.116.0.tgz",
|
||||
"integrity": "sha512-kGpVZTDHxFTJS3tu+rU0iTAZ+4U0bcLVjxwCk8f3gRhjw3qdCZjTBlgYvc4kGH2XccmAzbkKwXL/mrNHMGSc+A==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"tslib": "2.8.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/realtime-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/realtime-js/-/realtime-js-2.116.0.tgz",
|
||||
"integrity": "sha512-MHAnlXxi2s6yiJsZsQMfs2B3RFxeVfQWxerqYhIMqcCQV/FuY3LIeouPEkXw/ah7wUWMLYwempF9MOCUScyddg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@supabase/phoenix": "0.4.5",
|
||||
"tslib": "2.8.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/ssr": {
|
||||
"version": "0.12.7",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/ssr/-/ssr-0.12.7.tgz",
|
||||
"integrity": "sha512-wiBtEie1KkRJi9RrZWY3R2imRhX1JY7qMyUCH2z9AUk15gQebNEplM+urbCKamdxaTJLXUU6LlpkJsaxhojCEg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"cookie": "^1.0.2"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"@supabase/supabase-js": "^2.114.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/storage-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/storage-js/-/storage-js-2.116.0.tgz",
|
||||
"integrity": "sha512-6/3hR6vccBP6oGM5B6RfbwZcTCKmQOodd/ZWQdsw8yJsU5zO/a//oBL6yLnmgxcjnHSrelW8rsO7hL5DPybyUQ==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"iceberg-js": "^0.8.1",
|
||||
"tslib": "2.8.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@supabase/supabase-js": {
|
||||
"version": "2.116.0",
|
||||
"resolved": "https://registry.npmjs.org/@supabase/supabase-js/-/supabase-js-2.116.0.tgz",
|
||||
"integrity": "sha512-YyWmKXt2NspV9iO8FPnlswUFJIRnrLd3oTCb+3ZyYRuKZtBH0xCUDgnUqoyA0fGUxpM/UhfwDjYf/dht/9bp7g==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@supabase/auth-js": "2.116.0",
|
||||
"@supabase/functions-js": "2.116.0",
|
||||
"@supabase/postgrest-js": "2.116.0",
|
||||
"@supabase/realtime-js": "2.116.0",
|
||||
"@supabase/storage-js": "2.116.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"@opentelemetry/api": ">=1.0.0"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"@opentelemetry/api": {
|
||||
"optional": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/@swc/counter": {
|
||||
"version": "0.1.3",
|
||||
"resolved": "https://registry.npmjs.org/@swc/counter/-/counter-0.1.3.tgz",
|
||||
@@ -2761,6 +2919,19 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/cookie": {
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/cookie/-/cookie-1.1.1.tgz",
|
||||
"integrity": "sha512-ei8Aos7ja0weRpFzJnEA9UHJ/7XQmqglbRwnf2ATjcB9Wq874VKH9kfjjirM6UhU2/E5fFYadylyhFldcqSidQ==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
},
|
||||
"funding": {
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/express"
|
||||
}
|
||||
},
|
||||
"node_modules/cross-spawn": {
|
||||
"version": "7.0.6",
|
||||
"resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz",
|
||||
@@ -4244,6 +4415,15 @@
|
||||
"node": ">= 0.4"
|
||||
}
|
||||
},
|
||||
"node_modules/iceberg-js": {
|
||||
"version": "0.8.1",
|
||||
"resolved": "https://registry.npmjs.org/iceberg-js/-/iceberg-js-0.8.1.tgz",
|
||||
"integrity": "sha512-1dhVQZXhcHje7798IVM+xoo/1ZdVfzOMIc8/rgVSijRK38EDqOJoGula9N/8ZI5RD8QTxNQtK/Gozpr+qUqRRA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=20.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/ignore": {
|
||||
"version": "5.3.2",
|
||||
"resolved": "https://registry.npmjs.org/ignore/-/ignore-5.3.2.tgz",
|
||||
@@ -7012,6 +7192,458 @@
|
||||
"integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==",
|
||||
"license": "0BSD"
|
||||
},
|
||||
"node_modules/tsx": {
|
||||
"version": "4.23.13",
|
||||
"resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.13.tgz",
|
||||
"integrity": "sha512-BL5MGkRln6aDYhb0xbQlEAGw743BaZYWdbWtdJOBriYJboKgUUYCadFp2/FpBBZquBC/ezNBn7wMMPx7FDZUDw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"esbuild": "~0.28.0"
|
||||
},
|
||||
"bin": {
|
||||
"tsx": "dist/cli.mjs"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"fsevents": "~2.3.3"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/aix-ppc64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.2.tgz",
|
||||
"integrity": "sha512-XExcO+dvLKvVtNTibSTBej1NCAbaGhWn9Ww1ZPx80qsahhPFe/8jgWP0IchNe0F3HwkU7n8ejhH8bjonqht8mQ==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"aix"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/android-arm": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.2.tgz",
|
||||
"integrity": "sha512-kXXoiPVVGQcnIYGOeaovwOURpniDBpSq4A03qkQ+BMQqtGG6HYap3xne9C1O1yo4TR3qxlCX5IqqmX6fFo2Lqg==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"android"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/android-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-5YfKeeI8qWfBZIX+u2xZC3Zlb3Os/gLS2sbEKM+I4ZOcsWmHS2WLysCcQZDAFRslDUU5Oiq44gf6PYN1vGwG5A==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"android"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/android-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-O387ite7SzUyCcy3JQX4P4bLtEA7bLLkx+esve5JHnyYfNTxcVpXZo9jhdB0lTKN44gztELTdU7nS8Nr16Fs1Q==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"android"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/darwin-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-n4KqkOQrraxHJcgjM1RvwbigfQKIKJVpM7xp+KsxiyUSrRdIXnt73VhrPAx0fV44hgfmIVKjxMN9J1t5jySVkw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/darwin-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-uq6suIWYP37qzGddBKPw5QEQPi6HiLGsO7UmkpfyaYNQ3D+rN6w6WfwH+nuqcGXWvawGwxOEroO4YGnFh95azw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"darwin"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/freebsd-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-n+I0BTSRIoy+d6RPKnEVwql5UwBJolytvY4mAOIEJorKlqgPII8ix6slVVrfZ5Tnj7glIZvloylbB/EJPMWEXw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"freebsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/freebsd-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-78XJTJkvPs0kz2w61301PJjXl4g7q3JqiYMZ/M/yVI73EHBrCRTgkhu9oqG7vPqq+a/yadEW8aD+agKlk5xrmg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"freebsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-arm": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.2.tgz",
|
||||
"integrity": "sha512-XlDnu2q5yoqems+xay6wSAcg9DDD7K9RLKZEBOMZm3ckNpJBvOX20tSfby8KfrrhINDyv9V2YVZKY/SpoGJI8w==",
|
||||
"cpu": [
|
||||
"arm"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-pW4AC0P3it8c7do9MVM4p51FzHzdM/TZrerurgRcHJ2WTa1VQ1CIq18xncfpBJw4ojkiZZrKW2yIBWBP92j6Ug==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-ia32": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.2.tgz",
|
||||
"integrity": "sha512-CYbnj78HsIeA+DhgUKgFCfvNsTHFhMMrinUrMZpDXJXKN8T3XViTZ/+wtHeVxEWY8ewSzTFN+nRmSwO2tZaLUQ==",
|
||||
"cpu": [
|
||||
"ia32"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-loong64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.2.tgz",
|
||||
"integrity": "sha512-buwkd8nsph4R+ajRvw0qM5Hja/TXQow3ptzWO2EbG/cqcIkHloRrdlBtQlshyYGTNFvfkfJ5tpPLVkY4DtsPfQ==",
|
||||
"cpu": [
|
||||
"loong64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-mips64el": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.2.tgz",
|
||||
"integrity": "sha512-ZVykbDyk7519VwiNb9Lcj9m8XM6v5V9uKPvrEMkkEedVewf+0itkhahp4HDpgERXhwLRpWFypsGbG/J8s0QjJA==",
|
||||
"cpu": [
|
||||
"mips64el"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-ppc64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.2.tgz",
|
||||
"integrity": "sha512-CAXl+Dtd9UUuJd8pKKdwh6MLm3MUMiqMPmhZ3tTSXPqfyQ3vDl6R5hZdZ/kYojK4ofXtdfSv1tFq8XzWx3heNQ==",
|
||||
"cpu": [
|
||||
"ppc64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-riscv64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.2.tgz",
|
||||
"integrity": "sha512-GeXCej4IQtU1B+QlDV8W/RRvbzI3O/Stss+/bCXv4lZls5WGRtu2a+3JkA3i4qIUlMXpcHebWpF8AkJhATowuA==",
|
||||
"cpu": [
|
||||
"riscv64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-s390x": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.2.tgz",
|
||||
"integrity": "sha512-3H1weTYZPxt/WOhByszQZybS9w5lKzUn1FDMsgEChbHWQwHYQQRfBxgCcZvPhjHfKyJjIievvMmEUawJrdY9Dg==",
|
||||
"cpu": [
|
||||
"s390x"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/linux-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-4xTZr1FUmSoQW4XIWmit3tzQrUTZM+N3P0XV8xROKYF50XfI7xeO90+1bZvNwxIufQ9hDQVRJH5YhgPVF8A/HQ==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"linux"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/netbsd-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-lqnzCV+mM0gIADaKihiCg6ifgfU2L3h5E33rNQBN1Y4MaVGnzryzmvvf7UHxprpQdE8hpqLolJ9Rl+SkIRDpyw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"netbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/openbsd-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-QtiuPytchRyC4rwUKhexJdQKvDuZ6hWloi3igqPQNUJCS1/v9EiO3UTOXR6A3FoMo4fnAKbWJdqaIwhOzh8qEw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"openbsd"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/sunos-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-GPMSkTOtMnv2U2F8gxe4Io6qmVs+YKyp832Etqqxr0hFngmXQ3rzwytelm3GIn7T4VviRUlf3sOgBOiTdvaf7g==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"sunos"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/win32-arm64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.2.tgz",
|
||||
"integrity": "sha512-PIhhEkE9uPBleRBrQEJpUn7MBnibZzbGzYWPmY3x+YoVg/95zbjB4CxPPOQ8l5tYYM4mMaCthF8/1DIfBQQyWQ==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/win32-ia32": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.2.tgz",
|
||||
"integrity": "sha512-YmJbfTlvU7Sdn9BB+4PRES4oB6pxgS37MAONj+hBr/cpXS1aBPKXxNnDbu+QCWPj0o9dgyxeq79g6c5P8KeuYA==",
|
||||
"cpu": [
|
||||
"ia32"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/@esbuild/win32-x64": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.2.tgz",
|
||||
"integrity": "sha512-5ebpxr3nWMzrL/rnUI755Jkuee0bHL/Gq0WTF9lvcpv73wAp5eu8MfBUgWK9bhWvZjj7yX8etf/8tI8Ney695g==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"os": [
|
||||
"win32"
|
||||
],
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/tsx/node_modules/esbuild": {
|
||||
"version": "0.28.2",
|
||||
"resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.28.2.tgz",
|
||||
"integrity": "sha512-HKVLS8dvII+xoKW9kmqxbRKrnWEXfJJr/FZhhJmiqIB0e053QNYFqOBouTMO/k5sID4MvCiUCvv8b9M4h32wIA==",
|
||||
"dev": true,
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"bin": {
|
||||
"esbuild": "bin/esbuild"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@esbuild/aix-ppc64": "0.28.2",
|
||||
"@esbuild/android-arm": "0.28.2",
|
||||
"@esbuild/android-arm64": "0.28.2",
|
||||
"@esbuild/android-x64": "0.28.2",
|
||||
"@esbuild/darwin-arm64": "0.28.2",
|
||||
"@esbuild/darwin-x64": "0.28.2",
|
||||
"@esbuild/freebsd-arm64": "0.28.2",
|
||||
"@esbuild/freebsd-x64": "0.28.2",
|
||||
"@esbuild/linux-arm": "0.28.2",
|
||||
"@esbuild/linux-arm64": "0.28.2",
|
||||
"@esbuild/linux-ia32": "0.28.2",
|
||||
"@esbuild/linux-loong64": "0.28.2",
|
||||
"@esbuild/linux-mips64el": "0.28.2",
|
||||
"@esbuild/linux-ppc64": "0.28.2",
|
||||
"@esbuild/linux-riscv64": "0.28.2",
|
||||
"@esbuild/linux-s390x": "0.28.2",
|
||||
"@esbuild/linux-x64": "0.28.2",
|
||||
"@esbuild/netbsd-arm64": "0.28.2",
|
||||
"@esbuild/netbsd-x64": "0.28.2",
|
||||
"@esbuild/openbsd-arm64": "0.28.2",
|
||||
"@esbuild/openbsd-x64": "0.28.2",
|
||||
"@esbuild/openharmony-arm64": "0.28.2",
|
||||
"@esbuild/sunos-x64": "0.28.2",
|
||||
"@esbuild/win32-arm64": "0.28.2",
|
||||
"@esbuild/win32-ia32": "0.28.2",
|
||||
"@esbuild/win32-x64": "0.28.2"
|
||||
}
|
||||
},
|
||||
"node_modules/type-check": {
|
||||
"version": "0.4.0",
|
||||
"resolved": "https://registry.npmjs.org/type-check/-/type-check-0.4.0.tgz",
|
||||
@@ -7646,9 +8278,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/zod": {
|
||||
"version": "3.25.76",
|
||||
"resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz",
|
||||
"integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==",
|
||||
"version": "4.5.4",
|
||||
"resolved": "https://registry.npmjs.org/zod/-/zod-4.5.4.tgz",
|
||||
"integrity": "sha512-sC95tT5iHHH9gtpj6A81kh+NEaRAUFN+qlUPDUbRfOMvNf5QCBqsb3WgvnpVtK5Y+4UfA6KqufotuTvMGiTlsA==",
|
||||
"license": "MIT",
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/colinhacks"
|
||||
|
||||
+5
-1
@@ -8,15 +8,18 @@
|
||||
"dev": "next dev",
|
||||
"build": "next build",
|
||||
"start": "next start",
|
||||
"experiment:start-case": "tsx scripts/start-case-experiment-helper.cjs",
|
||||
"lint": "next lint",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest"
|
||||
},
|
||||
"dependencies": {
|
||||
"@supabase/ssr": "^0.12.7",
|
||||
"@supabase/supabase-js": "^2.116.0",
|
||||
"next": "^14.2.0",
|
||||
"react": "^18.3.0",
|
||||
"react-dom": "^18.3.0",
|
||||
"zod": "^3.23.0"
|
||||
"zod": "^4.5.4"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@playwright/test": "^1.62.1",
|
||||
@@ -28,6 +31,7 @@
|
||||
"eslint-config-next": "^14.2.0",
|
||||
"postcss": "^8.4.0",
|
||||
"tailwindcss": "^3.4.0",
|
||||
"tsx": "^4.23.13",
|
||||
"vitest": "^2.0.0"
|
||||
}
|
||||
}
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 70 KiB |
@@ -0,0 +1,179 @@
|
||||
You are a neutral analyst performing evidence-based situation reconstruction.
|
||||
|
||||
## Rules
|
||||
|
||||
1. Do NOT invent facts, context or causes. Only include information present in the scenario. For decomposition structure — splitting a concept into child categories — each child's distinguishing semantic content must be directly supported by meaning supplied by the user in their text (semantic-equivalent paraphrase is permitted; literal word-for-word matching is not required). A decomposition child is NOT permitted if producing it requires adding a new actor, category, segment, mechanism, cause, process, subtype, condition, system behaviour, operational distinction, or equivalent semantic proposition that the user did not supply. The following are NOT sufficient to justify creating new decomposition structure: plausible, likely, domain-typical, possibly implied, clearly implied, worth investigating, could be relevant, might explain. Those standards may inform a provisional interpretation but never decomposition.
|
||||
|
||||
2. First determine what kind of input has been supplied. Use only these classification types:
|
||||
observed_problem, unexplained_change, contradiction, decision_request, causal_claim,
|
||||
reported_claim, fault_report, ambiguous_statement, question, desired_outcome,
|
||||
insufficient_context, other
|
||||
|
||||
3. Choose reasoning modes from:
|
||||
establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate,
|
||||
validate_measurement, validate_claim, investigate_contradiction, clarify_meaning,
|
||||
decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other
|
||||
|
||||
4. Look for anchors: actor, system or object, expected outcome, observed outcome,
|
||||
previous state, current state, difference between groups, change over time, measurement,
|
||||
evidence source, proposed action.
|
||||
|
||||
5. Identify meaningful differences (e.g., some succeed while others fail; revenue rises while cash falls).
|
||||
|
||||
6. Keep multiple plausible interpretations separate where the evidence does not distinguish them. Model-generated possible interpretations are permitted ONLY when they satisfy all of the following conditions: (a) supported by supplied evidence; (b) clearly represented in plausibleInterpretations; (c) explicitly marked as provisional; (d) their supporting evidence is identified; (e) NOT presented as observation, supplied unknown, transition, relationship, or established cause. A plausible interpretation must NEVER be smuggled into decomposition structure. Do NOT generate plausible interpretations merely to fill a list — return an empty array [] if the evidence does not support useful, distinct interpretations.
|
||||
|
||||
7. When the user explicitly names multiple distinct possible explanations, causes, constraints, or dependencies for the situation, preserve those user-stated alternatives as separate importantUnknowns when they can sensibly be investigated independently. Do not collapse them into one "which factor", "relative contribution", or equivalent umbrella unknown. Do not turn a user-stated possibility into an asserted plausible interpretation — preserve its uncertain status. Only split concepts when the user has presented materially distinct dimensions that each warrant independent investigation.
|
||||
|
||||
8. Distinguish: what was said / what it may mean / why it may have been said.
|
||||
|
||||
9. If input is too ambiguous or contains no useful operational anchors, say so and ask for
|
||||
the single piece of context that would best distinguish plausible interpretations.
|
||||
|
||||
10. Do not split concepts merely to increase the number of unknowns — only separate when the user has presented materially distinct dimensions worth independent investigation.
|
||||
|
||||
## Preserve supplied relationships
|
||||
|
||||
Reconstruction must preserve not only supplied semantic units but also relationships the user supplies between them. Preserved relationship types include: dependency, condition, comparison, alternative, constraint, sequence, reported causal claim, decision contingency. Splitting two supplied concepts must NOT erase the relationship between them. A supplied unresolved decision dependency with no dedicated schema field MUST be preserved explicitly in summary and as an importantUnknown whose wording states the dependency, keeping its uncertainty intact. Do not invent new schema fields for this purpose.
|
||||
|
||||
## Explicit stop boundary for decomposition
|
||||
|
||||
Once the supplied meaning of an observation, uncertainty, relationship or transition has been faithfully represented, STOP. Do not recursively decompose unless the supplied user material itself contains further distinct semantic structure. For each candidate decomposition child: identify the supplied meaning that supports its distinguishing content — if that content adds no new semantic proposition beyond what the user supplied, allow it; if it requires adding new semantic content, stop and do not create the child.
|
||||
|
||||
## Normalisation and rate reasoning (apply whenever applicable)
|
||||
|
||||
When the scenario mentions counts, totals, frequencies, or volumes alongside changes in
|
||||
scale, volume, exposure, time, population, or output:
|
||||
|
||||
- ALWAYS consider whether a denominator or exposure metric is needed to normalise the count.
|
||||
- Distinguish between absolute count (total number observed) and rate (count per unit of exposure).
|
||||
- Two metrics rising at similar percentages does NOT imply that quality, performance, or safety
|
||||
has worsened — production growth may outpace complaint growth, meaning the per-unit rate
|
||||
could be stable or even improved.
|
||||
- Identify the possible denominator explicitly (e.g., "per unit produced", "per customer served",
|
||||
"per hour of operation").
|
||||
- State clearly: "The absolute count changed by X%, but without knowing the denominator we cannot
|
||||
determine whether the rate per unit has worsened, stayed stable, or improved."
|
||||
- Avoid treating correlation between two rising counts as evidence of a causal relationship.
|
||||
|
||||
## Interpretation discipline
|
||||
|
||||
- Do NOT generate plausible interpretations merely to fill a list. If the evidence does not
|
||||
support useful, distinct interpretations, return an empty array [].
|
||||
- Only include an interpretation when there is specific evidence that makes it distinguishable
|
||||
from alternatives and worth evaluating further.
|
||||
- Rank all reconstruction details by importance:
|
||||
- critical: essential to resolving the situation; without it conclusions cannot be drawn
|
||||
- important: materially affects understanding of the situation
|
||||
- supporting: adds context but not critical
|
||||
- incidental: minor detail, unlikely to affect conclusions
|
||||
|
||||
## Next question discipline
|
||||
|
||||
- Generate exactly ONE next question. Do NOT combine multiple questions.
|
||||
- The first and only question should target the single most useful missing comparison or data point.
|
||||
- Prefer narrow, specific questions over broad compound questions.
|
||||
- When counts have changed alongside scale/exposure, the highest-value question typically targets
|
||||
the rate-per-unit or equivalent normalised metric.
|
||||
- Do NOT generate speculative interpretations merely to justify a question.
|
||||
|
||||
## Confidence scale
|
||||
|
||||
- low — weak evidence, speculation, or missing information
|
||||
- medium — reasonable inference from available evidence
|
||||
- high — strong evidence, direct observation, or confirmed fact
|
||||
|
||||
## Importance scale (evidence records)
|
||||
|
||||
- incidental — minor detail, unlikely to affect conclusions
|
||||
- supporting — adds context but not critical
|
||||
- important — materially affects understanding of the situation
|
||||
- critical — essential to resolving the situation; without it conclusions cannot be drawn
|
||||
|
||||
## Expected information value (next question)
|
||||
|
||||
- low — marginally useful even if answered
|
||||
- medium — meaningfully clarifies the situation
|
||||
- high — would significantly distinguish between plausible explanations or fill a gap in understanding
|
||||
|
||||
## Next question selection criteria
|
||||
|
||||
Prefer questions that:
|
||||
- clarify a major difference
|
||||
- establish a baseline
|
||||
- explain an important transition
|
||||
- test an unsupported claim
|
||||
- distinguish between plausible explanations
|
||||
- request measurable evidence
|
||||
- identify who or what is affected
|
||||
- establish timing
|
||||
|
||||
Avoid questions that:
|
||||
- have already been answered
|
||||
- assume a cause
|
||||
- jump to a solution
|
||||
- ask about motive before the observable situation is understood
|
||||
- focus on incidental wording
|
||||
- are too broad to produce useful information
|
||||
- combine many unrelated questions
|
||||
|
||||
## Output format — return this exact JSON structure
|
||||
|
||||
Return a JSON object with exactly these four top-level keys (use **camelCase**):
|
||||
|
||||
```json
|
||||
{
|
||||
"inputClassification": {
|
||||
"primaryType": "<one of: observed_problem, unexplained_change, contradiction, decision_request, causal_claim, reported_claim, fault_report, ambiguous_statement, question, desired_outcome, insufficient_context, other>",
|
||||
"secondaryTypes": ["<optional additional types from the same list>"],
|
||||
"reasoningModes": ["<one or more of: establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate, validate_measurement, validate_claim, investigate_contradiction, clarify_meaning, decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other>"],
|
||||
"classificationReason": "<brief explanation of why you chose the primary type>",
|
||||
"confidence": "<low | medium | high>"
|
||||
},
|
||||
"reconstruction": {
|
||||
"summary": "<one-sentence overview of the situation>",
|
||||
"actors": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"systemsOrObjects": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"expectedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"observedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"differences": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"knownTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}],
|
||||
"unexplainedTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "..."}],
|
||||
"contradictions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"importantUnknowns": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"plausibleInterpretations": [{"id": "...", "description": "...", "supportingEvidenceIds": ["<ids that support this interpretation>"], "assumptionsRequired": [], "confidence": "<low|medium|high>"}]
|
||||
},
|
||||
"evidence": [
|
||||
{
|
||||
"id": "<any unique string>",
|
||||
"description": "...",
|
||||
"evidenceType": "<direct_observation | reported_statement | interpretation | assumption | inferred_relationship>",
|
||||
"source": "<optional — who/where this came from>",
|
||||
"attribution": null,
|
||||
"confidence": "<low | medium | high>",
|
||||
"importance": "<incidental | supporting | important | critical>"
|
||||
}
|
||||
],
|
||||
"nextQuestion": {
|
||||
"id": "<any unique string>",
|
||||
"question": "<one precise question>",
|
||||
"targets": ["<what this question targets — e.g. 'actor', 'system', 'expectedOutcome'>"],
|
||||
"reason": "<why answering this is important>",
|
||||
"expectedInformationValue": "<low | medium | high>",
|
||||
"reasoningMode": "<optional reasoning mode from the list above>"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
CRITICAL RULES for JSON output:
|
||||
1. Use **exactly** the key names shown above (camelCase, no snake_case).
|
||||
2. The four top-level keys must be: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`.
|
||||
3. Do NOT invent new top-level keys (no `anchors`, `confidence` at top level, `meaningful_differences`, etc.).
|
||||
4. Keep `actors`, `systemsOrObjects`, `expectedStates`, `observedStates`, `differences`, `contradictions`, `importantUnknowns` as arrays even if empty: [].
|
||||
5. Keep `plausibleInterpretations` as an array (can be []), same for `knownTransitions` and `unexplainedTransitions`.
|
||||
6. Each object in arrays must have at least `id`, `description`, `confidence`.
|
||||
7. **evidenceType**: classify each evidence item clearly as either a direct observation, a reported statement, an interpretation, an assumption, or an inferred relationship. Do not treat raw counts as proof of causal relationships — they may be inferred relationships only when supported by explicit reasoning about denominators or rates.
|
||||
|
||||
Scenario:
|
||||
{{SCENARIO}}
|
||||
|
||||
Return ONLY the JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.
|
||||
@@ -0,0 +1,180 @@
|
||||
You are a neutral analyst performing evidence-based situation reconstruction.
|
||||
|
||||
## Rules
|
||||
|
||||
1. Do NOT invent facts, context or causes. Only include information present in the scenario. For decomposition structure — splitting a concept into child categories — each child's distinguishing semantic content must be directly supported by meaning supplied by the user in their text (semantic-equivalent paraphrase is permitted; literal word-for-word matching is not required). A decomposition child is NOT permitted if producing it requires adding a new actor, category, segment, mechanism, cause, process, subtype, condition, system behaviour, operational distinction, or equivalent semantic proposition that the user did not supply. The following are NOT sufficient to justify creating new decomposition structure: plausible, likely, domain-typical, possibly implied, clearly implied, worth investigating, could be relevant, might explain. Those standards may inform a provisional interpretation but never decomposition.
|
||||
|
||||
2. First determine what kind of input has been supplied. Use only these classification types:
|
||||
observed_problem, unexplained_change, contradiction, decision_request, causal_claim,
|
||||
reported_claim, fault_report, ambiguous_statement, question, desired_outcome,
|
||||
insufficient_context, other
|
||||
|
||||
3. Choose reasoning modes from:
|
||||
establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate,
|
||||
validate_measurement, validate_claim, investigate_contradiction, clarify_meaning,
|
||||
decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other
|
||||
|
||||
4. Look for anchors: actor, system or object, expected outcome, observed outcome,
|
||||
previous state, current state, difference between groups, change over time, measurement,
|
||||
evidence source, proposed action.
|
||||
|
||||
5. Identify meaningful differences (e.g., some succeed while others fail; revenue rises while cash falls).
|
||||
|
||||
5a. Preserve materially distinct supplied outcome categories. When the user explicitly describes different kinds of observed outcome within an aggregate result, preserve that distinction when the categories could imply different underlying problems, investigations, or relevance to a contemplated intervention. Do not collapse them into a generic umbrella outcome merely because they contribute to the same aggregate count. Preserve only distinctions supplied by the user; do not invent additional categories, mechanisms, causes, or subtypes.
|
||||
|
||||
6. Keep multiple plausible interpretations separate where the evidence does not distinguish them. Model-generated possible interpretations are permitted ONLY when they satisfy all of the following conditions: (a) supported by supplied evidence; (b) clearly represented in plausibleInterpretations; (c) explicitly marked as provisional; (d) their supporting evidence is identified; (e) NOT presented as observation, supplied unknown, transition, relationship, or established cause. A plausible interpretation must NEVER be smuggled into decomposition structure. Do NOT generate plausible interpretations merely to fill a list — return an empty array [] if the evidence does not support useful, distinct interpretations.
|
||||
|
||||
7. When the user explicitly names multiple distinct possible explanations, causes, constraints, or dependencies for the situation, preserve those user-stated alternatives as separate importantUnknowns when they can sensibly be investigated independently. Do not collapse them into one "which factor", "relative contribution", or equivalent umbrella unknown. Do not turn a user-stated possibility into an asserted plausible interpretation — preserve its uncertain status. Only split concepts when the user has presented materially distinct dimensions that each warrant independent investigation.
|
||||
|
||||
8. Distinguish: what was said / what it may mean / why it may have been said.
|
||||
|
||||
9. If input is too ambiguous or contains no useful operational anchors, say so and ask for the single piece of context that would best distinguish plausible interpretations.
|
||||
|
||||
10. Do not split concepts merely to increase the number of unknowns — only separate when the user has presented materially distinct dimensions worth independent investigation.
|
||||
|
||||
## Preserve supplied relationships
|
||||
|
||||
Reconstruction must preserve not only supplied semantic units but also relationships the user supplies between them. Supplied semantic units remain represented in the existing reconstruction arrays. Supplied relationships between those units MUST be represented in `reconstruction.relationships`, using `id`, `fromId`, `toId`, `relationship`, `description`, and `confidence`.
|
||||
|
||||
Each relationship `fromId` and `toId` MUST reference IDs of semantic units already present elsewhere in `reconstruction`. Do NOT invent a relationship endpoint merely to complete a relationship. Preserve unresolved relationships as unresolved or provisional when the user leaves them unresolved. Keep useful relationship prose in `summary` where appropriate, but summary prose is not the authoritative representation of graph-critical relationships. Do not introduce an action recommendation or steering.
|
||||
|
||||
Complete the semantic-unit collections first; only then construct relationships. For every relationship, copy `fromId` and `toId` exactly from IDs already emitted in a reconstruction semantic-unit collection—never generate a new relationship-only ID or guess one from a description. If a supplied relationship needs a legitimate semantic concept not yet represented, add that semantic unit to the appropriate existing collection first, then reference its exact ID; if that concept cannot legitimately be represented under the provenance rules, omit the relationship rather than substituting a related endpoint. Before returning JSON, check every relationship by finding both endpoint IDs in emitted semantic units. The structural need for an endpoint is not evidence that its meaning is justified.
|
||||
|
||||
Read every relationship literally as `fromId → relationship → toId`. For directional types, endpoint meanings and the description MUST agree: `A depends_on B`, `A causes B`, `A may_cause B`, `A supports B`, and `A weakens B` mean `fromId=A` and `toId=B`. Do not assign stronger directionality to `contradicts`, `compares_with`, or `other`; `compares_with` may use either endpoint order. Before emitting a relationship, read it back as “FROM [relationship] TO”. If that reading contradicts the description, reverse the endpoints or choose the correct relationship type.
|
||||
|
||||
When the user explicitly makes a contemplated action, intervention, investment, or decision contingent on understanding an unresolved condition, preserve that dependency at the level supplied: represent both the unresolved appropriateness, fit, or timing of the intervention and the unresolved condition, then declare `intervention appropriateness depends_on unresolved condition`. Narrower supported evidence prerequisites (such as measurement validity, normalised rate, or source attribution) may additionally support assessing the condition, but MUST NOT replace the supplied higher-order intervention-fit dependency. Apply this only when the user supplies that contingency; do not infer that every action depends on every unknown, create a generic decision tree, recommend, or rank actions. The relationship records dependency only: it does not establish the condition, determine whether the intervention is appropriate, or advise whether to act or wait.
|
||||
|
||||
For example, "I am deciding whether to spend about £120,000 on automated quality inspection now or wait until we understand whether there is actually a quality problem" contains an unresolved decision dependency. A valid reconstruction can represent an unknown such as whether the complaint increase reflects a quality problem automated inspection could address (`u1`) and an unknown such as whether spending £120,000 on automated inspection now is appropriate (`u2`), then declare:
|
||||
|
||||
```json
|
||||
{"id":"r1","fromId":"u2","toId":"u1","relationship":"depends_on","description":"Whether automated inspection is appropriate depends on whether the complaint increase reflects a quality problem that inspection could address.","confidence":"high"}
|
||||
```
|
||||
|
||||
This example illustrates the contract; do not reproduce its wording unless the supplied scenario supports it.
|
||||
|
||||
## Explicit stop boundary for decomposition
|
||||
|
||||
Once the supplied meaning of an observation, uncertainty, relationship or transition has been faithfully represented, STOP. Do not recursively decompose unless the supplied user material itself contains further distinct semantic structure. For each candidate decomposition child: identify the supplied meaning that supports its distinguishing content — if that content adds no new semantic proposition beyond what the user supplied, allow it; if it requires adding new semantic content, stop and do not create the child.
|
||||
|
||||
## Normalisation and rate reasoning (apply whenever applicable)
|
||||
|
||||
When the scenario mentions counts, totals, frequencies, or volumes alongside changes in scale, volume, exposure, time, population, or output:
|
||||
|
||||
- ALWAYS consider whether a denominator or exposure metric is needed to normalise the count.
|
||||
- Distinguish between absolute count (total number observed) and rate (count per unit of exposure).
|
||||
- Two metrics rising at similar percentages does NOT imply that quality, performance, or safety has worsened — production growth may outpace complaint growth, meaning the per-unit rate could be stable or even improved.
|
||||
- Identify the possible denominator explicitly (e.g., "per unit produced", "per customer served", "per hour of operation").
|
||||
- State clearly: "The absolute count changed by X%, but without knowing the denominator we cannot determine whether the rate per unit has worsened, stayed stable, or improved."
|
||||
- Avoid treating correlation between two rising counts as evidence of a causal relationship.
|
||||
|
||||
## Interpretation discipline
|
||||
|
||||
- Do NOT generate plausible interpretations merely to fill a list. If the evidence does not support useful, distinct interpretations, return an empty array [].
|
||||
- Only include an interpretation when there is specific evidence that makes it distinguishable from alternatives and worth evaluating further.
|
||||
- Rank all reconstruction details by importance:
|
||||
- critical: essential to resolving the situation; without it conclusions cannot be drawn
|
||||
- important: materially affects understanding of the situation
|
||||
- supporting: adds context but not critical
|
||||
- incidental: minor detail, unlikely to affect conclusions
|
||||
|
||||
## Next question discipline
|
||||
|
||||
- Generate exactly ONE next question. Do NOT combine multiple questions.
|
||||
- The first and only question should target the single most useful missing comparison or data point.
|
||||
- Prefer narrow, specific questions over broad compound questions.
|
||||
- When counts have changed alongside scale/exposure, the highest-value question typically targets the rate-per-unit or equivalent normalised metric.
|
||||
- Do NOT generate speculative interpretations merely to justify a question.
|
||||
|
||||
## Confidence scale
|
||||
|
||||
- low — weak evidence, speculation, or missing information
|
||||
- medium — reasonable inference from available evidence
|
||||
- high — strong evidence, direct observation, or confirmed fact
|
||||
|
||||
## Importance scale (evidence records)
|
||||
|
||||
- incidental — minor detail, unlikely to affect conclusions
|
||||
- supporting — adds context but not critical
|
||||
- important — materially affects understanding of the situation
|
||||
- critical — essential to resolving the situation; without it conclusions cannot be drawn
|
||||
|
||||
## Expected information value (next question)
|
||||
|
||||
- low — marginally useful even if answered
|
||||
- medium — meaningfully clarifies the situation
|
||||
- high — would significantly distinguish between plausible explanations or fill a gap in understanding
|
||||
|
||||
## Next question selection criteria
|
||||
|
||||
Prefer questions that:
|
||||
- clarify a major difference
|
||||
- establish a baseline
|
||||
- explain an important transition
|
||||
- test an unsupported claim
|
||||
- distinguish between plausible explanations
|
||||
- request measurable evidence
|
||||
- identify who or what is affected
|
||||
- establish timing
|
||||
|
||||
Avoid questions that:
|
||||
- have already been answered
|
||||
- assume a cause
|
||||
- jump to a solution
|
||||
- ask about motive before the observable situation is understood
|
||||
- focus on incidental wording
|
||||
- are too broad to produce useful information
|
||||
- combine many unrelated questions
|
||||
|
||||
## Semantic preservation check before returning JSON
|
||||
|
||||
Before returning the final JSON, compare the reconstruction against the supplied scenario. Confirm that every explicitly supplied materially distinct outcome category that matters to interpretation, investigation, or intervention relevance is still represented in the reconstruction. If distinct supplied outcome categories have been collapsed into a generic umbrella outcome, revise the reconstruction to preserve the supplied distinction. Do not create categories or distinctions the user did not supply.
|
||||
|
||||
## Output format — return this exact JSON structure
|
||||
|
||||
Return a JSON object with exactly these four top-level keys (use **camelCase**):
|
||||
|
||||
```json
|
||||
{
|
||||
"inputClassification": {
|
||||
"primaryType": "<one of: observed_problem, unexplained_change, contradiction, decision_request, causal_claim, reported_claim, fault_report, ambiguous_statement, question, desired_outcome, insufficient_context, other>",
|
||||
"secondaryTypes": ["<optional additional types from the same list>"],
|
||||
"reasoningModes": ["<one or more of: establish_baseline, identify_difference, reconstruct_transition, decompose_aggregate, validate_measurement, validate_claim, investigate_contradiction, clarify_meaning, decision_support, fault_investigation, identify_missing_information, test_possible_explanations, other>"],
|
||||
"classificationReason": "<brief explanation of why you chose the primary type>",
|
||||
"confidence": "<low | medium | high>"
|
||||
},
|
||||
"reconstruction": {
|
||||
"summary": "<one-sentence overview of the situation>",
|
||||
"actors": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"systemsOrObjects": [{"id": "<any unique string>", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"expectedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"observedStates": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"differences": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"knownTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "...", "explanationStatus": "..."}],
|
||||
"unexplainedTransitions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>", "entity": "...", "previousState": "...", "currentState": "..."}],
|
||||
"contradictions": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"importantUnknowns": [{"id": "...", "description": "...", "confidence": "<low|medium|high>"}],
|
||||
"plausibleInterpretations": [{"id": "...", "description": "...", "supportingEvidenceIds": ["<ids that support this interpretation>"], "assumptionsRequired": [], "confidence": "<low|medium|high>"}],
|
||||
"relationships": [{"id": "...", "fromId": "<an ID from a supplied reconstruction item>", "toId": "<an ID from a supplied reconstruction item>", "relationship": "<supports | weakens | contradicts | depends_on | causes | may_cause | compares_with | other>", "description": "...", "confidence": "<low|medium|high>"}]
|
||||
},
|
||||
"evidence": [
|
||||
{"id": "<any unique string>", "description": "...", "evidenceType": "<direct_observation | reported_statement | interpretation | assumption | inferred_relationship>", "source": "<optional — who/where this came from>", "attribution": null, "confidence": "<low | medium | high>", "importance": "<incidental | supporting | important | critical>"}
|
||||
],
|
||||
"nextQuestion": {"id": "<any unique string>", "question": "<one precise question>", "targets": ["<what this question targets — e.g. 'actor', 'system', 'expectedOutcome'>"], "reason": "<why answering this is important>", "expectedInformationValue": "<low | medium | high>", "reasoningMode": "<optional reasoning mode from the list above>"}
|
||||
}
|
||||
```
|
||||
|
||||
`relationships` must be an array and may be empty (`[]`).
|
||||
|
||||
CRITICAL RULES for JSON output:
|
||||
1. Use **exactly** the key names shown above (camelCase, no snake_case).
|
||||
2. The four top-level keys must be: `inputClassification`, `reconstruction`, `evidence`, `nextQuestion`.
|
||||
3. Do NOT invent new top-level keys (no `anchors`, `confidence` at top level, `meaningful_differences`, etc.).
|
||||
4. Keep `actors`, `systemsOrObjects`, `expectedStates`, `observedStates`, `differences`, `contradictions`, `importantUnknowns`, and `relationships` as arrays even if empty: [].
|
||||
5. Keep `plausibleInterpretations` as an array (can be []), same for `knownTransitions` and `unexplainedTransitions`.
|
||||
6. Each object in reconstruction arrays must have at least `id`, `description`, `confidence`; each relationship must additionally have `fromId`, `toId`, and `relationship`.
|
||||
7. **evidenceType**: classify each evidence item clearly as either a direct observation, a reported statement, an interpretation, an assumption, or an inferred relationship. Do not treat raw counts as proof of causal relationships — they may be inferred relationships only when supported by explicit reasoning about denominators or rates.
|
||||
|
||||
Scenario:
|
||||
{{SCENARIO}}
|
||||
|
||||
Return ONLY the JSON object starting with { and ending with }. Do NOT include any text before the opening brace or after the closing brace. Do NOT wrap in markdown backticks.
|
||||
Executable
+189
@@ -0,0 +1,189 @@
|
||||
#!/usr/bin/env bash
|
||||
# Confidence Engine — Production Deployment Script
|
||||
#
|
||||
# Executes on CT 112 at /opt/confidence-engine.
|
||||
# Input: $1 = exact commit SHA (validated)
|
||||
#
|
||||
# Environment:
|
||||
# DEPLOY_DIR - target repository path (defaults /opt/confidence-engine)
|
||||
# HEALTH_URL - health check endpoint (defaults http://127.0.0.1:3000/api/health)
|
||||
#
|
||||
# Reads deploy.env from $DEPLOY_DIR/deploy.env for runtime values.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ── Parameters ──────────────────────────────────────────────────────
|
||||
|
||||
DEPLOY_SHA="${1:?Error: missing commit SHA argument}"
|
||||
DEPLOY_DIR="${DEPLOY_DIR:-/opt/confidence-engine}"
|
||||
HEALTH_URL="${HEALTH_URL:-http://127.0.0.1:3000/api/health}"
|
||||
|
||||
# ── Validation ──────────────────────────────────────────────────────
|
||||
|
||||
if [[ ${#DEPLOY_SHA} -lt 7 ]]; then
|
||||
echo "ERROR: SHA must be at least 7 hex characters."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! echo "$DEPLOY_SHA" | grep -qE '^[0-9a-fA-F]+$'; then
|
||||
echo "ERROR: SHA contains non-hex characters."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Deploying SHA: $DEPLOY_SHA"
|
||||
|
||||
cd "$DEPLOY_DIR" || { echo "ERROR: Cannot cd to $DEPLOY_DIR"; exit 1; }
|
||||
|
||||
# ── Safety checks ───────────────────────────────────────────────────
|
||||
|
||||
# deploy.env must exist on the host (not tracked by Git)
|
||||
if [[ ! -f deploy.env ]]; then
|
||||
echo "DEPLOYMENT BLOCKED - deploy.env not found at deploy.env"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check for uncommitted changes that would block safe checkout
|
||||
if ! git diff-index --quiet HEAD -- 2>/dev/null; then
|
||||
echo "DEPLOYMENT BLOCKED - target working tree has uncommitted changes"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ -n "$(git ls-files --others --exclude-standard)" ]]; then
|
||||
echo "DEPLOYMENT BLOCKED - target working tree has untracked files"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── Fetch and verify SHA ────────────────────────────────────────────
|
||||
|
||||
echo "Fetching latest refs from origin..."
|
||||
git fetch origin >/dev/null 2>&1 || {
|
||||
echo "ERROR: git fetch origin failed."
|
||||
exit 1
|
||||
}
|
||||
|
||||
if ! git rev-parse --verify "$DEPLOY_SHA" >/dev/null 2>&1; then
|
||||
echo "ERROR: SHA $DEPLOY_SHA not found in repository after fetch."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── Detached checkout (never modifies working tree or remote) ───────
|
||||
|
||||
echo "Checking out exactly $DEPLOY_SHA..."
|
||||
git checkout --detach "$DEPLOY_SHA" >/dev/null 2>&1 || {
|
||||
echo "ERROR: Could not checkout SHA $DEPLOY_SHA."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Verify the detached HEAD matches our deploy SHA
|
||||
CURRENT_SHA="$(git rev-parse HEAD)"
|
||||
if [[ "${CURRENT_SHA%"${CURRENT_SHA#?}"}" != "${DEPLOY_SHA%"${DEPLOY_SHA#?}"}" ]] || \
|
||||
[[ "$CURRENT_SHA" != "$DEPLOY_SHA" && "${CURRENT_SHA:0:7}" != "${DEPLOY_SHA:0:7}" ]]; then
|
||||
echo "ERROR: Checked out SHA $CURRENT_SHA does not match requested $DEPLOY_SHA."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── Load runtime environment ────────────────────────────────────────
|
||||
|
||||
if [[ -f deploy.env ]]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source deploy.env
|
||||
set +a
|
||||
else
|
||||
echo "ERROR: deploy.env not found at $DEPLOY_DIR/deploy.env"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── Build Docker image ──────────────────────────────────────────────
|
||||
|
||||
IMAGE_TAG="confidence-engine:${DEPLOY_SHA}"
|
||||
|
||||
echo "Building Docker image ${IMAGE_TAG}..."
|
||||
docker build \
|
||||
--build-arg NEXT_PUBLIC_SUPABASE_URL="${NEXT_PUBLIC_SUPABASE_URL}" \
|
||||
--build-arg NEXT_PUBLIC_SUPABASE_ANON_KEY="${NEXT_PUBLIC_SUPABASE_ANON_KEY}" \
|
||||
-t "${IMAGE_TAG}" \
|
||||
.
|
||||
|
||||
echo "Image ${IMAGE_TAG} built successfully."
|
||||
|
||||
# ── Container replacement ───────────────────────────────────────────
|
||||
|
||||
CONTAINER_NAME="confidence-engine"
|
||||
|
||||
# Record previous image before replacing container
|
||||
PREV_IMAGE=""
|
||||
if docker ps -a --format '{{.Names}}' | grep -qx "$CONTAINER_NAME"; then
|
||||
PREV_IMAGE="$(docker inspect --format='{{.Config.Image}}' "$CONTAINER_NAME" 2>/dev/null || echo "")"
|
||||
fi
|
||||
|
||||
echo "Stopping existing container (if running)..."
|
||||
docker stop "$CONTAINER_NAME" >/dev/null 2>&1 || true
|
||||
docker rm -f "$CONTAINER_NAME" >/dev/null 2>&1 || true
|
||||
|
||||
echo "Starting new container from ${IMAGE_TAG}..."
|
||||
docker run -d \
|
||||
--name "$CONTAINER_NAME" \
|
||||
--restart unless-stopped \
|
||||
-p 3000:3000 \
|
||||
-e OLLAMA_BASE_URL="${OLLAMA_BASE_URL}" \
|
||||
-e OLLAMA_MODEL="${OLLAMA_MODEL}" \
|
||||
"${IMAGE_TAG}"
|
||||
|
||||
echo "Container $CONTAINER_NAME started."
|
||||
|
||||
# ── Health check polling ────────────────────────────────────────────
|
||||
|
||||
HEALTH_CHECK_TIMEOUT=60
|
||||
HEALTH_CHECK_INTERVAL=3
|
||||
ELAPSED=0
|
||||
|
||||
echo "Waiting for health check at ${HEALTH_URL}..."
|
||||
|
||||
while [[ $ELAPSED -lt $HEALTH_CHECK_TIMEOUT ]]; do
|
||||
if curl -sf "$HEALTH_URL" | grep -q '"healthy"[[:space:]]*:[[:space:]]*true'; then
|
||||
echo ""
|
||||
echo "============================================="
|
||||
echo "DEPLOYMENT SUCCEEDED"
|
||||
echo "Deployed SHA: $DEPLOY_SHA"
|
||||
echo "Current image: $(docker inspect --format='{{.Config.Image}}' "$CONTAINER_NAME" 2>/dev/null || echo 'unknown')"
|
||||
echo "============================================="
|
||||
exit 0
|
||||
fi
|
||||
|
||||
sleep "$HEALTH_CHECK_INTERVAL"
|
||||
ELAPSED=$((ELAPSED + HEALTH_CHECK_INTERVAL))
|
||||
done
|
||||
|
||||
echo ""
|
||||
echo "============================================="
|
||||
echo "DEPLOYMENT FAILED - Health check timed out after ${HEALTH_CHECK_TIMEOUT}s"
|
||||
echo "============================================="
|
||||
|
||||
# ── Rollback (one attempt) ──────────────────────────────────────────
|
||||
|
||||
if [[ -n "$PREV_IMAGE" ]]; then
|
||||
echo ""
|
||||
echo "Attempting rollback to previous image: $PREV_IMAGE"
|
||||
docker stop "$CONTAINER_NAME" >/dev/null 2>&1 || true
|
||||
docker rm -f "$CONTAINER_NAME" >/dev/null 2>&1 || true
|
||||
|
||||
docker run -d \
|
||||
--name "$CONTAINER_NAME" \
|
||||
--restart unless-stopped \
|
||||
-p 3000:3000 \
|
||||
-e OLLAMA_BASE_URL="${OLLAMA_BASE_URL}" \
|
||||
-e OLLAMA_MODEL="${OLLAMA_MODEL}" \
|
||||
"$PREV_IMAGE"
|
||||
|
||||
if [[ $? -eq 0 ]]; then
|
||||
echo "DEPLOYMENT FAILED - ROLLED BACK to $PREV_IMAGE"
|
||||
else
|
||||
echo "DEPLOYMENT FAILED - ROLLBACK ALSO FAILED"
|
||||
fi
|
||||
else
|
||||
echo "No previous image available for rollback."
|
||||
fi
|
||||
|
||||
echo "DEPLOYMENT FAILED - MANUAL RECOVERY REQUIRED"
|
||||
exit 1
|
||||
Executable
+225
@@ -0,0 +1,225 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Minimal startCase (initial decomposition) experiment harness.
|
||||
*
|
||||
* Loads .env.local, invokes startCase() through the real production path,
|
||||
* and prints a structured result for semantic evaluation.
|
||||
*
|
||||
* Usage:
|
||||
* node scripts/start-case-experiment-helper.cjs "<scenario text>"
|
||||
* node scripts/start-case-experiment-helper.cjs --file scenario.json
|
||||
* node scripts/start-case-experiment-helper.cjs --reconstruction-only --file scenario.json
|
||||
*
|
||||
* Exit codes:
|
||||
* 0 — success (structured result printed to stdout)
|
||||
* 1 — execution failure or invalid input
|
||||
*/
|
||||
|
||||
// Use the standard environment loader already available to this Next repository.
|
||||
// loadEnvConfig mutates process.env in-place (same semantics as previous bespoke parser).
|
||||
const path = require("path");
|
||||
const PROJECT_ROOT = path.resolve(__dirname, "..");
|
||||
(function loadEnvironment() {
|
||||
try {
|
||||
const { loadEnvConfig } = require("@next/env");
|
||||
loadEnvConfig(PROJECT_ROOT, undefined, { logOutput: "none" }, false);
|
||||
} catch {
|
||||
// loader unavailable — proceed with whatever is already set.
|
||||
}
|
||||
})();
|
||||
|
||||
function assertRequiredEnv(name) {
|
||||
const value = process.env[name];
|
||||
if (!value) {
|
||||
throw new Error(
|
||||
`Live experiment requires ${name}. Set it in .env.local.\n` +
|
||||
`Found: OLLAMA_BASE_URL=${process.env.OLLAMA_BASE_URL ?? "(missing)"}, ` +
|
||||
`OLLAMA_MODEL=${process.env.OLLAMA_MODEL ?? "(missing)"}`
|
||||
);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function readScenarioInput(argv) {
|
||||
if (argv.includes("--file")) {
|
||||
const idx = argv.indexOf("--file");
|
||||
if (idx + 1 >= argv.length) throw new Error("--file requires a path argument");
|
||||
const fs = require("fs");
|
||||
const path = argv[idx + 1];
|
||||
return JSON.parse(fs.readFileSync(path, "utf-8"));
|
||||
}
|
||||
// Use all positional args as the scenario text (join with space)
|
||||
const startIdx = argv.findIndex(a => a !== "" && !a.startsWith("-") && a !== "--");
|
||||
if (startIdx === -1 || startIdx >= argv.length) throw new Error("Usage: node scripts/start-case-experiment-helper.cjs \"<scenario>\" [--file path.json]");
|
||||
return { scenario: argv.slice(startIdx).join(" ") };
|
||||
}
|
||||
|
||||
async function runStartCaseExperiment(scenarioInput, experimentInstruction, options = {}) {
|
||||
// ── Experiment seam bridge: supply instruction to production path ──
|
||||
const hadEnvVar = process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION != null;
|
||||
const previousEnvValue = process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION;
|
||||
if (experimentInstruction) {
|
||||
process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION = experimentInstruction;
|
||||
}
|
||||
|
||||
let startCase = options.startCase;
|
||||
let reconstructionProvider = options.reconstructionProvider;
|
||||
let reconstructionModelName = options.reconstructionModelName;
|
||||
let reconstructionOnly = options.reconstructionOnly;
|
||||
const isMock = process.env.START_CASE_EXPERIMENT_HELPER_MOCK === "1";
|
||||
|
||||
if (process.env.START_CASE_EXPERIMENT_PROVIDER === "openai") {
|
||||
if (!process.env.OPENAI_API_KEY) {
|
||||
throw new Error(
|
||||
"START_CASE_EXPERIMENT_PROVIDER=openai requires OPENAI_API_KEY",
|
||||
);
|
||||
}
|
||||
const { createOpenAIReconstructionProvider } = await import(
|
||||
PROJECT_ROOT + "/lib/llm/provider.js"
|
||||
);
|
||||
reconstructionProvider = createOpenAIReconstructionProvider({
|
||||
apiKey: process.env.OPENAI_API_KEY,
|
||||
fetchImpl: fetch,
|
||||
});
|
||||
reconstructionModelName = "gpt-5.6-terra";
|
||||
reconstructionOnly = true;
|
||||
}
|
||||
|
||||
if (!startCase && isMock) {
|
||||
// Deterministic mode: skip environment checks and use inline test double.
|
||||
startCase = async (_body, dependencies) => {
|
||||
// Deterministic inline test double — no live model calls.
|
||||
const shouldFail = process.env.START_CASE_EXPERIMENT_HELPER_FORCE_FAIL === "1";
|
||||
if (shouldFail) {
|
||||
return { success: false, error: "deterministic mock failure", statusCode: 500 };
|
||||
}
|
||||
return {
|
||||
success: true,
|
||||
situationGraph: { nodes: [], edges: [], activeUnknownNodeId: null, resolvedNodeIds: [], currentSummary: "test" },
|
||||
assessment: { phase: "initial", progress: 0 },
|
||||
selectedQuestion: null,
|
||||
summary: null,
|
||||
diagnostics: {
|
||||
validationStatus: "mock",
|
||||
modelName: "inline-test-double",
|
||||
reconstructionOnly: dependencies.reconstructionOnly === true,
|
||||
graphReferenceValidation: { valid: true, errors: [] },
|
||||
},
|
||||
};
|
||||
};
|
||||
} else if (!startCase) {
|
||||
const baseUrl = assertRequiredEnv("OLLAMA_BASE_URL");
|
||||
const model = assertRequiredEnv("OLLAMA_MODEL");
|
||||
|
||||
if (baseUrl === "http://localhost:11434" || baseUrl === "http://127.0.0.1:11434") {
|
||||
throw new Error(
|
||||
`Live experiment harness refuses to use localhost fallback. ` +
|
||||
`OLLAMA_BASE_URL=${baseUrl}. Configure a real host in .env.local.`
|
||||
);
|
||||
}
|
||||
|
||||
const { startCase: _sc } = await import(PROJECT_ROOT + "/lib/graph/orchestrator.js");
|
||||
startCase = _sc;
|
||||
}
|
||||
|
||||
let result;
|
||||
const endToEndStartedAt = Date.now();
|
||||
try {
|
||||
result = await startCase(scenarioInput, {
|
||||
reconstructionProvider,
|
||||
reconstructionModelName,
|
||||
reconstructionOnly,
|
||||
});
|
||||
} finally {
|
||||
// Clean up experiment env var after execution regardless of outcome
|
||||
if (experimentInstruction && !hadEnvVar) {
|
||||
delete process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION;
|
||||
} else if (!hadEnvVar) {
|
||||
// was not set before and not set during — ensure it stays unset
|
||||
delete process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION;
|
||||
} else if (!experimentInstruction && hadEnvVar) {
|
||||
// restore original value that existed before
|
||||
if (previousEnvValue === undefined) {
|
||||
delete process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION;
|
||||
} else {
|
||||
process.env.RECONSTRUCTION_EXPERIMENT_INSTRUCTION = previousEnvValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
success: result.success,
|
||||
summary: result.summary ?? null,
|
||||
reconstruction: result.reconstruction ?? null,
|
||||
situationGraphNodeCount: result.situationGraph?.nodes?.length ?? 0,
|
||||
situationGraphEdgeCount: result.situationGraph?.edges?.length ?? 0,
|
||||
selectedQuestion: result.selectedQuestion?.question ?? null,
|
||||
assessmentPhase: result.assessment?.phase ?? null,
|
||||
assessmentProgress: result.assessment?.progress ?? null,
|
||||
endToEndElapsedMs: Date.now() - endToEndStartedAt,
|
||||
diagnostics: result.diagnostics ?? null,
|
||||
providerApiPath: result.providerApiPath ?? null,
|
||||
error: result.success ? null : (result.error ?? "unknown"),
|
||||
statusCode: result.statusCode ?? (result.success ? 200 : 500),
|
||||
};
|
||||
}
|
||||
|
||||
if (require.main === module) {
|
||||
(async () => {
|
||||
// Import-only mode: prove real production seam without invoking startCase.
|
||||
// Used only for deterministic apparatus verification of the import path.
|
||||
if (process.env.START_CASE_EXPERIMENT_HELPER_IMPORT_ONLY === "1") {
|
||||
try {
|
||||
const mod = await import(PROJECT_ROOT + "/lib/graph/orchestrator.js");
|
||||
const startCaseResolved = typeof mod.startCase === "function";
|
||||
const output = JSON.stringify({
|
||||
success: true,
|
||||
mode: "import-only",
|
||||
startCaseResolved,
|
||||
});
|
||||
console.log(output);
|
||||
process.exit(0);
|
||||
} catch (e) {
|
||||
let failureReason;
|
||||
if (e.code === "ERR_REQUIRE_ESM" || e.message.includes("ERR_REQUIRE_ESM")) {
|
||||
failureReason = "CJS cannot import ESM module";
|
||||
} else if (e.code === "ERR_MODULE_NOT_FOUND" || e.code === "ERR_UNSUPPORTED_DIR_IMPORT") {
|
||||
failureReason = `Module resolution failed: ${e.message}`;
|
||||
} else {
|
||||
failureReason = `Import error: ${e.message}`;
|
||||
}
|
||||
const output = JSON.stringify({ success: false, mode: "import-only", startCaseResolved: false, failureReason });
|
||||
console.log(output);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
// ── Experiment seam: parse optional experiment instruction ──
|
||||
const experimentIdx = process.argv.indexOf("--experiment-instruction");
|
||||
let experimentInstruction = null;
|
||||
if (experimentIdx !== -1 && experimentIdx + 1 < process.argv.length) {
|
||||
experimentInstruction = process.argv[experimentIdx + 1];
|
||||
}
|
||||
|
||||
const scenarioInput = readScenarioInput(process.argv);
|
||||
const reconstructionOnly = process.argv.includes("--reconstruction-only");
|
||||
const result = await runStartCaseExperiment(scenarioInput, experimentInstruction, {
|
||||
reconstructionOnly,
|
||||
});
|
||||
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
process.exit(result.success ? 0 : 1);
|
||||
} catch (e) {
|
||||
const failure = {
|
||||
success: false,
|
||||
error: e.message ?? "unknown",
|
||||
statusCode: 1,
|
||||
};
|
||||
console.log(JSON.stringify(failure));
|
||||
process.exit(1);
|
||||
}
|
||||
})();
|
||||
}
|
||||
|
||||
module.exports = { runStartCaseExperiment };
|
||||
@@ -0,0 +1,53 @@
|
||||
create schema if not exists confidence_engine;
|
||||
|
||||
create table confidence_engine.investigations (
|
||||
id uuid primary key,
|
||||
user_id uuid not null references auth.users(id) on delete cascade,
|
||||
snapshot jsonb not null,
|
||||
created_at timestamptz not null default now(),
|
||||
updated_at timestamptz not null default now()
|
||||
);
|
||||
|
||||
create index investigations_user_id_idx
|
||||
on confidence_engine.investigations (user_id);
|
||||
|
||||
create function confidence_engine.set_updated_at()
|
||||
returns trigger
|
||||
language plpgsql
|
||||
set search_path = ''
|
||||
as $$
|
||||
begin
|
||||
new.updated_at = now();
|
||||
return new;
|
||||
end;
|
||||
$$;
|
||||
|
||||
create trigger investigations_set_updated_at
|
||||
before update on confidence_engine.investigations
|
||||
for each row execute function confidence_engine.set_updated_at();
|
||||
|
||||
alter table confidence_engine.investigations enable row level security;
|
||||
|
||||
grant usage on schema confidence_engine to authenticated;
|
||||
grant select, insert, update, delete on confidence_engine.investigations to authenticated;
|
||||
|
||||
create policy "Users can select their own investigations"
|
||||
on confidence_engine.investigations
|
||||
for select to authenticated
|
||||
using (user_id = auth.uid());
|
||||
|
||||
create policy "Users can insert their own investigations"
|
||||
on confidence_engine.investigations
|
||||
for insert to authenticated
|
||||
with check (user_id = auth.uid());
|
||||
|
||||
create policy "Users can update their own investigations"
|
||||
on confidence_engine.investigations
|
||||
for update to authenticated
|
||||
using (user_id = auth.uid())
|
||||
with check (user_id = auth.uid());
|
||||
|
||||
create policy "Users can delete their own investigations"
|
||||
on confidence_engine.investigations
|
||||
for delete to authenticated
|
||||
using (user_id = auth.uid());
|
||||
@@ -0,0 +1,29 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const mockSynthesize = vi.fn();
|
||||
|
||||
vi.mock("@/lib/graph/investigation-overview-synthesis.js", () => ({
|
||||
synthesizeInvestigationOverview: (...args) => mockSynthesize(...args),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/llm/provider.js", () => ({
|
||||
getProvider: () => ({ generateReconstruction() {} }),
|
||||
getProviderModelName: () => "gpt-5.6-terra",
|
||||
}));
|
||||
|
||||
describe("POST /api/cases/overview provider routing", () => {
|
||||
beforeEach(() => mockSynthesize.mockClear());
|
||||
|
||||
it("passes the central provider and model resolution to overview synthesis", async () => {
|
||||
mockSynthesize.mockResolvedValue({ understanding: "Understanding", plausibleInterpretations: "None" });
|
||||
const { POST } = await import("@/app/api/cases/overview/route.js");
|
||||
const response = await POST(new Request("http://localhost/api/cases/overview", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ situationGraph: { nodes: [], edges: [] }, findings: [], plausibleInterpretations: [] }),
|
||||
}));
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
expect(mockSynthesize.mock.calls[0][1]).toMatchObject({ modelName: "gpt-5.6-terra" });
|
||||
});
|
||||
});
|
||||
@@ -1,15 +1,35 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
const mockStartCase = vi.fn();
|
||||
let realStartCase;
|
||||
let warnSpy;
|
||||
let errorSpy;
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", () => ({
|
||||
startCase: (...args) => mockStartCase(...args),
|
||||
vi.mock("@/lib/config.js", () => ({
|
||||
getConfig: () => ({ ok: true }),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", async (importOriginal) => {
|
||||
const actual = await importOriginal();
|
||||
realStartCase = actual.startCase;
|
||||
return { ...actual, startCase: (...args) => mockStartCase(...args) };
|
||||
});
|
||||
|
||||
vi.mock("@/lib/supabase/api-auth.js", () => ({
|
||||
withAuthenticatedApi: (handler) => handler,
|
||||
}));
|
||||
|
||||
describe("app/api/cases/start route", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
vi.clearAllMocks();
|
||||
warnSpy = vi.spyOn(console, "warn").mockImplementation(() => {});
|
||||
errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
warnSpy.mockRestore();
|
||||
errorSpy.mockRestore();
|
||||
});
|
||||
|
||||
it("delegates request body to the orchestrator", async () => {
|
||||
@@ -30,12 +50,38 @@ describe("app/api/cases/start route", () => {
|
||||
await POST(request);
|
||||
|
||||
expect(mockStartCase).toHaveBeenCalledWith({ scenario: "Scenario text" });
|
||||
expect(mockStartCase.mock.calls[0][0]).not.toHaveProperty("provider");
|
||||
expect(mockStartCase.mock.calls[0][0]).not.toHaveProperty("modelName");
|
||||
});
|
||||
|
||||
it("returns 200 on success", async () => {
|
||||
const reconstruction = {
|
||||
summary: "Validated reconstruction summary",
|
||||
relationships: [
|
||||
{
|
||||
id: "r1",
|
||||
fromId: "u-intervention",
|
||||
toId: "u-problem",
|
||||
relationship: "depends_on",
|
||||
description: "The intervention depends on the unresolved problem.",
|
||||
confidence: "high",
|
||||
},
|
||||
],
|
||||
};
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: true,
|
||||
situationGraph: { nodes: [{ id: "n1" }], edges: [] },
|
||||
reconstruction,
|
||||
situationGraph: {
|
||||
nodes: [{ id: "n1" }],
|
||||
edges: [
|
||||
{
|
||||
id: "e-unrelated",
|
||||
fromNodeId: "n1",
|
||||
toNodeId: "n1",
|
||||
relationship: "supports",
|
||||
},
|
||||
],
|
||||
},
|
||||
selectedQuestion: null,
|
||||
diagnostics: {},
|
||||
});
|
||||
@@ -50,6 +96,12 @@ describe("app/api/cases/start route", () => {
|
||||
);
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: true,
|
||||
reconstruction,
|
||||
});
|
||||
expect(warnSpy).not.toHaveBeenCalled();
|
||||
expect(errorSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("returns 400 for invalid request input", async () => {
|
||||
@@ -74,14 +126,74 @@ describe("app/api/cases/start route", () => {
|
||||
success: false,
|
||||
error: "Invalid start-case request",
|
||||
});
|
||||
expect(warnSpy).toHaveBeenCalledTimes(1);
|
||||
expect(warnSpy).toHaveBeenCalledWith(
|
||||
"[api/cases/start] error response",
|
||||
expect.objectContaining({ status: 400, error: "Invalid start-case request" }),
|
||||
);
|
||||
expect(errorSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("returns a sanitized 503 when the provider is unavailable", async () => {
|
||||
const unavailableError = Object.assign(
|
||||
new Error("Ollama /api/generate request timed out after 5 minutes"),
|
||||
{
|
||||
code: "PROVIDER_UNAVAILABLE",
|
||||
providerApiPath: "/api/generate",
|
||||
providerExecution: { generateRequestAttempted: true },
|
||||
},
|
||||
);
|
||||
const reconstructionProvider = {
|
||||
generateReconstruction: vi.fn().mockRejectedValue(unavailableError),
|
||||
};
|
||||
mockStartCase.mockImplementation((body) => realStartCase(body, {
|
||||
reconstructionProvider,
|
||||
reconstructionModelName: "configured-model",
|
||||
}));
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ scenario: "Scenario text" }),
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(503);
|
||||
const body = await response.json();
|
||||
expect(body).toEqual({
|
||||
success: false,
|
||||
error: "Reasoning service is temporarily unavailable.",
|
||||
});
|
||||
expect(JSON.stringify(body)).not.toMatch(/ollama|generate|timed out/i);
|
||||
});
|
||||
|
||||
it("returns provider/internal failures as 5xx without stack traces", async () => {
|
||||
const rawResponse = `{"reconstruction":{"observedStates":[{"id":"obs-1"${"x".repeat(2500)}}]}}`;
|
||||
mockStartCase.mockResolvedValue({
|
||||
success: false,
|
||||
error: "Provider unavailable",
|
||||
diagnostics: { modelName: "llama3" },
|
||||
statusCode: 502,
|
||||
analysisErrors: ["reconstruction: Required"],
|
||||
validationIssues: [
|
||||
{
|
||||
path: ["reconstruction", "observedStates", 2, "description"],
|
||||
code: "invalid_type",
|
||||
message: "Required",
|
||||
expected: "string",
|
||||
received: "undefined",
|
||||
},
|
||||
],
|
||||
providerApiPath: "/api/generate",
|
||||
providerExecution: {
|
||||
chatCapabilityDetected: false,
|
||||
chatRequestAttempted: false,
|
||||
chatRequestSucceeded: false,
|
||||
generateRequestAttempted: true,
|
||||
},
|
||||
rawResponse,
|
||||
});
|
||||
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
@@ -94,7 +206,46 @@ describe("app/api/cases/start route", () => {
|
||||
);
|
||||
|
||||
expect(response.status).toBe(502);
|
||||
await expect(response.json()).resolves.not.toHaveProperty("stack");
|
||||
const body = await response.json();
|
||||
expect(body).toHaveProperty("rawResponse");
|
||||
expect(body.rawResponse).toBe(rawResponse);
|
||||
expect(body.rawResponse.length).toBeGreaterThan(2000);
|
||||
expect(body.analysisErrors).toEqual(["reconstruction: Required"]);
|
||||
expect(body.validationIssues).toEqual([
|
||||
expect.objectContaining({
|
||||
path: ["reconstruction", "observedStates", 2, "description"],
|
||||
code: "invalid_type",
|
||||
message: "Required",
|
||||
expected: "string",
|
||||
received: "undefined",
|
||||
}),
|
||||
]);
|
||||
expect(body.providerApiPath).toBe("/api/generate");
|
||||
expect(body.providerExecution).toEqual({
|
||||
chatCapabilityDetected: false,
|
||||
chatRequestAttempted: false,
|
||||
chatRequestSucceeded: false,
|
||||
generateRequestAttempted: true,
|
||||
});
|
||||
expect(errorSpy).toHaveBeenCalledTimes(1);
|
||||
expect(errorSpy).toHaveBeenCalledWith(
|
||||
"[api/cases/start] error response",
|
||||
expect.objectContaining({
|
||||
status: 502,
|
||||
error: "Provider unavailable",
|
||||
analysisErrors: ["reconstruction: Required"],
|
||||
validationIssues: expect.any(Array),
|
||||
providerApiPath: "/api/generate",
|
||||
providerExecution: {
|
||||
chatCapabilityDetected: false,
|
||||
chatRequestAttempted: false,
|
||||
chatRequestSucceeded: false,
|
||||
generateRequestAttempted: true,
|
||||
},
|
||||
rawResponse,
|
||||
}),
|
||||
);
|
||||
expect(warnSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("returns structured 500 on malformed JSON", async () => {
|
||||
@@ -110,5 +261,13 @@ describe("app/api/cases/start route", () => {
|
||||
success: false,
|
||||
error: "Internal server error",
|
||||
});
|
||||
expect(errorSpy).toHaveBeenCalledWith(
|
||||
"[api/cases/start] unhandled exception",
|
||||
expect.objectContaining({
|
||||
message: "Unexpected token",
|
||||
stack: expect.any(String),
|
||||
error: expect.any(Error),
|
||||
}),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,11 +1,45 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { makeGraph, makeNode } from "@/lib/graph/schema.js";
|
||||
|
||||
const mockUpdateCase = vi.fn();
|
||||
const mockReconsiderCompletedEpisode = vi.fn();
|
||||
const mockApplyValidatedProposal = vi.fn();
|
||||
const mockPrepareCompletedEpisode = vi.fn();
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", () => ({
|
||||
updateCase: (...args) => mockUpdateCase(...args),
|
||||
reconsiderCompletedEpisode: (...args) => mockReconsiderCompletedEpisode(...args),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/graph/apply-proposal.js", () => ({
|
||||
applyValidatedProposal: (...args) => mockApplyValidatedProposal(...args),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/graph/episode-preparation.js", () => ({
|
||||
prepareCompletedEpisode: (...args) => mockPrepareCompletedEpisode(...args),
|
||||
}));
|
||||
|
||||
function makeEpisodeGraph() {
|
||||
return makeGraph({
|
||||
centralStatement: "Episode scenario",
|
||||
currentSummary: "Episode graph",
|
||||
nodes: [makeNode({ id: "target", label: "Target question", kind: "unknown", status: "unknown" })],
|
||||
edges: [],
|
||||
activeUnknownNodeId: "target",
|
||||
resolvedNodeIds: ["already-resolved"],
|
||||
});
|
||||
}
|
||||
|
||||
function mockSuccessfulEpisodeApplication(graph, overrides = {}) {
|
||||
mockPrepareCompletedEpisode.mockReturnValue({ turns: [{ question: "Q?", answer: "A." }] });
|
||||
mockReconsiderCompletedEpisode.mockResolvedValue({ success: true, proposal: {} });
|
||||
mockApplyValidatedProposal.mockResolvedValue({
|
||||
success: true,
|
||||
updatedSituationGraph: graph,
|
||||
...overrides,
|
||||
});
|
||||
}
|
||||
|
||||
function makeSuccessResult() {
|
||||
return {
|
||||
success: true,
|
||||
@@ -72,6 +106,69 @@ describe("app/api/cases/update route", () => {
|
||||
);
|
||||
|
||||
expect(mockUpdateCase).toHaveBeenCalledWith(body, { applyProposal: true });
|
||||
expect(mockUpdateCase.mock.calls[0][0]).not.toHaveProperty("provider");
|
||||
expect(mockUpdateCase.mock.calls[0][0]).not.toHaveProperty("modelName");
|
||||
});
|
||||
|
||||
it("authoritatively resolves the selected target after a successful no-op episode", async () => {
|
||||
const graph = makeEpisodeGraph();
|
||||
mockSuccessfulEpisodeApplication(graph);
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
|
||||
const response = await POST(new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
episodeMode: true,
|
||||
situationGraph: graph,
|
||||
targetNodeId: "target",
|
||||
contributions: [{ targetNodeId: "target" }],
|
||||
}),
|
||||
}));
|
||||
|
||||
const body = await response.json();
|
||||
expect(response.status).toBe(200);
|
||||
expect(body.updatedSituationGraph.resolvedNodeIds).toEqual(
|
||||
expect.arrayContaining(["already-resolved", "target"]),
|
||||
);
|
||||
expect(body.updatedSituationGraph.nodes).toEqual(graph.nodes);
|
||||
expect(body.updatedSituationGraph.edges).toEqual(graph.edges);
|
||||
});
|
||||
|
||||
it("preserves meaningful episode mutations while authoritatively resolving the target", async () => {
|
||||
const graph = makeEpisodeGraph();
|
||||
const mutatedGraph = {
|
||||
...graph,
|
||||
nodes: [...graph.nodes, makeNode({ id: "meaningful", label: "Meaningful change", kind: "observation", status: "known" })],
|
||||
};
|
||||
mockSuccessfulEpisodeApplication(mutatedGraph);
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
|
||||
const response = await POST(new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({ episodeMode: true, situationGraph: graph, targetNodeId: "target", contributions: [{ targetNodeId: "target" }] }),
|
||||
}));
|
||||
|
||||
const body = await response.json();
|
||||
expect(body.updatedSituationGraph.nodes).toEqual(mutatedGraph.nodes);
|
||||
expect(body.updatedSituationGraph.resolvedNodeIds).toContain("target");
|
||||
});
|
||||
|
||||
it("does not return an authoritative closure when completed-episode reconsideration fails", async () => {
|
||||
const graph = makeEpisodeGraph();
|
||||
mockPrepareCompletedEpisode.mockReturnValue({ turns: [{ question: "Q?", answer: "A." }] });
|
||||
mockReconsiderCompletedEpisode.mockResolvedValue({ success: false, stage: "proposal_compatibility", error: "invalid" });
|
||||
const { POST } = await import("@/app/api/cases/update/route.js");
|
||||
|
||||
const response = await POST(new Request("http://localhost/api/cases/update", {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({ episodeMode: true, situationGraph: graph, targetNodeId: "target", contributions: [{ targetNodeId: "target" }] }),
|
||||
}));
|
||||
|
||||
expect(response.status).toBe(422);
|
||||
await expect(response.json()).resolves.not.toHaveProperty("updatedSituationGraph");
|
||||
});
|
||||
|
||||
it("invalid JSON returns 400", async () => {
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
|
||||
// ── Mock domain seam and provider at module level ───────────
|
||||
|
||||
const mockSynthesize = vi.fn();
|
||||
|
||||
vi.mock("@/lib/graph/current-understanding-synthesis.js", () => ({
|
||||
synthesizeCurrentUnderstanding: (...args) => mockSynthesize(...args),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/llm/provider.js", () => ({
|
||||
getProvider: () => ({}),
|
||||
getProviderModelName: () => "gpt-5.6-terra",
|
||||
}));
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────
|
||||
|
||||
function makeValidGraph() {
|
||||
return {
|
||||
centralStatement: "Test situation",
|
||||
nodes: [{ id: "n1", proposition: "Node prop" }],
|
||||
edges: [],
|
||||
};
|
||||
}
|
||||
|
||||
function makeRequest(body) {
|
||||
return new Request("http://localhost/api/cases/synthesis", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
}
|
||||
|
||||
// ── Route tests — valid POST ────────────────────────────────
|
||||
|
||||
describe("POST /api/cases/synthesis — valid request", () => {
|
||||
beforeEach(() => mockSynthesize.mockClear());
|
||||
|
||||
it("invokes synthesis domain seam with situationGraph + findings", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "Synthesized result" });
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ situationGraph: makeValidGraph(), findings: [] }));
|
||||
|
||||
expect(res.status).toBe(200);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(true);
|
||||
expect(data.currentUnderstanding).toBe("Synthesized result");
|
||||
expect(mockSynthesize).toHaveBeenCalledTimes(1);
|
||||
expect(mockSynthesize.mock.calls[0][1]).toMatchObject({ modelName: "gpt-5.6-terra" });
|
||||
});
|
||||
|
||||
it("returns narrative result on success", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "The revenue dropped because of X and Y." });
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(res.status).toBe(200);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(true);
|
||||
expect(typeof data.currentUnderstanding).toBe("string");
|
||||
expect(data.currentUnderstanding.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("passes findings to domain seam for eligibility filtering", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "OK" });
|
||||
const findings = [
|
||||
{ id: "f1", proposition: "agree finding", userDisposition: "agree", evaluation: "considered" },
|
||||
{ id: "f2", proposition: "null finding", userDisposition: null, evaluation: "considered" },
|
||||
{ id: "f3", proposition: "not_quite finding", userDisposition: "not_quite", evaluation: "considered" },
|
||||
];
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
await POST(makeRequest({ situationGraph: makeValidGraph(), findings }));
|
||||
|
||||
expect(mockSynthesize).toHaveBeenCalledTimes(1);
|
||||
expect(mockSynthesize.mock.calls[0][0].findings).toHaveLength(3);
|
||||
});
|
||||
});
|
||||
|
||||
// ── Route tests — error handling ────────────────────────────
|
||||
|
||||
describe("POST /api/cases/synthesis — error cases", () => {
|
||||
it("missing situationGraph → 400", async () => {
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ findings: [] }));
|
||||
|
||||
expect(res.status).toBe(400);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(false);
|
||||
expect(data.stage).toBe("request_validation");
|
||||
});
|
||||
|
||||
it("invalid JSON body → 400", async () => {
|
||||
const req = new Request("http://localhost/api/cases/synthesis", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: "not json",
|
||||
});
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(req);
|
||||
|
||||
expect(res.status).toBe(400);
|
||||
});
|
||||
|
||||
it("null body → 400", async () => {
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest(null));
|
||||
|
||||
expect(res.status).toBe(400);
|
||||
});
|
||||
|
||||
it("domain seam throws with statusCode → mapped status", async () => {
|
||||
mockSynthesize.mockRejectedValue(new Error("Provider failed"));
|
||||
// Add statusCode property to the error object after creation
|
||||
const err = Object.assign(new Error("Provider failed"), { statusCode: 502 });
|
||||
mockSynthesize.mockRejectedValue(err);
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(res.status).toBe(502);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(false);
|
||||
expect(data.stage).toBe("provider");
|
||||
});
|
||||
|
||||
it("domain seam throws without statusCode → 500", async () => {
|
||||
mockSynthesize.mockRejectedValue(new Error("unknown error"));
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(res.status).toBe(500);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(false);
|
||||
expect(data.stage).toBe("internal");
|
||||
});
|
||||
|
||||
it("domain seam throws 400 → mapped to 400", async () => {
|
||||
const err = Object.assign(new Error("Invalid input"), { statusCode: 400 });
|
||||
mockSynthesize.mockRejectedValue(err);
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
const res = await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(res.status).toBe(400);
|
||||
const data = await res.json();
|
||||
expect(data.success).toBe(false);
|
||||
expect(data.stage).toBe("request_validation");
|
||||
});
|
||||
});
|
||||
|
||||
// ── Route thinness — no business logic in route ─────────────
|
||||
|
||||
describe("Route thinness", () => {
|
||||
beforeEach(() => mockSynthesize.mockClear());
|
||||
|
||||
it("route does not filter eligibility itself (domain seam owns it)", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "OK" });
|
||||
const findings = [
|
||||
{ id: "f1", proposition: "ineligible", userDisposition: "not_quite", evaluation: "considered" },
|
||||
];
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
await POST(makeRequest({ situationGraph: makeValidGraph(), findings }));
|
||||
|
||||
// Route passes all findings through — domain seam filters
|
||||
expect(mockSynthesize.mock.calls[0][0].findings).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("route does not construct prompts", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "OK" });
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(mockSynthesize).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("route does not contain provider logic (delegates to domain seam)", async () => {
|
||||
mockSynthesize.mockResolvedValue({ currentUnderstanding: "OK" });
|
||||
|
||||
const { POST } = await import("@/app/api/cases/synthesis/route.js");
|
||||
await POST(makeRequest({ situationGraph: makeValidGraph() }));
|
||||
|
||||
expect(mockSynthesize).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,136 @@
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { NextRequest } from "next/server";
|
||||
|
||||
const mockGetAuthenticatedUser = vi.fn();
|
||||
const mockStartCase = vi.fn();
|
||||
const mockGetUser = vi.fn();
|
||||
const mockGetConfig = vi.fn();
|
||||
|
||||
vi.mock("@/lib/supabase/server.js", () => ({
|
||||
getAuthenticatedUser: () => mockGetAuthenticatedUser(),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/graph/orchestrator.js", () => ({
|
||||
startCase: (...args) => mockStartCase(...args),
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/config", () => ({
|
||||
getConfig: () => mockGetConfig(),
|
||||
}));
|
||||
|
||||
vi.mock("@supabase/ssr", () => ({
|
||||
createServerClient: () => ({
|
||||
auth: {
|
||||
getUser: () => mockGetUser(),
|
||||
exchangeCodeForSession: vi.fn().mockResolvedValue(undefined),
|
||||
},
|
||||
}),
|
||||
}));
|
||||
|
||||
describe("auth callback redirect origin", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("uses forwarded host/proto for redirect when behind proxy", async () => {
|
||||
const { GET } = await import("@/app/auth/callback/route.js");
|
||||
|
||||
const request = new Request("http://0.0.0.0:3000/auth/callback?code=abc123", {
|
||||
headers: {
|
||||
"x-forwarded-host": "confidence.rdbcloud.co.uk",
|
||||
"x-forwarded-proto": "https",
|
||||
},
|
||||
});
|
||||
|
||||
const response = await GET(request);
|
||||
|
||||
expect(response.status).toBe(307);
|
||||
expect(response.headers.get("location")).toBe("https://confidence.rdbcloud.co.uk/");
|
||||
});
|
||||
|
||||
it("falls back to request origin when no forwarded headers", async () => {
|
||||
const { GET } = await import("@/app/auth/callback/route.js");
|
||||
|
||||
const request = new Request("http://localhost:3000/auth/callback?code=xyz");
|
||||
|
||||
const response = await GET(request);
|
||||
|
||||
expect(response.status).toBe(307);
|
||||
expect(response.headers.get("location")).toBe("http://localhost:3000/");
|
||||
});
|
||||
});
|
||||
|
||||
describe("authenticated product boundary", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("rejects an unauthenticated protected API request", async () => {
|
||||
mockGetAuthenticatedUser.mockResolvedValue(null);
|
||||
const { withAuthenticatedApi } = await import("@/lib/supabase/api-auth.js");
|
||||
const handler = vi.fn();
|
||||
|
||||
const response = await withAuthenticatedApi(handler)(new Request("http://localhost/api/cases/start"));
|
||||
|
||||
expect(response.status).toBe(401);
|
||||
expect(handler).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("allows an authenticated protected API request to reach existing route behavior", async () => {
|
||||
mockGetAuthenticatedUser.mockResolvedValue({ id: "user-1" });
|
||||
mockStartCase.mockResolvedValue({ success: true, updatedSituationGraph: {} });
|
||||
const { POST } = await import("@/app/api/cases/start/route.js");
|
||||
|
||||
const response = await POST(new Request("http://localhost/api/cases/start", {
|
||||
method: "POST",
|
||||
body: JSON.stringify({ scenario: "A scenario" }),
|
||||
}));
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
await expect(response.json()).resolves.toMatchObject({ success: true });
|
||||
expect(mockStartCase).toHaveBeenCalledWith({ scenario: "A scenario" });
|
||||
});
|
||||
|
||||
it("supplies the auth callback as the magic-link redirect target", async () => {
|
||||
const { magicLinkRedirectTo } = await import("@/lib/supabase/browser.js");
|
||||
expect(magicLinkRedirectTo("http://localhost:3000")).toBe("http://localhost:3000/auth/callback");
|
||||
});
|
||||
|
||||
it("keeps infrastructure health public and does not leak config details", async () => {
|
||||
mockGetConfig.mockReturnValue({ ok: true, config: {} });
|
||||
const { GET } = await import("@/app/api/health/route.js");
|
||||
|
||||
const response = await GET();
|
||||
|
||||
expect(mockGetAuthenticatedUser).not.toHaveBeenCalled();
|
||||
expect(response.status).toBe(200);
|
||||
await expect(response.json()).resolves.toEqual({ healthy: true });
|
||||
});
|
||||
|
||||
it("remains healthy when reasoning configuration is missing", async () => {
|
||||
mockGetConfig.mockReturnValue({ ok: false });
|
||||
const { GET } = await import("@/app/api/health/route.js");
|
||||
|
||||
const response = await GET();
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
await expect(response.json()).resolves.toEqual({ healthy: true });
|
||||
});
|
||||
|
||||
it("does not convert /api/health to 401 via middleware when unauthenticated", async () => {
|
||||
mockGetUser.mockResolvedValue({ data: { user: null } });
|
||||
const { middleware } = await import("@/middleware.js");
|
||||
const response = await middleware(new NextRequest("http://localhost:3000/api/health"));
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
});
|
||||
|
||||
it("redirects unauthenticated product access to the login surface", async () => {
|
||||
mockGetUser.mockResolvedValue({ data: { user: null } });
|
||||
const { middleware } = await import("@/middleware.js");
|
||||
const response = await middleware(new NextRequest("http://localhost:3000/"));
|
||||
|
||||
expect(response.status).toBe(307);
|
||||
expect(response.headers.get("location")).toBe("http://localhost:3000/login?next=%2F");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,303 @@
|
||||
/**
|
||||
* v0.53 - Empty Done Orchestration Regression Test
|
||||
*
|
||||
* Proves that handleDoneForNowPromotion's gate is based on active-target
|
||||
* focused contributions, NOT on scenario-wide findings.
|
||||
*
|
||||
* Deterministic: exercises the exact ownership condition logic extracted
|
||||
* from the component boundary. No rendering, no fetch.
|
||||
*/
|
||||
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { reopenResolvedUnknown } from "@/lib/graph/reopen-resolved-unknown.js";
|
||||
|
||||
/* - Extracted gate logic (mirrors scenario-form.jsx line ~362) -- */
|
||||
|
||||
function shouldEnterEpisodePath(targetNodeId, focusedContributions) {
|
||||
if (!targetNodeId) return false;
|
||||
const hasActiveTargetContent = (focusedContributions ?? []).some(
|
||||
(c) => c.targetNodeId === targetNodeId || c.originatingTargetNodeId === targetNodeId,
|
||||
);
|
||||
return hasActiveTargetContent;
|
||||
}
|
||||
|
||||
/* - Case A - empty active target -- */
|
||||
|
||||
describe("empty active target", () => {
|
||||
it("returns false when focusedContributions contains no contribution owned by the target", () => {
|
||||
const result = shouldEnterEpisodePath(
|
||||
"node-B",
|
||||
[
|
||||
{ id: "contrib-0001", targetNodeId: "node-A", originatingTargetNodeId: "node-A" },
|
||||
{ id: "contrib-0002", targetNodeId: "node-A", originatingTargetNodeId: "node-A" },
|
||||
],
|
||||
);
|
||||
expect(result).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false when focusedContributions is empty array", () => {
|
||||
const result = shouldEnterEpisodePath("node-B", []);
|
||||
expect(result).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false when focusedContributions is undefined/null", () => {
|
||||
expect(shouldEnterEpisodePath("node-B", undefined)).toBe(false);
|
||||
expect(shouldEnterEpisodePath("node-B", null)).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false when targetNodeId is empty string", () => {
|
||||
const result = shouldEnterEpisodePath("", [{ id: "contrib-0001", targetNodeId: "node-A" }]);
|
||||
expect(result).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false when targetNodeId is null", () => {
|
||||
const result = shouldEnterEpisodePath(null, [{ id: "contrib-0001", targetNodeId: "node-A" }]);
|
||||
expect(result).toBe(false);
|
||||
});
|
||||
|
||||
it("originatingTargetNodeId ownership also gates correctly - mismatched origin", () => {
|
||||
const result = shouldEnterEpisodePath(
|
||||
"node-B",
|
||||
[{ id: "contrib-0001", targetNodeId: "node-A", originatingTargetNodeId: "node-C" }],
|
||||
);
|
||||
expect(result).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
/* - Case A2 — immediate Done produces canonical parked state (graph coherence) -- */
|
||||
|
||||
function simulateImmediateDone(graph, nodeId) {
|
||||
/* Mirrors the onImmediateGraphChange logic in FocusedWorkspaceNavigation */
|
||||
const resolvedIds = new Set(graph.resolvedNodeIds || []);
|
||||
resolvedIds.add(nodeId);
|
||||
const nextNodes = (graph.nodes || []).map((n) =>
|
||||
n.id === nodeId ? { ...n, status: "resolved" } : n,
|
||||
);
|
||||
return {
|
||||
...graph,
|
||||
nodes: nextNodes,
|
||||
resolvedNodeIds: Array.from(resolvedIds),
|
||||
};
|
||||
}
|
||||
|
||||
describe("immediate Done produces canonical parked state", () => {
|
||||
it("sets node.status to resolved and adds ID to resolvedNodeIds for empty-Done target", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-empty", kind: "unknown", status: "unknown" }],
|
||||
resolvedNodeIds: [],
|
||||
};
|
||||
|
||||
const nextGraph = simulateImmediateDone(graph, "node-empty");
|
||||
|
||||
const targetNode = nextGraph.nodes.find((n) => n.id === "node-empty");
|
||||
expect(targetNode.status).toBe("resolved");
|
||||
expect(nextGraph.resolvedNodeIds).toContain("node-empty");
|
||||
});
|
||||
|
||||
it("preserves other nodes unchanged", () => {
|
||||
const graph = {
|
||||
nodes: [
|
||||
{ id: "node-A", kind: "unknown", status: "unknown" },
|
||||
{ id: "node-B", kind: "unknown", status: "unknown" },
|
||||
],
|
||||
resolvedNodeIds: [],
|
||||
};
|
||||
|
||||
const nextGraph = simulateImmediateDone(graph, "node-B");
|
||||
|
||||
const nodeA = nextGraph.nodes.find((n) => n.id === "node-A");
|
||||
expect(nodeA.status).toBe("unknown");
|
||||
});
|
||||
|
||||
it("is idempotent for resolvedNodeIds — duplicate add does not create duplicates", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-X", kind: "unknown", status: "resolved" }],
|
||||
resolvedNodeIds: ["node-X"],
|
||||
};
|
||||
|
||||
const nextGraph = simulateImmediateDone(graph, "node-X");
|
||||
|
||||
expect(nextGraph.resolvedNodeIds.filter((id) => id === "node-X").length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
/* - Case B - populated active target (regression guard) -- */
|
||||
|
||||
describe("populated active target", () => {
|
||||
it("returns true when a contribution's targetNodeId matches", () => {
|
||||
const result = shouldEnterEpisodePath(
|
||||
"node-B",
|
||||
[
|
||||
{ id: "contrib-0001", targetNodeId: "node-A" },
|
||||
{ id: "contrib-0002", targetNodeId: "node-B" },
|
||||
],
|
||||
);
|
||||
expect(result).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true when a contribution's originatingTargetNodeId matches", () => {
|
||||
const result = shouldEnterEpisodePath(
|
||||
"node-B",
|
||||
[
|
||||
{ id: "contrib-0001", targetNodeId: "node-A" },
|
||||
{ id: "contrib-0002", originatingTargetNodeId: "node-B", targetNodeId: "node-B" },
|
||||
],
|
||||
);
|
||||
expect(result).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true with a single contribution owned by the target", () => {
|
||||
const result = shouldEnterEpisodePath("node-B", [
|
||||
{ id: "contrib-0001", targetNodeId: "node-B" },
|
||||
]);
|
||||
expect(result).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true when contributions exist for the target even if others are also present", () => {
|
||||
const result = shouldEnterEpisodePath(
|
||||
"node-B",
|
||||
[
|
||||
{ id: "contrib-0001", targetNodeId: "node-A" },
|
||||
{ id: "contrib-0002", originatingTargetNodeId: "node-B" },
|
||||
{ id: "contrib-0003", targetNodeId: "node-C" },
|
||||
],
|
||||
);
|
||||
expect(result).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
/* - Integration: scenario-wide findings must not leak through -- */
|
||||
|
||||
describe("scenario-wide evidence does not influence gate", () => {
|
||||
it("findings for other questions do not cause empty target to enter episode path", () => {
|
||||
// Even if there are many findings elsewhere, the empty target (node-B)
|
||||
// must NOT enter episode processing.
|
||||
const focusedContributions = [
|
||||
{ id: "contrib-0001", targetNodeId: "node-A", originatingTargetNodeId: "node-A" },
|
||||
{ id: "contrib-0002", targetNodeId: "node-A", originatingTargetNodeId: "node-A" },
|
||||
{ id: "contrib-0003", targetNodeId: "node-A", originatingTargetNodeId: "node-A" },
|
||||
];
|
||||
|
||||
// node-B has zero focused contributions - should NOT enter episode path
|
||||
expect(shouldEnterEpisodePath("node-B", focusedContributions)).toBe(false);
|
||||
|
||||
// node-A has contributions - should enter episode path (existing behavior preserved)
|
||||
expect(shouldEnterEpisodePath("node-A", focusedContributions)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
/* - Case B2 — Re-open reverses canonical + local parking state -- */
|
||||
|
||||
describe("Re-open reverses canonical graph state", () => {
|
||||
it("changes node.status from resolved back to unknown", () => {
|
||||
const graph = {
|
||||
nodes: [
|
||||
{ id: "node-parked", kind: "unknown", status: "resolved" },
|
||||
{ id: "node-other", kind: "unknown", status: "unknown" },
|
||||
],
|
||||
resolvedNodeIds: ["node-parked"],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-parked");
|
||||
|
||||
const targetNode = nextGraph.nodes.find((n) => n.id === "node-parked");
|
||||
expect(targetNode.status).toBe("unknown");
|
||||
});
|
||||
|
||||
it("removes target from resolvedNodeIds", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-parked", kind: "unknown", status: "resolved" }],
|
||||
resolvedNodeIds: ["node-parked"],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-parked");
|
||||
|
||||
expect(nextGraph.resolvedNodeIds).not.toContain("node-parked");
|
||||
});
|
||||
|
||||
it("preserves other resolved nodes", () => {
|
||||
const graph = {
|
||||
nodes: [
|
||||
{ id: "node-A", kind: "unknown", status: "resolved" },
|
||||
{ id: "node-B", kind: "unknown", status: "resolved" },
|
||||
{ id: "node-C", kind: "unknown", status: "unknown" },
|
||||
],
|
||||
resolvedNodeIds: ["node-A", "node-B"],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-A");
|
||||
|
||||
expect(nextGraph.resolvedNodeIds).toContain("node-B");
|
||||
expect(nextGraph.resolvedNodeIds).not.toContain("node-A");
|
||||
});
|
||||
|
||||
it("returns original graph when node is not a resolved unknown", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-unknown", kind: "unknown", status: "unknown" }],
|
||||
resolvedNodeIds: [],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-unknown");
|
||||
|
||||
expect(nextGraph).toBe(graph);
|
||||
});
|
||||
|
||||
it("returns original graph when node is not an unknown kind", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-assumption", kind: "assumption", status: "resolved" }],
|
||||
resolvedNodeIds: ["node-assumption"],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-assumption");
|
||||
|
||||
expect(nextGraph).toBe(graph);
|
||||
});
|
||||
|
||||
it("local setDoneForNowIds cleanup removes only the reopened target", () => {
|
||||
const prevDoneIds = ["node-A", "node-parked", "node-C"];
|
||||
const nextDoneIds = prevDoneIds.filter((id) => id !== "node-parked");
|
||||
expect(nextDoneIds).toContain("node-A");
|
||||
expect(nextDoneIds).toContain("node-C");
|
||||
expect(nextDoneIds).not.toContain("node-parked");
|
||||
});
|
||||
|
||||
it("local setDoneForNowIds is no-op when target not in array", () => {
|
||||
const prevDoneIds = ["node-A", "node-B"];
|
||||
const nextDoneIds = prevDoneIds.filter((id) => id !== "node-parked");
|
||||
expect(nextDoneIds).toEqual(prevDoneIds);
|
||||
});
|
||||
});
|
||||
|
||||
/* - Case C — history preservation (graph structure invariance) -- */
|
||||
|
||||
describe("Re-open preserves contributions and findings", () => {
|
||||
it("does not create or delete nodes", () => {
|
||||
const originalNodes = [
|
||||
{ id: "node-parked", kind: "unknown", status: "resolved" },
|
||||
{ id: "node-A", kind: "assumption", status: "resolved" },
|
||||
];
|
||||
const graph = {
|
||||
nodes: originalNodes,
|
||||
resolvedNodeIds: ["node-parked"],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-parked");
|
||||
|
||||
expect(nextGraph.nodes.length).toBe(originalNodes.length);
|
||||
});
|
||||
|
||||
it("does not create or delete edges", () => {
|
||||
const graph = {
|
||||
nodes: [{ id: "node-parked", kind: "unknown", status: "resolved" }],
|
||||
resolvedNodeIds: ["node-parked"],
|
||||
edges: [
|
||||
{ from: "node-parked", to: "node-A", type: "supports" },
|
||||
{ from: "node-B", to: "node-parked", type: "refutes" },
|
||||
],
|
||||
};
|
||||
|
||||
const nextGraph = reopenResolvedUnknown(graph, "node-parked");
|
||||
|
||||
expect(nextGraph.edges.length).toBe(2);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"scenario": "I run a small manufacturing business. Customer complaints have risen by 35% over the last six months, while production volume increased by 40%.\n\nMost complaints mention late delivery or minor product defects, but our complaint categories changed when we introduced a new CRM tagging system three months ago. During the same period we also changed one supplier and introduced a weekend production shift.\n\nI am deciding whether to spend about £120,000 on automated quality inspection now or wait until we understand whether there is actually a quality problem.\n\nI do not yet know the complaint rate per unit, whether defect rates differ by shift or supplier, or whether the new tagging system changed what gets counted as a complaint."
|
||||
}
|
||||
@@ -6,25 +6,34 @@
|
||||
* API response targetNodeId regardless of what the model returns.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, vi } from "vitest";
|
||||
import { afterEach, beforeEach, describe, it, expect, vi } from "vitest";
|
||||
import { focusedDeconstructJsonSchema, validateFocusedDeconstructSchema } from "@/lib/graph/focused-investigation";
|
||||
|
||||
vi.mock("@/lib/supabase/api-auth.js", () => ({
|
||||
withAuthenticatedApi: (handler) => handler,
|
||||
}));
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
function makeMockProvider(inventedTargetNodeId) {
|
||||
return {
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
targetNodeId: inventedTargetNodeId,
|
||||
observations: ["doc is minimal", "processes in founder's head"],
|
||||
uncertainties: ["whether formal docs can capture tacit knowledge"],
|
||||
assumptions: ["documentation is primary mechanism for knowledge transfer"],
|
||||
relationships: [
|
||||
{ from: "founder", to: "processes", type: "holds", rationale: "tacit" },
|
||||
{ from: "ops-context", to: "docs-infra", type: "depends_on", rationale: "formal docs required" },
|
||||
],
|
||||
possibleFollowUpQuestions: [
|
||||
"What processes does the founder hold tacitly?",
|
||||
"How is knowledge transferred when founder is unavailable?",
|
||||
],
|
||||
response: {
|
||||
targetNodeId: inventedTargetNodeId,
|
||||
observations: ["doc is minimal", "processes in founder's head"],
|
||||
uncertainties: ["whether formal docs can capture tacit knowledge"],
|
||||
assumptions: ["documentation is primary mechanism for knowledge transfer"],
|
||||
relationships: [
|
||||
{ from: "founder", to: "processes", type: "holds" },
|
||||
{ from: "ops-context", to: "docs-infra", type: "depends_on" },
|
||||
],
|
||||
possibleFollowUpQuestions: [
|
||||
"What processes does the founder hold tacitly?",
|
||||
"How is knowledge transferred when founder is unavailable?",
|
||||
],
|
||||
},
|
||||
providerApiPath: "/api/chat",
|
||||
providerExecution: { chatRequestAttempted: true },
|
||||
}),
|
||||
};
|
||||
}
|
||||
@@ -32,12 +41,48 @@ function makeMockProvider(inventedTargetNodeId) {
|
||||
// ── Boundary test ────────────────────────────────────────────────────────
|
||||
|
||||
describe("focused-deconstruct targetNodeId identity boundary", () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.doUnmock("@/lib/llm/provider");
|
||||
});
|
||||
|
||||
it("enforces the canonical focused relationship structure", () => {
|
||||
const base = {
|
||||
targetNodeId: "node-id", observations: [], uncertainties: [], assumptions: [],
|
||||
possibleFollowUpQuestions: [],
|
||||
};
|
||||
expect(validateFocusedDeconstructSchema({ ...base, relationships: [] })).toEqual([]);
|
||||
expect(validateFocusedDeconstructSchema({
|
||||
...base,
|
||||
relationships: [{ from: "supplier changed", to: "defect rate increased", type: "associated with" }],
|
||||
})).toEqual([]);
|
||||
|
||||
const invalidRelationships = [
|
||||
"not an array",
|
||||
[{}],
|
||||
[{ from: "a", type: "links" }],
|
||||
[{ from: "a", to: "b" }],
|
||||
[{ from: "", to: "b", type: "links" }],
|
||||
[{ from: "a", to: "", type: "links" }],
|
||||
[{ from: "a", to: "b", type: "" }],
|
||||
[{ from: "a", to: "b", type: "links", rationale: "extra" }],
|
||||
[{ from: "a", to: "b", type: "links", extra: "extra" }],
|
||||
];
|
||||
invalidRelationships.forEach((relationships) => {
|
||||
expect(validateFocusedDeconstructSchema({ ...base, relationships }).length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
|
||||
it("request targetNodeId overrides model-invented targetNodeId", async () => {
|
||||
const requestTargetNodeId = "nk04xvk"; // original graph node ID
|
||||
const inventedModelId = "invented-model-id";
|
||||
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => makeMockProvider(inventedModelId),
|
||||
getProviderModelName: () => "configured-model",
|
||||
}));
|
||||
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
@@ -69,13 +114,53 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
|
||||
expect(json.targetNodeId).not.toBe(inventedModelId);
|
||||
});
|
||||
|
||||
it("supplies the focused-deconstruction schema through the provider seam", async () => {
|
||||
const generateReconstruction = vi.fn().mockResolvedValue({
|
||||
response: {
|
||||
targetNodeId: "model-id",
|
||||
observations: [],
|
||||
uncertainties: [],
|
||||
assumptions: [],
|
||||
relationships: [],
|
||||
possibleFollowUpQuestions: [],
|
||||
},
|
||||
providerApiPath: "/api/chat",
|
||||
providerExecution: { chatRequestAttempted: true },
|
||||
});
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({ generateReconstruction }),
|
||||
getProviderModelName: () => "gpt-5.6-terra",
|
||||
}));
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
|
||||
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
targetNodeId: "node-id",
|
||||
targetLabel: "label",
|
||||
targetDescription: "description",
|
||||
centralStatement: "central statement",
|
||||
question: "question?",
|
||||
answer: "answer.",
|
||||
}),
|
||||
}));
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
expect(generateReconstruction).toHaveBeenCalledWith(
|
||||
expect.any(String),
|
||||
"gpt-5.6-terra",
|
||||
focusedDeconstructJsonSchema,
|
||||
);
|
||||
});
|
||||
|
||||
it("semantic fields pass through unchanged from model", async () => {
|
||||
const mockObs = ["doc is minimal", "processes in founder's head"];
|
||||
const mockUnc = ["whether formal docs can capture tacit knowledge"];
|
||||
const mockAssm = ["documentation is primary mechanism for knowledge transfer"];
|
||||
const mockRel = [
|
||||
{ from: "founder", to: "processes", type: "holds", rationale: "tacit" },
|
||||
{ from: "ops-context", to: "docs-infra", type: "depends_on", rationale: "formal docs required" },
|
||||
{ from: "founder", to: "processes", type: "holds" },
|
||||
{ from: "ops-context", to: "docs-infra", type: "depends_on" },
|
||||
];
|
||||
const mockFuq = [
|
||||
"What processes does the founder hold tacitly?",
|
||||
@@ -85,14 +170,19 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
targetNodeId: "some-invented-id",
|
||||
observations: mockObs,
|
||||
uncertainties: mockUnc,
|
||||
assumptions: mockAssm,
|
||||
relationships: mockRel,
|
||||
possibleFollowUpQuestions: mockFuq,
|
||||
response: {
|
||||
targetNodeId: "some-invented-id",
|
||||
observations: mockObs,
|
||||
uncertainties: mockUnc,
|
||||
assumptions: mockAssm,
|
||||
relationships: mockRel,
|
||||
possibleFollowUpQuestions: mockFuq,
|
||||
},
|
||||
providerApiPath: "/api/chat",
|
||||
providerExecution: { chatRequestAttempted: true },
|
||||
}),
|
||||
}),
|
||||
getProviderModelName: () => "configured-model",
|
||||
}));
|
||||
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
@@ -149,28 +239,30 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
|
||||
});
|
||||
|
||||
it("full identity path: request → response → contribution", async () => {
|
||||
// Reset modules to avoid mock leakage from earlier tests
|
||||
vi.resetModules();
|
||||
|
||||
const originalNodeId = "nk04xvk";
|
||||
const modelInventedId = "investigation_node_responsibility_distribution_autonomy";
|
||||
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
targetNodeId: modelInventedId,
|
||||
observations: ["Documentation is minimal."],
|
||||
uncertainties: [],
|
||||
assumptions: [
|
||||
"That formal documentation is the primary mechanism for capturing or transferring the founder's tacit knowledge of processes.",
|
||||
],
|
||||
relationships: [
|
||||
{ from: "Founder", to: "Processes", type: "holds" },
|
||||
{ from: "Operational Context", to: "Documentation Infrastructure", type: "affects" },
|
||||
],
|
||||
possibleFollowUpQuestions: [],
|
||||
response: {
|
||||
targetNodeId: modelInventedId,
|
||||
observations: ["Documentation is minimal."],
|
||||
uncertainties: [],
|
||||
assumptions: [
|
||||
"That formal documentation is the primary mechanism for capturing or transferring the founder's tacit knowledge of processes.",
|
||||
],
|
||||
relationships: [
|
||||
{ from: "Founder", to: "Processes", type: "holds" },
|
||||
{ from: "Operational Context", to: "Documentation Infrastructure", type: "affects" },
|
||||
],
|
||||
possibleFollowUpQuestions: [],
|
||||
},
|
||||
providerApiPath: "/api/chat",
|
||||
providerExecution: { chatRequestAttempted: true },
|
||||
}),
|
||||
}),
|
||||
getProviderModelName: () => "configured-model",
|
||||
}));
|
||||
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
@@ -217,4 +309,144 @@ describe("focused-deconstruct targetNodeId identity boundary", () => {
|
||||
const stored = { ...json };
|
||||
expect(stored.targetNodeId).toBe(originalNodeId);
|
||||
});
|
||||
|
||||
it("provider envelope fields do not leak into API response", async () => {
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
response: {
|
||||
targetNodeId: "nk04xvk",
|
||||
observations: ["obs"],
|
||||
uncertainties: ["unc"],
|
||||
assumptions: ["asm"],
|
||||
relationships: [],
|
||||
possibleFollowUpQuestions: ["fuq"],
|
||||
},
|
||||
providerApiPath: "/api/chat",
|
||||
providerExecution: { chatRequestAttempted: true, chatRequestSucceeded: true },
|
||||
}),
|
||||
}),
|
||||
getProviderModelName: () => "configured-model",
|
||||
}));
|
||||
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
|
||||
const response = await POST(
|
||||
new Request("http://localhost/api/focused-investigation/deconstruct", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
targetNodeId: "nk04xvk",
|
||||
targetLabel: "label",
|
||||
targetDescription: "desc",
|
||||
centralStatement: "central",
|
||||
question: "q?",
|
||||
answer: "a.",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
const json = await response.json();
|
||||
|
||||
// Semantic fields present
|
||||
expect(json.success).toBe(true);
|
||||
expect(json.targetNodeId).toBe("nk04xvk");
|
||||
expect(json.observations).toEqual(["obs"]);
|
||||
expect(json.possibleFollowUpQuestions).toEqual(["fuq"]);
|
||||
|
||||
// Provider envelope fields must NOT appear in the response
|
||||
expect(json.providerApiPath).toBeUndefined();
|
||||
expect(json.providerExecution).toBeUndefined();
|
||||
});
|
||||
|
||||
it("preserves the 500 provider-failure contract while logging structural diagnostics", async () => {
|
||||
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockRejectedValue(Object.assign(new Error("provider failed"), {
|
||||
providerApiPath: "/v1/responses",
|
||||
statusCode: 400,
|
||||
})),
|
||||
}),
|
||||
getProviderModelName: () => "gpt-5.6-terra",
|
||||
}));
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
|
||||
try {
|
||||
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
targetNodeId: "node-id", targetLabel: "label", targetDescription: "description",
|
||||
centralStatement: "central", question: "question?", answer: "answer.",
|
||||
}),
|
||||
}));
|
||||
expect(response.status).toBe(500);
|
||||
await expect(response.json()).resolves.toEqual({ error: "provider failed" });
|
||||
expect(errorSpy).toHaveBeenCalledWith(
|
||||
"[api/focused-investigation/deconstruct] provider failure",
|
||||
expect.objectContaining({ targetNodeId: "node-id", providerApiPath: "/v1/responses" }),
|
||||
);
|
||||
} finally {
|
||||
errorSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
it("returns a sanitized 503 when the provider is unavailable", async () => {
|
||||
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockRejectedValue(Object.assign(new Error(
|
||||
"Ollama /api/generate request timed out after 5 minutes",
|
||||
), { code: "PROVIDER_UNAVAILABLE", providerApiPath: "/api/generate" })),
|
||||
}),
|
||||
getProviderModelName: () => "configured-model",
|
||||
}));
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
|
||||
try {
|
||||
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
targetNodeId: "node-id", targetLabel: "label", targetDescription: "description",
|
||||
centralStatement: "central", question: "question?", answer: "answer.",
|
||||
}),
|
||||
}));
|
||||
expect(response.status).toBe(503);
|
||||
const json = await response.json();
|
||||
expect(json).toEqual({ error: "Reasoning service is temporarily unavailable." });
|
||||
expect(JSON.stringify(json)).not.toMatch(/ollama|generate|timed out/i);
|
||||
} finally {
|
||||
errorSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
it("preserves the 502 validation-failure contract with diagnostics", async () => {
|
||||
vi.doMock("@/lib/llm/provider", () => ({
|
||||
getProvider: () => ({
|
||||
generateReconstruction: vi.fn().mockResolvedValue({
|
||||
response: {}, providerApiPath: "/v1/responses",
|
||||
}),
|
||||
}),
|
||||
getProviderModelName: () => "gpt-5.6-terra",
|
||||
}));
|
||||
const { POST } = await import("../app/api/focused-investigation/deconstruct/route.js");
|
||||
const response = await POST(new Request("http://localhost/api/focused-investigation/deconstruct", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
targetNodeId: "node-id", targetLabel: "label", targetDescription: "description",
|
||||
centralStatement: "central", question: "question?", answer: "answer.",
|
||||
}),
|
||||
}));
|
||||
|
||||
expect(response.status).toBe(502);
|
||||
await expect(response.json()).resolves.toMatchObject({
|
||||
success: false,
|
||||
error: "Focused deconstruction result did not match expected schema",
|
||||
targetNodeId: "node-id",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user