Files
confidence-engine/tests/graph/focused-deconstruct-regression-cases.mjs
T

279 lines
14 KiB
JavaScript

export const FOCUSED_DECONSTRUCT_REGRESSION_CASES = [
{
id: "nearest-frontier-wins",
centralStatement:
"I need to understand what currently prevents more operational responsibility moving away from me.",
targetLabel: "Team capability constraints on task handoff boundaries",
targetDescription:
"Which routine work is already handled independently and what still causes work to return to the owner.",
question:
"What was the comparable state before team capability constraints on task handoff boundaries?",
answer:
"She hasn't had any formal training in this area, but she already handles the weekly supplier payments herself and has done that without me checking them for the last six months. She still brings failed payments or anything unusual to me.",
manualReviewCriteria:
`Must preserve the stated payment facts without strengthening them.
Nearest frontier is the distinction between routine payments and failed/unusual payments that return to the user.
Must not jump first to:
other tasks
broader delegation
transferability
future staffing
Exactly one uncertainty.
Exactly one follow-up question.`,
},
{
id: "no-genuine-assumption",
centralStatement:
"I am trying to understand whether there is evidence of duplicate supplier payments.",
targetLabel: "Evidence of duplicate supplier payments",
targetDescription:
"What the available transaction evidence currently establishes.",
question: "What have you checked so far?",
answer: "I reviewed the last six months of transaction logs and didn't find any duplicate payments.",
manualReviewCriteria:
`Observation must preserve exactly what was checked and what was not found.
Must not invent assumptions about:
the logs being complete
the review being perfect
duplicates being impossible
the payment process being reliable
assumptions should be [] unless the production model identifies something genuinely required by the answer's meaning.
Exactly one nearest uncertainty.
Exactly one next question.`,
},
{
id: "genuine-evidence-sufficiency-assumption",
centralStatement:
"I need to understand where responsibility can safely move away from me without creating unacceptable operational risk.",
targetLabel: "Team capability constraints on task handoff boundaries",
targetDescription:
"Which payment responsibilities can be handled independently and what currently limits further handoff.",
question: "Why do you still review supplier payments over £10,000 yourself?",
answer: "Because she's only ever handled the normal weekly payments, which are usually under £2,000.",
manualReviewCriteria:
`A genuine implicit reasoning dependency exists:
her lower-value experience is not yet sufficient evidence, by itself, for handing over the higher-value payments.
Exact wording need not match.
Must not strengthen into:
she is incapable
large payments are inherently unsafe
formal training is required
senior approval is mandatory
Exactly one nearest uncertainty.
Exactly one next question.`,
},
{
id: "juxtaposition-must-not-create-link",
centralStatement:
"I am collecting the facts that changed this week before deciding what matters.",
targetLabel: "Recent project changes",
targetDescription:
"Facts that changed recently without assuming how they relate.",
question: "What changed this week?",
answer: "The vendor quoted £50,000. We have three weeks left for approval.",
manualReviewCriteria:
`Both facts must be preserved.
Must not automatically assert:
the £50,000 quote caused the three-week constraint
the approval deadline is caused by cost
the project is unaffordable
the vendor quote threatens approval
Co-occurrence alone must not become a relationship or user-held assumption.
Exactly one uncertainty.
Exactly one next question.`,
},
{
id: "tentative-observation-fidelity",
centralStatement:
"I am trying to understand whether the delivery timeline is genuinely at risk.",
targetLabel: "Timeline risk",
targetDescription:
"What is known and uncertain about whether cost changes could affect timing.",
question: "What makes you think the timeline might need to change?",
answer: "I'm not certain, but I think we might need to reconsider the timeline if costs go up.",
manualReviewCriteria:
`Must preserve:
uncertainty
conditional language
tentative stance
Must not turn this into:
costs will rise
the timeline will change
cost increases cause delay
Subjective/tentative meaning must not become objective fact.
Exactly one nearest uncertainty.
Exactly one next question.`,
},
{
id: "genuine-dependency-frontier",
centralStatement:
"I need to understand which operational responsibilities still depend directly on me and why.",
targetLabel: "Payment approval dependency on owner access",
targetDescription:
"Why supplier payments cannot currently be completed when the owner is unavailable.",
question: "Why do supplier payments have to wait until you are available?",
answer: "Because the banking approval token is linked only to my login.",
manualReviewCriteria:
`Direct observation:
token is linked only to the user's login.
Genuine implicit dependency:
payment completion requires access to that token.
Must not strengthen into:
only the owner can ever approve
bank policy prohibits delegation
another login cannot be authorised
the dependency is permanent
the team lacks capability
Nearest frontier:
whether token access itself is the actual blocker or whether another requirement also exists.
Exactly one uncertainty.
Exactly one next question.`,
},
{
id: "structural-frontier-over-working-example-detail",
centralStatement:
"I want to reduce how dependent the business is on me for routine operational work, but I am not sure how much responsibility I can realistically hand over to my team yet. I want to understand what is genuinely preventing more delegation rather than just assuming I need to stay involved.",
targetLabel:
"Concrete capability gaps, process maturity levels, or approval authority restrictions limiting delegation",
targetDescription:
"What is actually stopping further delegation right now — not hypothetical constraints but the real bottleneck in the current evidence.",
question:
"What evidence would confirm or rule out concrete capability gaps, process maturity levels, or approval authority restrictions limiting delegation?",
answer:
"One member of the team already handles the weekly supplier payments herself. She has done that for the last six months without me checking the routine payments. She hasn't had any formal training in this area. If a payment fails or there is anything unusual, she brings it to me. I still handle other routine operational work myself, but I haven't yet broken all of that down into a complete list.",
manualReviewCriteria:
`Observation fidelity — the reasoning must preserve the stated facts without weakening them:
- weekly supplier payments are already handled independently by one team member
- those routine payments have operated for six months without routine review
- no formal training exists for that specific task
- failures or unusual cases are escalated back to the owner
- other routine work remains with the owner
- remaining routine work has not yet been fully identified / broken down
Frontier priority (the key invariant):
A local unresolved detail inside an already-working delegation example must NOT outrank a structural information gap required to progress the actual investigation.
Specifically:
exception mechanics: "Can she also handle failed/unusual payments?"
is SAME-BRANCH-DETAIL / VALID-LATER
while:
remaining-work inventory: "What other routine work is still being retained / not yet identified?"
is IMMEDIATE-FRONTIER
The expected reasoning should remain anchored to the missing remaining-work inventory (or an equivalent structural prerequisite) rather than promoting exception-mechanics detail.
Do not require exact wording such as "make a complete list". The semantic requirement is: identify what remaining routine work is actually being retained before reasoning about which capability/process/authority constraints prevent delegation.
Assumption restraint — the following must be rejected (OVERREACH):
- Hands-on experience with one task proves competence for similar work.
- Formal training is generally unnecessary for delegation readiness.
- Success on supplier payments establishes transferability to other routine work.
The user did not state or rely upon those propositions. assumptions: [] is acceptable if no genuinely user-held assumption exists.
Relationship restraint — a relationship equivalent to
"independent routine supplier-payment execution occurred despite lack of formal training"
may be acceptable if represented purely as a relationship between the two stated facts.
It must NOT be strengthened into:
- lack of formal training does not prevent delegation generally
- exception escalation makes delegation safe
- payment success proves readiness for other work`,
},
{
id: "tentative-condition-must-not-map-to-specific-items",
centralStatement:
"I want to reduce how dependent the business is on me for routine operational work, but I am not sure how much responsibility I can realistically hand over to my team yet. I want to understand what is genuinely preventing more delegation rather than just assuming I need to stay involved.",
targetLabel:
"Concrete capability gaps, process maturity levels, or approval authority restrictions limiting delegation",
targetDescription:
"What is actually stopping further delegation right now — not hypothetical constraints but the real bottleneck in the current evidence.",
question:
"What other routine tasks do you currently handle yourself, and does each one require specific training, a documented process, or a defined approval threshold before it can be assigned to someone else?",
answer:
"I still handle customer quotations, purchasing, some supplier issues, technical questions from customers, scheduling work, and checking some larger or unusual payments. I don't think all of those actually require me personally. Some probably just need a clear process or an approval limit. Technical questions may depend on experience, but I haven't separated which ones genuinely need my knowledge from the ones the team could answer. I also haven't documented who could currently take over each of these tasks.",
manualReviewCriteria:
`Observation fidelity — must preserve the stated facts without strengthening them:\n\n- the listed tasks (quotations, purchasing, supplier issues, technical questions, scheduling, payments)\n- user does not think all necessarily require them personally\n- "some probably" may need process/approval\n- technical-question boundary is unresolved\n- ownership/takeover mapping is undocumented\n\nTentative wording must remain tentative.\n"some probably just need..." must NOT become a definite "some only need..."\nunless equivalent uncertainty remains explicit.\n\nAssumption restraint — the following must be rejected as OVERREACH:\n- quotations are delegable with process/approval\n- payments are delegable with process/approval\n- specific named tasks are feasible to delegate because of the generic "some probably" statement\n- process/approval is sufficient\n- deep expertise is unnecessary\n\nThe invariant: A general or unspecified statement about "some" items must not be attached to particular items from a preceding enumeration unless the user makes that mapping.\nAlso: Tentative/probabilistic language must not be strengthened into definite feasibility/readiness claims.\nassumptions: [] is valid.\n\nFrontier: Do NOT over-specify the exact winning frontier.\ntechnical capability, ownership mapping and process/authority are same-level alternatives.\nThe selected uncertainty must be:\n- directly grounded in the answer text\n- investigation-relevant\n- does not depend on invented item-to-condition mappings\nDo not turn an underdetermined frontier into a false exact-answer test.`,
},
];
// ── Deterministic fixture-integrity assertions (zero live calls) ─────────
export const FOCUSED_DECONSTRUCT_REGRESSION_CASE_IDS = [
"nearest-frontier-wins",
"no-genuine-assumption",
"genuine-evidence-sufficiency-assumption",
"juxtaposition-must-not-create-link",
"tentative-observation-fidelity",
"genuine-dependency-frontier",
"structural-frontier-over-working-example-detail",
"tentative-condition-must-not-map-to-specific-items",
];
export function validateRegressionFixtureIntegrity() {
const errors = [];
if (FOCUSED_DECONSTRUCT_REGRESSION_CASES.length !== 8) {
errors.push(
`Expected exactly 8 regression cases; got ${FOCUSED_DECONSTRUCT_REGRESSION_CASES.length}`,
);
}
const ids = FOCUSED_DECONSTRUCT_REGRESSION_CASES.map((c) => c.id);
const uniqueIds = new Set(ids);
if (uniqueIds.size !== ids.length) {
errors.push(`Duplicate IDs found: ${ids.filter((id, i) => ids.indexOf(id) !== i).join(", ")}`);
}
for (const expectedId of FOCUSED_DECONSTRUCT_REGRESSION_CASE_IDS) {
if (!ids.includes(expectedId)) {
errors.push(`Missing expected case ID: ${expectedId}`);
}
}
// Verify no unknown IDs
for (const id of ids) {
if (!FOCUSED_DECONSTRUCT_REGRESSION_CASE_IDS.includes(id)) {
errors.push(`Unexpected case ID: ${id}`);
}
}
// Verify each case has required fields
const requiredFields = ["id", "centralStatement", "targetLabel", "targetDescription", "question", "answer", "manualReviewCriteria"];
for (const i in FOCUSED_DECONSTRUCT_REGRESSION_CASES) {
const case_ = FOCUSED_DECONSTRUCT_REGRESSION_CASES[i];
for (const field of requiredFields) {
if (!(field in case_) || typeof case_[field] !== "string" || !case_[field].trim()) {
errors.push(`Case ${i} (${case_.id}): missing or invalid field "${field}"`);
}
}
}
return errors;
}