diff --git a/docs/current-handoff.md b/docs/current-handoff.md index 2b9b2d7..c9da7ea 100644 --- a/docs/current-handoff.md +++ b/docs/current-handoff.md @@ -116,6 +116,12 @@ - This increment adds only a final semantic-preservation self-check, testing instruction salience rather than missing semantic specification. No live model calls were made. - Next boundary: one manual fixed-scenario Postman production run, scoring C first while checking A/B/D1/D2/E and unsupported expansion or causal strengthening for regressions. +## OpenAI reconstruction-provider experiment apparatus + +- Structural Ollama/provider issues were previously fixed; current semantic comparison now warrants testing model-compliance variance. +- An experiment-only OpenAI Responses API provider apparatus exists for `gpt-5.6-terra`, using the same current v0.5 prompt and canonical reconstruction schema. Production provider selection remains Ollama/Qwen. +- No live OpenAI calls have occurred. Next boundary: a bounded live semantic comparison, not production migration. + ## Current product architecture Three distinct routes, not a single page: diff --git a/lib/llm/provider.js b/lib/llm/provider.js index 04f3e16..aec4778 100644 --- a/lib/llm/provider.js +++ b/lib/llm/provider.js @@ -12,6 +12,14 @@ export function getProvider() { return new OllamaLlmProvider(); } +/** + * Experiment-only construction seam. Production provider selection remains Ollama. + * @param {{ apiKey?: string, fetchImpl?: typeof fetch }} [options] + */ +export function createOpenAIReconstructionProvider(options = {}) { + return new OpenAIReconstructionProvider(options); +} + function recoverJson(raw) { if (typeof raw !== "string") return raw; const trimmed = raw.trim(); @@ -294,3 +302,56 @@ class OllamaLlmProvider { } } } + +class OpenAIReconstructionProvider { + constructor({ apiKey = process.env.OPENAI_API_KEY, fetchImpl = fetch } = {}) { + this.apiKey = apiKey; + this.fetchImpl = fetchImpl; + } + + async generateReconstruction(prompt, modelName = "gpt-5.6-terra") { + if (!this.apiKey) throw new Error("OPENAI_API_KEY is not set"); + + const response = await this.fetchImpl("https://api.openai.com/v1/responses", { + method: "POST", + headers: { + Authorization: `Bearer ${this.apiKey}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + model: modelName, + input: prompt, + text: { + format: { + type: "json_schema", + name: "reconstruction", + strict: true, + schema: reconstructionJsonSchema, + }, + }, + }), + }); + + if (!response.ok) { + const body = await response.text(); + const error = new Error( + `OpenAI Responses API returned ${response.status}: ${body}`, + ); + error.providerApiPath = "/v1/responses"; + throw error; + } + + const data = await response.json(); + const outputText = data.output_text; + if (typeof outputText !== "string") { + const error = new Error("OpenAI Responses API returned no output_text"); + error.providerApiPath = "/v1/responses"; + throw error; + } + + return { + response: recoverJson(outputText), + providerApiPath: "/v1/responses", + }; + } +} diff --git a/tests/llm/provider.test.js b/tests/llm/provider.test.js index a8fe8a1..7a79675 100644 --- a/tests/llm/provider.test.js +++ b/tests/llm/provider.test.js @@ -1,5 +1,9 @@ import { describe, expect, it, vi } from "vitest"; -import { __resetChatSupportForTests, getProvider } from "@/lib/llm/provider.js"; +import { + __resetChatSupportForTests, + createOpenAIReconstructionProvider, + getProvider, +} from "@/lib/llm/provider.js"; describe("OllamaLlmProvider chat capability detection", () => { it("uses the configured model for the chat probe and keeps the chat path", async () => { @@ -114,4 +118,65 @@ describe("OllamaLlmProvider chat capability detection", () => { else process.env.OLLAMA_BASE_URL = originalBaseUrl; } }); +}); + +describe("OpenAI reconstruction provider experiment seam", () => { + it("uses the Responses API with the canonical strict reconstruction schema", async () => { + const fetchSpy = vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ output_text: '{"reconstruction":"result"}' }), + }); + const provider = createOpenAIReconstructionProvider({ + apiKey: "test-key", + fetchImpl: fetchSpy, + }); + + const result = await provider.generateReconstruction( + "current reconstruction prompt", + ); + const request = JSON.parse(fetchSpy.mock.calls[0][1].body); + const schemaText = JSON.stringify(request.text.format.schema); + + expect(fetchSpy).toHaveBeenCalledWith( + "https://api.openai.com/v1/responses", + expect.objectContaining({ method: "POST" }), + ); + expect(request).toMatchObject({ + model: "gpt-5.6-terra", + input: "current reconstruction prompt", + text: { + format: { + type: "json_schema", + name: "reconstruction", + strict: true, + }, + }, + }); + expect(schemaText).toContain("relationship"); + expect(schemaText).toContain("evidenceType"); + expect(schemaText).toContain("primaryType"); + expect(schemaText).toContain("secondaryTypes"); + expect(schemaText).toContain("confidence"); + expect(schemaText).toContain("importance"); + expect(result).toEqual({ + response: { reconstruction: "result" }, + providerApiPath: "/v1/responses", + }); + }); + + it("surfaces Responses API failures without inventing reconstruction content", async () => { + const provider = createOpenAIReconstructionProvider({ + apiKey: "test-key", + fetchImpl: vi.fn().mockResolvedValue({ + ok: false, + status: 429, + text: async () => "rate limited", + }), + }); + + await expect(provider.generateReconstruction("prompt")).rejects.toMatchObject({ + providerApiPath: "/v1/responses", + message: "OpenAI Responses API returned 429: rate limited", + }); + }); }); \ No newline at end of file